Update 2026-09-03 00:10

This commit is contained in:
Sidra 2026-09-03 00:10:18 +04:00
commit 6e939ce6c3
52 changed files with 9529 additions and 0 deletions

276
.env.example Normal file
View File

@ -0,0 +1,276 @@
# =============================================================================
# AGIBOT A3 - Voice Control | configuration
# =============================================================================
# Copy this file to `.env` and edit that copy. Nothing in the source code needs
# to change to switch between the simulator and the real robot.
#
# Restart the server after editing, or press "Reload config" in the dashboard.
# =============================================================================
# -----------------------------------------------------------------------------
# 1. WEB SERVER (this PC)
# -----------------------------------------------------------------------------
# 127.0.0.1 = only this PC can open the dashboard (recommended).
# 0.0.0.0 = other devices on your LAN can open it too (tablet, phone).
HOST=127.0.0.1
PORT=8000
LOG_LEVEL=info
# Only needed if you serve the frontend from a different origin. Usually empty.
CORS_ORIGINS=
# -----------------------------------------------------------------------------
# 2. ROBOT MODE <<< THIS IS THE SWITCH >>>
# -----------------------------------------------------------------------------
# mock = simulated robot, no hardware required (default; use this today)
# real = talk to a physical AGIBOT A3 over the network
ROBOT_MODE=mock
ROBOT_NAME=AGIBOT A3
ROBOT_MODEL=AgiBot A3
# -----------------------------------------------------------------------------
# 3. ROBOT ADDRESS <<< PUT THE ROBOT IP HERE WHEN YOU HAVE IT >>>
# -----------------------------------------------------------------------------
# Leave ROBOT_IP empty while ROBOT_MODE=mock.
# When the robot is powered on and on the same LAN:
# ROBOT_MODE=real
# ROBOT_IP=192.168.1.50 <- the robot's actual IP on YOUR network
# ROBOT_PORT=59301 <- AgiBot's documented A3 TTS RPC port (HDU)
#
# NOTE: AgiBot's own examples use 10.42.10.10 - that is the robot's INTERNAL
# address for its head unit (HDU) and is not reachable from your PC. Use the
# HDU's address on your WiFi/LAN. Find it with: python scripts/discover_robot.py
ROBOT_IP=
ROBOT_PORT=59301
ROBOT_USE_TLS=false
# Timing (seconds). Keep connect timeout short so the UI never feels stuck.
ROBOT_CONNECT_TIMEOUT=3.0
ROBOT_REQUEST_TIMEOUT=8.0
ROBOT_HEALTH_INTERVAL=5.0
ROBOT_RECONNECT_MIN_DELAY=1.0
ROBOT_RECONNECT_MAX_DELAY=15.0
# -----------------------------------------------------------------------------
# 4. HOW TO TALK TO THE A3
# -----------------------------------------------------------------------------
# Only used when ROBOT_MODE=real.
#
# aimdk - AgiBot's documented A3 speech RPC <<< USE THIS, it is the default >>>
# http - a generic HTTP/REST endpoint you specify yourself
# ws - a WebSocket endpoint you specify yourself
# ros2 - a ROS 2 topic or service (requires ROS 2 on this PC)
# ssh - run a command on the robot over SSH (last-resort fallback)
#
# The `aimdk` transport implements the interface AgiBot documents for the A3:
# POST http://<robot>:59301/rpc/aimdk.protocol.TTSService/PlayTTS
# Content-Type: application/json
# {"text": "...", "priority_level": "INTERACTION_L6", "domain": "...",
# "trace_id": "...", "is_interrupted": true}
# Docs: https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/07-02-audio_play
#
# The defaults below come from that documentation, but they are still settings
# rather than hard-coded values: AgiBot does not guarantee ports or names across
# firmware, and there is no endpoint-discovery API. If your unit differs, change
# it here - never in the source. See docs/AGIBOT_A3_INTEGRATION.md.
A3_TRANSPORT=aimdk
# --- 4a. AimDK transport (recommended) ---------------------------------------
# Uses ROBOT_IP and ROBOT_PORT above.
A3_AIMDK_SERVICE=aimdk.protocol.TTSService
A3_AIMDK_PLAY_METHOD=PlayTTS
A3_AIMDK_STOP_METHOD=StopTTSTraceId
A3_AIMDK_STATUS_METHOD=GetAudioStatus
# Playback priority. Only INTERACTION_L6 appears in AgiBot's A3 examples.
A3_AIMDK_PRIORITY=INTERACTION_L6
# Free-form caller tag sent with each request, so robot-side logs show who spoke.
A3_AIMDK_DOMAIN=voice_control
# true = a new utterance interrupts whatever is currently playing.
A3_AIMDK_INTERRUPT=true
# Hard request limit documented by AgiBot: 1024 BYTES of UTF-8 (not characters).
# Longer text is split automatically on sentence boundaries and sent in order.
A3_AIMDK_MAX_BYTES=1024
# --- 4b. Generic HTTP / REST transport ---------------------------------------
# Only needed if your unit does NOT use the AimDK interface above.
# Path is relative to http://ROBOT_IP:ROBOT_PORT
# Example shape only - replace with the documented endpoint:
# A3_HTTP_SPEAK_PATH=/api/v1/tts/speak
A3_HTTP_SPEAK_PATH=
A3_HTTP_SPEAK_METHOD=POST
# JSON body template. {text} is replaced with the operator's text.
# Placeholders: {text} {id} {voice} {language} {volume} {speed}
# A key whose value resolves to nothing is dropped from the request.
A3_HTTP_SPEAK_PAYLOAD={"text": "{text}"}
# Optional: endpoint that interrupts speech.
A3_HTTP_STOP_PATH=
A3_HTTP_STOP_METHOD=POST
A3_HTTP_STOP_PAYLOAD={}
# Optional: cheap endpoint polled for the connection indicator.
# If left empty the app falls back to a TCP connect test on ROBOT_PORT.
A3_HTTP_STATUS_PATH=
A3_HTTP_STATUS_METHOD=GET
# Optional: extra headers and auth.
A3_HTTP_HEADERS={}
A3_HTTP_AUTH_TOKEN=
# Optional: dotted path to a success flag in the JSON response, e.g. "code" or
# "result.success". Leave empty to trust the HTTP status code alone.
A3_HTTP_SUCCESS_FIELD=
# --- 4c. WebSocket transport --------------------------------------------------
A3_WS_PATH=
A3_WS_SPEAK_PAYLOAD={"text": "{text}"}
A3_WS_STOP_PAYLOAD={}
A3_WS_PING_INTERVAL=20
# Optional: how the robot announces "finished speaking". With these set, the
# dashboard shows real completion instead of an estimate.
# A3_WS_DONE_FIELD=event
# A3_WS_DONE_VALUE=speech_end
A3_WS_DONE_FIELD=
A3_WS_DONE_VALUE=
# --- 4d. ROS 2 transport ------------------------------------------------------
# Requires ROS 2 installed on this PC and the same ROS_DOMAIN_ID as the robot.
A3_ROS_DOMAIN_ID=0
A3_ROS_SPEAK_TOPIC=
A3_ROS_SPEAK_MSG_TYPE=std_msgs/msg/String
A3_ROS_SPEAK_MSG_FIELD=data
A3_ROS_STOP_TOPIC=
A3_ROS_USE_SERVICE=false
A3_ROS_SERVICE_NAME=
A3_ROS_SERVICE_TYPE=
# --- 4e. SSH transport (last resort) --------------------------------------------
# Runs a command on the robot's own Linux computer. {text} is shell-quoted.
# A3_SSH_SPEAK_COMMAND=<the command that makes your robot speak> {text}
A3_SSH_USER=root
A3_SSH_PORT=22
A3_SSH_KEY_PATH=
A3_SSH_PASSWORD=
A3_SSH_SPEAK_COMMAND=
A3_SSH_STOP_COMMAND=
A3_SSH_PROBE_COMMAND=true
# --- 4f. Optional TTS parameters (sent only when set) -------------------------
A3_VOICE=
A3_LANGUAGE=
A3_VOLUME=
A3_SPEED=
# -----------------------------------------------------------------------------
# 5. SPEECH BEHAVIOUR
# -----------------------------------------------------------------------------
SPEECH_MAX_LENGTH=1000
SPEECH_MIN_LENGTH=1
# true = pressing Speak while talking interrupts and says the new text
# false = pressing Speak while talking is rejected with "already speaking"
SPEECH_ALLOW_INTERRUPT=true
SPEECH_HISTORY_LIMIT=100
# -----------------------------------------------------------------------------
# 6. MOCK ROBOT (ROBOT_MODE=mock only)
# -----------------------------------------------------------------------------
MOCK_CONNECT_DELAY_MS=350
MOCK_NETWORK_LATENCY_MS=45
MOCK_PROCESSING_MS=180
MOCK_WORDS_PER_MINUTE=150
# Set to 0.2 to make 1 in 5 utterances fail, for testing error handling.
MOCK_FAILURE_RATE=0.0
# Set to true to make the simulated link drop occasionally, for testing reconnect.
MOCK_FLAKY_CONNECTION=false
# --- PC speaker playback (simulator only) ------------------------------------
# true = the simulator actually SPEAKS the text through this laptop's speakers,
# so you can rehearse a demo before the robot is on the network.
# When this is on, the real audio drives the UI: "Completed" appears exactly when
# the sound stops, and Stop cuts the voice mid-word.
#
# Windows uses the built-in SAPI voices - nothing to install.
# macOS uses `say`. Linux needs: sudo apt install espeak-ng
MOCK_LOCAL_AUDIO=true
# Which PC voice to use (matched against the voice name, case-insensitive).
# python scripts/voices.py list what is installed
# python scripts/voices.py --demo hear each one
#
# NOTE: this is the SIMULATOR's voice, not the robot's - the real A3 synthesises
# speech on-board. The robot's configured voice is "Yunxiao": a MALE TEENAGER,
# multi-language. Windows ships no teenage voice, so the closest approximation is
# its lighter adult male voice (Mark) pitched up a little. It is a stand-in, not
# a match. Run `python scripts/voices.py --demo` to hear the alternatives.
MOCK_VOICE=Mark
# Speaking rate, -10 (slowest) to 10 (fastest). Try -1 or -2 for a noisy room.
MOCK_SPEECH_RATE=0
# Volume, 0-100.
MOCK_SPEECH_VOLUME=100
# Pitch, -10 (deepest) to 10 (highest). Left at 0: raising it makes the voice
# sound affected rather than natural, which is the opposite of what a service
# robot should sound like.
MOCK_SPEECH_PITCH=0
# -----------------------------------------------------------------------------
# 7. GEMINI CLOUD VOICE (simulator only - nothing here touches the real robot)
# -----------------------------------------------------------------------------
# The built-in Windows voice is instant but robotic. Gemini gives the simulator a
# natural neural voice, at the cost of a network round trip.
#
# system = built-in OS voice - instant, robotic
# gemini = Gemini neural voice - natural, ~4s to synthesise new text
MOCK_VOICE_ENGINE=system
# Get a key at https://aistudio.google.com/apikey
GEMINI_API_KEY=
GEMINI_TTS_MODEL=gemini-3.1-flash-tts-preview
# Prebuilt voice. Iapetus ("Clear") is the default: professional and straight,
# which is how a tour-guide robot actually speaks. Charon ("Informative") is the
# same register but deeper and more adult - swap it in if you prefer that.
# Puck / Fenrir read younger but bouncier. Hear them all:
# python scripts/voices.py
GEMINI_VOICE=Iapetus
# Optional acting direction. LEAVE THIS EMPTY. Anything here makes the model
# *perform* the line instead of simply saying it, which sounds theatrical and
# fake for a service robot. Only set it for a deliberate stage character.
GEMINI_TTS_STYLE=
# 0 = never split. Every line is spoken as ONE clip, in one continuous take.
# Set a character count (e.g. 280) only if you want very long paragraphs to
# start speaking before the whole thing is synthesised - the pieces are then
# synthesised separately, and the join between them can be audible.
GEMINI_CHUNK_CHARS=0
# LATENCY: new text takes ~4s (short) to ~8s (paragraph) to synthesise. Audio is
# cached on disk in .voice-cache/, so a repeated line is INSTANT and stays
# instant across restarts. Before a live demo, warm your lines:
# python scripts/warm_voice.py --file demo_lines.txt
# If Gemini is unreachable the simulator falls back to the built-in voice, so a
# network problem never leaves you with silence.

13
.gitignore vendored Normal file
View File

@ -0,0 +1,13 @@
# Local configuration - contains the robot address and any credentials.
.env
__pycache__/
*.py[cod]
.venv/
venv/
*.log
.DS_Store
# Saved speech audio (regenerated on demand).
audio_library/

392
README.md Normal file
View File

@ -0,0 +1,392 @@
# AGIBOT A3 · Voice Control
A local web dashboard for making an **AGIBOT A3** humanoid speak typed text through
its own built-in speaker.
```
PC Browser → Local Web App → Robot Service Layer → A3 Adapter → AGIBOT A3 → Speaker
```
It runs today, with **no robot attached**, against a built-in simulator. When the
robot arrives you change the IP in one file and switch one setting.
**The A3's speech interface is already implemented.** AgiBot publicly documents a
native text-to-speech HTTP endpoint on the robot, and this project speaks it:
```http
POST http://<robot>:59301/rpc/aimdk.protocol.TTSService/PlayTTS
Content-Type: application/json
{"text": "...", "priority_level": "INTERACTION_L6", "domain": "voice_control",
"trace_id": "...", "is_interrupted": true}
```
Text goes in, the robot's own TTS comes out of its own speaker. **No audio is
generated or transferred by the PC**, which is what makes it fast. Full detail,
sources and open questions: **[docs/AGIBOT_A3_INTEGRATION.md](docs/AGIBOT_A3_INTEGRATION.md)**.
<br>
## Quick start
```bat
start.bat
```
…then open **http://localhost:8000**.
Or manually:
```bash
pip install -r requirements.txt
python backend/main.py
```
The app starts in **Mock Mode**: the dashboard, the API, the status pipeline, the
history and every error path work exactly as they will with the real robot — the
robot itself is simulated.
**You will hear it.** The simulator speaks each utterance through this PC's
speakers, so you can rehearse a demo — wording, pacing, the Stop button — before
the A3 is on the network. The real audio drives the UI: *Completed* appears when
the sound actually stops, and **Stop** cuts the voice mid-word.
Two voice engines, set by `MOCK_VOICE_ENGINE` in `.env`:
| | `system` | `gemini` |
| --- | --- | --- |
| Sound | robotic, built-in OS voice | **natural neural voice** |
| New text | instant | ~4 s (sentence) 8 s (paragraph) |
| Repeated text | instant | **instant** — cached on disk |
| Needs | nothing | `GEMINI_API_KEY` + internet |
```bash
python scripts/voices.py # list both engines' voices
python scripts/voices.py --demo # hear the system ones
```
### Saved audio
Every line spoken with the neural voice is saved to **`audio_library/`** as an
ordinary `.wav` named after its text:
```
audio_library/
welcome-to-our-showroom-78e24feb.wav
please-follow-me-to-the-first-exhibit-653c14ff.wav
index.json
```
That means a line spoken once **replays instantly** — no synthesis, no network,
no wait — and the files are yours: play them in any media player, drop them into
a video edit, or put them on a stand's playlist.
The dashboard's **Saved audio** panel lists them with play and download buttons,
and history rows that already have audio get a ▶ for instant replay.
**Before a live demo, warm your lines** so nothing waits on synthesis:
```bash
python scripts/warm_voice.py --file demo_lines.txt
python scripts/warm_voice.py --stats
```
If Gemini is unreachable — no network, bad key, quota — the simulator silently
falls back to the built-in voice, so a network problem never leaves you with
silence.
Pick one with `MOCK_VOICE=` in `.env`; silence it entirely with
`MOCK_LOCAL_AUDIO=false`.
This is the *simulator's* voice — the real A3 synthesises its own speech
on-board. The robot's configured voice is **"Yunxiao"** (teenager, male,
multi-language). Windows ships no teenage voice, so the default here is its
lighter adult male voice (**Mark**) pitched up (`MOCK_SPEECH_PITCH=3`) to read
younger. It is an approximation for rehearsal, not a match.
Verify the whole stack end to end at any time:
```bash
python scripts/selftest.py # 31 checks: REST, WebSocket, speak, stop, errors, history
```
### Testing the real robot path before the robot exists
Mock mode tests the *application*. To exercise the actual A3 wire protocol —
the URL, the JSON body, the `trace_id` round trip, byte-limit chunking, Stop —
run the local stand-in:
```bash
python scripts/fake_a3_server.py # implements the documented A3 contract on :59301
```
then set `ROBOT_MODE=real`, `ROBOT_IP=127.0.0.1` in `.env`. Every utterance
prints in the stand-in's console. This project's `aimdk` transport passes all 31
checks against it.
<br>
## Using the dashboard
| Action | How |
| --- | --- |
| Speak | Type, then click **Speak** or press **Ctrl + Enter** |
| Stop | **Stop** button or **Esc** |
| Clear the box | **Clear** button |
| Repeat something | Click any entry in **History** — it goes back in the box |
| Reconnect | The ↻ button in the header |
`Enter` inserts a newline and does **not** send — a half-typed sentence should
never reach the robot's speaker mid-demo.
The header always states the truth about the connection: green *Robot Connected*,
amber *Connecting…*, red *Robot Disconnected*, plus the measured round-trip
latency to the robot.
<br>
## Project layout
```
A3_text_to_speach/
├── frontend/ no build step - plain ES modules, served by the backend
│ ├── index.html
│ ├── styles/main.css
│ └── js/
│ ├── app.js wiring + speech lifecycle
│ ├── api.js REST client
│ ├── socket.js WebSocket client (auto-reconnect, RTT ping)
│ └── ui.js all DOM rendering
├── backend/
│ ├── main.py entry point: app factory, static serving, error mapping
│ ├── api/
│ │ ├── routes.py REST endpoints
│ │ ├── websocket.py live status/lifecycle channel
│ │ └── schemas.py
│ ├── services/
│ │ ├── speech_service.py validation, single-flight, interrupt, latency
│ │ └── history.py
│ ├── robot/
│ │ ├── base.py ← the only contract the app depends on
│ │ ├── factory.py mode → adapter
│ │ ├── manager.py connect / health-poll / backoff reconnect
│ │ ├── mock_robot.py the simulator
│ │ ├── agibot_a3.py ← THE INTEGRATION POINT
│ │ └── transports/
│ │ ├── aimdk_transport.py ← AgiBot's documented A3 speech RPC (default)
│ │ └── http · ws · ros2 · ssh fallbacks, fully config-driven
│ ├── core/ event bus, text utils, logging
│ └── config/settings.py every tunable, loaded from .env
├── docs/
│ ├── AGIBOT_A3_INTEGRATION.md how A3 speech works + what to fill in
│ └── NETWORK.md PC ↔ robot networking and troubleshooting
├── scripts/
│ ├── voices.py list / audition the simulator's voices
│ ├── warm_voice.py pre-synthesise demo lines (instant playback)
│ ├── selftest.py end-to-end test (31 checks)
│ ├── discover_robot.py probe a robot IP for its speech interface
│ └── fake_a3_server.py local stand-in for the A3's RPC, to test the real path
├── audio_library/ saved .wav files + index (instant replay)
├── pronunciation.json how the simulator says tricky names
├── .env ← YOUR CONFIGURATION (robot IP goes here)
├── .env.example documented template
├── requirements.txt
└── start.bat / start.sh
```
### Why it is layered this way
Each arrow is a seam you can replace without touching the others:
```
Browser
│ REST for commands, WebSocket for state
Backend API (api/)
│ domain objects only - no HTTP, no sockets
Speech Service (services/)
│ RobotAdapter interface
Robot Adapter (robot/mock_robot.py | robot/agibot_a3.py)
│ SpeechTransport interface
Transport (robot/transports/*)
AGIBOT A3
```
Nothing above `robot/` knows a robot SDK exists. Adding an AgiBot X2, a second
robot, or a different TTS engine means writing one adapter and adding one line to
`factory.py` — the UI, the API and the service layer are untouched.
<br>
## Configuration
Everything lives in **`.env`** (created from `.env.example` on first run). The
source code contains no IP address, port or endpoint.
```env
ROBOT_MODE=mock # mock | real ← the switch
ROBOT_IP= # ← the robot's IP goes here
ROBOT_PORT=59301 # AgiBot's documented A3 TTS RPC port
A3_TRANSPORT=aimdk # aimdk | http | ws | ros2 | ssh
```
After editing `.env`, either restart the server or call:
```bash
curl -X POST http://localhost:8000/api/config/reload
```
which rebuilds the robot adapter in place — no restart, no code change.
### Mock-mode knobs worth knowing
| Setting | Effect |
| --- | --- |
| `MOCK_FAILURE_RATE=0.3` | 3 in 10 utterances fail — exercises the error UI |
| `MOCK_FLAKY_CONNECTION=true` | the link drops periodically — exercises reconnect |
| `MOCK_LOCAL_AUDIO=true` | **on by default** — actually speaks through this PC's speakers |
| `MOCK_VOICE_ENGINE=gemini` | natural neural voice instead of the robotic built-in one |
| `GEMINI_VOICE=Puck` | which neural voice (Puck/Fenrir read youngest + male) |
| `MOCK_VOICE=Mark` | which *system* voice to use, when engine is `system` |
| `MOCK_SPEECH_PITCH=3` | raise the pitch — approximates the robot's young male voice |
| `MOCK_SPEECH_RATE=-2` | slow the simulator down for a noisy room (10…10) |
| `SPEECH_ALLOW_INTERRUPT=false` | a second Speak while talking is rejected instead of interrupting |
<br>
## API
| Method | Path | Purpose |
| --- | --- | --- |
| `GET` | `/api/health` | web app liveness (always 200, even with the robot offline) |
| `GET` | `/api/robot/status` | connection state, latency, uptime, config problems |
| `POST` | `/api/robot/speak` | `{"text": "..."}` → speak it |
| `POST` | `/api/robot/stop` | interrupt the current utterance |
| `POST` | `/api/robot/reconnect` | retry the connection immediately |
| `GET` | `/api/robot/diagnostics` | adapter + transport detail, for troubleshooting |
| `GET` | `/api/speech/history` | recent utterances |
| `DELETE` | `/api/speech/history` | clear it |
| `GET` | `/api/audio` | saved clips + library stats |
| `GET` | `/api/audio/{id}/file` | the `.wav` itself (play or download) |
| `DELETE` | `/api/audio/{id}` | delete one clip |
| `DELETE` | `/api/audio` | delete all saved audio |
| `GET` | `/api/config` | non-secret configuration the UI needs |
| `POST` | `/api/config/reload` | re-read `.env` and rebuild the adapter |
| `WS` | `/ws` | live status, speech lifecycle, history updates |
Interactive docs: **http://localhost:8000/api/docs**
```bash
curl -X POST http://localhost:8000/api/robot/speak \
-H "Content-Type: application/json" \
-d "{\"text\": \"Hello, welcome to our showroom.\"}"
```
```json
{ "success": true, "status": "processing",
"requestId": "3c7cfe41cf7d", "ackLatencyMs": 56 }
```
**`/api/robot/speak` returns when the robot has *accepted* the utterance, not when
it stops talking.** A 12-second sentence must not look like a 12-second-slow
button. The rest of the lifecycle — `speaking`, `completed`, failures — arrives on
the WebSocket.
<br>
## How the speed is achieved
| Decision | Why |
| --- | --- |
| Text is sent to the robot; audio is never generated on the PC | no synthesis, no file, no upload — one small HTTP request |
| One persistent connection, opened at startup and reused | no TCP/TLS handshake per utterance |
| HTTP response returns at acknowledgement | button feels instant; speaking progress streams separately |
| WebSocket push, zero polling | the UI changes the moment the robot does |
| Health probe is separate from the speech path | a slow health check can never delay a Speak |
| Short connect timeout (3 s), long request timeout (8 s) | an offline robot fails fast; a busy one is not cut off |
<br>
## Error handling
| Situation | What you see |
| --- | --- |
| Robot offline | `Robot is offline. Check the robot IP address and network connection.` |
| `ROBOT_IP` empty while `ROBOT_MODE=real` | a configuration banner naming the exact `.env` key |
| Invalid IP / hostname | configuration error before any connection is attempted |
| Connection timeout | fails after 3 s, UI stays responsive, retry continues in the background |
| TTS failure | `Speech request failed.` + the robot's own reason in the server log |
| Empty text | Speak is disabled; the API returns `400` |
| Text too long | rejected with the actual limit named |
| Disconnect mid-utterance | status flips immediately and the utterance is marked failed |
| Backend stopped | the page says so rather than showing a stale green light |
The robot being unreachable is treated as normal, not exceptional: the server
keeps serving the dashboard and retries with capped exponential backoff.
<br>
## Security
The browser never talks to the robot.
```
Browser → local backend (127.0.0.1) → robot
```
- The server binds to `127.0.0.1` by default — nothing is exposed to the LAN.
- Robot credentials stay in `.env` on the PC; `/api/config` returns a filtered
view with no tokens, passwords or key paths.
- No robot port is proxied or forwarded to the page.
Set `HOST=0.0.0.0` only if you deliberately want to open the dashboard from a
tablet on the same network.
<br>
## Connecting the real robot
Full procedure: **[docs/AGIBOT_A3_INTEGRATION.md](docs/AGIBOT_A3_INTEGRATION.md)**
· Networking and troubleshooting: **[docs/NETWORK.md](docs/NETWORK.md)**
Short version, once you have the robot's IP:
```bash
ping 192.168.1.50 # 1. is it reachable
python scripts/discover_robot.py 192.168.1.50 # 2. what does it expose
```
```bash
# 3. the decisive test - this makes the robot talk
python scripts/discover_robot.py 192.168.1.50 --speak "Hello, I am Expedition A3"
```
then in `.env`**three lines, no code changes**:
```env
ROBOT_MODE=real
ROBOT_IP=192.168.1.50
ROBOT_PORT=59301
```
and restart (or `curl -X POST http://localhost:8000/api/config/reload`).
> Do **not** use `10.42.10.10`. It appears throughout AgiBot's examples but is
> the robot's *internal* address — your PC cannot reach it. Use the head unit's
> address on your own network.
<br>
## Requirements
- **Python 3.9+** (developed and tested on 3.11)
- Windows, macOS or Linux
- A browser from the last few years
Runtime dependencies: `fastapi`, `uvicorn`, `python-dotenv`, `pydantic`, `httpx`,
`websockets`. Optional: `pyttsx3` (mock audio), `paramiko` (SSH password auth),
`rclpy` (ROS 2 transport — comes from a ROS 2 install, not pip).

0
backend/__init__.py Normal file
View File

0
backend/api/__init__.py Normal file
View File

172
backend/api/routes.py Normal file
View File

@ -0,0 +1,172 @@
"""HTTP API.
The browser talks only to this backend; the backend talks to the robot. No robot
address, port or credential ever reaches the page.
"""
from __future__ import annotations
import logging
import platform
import time
from typing import Any, Dict
from fastapi import APIRouter, HTTPException, Request
from fastapi.responses import FileResponse
from ..config.settings import reload_settings
from ..core.events import EventType
from ..robot.manager import RobotManager
from ..services.audio_library import get_audio_library
from ..services.speech_service import SpeechService
from .schemas import SimpleResponse, SpeakRequest, SpeakResponse
logger = logging.getLogger(__name__)
router = APIRouter(prefix="/api", tags=["robot"])
STARTED_AT = time.time()
def _manager(request: Request) -> RobotManager:
return request.app.state.robot_manager
def _speech(request: Request) -> SpeechService:
return request.app.state.speech_service
# --------------------------------------------------------------------------- #
# health / config
# --------------------------------------------------------------------------- #
@router.get("/health")
async def health(request: Request) -> Dict[str, Any]:
"""Liveness of the *web app* - always 200, even with the robot offline."""
return {
"status": "ok",
"uptimeSeconds": round(time.time() - STARTED_AT, 1),
"python": platform.python_version(),
"mode": request.app.state.settings.robot.mode,
}
@router.get("/config")
async def get_config(request: Request) -> Dict[str, Any]:
return request.app.state.settings.public_dict()
@router.post("/config/reload")
async def post_config_reload(request: Request) -> Dict[str, Any]:
"""Re-read .env and rebuild the robot adapter without restarting the server.
This is what you call after pasting the robot's IP into .env.
"""
settings = reload_settings()
request.app.state.settings = settings
_speech(request).update_settings(settings)
await _manager(request).rebuild(settings)
request.app.state.event_bus.publish(EventType.CONFIG_UPDATED, settings.public_dict())
logger.info("configuration reloaded: mode=%s address=%s",
settings.robot.mode, settings.robot.address)
return {"success": True, "config": settings.public_dict()}
# --------------------------------------------------------------------------- #
# robot
# --------------------------------------------------------------------------- #
@router.get("/robot/status")
async def robot_status(request: Request) -> Dict[str, Any]:
status = _manager(request).status_dict()
status["busy"] = _speech(request).is_busy
status["activeRequestId"] = _speech(request).active_request_id
return status
@router.get("/robot/diagnostics")
async def robot_diagnostics(request: Request) -> Dict[str, Any]:
manager = _manager(request)
detail = await manager.adapter.describe()
return {
"status": manager.status_dict(),
"adapter": detail,
"config": request.app.state.settings.public_dict(),
}
@router.post("/robot/speak", response_model=SpeakResponse)
async def robot_speak(payload: SpeakRequest, request: Request) -> Dict[str, Any]:
"""Send text to the robot.
Returns once the robot has *acknowledged* the utterance. Speaking progress and
completion arrive over the WebSocket at /ws.
"""
return await _speech(request).speak(payload.text, payload.voice, payload.language)
@router.post("/robot/stop", response_model=SimpleResponse)
async def robot_stop(request: Request) -> Dict[str, Any]:
return await _speech(request).stop()
@router.post("/robot/reconnect")
async def robot_reconnect(request: Request) -> Dict[str, Any]:
manager = _manager(request)
manager.request_reconnect()
return {"success": True, "status": manager.state.value}
# --------------------------------------------------------------------------- #
# history
# --------------------------------------------------------------------------- #
@router.get("/speech/history")
async def speech_history(request: Request) -> Dict[str, Any]:
service = _speech(request)
items = service.annotate_saved(service.history.list())
return {"items": items, "count": len(items)}
@router.delete("/speech/history")
async def speech_history_delete(request: Request) -> Dict[str, Any]:
return _speech(request).clear_history()
@router.post("/speech/history/clear")
async def speech_history_clear(request: Request) -> Dict[str, Any]:
"""Same as DELETE - kept because it is easier to call from a plain form/curl."""
return _speech(request).clear_history()
# --------------------------------------------------------------------------- #
# saved audio
# --------------------------------------------------------------------------- #
@router.get("/audio")
async def audio_list() -> Dict[str, Any]:
"""Every line that has been synthesised and saved as a .wav."""
library = get_audio_library()
return {"items": library.list(), "stats": library.stats()}
@router.get("/audio/{audio_id}/file")
async def audio_file(audio_id: str):
"""The .wav itself - playable in the browser, or downloadable."""
entry = get_audio_library().get(audio_id)
if entry is None:
raise HTTPException(status_code=404, detail="No saved audio with that id.")
path = get_audio_library().root / entry["file"]
return FileResponse(
str(path), media_type="audio/wav", filename=entry["file"],
headers={"Cache-Control": "public, max-age=31536000"},
)
@router.delete("/audio/{audio_id}")
async def audio_delete(audio_id: str) -> Dict[str, Any]:
if not get_audio_library().delete(audio_id):
raise HTTPException(status_code=404, detail="No saved audio with that id.")
return {"success": True}
@router.delete("/audio")
async def audio_clear() -> Dict[str, Any]:
"""Delete every saved clip. They are re-created on demand."""
return {"success": True, "removed": get_audio_library().clear()}

32
backend/api/schemas.py Normal file
View File

@ -0,0 +1,32 @@
"""Request/response models for the HTTP API."""
from __future__ import annotations
from typing import Optional
from pydantic import BaseModel, Field
class SpeakRequest(BaseModel):
text: str = Field(..., description="What the robot should say.")
voice: Optional[str] = Field(None, description="Optional voice id, if the robot supports it.")
language: Optional[str] = Field(None, description="Optional language/locale hint.")
class SpeakResponse(BaseModel):
success: bool
status: str
requestId: str
text: str
ackLatencyMs: Optional[int] = None
class SimpleResponse(BaseModel):
success: bool
status: Optional[str] = None
class ErrorResponse(BaseModel):
success: bool = False
error: str
errorCode: str

114
backend/api/websocket.py Normal file
View File

@ -0,0 +1,114 @@
"""Real-time channel to the dashboard.
One WebSocket replaces all polling: connection state, speech lifecycle, errors and
history updates are pushed the moment they happen. The socket also answers
client pings so the UI can display an honest browser->backend round-trip time.
"""
from __future__ import annotations
import asyncio
import json
import logging
from typing import Any, Dict
from fastapi import APIRouter, WebSocket, WebSocketDisconnect
from ..core.events import Event, EventBus, EventType
logger = logging.getLogger(__name__)
ws_router = APIRouter()
@ws_router.websocket("/ws")
async def dashboard_socket(websocket: WebSocket) -> None:
await websocket.accept()
app = websocket.app
bus: EventBus = app.state.event_bus
manager = app.state.robot_manager
speech = app.state.speech_service
settings = app.state.settings
subscription = bus.subscribe(name="ws")
logger.debug("dashboard socket opened (subscribers=%d)", bus.subscriber_count)
try:
# Prime the page with everything it needs for a correct first paint.
await _send(
websocket,
{
"type": "hello",
"data": {
"config": settings.public_dict(),
"status": _status(manager, speech),
"history": speech.annotate_saved(speech.history.list()),
},
},
)
pump = asyncio.create_task(
_pump(websocket, subscription, speech), name="ws-pump"
)
try:
while True:
raw = await websocket.receive_text()
await _handle_client_message(websocket, raw, manager, speech)
finally:
pump.cancel()
try:
await pump
except (asyncio.CancelledError, Exception):
pass
except WebSocketDisconnect:
logger.debug("dashboard socket closed by client")
except Exception as exc: # pragma: no cover - transport level
logger.debug("dashboard socket error: %s", exc)
finally:
subscription.close()
async def _pump(websocket: WebSocket, subscription, speech) -> None:
"""Forward bus events to this browser until cancelled."""
while True:
event: Event = await subscription.get()
if event is None: # pragma: no cover
continue
payload = event.to_dict()
if event.type == EventType.ROBOT_STATUS:
# The robot layer does not know about the speech queue; add it here so
# every status the page sees carries a current busy flag.
payload["data"] = dict(payload["data"])
payload["data"]["busy"] = speech.is_busy
payload["data"]["activeRequestId"] = speech.active_request_id
await _send(websocket, payload)
async def _handle_client_message(websocket: WebSocket, raw: str, manager, speech) -> None:
try:
message: Dict[str, Any] = json.loads(raw)
except json.JSONDecodeError:
return
kind = message.get("type")
if kind == "ping":
# Echo the client's timestamp so it can compute RTT without clock sync.
await _send(websocket, {"type": "pong", "data": {"t": message.get("t")}})
elif kind == "status":
await _send(
websocket, {"type": EventType.ROBOT_STATUS, "data": _status(manager, speech)}
)
elif kind == "reconnect":
manager.request_reconnect()
def _status(manager, speech) -> Dict[str, Any]:
status = manager.status_dict()
status["busy"] = speech.is_busy
status["activeRequestId"] = speech.active_request_id
return status
async def _send(websocket: WebSocket, payload: Dict[str, Any]) -> None:
await websocket.send_text(json.dumps(payload, ensure_ascii=False, default=str))

View File

540
backend/config/settings.py Normal file
View File

@ -0,0 +1,540 @@
"""Typed application configuration, loaded once from the environment / .env file.
Everything that could differ between "my laptop with no robot" and "the show floor
with a live AGIBOT A3" lives here. No module outside this package reads os.environ
directly, and no robot address is ever hard-coded in source.
"""
from __future__ import annotations
import ipaddress
import json
import os
import re
from dataclasses import dataclass, field
from pathlib import Path
from typing import Any, Dict, List, Literal, Optional
from dotenv import load_dotenv
PROJECT_ROOT = Path(__file__).resolve().parents[2]
ENV_FILE = PROJECT_ROOT / ".env"
RobotMode = Literal["mock", "real"]
Transport = Literal["aimdk", "http", "ws", "ros2", "ssh"]
TRANSPORTS = ("aimdk", "http", "ws", "ros2", "ssh")
# --------------------------------------------------------------------------- #
# env helpers
# --------------------------------------------------------------------------- #
def _raw(key: str) -> Optional[str]:
value = os.environ.get(key)
if value is None:
return None
value = value.strip()
# Treat `KEY=` and quoted-empty as "unset" rather than "empty string".
if value in ('', '""', "''"):
return None
if len(value) >= 2 and value[0] == value[-1] and value[0] in ('"', "'"):
value = value[1:-1]
return value or None
def env_str(key: str, default: str = "") -> str:
value = _raw(key)
return default if value is None else value
def env_opt(key: str) -> Optional[str]:
return _raw(key)
def env_int(key: str, default: int) -> int:
value = _raw(key)
if value is None:
return default
try:
return int(value)
except ValueError:
return default
def env_float(key: str, default: float) -> float:
value = _raw(key)
if value is None:
return default
try:
return float(value)
except ValueError:
return default
def env_bool(key: str, default: bool) -> bool:
value = _raw(key)
if value is None:
return default
return value.lower() in ("1", "true", "yes", "on", "y")
def env_json(key: str, default: Any) -> Any:
"""Parse a JSON-valued env var; fall back to `default` on malformed input."""
value = _raw(key)
if value is None:
return default
try:
return json.loads(value)
except json.JSONDecodeError:
return default
def env_list(key: str, default: Optional[List[str]] = None) -> List[str]:
value = _raw(key)
if value is None:
return list(default or [])
return [item.strip() for item in value.split(",") if item.strip()]
# --------------------------------------------------------------------------- #
# configuration issues (surfaced in the UI instead of crashing the app)
# --------------------------------------------------------------------------- #
@dataclass(frozen=True)
class ConfigIssue:
level: Literal["error", "warning"]
key: str
message: str
def to_dict(self) -> Dict[str, str]:
return {"level": self.level, "key": self.key, "message": self.message}
# --------------------------------------------------------------------------- #
# settings groups
# --------------------------------------------------------------------------- #
@dataclass(frozen=True)
class ServerSettings:
host: str
port: int
log_level: str
cors_origins: List[str]
@dataclass(frozen=True)
class RobotSettings:
"""Vendor-neutral robot connection settings."""
mode: RobotMode
name: str
model: str
ip: Optional[str]
port: int
use_tls: bool
connect_timeout: float
request_timeout: float
health_interval: float
reconnect_min_delay: float
reconnect_max_delay: float
@property
def scheme(self) -> str:
return "https" if self.use_tls else "http"
@property
def ws_scheme(self) -> str:
return "wss" if self.use_tls else "ws"
@property
def base_url(self) -> str:
return "{0}://{1}:{2}".format(self.scheme, self.ip, self.port)
@property
def ws_base_url(self) -> str:
return "{0}://{1}:{2}".format(self.ws_scheme, self.ip, self.port)
@property
def address(self) -> str:
return "{0}:{1}".format(self.ip, self.port) if self.ip else "(not configured)"
@dataclass(frozen=True)
class A3Settings:
"""AGIBOT A3 adapter settings.
IMPORTANT: every endpoint below is *configuration*, never a guess baked into
code. See docs/AGIBOT_A3_INTEGRATION.md - once the interface for your unit is
confirmed you fill these in and the adapter works unchanged.
"""
transport: Transport
# -- AimDK transport (AgiBot's documented A3 speech RPC) ------------------
# POST http://<robot>:<port>/rpc/<service>/<method>
# https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/07-02-audio_play
aimdk_service: str
aimdk_play_method: str
aimdk_stop_method: str
aimdk_status_method: str
aimdk_priority: str
aimdk_domain: str
aimdk_interrupt: bool
aimdk_max_bytes: int
# -- HTTP / REST transport ------------------------------------------------
http_speak_path: str
http_speak_method: str
http_speak_payload: Any
http_stop_path: str
http_stop_method: str
http_stop_payload: Any
http_status_path: str
http_status_method: str
http_headers: Dict[str, str]
http_auth_token: Optional[str]
http_success_field: Optional[str]
# -- WebSocket transport --------------------------------------------------
ws_path: str
ws_speak_payload: Any
ws_stop_payload: Any
ws_ping_interval: float
ws_done_field: Optional[str]
ws_done_value: Optional[str]
# -- ROS 2 transport ------------------------------------------------------
ros_domain_id: int
ros_speak_topic: str
ros_speak_msg_type: str
ros_speak_msg_field: str
ros_stop_topic: str
ros_use_service: bool
ros_service_name: str
ros_service_type: str
# -- SSH / on-robot command transport -------------------------------------
ssh_user: str
ssh_password: Optional[str]
ssh_key_path: Optional[str]
ssh_port: int
ssh_speak_command: str
ssh_stop_command: str
ssh_probe_command: str
# -- shared TTS parameters (only sent when set) ---------------------------
voice: Optional[str]
language: Optional[str]
volume: Optional[float]
speed: Optional[float]
@dataclass(frozen=True)
class MockSettings:
connect_delay_ms: int
network_latency_ms: int
processing_ms: int
words_per_minute: int
failure_rate: float
local_audio: bool
flaky_connection: bool
voice: Optional[str]
speech_rate: int
speech_volume: int
speech_pitch: int
voice_engine: str
gemini_api_key: Optional[str]
gemini_model: str
gemini_voice: str
gemini_style: str
gemini_chunk_chars: int
pronunciation: bool
@dataclass(frozen=True)
class SpeechSettings:
max_length: int
min_length: int
allow_interrupt: bool
history_limit: int
@dataclass(frozen=True)
class Settings:
server: ServerSettings
robot: RobotSettings
a3: A3Settings
mock: MockSettings
speech: SpeechSettings
env_file: str
issues: List[ConfigIssue] = field(default_factory=list)
@property
def is_mock(self) -> bool:
return self.robot.mode == "mock"
@property
def has_blocking_issue(self) -> bool:
return any(issue.level == "error" for issue in self.issues)
def public_dict(self) -> Dict[str, Any]:
"""Safe-to-expose subset for the browser. Never leaks credentials."""
return {
"mode": self.robot.mode,
"robotName": self.robot.name,
"robotModel": self.robot.model,
"transport": self.a3.transport if self.robot.mode == "real" else "mock",
"address": self.robot.address,
"ipConfigured": bool(self.robot.ip),
"maxLength": self.speech.max_length,
"allowInterrupt": self.speech.allow_interrupt,
"historyLimit": self.speech.history_limit,
"healthInterval": self.robot.health_interval,
"envFile": self.env_file,
"issues": [issue.to_dict() for issue in self.issues],
}
# --------------------------------------------------------------------------- #
# validation
# --------------------------------------------------------------------------- #
def _looks_like_host(value: str) -> bool:
try:
ipaddress.ip_address(value)
return True
except ValueError:
pass
# Permissive hostname check - mDNS names such as "agibot-a3.local" are valid.
return bool(re.fullmatch(r"[A-Za-z0-9]([A-Za-z0-9\-._]{0,251}[A-Za-z0-9])?", value))
def _validate(robot: RobotSettings, a3: A3Settings, speech: SpeechSettings) -> List[ConfigIssue]:
issues: List[ConfigIssue] = []
if robot.mode not in ("mock", "real"):
issues.append(
ConfigIssue("error", "ROBOT_MODE", "Unknown ROBOT_MODE '%s'. Use 'mock' or 'real'." % robot.mode)
)
return issues
if robot.mode == "real":
if not robot.ip:
issues.append(
ConfigIssue(
"error",
"ROBOT_IP",
"ROBOT_MODE=real but ROBOT_IP is empty. Set the robot's LAN IP address in .env.",
)
)
elif not _looks_like_host(robot.ip):
issues.append(
ConfigIssue("error", "ROBOT_IP", "'%s' is not a valid IP address or hostname." % robot.ip)
)
if not 0 < robot.port < 65536:
issues.append(
ConfigIssue("error", "ROBOT_PORT", "ROBOT_PORT must be 1-65535, got %s." % robot.port)
)
if a3.transport not in TRANSPORTS:
issues.append(
ConfigIssue(
"error",
"A3_TRANSPORT",
"Unknown A3_TRANSPORT '%s'. Use one of: %s."
% (a3.transport, ", ".join(TRANSPORTS)),
)
)
elif a3.transport == "aimdk":
if not a3.aimdk_service or not a3.aimdk_play_method:
issues.append(
ConfigIssue(
"error",
"A3_AIMDK_SERVICE",
"AimDK transport needs a service name and a play method.",
)
)
if robot.port != 59301:
issues.append(
ConfigIssue(
"warning",
"ROBOT_PORT",
"AgiBot documents the A3 TTS RPC on port 59301; ROBOT_PORT is %s. "
"Confirm the port for your firmware." % robot.port,
)
)
elif a3.transport == "http" and not a3.http_speak_path:
issues.append(
ConfigIssue(
"error",
"A3_HTTP_SPEAK_PATH",
"HTTP transport selected but no speak endpoint is configured. "
"Fill it in from the robot's API reference - see docs/AGIBOT_A3_INTEGRATION.md.",
)
)
elif a3.transport == "ws" and not a3.ws_path:
issues.append(
ConfigIssue("error", "A3_WS_PATH", "WebSocket transport selected but A3_WS_PATH is empty.")
)
elif a3.transport == "ros2" and not (a3.ros_speak_topic or a3.ros_service_name):
issues.append(
ConfigIssue(
"error",
"A3_ROS_SPEAK_TOPIC",
"ROS 2 transport selected but neither a topic nor a service name is configured.",
)
)
elif a3.transport == "ssh":
if not a3.ssh_speak_command:
issues.append(
ConfigIssue(
"error",
"A3_SSH_SPEAK_COMMAND",
"SSH transport selected but A3_SSH_SPEAK_COMMAND is empty.",
)
)
if not a3.ssh_password and not a3.ssh_key_path:
issues.append(
ConfigIssue(
"warning",
"A3_SSH_PASSWORD",
"No SSH password or key configured; authentication will rely on an ssh agent.",
)
)
if speech.max_length < 1:
issues.append(ConfigIssue("error", "SPEECH_MAX_LENGTH", "SPEECH_MAX_LENGTH must be >= 1."))
return issues
# --------------------------------------------------------------------------- #
# loading
# --------------------------------------------------------------------------- #
def load_settings(env_file: Optional[Path] = None, override: bool = True) -> Settings:
"""Read .env + process environment into an immutable Settings object."""
path = Path(env_file) if env_file else ENV_FILE
if path.exists():
load_dotenv(path, override=override)
server = ServerSettings(
host=env_str("HOST", "127.0.0.1"),
port=env_int("PORT", 8000),
log_level=env_str("LOG_LEVEL", "info").lower(),
cors_origins=env_list("CORS_ORIGINS", []),
)
robot = RobotSettings(
mode=env_str("ROBOT_MODE", "mock").lower(), # type: ignore[arg-type]
name=env_str("ROBOT_NAME", "AGIBOT A3"),
model=env_str("ROBOT_MODEL", "AgiBot A3"),
ip=env_opt("ROBOT_IP"),
port=env_int("ROBOT_PORT", 59301),
use_tls=env_bool("ROBOT_USE_TLS", False),
connect_timeout=env_float("ROBOT_CONNECT_TIMEOUT", 3.0),
request_timeout=env_float("ROBOT_REQUEST_TIMEOUT", 8.0),
health_interval=env_float("ROBOT_HEALTH_INTERVAL", 5.0),
reconnect_min_delay=env_float("ROBOT_RECONNECT_MIN_DELAY", 1.0),
reconnect_max_delay=env_float("ROBOT_RECONNECT_MAX_DELAY", 15.0),
)
a3 = A3Settings(
transport=env_str("A3_TRANSPORT", "aimdk").lower(), # type: ignore[arg-type]
aimdk_service=env_str("A3_AIMDK_SERVICE", "aimdk.protocol.TTSService"),
aimdk_play_method=env_str("A3_AIMDK_PLAY_METHOD", "PlayTTS"),
aimdk_stop_method=env_str("A3_AIMDK_STOP_METHOD", "StopTTSTraceId"),
aimdk_status_method=env_str("A3_AIMDK_STATUS_METHOD", "GetAudioStatus"),
aimdk_priority=env_str("A3_AIMDK_PRIORITY", "INTERACTION_L6"),
aimdk_domain=env_str("A3_AIMDK_DOMAIN", "voice_control"),
aimdk_interrupt=env_bool("A3_AIMDK_INTERRUPT", True),
aimdk_max_bytes=env_int("A3_AIMDK_MAX_BYTES", 1024),
http_speak_path=env_str("A3_HTTP_SPEAK_PATH", ""),
http_speak_method=env_str("A3_HTTP_SPEAK_METHOD", "POST").upper(),
http_speak_payload=env_json("A3_HTTP_SPEAK_PAYLOAD", {"text": "{text}"}),
http_stop_path=env_str("A3_HTTP_STOP_PATH", ""),
http_stop_method=env_str("A3_HTTP_STOP_METHOD", "POST").upper(),
http_stop_payload=env_json("A3_HTTP_STOP_PAYLOAD", {}),
http_status_path=env_str("A3_HTTP_STATUS_PATH", ""),
http_status_method=env_str("A3_HTTP_STATUS_METHOD", "GET").upper(),
http_headers=env_json("A3_HTTP_HEADERS", {}),
http_auth_token=env_opt("A3_HTTP_AUTH_TOKEN"),
http_success_field=env_opt("A3_HTTP_SUCCESS_FIELD"),
ws_path=env_str("A3_WS_PATH", ""),
ws_speak_payload=env_json("A3_WS_SPEAK_PAYLOAD", {"text": "{text}"}),
ws_stop_payload=env_json("A3_WS_STOP_PAYLOAD", {}),
ws_ping_interval=env_float("A3_WS_PING_INTERVAL", 20.0),
ws_done_field=env_opt("A3_WS_DONE_FIELD"),
ws_done_value=env_opt("A3_WS_DONE_VALUE"),
ros_domain_id=env_int("A3_ROS_DOMAIN_ID", 0),
ros_speak_topic=env_str("A3_ROS_SPEAK_TOPIC", ""),
ros_speak_msg_type=env_str("A3_ROS_SPEAK_MSG_TYPE", "std_msgs/msg/String"),
ros_speak_msg_field=env_str("A3_ROS_SPEAK_MSG_FIELD", "data"),
ros_stop_topic=env_str("A3_ROS_STOP_TOPIC", ""),
ros_use_service=env_bool("A3_ROS_USE_SERVICE", False),
ros_service_name=env_str("A3_ROS_SERVICE_NAME", ""),
ros_service_type=env_str("A3_ROS_SERVICE_TYPE", ""),
ssh_user=env_str("A3_SSH_USER", "root"),
ssh_password=env_opt("A3_SSH_PASSWORD"),
ssh_key_path=env_opt("A3_SSH_KEY_PATH"),
ssh_port=env_int("A3_SSH_PORT", 22),
ssh_speak_command=env_str("A3_SSH_SPEAK_COMMAND", ""),
ssh_stop_command=env_str("A3_SSH_STOP_COMMAND", ""),
ssh_probe_command=env_str("A3_SSH_PROBE_COMMAND", "true"),
voice=env_opt("A3_VOICE"),
language=env_opt("A3_LANGUAGE"),
volume=env_float("A3_VOLUME", 1.0) if env_opt("A3_VOLUME") else None,
speed=env_float("A3_SPEED", 1.0) if env_opt("A3_SPEED") else None,
)
mock = MockSettings(
connect_delay_ms=env_int("MOCK_CONNECT_DELAY_MS", 350),
network_latency_ms=env_int("MOCK_NETWORK_LATENCY_MS", 45),
processing_ms=env_int("MOCK_PROCESSING_MS", 180),
words_per_minute=env_int("MOCK_WORDS_PER_MINUTE", 150),
failure_rate=env_float("MOCK_FAILURE_RATE", 0.0),
local_audio=env_bool("MOCK_LOCAL_AUDIO", False),
flaky_connection=env_bool("MOCK_FLAKY_CONNECTION", False),
voice=env_opt("MOCK_VOICE"),
speech_rate=env_int("MOCK_SPEECH_RATE", 0),
speech_volume=env_int("MOCK_SPEECH_VOLUME", 100),
speech_pitch=env_int("MOCK_SPEECH_PITCH", 0),
voice_engine=env_str("MOCK_VOICE_ENGINE", "system").lower(),
gemini_api_key=env_opt("GEMINI_API_KEY"),
gemini_model=env_str("GEMINI_TTS_MODEL", "gemini-3.1-flash-tts-preview"),
gemini_voice=env_str("GEMINI_VOICE", "Puck"),
gemini_style=env_str("GEMINI_TTS_STYLE", ""),
gemini_chunk_chars=env_int("GEMINI_CHUNK_CHARS", 0),
pronunciation=env_bool("SPEECH_PRONUNCIATION", True),
)
speech = SpeechSettings(
max_length=env_int("SPEECH_MAX_LENGTH", 1000),
min_length=env_int("SPEECH_MIN_LENGTH", 1),
allow_interrupt=env_bool("SPEECH_ALLOW_INTERRUPT", True),
history_limit=env_int("SPEECH_HISTORY_LIMIT", 100),
)
return Settings(
server=server,
robot=robot,
a3=a3,
mock=mock,
speech=speech,
env_file=str(path),
issues=_validate(robot, a3, speech),
)
_settings: Optional[Settings] = None
def get_settings() -> Settings:
global _settings
if _settings is None:
_settings = load_settings()
return _settings
def reload_settings() -> Settings:
"""Re-read .env from disk (used by the /api/config/reload endpoint)."""
global _settings
_settings = load_settings(override=True)
return _settings

0
backend/core/__init__.py Normal file
View File

125
backend/core/events.py Normal file
View File

@ -0,0 +1,125 @@
"""A tiny in-process publish/subscribe bus.
The robot layer publishes; the WebSocket layer subscribes and forwards to the
browser. Keeping this in the middle means the robot adapters never know that a
browser exists, and the API layer never polls the robot.
Delivery is best-effort and non-blocking: a slow or wedged browser tab can never
stall the robot loop. If a subscriber's queue overflows we drop its oldest event
and mark the stream as lossy rather than applying backpressure upstream.
"""
from __future__ import annotations
import asyncio
import logging
import time
from dataclasses import dataclass, field
from typing import Any, Dict, List, Optional
logger = logging.getLogger(__name__)
MAX_QUEUE = 256
class EventType(str):
"""String constants - kept as plain strings so they serialise for free."""
ROBOT_STATUS = "robot.status"
SPEECH_PROGRESS = "speech.progress"
SPEECH_RESULT = "speech.result"
HISTORY_UPDATED = "history.updated"
CONFIG_UPDATED = "config.updated"
NOTICE = "notice"
@dataclass
class Event:
type: str
data: Dict[str, Any] = field(default_factory=dict)
at: float = field(default_factory=time.time)
def to_dict(self) -> Dict[str, Any]:
return {"type": self.type, "data": self.data, "at": self.at}
class Subscription:
"""An async iterator over bus events, scoped to one subscriber."""
def __init__(self, bus: "EventBus", name: str) -> None:
self._bus = bus
self._queue: asyncio.Queue = asyncio.Queue(maxsize=MAX_QUEUE)
self.name = name
self.dropped = 0
def _offer(self, event: Event) -> None:
try:
self._queue.put_nowait(event)
except asyncio.QueueFull:
self.dropped += 1
try: # make room, keep the newest - stale status is worse than none
self._queue.get_nowait()
self._queue.put_nowait(event)
except (asyncio.QueueEmpty, asyncio.QueueFull): # pragma: no cover
pass
async def get(self, timeout: Optional[float] = None) -> Optional[Event]:
if timeout is None:
return await self._queue.get()
try:
return await asyncio.wait_for(self._queue.get(), timeout=timeout)
except asyncio.TimeoutError:
return None
def __enter__(self) -> "Subscription":
return self
def __exit__(self, *exc: Any) -> None:
self.close()
def close(self) -> None:
self._bus.unsubscribe(self)
class EventBus:
def __init__(self) -> None:
self._subscribers: List[Subscription] = []
self._last: Dict[str, Event] = {}
# -- subscription -------------------------------------------------------- #
def subscribe(self, name: str = "anonymous") -> Subscription:
sub = Subscription(self, name)
self._subscribers.append(sub)
logger.debug("event subscriber added: %s (total=%d)", name, len(self._subscribers))
return sub
def unsubscribe(self, sub: Subscription) -> None:
if sub in self._subscribers:
self._subscribers.remove(sub)
logger.debug("event subscriber removed: %s (total=%d)", sub.name, len(self._subscribers))
@property
def subscriber_count(self) -> int:
return len(self._subscribers)
# -- publishing ---------------------------------------------------------- #
def publish(self, event_type: str, data: Optional[Dict[str, Any]] = None) -> Event:
event = Event(type=event_type, data=data or {})
self._last[event_type] = event
for sub in list(self._subscribers):
sub._offer(event)
return event
def last(self, event_type: str) -> Optional[Event]:
"""Most recent event of a type - used to prime a newly-opened socket."""
return self._last.get(event_type)
_bus: Optional[EventBus] = None
def get_event_bus() -> EventBus:
global _bus
if _bus is None:
_bus = EventBus()
return _bus

43
backend/core/logging.py Normal file
View File

@ -0,0 +1,43 @@
"""Console logging setup."""
from __future__ import annotations
import logging
import sys
_FORMAT = "%(asctime)s %(levelname)-7s %(name)-28s %(message)s"
_DATEFMT = "%H:%M:%S"
def ensure_utf8_stdout() -> None:
"""Make stdout/stderr able to print non-ASCII on a Windows console.
Windows terminals default to a legacy code page (cp1252 in Western locales),
and printing a single Chinese character raises UnicodeEncodeError. The robot's
own documentation and error strings are Chinese, so an error response would
otherwise crash the log call that was trying to report it - turning a handled
failure into an unhandled one.
"""
for stream in (sys.stdout, sys.stderr):
try:
if stream is not None and hasattr(stream, "reconfigure"):
stream.reconfigure(encoding="utf-8", errors="replace") # type: ignore[union-attr]
except Exception: # pragma: no cover - never let logging setup fail
pass
def configure_logging(level: str = "info") -> None:
ensure_utf8_stdout()
numeric = getattr(logging, level.upper(), logging.INFO)
handler = logging.StreamHandler(sys.stdout)
handler.setFormatter(logging.Formatter(_FORMAT, datefmt=_DATEFMT))
root = logging.getLogger()
root.handlers = [handler]
root.setLevel(numeric)
# Access logs for a single-user local dashboard are noise.
logging.getLogger("uvicorn.access").setLevel(logging.WARNING)
logging.getLogger("httpx").setLevel(logging.WARNING)
logging.getLogger("httpcore").setLevel(logging.WARNING)
logging.getLogger("websockets").setLevel(logging.WARNING)

View File

@ -0,0 +1,112 @@
"""Pronunciation fixes for the simulator's speech.
The built-in OS voices are clear and instant, but they mispronounce brand names
and initialisms - "AGIBOT" is the obvious one. Rather than abandoning a voice
that otherwise sounds good, we respell the problem words phonetically just
before they are spoken.
The text the *robot* receives is never touched. This rewrite happens at the very
last step, inside the PC audio layer, so:
* the dashboard, history and API all keep the operator's original wording;
* the real AGIBOT A3 always gets the real text - it has its own TTS and its
own idea of how its name sounds.
Rules live in `pronunciation.json` at the project root so they can be corrected
without touching code:
{
"AGIBOT": "Ah-jee-bot",
"A3": "A three"
}
Matching is case-insensitive and respects word boundaries, so "AGIBOT" will not
corrupt a longer word that happens to contain it.
"""
from __future__ import annotations
import json
import logging
import re
from pathlib import Path
from typing import Dict, List, Optional, Tuple
logger = logging.getLogger(__name__)
RULES_FILE = "pronunciation.json"
#: Shipped defaults. Deliberately small - guessing at pronunciations nobody
#: asked for is worse than leaving a word alone. Extend via pronunciation.json.
DEFAULT_RULES: Dict[str, str] = {
"AGIBOT": "Ah-jee-bot",
"AgiBot": "Ah-jee-bot",
}
class Pronouncer:
"""Applies respelling rules to text just before synthesis."""
def __init__(self, rules: Optional[Dict[str, str]] = None) -> None:
self._compiled: List[Tuple[re.Pattern, str]] = []
self.rules: Dict[str, str] = {}
self.load(rules if rules is not None else DEFAULT_RULES)
def load(self, rules: Dict[str, str]) -> None:
self.rules = dict(rules or {})
self._compiled = []
# Longest first, so a specific phrase wins over a word inside it.
for phrase in sorted(self.rules, key=len, reverse=True):
replacement = self.rules[phrase]
if not phrase:
continue
# \b does not anchor against digits/symbols the way we need for
# names like "A3", so use explicit look-around on word characters.
pattern = re.compile(
r"(?<![0-9A-Za-z]){0}(?![0-9A-Za-z])".format(re.escape(phrase)),
re.IGNORECASE,
)
self._compiled.append((pattern, replacement))
def apply(self, text: str) -> str:
if not text or not self._compiled:
return text
result = text
for pattern, replacement in self._compiled:
result = pattern.sub(replacement, result)
return result
def __len__(self) -> int:
return len(self.rules)
def load_rules(project_root: Path) -> Dict[str, str]:
"""Read pronunciation.json, falling back to the shipped defaults."""
path = Path(project_root) / RULES_FILE
if not path.exists():
return dict(DEFAULT_RULES)
try:
data = json.loads(path.read_text(encoding="utf-8"))
except (json.JSONDecodeError, OSError) as exc:
logger.warning("%s is not valid JSON (%s); using default pronunciations", path, exc)
return dict(DEFAULT_RULES)
if not isinstance(data, dict):
logger.warning("%s must contain a JSON object of {word: respelling}", path)
return dict(DEFAULT_RULES)
# Keys starting with "_" are comments, so the file can document itself.
rules = {
str(k): str(v)
for k, v in data.items()
if str(k).strip() and not str(k).startswith("_")
}
logger.info("loaded %d pronunciation rule(s) from %s", len(rules), path.name)
return rules
def build(project_root: Path, enabled: bool = True) -> Optional[Pronouncer]:
"""A Pronouncer, or None when the feature is switched off."""
if not enabled:
return None
return Pronouncer(load_rules(project_root))

39
backend/core/text.py Normal file
View File

@ -0,0 +1,39 @@
"""Text helpers shared by the adapters."""
from __future__ import annotations
import re
_WHITESPACE = re.compile(r"\s+")
def normalise(text: str) -> str:
"""Collapse whitespace and trim. Keeps the robot from reading stray newlines."""
return _WHITESPACE.sub(" ", text or "").strip()
def is_cjk(char: str) -> bool:
code = ord(char)
return (
0x4E00 <= code <= 0x9FFF # CJK unified ideographs
or 0x3400 <= code <= 0x4DBF # extension A
or 0x3040 <= code <= 0x30FF # kana
or 0xAC00 <= code <= 0xD7AF # hangul
)
def estimate_speech_seconds(text: str, words_per_minute: int = 150) -> float:
"""Rough spoken duration.
Used only when the robot does not report utterance completion itself, so the
dashboard can still return to "Ready" at a believable moment instead of
hanging on "Speaking..." forever.
"""
cleaned = normalise(text)
if not cleaned:
return 0.0
cjk_chars = sum(1 for ch in cleaned if is_cjk(ch))
latin_words = len([w for w in cleaned.split() if any(not is_cjk(c) for c in w)])
# ~2.5 CJK characters per second is a typical TTS cadence.
seconds = (latin_words / max(60, words_per_minute)) * 60.0 + (cjk_chars / 2.5)
return max(0.8, seconds)

175
backend/main.py Normal file
View File

@ -0,0 +1,175 @@
"""Application entry point.
python backend/main.py -> http://localhost:8000
Serves the dashboard, the JSON API and the WebSocket from one process, so there
is nothing to orchestrate on the demo machine.
"""
from __future__ import annotations
import logging
import os
import sys
from contextlib import asynccontextmanager
from pathlib import Path
# Allow `python backend/main.py` as well as `python -m backend.main`.
if __package__ in (None, ""): # pragma: no cover
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from fastapi import FastAPI, Request # noqa: E402
from fastapi.middleware.cors import CORSMiddleware # noqa: E402
from fastapi.responses import JSONResponse # noqa: E402
from fastapi.staticfiles import StaticFiles # noqa: E402
from backend.api.routes import router as api_router # noqa: E402
from backend.api.websocket import ws_router # noqa: E402
from backend.config.settings import PROJECT_ROOT, get_settings # noqa: E402
from backend.core.events import get_event_bus # noqa: E402
from backend.core.logging import configure_logging # noqa: E402
from backend.robot.base import ( # noqa: E402
RobotBusy,
RobotError,
RobotNotConfigured,
RobotTimeout,
RobotUnreachable,
SpeechFailed,
TransportNotAvailable,
)
from backend.robot.manager import RobotManager # noqa: E402
from backend.services.speech_service import SpeechService, ValidationError # noqa: E402
logger = logging.getLogger("agibot.app")
FRONTEND_DIR = PROJECT_ROOT / "frontend"
# HTTP status for each robot-layer failure. Anything unmapped becomes 500.
STATUS_FOR_ERROR = {
ValidationError: 400,
RobotBusy: 409,
TransportNotAvailable: 501,
SpeechFailed: 502,
RobotNotConfigured: 503,
RobotUnreachable: 503,
RobotTimeout: 504,
}
def _status_for(exc: RobotError) -> int:
for error_type, status in STATUS_FOR_ERROR.items():
if isinstance(exc, error_type):
return status
return 500
@asynccontextmanager
async def lifespan(app: FastAPI):
settings = app.state.settings
manager: RobotManager = app.state.robot_manager
banner(settings)
await manager.start()
try:
yield
finally:
await app.state.speech_service.shutdown()
await manager.stop()
logger.info("shutdown complete")
def create_app() -> FastAPI:
settings = get_settings()
configure_logging(settings.server.log_level)
app = FastAPI(
title="AGIBOT A3 Voice Control",
description="Local dashboard for sending speech to an AGIBOT A3 humanoid.",
version="1.0.0",
lifespan=lifespan,
docs_url="/api/docs",
openapi_url="/api/openapi.json",
)
bus = get_event_bus()
manager = RobotManager(settings, bus)
app.state.settings = settings
app.state.event_bus = bus
app.state.robot_manager = manager
app.state.speech_service = SpeechService(settings, manager, bus)
if settings.server.cors_origins:
app.add_middleware(
CORSMiddleware,
allow_origins=settings.server.cors_origins,
allow_methods=["*"],
allow_headers=["*"],
)
@app.exception_handler(RobotError)
async def robot_error_handler(request: Request, exc: RobotError) -> JSONResponse:
status = _status_for(exc)
if status >= 500:
logger.warning("%s -> %s (%s)", request.url.path, exc.code, exc)
return JSONResponse(
status_code=status,
content={"success": False, "error": exc.user_message, "errorCode": exc.code},
)
app.include_router(api_router)
app.include_router(ws_router)
if FRONTEND_DIR.is_dir():
# Mounted last so /api/* and /ws win.
app.mount("/", StaticFiles(directory=str(FRONTEND_DIR), html=True), name="frontend")
else: # pragma: no cover
logger.error("frontend directory missing: %s", FRONTEND_DIR)
return app
def banner(settings) -> None:
mode = settings.robot.mode.upper()
target = "local simulation" if settings.is_mock else "{0} via {1}".format(
settings.robot.address, settings.a3.transport
)
url = "http://{0}:{1}".format(
"localhost" if settings.server.host in ("0.0.0.0", "127.0.0.1") else settings.server.host,
settings.server.port,
)
lines = [
"",
" AGIBOT A3 - Voice Control",
" " + "-" * 46,
" Mode : {0}".format(mode),
" Robot : {0}".format(target),
" Dashboard : {0}".format(url),
" Config : {0}".format(settings.env_file),
]
for issue in settings.issues:
lines.append(" {0:<10}: [{1}] {2}".format("Config", issue.level.upper(), issue.message))
if settings.is_mock:
lines.append(" Note : running against the MOCK robot - no hardware needed.")
lines.append("")
print("\n".join(lines))
app = create_app()
def main() -> None:
import uvicorn
settings = get_settings()
uvicorn.run(
"backend.main:app",
host=settings.server.host,
port=settings.server.port,
reload=os.environ.get("DEV_RELOAD", "").lower() in ("1", "true", "yes"),
log_level=settings.server.log_level,
access_log=False,
)
if __name__ == "__main__":
main()

View File

299
backend/robot/agibot_a3.py Normal file
View File

@ -0,0 +1,299 @@
"""AGIBOT A3 adapter.
============================================================================
THIS FILE IS THE INTEGRATION POINT.
============================================================================
Everything else in this project is finished and testable today. When the robot's
speech interface is confirmed, the change is limited to:
1. .env - transport, address, endpoint/topic/command and payload
2. (only if the wire format is unusual) a new file under robot/transports/
No endpoint, port, ROS topic or SDK symbol is invented in this code. The adapter
deliberately holds *no* opinion about the A3's protocol; it holds the parts that
are true regardless of protocol:
- one persistent connection, reused across utterances (latency)
- ack-latency measurement, so "how fast is it really" is answerable
- the SENDING -> PROCESSING -> SPEAKING -> COMPLETED lifecycle the UI renders
- completion by robot event when the robot reports it, by estimate when it does not
- interruption, timeouts and error mapping that never wedge the web app
See docs/AGIBOT_A3_INTEGRATION.md for the discovery procedure to run against the
real unit, and for the exact questions to put to AgiBot support.
"""
from __future__ import annotations
import asyncio
import logging
import time
from typing import Any, Dict, Optional
from ..config.settings import Settings
from ..core.text import estimate_speech_seconds
from .base import (
ProgressCallback,
RobotAdapter,
RobotCapabilities,
RobotError,
RobotInfo,
RobotNotConfigured,
RobotUnreachable,
SpeechProgress,
SpeechRequest,
SpeechResult,
SpeechStage,
)
from .transports import create_transport
from .transports.base import SpeechTransport
logger = logging.getLogger(__name__)
_TICK = 0.05
class AgibotA3(RobotAdapter):
"""Real-robot adapter, delegating the wire protocol to a configured transport."""
def __init__(self, settings: Settings) -> None:
self._settings = settings
self._robot = settings.robot
self._a3 = settings.a3
self._transport: Optional[SpeechTransport] = None
self._connected = False
self._speaking = False
self._cancel = asyncio.Event()
self._last_error: Optional[str] = None
# -- identity ------------------------------------------------------------ #
@property
def info(self) -> RobotInfo:
transport = self._transport
return RobotInfo(
name=self._robot.name,
model=self._robot.model,
mode="real",
transport=self._a3.transport,
address=self._robot.address,
capabilities=RobotCapabilities(
native_tts=True,
stop=self._can_stop(),
progress_events=True,
reports_completion=bool(transport and transport.reports_completion),
voice_selection=bool(self._a3.voice),
volume_control=self._a3.volume is not None,
),
)
def _can_stop(self) -> bool:
"""Only advertise Stop when a stop path is actually configured."""
transport = self._a3.transport
if transport == "aimdk":
return bool(self._a3.aimdk_stop_method)
if transport == "http":
return bool(self._a3.http_stop_path)
if transport == "ws":
return bool(self._a3.ws_stop_payload)
if transport == "ros2":
return bool(self._a3.ros_stop_topic)
if transport == "ssh":
return bool(self._a3.ssh_stop_command)
return False
@property
def is_connected(self) -> bool:
return self._connected
@property
def is_speaking(self) -> bool:
return self._speaking
@property
def last_error(self) -> Optional[str]:
return self._last_error
# -- lifecycle ----------------------------------------------------------- #
async def connect(self) -> None:
if not self._robot.ip:
raise RobotNotConfigured()
if self._settings.has_blocking_issue:
first = next(i for i in self._settings.issues if i.level == "error")
raise RobotError(first.message, user_message=first.message)
if self._transport is None:
self._transport = create_transport(self._settings)
try:
await self._transport.open()
ok = await self._transport.probe()
except RobotError:
self._connected = False
raise
except Exception as exc:
self._connected = False
raise RobotUnreachable(
"transport {0} failed to open: {1}".format(self._a3.transport, exc)
) from exc
if not ok:
self._connected = False
raise RobotUnreachable(
"probe against {0} failed".format(self._robot.address)
)
self._connected = True
self._last_error = None
logger.info(
"AGIBOT A3 connected via %s at %s", self._a3.transport, self._robot.address
)
async def disconnect(self) -> None:
self._cancel.set()
self._connected = False
self._speaking = False
if self._transport is not None:
try:
await self._transport.close()
except Exception: # pragma: no cover
logger.debug("transport close raised", exc_info=True)
async def health_check(self) -> bool:
if self._transport is None:
return False
try:
ok = await self._transport.probe()
except Exception as exc:
logger.debug("health check raised: %s", exc)
ok = False
self._connected = ok
return ok
# -- speech -------------------------------------------------------------- #
async def speak(self, request: SpeechRequest, on_progress: ProgressCallback) -> SpeechResult:
if self._transport is None or not self._connected:
raise RobotUnreachable("robot is not connected")
self._cancel.clear()
self._speaking = True
started = time.perf_counter()
try:
ack = await self._transport.speak(request, on_progress)
ack_ms = ack.ack_latency_ms
await on_progress(
SpeechProgress(
request.id,
SpeechStage.PROCESSING,
"Robot accepted the request",
detail=ack.detail,
elapsed_ms=ack_ms,
)
)
estimated = estimate_speech_seconds(request.text)
await on_progress(
SpeechProgress(
request.id,
SpeechStage.SPEAKING,
"Speaking...",
detail=None if self._reports_completion else "~{0:.1f}s".format(estimated),
elapsed_ms=int((time.perf_counter() - started) * 1000),
)
)
cancelled = await self._await_end(estimated)
total_ms = int((time.perf_counter() - started) * 1000)
if cancelled:
await on_progress(
SpeechProgress(request.id, SpeechStage.CANCELLED, "Speech stopped.")
)
return SpeechResult(
request_id=request.id,
success=False,
stage=SpeechStage.CANCELLED,
ack_latency_ms=ack_ms,
total_ms=total_ms,
error_code="cancelled",
error="Stopped by operator.",
)
await on_progress(
SpeechProgress(request.id, SpeechStage.COMPLETED, "Ready", elapsed_ms=total_ms)
)
return SpeechResult(
request_id=request.id,
success=True,
stage=SpeechStage.COMPLETED,
ack_latency_ms=ack_ms,
total_ms=total_ms,
)
except RobotError as exc:
self._last_error = str(exc)
raise
finally:
self._speaking = False
@property
def _reports_completion(self) -> bool:
return bool(self._transport and self._transport.reports_completion)
async def _await_end(self, estimated_seconds: float) -> bool:
"""Wait for the utterance to finish. Returns True if it was cancelled.
If the transport delivers a real completion event we use it; otherwise we
fall back to the estimated duration so the UI still returns to Ready.
"""
transport = self._transport
if self._reports_completion and hasattr(transport, "await_completion"):
budget = max(estimated_seconds * 3, 10.0)
done_task = asyncio.ensure_future(transport.await_completion(budget)) # type: ignore[union-attr]
cancel_task = asyncio.ensure_future(self._cancel.wait())
try:
await asyncio.wait(
{done_task, cancel_task}, return_when=asyncio.FIRST_COMPLETED
)
return cancel_task.done() and not done_task.done()
finally:
for task in (done_task, cancel_task):
if not task.done():
task.cancel()
if self._reports_completion:
# e.g. SSH: transport.speak() already blocked until playback ended.
return self._cancel.is_set()
deadline = time.perf_counter() + estimated_seconds
while True:
remaining = deadline - time.perf_counter()
if remaining <= 0:
return False
if self._cancel.is_set():
return True
await asyncio.sleep(min(_TICK, remaining))
async def stop_speaking(self) -> bool:
self._cancel.set()
if self._transport is None:
return False
try:
return await self._transport.stop()
except Exception as exc:
logger.warning("stop failed: %s", exc)
return False
async def describe(self) -> Dict[str, Any]:
detail: Dict[str, Any] = {
"transport": self._a3.transport,
"address": self._robot.address,
"connected": self._connected,
"lastError": self._last_error,
}
if self._transport is not None:
try:
detail["transportDetail"] = await self._transport.describe()
except Exception: # pragma: no cover
pass
return detail

242
backend/robot/base.py Normal file
View File

@ -0,0 +1,242 @@
"""Robot abstraction layer.
This is the ONLY contract the rest of the application knows about. Nothing above
this layer (services, API, frontend) may import a vendor SDK, an HTTP client or a
ROS package. Adding a new robot means adding one file that implements
`RobotAdapter` and registering it in `factory.py`.
"""
from __future__ import annotations
import abc
import time
import uuid
from dataclasses import dataclass, field
from enum import Enum
from typing import Any, Awaitable, Callable, Dict, Optional
# --------------------------------------------------------------------------- #
# enums
# --------------------------------------------------------------------------- #
class RobotState(str, Enum):
DISCONNECTED = "disconnected"
CONNECTING = "connecting"
CONNECTED = "connected"
ERROR = "error"
class SpeechStage(str, Enum):
"""Lifecycle of a single utterance, mirrored 1:1 in the UI."""
QUEUED = "queued"
SENDING = "sending" # request leaving the PC
PROCESSING = "processing" # robot accepted it, synthesising
SPEAKING = "speaking" # audio is coming out of the speaker
COMPLETED = "completed"
CANCELLED = "cancelled"
FAILED = "failed"
TERMINAL_STAGES = {SpeechStage.COMPLETED, SpeechStage.CANCELLED, SpeechStage.FAILED}
# --------------------------------------------------------------------------- #
# errors
# --------------------------------------------------------------------------- #
class RobotError(Exception):
"""Base class for every robot-layer failure.
`user_message` is safe to show in the browser; `str(exc)` may contain detail
useful in the server log but noisy for an operator during a live demo.
"""
code = "robot_error"
user_message = "The robot reported an error."
def __init__(self, message: str = "", user_message: Optional[str] = None) -> None:
super().__init__(message or self.user_message)
if user_message:
self.user_message = user_message
class RobotNotConfigured(RobotError):
code = "not_configured"
user_message = "Robot is not configured. Set ROBOT_IP in the .env file."
class RobotUnreachable(RobotError):
code = "unreachable"
user_message = "Robot is offline. Check the robot IP address and network connection."
class RobotTimeout(RobotError):
code = "timeout"
user_message = "The robot did not respond in time."
class RobotBusy(RobotError):
code = "busy"
user_message = "The robot is already speaking."
class SpeechFailed(RobotError):
code = "speech_failed"
user_message = "Speech request failed."
class TransportNotAvailable(RobotError):
code = "transport_unavailable"
user_message = "The selected robot transport is not available on this PC."
# --------------------------------------------------------------------------- #
# value objects
# --------------------------------------------------------------------------- #
@dataclass(frozen=True)
class RobotCapabilities:
"""What this adapter can actually do, so the UI never offers a dead control."""
native_tts: bool = True
stop: bool = True
progress_events: bool = True
reports_completion: bool = True
voice_selection: bool = False
volume_control: bool = False
@dataclass(frozen=True)
class RobotInfo:
name: str
model: str
mode: str
transport: str
address: str
capabilities: RobotCapabilities
def to_dict(self) -> Dict[str, Any]:
return {
"name": self.name,
"model": self.model,
"mode": self.mode,
"transport": self.transport,
"address": self.address,
"capabilities": {
"nativeTts": self.capabilities.native_tts,
"stop": self.capabilities.stop,
"progressEvents": self.capabilities.progress_events,
"reportsCompletion": self.capabilities.reports_completion,
"voiceSelection": self.capabilities.voice_selection,
"volumeControl": self.capabilities.volume_control,
},
}
@dataclass
class SpeechRequest:
text: str
id: str = field(default_factory=lambda: uuid.uuid4().hex[:12])
voice: Optional[str] = None
language: Optional[str] = None
created_at: float = field(default_factory=time.time)
@property
def preview(self) -> str:
text = " ".join(self.text.split())
return text if len(text) <= 80 else text[:77] + "..."
@dataclass
class SpeechProgress:
"""One lifecycle update for an utterance, pushed straight to the browser."""
request_id: str
stage: SpeechStage
message: str = ""
detail: Optional[str] = None
error_code: Optional[str] = None
elapsed_ms: Optional[int] = None
at: float = field(default_factory=time.time)
def to_dict(self) -> Dict[str, Any]:
return {
"requestId": self.request_id,
"stage": self.stage.value,
"message": self.message,
"detail": self.detail,
"errorCode": self.error_code,
"elapsedMs": self.elapsed_ms,
"at": self.at,
}
@dataclass
class SpeechResult:
request_id: str
success: bool
stage: SpeechStage
# Time from "user pressed Speak" to the robot acknowledging the request. This
# is the number that matters for a live demo - it is the perceived delay.
ack_latency_ms: Optional[int] = None
total_ms: Optional[int] = None
error_code: Optional[str] = None
error: Optional[str] = None
def to_dict(self) -> Dict[str, Any]:
return {
"requestId": self.request_id,
"success": self.success,
"stage": self.stage.value,
"ackLatencyMs": self.ack_latency_ms,
"totalMs": self.total_ms,
"errorCode": self.error_code,
"error": self.error,
}
ProgressCallback = Callable[[SpeechProgress], Awaitable[None]]
# --------------------------------------------------------------------------- #
# adapter contract
# --------------------------------------------------------------------------- #
class RobotAdapter(abc.ABC):
"""Every robot backend implements exactly this."""
@property
@abc.abstractmethod
def info(self) -> RobotInfo:
"""Static description of the robot / connection."""
@abc.abstractmethod
async def connect(self) -> None:
"""Establish (or verify) the connection. Raises RobotError on failure."""
@abc.abstractmethod
async def disconnect(self) -> None:
"""Release sockets/clients. Must never raise."""
@abc.abstractmethod
async def health_check(self) -> bool:
"""Cheap liveness probe used by the connection manager."""
@abc.abstractmethod
async def speak(self, request: SpeechRequest, on_progress: ProgressCallback) -> SpeechResult:
"""Send text to the robot and report progress until the utterance ends."""
@abc.abstractmethod
async def stop_speaking(self) -> bool:
"""Interrupt the current utterance. Returns True if a stop was issued."""
@property
@abc.abstractmethod
def is_connected(self) -> bool:
"""Last known connection state - must not perform I/O."""
@property
def is_speaking(self) -> bool:
return False
async def describe(self) -> Dict[str, Any]:
"""Optional richer diagnostics for the /api/robot/diagnostics endpoint."""
return {}

46
backend/robot/factory.py Normal file
View File

@ -0,0 +1,46 @@
"""Robot adapter registry.
Adding a robot (AgiBot X2, another vendor, a different TTS backend) means writing
one adapter and adding one line here. Nothing above this layer changes.
"""
from __future__ import annotations
from typing import Callable, Dict
from ..config.settings import Settings
from .base import RobotAdapter, RobotError
def _mock(settings: Settings) -> RobotAdapter:
from .mock_robot import MockRobot
return MockRobot(settings)
def _agibot_a3(settings: Settings) -> RobotAdapter:
from .agibot_a3 import AgibotA3
return AgibotA3(settings)
_REGISTRY: Dict[str, Callable[[Settings], RobotAdapter]] = {
"mock": _mock,
"real": _agibot_a3,
}
def available_modes() -> list:
return sorted(_REGISTRY)
def create_robot(settings: Settings) -> RobotAdapter:
builder = _REGISTRY.get(settings.robot.mode)
if builder is None:
raise RobotError(
"unknown ROBOT_MODE '{0}'".format(settings.robot.mode),
user_message="Unknown ROBOT_MODE '{0}'. Valid values: {1}.".format(
settings.robot.mode, ", ".join(available_modes())
),
)
return builder(settings)

View File

@ -0,0 +1,433 @@
"""Gemini cloud voice for the simulator.
Gives the mock robot a natural neural voice instead of the flat built-in Windows
one, so a rehearsal sounds like the real thing.
**Simulator only.** The real AGIBOT A3 synthesises its own speech on-board;
nothing here is ever used against the robot.
API (verified against a live call on 2026-09-02, not copied from a doc summary):
POST https://generativelanguage.googleapis.com/v1beta/interactions
x-goog-api-key: <key>
{"model": "...-tts-preview", "input": "...",
"response_format": {"type": "audio"},
"generation_config": {"speech_config": [{"voice": "Puck"}]}}
-> steps[0].content[0].data base64 PCM
steps[0].content[0].mime_type audio/l16; rate=24000; channels=1
The audio lives under `steps[]`, not the `output_audio` field some docs describe
- that path was checked against a real response.
THE LATENCY PROBLEM, AND THE TWO THINGS DONE ABOUT IT
-----------------------------------------------------
Measured on this account: ~4 s to synthesise a short sentence, ~8-10 s for a long
paragraph. Unusable for a live demo if taken naively. So:
1. **Saved audio.** Every line is written to `audio_library/` as an ordinary
.wav named after its text, so a repeated line replays instantly, survives
restarts, works with no internet, and can be played outside this app.
Rehearsed lines can be built ahead of time - see scripts/warm_voice.py.
2. **One clip per utterance.** Text is NOT split by default: the whole line is
synthesised in a single take, so the delivery is continuous and the library
holds one file per thing you said. Optional sentence pipelining is available
for very long text via GEMINI_CHUNK_CHARS, at the cost of an audible seam.
And it must never break the demo: any failure - no network, bad key, quota -
falls back to the built-in system voice rather than producing silence.
"""
from __future__ import annotations
import base64
import json
import logging
import os
import platform
import re
import struct
import subprocess
import threading
import time
import urllib.error
import urllib.request
from pathlib import Path
from typing import Dict, List, Optional, Tuple
from ..services.audio_library import AudioLibrary
from .local_audio import LocalVoice
logger = logging.getLogger(__name__)
ENDPOINT = "https://generativelanguage.googleapis.com/v1beta/interactions"
#: Prebuilt voices with the tone Google documents for each.
VOICES: Dict[str, str] = {
"Zephyr": "Bright", "Puck": "Upbeat", "Charon": "Informative", "Kore": "Firm",
"Fenrir": "Excitable", "Leda": "Youthful", "Orus": "Firm", "Aoede": "Breezy",
"Callirrhoe": "Easy-going", "Autonoe": "Bright", "Enceladus": "Breathy",
"Iapetus": "Clear", "Umbriel": "Easy-going", "Algieba": "Smooth",
"Despina": "Smooth", "Erinome": "Clear", "Algenib": "Gravelly",
"Rasalgethi": "Informative", "Laomedeia": "Upbeat", "Achernar": "Soft",
"Alnilam": "Firm", "Schedar": "Even", "Gacrux": "Mature",
"Pulcherrima": "Forward", "Achird": "Friendly", "Zubenelgenubi": "Casual",
"Vindemiatrix": "Gentle", "Sadachbia": "Lively", "Sadaltager": "Knowledgeable",
"Sulafat": "Warm",
}
#: 0 = never split. Every utterance is synthesised as ONE clip, in one take.
#:
#: Splitting was an optimisation: a request costs ~3.5 s fixed plus ~45 ms per
#: character, so starting playback after the first sentence reaches audio sooner
#: on long text. But each piece is a separate synthesis - separate delivery,
#: separate file - and that is audible. One continuous take is worth more than
#: a second or two of head start.
#:
#: Set GEMINI_CHUNK_CHARS in .env to a character count to re-enable pipelining
#: for very long text.
#:
#: NOTE: unrelated to the robot's own 1024-BYTE limit in aimdk_transport.py.
#: That one is a hard API constraint; this is a latency choice.
_CHUNK_TARGET_CHARS = 0
_SENTENCE_END = re.compile(r"(?<=[.!?。!?;:])\s+|\n+")
def wav_from_pcm(pcm: bytes, sample_rate: int = 24000, channels: int = 1,
bits: int = 16) -> bytes:
"""Wrap raw PCM in a 44-byte WAV header so ordinary players accept it."""
byte_rate = sample_rate * channels * bits // 8
block_align = channels * bits // 8
return b"".join([
b"RIFF", struct.pack("<I", 36 + len(pcm)), b"WAVE",
b"fmt ", struct.pack("<IHHIIHH", 16, 1, channels, sample_rate,
byte_rate, block_align, bits),
b"data", struct.pack("<I", len(pcm)), pcm,
])
def split_sentences(text: str, target: int = _CHUNK_TARGET_CHARS) -> List[str]:
"""Split into speakable chunks on sentence boundaries.
`target <= 0` disables splitting entirely: the whole utterance is synthesised
as ONE clip, in one continuous take. That is the default, because a split
line is synthesised as separate requests - each piece gets its own delivery
and its own file, and the join between them can be audible.
"""
text = " ".join((text or "").split())
if not text:
return []
if target <= 0 or len(text) <= target:
return [text]
pieces = [p.strip() for p in _SENTENCE_END.split(text) if p and p.strip()]
chunks: List[str] = []
current = ""
for piece in pieces:
if not current:
current = piece
elif len(current) + 1 + len(piece) <= target:
current += " " + piece
else:
chunks.append(current)
current = piece
if current:
chunks.append(current)
return chunks
# --------------------------------------------------------------------------- #
# playback
# --------------------------------------------------------------------------- #
class _Player:
"""Plays one WAV without blocking; can be stopped mid-playback."""
def __init__(self) -> None:
self._windows = platform.system().lower().startswith("win")
self._process: Optional[subprocess.Popen] = None
self._ends_at = 0.0
self._lock = threading.Lock()
def play(self, path: str, duration: float) -> None:
if self._windows:
import winsound
winsound.PlaySound(path, winsound.SND_FILENAME | winsound.SND_ASYNC)
# winsound has no completion callback, but the exact duration is
# known from the PCM length, so the deadline is precise.
with self._lock:
self._ends_at = time.monotonic() + duration
return
binary = "afplay" if platform.system().lower() == "darwin" else "aplay"
try:
with self._lock:
self._process = subprocess.Popen(
[binary, path], stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL, stdin=subprocess.DEVNULL)
except FileNotFoundError:
logger.warning("no audio player found (%s); install it to hear the simulator", binary)
def stop(self) -> None:
if self._windows:
try:
import winsound
winsound.PlaySound(None, winsound.SND_PURGE)
except Exception: # pragma: no cover
pass
with self._lock:
self._ends_at = 0.0
return
with self._lock:
process, self._process = self._process, None
if process is not None and process.poll() is None:
try:
process.kill()
except Exception: # pragma: no cover
pass
def is_playing(self) -> bool:
if self._windows:
with self._lock:
return time.monotonic() < self._ends_at
with self._lock:
process = self._process
return process is not None and process.poll() is None
# --------------------------------------------------------------------------- #
# the voice
# --------------------------------------------------------------------------- #
class GeminiVoice(LocalVoice):
"""Neural TTS through the Gemini API, saved to disk, with a local fallback."""
name = "gemini"
def __init__(self, api_key: str, model: str, voice: str, style: str = "",
timeout: float = 60.0, cache_dir: Optional[str] = None,
library: Optional[AudioLibrary] = None,
chunk_chars: int = _CHUNK_TARGET_CHARS,
fallback: Optional[LocalVoice] = None) -> None:
if not api_key:
raise ValueError("GEMINI_API_KEY is empty")
self._key = api_key
self._model = model
self._voice = voice
self._style = style
self._timeout = timeout
# 0 (the default) means never split - one utterance, one clip.
self._chunk_chars = max(0, int(chunk_chars))
self._fallback = fallback
self._using_fallback = False
self._warned = False
# Audio is saved as ordinary .wav files, so a line spoken once can be
# replayed instantly, played outside this app, or used offline.
if library is not None:
self.library = library
else:
root = cache_dir or os.path.join(
os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))),
"audio_library",
)
self.library = AudioLibrary(Path(root))
self._player = _Player()
self._sequence_stop = threading.Event()
self._sequence: Optional[threading.Thread] = None
self._active = threading.Event()
# -- LocalVoice ---------------------------------------------------------- #
def prepare(self, text: str) -> bool:
"""Synthesise the first chunk (and warm the rest in the background).
Returns False when the cloud voice is unusable, so playback falls back
to the system voice.
"""
chunks = split_sentences(text, self._chunk_chars)
if not chunks:
return False
# Start the later chunks BEFORE synthesising the first one, so they are
# fetched in parallel. Doing it afterwards leaves an audible gap between
# sentences on a cold run, because chunk 2 only starts once chunk 1 is
# already playing and finishes after the audio has run out.
prefetch: Optional[threading.Thread] = None
if len(chunks) > 1:
prefetch = threading.Thread(
target=self._warm_rest, args=(chunks[1:],),
name="gemini-prefetch", daemon=True,
)
prefetch.start()
try:
self._ensure(chunks[0])
except Exception as exc:
self._using_fallback = True
if not self._warned:
self._warned = True
logger.warning(
"Gemini voice unavailable (%s) - falling back to the built-in "
"system voice. The demo continues.", exc)
else:
logger.debug("Gemini synthesis failed: %s", exc)
return False
self._using_fallback = False
return True
def start(self, text: str) -> None:
self.stop()
chunks = split_sentences(text, self._chunk_chars)
if not chunks:
return
if self._using_fallback or not self._is_cached(chunks[0]):
if not self.prepare(text):
if self._fallback is not None:
self._using_fallback = True
self._fallback.start(text)
return
self._sequence_stop.clear()
self._active.set()
self._sequence = threading.Thread(
target=self._play_sequence, args=(chunks,),
name="gemini-playback", daemon=True,
)
self._sequence.start()
def stop(self) -> None:
self._sequence_stop.set()
self._player.stop()
self._active.clear()
if self._fallback is not None:
try:
self._fallback.stop()
except Exception: # pragma: no cover
pass
def is_speaking(self) -> bool:
if self._using_fallback and self._fallback is not None:
return self._fallback.is_speaking()
return self._active.is_set()
def close(self) -> None:
self.stop()
if self._fallback is not None:
try:
self._fallback.close()
except Exception: # pragma: no cover
pass
# -- playback sequencing ------------------------------------------------- #
def _play_sequence(self, chunks: List[str]) -> None:
"""Play each chunk in turn, waiting for later ones to finish synthesis."""
try:
for chunk in chunks:
if self._sequence_stop.is_set():
return
try:
path, duration = self._ensure(chunk)
except Exception as exc:
logger.warning("Gemini synthesis failed mid-utterance: %s", exc)
return
if self._sequence_stop.is_set():
return
self._player.play(path, duration)
while self._player.is_playing():
if self._sequence_stop.is_set():
self._player.stop()
return
time.sleep(0.03)
finally:
self._active.clear()
def _warm_rest(self, chunks: List[str]) -> None:
for chunk in chunks:
if self._sequence_stop.is_set():
return
try:
self._ensure(chunk)
except Exception as exc: # pragma: no cover - retried at play time
logger.debug("prefetch failed: %s", exc)
return
# -- saved audio + synthesis --------------------------------------------- #
def _is_cached(self, text: str) -> bool:
return self.library.find(text, self._voice, self._model, self._style) is not None
def _ensure(self, text: str) -> Tuple[str, float]:
"""Return (wav path, duration), synthesising only if not already saved."""
entry = self.library.find(text, self._voice, self._model, self._style)
if entry is None:
pcm, sample_rate, channels = self._synthesise(text)
entry = self.library.save(
text, wav_from_pcm(pcm, sample_rate, channels),
self._voice, self._model, self._style,
)
return str(self.library.root / entry["file"]), entry["durationSeconds"]
def _synthesise(self, text: str) -> Tuple[bytes, int, int]:
prompt = "{0} {1}".format(self._style, text).strip() if self._style else text
body = json.dumps({
"model": self._model,
"input": prompt,
"response_format": {"type": "audio"},
"generation_config": {"speech_config": [{"voice": self._voice}]},
}).encode("utf-8")
request = urllib.request.Request(
ENDPOINT, data=body,
headers={"x-goog-api-key": self._key, "Content-Type": "application/json"},
)
started = time.perf_counter()
try:
with urllib.request.urlopen(request, timeout=self._timeout) as response:
payload = json.loads(response.read())
except urllib.error.HTTPError as exc:
detail = exc.read().decode("utf-8", "replace")[:300]
raise RuntimeError("HTTP {0}: {1}".format(exc.code, detail)) from exc
pcm, sample_rate, channels = _extract_audio(payload)
logger.info(
"Gemini '%s': %.1fs of audio in %.2fs for %d chars",
self._voice, len(pcm) / float(sample_rate * channels * 2),
time.perf_counter() - started, len(text),
)
return pcm, sample_rate, channels
# -- helpers ------------------------------------------------------------- #
def warm(self, text: str) -> int:
"""Pre-synthesise every chunk of `text`. Returns how many were fetched."""
fetched = 0
for chunk in split_sentences(text, self._chunk_chars):
if not self._is_cached(chunk):
self._ensure(chunk)
fetched += 1
return fetched
def cache_stats(self) -> Dict[str, object]:
stats = self.library.stats()
return {"entries": stats["count"], "bytes": stats["bytes"], "dir": stats["dir"]}
def clear_cache(self) -> None:
self.library.clear()
def _extract_audio(payload: dict) -> Tuple[bytes, int, int]:
"""Pull PCM out of an interactions response.
Walks the steps rather than indexing a fixed path, so an extra step or a
reordered response does not break playback.
"""
for step in payload.get("steps", []) or []:
for part in step.get("content", []) or []:
if part.get("type") != "audio" or not part.get("data"):
continue
pcm = base64.b64decode(part["data"])
return pcm, int(part.get("sample_rate") or 24000), int(part.get("channels") or 1)
raise RuntimeError("no audio in response (status={0})".format(payload.get("status")))

View File

@ -0,0 +1,453 @@
"""PC-speaker playback for the simulator.
Lets the mock robot actually *say* the text out of this laptop's speakers, so the
whole demo can be rehearsed - wording, pacing, the Stop button - before the A3 is
on the network.
This is simulation only. It has nothing to do with the real robot: the A3
synthesises its own speech on-board and no audio ever leaves the PC (see
docs/AGIBOT_A3_INTEGRATION.md).
Backends, in order of preference:
Windows SAPI5 through comtypes - in-process, ~20 ms to start, real interrupt
Windows PowerShell System.Speech - fallback, no packages needed at all
macOS `say`
Linux `espeak-ng` / `espeak` / `spd-say`
Every backend supports three things the simulator needs: start without blocking,
report whether audio is still playing, and **stop immediately**. Without that last
one the Stop button would lie - the UI would say "stopped" while the laptop kept
talking.
"""
from __future__ import annotations
import logging
import platform
import queue
import shutil
import subprocess
import threading
from typing import List, Optional
logger = logging.getLogger(__name__)
# SAPI ISpVoice::Speak flags
_SVSF_ASYNC = 1
_SVSF_PURGE_BEFORE_SPEAK = 2
_SVSF_IS_XML = 8
def _xml_escape(text: str) -> str:
"""Escape text going into SAPI/SSML markup.
Without this, an operator typing `Tom & Jerry` or `a < b` would produce
malformed markup and the utterance would be mangled or dropped.
"""
return (
text.replace("&", "&amp;")
.replace("<", "&lt;")
.replace(">", "&gt;")
.replace('"', "&quot;")
)
class LocalVoice:
"""A speech engine available to the simulator."""
name = "none"
def prepare(self, text: str) -> bool:
"""Optional: do slow work (e.g. a cloud round trip) before playback.
Called during the "Processing" stage so that "Speaking" is only
announced once audio can really start. Local engines need nothing.
"""
return True
def start(self, text: str) -> None:
raise NotImplementedError
def stop(self) -> None:
raise NotImplementedError
def is_speaking(self) -> bool:
raise NotImplementedError
def close(self) -> None:
pass
# --------------------------------------------------------------------------- #
# Windows - SAPI5 via COM
# --------------------------------------------------------------------------- #
class SapiVoice(LocalVoice):
"""Windows SAPI5.
COM objects are apartment-bound, so one dedicated thread creates the voice
and is the only thread that ever touches it. Public methods just hand it work
through a queue and read plain flags, which makes them safe to call from the
asyncio loop.
"""
name = "sapi"
def __init__(self, voice_hint: Optional[str] = None, rate: int = 0,
volume: int = 100, pitch: int = 0) -> None:
self._voice_hint = voice_hint
self._rate = max(-10, min(10, rate))
self._volume = max(0, min(100, volume))
# SAPI pitch: -10..10. Raising it turns an adult male voice into a
# younger-sounding one, which is the closest a stock Windows voice gets
# to a teenage character voice.
self._pitch = max(-10, min(10, pitch))
self._commands: "queue.Queue" = queue.Queue()
self._speaking = threading.Event()
self._stop_flag = threading.Event()
self._ready = threading.Event()
self._error: Optional[str] = None
self._closing = False
self._thread = threading.Thread(target=self._run, name="sapi-voice", daemon=True)
self._thread.start()
self._ready.wait(timeout=8)
if self._error:
raise RuntimeError(self._error)
# -- public API ---------------------------------------------------------- #
def start(self, text: str) -> None:
self._stop_flag.clear()
# Set the flag here, not on the worker thread: a caller that polls
# is_speaking() immediately must never see "idle" before we begin.
self._speaking.set()
self._commands.put(text)
def stop(self) -> None:
self._stop_flag.set()
def is_speaking(self) -> bool:
return self._speaking.is_set()
def close(self) -> None:
self._closing = True
self._stop_flag.set()
self._commands.put(None)
# -- worker thread ------------------------------------------------------- #
def _run(self) -> None:
try:
import comtypes
import comtypes.client
except ImportError as exc: # pragma: no cover - checked by the factory
self._error = "comtypes not installed: {0}".format(exc)
self._ready.set()
return
try:
comtypes.CoInitialize()
except Exception: # pragma: no cover - already initialised is fine
pass
try:
voice = comtypes.client.CreateObject("SAPI.SpVoice")
self._select_voice(voice)
voice.Rate = self._rate
voice.Volume = self._volume
except Exception as exc:
self._error = "cannot create SAPI voice: {0}".format(exc)
self._ready.set()
return
self._ready.set()
logger.info("local audio: Windows SAPI ready (rate=%s volume=%s)", self._rate, self._volume)
while not self._closing:
text = self._commands.get()
if text is None:
break
try:
if self._pitch:
voice.Speak(
'<pitch middle="{0}">{1}</pitch>'.format(self._pitch, _xml_escape(text)),
_SVSF_ASYNC | _SVSF_IS_XML,
)
else:
voice.Speak(text, _SVSF_ASYNC)
# Poll rather than block, so a Stop lands within ~50 ms.
while not voice.WaitUntilDone(50):
if self._stop_flag.is_set():
voice.Speak("", _SVSF_PURGE_BEFORE_SPEAK)
break
except Exception as exc: # pragma: no cover - device may vanish
logger.warning("local audio playback failed: %s", exc)
finally:
self._speaking.clear()
try:
import comtypes
comtypes.CoUninitialize()
except Exception: # pragma: no cover
pass
def _select_voice(self, voice) -> None:
"""Pick the voice whose description matches the configured hint."""
if not self._voice_hint:
return
hint = self._voice_hint.lower()
try:
for description, token, source in _sapi_tokens():
if hint in description.lower():
voice.Voice = token
logger.info("local audio: using voice '%s' (%s)", description, source)
return
logger.warning(
"local audio: no voice matching '%s'; using the system default. "
"Run 'python scripts/voices.py' to see what is installed.",
self._voice_hint,
)
except Exception as exc: # pragma: no cover
logger.debug("voice selection failed: %s", exc)
#: Windows keeps two voice registries. The "OneCore" set is the newer, better
#: sounding one and is invisible to SpVoice.GetVoices(), so look there first.
_ONECORE_KEY = r"HKEY_LOCAL_MACHINE\SOFTWARE\Microsoft\Speech_OneCore\Voices"
def _sapi_tokens():
"""[(description, token, source)] for every SAPI voice, best set first.
Must be called on a thread where COM is initialised.
"""
import comtypes.client
found = []
seen = set()
try:
category = comtypes.client.CreateObject("SAPI.SpObjectTokenCategory")
category.SetId(_ONECORE_KEY, False)
tokens = category.EnumerateTokens()
for index in range(tokens.Count):
token = tokens.Item(index)
description = token.GetDescription()
found.append((description, token, "onecore"))
seen.add(description)
except Exception as exc: # pragma: no cover - older Windows
logger.debug("OneCore voices unavailable: %s", exc)
try:
voice = comtypes.client.CreateObject("SAPI.SpVoice")
tokens = voice.GetVoices()
for index in range(tokens.Count):
token = tokens.Item(index)
description = token.GetDescription()
if description not in seen:
found.append((description, token, "classic"))
except Exception as exc: # pragma: no cover
logger.debug("classic voices unavailable: %s", exc)
return found
def list_local_voices():
"""[(description, source)] - for `python scripts/voices.py`. Never raises."""
if not platform.system().lower().startswith("win"):
return []
try:
import comtypes
try:
comtypes.CoInitialize()
except Exception: # pragma: no cover
pass
return [(description, source) for description, _, source in _sapi_tokens()]
except Exception: # pragma: no cover
return []
# --------------------------------------------------------------------------- #
# Everything else - an external command we can kill
# --------------------------------------------------------------------------- #
class CommandVoice(LocalVoice):
"""Speak by running a command. Stop = kill the process."""
def __init__(self, name: str, argv_builder) -> None:
self.name = name
self._argv = argv_builder
self._process: Optional[subprocess.Popen] = None
self._lock = threading.Lock()
def start(self, text: str) -> None:
self.stop()
argv = self._argv(text)
try:
with self._lock:
self._process = subprocess.Popen(
argv,
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL,
stdin=subprocess.DEVNULL,
)
except Exception as exc: # pragma: no cover - depends on host
logger.warning("local audio command failed (%s): %s", argv[0], exc)
self._process = None
def stop(self) -> None:
with self._lock:
process = self._process
self._process = None
if process is not None and process.poll() is None:
try:
process.kill()
except Exception: # pragma: no cover
pass
def is_speaking(self) -> bool:
with self._lock:
process = self._process
return process is not None and process.poll() is None
def close(self) -> None:
self.stop()
def _powershell_voice(rate: int, volume: int, pitch: int = 0,
voice_hint: Optional[str] = None) -> CommandVoice:
"""Windows fallback using System.Speech - present on every Windows install."""
rate = max(-10, min(10, rate))
volume = max(0, min(100, volume))
pitch = max(-10, min(10, pitch))
def argv(text: str) -> List[str]:
select = ""
if voice_hint:
# SelectVoiceByHints has no substring form; fall back silently when
# the named voice is absent rather than throwing.
select = (
"try {{ $s.SelectVoice(($s.GetInstalledVoices() | "
"Where-Object {{ $_.VoiceInfo.Name -like '*{0}*' }} | "
"Select-Object -First 1).VoiceInfo.Name) }} catch {{}};"
).format(voice_hint.replace("'", "''"))
if pitch:
# System.Speech exposes pitch only through SSML prosody.
body = (
"<speak version='1.0' xmlns='http://www.w3.org/2001/10/synthesis' "
"xml:lang='en-US'><prosody pitch='{0:+d}%'>{1}</prosody></speak>"
).format(pitch * 5, _xml_escape(text))
speak = "$s.SpeakSsml('{0}')".format(body.replace("'", "''"))
else:
speak = "$s.Speak('{0}')".format(text.replace("'", "''"))
script = (
"Add-Type -AssemblyName System.Speech;"
"$s = New-Object System.Speech.Synthesis.SpeechSynthesizer;"
"{select}$s.Rate = {rate}; $s.Volume = {volume};{speak}"
).format(select=select, rate=rate, volume=volume, speak=speak)
return ["powershell", "-NoProfile", "-NonInteractive", "-Command", script]
return CommandVoice("powershell", argv)
def _macos_voice(rate: int) -> CommandVoice:
# `say` takes words per minute; map the -10..10 scale onto a sane range.
wpm = max(90, min(320, 180 + rate * 12))
def argv(text: str) -> List[str]:
return ["say", "-r", str(wpm), text]
return CommandVoice("say", argv)
def _linux_voice(rate: int, volume: int) -> Optional[CommandVoice]:
if shutil.which("espeak-ng") or shutil.which("espeak"):
binary = "espeak-ng" if shutil.which("espeak-ng") else "espeak"
wpm = max(80, min(320, 175 + rate * 12))
amplitude = max(0, min(200, int(volume * 2)))
def argv(text: str) -> List[str]:
return [binary, "-s", str(wpm), "-a", str(amplitude), text]
return CommandVoice(binary, argv)
if shutil.which("spd-say"):
def argv(text: str) -> List[str]:
return ["spd-say", "-w", "-r", str(max(-100, min(100, rate * 10))), text]
return CommandVoice("spd-say", argv)
return None
# --------------------------------------------------------------------------- #
# factory
# --------------------------------------------------------------------------- #
def create_local_voice(
voice_hint: Optional[str] = None, rate: int = 0, volume: int = 100, pitch: int = 0,
pronouncer=None,
) -> Optional[LocalVoice]:
"""Best available PC voice, or None if this machine cannot speak.
Never raises: local audio is a convenience, and losing it must not stop the
simulator from running.
"""
voice = _create(voice_hint, rate, volume, pitch)
if voice is not None and pronouncer is not None:
# OS voices mispronounce brand names; fix the text at the last moment.
return Respelling(voice, pronouncer)
return voice
class Respelling(LocalVoice):
"""Wraps a voice, respelling text on its way to the synthesiser."""
def __init__(self, inner: LocalVoice, pronouncer) -> None:
self._inner = inner
self._pronouncer = pronouncer
self.name = inner.name
def prepare(self, text: str) -> bool:
return self._inner.prepare(self._pronouncer.apply(text))
def start(self, text: str) -> None:
self._inner.start(self._pronouncer.apply(text))
def stop(self) -> None:
self._inner.stop()
def is_speaking(self) -> bool:
return self._inner.is_speaking()
def close(self) -> None:
self._inner.close()
def _create(voice_hint, rate, volume, pitch):
system = platform.system().lower()
if system.startswith("win"):
try:
return SapiVoice(voice_hint, rate, volume, pitch)
except Exception as exc:
logger.info("SAPI unavailable (%s); falling back to PowerShell", exc)
if shutil.which("powershell"):
return _powershell_voice(rate, volume, pitch, voice_hint)
logger.warning("local audio: no speech engine found on this Windows PC")
return None
if system == "darwin":
if shutil.which("say"):
return _macos_voice(rate)
logger.warning("local audio: 'say' not found")
return None
voice = _linux_voice(rate, volume)
if voice is None:
logger.warning(
"local audio: install espeak-ng (apt install espeak-ng) to hear the simulator"
)
return voice

197
backend/robot/manager.py Normal file
View File

@ -0,0 +1,197 @@
"""Connection supervisor.
Owns the adapter's lifecycle so the web app never has to care whether the robot
is up. One background task:
disconnected -> connecting -> connected -> (health poll) -> disconnected -> ...
Connection failures are normal, not exceptional: they are published as status
events and retried with capped exponential backoff. The HTTP server keeps serving
the dashboard throughout, which is the whole point of the offline-first design.
"""
from __future__ import annotations
import asyncio
import logging
import time
from typing import Any, Dict, Optional
from ..config.settings import Settings
from ..core.events import EventBus, EventType
from .base import RobotAdapter, RobotError, RobotState
from .factory import create_robot
logger = logging.getLogger(__name__)
class RobotManager:
def __init__(self, settings: Settings, bus: EventBus) -> None:
self._settings = settings
self._bus = bus
self._adapter: RobotAdapter = create_robot(settings)
self._state = RobotState.DISCONNECTED
self._error: Optional[str] = None
self._error_code: Optional[str] = None
self._latency_ms: Optional[float] = None
self._connected_since: Optional[float] = None
self._attempts = 0
self._task: Optional[asyncio.Task] = None
self._wake = asyncio.Event()
self._closing = False
# -- accessors ----------------------------------------------------------- #
@property
def adapter(self) -> RobotAdapter:
return self._adapter
@property
def state(self) -> RobotState:
return self._state
@property
def is_connected(self) -> bool:
return self._state == RobotState.CONNECTED and self._adapter.is_connected
def status_dict(self) -> Dict[str, Any]:
info = self._adapter.info
return {
"state": self._state.value,
"connected": self.is_connected,
"speaking": self._adapter.is_speaking,
"robot": info.to_dict(),
"latencyMs": round(self._latency_ms, 1) if self._latency_ms is not None else None,
"uptimeSeconds": (
round(time.time() - self._connected_since, 1) if self._connected_since else None
),
"attempts": self._attempts,
"error": self._error,
"errorCode": self._error_code,
"configIssues": [issue.to_dict() for issue in self._settings.issues],
}
# -- lifecycle ----------------------------------------------------------- #
async def start(self) -> None:
self._closing = False
self._task = asyncio.create_task(self._supervise(), name="robot-supervisor")
self._publish()
async def stop(self) -> None:
self._closing = True
self._wake.set()
if self._task is not None:
self._task.cancel()
try:
await self._task
except (asyncio.CancelledError, Exception):
pass
self._task = None
try:
await self._adapter.disconnect()
except Exception: # pragma: no cover
logger.debug("adapter disconnect raised", exc_info=True)
self._set_state(RobotState.DISCONNECTED)
def request_reconnect(self) -> None:
"""Ask the supervisor to retry immediately (used by the API)."""
self._wake.set()
async def rebuild(self, settings: Settings) -> None:
"""Swap in a new configuration without restarting the process."""
await self.stop()
self._settings = settings
self._adapter = create_robot(settings)
self._error = None
self._error_code = None
self._latency_ms = None
self._attempts = 0
await self.start()
# -- supervisor ---------------------------------------------------------- #
async def _supervise(self) -> None:
delay = self._settings.robot.reconnect_min_delay
while not self._closing:
try:
if not self._adapter.is_connected:
self._attempts += 1
self._set_state(RobotState.CONNECTING)
started = time.perf_counter()
await self._adapter.connect()
self._latency_ms = (time.perf_counter() - started) * 1000
self._connected_since = time.time()
self._error = None
self._error_code = None
delay = self._settings.robot.reconnect_min_delay
self._set_state(RobotState.CONNECTED)
logger.info("robot connected (%.0f ms)", self._latency_ms)
await self._sleep_or_wake(self._settings.robot.health_interval)
if self._closing:
break
started = time.perf_counter()
healthy = await self._adapter.health_check()
elapsed = (time.perf_counter() - started) * 1000
if healthy:
# Smooth the latency reading so the UI badge does not flicker.
self._latency_ms = (
elapsed if self._latency_ms is None else self._latency_ms * 0.7 + elapsed * 0.3
)
if self._state != RobotState.CONNECTED:
self._set_state(RobotState.CONNECTED)
else:
self._publish() # keep latency fresh in the dashboard
else:
logger.warning("robot health check failed - reconnecting")
self._connected_since = None
self._error = "Lost connection to the robot."
self._error_code = "unreachable"
self._set_state(RobotState.DISCONNECTED)
await self._safe_disconnect()
except asyncio.CancelledError:
raise
except RobotError as exc:
self._on_failure(exc.user_message, exc.code)
delay = await self._backoff(delay)
except Exception as exc: # never let the supervisor die
logger.exception("unexpected supervisor error")
self._on_failure("Unexpected robot error: {0}".format(exc), "internal")
delay = await self._backoff(delay)
def _on_failure(self, message: str, code: str) -> None:
self._connected_since = None
self._latency_ms = None
self._error = message
self._error_code = code
self._set_state(RobotState.ERROR if code == "not_configured" else RobotState.DISCONNECTED)
async def _backoff(self, delay: float) -> float:
await self._sleep_or_wake(delay)
return min(delay * 2, self._settings.robot.reconnect_max_delay)
async def _sleep_or_wake(self, seconds: float) -> None:
"""Sleep, but return early if someone calls request_reconnect()."""
try:
await asyncio.wait_for(self._wake.wait(), timeout=max(0.1, seconds))
except asyncio.TimeoutError:
return
finally:
self._wake.clear()
async def _safe_disconnect(self) -> None:
try:
await self._adapter.disconnect()
except Exception: # pragma: no cover
logger.debug("disconnect during recovery raised", exc_info=True)
# -- events -------------------------------------------------------------- #
def _set_state(self, state: RobotState) -> None:
changed = state != self._state
self._state = state
if changed:
logger.info("robot state -> %s", state.value)
self._publish()
def _publish(self) -> None:
self._bus.publish(EventType.ROBOT_STATUS, self.status_dict())

365
backend/robot/mock_robot.py Normal file
View File

@ -0,0 +1,365 @@
"""Mock AGIBOT A3.
A stand-in that reproduces the *shape and timing* of a real robot TTS exchange:
a network hop, a synthesis pause, a speaking window proportional to the text, then
completion - all interruptible. The frontend cannot tell the difference, which is
the point: everything except the wire protocol gets tested before the robot lands.
Optional: set MOCK_LOCAL_AUDIO=true to hear the text through the PC's own
speakers, so the whole demo can be rehearsed before the robot is on the network.
When that is on, the *real* playback drives the timeline - "Completed" appears
exactly when the sound stops, and Stop cuts the audio mid-word. See local_audio.py.
"""
from __future__ import annotations
import asyncio
import logging
import random
import time
from typing import Any, Dict, Optional
from ..config.settings import Settings
from .local_audio import LocalVoice, create_local_voice
from .base import (
ProgressCallback,
RobotAdapter,
RobotCapabilities,
RobotInfo,
RobotUnreachable,
SpeechFailed,
SpeechProgress,
SpeechRequest,
SpeechResult,
SpeechStage,
)
logger = logging.getLogger(__name__)
# Granularity of the simulated speaking window. Small enough that Stop feels
# instant, large enough not to spin the event loop.
_TICK = 0.05
class MockRobot(RobotAdapter):
"""Simulated robot. Never touches the network."""
def __init__(self, settings: Settings) -> None:
self._settings = settings
self._mock = settings.mock
self._connected = False
self._speaking = False
self._cancel = asyncio.Event()
self._health_calls = 0
self._rng = random.Random(20260902)
self._local_voice: Optional[LocalVoice] = None
self._voice_ready = False
# -- identity ------------------------------------------------------------ #
@property
def info(self) -> RobotInfo:
return RobotInfo(
name=self._settings.robot.name,
model=self._settings.robot.model + " (simulated)",
mode="mock",
transport="mock",
address="local simulation",
capabilities=RobotCapabilities(
native_tts=True,
stop=True,
progress_events=True,
reports_completion=True,
voice_selection=False,
volume_control=False,
),
)
@property
def is_connected(self) -> bool:
return self._connected
@property
def is_speaking(self) -> bool:
return self._speaking
# -- lifecycle ----------------------------------------------------------- #
async def connect(self) -> None:
await asyncio.sleep(self._mock.connect_delay_ms / 1000)
if self._mock.flaky_connection and self._rng.random() < 0.25:
self._connected = False
raise RobotUnreachable("simulated connection failure")
self._connected = True
logger.info("MockRobot connected (simulated)")
async def disconnect(self) -> None:
self._cancel.set()
self._connected = False
self._speaking = False
if self._local_voice is not None:
try:
self._local_voice.close()
except Exception: # pragma: no cover
pass
self._local_voice = None
self._voice_ready = False
logger.info("MockRobot disconnected (simulated)")
async def health_check(self) -> bool:
self._health_calls += 1
# Cost a plausible round trip so the dashboard's latency readout looks
# like a network number rather than a suspicious 0 ms.
await asyncio.sleep(self._mock.network_latency_ms / 2000)
# Simulate an intermittent link so reconnect logic gets exercised offline.
if self._mock.flaky_connection and self._health_calls % 17 == 0:
self._connected = False
return False
return self._connected
# -- speech -------------------------------------------------------------- #
async def speak(self, request: SpeechRequest, on_progress: ProgressCallback) -> SpeechResult:
if not self._connected:
raise RobotUnreachable("mock robot is not connected")
self._cancel.clear()
self._speaking = True
started = time.perf_counter()
ack_ms: Optional[int] = None
try:
# 1. request leaves the PC
await on_progress(
SpeechProgress(request.id, SpeechStage.SENDING, "Sending to robot...")
)
await self._sleep(self._mock.network_latency_ms / 1000)
if self._mock.failure_rate and self._rng.random() < self._mock.failure_rate:
raise SpeechFailed(
"simulated TTS failure",
user_message="Speech request failed. The robot rejected the utterance.",
)
# 2. robot has the text and is synthesising
ack_ms = int((time.perf_counter() - started) * 1000)
await on_progress(
SpeechProgress(
request.id,
SpeechStage.PROCESSING,
"Robot is preparing speech...",
elapsed_ms=ack_ms,
)
)
await self._sleep(self._mock.processing_ms / 1000)
# A cloud voice needs a network round trip. Do it here, while the UI
# still says "preparing speech", so that "Speaking..." is only shown
# once audio can actually start.
voice = self._voice()
if voice is not None and not self._cancel.is_set():
await asyncio.get_running_loop().run_in_executor(
None, voice.prepare, request.text
)
# 3. audio starts
duration = self._estimate_duration(request.text)
await on_progress(
SpeechProgress(
request.id,
SpeechStage.SPEAKING,
"Speaking...",
detail="~{0:.1f}s".format(duration),
elapsed_ms=int((time.perf_counter() - started) * 1000),
)
)
# With PC audio on, the real playback drives the timeline, so the UI
# returns to Ready exactly when the sound stops. Otherwise fall back
# to the estimate.
if self._start_local_audio(request.text):
cancelled = await self._wait_for_audio(duration)
else:
cancelled = await self._sleep(duration)
if cancelled:
await on_progress(
SpeechProgress(request.id, SpeechStage.CANCELLED, "Speech stopped.")
)
return SpeechResult(
request_id=request.id,
success=False,
stage=SpeechStage.CANCELLED,
ack_latency_ms=ack_ms,
total_ms=int((time.perf_counter() - started) * 1000),
error_code="cancelled",
error="Stopped by operator.",
)
total_ms = int((time.perf_counter() - started) * 1000)
await on_progress(
SpeechProgress(request.id, SpeechStage.COMPLETED, "Ready", elapsed_ms=total_ms)
)
return SpeechResult(
request_id=request.id,
success=True,
stage=SpeechStage.COMPLETED,
ack_latency_ms=ack_ms,
total_ms=total_ms,
)
finally:
self._speaking = False
async def stop_speaking(self) -> bool:
if not self._speaking:
return False
self._cancel.set()
# Cut the audio now rather than waiting for the polling loop, so the
# speakers go quiet the instant Stop is pressed.
if self._local_voice is not None:
try:
self._local_voice.stop()
except Exception: # pragma: no cover
pass
return True
async def describe(self) -> Dict[str, Any]:
return {
"simulation": {
"connectDelayMs": self._mock.connect_delay_ms,
"networkLatencyMs": self._mock.network_latency_ms,
"processingMs": self._mock.processing_ms,
"wordsPerMinute": self._mock.words_per_minute,
"failureRate": self._mock.failure_rate,
"flakyConnection": self._mock.flaky_connection,
"localAudio": self._mock.local_audio,
"voiceBackend": self._local_voice.name if self._local_voice else None,
"voiceHint": self._mock.voice,
"voicePitch": self._mock.speech_pitch,
},
"healthChecks": self._health_calls,
}
# -- internals ----------------------------------------------------------- #
async def _sleep(self, seconds: float) -> bool:
"""Sleep, but wake immediately on Stop. Returns True if interrupted."""
deadline = time.perf_counter() + seconds
while True:
remaining = deadline - time.perf_counter()
if remaining <= 0:
return False
if self._cancel.is_set():
return True
await asyncio.sleep(min(_TICK, remaining))
def _estimate_duration(self, text: str) -> float:
"""Approximate how long a human-sounding voice would take to read `text`."""
words = len(text.split())
cjk = sum(1 for ch in text if "" <= ch <= "鿿")
if cjk > words: # Chinese/Japanese text has few whitespace-delimited words
words = max(words, cjk // 2)
wpm = max(60, self._mock.words_per_minute)
return max(0.8, (words / wpm) * 60.0)
def _start_local_audio(self, text: str) -> bool:
"""Begin PC-speaker playback. Returns True if audio actually started."""
voice = self._voice()
if voice is None:
return False
try:
voice.start(text)
return True
except Exception as exc: # pragma: no cover - depends on host audio
logger.warning("local audio failed, falling back to silent timing: %s", exc)
return False
def _voice(self) -> Optional[LocalVoice]:
"""Create the PC voice on first use; never fail the utterance over it."""
if not self._mock.local_audio:
return None
if self._voice_ready:
return self._local_voice
self._voice_ready = True
try:
system_voice = create_local_voice(
voice_hint=self._mock.voice,
rate=self._mock.speech_rate,
volume=self._mock.speech_volume,
pitch=self._mock.speech_pitch,
)
# Respell once, around whichever engine ends up in front, so both
# the cloud voice and the fallback say names the same way.
self._local_voice = self._wrap_pronunciation(self._wrap_cloud(system_voice))
if self._local_voice is None:
logger.warning(
"MOCK_LOCAL_AUDIO is on but this PC has no usable speech engine"
)
except Exception as exc: # pragma: no cover
logger.warning("local audio unavailable: %s", exc)
self._local_voice = None
return self._local_voice
def _wrap_pronunciation(self, voice):
"""Fix brand-name pronunciation just before synthesis, for any engine."""
if voice is None or not self._mock.pronunciation:
return voice
from ..config.settings import PROJECT_ROOT
from ..core.pronunciation import build as build_pronouncer
from .local_audio import Respelling
pronouncer = build_pronouncer(PROJECT_ROOT, True)
if pronouncer is None or not len(pronouncer):
return voice
return Respelling(voice, pronouncer)
def _wrap_cloud(self, system_voice):
"""Put the Gemini voice in front of the system voice, if configured.
The system voice stays as the fallback so a network or quota problem
degrades to a robotic voice rather than to silence mid-demo.
"""
if self._mock.voice_engine != "gemini":
return system_voice
if not self._mock.gemini_api_key:
logger.warning(
"MOCK_VOICE_ENGINE=gemini but GEMINI_API_KEY is not set; "
"using the built-in system voice"
)
return system_voice
try:
from ..services.audio_library import get_audio_library
from .gemini_voice import GeminiVoice
voice = GeminiVoice(
library=get_audio_library(),
api_key=self._mock.gemini_api_key,
model=self._mock.gemini_model,
voice=self._mock.gemini_voice,
style=self._mock.gemini_style,
chunk_chars=self._mock.gemini_chunk_chars,
fallback=system_voice,
)
logger.info(
"local audio: Gemini voice '%s' (%s), fallback=%s",
self._mock.gemini_voice, self._mock.gemini_model,
system_voice.name if system_voice else "none",
)
return voice
except Exception as exc:
logger.warning("Gemini voice unavailable (%s); using the system voice", exc)
return system_voice
async def _wait_for_audio(self, fallback_seconds: float) -> bool:
"""Wait until the speakers go quiet. Returns True if Stop interrupted it."""
voice = self._local_voice
assert voice is not None
# Never wait forever if a backend misreports its state.
deadline = time.perf_counter() + max(fallback_seconds * 4, 30.0)
while time.perf_counter() < deadline:
if self._cancel.is_set():
try:
voice.stop()
except Exception: # pragma: no cover
pass
return True
if not voice.is_speaking():
return False
await asyncio.sleep(_TICK)
return False

View File

@ -0,0 +1,67 @@
"""Transport registry for the AGIBOT A3 adapter."""
from __future__ import annotations
from typing import Callable, Dict
from ...config.settings import Settings
from ..base import RobotError
from .base import SpeechTransport, TransportAck
_BUILDERS: Dict[str, Callable[[Settings], SpeechTransport]] = {}
def _aimdk(settings: Settings) -> SpeechTransport:
from .aimdk_transport import AimdkTransport
return AimdkTransport(settings)
def _http(settings: Settings) -> SpeechTransport:
from .http_transport import HttpTransport
return HttpTransport(settings)
def _ws(settings: Settings) -> SpeechTransport:
from .ws_transport import WebSocketTransport
return WebSocketTransport(settings)
def _ros2(settings: Settings) -> SpeechTransport:
from .ros2_transport import Ros2Transport
return Ros2Transport(settings)
def _ssh(settings: Settings) -> SpeechTransport:
from .ssh_transport import SshTransport
return SshTransport(settings)
_BUILDERS.update(
{"aimdk": _aimdk, "http": _http, "ws": _ws, "ros2": _ros2, "ssh": _ssh}
)
def available_transports() -> list:
return sorted(_BUILDERS)
def create_transport(settings: Settings) -> SpeechTransport:
"""Instantiate the transport named by A3_TRANSPORT (imported lazily)."""
name = settings.a3.transport
builder = _BUILDERS.get(name)
if builder is None:
raise RobotError(
"unknown transport '{0}'".format(name),
user_message="Unknown A3_TRANSPORT '{0}'. Valid values: {1}.".format(
name, ", ".join(available_transports())
),
)
return builder(settings)
__all__ = ["SpeechTransport", "TransportAck", "create_transport", "available_transports"]

View File

@ -0,0 +1,354 @@
"""AimDK HTTP JSON-RPC transport - the documented AGIBOT A3 speech path.
WHAT THIS IMPLEMENTS
--------------------
AgiBot's A3 developer guide documents an HTTP JSON-RPC interface served by the
robot's HDU (head unit) through their AimRT runtime:
POST http://<robot>:<port>/rpc/<service_name>/<method_name>
Content-Type: application/json
and, on that interface, a text-to-speech service:
POST http://<robot>:59301/rpc/aimdk.protocol.TTSService/PlayTTS
{"text": "...", "priority_level": "INTERACTION_L6", "domain": "...",
"trace_id": "...", "is_interrupted": true}
-> {"trace_id": "<yours>_<suffix>", "is_sucess": true, "error_message": "", ...}
Docs (public, no login required):
https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/07-02-audio_play
https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/03-second_develop_interface_overview
This is the fastest available path: text in, robot's own TTS out. No audio is
generated or transferred by this PC.
FOUR TRAPS THIS FILE HANDLES DELIBERATELY
-----------------------------------------
1. `is_sucess` - the success flag is spelled with one 'c' in AgiBot's docs. Both
spellings are read, so a corrected firmware does not break us.
2. The **returned** trace_id differs from the one you send (a random suffix is
appended). Stop only works with the returned value, so it is what we keep.
3. A 1024-byte request limit on `text`. Long input is chunked on sentence
boundaries by UTF-8 byte length - a 1000-character Chinese sentence is ~3000
bytes and would otherwise be rejected.
4. AgiBot's constraints page limits status polling to <= 0.2 Hz and warns that
high-frequency RPC can destabilise the robot. So the liveness probe is a TCP
connect, never an RPC, and completion is estimated rather than polled.
EVERYTHING IS STILL CONFIGURABLE. Service and method names, the port, the
priority level and the byte limit are all .env keys, because AimRT publishes no
endpoint-discovery route and AgiBot does not guarantee port stability across
firmware. If your unit differs, you edit .env - not this file.
"""
from __future__ import annotations
import asyncio
import logging
import time
from typing import Any, Dict, List, Optional, Tuple
from ...config.settings import Settings
from ..base import (
ProgressCallback,
RobotTimeout,
RobotUnreachable,
SpeechFailed,
SpeechProgress,
SpeechRequest,
SpeechStage,
TransportNotAvailable,
)
from .base import SpeechTransport, TransportAck, now_ms
logger = logging.getLogger(__name__)
try:
import httpx
except ImportError: # pragma: no cover
httpx = None # type: ignore[assignment]
# Sentence-ish boundaries, most preferred first. Covers CJK punctuation because
# the robot's own examples are Chinese.
_BREAKS = ("\n", "", "", "", ". ", "! ", "? ", "", "; ", "", ", ", " ")
# Values AgiBot's docs treat as "no error" in the RPC envelope.
_OK_CODES = ("", "0", 0, None)
class AimdkTransport(SpeechTransport):
"""AGIBOT A3 native TTS over AimRT HTTP JSON-RPC."""
name = "aimdk"
reports_completion = False # PlayTTS acks acceptance, not end-of-audio
def __init__(self, settings: Settings) -> None:
self._settings = settings
self._robot = settings.robot
self._a3 = settings.a3
self._client: Optional["httpx.AsyncClient"] = None
#: trace_id the robot gave us for the utterance in flight - Stop needs it
self._active_trace: Optional[str] = None
self._last_response: Any = None
# -- lifecycle ----------------------------------------------------------- #
async def open(self) -> None:
if httpx is None: # pragma: no cover
raise TransportNotAvailable(
"httpx is not installed",
user_message="This transport requires 'httpx'. Run: pip install -r requirements.txt",
)
if self._client is not None:
return
headers = {
"Content-Type": "application/json", # mandatory per AgiBot's docs
"Connection": "keep-alive",
# AimRT honours a whole-second Timeout header; its own default is 5 s.
"Timeout": str(max(1, int(self._robot.request_timeout))),
}
headers.update(self._a3.http_headers or {})
if self._a3.http_auth_token:
headers.setdefault("Authorization", "Bearer " + self._a3.http_auth_token)
self._client = httpx.AsyncClient(
base_url=self._robot.base_url,
headers=headers,
timeout=httpx.Timeout(self._robot.request_timeout, connect=self._robot.connect_timeout),
limits=httpx.Limits(max_keepalive_connections=4, max_connections=8),
)
logger.info(
"AimDK transport ready: %s/rpc/%s/%s",
self._robot.base_url, self._a3.aimdk_service, self._a3.aimdk_play_method,
)
async def close(self) -> None:
if self._client is not None:
try:
await self._client.aclose()
except Exception: # pragma: no cover
pass
self._client = None
# -- health -------------------------------------------------------------- #
async def probe(self) -> bool:
"""TCP connect only.
Deliberately not an RPC call: AgiBot's constraints page caps status RPC
polling at 0.2 Hz, and this runs on the health-check interval. A socket
connect proves the service port is listening without adding RPC load.
"""
if not self._robot.ip:
return False
try:
fut = asyncio.open_connection(self._robot.ip, self._robot.port)
_, writer = await asyncio.wait_for(fut, timeout=self._robot.connect_timeout)
writer.close()
try:
await writer.wait_closed()
except Exception: # pragma: no cover
pass
return True
except Exception as exc:
logger.debug("AimDK probe to %s failed: %s", self._robot.address, exc)
return False
# -- speech -------------------------------------------------------------- #
async def speak(self, request: SpeechRequest, on_progress: ProgressCallback) -> TransportAck:
if self._client is None:
await self.open()
chunks = chunk_text(request.text, self._a3.aimdk_max_bytes)
if not chunks:
raise SpeechFailed("nothing to say", user_message="Please enter some text.")
await on_progress(SpeechProgress(request.id, SpeechStage.SENDING, "Sending to robot..."))
started = time.perf_counter()
ack_ms: Optional[int] = None
detail: Optional[str] = None
for index, chunk in enumerate(chunks):
payload = {
"text": chunk,
"priority_level": self._a3.aimdk_priority,
"domain": self._a3.aimdk_domain,
"trace_id": "{0}{1}".format(request.id, "" if index == 0 else "-%d" % index),
# Only the first chunk may interrupt whatever was playing; the
# rest must queue behind it or the utterance cuts itself off.
"is_interrupted": self._a3.aimdk_interrupt if index == 0 else False,
}
trace, chunk_detail = await self._call(
self._a3.aimdk_play_method, payload, what="PlayTTS"
)
if index == 0:
ack_ms = now_ms(started)
detail = chunk_detail
if len(chunks) > 1:
detail = "{0} parts".format(len(chunks))
if trace:
self._active_trace = trace
return TransportAck(
accepted=True,
ack_latency_ms=ack_ms if ack_ms is not None else now_ms(started),
detail=detail,
raw=self._last_response,
)
async def stop(self) -> bool:
"""Stop the utterance using the trace_id the robot returned to us."""
trace = self._active_trace
if not trace:
logger.info("no active trace_id - nothing to stop")
return False
if self._client is None:
await self.open()
try:
await self._call(self._a3.aimdk_stop_method, {"trace_id": trace}, what="StopTTS")
return True
except Exception as exc:
logger.warning("StopTTS failed: %s", exc)
return False
finally:
self._active_trace = None
async def describe(self) -> Dict[str, Any]:
return {
"baseUrl": self._robot.base_url,
"service": self._a3.aimdk_service,
"playMethod": self._a3.aimdk_play_method,
"stopMethod": self._a3.aimdk_stop_method,
"priority": self._a3.aimdk_priority,
"domain": self._a3.aimdk_domain,
"maxBytesPerChunk": self._a3.aimdk_max_bytes,
"activeTraceId": self._active_trace,
"lastResponse": self._last_response,
"docs": "https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/07-02-audio_play",
}
# -- internals ----------------------------------------------------------- #
def _route(self, method: str) -> str:
# AimRT strips the "pb:/" func-name prefix when building the HTTP route -
# the URL is /rpc/<service>/<method> and must never contain "pb:/".
return "/rpc/{0}/{1}".format(self._a3.aimdk_service.lstrip("/"), method)
async def _call(self, method: str, payload: Dict[str, Any], what: str) -> Tuple[Optional[str], Optional[str]]:
"""One RPC. Returns (trace_id, human detail). Raises on any failure."""
assert self._client is not None
route = self._route(method)
try:
response = await self._client.post(route, json=payload)
except Exception as exc:
if httpx is not None and isinstance(exc, httpx.TimeoutException):
raise RobotTimeout(
"timeout calling {0}{1}".format(self._robot.base_url, route)
) from exc
raise RobotUnreachable(
"cannot reach {0}{1}: {2}".format(self._robot.base_url, route, exc)
) from exc
# AimRT maps an RPC-level failure to HTTP 500 and an unknown route to 404.
if response.status_code == 404:
raise SpeechFailed(
"route {0} not found (HTTP 404)".format(route),
user_message=(
"The robot does not expose {0} at {1}. Check A3_AIMDK_SERVICE / "
"A3_AIMDK_PLAY_METHOD and the robot's firmware version."
).format(method, route),
)
if response.status_code >= 400:
raise SpeechFailed(
"{0} returned HTTP {1}: {2}".format(what, response.status_code, response.text[:300]),
user_message="Speech request failed (robot returned HTTP {0}).".format(
response.status_code
),
)
try:
body = response.json()
except Exception:
body = {"raw": response.text[:300]}
self._last_response = body
if not isinstance(body, dict):
return None, None
# Envelope check, when the firmware wraps replies in a header block.
header = body.get("header")
if isinstance(header, dict):
code = header.get("code")
if code not in _OK_CODES:
raise SpeechFailed(
"{0} rejected: code={1} msg={2}".format(what, code, header.get("msg")),
user_message="Speech request failed: {0}".format(
header.get("msg") or "robot returned error code {0}".format(code)
),
)
# AgiBot's documented field is `is_sucess` (one 'c'). Read both spellings
# so a future firmware that fixes the typo keeps working.
success = body.get("is_sucess", body.get("is_success"))
if success is False:
raise SpeechFailed(
"{0} rejected: {1}".format(what, body.get("error_message")),
user_message="Speech request failed: {0}".format(
body.get("error_message") or "the robot rejected the utterance."
),
)
trace = body.get("trace_id")
return (trace if isinstance(trace, str) else None), None
# --------------------------------------------------------------------------- #
# chunking
# --------------------------------------------------------------------------- #
def chunk_text(text: str, max_bytes: int) -> List[str]:
"""Split `text` into pieces of at most `max_bytes` UTF-8 bytes.
The limit AgiBot documents is a *byte* limit, so counting characters is not
enough - Chinese text is three bytes per character. Splits prefer sentence
ends, then clause punctuation, then spaces, and finally cut mid-word rather
than emit an over-length chunk.
"""
text = (text or "").strip()
if not text:
return []
budget = max(64, int(max_bytes) - 2) # small margin for encoding edge cases
if len(text.encode("utf-8")) <= budget:
return [text]
chunks: List[str] = []
remaining = text
while remaining:
if len(remaining.encode("utf-8")) <= budget:
chunks.append(remaining.strip())
break
window = _prefix_within_bytes(remaining, budget)
split_at = -1
for marker in _BREAKS:
found = window.rfind(marker)
if found > budget // 4: # avoid pathologically tiny chunks
split_at = found + len(marker)
break
if split_at <= 0:
split_at = len(window)
piece = remaining[:split_at].strip()
if piece:
chunks.append(piece)
remaining = remaining[split_at:].lstrip()
return [chunk for chunk in chunks if chunk]
def _prefix_within_bytes(text: str, budget: int) -> str:
"""Longest prefix of `text` that fits in `budget` UTF-8 bytes."""
encoded = text.encode("utf-8")
if len(encoded) <= budget:
return text
# Decode back, dropping any partial trailing character.
return encoded[:budget].decode("utf-8", "ignore")

View File

@ -0,0 +1,136 @@
"""Transport contract for the AGIBOT A3 adapter.
A *transport* knows how to move one "speak this text" command onto the wire and
back. It knows nothing about history, the UI, or the event bus.
Why the payloads are templates rather than code
-----------------------------------------------
The exact request body an A3 expects is defined by AgiBot's SDK/API reference for
your specific unit and firmware. Inventing one would be worse than useless, so the
shape is supplied as configuration:
A3_HTTP_SPEAK_PATH=/some/documented/path
A3_HTTP_SPEAK_PAYLOAD={"text": "{text}", "voice": "{voice}"}
Placeholders available in any template string:
{text} {id} {voice} {language} {volume} {speed}
A template string that is *exactly* one placeholder keeps the value's native type
(so {volume} becomes a JSON number, not the string "0.8"). Keys whose rendered
value is null are dropped, so unset optional parameters are simply not sent.
"""
from __future__ import annotations
import abc
import time
from dataclasses import dataclass
from typing import Any, Dict, Optional
from urllib.parse import quote
from ..base import ProgressCallback, SpeechRequest
@dataclass
class TransportAck:
"""What the robot said when it accepted (or rejected) the utterance."""
accepted: bool
ack_latency_ms: int
detail: Optional[str] = None
raw: Optional[Any] = None
class SpeechTransport(abc.ABC):
"""One wire protocol for delivering text to the robot."""
name: str = "transport"
#: True when the robot itself tells us the utterance finished. When False the
#: adapter estimates the speaking duration instead of guessing completion.
reports_completion: bool = False
@abc.abstractmethod
async def open(self) -> None:
"""Set up any persistent client/socket. Raises RobotError on failure."""
@abc.abstractmethod
async def close(self) -> None:
"""Tear down. Must never raise."""
@abc.abstractmethod
async def probe(self) -> bool:
"""Cheap liveness check against the robot."""
@abc.abstractmethod
async def speak(
self, request: SpeechRequest, on_progress: ProgressCallback
) -> TransportAck:
"""Deliver the utterance. Return as soon as the robot acknowledges it."""
@abc.abstractmethod
async def stop(self) -> bool:
"""Ask the robot to stop speaking. Returns True if a stop was delivered."""
async def describe(self) -> Dict[str, Any]:
return {}
# --------------------------------------------------------------------------- #
# payload templating
# --------------------------------------------------------------------------- #
def build_context(request: SpeechRequest, defaults: Dict[str, Any]) -> Dict[str, Any]:
ctx: Dict[str, Any] = {
"text": request.text,
"id": request.id,
"voice": request.voice if request.voice is not None else defaults.get("voice"),
"language": request.language if request.language is not None else defaults.get("language"),
"volume": defaults.get("volume"),
"speed": defaults.get("speed"),
}
return ctx
def render_payload(template: Any, ctx: Dict[str, Any], drop_null: bool = True) -> Any:
"""Recursively substitute placeholders into a parsed-JSON template."""
if isinstance(template, str):
stripped = template.strip()
if stripped.startswith("{") and stripped.endswith("}"):
key = stripped[1:-1]
if key in ctx:
return ctx[key] # preserve native type
rendered = template
for key, value in ctx.items():
token = "{" + key + "}"
if token in rendered:
rendered = rendered.replace(token, "" if value is None else str(value))
return rendered
if isinstance(template, dict):
out: Dict[str, Any] = {}
for key, value in template.items():
resolved = render_payload(value, ctx, drop_null)
if drop_null and resolved is None:
continue # unset optional parameter - do not send the key at all
out[key] = resolved
return out
if isinstance(template, list):
items = [render_payload(item, ctx, drop_null) for item in template]
return [item for item in items if not (drop_null and item is None)]
return template
def render_path(template: str, ctx: Dict[str, Any]) -> str:
"""Render a URL path/query template with percent-encoded values."""
rendered = template
for key, value in ctx.items():
token = "{" + key + "}"
if token in rendered:
rendered = rendered.replace(token, quote("" if value is None else str(value), safe=""))
return rendered
def now_ms(since: float) -> int:
return int((time.perf_counter() - since) * 1000)

View File

@ -0,0 +1,232 @@
"""HTTP/REST transport.
Lowest-friction path when the robot exposes an HTTP API: one keep-alive
connection is opened at startup and reused for every utterance, so a Speak press
costs a single request/response round trip with no TCP or TLS handshake.
The endpoint, method and body come from .env - see transports/base.py for why.
"""
from __future__ import annotations
import asyncio
import logging
import time
from typing import Any, Dict, Optional
from ...config.settings import Settings
from ..base import (
ProgressCallback,
RobotTimeout,
RobotUnreachable,
SpeechFailed,
SpeechProgress,
SpeechRequest,
SpeechStage,
TransportNotAvailable,
)
from .base import SpeechTransport, TransportAck, build_context, now_ms, render_path, render_payload
logger = logging.getLogger(__name__)
try: # httpx is a hard requirement for this transport only
import httpx
except ImportError: # pragma: no cover
httpx = None # type: ignore[assignment]
class HttpTransport(SpeechTransport):
name = "http"
reports_completion = False # a REST ack rarely means "finished speaking"
def __init__(self, settings: Settings) -> None:
self._settings = settings
self._robot = settings.robot
self._a3 = settings.a3
self._client: Optional["httpx.AsyncClient"] = None
# -- lifecycle ----------------------------------------------------------- #
async def open(self) -> None:
if httpx is None:
raise TransportNotAvailable(
"httpx is not installed",
user_message="HTTP transport requires the 'httpx' package. Run: pip install -r requirements.txt",
)
if self._client is not None:
return
headers = {"Content-Type": "application/json", "Connection": "keep-alive"}
headers.update(self._a3.http_headers or {})
if self._a3.http_auth_token:
headers.setdefault("Authorization", "Bearer " + self._a3.http_auth_token)
self._client = httpx.AsyncClient(
base_url=self._robot.base_url,
headers=headers,
timeout=httpx.Timeout(
self._robot.request_timeout, connect=self._robot.connect_timeout
),
# Keep the socket hot between utterances - this is the latency win.
limits=httpx.Limits(max_keepalive_connections=4, max_connections=8),
verify=False if self._robot.use_tls else True,
)
logger.info("HTTP transport ready for %s", self._robot.base_url)
async def close(self) -> None:
if self._client is not None:
try:
await self._client.aclose()
except Exception: # pragma: no cover
pass
self._client = None
# -- health -------------------------------------------------------------- #
async def probe(self) -> bool:
"""Prefer a documented status endpoint; fall back to a TCP connect."""
if self._a3.http_status_path and self._client is not None:
try:
response = await self._client.request(
self._a3.http_status_method,
self._a3.http_status_path,
timeout=self._robot.connect_timeout,
)
return response.status_code < 500
except Exception as exc:
logger.debug("HTTP probe failed: %s", exc)
return False
return await self._tcp_probe()
async def _tcp_probe(self) -> bool:
"""No status endpoint configured - at least verify the port is open."""
if not self._robot.ip:
return False
try:
fut = asyncio.open_connection(self._robot.ip, self._robot.port)
reader, writer = await asyncio.wait_for(fut, timeout=self._robot.connect_timeout)
writer.close()
try:
await writer.wait_closed()
except Exception: # pragma: no cover
pass
return True
except Exception as exc:
logger.debug("TCP probe %s failed: %s", self._robot.address, exc)
return False
# -- speech -------------------------------------------------------------- #
async def speak(self, request: SpeechRequest, on_progress: ProgressCallback) -> TransportAck:
if self._client is None:
await self.open()
assert self._client is not None
ctx = build_context(request, self._defaults())
path = render_path(self._a3.http_speak_path, ctx)
payload = render_payload(self._a3.http_speak_payload, ctx)
method = self._a3.http_speak_method
await on_progress(SpeechProgress(request.id, SpeechStage.SENDING, "Sending to robot..."))
started = time.perf_counter()
try:
if method in ("GET", "DELETE", "HEAD"):
response = await self._client.request(method, path)
else:
response = await self._client.request(method, path, json=payload)
except Exception as exc: # httpx timeout/connect errors
if httpx is not None and isinstance(exc, httpx.TimeoutException):
raise RobotTimeout(
"timeout calling {0}{1}: {2}".format(self._robot.base_url, path, exc)
) from exc
raise RobotUnreachable(
"cannot reach {0}{1}: {2}".format(self._robot.base_url, path, exc)
) from exc
ack_ms = now_ms(started)
if response.status_code >= 400:
raise SpeechFailed(
"robot returned HTTP {0}: {1}".format(response.status_code, response.text[:400]),
user_message="Speech request failed (robot returned HTTP {0}).".format(
response.status_code
),
)
raw: Any = None
try:
raw = response.json()
except Exception:
raw = response.text[:400]
# Optional: a documented success flag inside the JSON body.
if self._a3.http_success_field and isinstance(raw, dict):
value = _dig(raw, self._a3.http_success_field)
if value is not None and not _truthy(value):
raise SpeechFailed(
"robot reported failure: {0}".format(raw),
user_message="Speech request failed. The robot rejected the utterance.",
)
return TransportAck(accepted=True, ack_latency_ms=ack_ms, raw=raw)
async def stop(self) -> bool:
if not self._a3.http_stop_path:
logger.info("no A3_HTTP_STOP_PATH configured - cannot stop robot-side playback")
return False
if self._client is None:
await self.open()
assert self._client is not None
ctx = build_context(SpeechRequest(text=""), self._defaults())
path = render_path(self._a3.http_stop_path, ctx)
payload = render_payload(self._a3.http_stop_payload, ctx)
try:
if self._a3.http_stop_method in ("GET", "DELETE", "HEAD"):
response = await self._client.request(self._a3.http_stop_method, path)
else:
response = await self._client.request(self._a3.http_stop_method, path, json=payload)
return response.status_code < 400
except Exception as exc:
logger.warning("stop request failed: %s", exc)
return False
async def describe(self) -> Dict[str, Any]:
return {
"baseUrl": self._robot.base_url,
"speak": "{0} {1}".format(self._a3.http_speak_method, self._a3.http_speak_path or "(unset)"),
"stop": "{0} {1}".format(self._a3.http_stop_method, self._a3.http_stop_path or "(unset)"),
"status": "{0} {1}".format(self._a3.http_status_method, self._a3.http_status_path or "(TCP probe)"),
"keepAlive": True,
}
def _defaults(self) -> Dict[str, Any]:
return {
"voice": self._a3.voice,
"language": self._a3.language,
"volume": self._a3.volume,
"speed": self._a3.speed,
}
def _dig(data: Dict[str, Any], dotted: str) -> Any:
node: Any = data
for part in dotted.split("."):
if not isinstance(node, dict) or part not in node:
return None
node = node[part]
return node
def _truthy(value: Any) -> bool:
"""Interpret a success field.
Deliberately liberal: robot APIs variously signal success as `true`, `"ok"`,
`0` (a result code) or `200` (an embedded HTTP status).
"""
if isinstance(value, bool):
return value
if isinstance(value, int):
return value in (0, 200, 1)
if isinstance(value, float):
return value in (0.0, 1.0, 200.0)
if isinstance(value, str):
return value.strip().lower() in ("0", "1", "200", "true", "ok", "success", "succeeded", "yes")
return bool(value)

View File

@ -0,0 +1,236 @@
"""ROS 2 transport.
Used when the robot exposes its speech interface as a ROS 2 topic or service on
the shared DDS network rather than as a web API.
`rclpy` is imported lazily: the rest of this project runs on a plain Windows PC
with no ROS installation, and only this file needs one. The message/service type
is resolved from its ROS type string at runtime, so no interface is hard-coded:
A3_ROS_SPEAK_TOPIC=/the/documented/topic
A3_ROS_SPEAK_MSG_TYPE=std_msgs/msg/String
A3_ROS_SPEAK_MSG_FIELD=data
DDS discovery is multicast-based, so the PC and the robot must be on the same L2
network segment and share ROS_DOMAIN_ID. See docs/NETWORK.md.
"""
from __future__ import annotations
import asyncio
import importlib
import logging
import os
import threading
import time
from typing import Any, Dict, Optional
from ...config.settings import Settings
from ..base import (
ProgressCallback,
RobotUnreachable,
SpeechFailed,
SpeechProgress,
SpeechRequest,
SpeechStage,
TransportNotAvailable,
)
from .base import SpeechTransport, TransportAck, build_context, now_ms
logger = logging.getLogger(__name__)
def _resolve_type(type_string: str) -> Any:
"""'std_msgs/msg/String' -> the std_msgs.msg.String class."""
parts = type_string.replace(".", "/").split("/")
if len(parts) == 2: # tolerate 'pkg/Type'
package, name = parts
module_path = package + ".msg"
elif len(parts) == 3:
package, kind, name = parts
module_path = package + "." + kind
else:
raise SpeechFailed(
"cannot parse ROS type '{0}'".format(type_string),
user_message="Invalid ROS message type in configuration: " + type_string,
)
module = importlib.import_module(module_path)
return getattr(module, name)
class Ros2Transport(SpeechTransport):
name = "ros2"
reports_completion = False
def __init__(self, settings: Settings) -> None:
self._settings = settings
self._a3 = settings.a3
self._robot = settings.robot
self._rclpy: Any = None
self._node: Any = None
self._publisher: Any = None
self._stop_publisher: Any = None
self._client: Any = None
self._executor: Any = None
self._spin_thread: Optional[threading.Thread] = None
# -- lifecycle ----------------------------------------------------------- #
async def open(self) -> None:
if self._node is not None:
return
await asyncio.get_running_loop().run_in_executor(None, self._open_blocking)
def _open_blocking(self) -> None:
try:
import rclpy
from rclpy.executors import SingleThreadedExecutor
from rclpy.qos import QoSProfile, ReliabilityPolicy
except ImportError as exc:
raise TransportNotAvailable(
"rclpy not importable: {0}".format(exc),
user_message=(
"ROS 2 transport selected but rclpy is not installed on this PC. "
"Install ROS 2 and run the app from a sourced ROS environment, "
"or choose a different A3_TRANSPORT."
),
) from exc
os.environ.setdefault("ROS_DOMAIN_ID", str(self._a3.ros_domain_id))
self._rclpy = rclpy
if not rclpy.ok():
rclpy.init(args=None)
self._node = rclpy.create_node("agibot_voice_bridge")
qos = QoSProfile(depth=10, reliability=ReliabilityPolicy.RELIABLE)
if self._a3.ros_use_service:
service_type = _resolve_type(self._a3.ros_service_type)
self._client = self._node.create_client(service_type, self._a3.ros_service_name)
else:
msg_type = _resolve_type(self._a3.ros_speak_msg_type)
self._publisher = self._node.create_publisher(msg_type, self._a3.ros_speak_topic, qos)
if self._a3.ros_stop_topic:
self._stop_publisher = self._node.create_publisher(
msg_type, self._a3.ros_stop_topic, qos
)
self._executor = SingleThreadedExecutor()
self._executor.add_node(self._node)
self._spin_thread = threading.Thread(
target=self._executor.spin, name="rclpy-spin", daemon=True
)
self._spin_thread.start()
logger.info("ROS 2 transport ready (domain %s)", os.environ.get("ROS_DOMAIN_ID"))
async def close(self) -> None:
node, executor, rclpy = self._node, self._executor, self._rclpy
self._node = self._executor = self._publisher = self._client = None
if executor is not None:
try:
executor.shutdown()
except Exception: # pragma: no cover
pass
if node is not None:
try:
node.destroy_node()
except Exception: # pragma: no cover
pass
if rclpy is not None:
try:
if rclpy.ok():
rclpy.shutdown()
except Exception: # pragma: no cover
pass
# -- health -------------------------------------------------------------- #
async def probe(self) -> bool:
if self._node is None:
return False
try:
if self._client is not None:
return bool(self._client.service_is_ready())
if self._publisher is not None:
# A subscriber on the speech topic means the robot side is alive.
return self._publisher.get_subscription_count() > 0
except Exception as exc:
logger.debug("ros2 probe failed: %s", exc)
return False
# -- speech -------------------------------------------------------------- #
async def speak(self, request: SpeechRequest, on_progress: ProgressCallback) -> TransportAck:
if self._node is None:
await self.open()
ctx = build_context(request, self._defaults())
await on_progress(SpeechProgress(request.id, SpeechStage.SENDING, "Publishing to ROS 2..."))
started = time.perf_counter()
loop = asyncio.get_running_loop()
try:
await loop.run_in_executor(None, self._publish_blocking, ctx)
except TransportNotAvailable:
raise
except Exception as exc:
raise RobotUnreachable("ROS 2 publish failed: {0}".format(exc)) from exc
return TransportAck(accepted=True, ack_latency_ms=now_ms(started))
def _publish_blocking(self, ctx: Dict[str, Any]) -> None:
if self._client is not None:
request_type = _resolve_type(self._a3.ros_service_type).Request
req = request_type()
setattr(req, self._a3.ros_speak_msg_field, ctx["text"])
if not self._client.wait_for_service(timeout_sec=self._robot.connect_timeout):
raise RobotUnreachable(
"ROS service {0} not available".format(self._a3.ros_service_name)
)
future = self._client.call_async(req)
deadline = time.time() + self._robot.request_timeout
while not future.done() and time.time() < deadline:
time.sleep(0.01)
return
assert self._publisher is not None
msg_type = _resolve_type(self._a3.ros_speak_msg_type)
msg = msg_type()
field = self._a3.ros_speak_msg_field
if not hasattr(msg, field):
raise SpeechFailed(
"message type {0} has no field '{1}'".format(self._a3.ros_speak_msg_type, field),
user_message="Configured ROS message field '{0}' does not exist on {1}.".format(
field, self._a3.ros_speak_msg_type
),
)
setattr(msg, field, ctx["text"])
self._publisher.publish(msg)
async def stop(self) -> bool:
if self._stop_publisher is None:
return False
try:
msg_type = _resolve_type(self._a3.ros_speak_msg_type)
msg = msg_type()
setattr(msg, self._a3.ros_speak_msg_field, "")
self._stop_publisher.publish(msg)
return True
except Exception as exc:
logger.warning("ros2 stop failed: %s", exc)
return False
async def describe(self) -> Dict[str, Any]:
return {
"domainId": self._a3.ros_domain_id,
"mode": "service" if self._a3.ros_use_service else "topic",
"topic": self._a3.ros_speak_topic,
"messageType": self._a3.ros_speak_msg_type,
"service": self._a3.ros_service_name,
"nodeUp": self._node is not None,
}
def _defaults(self) -> Dict[str, Any]:
return {
"voice": self._a3.voice,
"language": self._a3.language,
"volume": self._a3.volume,
"speed": self._a3.speed,
}

View File

@ -0,0 +1,192 @@
"""SSH / on-robot command transport.
The universal fallback. Every AgiBot humanoid runs a Linux compute module, so if
no network speech API is exposed, the robot can still be made to speak by running
its own TTS or audio-playback command over SSH.
Nothing here is A3-specific - you supply the command that actually works on your
unit, discovered by logging in once and trying it by hand:
A3_SSH_SPEAK_COMMAND=<the command that speaks on your robot> {text}
`{text}` is shell-quoted before substitution, so arbitrary operator input cannot
break out of the argument. Authentication uses the OpenSSH client (key or agent);
password auth is used only when `paramiko` is installed.
"""
from __future__ import annotations
import asyncio
import logging
import shlex
import time
from typing import Any, Dict, List, Optional
from ...config.settings import Settings
from ..base import (
ProgressCallback,
RobotTimeout,
RobotUnreachable,
SpeechFailed,
SpeechProgress,
SpeechRequest,
SpeechStage,
)
from .base import SpeechTransport, TransportAck, now_ms
logger = logging.getLogger(__name__)
try:
import paramiko # optional - only needed for password authentication
except ImportError: # pragma: no cover
paramiko = None # type: ignore[assignment]
class SshTransport(SpeechTransport):
name = "ssh"
reports_completion = True # the command returns when playback ends
def __init__(self, settings: Settings) -> None:
self._settings = settings
self._robot = settings.robot
self._a3 = settings.a3
self._current: Optional[asyncio.subprocess.Process] = None
# -- lifecycle ----------------------------------------------------------- #
async def open(self) -> None:
if not await self.probe():
raise RobotUnreachable(
"SSH probe to {0}@{1} failed".format(self._a3.ssh_user, self._robot.ip)
)
async def close(self) -> None:
process = self._current
self._current = None
if process is not None and process.returncode is None:
try:
process.terminate()
except Exception: # pragma: no cover
pass
async def probe(self) -> bool:
try:
code, _, _ = await self._run(
self._a3.ssh_probe_command or "true", timeout=self._robot.connect_timeout + 2
)
return code == 0
except Exception as exc:
logger.debug("ssh probe failed: %s", exc)
return False
# -- speech -------------------------------------------------------------- #
async def speak(self, request: SpeechRequest, on_progress: ProgressCallback) -> TransportAck:
command = self._a3.ssh_speak_command.replace("{text}", shlex.quote(request.text))
command = command.replace("{id}", shlex.quote(request.id))
await on_progress(SpeechProgress(request.id, SpeechStage.SENDING, "Sending to robot..."))
started = time.perf_counter()
code, out, err = await self._run(command, timeout=self._robot.request_timeout + 60)
if code != 0:
raise SpeechFailed(
"remote command exited {0}: {1}".format(code, (err or out)[:400]),
user_message="Speech request failed. The robot's speech command returned an error.",
)
return TransportAck(accepted=True, ack_latency_ms=now_ms(started), detail=(out or "")[:200])
async def stop(self) -> bool:
if not self._a3.ssh_stop_command:
return False
try:
code, _, _ = await self._run(self._a3.ssh_stop_command, timeout=10)
return code == 0
except Exception as exc:
logger.warning("ssh stop failed: %s", exc)
return False
async def describe(self) -> Dict[str, Any]:
return {
"host": "{0}@{1}:{2}".format(self._a3.ssh_user, self._robot.ip, self._a3.ssh_port),
"auth": "paramiko-password" if self._use_paramiko else "openssh-key/agent",
"speakCommand": self._a3.ssh_speak_command or "(unset)",
}
# -- internals ----------------------------------------------------------- #
@property
def _use_paramiko(self) -> bool:
return bool(self._a3.ssh_password) and paramiko is not None
async def _run(self, command: str, timeout: float):
if not self._robot.ip:
raise RobotUnreachable("ROBOT_IP is not set")
if self._use_paramiko:
return await asyncio.get_running_loop().run_in_executor(
None, self._run_paramiko, command, timeout
)
return await self._run_openssh(command, timeout)
def _ssh_argv(self, command: str) -> List[str]:
argv = [
"ssh",
"-p",
str(self._a3.ssh_port),
"-o",
"BatchMode=yes",
"-o",
"StrictHostKeyChecking=accept-new",
"-o",
"ConnectTimeout={0}".format(int(max(1, self._robot.connect_timeout))),
]
if self._a3.ssh_key_path:
argv += ["-i", self._a3.ssh_key_path]
argv.append("{0}@{1}".format(self._a3.ssh_user, self._robot.ip))
argv.append(command)
return argv
async def _run_openssh(self, command: str, timeout: float):
try:
process = await asyncio.create_subprocess_exec(
*self._ssh_argv(command),
stdout=asyncio.subprocess.PIPE,
stderr=asyncio.subprocess.PIPE,
)
except FileNotFoundError as exc:
raise RobotUnreachable(
"the 'ssh' client was not found on this PC",
) from exc
self._current = process
try:
stdout, stderr = await asyncio.wait_for(process.communicate(), timeout=timeout)
except asyncio.TimeoutError as exc:
process.kill()
raise RobotTimeout("remote command timed out after {0}s".format(timeout)) from exc
finally:
self._current = None
return (
process.returncode,
stdout.decode("utf-8", "replace"),
stderr.decode("utf-8", "replace"),
)
def _run_paramiko(self, command: str, timeout: float): # pragma: no cover - needs a host
assert paramiko is not None
client = paramiko.SSHClient()
client.set_missing_host_key_policy(paramiko.AutoAddPolicy())
try:
client.connect(
hostname=self._robot.ip,
port=self._a3.ssh_port,
username=self._a3.ssh_user,
password=self._a3.ssh_password,
key_filename=self._a3.ssh_key_path,
timeout=self._robot.connect_timeout,
allow_agent=False,
look_for_keys=False,
)
_, stdout, stderr = client.exec_command(command, timeout=timeout)
out = stdout.read().decode("utf-8", "replace")
err = stderr.read().decode("utf-8", "replace")
return stdout.channel.recv_exit_status(), out, err
finally:
client.close()

View File

@ -0,0 +1,226 @@
"""WebSocket transport.
The lowest-latency option when the robot offers it: the socket is opened once at
startup and held, so pressing Speak costs one frame write - no handshake, no
connection setup. It is also the only transport that can naturally deliver
*unsolicited* robot events ("speaking started", "speech finished") back to the UI.
If the robot's event schema is known, set:
A3_WS_DONE_FIELD=event # dotted path inside the JSON message
A3_WS_DONE_VALUE=speech_end # value that means "utterance finished"
and the dashboard will show true completion instead of an estimate.
"""
from __future__ import annotations
import asyncio
import json
import logging
import time
from typing import Any, Dict, Optional
from ...config.settings import Settings
from ..base import (
ProgressCallback,
RobotTimeout,
RobotUnreachable,
SpeechProgress,
SpeechRequest,
SpeechStage,
TransportNotAvailable,
)
from .base import SpeechTransport, TransportAck, build_context, now_ms, render_payload
logger = logging.getLogger(__name__)
try:
from websockets.asyncio.client import connect as ws_connect # websockets >= 13
except Exception: # pragma: no cover - older releases
try:
from websockets.client import connect as ws_connect # type: ignore[no-redef]
except Exception:
ws_connect = None # type: ignore[assignment]
class WebSocketTransport(SpeechTransport):
name = "ws"
def __init__(self, settings: Settings) -> None:
self._settings = settings
self._robot = settings.robot
self._a3 = settings.a3
self._ws: Any = None
self._reader: Optional[asyncio.Task] = None
self._completion: Optional[asyncio.Future] = None
self._lock = asyncio.Lock()
self._last_message: Any = None
@property
def reports_completion(self) -> bool: # type: ignore[override]
return bool(self._a3.ws_done_field)
# -- lifecycle ----------------------------------------------------------- #
@property
def url(self) -> str:
path = self._a3.ws_path or "/"
if not path.startswith("/"):
path = "/" + path
return self._robot.ws_base_url + path
async def open(self) -> None:
if ws_connect is None: # pragma: no cover
raise TransportNotAvailable(
"websockets package missing",
user_message="WebSocket transport requires the 'websockets' package.",
)
if self._ws is not None:
return
try:
self._ws = await asyncio.wait_for(
ws_connect(
self.url,
ping_interval=self._a3.ws_ping_interval or None,
ping_timeout=self._robot.request_timeout,
open_timeout=self._robot.connect_timeout,
max_queue=32,
),
timeout=self._robot.connect_timeout + 1,
)
except asyncio.TimeoutError as exc:
raise RobotTimeout("timed out opening {0}".format(self.url)) from exc
except Exception as exc:
raise RobotUnreachable("cannot open {0}: {1}".format(self.url, exc)) from exc
self._reader = asyncio.create_task(self._read_loop(), name="a3-ws-reader")
logger.info("WebSocket transport connected to %s", self.url)
async def close(self) -> None:
if self._reader is not None:
self._reader.cancel()
try:
await self._reader
except (asyncio.CancelledError, Exception):
pass
self._reader = None
if self._ws is not None:
try:
await self._ws.close()
except Exception: # pragma: no cover
pass
self._ws = None
async def probe(self) -> bool:
if self._ws is None:
return False
# A closed socket surfaces here without an extra round trip.
reader = self._reader
if reader is not None and reader.done():
return False
try:
pong = await self._ws.ping()
await asyncio.wait_for(pong, timeout=self._robot.connect_timeout)
return True
except Exception as exc:
logger.debug("ws probe failed: %s", exc)
return False
# -- speech -------------------------------------------------------------- #
async def speak(self, request: SpeechRequest, on_progress: ProgressCallback) -> TransportAck:
async with self._lock:
if self._ws is None:
await self.open()
assert self._ws is not None
ctx = build_context(request, self._defaults())
payload = render_payload(self._a3.ws_speak_payload, ctx)
message = json.dumps(payload, ensure_ascii=False)
loop = asyncio.get_running_loop()
self._completion = loop.create_future() if self.reports_completion else None
await on_progress(SpeechProgress(request.id, SpeechStage.SENDING, "Sending to robot..."))
started = time.perf_counter()
try:
await self._ws.send(message)
except Exception as exc:
await self.close()
raise RobotUnreachable("websocket send failed: {0}".format(exc)) from exc
return TransportAck(accepted=True, ack_latency_ms=now_ms(started))
async def await_completion(self, timeout: float) -> bool:
"""Block until the robot reports the utterance finished (if it does)."""
if self._completion is None:
return False
try:
await asyncio.wait_for(asyncio.shield(self._completion), timeout=timeout)
return True
except asyncio.TimeoutError:
return False
except Exception:
return False
async def stop(self) -> bool:
if self._ws is None or not self._a3.ws_stop_payload:
return False
ctx = build_context(SpeechRequest(text=""), self._defaults())
payload = render_payload(self._a3.ws_stop_payload, ctx)
try:
await self._ws.send(json.dumps(payload, ensure_ascii=False))
return True
except Exception as exc:
logger.warning("ws stop failed: %s", exc)
return False
async def describe(self) -> Dict[str, Any]:
return {
"url": self.url,
"connected": self._ws is not None,
"reportsCompletion": self.reports_completion,
"lastMessage": self._last_message,
}
# -- internals ----------------------------------------------------------- #
async def _read_loop(self) -> None:
assert self._ws is not None
try:
async for raw in self._ws:
self._last_message = raw if isinstance(raw, str) else "<binary>"
if not self._a3.ws_done_field:
continue
try:
message = json.loads(raw)
except Exception:
continue
value = _dig(message, self._a3.ws_done_field)
if value is None:
continue
expected = self._a3.ws_done_value
if expected is None or str(value) == expected:
fut = self._completion
if fut is not None and not fut.done():
fut.set_result(True)
except asyncio.CancelledError:
raise
except Exception as exc:
logger.info("websocket reader closed: %s", exc)
self._ws = None
def _defaults(self) -> Dict[str, Any]:
return {
"voice": self._a3.voice,
"language": self._a3.language,
"volume": self._a3.volume,
"speed": self._a3.speed,
}
def _dig(data: Any, dotted: str) -> Any:
node = data
for part in dotted.split("."):
if not isinstance(node, dict) or part not in node:
return None
node = node[part]
return node

View File

View File

@ -0,0 +1,257 @@
"""Saved speech audio.
Every line the cloud voice synthesises is written here as an ordinary `.wav`
file with a readable name, plus an `index.json` describing it. That gives three
things a hidden cache could not:
* **Instant replay.** A saved line plays with no synthesis, no network, no
wait - just read the file and play it.
* **Files you can actually use.** Open the folder, double-click a `.wav`, drop
one into a video edit or a stand's playlist. Nothing is locked in.
* **Works offline.** Once a line is saved, the demo no longer needs internet
for that line.
Layout:
audio_library/
index.json
good-afternoon-and-welcome-to-our-a1b2c3d4.wav
please-follow-me-to-the-first-9f8e7d6c.wav
The id is a hash of (text, voice, model, style), so the same line spoken by a
different voice is a different entry, and re-saving an identical line is free.
NOTE: only the cloud voice produces files. The built-in OS voices play straight
to the sound card and never hand us audio, so utterances spoken by them are not
saved. `MOCK_VOICE_ENGINE=gemini` is what fills this library.
"""
from __future__ import annotations
import hashlib
import json
import logging
import os
import re
import struct
import threading
import time
from pathlib import Path
from typing import Any, Dict, List, Optional
logger = logging.getLogger(__name__)
INDEX_FILE = "index.json"
_SLUG_STRIP = re.compile(r"[^a-z0-9]+")
def slugify(text: str, limit: int = 40) -> str:
"""A short, filesystem-safe, human-readable stem for a filename."""
ascii_text = text.encode("ascii", "ignore").decode("ascii").lower()
slug = _SLUG_STRIP.sub("-", ascii_text).strip("-")
if len(slug) > limit:
slug = slug[:limit].rsplit("-", 1)[0] or slug[:limit]
return slug or "utterance"
def wav_duration(path: Path) -> float:
"""Duration of a 16-bit PCM WAV, from its header."""
try:
with open(path, "rb") as handle:
header = handle.read(44)
if len(header) < 44:
return 0.0
channels, sample_rate = struct.unpack("<HI", header[22:28])
data_bytes = struct.unpack("<I", header[40:44])[0]
return round(data_bytes / float(max(1, sample_rate * channels * 2)), 2)
except Exception: # pragma: no cover
return 0.0
class AudioLibrary:
"""Saved `.wav` files plus their index. Safe to share across threads."""
def __init__(self, root: Path) -> None:
self.root = Path(root)
self.root.mkdir(parents=True, exist_ok=True)
self._lock = threading.RLock()
self._index: Dict[str, Dict[str, Any]] = {}
self._stamp: Optional[tuple] = None
self._load_index()
# -- identity ------------------------------------------------------------ #
@staticmethod
def make_id(text: str, voice: str, model: str, style: str = "") -> str:
raw = "\x00".join([model or "", voice or "", style or "", text or ""])
return hashlib.sha256(raw.encode("utf-8")).hexdigest()[:16]
# -- index --------------------------------------------------------------- #
@property
def index_path(self) -> Path:
return self.root / INDEX_FILE
def _stamp_of_index(self) -> Optional[tuple]:
try:
stat = self.index_path.stat()
return (stat.st_mtime_ns, stat.st_size)
except OSError:
return None
def _refresh(self) -> None:
"""Re-read the index if another process changed it.
`warm_voice.py` and the server both open this library. Without this the
server would keep a stale copy in memory and write it back over the
other process's changes - resurrecting entries whose files were deleted.
"""
if self._stamp_of_index() != self._stamp:
self._load_index()
def _load_index(self) -> None:
self._stamp = self._stamp_of_index()
if not self.index_path.exists():
self._index = {}
return
try:
data = json.loads(self.index_path.read_text(encoding="utf-8"))
entries = data.get("entries", data) if isinstance(data, dict) else {}
if isinstance(entries, dict):
# Drop entries whose file was deleted by hand.
self._index = {
key: value for key, value in entries.items()
if (self.root / value.get("file", "")).exists()
}
except (json.JSONDecodeError, OSError) as exc:
logger.warning("audio index unreadable (%s); starting a fresh one", exc)
self._index = {}
def _save_index(self) -> None:
payload = {"version": 1, "entries": self._index}
temporary = self.index_path.with_suffix(".tmp")
try:
temporary.write_text(
json.dumps(payload, indent=1, ensure_ascii=False), encoding="utf-8"
)
os.replace(temporary, self.index_path)
self._stamp = self._stamp_of_index()
except OSError as exc: # pragma: no cover
logger.warning("could not write audio index: %s", exc)
# -- lookup -------------------------------------------------------------- #
def get(self, audio_id: str) -> Optional[Dict[str, Any]]:
with self._lock:
self._refresh()
entry = self._index.get(audio_id)
if entry is None:
return None
if not (self.root / entry["file"]).exists():
self._index.pop(audio_id, None)
self._save_index()
return None
return dict(entry)
def find(self, text: str, voice: str, model: str, style: str = "") -> Optional[Dict[str, Any]]:
return self.get(self.make_id(text, voice, model, style))
def path_of(self, audio_id: str) -> Optional[Path]:
entry = self.get(audio_id)
return self.root / entry["file"] if entry else None
def list(self, newest_first: bool = True) -> List[Dict[str, Any]]:
with self._lock:
self._refresh()
# Only list clips whose file is really there - the panel offers a
# play button per row, and a row that 404s is worse than no row.
alive, missing = [], []
for key, value in self._index.items():
(alive if (self.root / value.get("file", "")).exists() else missing).append(
(key, value)
)
if missing:
for key, _ in missing:
self._index.pop(key, None)
self._save_index()
items = [dict(value) for _, value in alive]
items.sort(key=lambda item: item.get("createdAt", 0), reverse=newest_first)
return items
# -- writing ------------------------------------------------------------- #
def save(self, text: str, wav_bytes: bytes, voice: str, model: str,
style: str = "") -> Dict[str, Any]:
audio_id = self.make_id(text, voice, model, style)
with self._lock:
self._refresh()
existing = self.get(audio_id)
if existing is not None:
return existing
filename = "{0}-{1}.wav".format(slugify(text), audio_id[:8])
destination = self.root / filename
temporary = destination.with_suffix(".part")
temporary.write_bytes(wav_bytes)
os.replace(temporary, destination) # never index a half-written file
entry = {
"id": audio_id,
"text": text,
"file": filename,
"voice": voice,
"model": model,
"style": style,
"createdAt": time.time(),
"durationSeconds": wav_duration(destination),
"bytes": destination.stat().st_size,
}
with self._lock:
self._index[audio_id] = entry
self._save_index()
return dict(entry)
# -- housekeeping -------------------------------------------------------- #
def delete(self, audio_id: str) -> bool:
with self._lock:
self._refresh()
entry = self._index.pop(audio_id, None)
if entry is None:
return False
self._save_index()
try:
(self.root / entry["file"]).unlink()
except OSError: # pragma: no cover
pass
return True
def clear(self) -> int:
with self._lock:
self._refresh()
entries = list(self._index.values())
self._index = {}
self._save_index()
for entry in entries:
try:
(self.root / entry["file"]).unlink()
except OSError: # pragma: no cover
pass
return len(entries)
def stats(self) -> Dict[str, Any]:
items = self.list()
return {
"count": len(items),
"bytes": sum(item.get("bytes", 0) for item in items),
"seconds": round(sum(item.get("durationSeconds", 0) for item in items), 1),
"dir": str(self.root),
}
_library: Optional[AudioLibrary] = None
def get_audio_library(root: Optional[Path] = None) -> AudioLibrary:
"""Process-wide library, so the API and the voice share one index."""
global _library
if _library is None:
if root is None:
root = Path(__file__).resolve().parents[2] / "audio_library"
_library = AudioLibrary(root)
return _library

View File

@ -0,0 +1,63 @@
"""Speech history.
Deliberately in-memory and bounded: this is an operator convenience for a live
demo ("say that again"), not an audit log. It resets when the server restarts.
"""
from __future__ import annotations
import threading
import time
from collections import deque
from typing import Any, Deque, Dict, List, Optional
from ..robot.base import SpeechStage
class SpeechHistory:
def __init__(self, limit: int = 100) -> None:
self._items: Deque[Dict[str, Any]] = deque(maxlen=max(1, limit))
self._index: Dict[str, Dict[str, Any]] = {}
self._lock = threading.Lock()
def add(self, request_id: str, text: str) -> Dict[str, Any]:
entry: Dict[str, Any] = {
"id": request_id,
"text": text,
"at": time.time(),
"stage": SpeechStage.SENDING.value,
"success": None,
"ackLatencyMs": None,
"totalMs": None,
"error": None,
}
with self._lock:
if len(self._items) == self._items.maxlen:
# Newest is at the left, so the deque drops the rightmost entry.
self._index.pop(self._items[-1]["id"], None)
self._items.appendleft(entry)
self._index[request_id] = entry
return entry
def update(self, request_id: str, **fields: Any) -> Optional[Dict[str, Any]]:
with self._lock:
entry = self._index.get(request_id)
if entry is None:
return None
entry.update(fields)
return dict(entry)
def list(self) -> List[Dict[str, Any]]:
with self._lock:
return [dict(item) for item in self._items]
def clear(self) -> int:
with self._lock:
count = len(self._items)
self._items.clear()
self._index.clear()
return count
def __len__(self) -> int:
with self._lock:
return len(self._items)

View File

@ -0,0 +1,319 @@
"""Speech orchestration.
The one deliberate design decision worth stating: POST /api/robot/speak returns as
soon as the *robot has acknowledged* the utterance - not when speaking finishes.
The remaining lifecycle (speaking -> completed) streams over the WebSocket.
That matters for a live demo. Blocking the HTTP response until the robot stops
talking would make a 12-second sentence look like a 12-second-slow button, and it
would tie a browser request to the speaking duration for no benefit.
"""
from __future__ import annotations
import asyncio
import logging
import time
from typing import Any, Dict, Optional
from ..config.settings import Settings
from ..core.events import EventBus, EventType
from ..core.text import normalise
from ..robot.base import (
RobotBusy,
RobotError,
RobotUnreachable,
SpeechProgress,
SpeechRequest,
SpeechResult,
SpeechStage,
)
from ..robot.manager import RobotManager
from .history import SpeechHistory
logger = logging.getLogger(__name__)
class ValidationError(RobotError):
code = "invalid_text"
user_message = "Please enter some text for the robot to say."
def _swallow_exception(task: "asyncio.Task") -> None:
"""Retrieve a finished task's exception so asyncio stops complaining."""
if not task.cancelled():
task.exception()
class SpeechService:
def __init__(self, settings: Settings, manager: RobotManager, bus: EventBus) -> None:
self._settings = settings
self._manager = manager
self._bus = bus
self.history = SpeechHistory(settings.speech.history_limit)
self._lock = asyncio.Lock()
self._active_task: Optional[asyncio.Task] = None
self._active_id: Optional[str] = None
def update_settings(self, settings: Settings) -> None:
"""Adopt reloaded configuration (limits, interrupt policy) in place."""
self._settings = settings
# -- state --------------------------------------------------------------- #
@property
def is_busy(self) -> bool:
return self._active_task is not None and not self._active_task.done()
@property
def active_request_id(self) -> Optional[str]:
return self._active_id if self.is_busy else None
# -- validation ---------------------------------------------------------- #
def validate(self, text: Optional[str]) -> str:
cleaned = normalise(text or "")
limits = self._settings.speech
if len(cleaned) < max(1, limits.min_length):
raise ValidationError("empty text")
if len(cleaned) > limits.max_length:
raise ValidationError(
"text too long: {0} chars".format(len(cleaned)),
user_message=(
"Text is too long ({0} characters). The limit is {1}. "
"Split it into shorter sentences.".format(len(cleaned), limits.max_length)
),
)
return cleaned
# -- speak --------------------------------------------------------------- #
async def speak(self, text: Optional[str], voice: Optional[str] = None,
language: Optional[str] = None) -> Dict[str, Any]:
cleaned = self.validate(text)
if not self._manager.is_connected:
status = self._manager.status_dict()
raise RobotUnreachable(
"robot not connected (state={0})".format(status["state"]),
user_message=(
status.get("error")
or "Robot is offline. Check the robot IP address and network connection."
),
)
async with self._lock:
if self.is_busy:
if not self._settings.speech.allow_interrupt:
raise RobotBusy()
logger.info("interrupting utterance %s", self._active_id)
await self._cancel_active()
request = SpeechRequest(text=cleaned, voice=voice, language=language)
entry = self.history.add(request.id, cleaned)
self._bus.publish(EventType.HISTORY_UPDATED, {"entry": entry, "action": "add"})
ack = asyncio.Event()
state: Dict[str, Any] = {"ackLatencyMs": None, "stage": SpeechStage.SENDING}
async def on_progress(progress: SpeechProgress) -> None:
state["stage"] = progress.stage
if progress.stage is not SpeechStage.SENDING:
if state["ackLatencyMs"] is None:
state["ackLatencyMs"] = progress.elapsed_ms
ack.set()
self._bus.publish(EventType.SPEECH_PROGRESS, progress.to_dict())
self.history.update(request.id, stage=progress.stage.value)
started = time.perf_counter()
self._active_id = request.id
self._active_task = asyncio.create_task(
self._run(request, on_progress), name="speak-" + request.id
)
# The HTTP call usually returns before the task finishes, so nobody is
# left to await it. Consume any exception here to keep asyncio from
# logging "Task exception was never retrieved" - it is already
# reported to the browser as a speech.progress / speech.result event.
self._active_task.add_done_callback(_swallow_exception)
# Wait only for the acknowledgement, then hand control back to the UI.
ack_wait = asyncio.ensure_future(ack.wait())
done, _ = await asyncio.wait(
{ack_wait, self._active_task},
timeout=self._settings.robot.request_timeout + 2,
return_when=asyncio.FIRST_COMPLETED,
)
if not ack_wait.done():
ack_wait.cancel()
# The utterance failed before it was ever acknowledged - surface it now
# as an HTTP error rather than only as a WebSocket event.
if self._active_task in done:
exc = self._active_task.exception()
if exc is not None:
raise exc
result = self._active_task.result()
return self._response(request, result.stage, result.ack_latency_ms, started)
if not ack.is_set():
await self._cancel_active()
raise RobotError(
"no acknowledgement within timeout",
user_message="The robot did not respond in time. The request was cancelled.",
)
return self._response(request, state["stage"], state["ackLatencyMs"], started)
def _response(self, request: SpeechRequest, stage: Any, ack_ms: Optional[int],
started: float) -> Dict[str, Any]:
stage_value = stage.value if hasattr(stage, "value") else str(stage)
return {
"success": stage_value not in (SpeechStage.FAILED.value, SpeechStage.CANCELLED.value),
"status": stage_value,
"requestId": request.id,
"text": request.text,
"ackLatencyMs": ack_ms if ack_ms is not None else int((time.perf_counter() - started) * 1000),
}
async def _run(self, request: SpeechRequest, on_progress) -> SpeechResult:
"""Drive the utterance to completion and record the outcome."""
try:
result = await self._manager.adapter.speak(request, on_progress)
except asyncio.CancelledError:
entry = self.history.update(
request.id, stage=SpeechStage.CANCELLED.value, success=False, error="Stopped."
)
self._publish_history(entry)
raise
except RobotError as exc:
logger.warning("speech %s failed: %s", request.id, exc)
await on_progress(
SpeechProgress(
request.id,
SpeechStage.FAILED,
exc.user_message,
error_code=exc.code,
)
)
entry = self.history.update(
request.id,
stage=SpeechStage.FAILED.value,
success=False,
error=exc.user_message,
)
self._publish_history(entry)
self._bus.publish(
EventType.SPEECH_RESULT,
{
"requestId": request.id,
"success": False,
"stage": SpeechStage.FAILED.value,
"errorCode": exc.code,
"error": exc.user_message,
},
)
raise
except Exception as exc: # pragma: no cover - defensive
logger.exception("unexpected speech failure")
await on_progress(
SpeechProgress(request.id, SpeechStage.FAILED, "Speech request failed.",
error_code="internal", detail=str(exc))
)
entry = self.history.update(
request.id, stage=SpeechStage.FAILED.value, success=False,
error="Speech request failed.",
)
self._publish_history(entry)
raise
entry = self.history.update(
request.id,
stage=result.stage.value,
success=result.success,
ackLatencyMs=result.ack_latency_ms,
totalMs=result.total_ms,
error=result.error,
)
self._publish_history(entry)
self._bus.publish(EventType.SPEECH_RESULT, result.to_dict())
return result
def _publish_history(self, entry: Optional[Dict[str, Any]]) -> None:
if entry is not None:
self._bus.publish(EventType.HISTORY_UPDATED, {"entry": entry, "action": "update"})
# -- stop ---------------------------------------------------------------- #
async def stop(self) -> Dict[str, Any]:
stopped_robot = False
try:
stopped_robot = await self._manager.adapter.stop_speaking()
except Exception as exc:
logger.warning("robot stop failed: %s", exc)
had_active = self.is_busy
if had_active:
# Give the adapter a moment to unwind cleanly before force-cancelling.
try:
await asyncio.wait_for(asyncio.shield(self._active_task), timeout=1.5)
except (asyncio.TimeoutError, Exception):
await self._cancel_active()
return {
"success": True,
"status": "stopped",
"hadActiveSpeech": had_active,
"robotAcknowledged": stopped_robot,
}
async def _cancel_active(self) -> None:
task = self._active_task
if task is None or task.done():
self._active_task = None
return
task.cancel()
try:
await task
except (asyncio.CancelledError, Exception):
pass
finally:
self._active_task = None
# -- history ------------------------------------------------------------- #
def annotate_saved(self, items):
"""Mark history entries whose audio is already on disk.
Lets the dashboard show a replay button only where playback really is
instant, instead of promising something it cannot deliver.
"""
try:
from .audio_library import get_audio_library
library = get_audio_library()
mock = self._settings.mock
if mock.voice_engine != "gemini":
return items
from ..core.pronunciation import build as build_pronouncer
from ..config.settings import PROJECT_ROOT
from ..robot.gemini_voice import split_sentences
pronouncer = build_pronouncer(PROJECT_ROOT, mock.pronunciation)
for item in items:
text = item.get("text") or ""
if pronouncer is not None:
text = pronouncer.apply(text)
chunks = split_sentences(text)
entries = [
library.find(c, mock.gemini_voice, mock.gemini_model, mock.gemini_style)
for c in chunks
]
found = [e for e in entries if e]
item["audioSaved"] = bool(chunks) and len(found) == len(chunks)
item["audioIds"] = [e["id"] for e in found]
except Exception: # pragma: no cover - never break history over this
logger.debug("could not annotate saved audio", exc_info=True)
return items
def clear_history(self) -> Dict[str, Any]:
removed = self.history.clear()
self._bus.publish(EventType.HISTORY_UPDATED, {"action": "clear"})
return {"success": True, "removed": removed}
async def shutdown(self) -> None:
await self._cancel_active()

View File

@ -0,0 +1,382 @@
# AGIBOT A3 — how the speech integration works
**Short answer: the A3 has a documented, LAN-callable, native text-to-speech HTTP
endpoint, and this project talks to it directly.** No SDK, no login, no ROS, no
audio file transfer.
```
Browser → this backend → HTTP JSON-RPC → A3 head unit (HDU) → built-in TTS → speaker
```
Everything below is sourced. Claims that could not be verified are tagged
**[UNVERIFIED]** or **[UNKNOWN]** rather than smoothed over, and anything found on
a *different* AgiBot model is labelled as such — an A2/X2/A3-Ultra fact is not an
A3 fact.
> Research date: **2 September 2026**, against AgiBot doc tree **A3 v3.1 / v3.2**.
> If your unit's firmware differs, re-run the discovery procedure in §6.
---
## 1. The interface
AgiBot's A3 developer guide documents two developer transports on the robot:
| Transport | AgiBot's stated use | Notes |
| --- | --- | --- |
| **HTTP JSON-RPC** | "low-frequency, many-to-one calls" | `POST http://{IP}:{PORT}/rpc/{service}/{method}`, `Content-Type: application/json` **mandatory** |
| **ROS 2 topic** | "high-frequency, many-to-many calls" | ROS 2 Jazzy on Fast DDS; default QoS `keep_last`/depth 10/**`best_effort`** |
Source: [03-second_develop_interface_overview](https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/03-second_develop_interface_overview)
The same page explicitly sanctions calling it from another machine:
> *"For remote calls, replace it with the reachable IP of the corresponding
> device in the current network."*
> *"…there are no restrictions [on language]. As long as you can send and receive
> HTTP requests, you can call all HTTP JSON RPC interfaces."*
**There is no documented gRPC or WebSocket interface on the robot.** (WebSocket
appears only in AgiBot's separate *cloud* LinkSoul agent SDK — see §3.)
### The TTS call
On the **HDU** (the head unit — it owns all audio), port **59301**:
```http
POST http://<ROBOT_IP>:59301/rpc/aimdk.protocol.TTSService/PlayTTS
Content-Type: application/json
{"text": "Hello, welcome to our showroom.",
"priority_level": "INTERACTION_L6",
"domain": "voice_control",
"trace_id": "abc123",
"is_interrupted": true}
```
```json
{"text": "...", "priority_level": "INTERACTION_L6", "priority_weight": 0,
"domain": "voice_control", "trace_id": "abc123_18bZZLTk5VfJGSy8Cylsu4",
"is_sucess": true, "error_message": "", "estimated_duration": 0}
```
Source: [07-02-audio_play](https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/07-02-audio_play)
· runnable `curl` in AgiBot's own [quick start](https://open.agibot.com/docs/en/aimdk/a3/v3_1/dev_guide/06-second_develop_quick_start)
Companion methods on the same host and port:
| Method | Body | Purpose |
| --- | --- | --- |
| `StopTTSTraceId` | `{"trace_id": "..."}` | interrupt an utterance |
| `GetAudioStatus` | `{"trace_id": "..."}` | `TTSStatusType_Begin\|_Playing\|_End\|_Stop\|_Error\|_InQue\|_NOTInQue` |
| `PlayMediaFile` | `{"file_name": "..."}` | play a file from `/agibot/data/var/interaction/audio/` |
Volume lives on a **different port, 56666**, service `aimdk.protocol.HalAudioService`
(`GetAudioVolume` / `SetAudioVolume`, range 0100).
> ⚠️ **AgiBot's own hardware warning: do not set volume above 70 — "exceeding
> this range may damage the speaker."** This project does not change volume, so
> it cannot trip that. If you add volume control later, clamp it at 70.
---
## 2. Four traps, and how this project handles each
These are the things that make a naive client fail. All four are handled in
[backend/robot/transports/aimdk_transport.py](../backend/robot/transports/aimdk_transport.py).
| # | The trap | What we do |
| --- | --- | --- |
| 1 | The success flag is spelled **`is_sucess`** — one `c` — in AgiBot's documentation. | Read `is_sucess` **and** `is_success`, so a corrected firmware still works. |
| 2 | The **returned `trace_id` is not the one you sent** — the robot appends a random suffix. Stop only works with the returned value. | The returned id is captured and is what `Stop` uses. |
| 3 | **1024 *bytes*** of UTF-8 is a hard limit (~200 characters). A 1000-character Chinese sentence is ~3000 bytes and is rejected outright. | Text is chunked on sentence boundaries by **byte** length; only the first chunk may interrupt, the rest queue behind it. |
| 4 | Status polling is capped at **≤ 0.2 Hz**; AgiBot warns high-frequency RPC can destabilise the robot. | The liveness probe is a **TCP connect, never an RPC**, and completion is estimated rather than polled. |
Two more behaviours worth knowing, from reading AimRT's own source
([http_rpc_backend.cc](https://raw.githubusercontent.com/AimRT/AimRT/main/src/plugins/net_plugin/http/http_rpc_backend.cc)):
- The route is `"/rpc" + <func name without prefix>` — the `pb:/` prefix seen in
AimRT function names is **stripped**. Never put `pb:/` in a URL.
- **An RPC-level failure is HTTP 500, not a JSON error.** Unknown route is 404.
**HTTP 200 means the transport worked, not that the robot spoke**
`is_sucess` only confirms the priority check passed.
---
## 3. Why this path, and not the others
| Path | Latency | Reliability | Complexity | Verdict |
| --- | --- | --- | --- | --- |
| **1. HTTP JSON-RPC `PlayTTS`** (this project) | 1 request RTT to enqueue (< 20 ms LAN); time-to-first-audio dominated by the robot's own synthesis | Highest AgiBot's own primary documented path | Near zero one HTTP POST | ** Chosen** |
| 2. Own TTS → WAV → SCP → `PlayFile` | 14 s cold; ~1 RTT for pre-cached phrases | Good; documented offline fallback | Moderate — SSH creds, file lifecycle | Fallback if #1 needs internet and you are air-gapped |
| 3. Own TTS → PCM → ROS 2 `/audiohal/audio/playback` | Best possible — tens of ms, true streaming | High ceiling, high risk (best-effort QoS, focus arbitration) | **High** — ROS 2 Jazzy on the PC, login-gated aarch64 packages, robot reboot into `only_voice` | Only if you need barge-in streaming |
| 4. LinkSoul cloud agent SDK (WebSocket) | Worst — PC → AgiBot cloud → robot | n/a | Commercial credentials, cloud dependency | Wrong tool: a conversational agent layer, not a "say this now" primitive |
| 5. SSH + direct ALSA write | — | — | — | **Do not ship.** No AgiBot document mentions ALSA. `hal_audio` owns the device behind an explicit focus/priority protocol; bypassing it also bypasses the volume damage limit. |
Path 1 wins on every axis the brief asked for: lowest latency, highest
reliability, simplest deployment, best-documented A3 compatibility, and it
scales because it is stateless HTTP.
**Critically, it sends *text*.** The robot synthesises. Nothing is rendered,
encoded or transferred by the PC — which is exactly the architecture the
"fast speech" requirement asks for.
---
## 4. What this project does with it
`A3_TRANSPORT=aimdk` (the default) selects
[aimdk_transport.py](../backend/robot/transports/aimdk_transport.py):
- one **keep-alive** `httpx` client opened at startup and reused for every
utterance — no handshake per Speak;
- a `Timeout:` header (AimRT honours whole seconds) **and** a client-side
timeout, so a wedged robot cannot hang the dashboard;
- ack latency measured and reported to the UI;
- `Stop` via `StopTTSTraceId` using the robot-returned `trace_id`;
- errors mapped to messages an operator can act on.
**Every documented value is a setting, not a literal**, because AimRT has no
service-discovery endpoint and AgiBot does not guarantee port or name stability
across firmware:
```env
A3_AIMDK_SERVICE=aimdk.protocol.TTSService
A3_AIMDK_PLAY_METHOD=PlayTTS
A3_AIMDK_STOP_METHOD=StopTTSTraceId
A3_AIMDK_PRIORITY=INTERACTION_L6
A3_AIMDK_MAX_BYTES=1024
```
If your firmware renames something, you edit `.env`. If it uses a wholly
different shape, `A3_TRANSPORT=http` gives you a fully templated HTTP client, and
`ws` / `ros2` / `ssh` are there as further fallbacks.
---
## 5. Connecting the robot — the exact steps
### 5.1 Where the IP goes
**One file: `.env`, in the project root.** Three lines:
```env
ROBOT_MODE=real
ROBOT_IP=192.168.1.50 ← the robot's IP on YOUR network
ROBOT_PORT=59301
```
> **Do not use `10.42.10.10`.** That address appears throughout AgiBot's
> examples, but it is the HDU's address on the robot's *internal* `eth_hdu`
> network. Your PC cannot reach it. Use the HDU's address on your WiFi/LAN.
Then either restart the server, or without restarting:
```bash
curl -X POST http://localhost:8000/api/config/reload
```
### 5.2 The order to do things in
```bash
# 1. reachable at all?
ping 192.168.1.50
# 2. what does it expose? (probes the documented ports and routes, read-only)
python scripts/discover_robot.py 192.168.1.50
# 3. the decisive test - does it actually speak?
python scripts/discover_robot.py 192.168.1.50 --speak "Hello, I am Expedition A3"
# 4. point the app at it, then
python scripts/selftest.py
```
### 5.3 What each failure means
| Result of step 3 | Meaning | Next step |
| --- | --- | --- |
| **The robot speaks** | Path 1 works. Done. | Set `.env`, restart, demo. |
| Connection refused / timeout | 59301 is not bound on the WLAN interface | §6 Step 2 — SSH in and check `ss -ltnp` |
| **HTTP 404** | Port is right, route is wrong | Check `A3_AIMDK_SERVICE` / `A3_AIMDK_PLAY_METHOD`; confirm firmware version |
| **HTTP 500** | Route exists, the RPC failed — **the port is reachable** | Read the response body; check `priority_level` |
| `is_sucess: false` | The robot rejected it | Read `error_message` — usually priority or text length |
| 200 + success but silence | Accepted but not played | Check robot volume, and whether TTS needs internet (§7) |
---
## 6. Discovery procedure for the live robot
`scripts/discover_robot.py` automates the PC-side checks. The rest needs an SSH
session on the robot.
```bash
# From the PC - which of the documented ports answer?
python scripts/discover_robot.py <ROBOT_IP>
# Equivalent by hand. A 404 proves an AimRT HTTP server is listening;
# connection-refused proves it is not.
for p in 59301 56666 51049; do
curl -s -o /dev/null -w "$p -> %{http_code}\n" -m 3 \
"http://<ROBOT_IP>:$p/rpc/does.not.Exist/Nope"
done
```
On the robot (username `agi` is **[UNVERIFIED]** — confirm with AgiBot):
```bash
# Is the RPC server bound to 0.0.0.0 or only to the internal 10.42.10.10?
ss -ltnp | grep -E '59301|56666|51049'
ip -br a # the HDU's real WLAN address
# The authoritative port map, from firmware config
grep -rn 'listen_port' /agibot/software/v0/ 2>/dev/null | head -50
# ROS environment (resolves ROS_DOMAIN_ID, only needed for the ros2 transport)
source /agibot/software/v0/entry/env/env.sh || source /agibot/software/v0/entry/env.sh
echo "DOMAIN=$ROS_DOMAIN_ID RMW=$RMW_IMPLEMENTATION"
# The real audio surface — this substitutes for the missing discovery API
ros2 topic list | grep -Ei 'audio|tts|interaction'
ros2 topic echo --once /interaction/tts_status # while a PlayTTS is running
```
> **Do not put any of these in a loop.** AgiBot's
> [constraints page](https://open.agibot.com/docs/aimdk/a3/v3_2/dev_guide/04-second_develop_constraints_overview)
> caps status RPCs at ≤ 0.2 Hz, forbids deploying on the MDU ("risks the robot
> falling"), forbids `apt` install/uninstall, and states violations "may void
> warranty."
---
## 7. Open questions — send these to AgiBot
The five that actually affect this project, in priority order. (Chinese versions
follow, since AgiBot support answers faster in Chinese.)
1. **Is port 59301 on the HDU bound to all interfaces and reachable from a PC on
the same WiFi/LAN using the HDU's WLAN IP — or only on the internal
`eth_hdu` 10.42.10.x segment?**
*HDU 的 59301 端口是否监听在所有网络接口上?能否从同一局域网的 PC 通过 HDU 的 WLAN IP 访问?还是仅绑定在内部 eth_hdu (10.42.10.x) 网段?*
2. **Does `TTSService/PlayTTS` require internet access?** Your English v3.2 page
says it "requires network connection". Which engine performs synthesis, and is
there a fully on-device/offline TTS mode? What happens with LAN but no internet?
*`PlayTTS` 是否必须联网?语音合成由哪家引擎提供?是否有完全离线的端侧 TTS 方案?只有局域网、没有外网时会怎样?*
3. **Is there any authentication, token or IP allow-list on the `/rpc/`
endpoints?** If not, what do you recommend on a shared network? (AimRT sets
`Access-Control-Allow-Origin: *` unconditionally, so any page on the LAN can
call these.)
*`/rpc/` 接口是否有鉴权、token 或 IP 白名单?如果没有,在共享网络下推荐如何防护?*
4. **Can we select voice/timbre, language, speaking rate or emotion per
`PlayTTS` call?** Is `SetTtsParameters` or an equivalent available on the A3,
and at which endpoint? **Our unit's voice is currently set to "Yunxiao"
(teenager, male, multi-language) — is that selectable per request, or only in
the console?**
*能否在每次 PlayTTS 调用时指定音色、语言、语速或情感A3 是否提供 SetTtsParameters接口地址是什么
我们的机器人当前音色为"Yunxiao"(少年、男声、多语言),该音色能否在每次调用时指定,还是只能在控制台配置?*
> **Operator-reported, 2026-09-02:** the robot's configured voice is
> **"Yunxiao"** — labelled *teenager | male*, *multi-language*, and marked as
> the default (默). It was shown in a voice-library UI offering per-voice
> "Speech Synthesis", which is consistent with AgiBot's LinkSoul timbre
> library (3040 timbres + voice cloning) — though that link is inference, not
> something any published document states. This is **not** in AgiBot's docs;
> it comes from the console for this specific unit. It suggests voice
> selection happens on a platform/console rather than in the `PlayTTS` body,
> which is exactly what question 4 needs to settle.
5. **Please confirm the complete `priority_level` enum** accepted by A3
`PlayTTS`, and the semantics of `priority_weight` and `is_interrupted`. Your
X2 documentation publishes seven levels — do the same apply to the A3?
*请提供 A3 PlayTTS 支持的完整 priority_level 枚举,以及 priority_weight 和 is_interrupted 的语义。X2 文档公布了七个等级A3 是否相同?*
Lower priority, worth asking in the same message: the response envelope
(is there a `header.code`/`header.msg` wrapper?); whether the AimDK SDK download
needs a verified-customer account; whether ports 59301/56666/51049 are stable
across firmware; and the `PlayMediaFile` sample rate — **your English v3.2 page
says 24 kHz and your Chinese v3.2 page says 16 kHz for the same call.**
---
## 8. Known contradictions in AgiBot's own documentation
Recorded so nobody wastes an afternoon on them.
| # | Contradiction | Impact here |
| --- | --- | --- |
| 1 | `PlayMediaFile` sample rate: **EN v3.2 says 24 kHz, ZH v3.2 says 16 kHz**, same page | None — we send text, not files |
| 2 | Microphone: v3.1 and v3.2 dev guides say the built-in mic is "not yet configured or enabled"; the User Manual and product page advertise an 8-mic array | None — output only |
| 3 | `env.sh` path: `/agibot/software/v0/entry/env.sh` vs `…/entry/env/env.sh` | Only matters for the `ros2` transport |
| 4 | ROS 2 distro: AgiBot says the A3 is **Jazzy**; AimRT's `ros2_plugin` doc says Humble | Only matters for the `ros2` transport |
| 5 | SDK wheel: the A3 v3.2 guide instructs `a2_aimdk-3.0.0…whl` in most places and `a3_aimdk-3.0.0…whl` in one — an **A2 artifact inside the A3 doc tree** | None — Path 1 needs no SDK |
**`ROS_DOMAIN_ID` for the A3 is [UNKNOWN]** — it is not printed in any A3
document. A2 uses 232 (**different model**), and one third-party A3-Ultra
deployment also uses 232 — suggestive, not proof. Read it from `env.sh` on the
robot.
---
## 9. Security note worth raising internally
No authentication scheme is documented for the `/rpc/` endpoints, and AimRT sets
`Access-Control-Allow-Origin: *` unconditionally with OPTIONS preflight handled.
**Taken together: any host — or any web page opened in any browser — on the same
LAN as the robot can make it speak.**
This project does not widen that exposure. The browser talks only to
`127.0.0.1:8000`; robot credentials and addresses stay server-side. But on a
shared or public network, treat the robot's RPC ports as unauthenticated and
segment the network accordingly.
---
## 10. Testing the real path before the robot arrives
`ROBOT_MODE=mock` tests the application. To test **this transport** — the URL
shape, the JSON body, the `trace_id` round trip, chunking, Stop, and the error
paths — run the local stand-in:
```bash
python scripts/fake_a3_server.py # implements the documented contract on :59301
```
then in `.env`:
```env
ROBOT_MODE=real
ROBOT_IP=127.0.0.1
ROBOT_PORT=59301
```
and use the dashboard normally. Every utterance prints in the stand-in's console.
> It is a **test double written from public documentation**, not the robot.
> Passing against it proves the client is well-formed — not that the firmware
> behaves identically. Verify on the real unit.
---
## 11. Sources
**Official AgiBot, A3-specific**
- Audio interfaces (TTS, status, stop, volume, ROS 2 audio) — [EN](https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/07-02-audio_play) · [ZH](https://open.agibot.com/docs/aimdk/a3/v3_2/dev_guide/07-02-audio_play)
- Interface overview (the two transports, URL shape, remote-call sanction) — [EN](https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/03-second_develop_interface_overview)
- Quick start (the runnable `PlayTTS` curl) — [EN](https://open.agibot.com/docs/en/aimdk/a3/v3_1/dev_guide/06-second_develop_quick_start)
- Resource manager (`PlayFile`, WAV/PCM rules, resource dirs) — [EN](https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/07-05-resource_manager)
- Interactive guide (`audio_msgs/msg/AudioPlayback`, focus services, `only_voice`) — [EN](https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/08-interactive_guide)
- Development constraints (polling limits, MDU, apt, warranty) — [ZH](https://open.agibot.com/docs/aimdk/a3/v3_2/dev_guide/04-second_develop_constraints_overview)
- Hardware/software overview (HDU/MDU, internal IPs, Ubuntu 24.04) — [ZH](https://open.agibot.com/docs/aimdk/a3/v3_1/dev_guide/01-a3_overview)
- AimDK docs root — https://open.agibot.com/docs/aimdk (SDK downloads are login-gated)
**AimRT framework** (transport semantics, read from source)
- [net_plugin docs](https://docs.aimrt.org/tutorials/plugins/net_plugin.html) · [http_rpc_backend.cc](https://raw.githubusercontent.com/AimRT/AimRT/main/src/plugins/net_plugin/http/http_rpc_backend.cc)
**Different model — do not assume it applies to the A3**
- X2 voice interface, full 7-value priority enum — https://x2-aimdk.agibot.com/en/dev/Interface/interactor/voice.html
**Third-party corroboration** (an A3 **Ultra** deployment, not a specification)
- [tzj-hub/agibot_A3_sells_cars](https://github.com/tzj-hub/agibot_A3_sells_cars) — same base URL, methods and payload shape; reads `is_sucess` with an `is_success` fallback; its `a3_interfaces.py` is the only source anywhere for the `header.code`/`header.msg` error envelope, which is why this project reads that envelope defensively rather than depending on it.

196
docs/NETWORK.md Normal file
View File

@ -0,0 +1,196 @@
# Network setup: PC ↔ AGIBOT A3
Everything here is standard LAN networking. Nothing in this file assumes anything
about the A3's software — that lives in
[AGIBOT_A3_INTEGRATION.md](AGIBOT_A3_INTEGRATION.md).
```
┌──────────────┐ ┌──────────────┐
│ Your PC │ │ AGIBOT A3 │
│ │ same LAN / subnet │ │
│ browser │ ───────────────────────► │ speech svc │
│ ↓ │ 192.168.x.x │ ↓ │
│ backend │ │ speaker │
└──────────────┘ └──────────────┘
```
The PC and the robot must be able to reach each other by IP. That is the whole
requirement.
---
## 1. Put both machines on the same network
Ranked by how well they work in practice:
| Setup | Latency | Notes |
| --- | --- | --- |
| **Wired Ethernet, same switch** | best, ~0.21 ms | Ideal for a demo. Nothing to go wrong. |
| **Both on the same Wi-Fi AP** | ~220 ms, variable | Fine, but a crowded venue Wi-Fi is the #1 cause of a bad demo. |
| **PC wired, robot Wi-Fi (same subnet)** | mixed | Works. Check they are on the same subnet, not two VLANs. |
| **Direct cable PC ↔ robot** | best | Needs static IPs on both, or link-local. Useful when there is no venue network. |
| Different subnets / guest Wi-Fi | — | Usually blocked. Guest networks isolate clients from each other. |
> **Demo advice:** if you can run a cable, run a cable. Client isolation on a
> conference Wi-Fi will silently block PC→robot traffic while both devices show
> "connected to the internet".
---
## 2. Find the robot's IP address
Any of these, easiest first:
1. **The robot's own screen / app / teach pendant** — usually shows the IP in a
network or system settings page. Most reliable.
2. **Your router's DHCP client list** — log into the router (often
`192.168.1.1`), look for a newly-connected device.
3. **Scan the subnet from the PC.** First find your own subnet:
```powershell
ipconfig # look at IPv4 Address, e.g. 192.168.1.23
```
Then sweep it:
```powershell
# Windows: ping every host, then read the ARP table
1..254 | ForEach-Object { Start-Process -WindowStyle Hidden ping "192.168.1.$_" -ArgumentList "-n 1 -w 200" }
Start-Sleep 5
arp -a
```
```bash
# if you have nmap (any OS) - much better
nmap -sn 192.168.1.0/24
```
4. **mDNS**, if the robot advertises itself:
```bash
ping agibot-a3.local
```
A hostname works anywhere this project asks for `ROBOT_IP`.
> Ask whoever commissions the robot to give it a **DHCP reservation** (a fixed
> IP tied to its MAC address). Otherwise the IP can change on reboot and you will
> be editing `.env` before every demo.
---
## 3. Test reachability
```bash
ping 192.168.1.50
```
| Result | Meaning |
| --- | --- |
| Replies with a time | The robot is reachable. Continue. |
| `Request timed out` | Wrong IP, robot off, different subnet, or ICMP blocked (see below). |
| `Destination host unreachable` | No route — you are on a different subnet. |
**ICMP being blocked does not mean the robot is unreachable.** Some robots drop
ping but still answer on their service port. Test the port directly:
```powershell
Test-NetConnection 192.168.1.50 -Port 8080
```
```bash
nc -vz 192.168.1.50 8080 # Linux/macOS
```
Then let this project probe it properly:
```bash
python scripts/discover_robot.py 192.168.1.50
```
---
## 4. Configure this app
In `.env`:
```env
ROBOT_MODE=real
ROBOT_IP=192.168.1.50
ROBOT_PORT=8080
```
Restart the server, or:
```bash
curl -X POST http://localhost:8000/api/config/reload
```
The header pill turns green within a few seconds if the connection succeeds. If
it does not, `GET /api/robot/diagnostics` says exactly what was attempted and
what failed.
---
## 5. Firewall
The PC makes **outbound** connections to the robot, so a Windows Firewall inbound
rule is usually unnecessary. Two exceptions:
- **`HOST=0.0.0.0`** — you want to open the dashboard from another device.
Windows will prompt to allow Python on private networks the first time; allow
it for *Private* networks only, never *Public*.
- **ROS 2 / DDS transport** — DDS uses multicast discovery and a wide range of
UDP ports **inbound** to the PC. This is the one case where the firewall
usually needs a rule:
```powershell
# run as Administrator, only if you use A3_TRANSPORT=ros2
New-NetFirewallRule -DisplayName "ROS2 DDS" -Direction Inbound -Protocol UDP `
-LocalPort 7400-7600 -Action Allow -Profile Private
```
DDS discovery is multicast and does **not** cross subnets or most Wi-Fi APs.
ROS 2 realistically requires the PC and robot on the same wired L2 segment,
with the same `ROS_DOMAIN_ID`.
---
## 6. Latency expectations
Measured by this app and shown in the header. What to expect on a healthy LAN:
| Hop | Typical |
| --- | --- |
| Browser → local backend | < 2 ms (loopback) |
| Backend → robot, wired | 0.53 ms |
| Backend → robot, Wi-Fi | 230 ms, occasionally spiking |
| Robot's own TTS synthesis start | the dominant term — tens to hundreds of ms |
The network is almost never the bottleneck; the robot's speech synthesis is. That
is exactly why this project sends **text** and lets the robot synthesise, instead
of generating audio on the PC and transferring it.
If the header shows latency above ~50 ms on Wi-Fi, move to Ethernet before
blaming the software.
---
## 7. Troubleshooting
| Symptom | Cause | Fix |
| --- | --- | --- |
| Pill stuck on *Connecting…* | Wrong IP or port, robot booting | `ping`, then `discover_robot.py` |
| `Robot is offline` immediately | Nothing listening on `ROBOT_PORT` | Confirm the port with the discovery script |
| Ping works, app says offline | Right host, wrong port or path | Check `ROBOT_PORT` and `A3_HTTP_SPEAK_PATH` |
| Works, then drops after minutes | Wi-Fi roaming, DHCP lease change | Wired connection + DHCP reservation |
| Connects but nothing is heard | Robot volume, muted speaker, wrong audio sink | Check the robot's own volume first |
| Fine on the bench, fails at the venue | Client isolation on guest Wi-Fi | Bring your own switch/router |
| ROS 2 sees no topics | Different `ROS_DOMAIN_ID`, multicast blocked | Same domain ID, same wired segment |
Two commands answer most questions:
```bash
python scripts/discover_robot.py <ROBOT_IP> # what the robot exposes
curl http://localhost:8000/api/robot/diagnostics # what this app tried and saw
```

168
frontend/index.html Normal file
View File

@ -0,0 +1,168 @@
<!DOCTYPE html>
<html lang="en" data-loading="true">
<head>
<meta charset="utf-8" />
<meta name="viewport" content="width=device-width, initial-scale=1" />
<meta name="color-scheme" content="dark" />
<title>AGIBOT A3 · Voice Control</title>
<link rel="icon" href="data:image/svg+xml,<svg xmlns='http://www.w3.org/2000/svg' viewBox='0 0 100 100'><text y='.9em' font-size='90'>🤖</text></svg>" />
<link rel="stylesheet" href="/styles/main.css" />
</head>
<body>
<!-- ================= HEADER ================= -->
<header class="topbar">
<div class="topbar__brand">
<span class="brand__mark" aria-hidden="true">
<svg viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="1.6" stroke-linecap="round" stroke-linejoin="round">
<rect x="4" y="8" width="16" height="12" rx="3" />
<path d="M12 4v4" /><circle cx="12" cy="3" r="1.4" />
<circle cx="9" cy="14" r="1.2" fill="currentColor" stroke="none" />
<circle cx="15" cy="14" r="1.2" fill="currentColor" stroke="none" />
<path d="M9.5 17.5h5" />
</svg>
</span>
<div class="brand__text">
<h1>AGIBOT A3</h1>
<p>Voice Control</p>
</div>
<span class="mode-badge" id="modeBadge" title="Robot mode">--</span>
</div>
<div class="topbar__status">
<!-- starts amber, not red: before the socket opens we genuinely do not
know the robot's state, and a red flash on every load reads as a fault -->
<div class="conn" id="connPill" data-state="connecting" title="Robot connection">
<span class="conn__dot" aria-hidden="true"></span>
<span class="conn__label" id="connLabel">Starting…</span>
</div>
<div class="chip" id="latencyChip" hidden title="Round-trip time to the robot">
<svg viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" aria-hidden="true"><path d="M13 2 4 14h7l-1 8 9-12h-7l1-8Z"/></svg>
<span id="latencyValue"></span>
</div>
<button class="icon-btn" id="reconnectBtn" title="Reconnect to robot" aria-label="Reconnect to robot">
<svg viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true">
<path d="M21 12a9 9 0 1 1-2.64-6.36" /><path d="M21 3v6h-6" />
</svg>
</button>
</div>
</header>
<!-- ================= CONFIG BANNER ================= -->
<div class="banner" id="configBanner" hidden role="alert">
<svg viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" aria-hidden="true"><path d="M12 9v4"/><path d="M12 17h.01"/><path d="M10.3 3.9 1.8 18a2 2 0 0 0 1.7 3h17a2 2 0 0 0 1.7-3L13.7 3.9a2 2 0 0 0-3.4 0Z"/></svg>
<div class="banner__body"><strong id="bannerTitle">Configuration</strong><span id="bannerText"></span></div>
</div>
<!-- ================= MAIN ================= -->
<main class="layout">
<!-- ---------- speech panel ---------- -->
<section class="panel panel--speech" aria-label="Speech">
<div class="panel__head">
<h2>Say something</h2>
<span class="panel__hint">The robot speaks through its built-in speaker</span>
</div>
<div class="composer">
<textarea
id="textInput"
class="composer__input"
placeholder="Type what you want the robot to say…"
spellcheck="true"
autocomplete="off"
rows="7"
maxlength="1000"></textarea>
<div class="composer__meta">
<span class="kbd-hint"><kbd>Ctrl</kbd><span>+</span><kbd>Enter</kbd> to speak · <kbd>Esc</kbd> to stop</span>
<span class="counter" id="counter"><b id="counterNow">0</b> / <span id="counterMax">1000</span></span>
</div>
</div>
<div class="actions">
<button class="btn btn--speak" id="speakBtn" disabled>
<span class="btn__icon" aria-hidden="true">
<svg viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round">
<path d="M11 5 6 9H2v6h4l5 4V5Z" /><path d="M15.5 8.5a5 5 0 0 1 0 7" /><path d="M18.5 5.5a9 9 0 0 1 0 13" />
</svg>
</span>
<span class="btn__label" id="speakLabel">Speak</span>
<span class="btn__spinner" aria-hidden="true"></span>
</button>
<button class="btn btn--ghost" id="stopBtn" disabled>
<svg viewBox="0 0 24 24" fill="currentColor" aria-hidden="true"><rect x="6" y="6" width="12" height="12" rx="2" /></svg>
Stop
</button>
<button class="btn btn--ghost" id="clearBtn">
<svg viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" aria-hidden="true"><path d="M3 6h18"/><path d="M8 6V4h8v2"/><path d="m19 6-1 14H6L5 6"/></svg>
Clear
</button>
</div>
<!-- pipeline -->
<div class="pipeline" id="pipeline" data-stage="idle">
<div class="pipeline__track" aria-hidden="true"><i id="pipelineFill"></i></div>
<ol class="pipeline__steps">
<li data-step="sending"><span class="dot"></span>Sending</li>
<li data-step="processing"><span class="dot"></span>Processing</li>
<li data-step="speaking"><span class="dot"></span>Playing</li>
<li data-step="completed"><span class="dot"></span>Completed</li>
</ol>
</div>
<!-- live status -->
<div class="live" id="live" data-tone="idle" role="status" aria-live="polite">
<span class="live__wave" id="liveWave" aria-hidden="true"><i></i><i></i><i></i><i></i><i></i></span>
<span class="live__text" id="liveText">Ready</span>
<span class="live__detail" id="liveDetail"></span>
</div>
</section>
<!-- ---------- sidebar ---------- -->
<aside class="sidebar">
<section class="panel panel--status" aria-label="Robot status">
<div class="panel__head"><h2>Robot</h2></div>
<dl class="facts">
<div class="fact"><dt>State</dt><dd id="factState"></dd></div>
<div class="fact"><dt>Mode</dt><dd id="factMode"></dd></div>
<div class="fact"><dt>Link</dt><dd id="factTransport"></dd></div>
<div class="fact"><dt>Address</dt><dd id="factAddress" class="mono"></dd></div>
<div class="fact"><dt>Latency</dt><dd id="factLatency"></dd></div>
<div class="fact"><dt>Uptime</dt><dd id="factUptime"></dd></div>
</dl>
<p class="panel__note" id="statusNote" hidden></p>
</section>
<section class="panel panel--audio" aria-label="Saved audio">
<div class="panel__head">
<h2>Saved audio <span class="count" id="audioCount">0</span></h2>
<button class="link-btn" id="clearAudioBtn">Clear</button>
</div>
<p class="panel__note" id="audioNote">
Every line spoken with the neural voice is saved as a .wav file, so it
replays instantly with no synthesis and no internet.
</p>
<ul class="audio-list" id="audioList"></ul>
</section>
<section class="panel panel--history" aria-label="Speech history">
<div class="panel__head">
<h2>History <span class="count" id="historyCount">0</span></h2>
<button class="link-btn" id="clearHistoryBtn">Clear</button>
</div>
<ul class="history" id="historyList"></ul>
<p class="empty" id="historyEmpty">Nothing spoken yet. Your recent utterances appear here — click one to reuse it.</p>
</section>
</aside>
</main>
<div class="toasts" id="toasts" aria-live="assertive"></div>
<script type="module" src="/js/app.js"></script>
</body>
</html>

69
frontend/js/api.js Normal file
View File

@ -0,0 +1,69 @@
/**
* Thin REST client for the local backend.
* The page never talks to the robot directly - only to this origin.
*/
class ApiError extends Error {
constructor(message, code, status) {
super(message);
this.name = 'ApiError';
this.code = code || 'error';
this.status = status || 0;
}
}
async function request(path, options = {}) {
let response;
try {
response = await fetch(path, {
headers: { 'Content-Type': 'application/json' },
...options,
});
} catch (err) {
// The backend itself is unreachable - a different failure from "robot offline".
throw new ApiError(
'Cannot reach the local server. Is backend/main.py still running?',
'backend_down',
0,
);
}
let body = null;
const text = await response.text();
if (text) {
try { body = JSON.parse(text); } catch { body = { error: text }; }
}
if (!response.ok) {
const message =
(body && (body.error || body.detail)) ||
`Request failed (HTTP ${response.status})`;
throw new ApiError(
typeof message === 'string' ? message : JSON.stringify(message),
(body && body.errorCode) || 'http_error',
response.status,
);
}
return body;
}
export const api = {
status: () => request('/api/robot/status'),
config: () => request('/api/config'),
diagnostics: () => request('/api/robot/diagnostics'),
reloadConfig: () => request('/api/config/reload', { method: 'POST' }),
reconnect: () => request('/api/robot/reconnect', { method: 'POST' }),
speak: (text) => request('/api/robot/speak', {
method: 'POST',
body: JSON.stringify({ text }),
}),
stop: () => request('/api/robot/stop', { method: 'POST' }),
history: () => request('/api/speech/history'),
clearHistory: () => request('/api/speech/history', { method: 'DELETE' }),
audio: () => request('/api/audio'),
audioUrl: (id) => `/api/audio/${id}/file`,
deleteAudio: (id) => request(`/api/audio/${id}`, { method: 'DELETE' }),
clearAudio: () => request('/api/audio', { method: 'DELETE' }),
};
export { ApiError };

373
frontend/js/app.js Normal file
View File

@ -0,0 +1,373 @@
/**
* Application wiring.
*
* Flow: user text -> POST /api/robot/speak -> backend -> robot
* robot lifecycle -> backend event bus -> WebSocket -> this file -> UI
*
* The POST returns as soon as the robot acknowledges the utterance; everything
* after that ("speaking", "completed", failures) arrives over the socket.
*/
import { api, ApiError } from './api.js';
import { RobotSocket } from './socket.js';
import * as ui from './ui.js';
const DRAFT_KEY = 'agibot-a3:draft';
const RESET_DELAY = 2200;
const state = {
config: { maxLength: 1000, mode: 'mock' },
status: { state: 'connecting', connected: false },
history: [],
audio: { items: [], stats: {} },
auditioning: null,
activeRequestId: null,
stage: 'idle',
resetTimer: null,
};
const socket = new RobotSocket();
// =========================================================================
// derived UI state
// =========================================================================
function syncControls() {
const length = ui.el.textInput.value.trim().length;
const busy = Boolean(state.activeRequestId);
const connected = Boolean(state.status.connected);
ui.setControls({
canSpeak: connected && length > 0 && length <= state.config.maxLength,
busy,
canStop: busy,
});
ui.renderCounter(ui.el.textInput.value.length, state.config.maxLength);
}
function applyStatus(status) {
state.status = { ...state.status, ...status };
ui.renderConnection(state.status);
ui.renderFacts(state.status);
// The robot vanished mid-utterance: stop pretending it is still speaking.
if (!state.status.connected && state.activeRequestId) {
finishUtterance('error', 'Robot disconnected during speech.');
}
// Page opened (or reloaded) while the robot is mid-sentence - adopt it rather
// than showing "Ready" over a talking robot.
if (state.status.busy && state.status.activeRequestId && !state.activeRequestId) {
state.activeRequestId = state.status.activeRequestId;
state.stage = 'speaking';
clearTimeout(state.resetTimer);
ui.renderPipeline('speaking');
ui.renderLive('speaking', 'Speaking…');
}
syncControls();
}
// =========================================================================
// speech lifecycle
// =========================================================================
const STAGE_VIEW = {
queued: { tone: 'busy', text: 'Queued…' },
sending: { tone: 'busy', text: 'Sending to robot…' },
processing: { tone: 'busy', text: 'Robot is preparing speech…' },
speaking: { tone: 'speaking', text: 'Speaking…' },
completed: { tone: 'ok', text: 'Completed' },
cancelled: { tone: 'error', text: 'Stopped' },
failed: { tone: 'error', text: 'Speech failed' },
};
function onProgress(progress) {
if (state.activeRequestId && progress.requestId !== state.activeRequestId) return;
state.activeRequestId = progress.requestId;
const stage = progress.stage;
state.stage = stage;
ui.renderPipeline(stage);
const view = STAGE_VIEW[stage] || { tone: 'busy', text: stage };
const detail = progress.detail
|| (typeof progress.elapsedMs === 'number' ? `${progress.elapsedMs} ms` : '');
ui.renderLive(view.tone, progress.message || view.text, detail);
if (['completed', 'failed', 'cancelled'].includes(stage)) {
if (stage === 'completed') refreshAudio();
if (stage === 'failed') ui.toast(progress.message || 'Speech request failed.', 'error');
finishUtterance(view.tone, progress.message || view.text, detail);
}
syncControls();
}
function finishUtterance(tone, message, detail = '') {
state.activeRequestId = null;
ui.renderLive(tone, message, detail);
syncControls();
clearTimeout(state.resetTimer);
state.resetTimer = setTimeout(() => {
if (state.activeRequestId) return; // a new utterance already started
state.stage = 'idle';
ui.renderPipeline('idle');
ui.renderLive('idle', state.status.connected ? 'Ready' : 'Robot offline');
}, RESET_DELAY);
}
// =========================================================================
// history
// =========================================================================
function applyHistoryEvent(payload) {
if (payload.action === 'clear') {
state.history = [];
} else if (payload.entry) {
const index = state.history.findIndex((item) => item.id === payload.entry.id);
if (index === -1) state.history.unshift(payload.entry);
else state.history[index] = payload.entry;
const limit = state.config.historyLimit || 100;
if (state.history.length > limit) state.history.length = limit;
}
ui.renderHistory(state.history, useHistoryText, replayLine);
}
/** Speak a history line again. Its audio is saved, so this is instant. */
async function replayLine(item) {
ui.el.textInput.value = item.text;
saveDraft();
await speak();
}
function useHistoryText(text) {
ui.el.textInput.value = text;
ui.el.textInput.focus();
ui.el.textInput.setSelectionRange(text.length, text.length);
saveDraft();
syncControls();
}
// =========================================================================
// actions
// =========================================================================
async function speak() {
const text = ui.el.textInput.value.trim();
if (!text) {
ui.toast('Type something for the robot to say first.', 'error');
ui.el.textInput.focus();
return;
}
if (!state.status.connected) {
ui.toast(
state.status.error || 'Robot is offline. Check the robot IP and network connection.',
'error',
);
return;
}
clearTimeout(state.resetTimer);
ui.renderPipeline('sending');
ui.renderLive('busy', 'Sending to robot…');
ui.setControls({ canSpeak: false, busy: true, canStop: true });
try {
const result = await api.speak(text);
state.activeRequestId = result.requestId;
syncControls();
} catch (err) {
state.activeRequestId = null;
const message = err instanceof ApiError ? err.message : 'Speech request failed.';
ui.toast(message, 'error');
ui.renderPipeline('failed');
finishUtterance('error', message);
}
}
async function stop() {
try {
await api.stop();
ui.renderLive('error', 'Stopping…');
} catch (err) {
ui.toast(err.message || 'Could not stop the robot.', 'error');
}
}
function clearText() {
ui.el.textInput.value = '';
saveDraft();
ui.el.textInput.focus();
syncControls();
}
// =========================================================================
// saved audio
// =========================================================================
async function refreshAudio() {
try {
const data = await api.audio();
state.audio = { items: data.items || [], stats: data.stats || {} };
ui.renderAudio(state.audio.items, state.audio.stats, {
fileUrl: api.audioUrl,
onPlayState: (audio) => {
// Only one clip auditions at a time.
if (state.auditioning && state.auditioning !== audio) {
state.auditioning.pause();
state.auditioning.currentTime = 0;
}
state.auditioning = audio;
},
});
} catch {
/* the panel is a convenience; never let it break the page */
}
}
async function clearAudio() {
try {
const result = await api.clearAudio();
ui.toast(`Deleted ${result.removed} saved clip(s).`, 'ok', 3000);
await refreshAudio();
// Rows can no longer promise instant replay.
state.history = state.history.map((item) => ({ ...item, audioSaved: false }));
ui.renderHistory(state.history, useHistoryText, replayLine);
} catch (err) {
ui.toast(err.message || 'Could not clear saved audio.', 'error');
}
}
async function clearHistory() {
try {
await api.clearHistory();
state.history = [];
ui.renderHistory(state.history, useHistoryText, replayLine);
} catch (err) {
ui.toast(err.message || 'Could not clear history.', 'error');
}
}
async function reconnect() {
ui.el.reconnectBtn.dataset.spin = 'true';
try {
await api.reconnect();
ui.toast('Reconnecting to the robot…', 'info', 2500);
} catch (err) {
ui.toast(err.message || 'Reconnect failed.', 'error');
} finally {
setTimeout(() => { ui.el.reconnectBtn.dataset.spin = 'false'; }, 900);
}
}
// =========================================================================
// draft persistence (per-browser convenience only)
// =========================================================================
function saveDraft() {
try { localStorage.setItem(DRAFT_KEY, ui.el.textInput.value); } catch { /* private mode */ }
}
function loadDraft() {
try {
const draft = localStorage.getItem(DRAFT_KEY);
if (draft) ui.el.textInput.value = draft;
} catch { /* ignore */ }
}
// =========================================================================
// events
// =========================================================================
function bindDom() {
ui.el.speakBtn.addEventListener('click', speak);
ui.el.stopBtn.addEventListener('click', stop);
ui.el.clearBtn.addEventListener('click', clearText);
ui.el.clearHistoryBtn.addEventListener('click', clearHistory);
ui.el.clearAudioBtn.addEventListener('click', clearAudio);
ui.el.reconnectBtn.addEventListener('click', reconnect);
ui.el.textInput.addEventListener('input', () => { saveDraft(); syncControls(); });
// Ctrl/Cmd+Enter speaks. A bare Enter inserts a newline, on purpose - nobody
// wants a half-typed sentence going out of the robot's speaker mid-demo.
ui.el.textInput.addEventListener('keydown', (event) => {
if ((event.ctrlKey || event.metaKey) && event.key === 'Enter') {
event.preventDefault();
speak();
}
});
document.addEventListener('keydown', (event) => {
if (event.key === 'Escape' && state.activeRequestId) {
event.preventDefault();
stop();
}
});
}
function bindSocket() {
socket.addEventListener('hello', (event) => {
const { config, status, history } = event.detail;
state.config = { ...state.config, ...config };
ui.renderConfig(state.config);
applyStatus(status);
state.history = history || [];
ui.renderHistory(state.history, useHistoryText, replayLine);
refreshAudio();
document.documentElement.removeAttribute('data-loading');
});
socket.addEventListener('robot.status', (event) => applyStatus(event.detail));
socket.addEventListener('speech.progress', (event) => onProgress(event.detail));
socket.addEventListener('history.updated', (event) => applyHistoryEvent(event.detail));
socket.addEventListener('config.updated', (event) => {
state.config = { ...state.config, ...event.detail };
ui.renderConfig(state.config);
});
socket.addEventListener('rtt', (event) => {
const rtt = event.detail.rtt;
if (typeof rtt === 'number') {
ui.el.latencyChip.title = `Robot round-trip. Browser to backend: ${rtt} ms`;
}
});
socket.addEventListener('link', (event) => {
if (event.detail.up) return;
// The backend went away - say so rather than leaving a stale green light.
applyStatus({
state: 'disconnected',
connected: false,
error: 'Lost connection to the local server.',
});
});
}
// =========================================================================
// boot
// =========================================================================
async function boot() {
bindDom();
bindSocket();
loadDraft();
syncControls();
socket.connect();
// Fallback for the rare case the socket cannot open at all.
setTimeout(async () => {
if (socket.isOpen) return;
try {
const [config, status, history] = await Promise.all([
api.config(), api.status(), api.history(),
]);
state.config = { ...state.config, ...config };
ui.renderConfig(state.config);
applyStatus(status);
state.history = history.items || [];
ui.renderHistory(state.history, useHistoryText, replayLine);
refreshAudio();
document.documentElement.removeAttribute('data-loading');
} catch {
ui.toast('Cannot reach the local server. Is backend/main.py running?', 'error', 12000);
}
}, 1200);
ui.el.textInput.focus();
}
boot();

111
frontend/js/socket.js Normal file
View File

@ -0,0 +1,111 @@
/**
* WebSocket client for live robot state.
*
* Replaces polling entirely: the backend pushes connection state, speech
* lifecycle and history updates. Reconnects on its own with capped backoff, so
* restarting the server does not require reloading the page.
*/
const PING_INTERVAL = 5000;
const MIN_BACKOFF = 500;
const MAX_BACKOFF = 8000;
export class RobotSocket extends EventTarget {
constructor() {
super();
this.ws = null;
this.backoff = MIN_BACKOFF;
this.pingTimer = null;
this.rtt = null;
this.closedByUs = false;
}
get url() {
const scheme = location.protocol === 'https:' ? 'wss' : 'ws';
return `${scheme}://${location.host}/ws`;
}
get isOpen() {
return this.ws && this.ws.readyState === WebSocket.OPEN;
}
connect() {
this.closedByUs = false;
if (this.ws && (this.ws.readyState === WebSocket.OPEN || this.ws.readyState === WebSocket.CONNECTING)) {
return;
}
let ws;
try {
ws = new WebSocket(this.url);
} catch {
this.#scheduleReconnect();
return;
}
this.ws = ws;
ws.onopen = () => {
this.backoff = MIN_BACKOFF;
this.#emit('link', { up: true });
this.#startPing();
};
ws.onmessage = (event) => {
let message;
try { message = JSON.parse(event.data); } catch { return; }
if (message.type === 'pong') {
const sent = message.data && message.data.t;
if (typeof sent === 'number') this.rtt = Math.max(0, Math.round(performance.now() - sent));
this.#emit('rtt', { rtt: this.rtt });
return;
}
this.#emit(message.type, message.data || {});
};
ws.onclose = () => {
this.#stopPing();
this.#emit('link', { up: false });
if (!this.closedByUs) this.#scheduleReconnect();
};
ws.onerror = () => { /* onclose always follows; handled there */ };
}
close() {
this.closedByUs = true;
this.#stopPing();
if (this.ws) this.ws.close();
}
send(type, data = {}) {
if (!this.isOpen) return false;
this.ws.send(JSON.stringify({ type, ...data }));
return true;
}
requestStatus() { this.send('status'); }
requestReconnect() { this.send('reconnect'); }
// -- internals ------------------------------------------------------------
#emit(type, detail) {
this.dispatchEvent(new CustomEvent(type, { detail }));
}
#startPing() {
this.#stopPing();
const ping = () => this.send('ping', { t: performance.now() });
ping();
this.pingTimer = setInterval(ping, PING_INTERVAL);
}
#stopPing() {
if (this.pingTimer) { clearInterval(this.pingTimer); this.pingTimer = null; }
}
#scheduleReconnect() {
const delay = this.backoff;
this.backoff = Math.min(this.backoff * 2, MAX_BACKOFF);
setTimeout(() => this.connect(), delay);
}
}

367
frontend/js/ui.js Normal file
View File

@ -0,0 +1,367 @@
/** DOM rendering. All element lookups and mutations live here. */
const $ = (id) => document.getElementById(id);
export const el = {
modeBadge: $('modeBadge'),
connPill: $('connPill'),
connLabel: $('connLabel'),
latencyChip: $('latencyChip'),
latencyValue: $('latencyValue'),
reconnectBtn: $('reconnectBtn'),
banner: $('configBanner'),
bannerTitle: $('bannerTitle'),
bannerText: $('bannerText'),
textInput: $('textInput'),
counter: $('counter'),
counterNow: $('counterNow'),
counterMax: $('counterMax'),
speakBtn: $('speakBtn'),
speakLabel: $('speakLabel'),
stopBtn: $('stopBtn'),
clearBtn: $('clearBtn'),
pipeline: $('pipeline'),
pipelineFill: $('pipelineFill'),
live: $('live'),
liveText: $('liveText'),
liveDetail: $('liveDetail'),
factState: $('factState'),
factMode: $('factMode'),
factTransport: $('factTransport'),
factAddress: $('factAddress'),
factLatency: $('factLatency'),
factUptime: $('factUptime'),
statusNote: $('statusNote'),
audioList: $('audioList'),
audioCount: $('audioCount'),
audioNote: $('audioNote'),
clearAudioBtn: $('clearAudioBtn'),
historyList: $('historyList'),
historyEmpty: $('historyEmpty'),
historyCount: $('historyCount'),
clearHistoryBtn: $('clearHistoryBtn'),
toasts: $('toasts'),
};
const CONNECTION_LABEL = {
connected: 'Robot Connected',
connecting: 'Connecting…',
disconnected: 'Robot Disconnected',
error: 'Connection Error',
};
const PIPELINE_PROGRESS = {
idle: 0, queued: 8, sending: 26, processing: 52, speaking: 80,
completed: 100, failed: 100, cancelled: 100,
};
const STEP_ORDER = ['sending', 'processing', 'speaking', 'completed'];
// -------------------------------------------------------------------------
// header / connection
// -------------------------------------------------------------------------
export function renderConnection(status) {
const state = status.state || 'disconnected';
el.connPill.dataset.state = state;
el.connLabel.textContent = CONNECTION_LABEL[state] || state;
const latency = status.latencyMs;
if (typeof latency === 'number' && state === 'connected') {
el.latencyChip.hidden = false;
el.latencyValue.textContent = `${latency < 10 ? latency.toFixed(1) : Math.round(latency)} ms`;
} else {
el.latencyChip.hidden = true;
}
}
export function renderConfig(config) {
if (!config) return;
const mode = (config.mode || 'mock').toLowerCase();
el.modeBadge.textContent = mode === 'real' ? 'Live Robot' : 'Mock Mode';
el.modeBadge.dataset.mode = mode;
const max = config.maxLength || 1000;
el.counterMax.textContent = String(max);
el.textInput.maxLength = max;
const issues = config.issues || [];
if (issues.length) {
const worst = issues.find((i) => i.level === 'error') || issues[0];
el.banner.hidden = false;
el.banner.dataset.level = worst.level;
el.bannerTitle.textContent =
worst.level === 'error' ? 'Configuration error' : 'Configuration warning';
el.bannerText.textContent = `${worst.message} (${worst.key} in ${config.envFile || '.env'})`;
} else {
el.banner.hidden = true;
}
}
export function renderFacts(status) {
const robot = status.robot || {};
const state = status.state || 'disconnected';
el.factState.textContent = CONNECTION_LABEL[state] || state;
el.factState.dataset.tone =
state === 'connected' ? 'ok' : state === 'connecting' ? 'warn' : 'error';
el.factMode.textContent = robot.mode === 'real' ? 'Live robot' : 'Simulation';
el.factTransport.textContent = (robot.transport || '').toUpperCase();
el.factAddress.textContent = robot.address || '';
el.factAddress.title = robot.address || '';
el.factLatency.textContent =
typeof status.latencyMs === 'number' ? `${Math.round(status.latencyMs)} ms` : '';
el.factUptime.textContent =
typeof status.uptimeSeconds === 'number' ? formatDuration(status.uptimeSeconds) : '';
if (status.error) {
el.statusNote.hidden = false;
el.statusNote.textContent = status.error;
} else {
el.statusNote.hidden = true;
}
}
// -------------------------------------------------------------------------
// composer
// -------------------------------------------------------------------------
export function renderCounter(length, max) {
el.counterNow.textContent = String(length);
const ratio = max ? length / max : 0;
el.counter.dataset.level = ratio >= 1 ? 'over' : ratio > 0.85 ? 'warn' : 'ok';
}
export function setControls({ canSpeak, busy, canStop }) {
el.speakBtn.disabled = !canSpeak;
el.speakBtn.dataset.busy = busy ? 'true' : 'false';
el.speakLabel.textContent = busy ? 'Speaking…' : 'Speak';
el.stopBtn.disabled = !canStop;
}
// -------------------------------------------------------------------------
// pipeline + live status
// -------------------------------------------------------------------------
export function renderPipeline(stage) {
const key = PIPELINE_PROGRESS[stage] === undefined ? 'idle' : stage;
el.pipeline.dataset.stage = key;
el.pipelineFill.style.inset = `0 ${100 - PIPELINE_PROGRESS[key]}% 0 0`;
const activeIndex = STEP_ORDER.indexOf(
key === 'queued' ? 'sending' : key === 'failed' || key === 'cancelled' ? 'speaking' : key,
);
el.pipeline.querySelectorAll('li').forEach((li, index) => {
if (key === 'idle') { li.removeAttribute('data-on'); return; }
if (index < activeIndex) li.dataset.on = 'done';
else if (index === activeIndex) li.dataset.on = 'active';
else li.removeAttribute('data-on');
});
}
export function renderLive(tone, text, detail = '') {
el.live.dataset.tone = tone;
el.liveText.textContent = text;
el.liveDetail.textContent = detail || '';
}
// -------------------------------------------------------------------------
// history
// -------------------------------------------------------------------------
export function renderHistory(items, onPick, onReplay = () => {}) {
el.historyCount.textContent = String(items.length);
el.historyEmpty.hidden = items.length > 0;
el.historyList.innerHTML = '';
for (const item of items) {
const li = document.createElement('li');
li.className = 'hitem';
li.dataset.ok = item.success === null || item.success === undefined
? (isTerminal(item.stage) ? 'false' : 'pending')
: String(Boolean(item.success));
li.title = 'Click to put this text back in the box';
li.tabIndex = 0;
const bar = document.createElement('span');
bar.className = 'hitem__bar';
const body = document.createElement('div');
body.className = 'hitem__body';
const text = document.createElement('div');
text.className = 'hitem__text';
text.textContent = item.text;
const meta = document.createElement('div');
meta.className = 'hitem__meta';
meta.append(spanOf(formatTime(item.at)));
if (typeof item.ackLatencyMs === 'number') meta.append(spanOf(`${item.ackLatencyMs} ms`));
if (item.error) {
const err = spanOf(item.error);
err.className = 'err';
meta.append(err);
} else if (!isTerminal(item.stage)) {
meta.append(spanOf(item.stage));
}
body.append(text, meta);
li.append(bar, body);
if (item.audioSaved) {
const replay = document.createElement('button');
replay.className = 'hitem__replay';
replay.title = 'Play the saved audio instantly';
replay.setAttribute('aria-label', 'Replay this line');
replay.innerHTML = '<svg viewBox="0 0 24 24" fill="currentColor"><path d="M8 5v14l11-7z"/></svg>';
replay.addEventListener('click', (event) => {
event.stopPropagation(); // do not also load it into the box
onReplay(item);
});
li.append(replay);
}
const pick = () => onPick(item.text);
li.addEventListener('click', pick);
li.addEventListener('keydown', (event) => {
if (event.key === 'Enter' || event.key === ' ') { event.preventDefault(); pick(); }
});
el.historyList.append(li);
}
}
function isTerminal(stage) {
return ['completed', 'failed', 'cancelled'].includes(stage);
}
function spanOf(text) {
const span = document.createElement('span');
span.textContent = text;
return span;
}
// -------------------------------------------------------------------------
// toasts
// -------------------------------------------------------------------------
export function toast(message, tone = 'info', ttl = 5000) {
const node = document.createElement('div');
node.className = 'toast';
node.dataset.tone = tone;
const dot = document.createElement('span');
dot.className = 'toast__dot';
const text = document.createElement('div');
text.textContent = message;
node.append(dot, text);
el.toasts.append(node);
const remove = () => {
node.classList.add('is-out');
setTimeout(() => node.remove(), 220);
};
const timer = setTimeout(remove, ttl);
node.addEventListener('click', () => { clearTimeout(timer); remove(); });
return node;
}
// -------------------------------------------------------------------------
// helpers
// -------------------------------------------------------------------------
export function formatTime(epochSeconds) {
if (!epochSeconds) return '';
return new Date(epochSeconds * 1000).toLocaleTimeString([], {
hour: '2-digit', minute: '2-digit', second: '2-digit',
});
}
export function formatDuration(seconds) {
const total = Math.max(0, Math.round(seconds));
if (total < 60) return `${total}s`;
const minutes = Math.floor(total / 60);
if (minutes < 60) return `${minutes}m ${total % 60}s`;
return `${Math.floor(minutes / 60)}h ${minutes % 60}m`;
}
// -------------------------------------------------------------------------
// saved audio
// -------------------------------------------------------------------------
const PLAY_ICON = '<svg viewBox="0 0 24 24" fill="currentColor"><path d="M8 5v14l11-7z"/></svg>';
const STOP_ICON = '<svg viewBox="0 0 24 24" fill="currentColor"><rect x="6" y="6" width="12" height="12" rx="2"/></svg>';
const DOWNLOAD_ICON =
'<svg viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" ' +
'stroke-linecap="round" stroke-linejoin="round"><path d="M12 3v12"/>' +
'<path d="m7 12 5 5 5-5"/><path d="M5 21h14"/></svg>';
export function renderAudio(items, stats, { fileUrl, onPlayState }) {
el.audioCount.textContent = String(items.length);
el.audioList.innerHTML = '';
if (!items.length) {
el.audioNote.textContent =
'Nothing saved yet. Speak a line with the neural voice and its audio is ' +
'kept here as a .wav, so it replays instantly with no synthesis.';
return;
}
// Keep the note to one short line; the full path would overflow the panel,
// so it goes in the tooltip instead.
el.audioNote.textContent =
`${items.length} clip(s) · ${(stats.seconds || 0).toFixed(0)}s · ` +
`${((stats.bytes || 0) / 1048576).toFixed(1)} MB · audio_library/`;
el.audioNote.title = stats.dir || '';
for (const item of items) {
const li = document.createElement('li');
li.className = 'aitem';
const button = document.createElement('button');
button.className = 'aitem__play';
button.innerHTML = PLAY_ICON;
button.title = 'Play in this browser';
// Played in the page, not through the backend: this is you auditioning a
// saved clip, not the robot speaking. Keeping them separate means the
// dashboard's Speaking state never lies.
const audio = new Audio(fileUrl(item.id));
button.addEventListener('click', () => {
if (!audio.paused) { audio.pause(); audio.currentTime = 0; return; }
onPlayState(audio);
audio.play().catch(() => {});
});
audio.addEventListener('play', () => { button.dataset.playing = 'true'; button.innerHTML = STOP_ICON; });
const reset = () => { button.dataset.playing = 'false'; button.innerHTML = PLAY_ICON; };
audio.addEventListener('pause', reset);
audio.addEventListener('ended', reset);
const body = document.createElement('div');
body.className = 'aitem__body';
const text = document.createElement('div');
text.className = 'aitem__text';
text.textContent = item.text;
text.title = item.text;
const meta = document.createElement('div');
meta.className = 'aitem__meta';
meta.textContent = `${(item.durationSeconds || 0).toFixed(1)}s · ${item.voice} · ${item.file}`;
meta.title = item.file;
body.append(text, meta);
const download = document.createElement('a');
download.className = 'aitem__dl';
download.href = fileUrl(item.id);
download.download = item.file;
download.title = 'Download this .wav';
download.innerHTML = DOWNLOAD_ICON;
li.append(button, body, download);
el.audioList.append(li);
}
}

565
frontend/styles/main.css Normal file
View File

@ -0,0 +1,565 @@
/* =========================================================================
AGIBOT A3 · Voice Control
A dark, single-purpose control surface. Committed to one theme on purpose:
this is meant to be projected or shown on a stand next to the robot.
========================================================================= */
:root {
/* surfaces */
--bg: #080a0f;
--bg-glow-a: rgba(34, 211, 238, 0.10);
--bg-glow-b: rgba(59, 130, 246, 0.08);
--surface: #11151f;
--surface-2: #161b28;
--surface-3: #1c2230;
--border: rgba(255, 255, 255, 0.075);
--border-strong: rgba(255, 255, 255, 0.14);
/* text */
--text: #e8ecf4;
--text-dim: #9aa4bb;
--text-faint: #626c85;
/* brand + state */
--accent: #22d3ee;
--accent-deep: #0891b2;
--accent-soft: rgba(34, 211, 238, 0.14);
--ok: #34d399;
--warn: #fbbf24;
--danger: #f87171;
--busy: #818cf8;
--radius: 14px;
--radius-sm: 10px;
--shadow: 0 1px 2px rgba(0,0,0,.4), 0 12px 32px rgba(0,0,0,.32);
--font: "Inter", "Segoe UI Variable Text", "Segoe UI", system-ui, -apple-system,
"Helvetica Neue", Arial, sans-serif;
--mono: "JetBrains Mono", "Cascadia Mono", "SF Mono", Consolas, monospace;
}
* { box-sizing: border-box; }
/* Author `display` rules below outrank the UA stylesheet's [hidden], so restate
it with priority - otherwise .banner / .chip stay visible when hidden=true. */
[hidden] { display: none !important; }
html, body { height: 100%; }
body {
margin: 0;
font-family: var(--font);
font-size: 15px;
line-height: 1.5;
color: var(--text);
background:
radial-gradient(1100px 620px at 12% -10%, var(--bg-glow-a), transparent 60%),
radial-gradient(900px 520px at 92% 0%, var(--bg-glow-b), transparent 55%),
var(--bg);
background-attachment: fixed;
-webkit-font-smoothing: antialiased;
text-rendering: optimizeLegibility;
}
h1, h2 { margin: 0; font-weight: 600; letter-spacing: -0.015em; }
.mono { font-family: var(--mono); font-size: 13px; }
/* =========================================================================
HEADER
========================================================================= */
.topbar {
position: sticky; top: 0; z-index: 40;
display: flex; align-items: center; justify-content: space-between; gap: 16px;
padding: 14px clamp(16px, 4vw, 40px);
background: rgba(8, 10, 15, 0.72);
backdrop-filter: blur(14px);
-webkit-backdrop-filter: blur(14px);
border-bottom: 1px solid var(--border);
}
.topbar__brand { display: flex; align-items: center; gap: 13px; min-width: 0; }
.brand__mark {
display: grid; place-items: center;
width: 40px; height: 40px; flex: none;
border-radius: 11px;
color: var(--accent);
background: linear-gradient(160deg, var(--surface-3), var(--surface));
border: 1px solid var(--border-strong);
box-shadow: inset 0 1px 0 rgba(255,255,255,.05);
}
.brand__mark svg { width: 23px; height: 23px; }
.brand__text h1 { font-size: 16px; letter-spacing: 0.02em; }
.brand__text p { margin: 0; font-size: 12px; color: var(--text-faint); letter-spacing: .06em; text-transform: uppercase; }
.mode-badge {
margin-left: 6px; padding: 3px 9px;
font-size: 10.5px; font-weight: 700; letter-spacing: .1em; text-transform: uppercase;
border-radius: 6px;
color: var(--warn);
background: rgba(251, 191, 36, .12);
border: 1px solid rgba(251, 191, 36, .3);
white-space: nowrap;
}
.mode-badge[data-mode="real"] { color: var(--ok); background: rgba(52,211,153,.12); border-color: rgba(52,211,153,.3); }
.topbar__status { display: flex; align-items: center; gap: 10px; }
/* connection pill */
.conn {
display: inline-flex; align-items: center; gap: 9px;
padding: 7px 14px 7px 12px;
border-radius: 999px;
font-size: 13px; font-weight: 500; white-space: nowrap;
background: var(--surface-2);
border: 1px solid var(--border);
transition: border-color .25s, background .25s, color .25s;
}
.conn__dot {
width: 8px; height: 8px; border-radius: 50%;
background: var(--text-faint);
box-shadow: 0 0 0 0 transparent;
transition: background .25s;
}
.conn[data-state="connected"] { color: var(--ok); border-color: rgba(52,211,153,.32); background: rgba(52,211,153,.09); }
.conn[data-state="connected"] .conn__dot { background: var(--ok); box-shadow: 0 0 10px rgba(52,211,153,.75); }
.conn[data-state="connecting"] { color: var(--warn); border-color: rgba(251,191,36,.32); background: rgba(251,191,36,.09); }
.conn[data-state="connecting"] .conn__dot { background: var(--warn); animation: pulse 1.1s ease-in-out infinite; }
.conn[data-state="disconnected"] { color: var(--danger); border-color: rgba(248,113,113,.3); background: rgba(248,113,113,.08); }
.conn[data-state="disconnected"] .conn__dot { background: var(--danger); }
.conn[data-state="error"] { color: var(--danger); border-color: rgba(248,113,113,.45); background: rgba(248,113,113,.12); }
.conn[data-state="error"] .conn__dot { background: var(--danger); animation: pulse 1.4s ease-in-out infinite; }
@keyframes pulse {
0%, 100% { opacity: 1; transform: scale(1); }
50% { opacity: .35; transform: scale(.82); }
}
.chip {
display: inline-flex; align-items: center; gap: 6px;
padding: 6px 11px; border-radius: 999px;
font-family: var(--mono); font-size: 12px; color: var(--text-dim);
background: var(--surface-2); border: 1px solid var(--border);
}
.chip svg { width: 12px; height: 12px; color: var(--accent); }
.icon-btn {
display: grid; place-items: center;
width: 36px; height: 36px; flex: none;
border-radius: 10px; cursor: pointer;
color: var(--text-dim);
background: var(--surface-2);
border: 1px solid var(--border);
transition: color .18s, border-color .18s, background .18s, transform .18s;
}
.icon-btn svg { width: 17px; height: 17px; }
.icon-btn:hover { color: var(--accent); border-color: var(--border-strong); background: var(--surface-3); }
.icon-btn:active { transform: scale(.94); }
.icon-btn[data-spin="true"] svg { animation: spin .9s linear infinite; }
@keyframes spin { to { transform: rotate(360deg); } }
/* =========================================================================
CONFIG BANNER
========================================================================= */
.banner {
display: flex; align-items: flex-start; gap: 11px;
margin: 16px clamp(16px, 4vw, 40px) 0;
padding: 13px 16px;
border-radius: var(--radius-sm);
font-size: 13.5px;
color: #fde68a;
background: rgba(251, 191, 36, .07);
border: 1px solid rgba(251, 191, 36, .28);
}
.banner[data-level="error"] { color: #fecaca; background: rgba(248,113,113,.08); border-color: rgba(248,113,113,.3); }
.banner svg { width: 17px; height: 17px; flex: none; margin-top: 1px; }
.banner__body { display: flex; flex-direction: column; gap: 2px; min-width: 0; }
.banner__body strong { font-weight: 650; }
.banner__body span { color: inherit; opacity: .88; }
/* =========================================================================
LAYOUT
========================================================================= */
.layout {
display: grid;
grid-template-columns: minmax(0, 1.65fr) minmax(300px, 0.85fr);
gap: 20px;
max-width: 1240px;
margin: 0 auto;
padding: 22px clamp(16px, 4vw, 40px) 48px;
align-items: start;
}
.sidebar { display: flex; flex-direction: column; gap: 20px; }
.panel {
background: linear-gradient(180deg, var(--surface-2), var(--surface));
border: 1px solid var(--border);
border-radius: var(--radius);
box-shadow: var(--shadow);
padding: 20px;
}
.panel--speech { padding: 22px; }
.panel__head {
display: flex; align-items: baseline; justify-content: space-between; gap: 12px;
margin-bottom: 14px;
}
.panel__head h2 { font-size: 15px; letter-spacing: -0.01em; }
.panel__hint { font-size: 12.5px; color: var(--text-faint); }
.panel__note {
margin: 12px 0 0; font-size: 12.5px; color: var(--text-faint); line-height: 1.5;
overflow-wrap: anywhere; /* long file paths must not push the panel wide */
}
.count {
display: inline-block; margin-left: 6px; padding: 1px 7px;
font-size: 11px; font-weight: 600; color: var(--text-dim);
background: var(--surface-3); border-radius: 999px; border: 1px solid var(--border);
}
/* =========================================================================
COMPOSER
========================================================================= */
.composer {
border: 1px solid var(--border);
border-radius: var(--radius-sm);
background: #0d111a;
transition: border-color .2s, box-shadow .2s;
}
.composer:focus-within {
border-color: rgba(34, 211, 238, .45);
box-shadow: 0 0 0 3px var(--accent-soft);
}
.composer__input {
display: block; width: 100%;
padding: 16px 17px 10px;
border: 0; outline: none; resize: vertical;
min-height: 150px;
font-family: inherit; font-size: 16.5px; line-height: 1.6;
color: var(--text); background: transparent;
}
.composer__input::placeholder { color: var(--text-faint); }
.composer__meta {
display: flex; align-items: center; justify-content: space-between; gap: 12px;
padding: 9px 15px 11px;
font-size: 12px; color: var(--text-faint);
border-top: 1px solid var(--border);
}
.kbd-hint { display: inline-flex; align-items: center; gap: 4px; flex-wrap: wrap; }
kbd {
padding: 2px 6px; border-radius: 5px;
font-family: var(--mono); font-size: 10.5px; color: var(--text-dim);
background: var(--surface-3); border: 1px solid var(--border-strong);
box-shadow: 0 1px 0 rgba(0,0,0,.4);
}
.counter { font-family: var(--mono); font-size: 12px; white-space: nowrap; }
.counter b { color: var(--text-dim); font-weight: 500; }
.counter[data-level="warn"] b { color: var(--warn); }
.counter[data-level="over"] b { color: var(--danger); }
/* =========================================================================
ACTIONS
========================================================================= */
.actions { display: flex; gap: 10px; margin-top: 16px; flex-wrap: wrap; }
.btn {
display: inline-flex; align-items: center; justify-content: center; gap: 9px;
padding: 0 20px; height: 46px;
font-family: inherit; font-size: 14.5px; font-weight: 550;
color: var(--text);
border-radius: 11px; border: 1px solid var(--border-strong);
background: var(--surface-3);
cursor: pointer;
transition: background .18s, border-color .18s, color .18s, transform .12s, opacity .18s;
}
.btn svg { width: 17px; height: 17px; }
.btn:hover:not(:disabled) { background: #232a3a; border-color: rgba(255,255,255,.2); }
.btn:active:not(:disabled) { transform: translateY(1px); }
.btn:disabled { opacity: .42; cursor: not-allowed; }
.btn--ghost { color: var(--text-dim); }
.btn--ghost:hover:not(:disabled) { color: var(--text); }
/* the one button that matters */
.btn--speak {
flex: 1 1 260px;
height: 54px;
padding: 0 30px;
font-size: 16px; font-weight: 650; letter-spacing: .01em;
color: #04222a;
border: 0;
background: linear-gradient(135deg, #67e8f9 0%, var(--accent) 45%, var(--accent-deep) 100%);
box-shadow: 0 6px 22px rgba(34, 211, 238, .26), inset 0 1px 0 rgba(255,255,255,.35);
}
.btn--speak:hover:not(:disabled) {
filter: brightness(1.07);
box-shadow: 0 8px 30px rgba(34, 211, 238, .36), inset 0 1px 0 rgba(255,255,255,.4);
}
.btn--speak:disabled { background: var(--surface-3); color: var(--text-faint); box-shadow: none; }
.btn__icon { display: inline-grid; place-items: center; }
.btn__spinner { display: none; width: 16px; height: 16px; border-radius: 50%;
border: 2px solid rgba(4,34,42,.28); border-top-color: #04222a; animation: spin .7s linear infinite; }
.btn--speak[data-busy="true"] .btn__icon { display: none; }
.btn--speak[data-busy="true"] .btn__spinner { display: block; }
/* =========================================================================
PIPELINE
========================================================================= */
.pipeline { margin-top: 22px; }
.pipeline__track {
position: relative; height: 3px; border-radius: 3px;
background: var(--surface-3); overflow: hidden;
}
.pipeline__track i {
position: absolute; inset: 0 100% 0 0;
background: linear-gradient(90deg, var(--accent-deep), var(--accent));
transition: inset .45s cubic-bezier(.4, 0, .2, 1);
}
.pipeline[data-stage="failed"] .pipeline__track i,
.pipeline[data-stage="cancelled"] .pipeline__track i { background: var(--danger); }
.pipeline__steps {
display: grid; grid-template-columns: repeat(4, 1fr);
margin: 11px 0 0; padding: 0; list-style: none;
font-size: 12.5px; color: var(--text-faint);
}
.pipeline__steps li { display: flex; align-items: center; gap: 7px; transition: color .3s; }
.pipeline__steps li:last-child { justify-content: flex-end; }
.pipeline__steps li:nth-child(2) { justify-content: center; }
.pipeline__steps li:nth-child(3) { justify-content: center; }
.pipeline__steps .dot {
width: 7px; height: 7px; border-radius: 50%;
background: var(--surface-3); border: 1px solid var(--border-strong);
transition: background .3s, border-color .3s, box-shadow .3s;
}
.pipeline__steps li[data-on="done"] { color: var(--text-dim); }
.pipeline__steps li[data-on="done"] .dot { background: var(--accent-deep); border-color: var(--accent-deep); }
.pipeline__steps li[data-on="active"] { color: var(--accent); font-weight: 550; }
.pipeline__steps li[data-on="active"] .dot { background: var(--accent); border-color: var(--accent);
box-shadow: 0 0 0 4px var(--accent-soft); }
.pipeline[data-stage="failed"] li[data-on="active"],
.pipeline[data-stage="cancelled"] li[data-on="active"] { color: var(--danger); }
.pipeline[data-stage="failed"] li[data-on="active"] .dot,
.pipeline[data-stage="cancelled"] li[data-on="active"] .dot {
background: var(--danger); border-color: var(--danger); box-shadow: 0 0 0 4px rgba(248,113,113,.16);
}
/* =========================================================================
LIVE STATUS
========================================================================= */
.live {
display: flex; align-items: center; gap: 11px;
margin-top: 18px; padding: 13px 16px;
border-radius: var(--radius-sm);
background: var(--surface); border: 1px solid var(--border);
font-size: 14px;
transition: border-color .3s, background .3s;
}
.live__text { font-weight: 550; }
.live__detail { color: var(--text-faint); font-size: 12.5px; font-family: var(--mono); margin-left: auto; }
.live[data-tone="idle"] { color: var(--text-dim); }
.live[data-tone="busy"] { color: var(--accent); border-color: rgba(34,211,238,.3); background: rgba(34,211,238,.05); }
.live[data-tone="speaking"] { color: var(--accent); border-color: rgba(34,211,238,.45); background: rgba(34,211,238,.07); }
.live[data-tone="ok"] { color: var(--ok); border-color: rgba(52,211,153,.28); background: rgba(52,211,153,.05); }
.live[data-tone="error"] { color: var(--danger); border-color: rgba(248,113,113,.32); background: rgba(248,113,113,.06); }
/* speaking waveform */
.live__wave { display: none; align-items: flex-end; gap: 3px; height: 18px; }
.live[data-tone="speaking"] .live__wave { display: flex; }
.live__wave i {
display: block; width: 3px; border-radius: 2px;
background: currentColor;
animation: wave 900ms ease-in-out infinite;
}
.live__wave i:nth-child(1) { height: 7px; animation-delay: 0ms; }
.live__wave i:nth-child(2) { height: 14px; animation-delay: 120ms; }
.live__wave i:nth-child(3) { height: 18px; animation-delay: 240ms; }
.live__wave i:nth-child(4) { height: 11px; animation-delay: 360ms; }
.live__wave i:nth-child(5) { height: 6px; animation-delay: 480ms; }
@keyframes wave {
0%, 100% { transform: scaleY(.35); opacity: .55; }
50% { transform: scaleY(1); opacity: 1; }
}
/* =========================================================================
FACTS
========================================================================= */
.facts { margin: 0; display: flex; flex-direction: column; gap: 1px; }
.fact {
display: flex; align-items: center; justify-content: space-between; gap: 12px;
padding: 8px 0;
border-bottom: 1px solid rgba(255,255,255,.045);
}
.fact:last-child { border-bottom: 0; }
.fact dt { font-size: 12.5px; color: var(--text-faint); }
.fact dd { margin: 0; font-size: 13px; font-weight: 500; text-align: right;
overflow: hidden; text-overflow: ellipsis; white-space: nowrap; max-width: 62%; }
.fact dd[data-tone="ok"] { color: var(--ok); }
.fact dd[data-tone="warn"] { color: var(--warn); }
.fact dd[data-tone="error"] { color: var(--danger); }
/* =========================================================================
HISTORY
========================================================================= */
.link-btn {
padding: 3px 8px; border: 0; border-radius: 6px; cursor: pointer;
font-family: inherit; font-size: 12.5px; color: var(--text-faint);
background: transparent; transition: color .18s, background .18s;
}
.link-btn:hover { color: var(--danger); background: rgba(248,113,113,.09); }
.history {
list-style: none; margin: 0; padding: 0;
display: flex; flex-direction: column; gap: 7px;
max-height: 340px; overflow-y: auto;
scrollbar-width: thin; scrollbar-color: var(--surface-3) transparent;
}
.history::-webkit-scrollbar { width: 7px; }
.history::-webkit-scrollbar-thumb { background: var(--surface-3); border-radius: 4px; }
.hitem {
display: flex; align-items: flex-start; gap: 10px;
padding: 9px 11px;
border-radius: var(--radius-sm);
background: var(--surface); border: 1px solid var(--border);
cursor: pointer;
transition: border-color .16s, background .16s, transform .16s;
}
.hitem:hover { border-color: var(--border-strong); background: var(--surface-3); }
.hitem:active { transform: scale(.99); }
.hitem__bar { width: 3px; align-self: stretch; border-radius: 2px; background: var(--text-faint); flex: none; }
.hitem[data-ok="true"] .hitem__bar { background: var(--ok); }
.hitem[data-ok="false"] .hitem__bar { background: var(--danger); }
.hitem[data-ok="pending"] .hitem__bar { background: var(--accent); animation: pulse 1s ease-in-out infinite; }
.hitem__body { min-width: 0; flex: 1; }
.hitem__text {
font-size: 13px; line-height: 1.45; color: var(--text);
display: -webkit-box; -webkit-line-clamp: 2; -webkit-box-orient: vertical; overflow: hidden;
}
.hitem__meta {
display: flex; align-items: center; gap: 8px; margin-top: 4px;
font-family: var(--mono); font-size: 11px; color: var(--text-faint);
}
.hitem__meta .err { color: var(--danger); font-family: var(--font); }
.empty { margin: 4px 0 0; font-size: 12.5px; color: var(--text-faint); line-height: 1.55; }
/* =========================================================================
TOASTS
========================================================================= */
.toasts {
position: fixed; right: 20px; bottom: 20px; z-index: 90;
display: flex; flex-direction: column; gap: 9px;
max-width: min(380px, calc(100vw - 40px));
}
.toast {
display: flex; align-items: flex-start; gap: 10px;
padding: 12px 14px;
border-radius: var(--radius-sm);
font-size: 13.5px; line-height: 1.45;
color: var(--text);
background: var(--surface-2);
border: 1px solid var(--border-strong);
box-shadow: var(--shadow);
animation: toast-in .25s cubic-bezier(.2, .8, .3, 1);
}
.toast[data-tone="error"] { border-color: rgba(248,113,113,.42); background: #1d1316; }
.toast[data-tone="ok"] { border-color: rgba(52,211,153,.38); background: #0f1a17; }
.toast__dot { width: 7px; height: 7px; border-radius: 50%; margin-top: 6px; flex: none; background: var(--text-faint); }
.toast[data-tone="error"] .toast__dot { background: var(--danger); }
.toast[data-tone="ok"] .toast__dot { background: var(--ok); }
.toast.is-out { animation: toast-out .2s ease-in forwards; }
@keyframes toast-in { from { opacity: 0; transform: translateY(10px) scale(.97); } }
@keyframes toast-out { to { opacity: 0; transform: translateY(6px) scale(.98); } }
/* =========================================================================
RESPONSIVE
========================================================================= */
@media (max-width: 980px) {
.layout { grid-template-columns: 1fr; }
.sidebar { order: 2; }
}
@media (max-width: 620px) {
.topbar { flex-wrap: wrap; gap: 10px; }
.brand__text p { display: none; }
.panel__hint { display: none; }
.actions { flex-direction: column; }
.btn { width: 100%; }
.pipeline__steps { font-size: 11.5px; }
.toasts { right: 12px; left: 12px; bottom: 12px; max-width: none; }
}
@media (prefers-reduced-motion: reduce) {
*, *::before, *::after { animation-duration: .001ms !important; transition-duration: .001ms !important; }
}
/* =========================================================================
SAVED AUDIO
========================================================================= */
.audio-list {
list-style: none; margin: 12px 0 0; padding: 0;
display: flex; flex-direction: column; gap: 6px;
max-height: 260px; overflow-y: auto;
scrollbar-width: thin; scrollbar-color: var(--surface-3) transparent;
}
.audio-list::-webkit-scrollbar { width: 7px; }
.audio-list::-webkit-scrollbar-thumb { background: var(--surface-3); border-radius: 4px; }
.aitem {
display: flex; align-items: center; gap: 9px;
padding: 7px 9px;
border-radius: var(--radius-sm);
background: var(--surface); border: 1px solid var(--border);
transition: border-color .16s, background .16s;
}
.aitem:hover { border-color: var(--border-strong); background: var(--surface-3); }
.aitem__play {
display: grid; place-items: center;
width: 26px; height: 26px; flex: none;
border-radius: 50%; cursor: pointer;
color: #04222a; background: var(--accent); border: 0;
transition: filter .15s, transform .12s;
}
.aitem__play svg { width: 12px; height: 12px; }
.aitem__play:hover { filter: brightness(1.1); }
.aitem__play:active { transform: scale(.92); }
.aitem__play[data-playing="true"] { background: var(--warn); }
.aitem__body { min-width: 0; flex: 1; }
.aitem__text {
font-size: 12.5px; line-height: 1.4; color: var(--text);
overflow: hidden; text-overflow: ellipsis; white-space: nowrap;
}
.aitem__meta {
font-family: var(--mono); font-size: 10.5px; color: var(--text-faint); margin-top: 2px;
overflow: hidden; text-overflow: ellipsis; white-space: nowrap;
}
.aitem__dl {
flex: none; display: grid; place-items: center;
width: 24px; height: 24px; border-radius: 6px;
color: var(--text-faint); text-decoration: none;
transition: color .15s, background .15s;
}
.aitem__dl svg { width: 13px; height: 13px; }
.aitem__dl:hover { color: var(--accent); background: var(--surface-3); }
/* replay marker on history rows whose audio is already saved */
.hitem__replay {
flex: none; display: grid; place-items: center;
width: 24px; height: 24px; border-radius: 50%;
border: 0; cursor: pointer;
color: var(--accent); background: var(--accent-soft);
transition: filter .15s, transform .12s;
}
.hitem__replay svg { width: 11px; height: 11px; }
.hitem__replay:hover { filter: brightness(1.25); }
.hitem__replay:active { transform: scale(.9); }

6
pronunciation.json Normal file
View File

@ -0,0 +1,6 @@
{
"_comment": "How the SIMULATOR should pronounce tricky words. Edit freely, then restart. The real robot never sees these - it always gets your original text. Matching is case-insensitive, on whole words only.",
"AGIBOT": "A G I bot",
"AgiBot": "A G I bot"
}

28
requirements.txt Normal file
View File

@ -0,0 +1,28 @@
# AGIBOT A3 - Voice Control : Python dependencies
# Install with: pip install -r requirements.txt
# --- core (always needed) ---------------------------------------------------
fastapi>=0.110
uvicorn[standard]>=0.27
python-dotenv>=1.0
pydantic>=2.5
# --- transports -------------------------------------------------------------
httpx>=0.27 # HTTP/REST transport (keep-alive client)
websockets>=12.0 # WebSocket transport + uvicorn ws support
# --- simulator audio (MOCK_LOCAL_AUDIO=true) --------------------------------
# Lets the simulator speak through this PC's speakers. Windows only; on other
# systems the app uses built-in commands and needs nothing installed.
comtypes>=1.2 ; sys_platform == "win32"
# --- optional ---------------------------------------------------------------
# paramiko>=3.4 # A3_TRANSPORT=ssh with password authentication
#
# Simulator audio without comtypes: Windows falls back to PowerShell's
# System.Speech (always present, just slower to start). macOS uses `say`.
# Linux needs: sudo apt install espeak-ng
#
# ROS 2 transport (A3_TRANSPORT=ros2) needs `rclpy`, which is NOT installed via
# pip - it comes from a ROS 2 installation. Run the app from a sourced ROS 2
# environment if you use that transport.

303
scripts/discover_robot.py Normal file
View File

@ -0,0 +1,303 @@
"""Probe an AGIBOT A3 to find out what it actually exposes.
Run this the moment you have the robot's IP, before touching .env:
python scripts/discover_robot.py 192.168.1.50
It is READ-ONLY by default: it pings, checks the ports AgiBot documents, and asks
each one for a deliberately nonexistent route. A **404 proves an AimRT HTTP
server is listening** on that port; connection-refused proves it is not. Nothing
is played and nothing is changed.
To run the decisive end-to-end test - which makes the robot actually talk:
python scripts/discover_robot.py 192.168.1.50 --speak "Hello, I am Expedition A3"
Port and route expectations come from AgiBot's A3 developer guide:
https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/07-02-audio_play
https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/03-second_develop_interface_overview
They are documented facts for A3 v3.1/v3.2 firmware, not guarantees: AgiBot does
not promise port stability, and AimRT has no service-discovery endpoint. Anything
this script cannot confirm, it reports as unknown rather than guessing.
"""
from __future__ import annotations
import argparse
import asyncio
import json
import platform
import subprocess
import sys
import time
from typing import Any, Dict, List, Optional, Tuple
try:
import httpx
except ImportError: # pragma: no cover
print("Install dependencies first: pip install -r requirements.txt")
raise SystemExit(2)
def _use_utf8_console() -> None:
"""Windows consoles default to a legacy code page; the robot speaks Chinese."""
for stream in (sys.stdout, sys.stderr):
try:
if stream is not None and hasattr(stream, "reconfigure"):
stream.reconfigure(encoding="utf-8", errors="replace")
except Exception:
pass
_use_utf8_console()
# Ports with published evidence. Audio ones first; the motion/mapping ports are
# probed only so the report is an honest map - they have nothing to do with TTS.
PORTS: List[Tuple[int, str, str]] = [
(59301, "TTS / agent RPC (HDU)", "audio"),
(56666, "HalAudioService - volume, PlayFile (HDU)", "audio"),
(51049, "ResourceService - audio resources (HDU)", "audio"),
(56444, "MotionCommandService (MDU)", "other"),
(56322, "motion control (MDU)", "other"),
(22, "SSH", "other"),
(8080, "common HTTP alternative", "other"),
]
TTS_SERVICE = "aimdk.protocol.TTSService"
NONEXISTENT_ROUTE = "/rpc/does.not.Exist/Nope"
OK = " [ OK ]"
NO = " [ -- ]"
WARN = " [ !! ]"
def head(title: str) -> None:
print("\n" + title)
print(" " + "-" * (len(title) + 2))
# --------------------------------------------------------------------------- #
# 1. ICMP
# --------------------------------------------------------------------------- #
def ping(host: str) -> Optional[float]:
flag = "-n" if platform.system().lower().startswith("win") else "-c"
started = time.perf_counter()
try:
result = subprocess.run(
["ping", flag, "2", host],
stdout=subprocess.PIPE, stderr=subprocess.STDOUT, timeout=12,
)
except Exception:
return None
if result.returncode != 0:
return None
return (time.perf_counter() - started) * 1000 / 2
# --------------------------------------------------------------------------- #
# 2. TCP
# --------------------------------------------------------------------------- #
async def tcp_open(host: str, port: int, timeout: float = 2.0) -> Optional[float]:
started = time.perf_counter()
try:
fut = asyncio.open_connection(host, port)
_, writer = await asyncio.wait_for(fut, timeout=timeout)
writer.close()
try:
await writer.wait_closed()
except Exception:
pass
return (time.perf_counter() - started) * 1000
except Exception:
return None
# --------------------------------------------------------------------------- #
# 3. Is it an AimRT HTTP server?
# --------------------------------------------------------------------------- #
async def aimrt_probe(client: "httpx.AsyncClient", host: str, port: int) -> Dict[str, Any]:
"""Ask for a route that cannot exist.
AimRT answers 404 for an unknown route, which is a positive identification
without invoking anything.
"""
url = "http://{0}:{1}{2}".format(host, port, NONEXISTENT_ROUTE)
try:
response = await client.post(url, json={}, headers={"Content-Type": "application/json"})
except Exception as exc:
return {"reachable": False, "detail": type(exc).__name__}
return {
"reachable": True,
"status": response.status_code,
"aimrt": response.status_code in (404, 500),
"cors": response.headers.get("access-control-allow-origin"),
"server": response.headers.get("server"),
"body": response.text[:160],
}
# --------------------------------------------------------------------------- #
# 4. The decisive test
# --------------------------------------------------------------------------- #
async def try_speak(client: "httpx.AsyncClient", host: str, port: int, text: str) -> Dict[str, Any]:
url = "http://{0}:{1}/rpc/{2}/PlayTTS".format(host, port, TTS_SERVICE)
payload = {
"text": text,
"priority_level": "INTERACTION_L6",
"domain": "discovery_probe",
"trace_id": "probe{0}".format(int(time.time())),
"is_interrupted": True,
}
started = time.perf_counter()
try:
response = await client.post(url, json=payload, headers={"Content-Type": "application/json"})
except Exception as exc:
return {"ok": False, "error": "{0}: {1}".format(type(exc).__name__, exc)}
elapsed = (time.perf_counter() - started) * 1000
try:
body = response.json()
except Exception:
body = {"raw": response.text[:300]}
success = None
if isinstance(body, dict):
success = body.get("is_sucess", body.get("is_success"))
return {
"ok": response.status_code == 200 and success is not False,
"status": response.status_code,
"ms": round(elapsed, 1),
"success_flag": success,
"trace_id": body.get("trace_id") if isinstance(body, dict) else None,
"sent_trace": payload["trace_id"],
"has_header_envelope": isinstance(body, dict) and isinstance(body.get("header"), dict),
"body": body,
}
# --------------------------------------------------------------------------- #
# main
# --------------------------------------------------------------------------- #
async def run(host: str, text: Optional[str], tts_port: int) -> int:
print("\n" + "=" * 72)
print(" AGIBOT A3 discovery -> {0}".format(host))
print("=" * 72)
# ---- 1. ping -----------------------------------------------------------
head("1. Reachability")
rtt = ping(host)
if rtt is None:
print(NO + " ping got no reply.")
print(" Not conclusive - some robots drop ICMP but still serve their ports.")
else:
print(OK + " ping replies (~{0:.0f} ms round trip)".format(rtt))
# ---- 2. ports ----------------------------------------------------------
head("2. Ports")
open_ports: List[int] = []
results = await asyncio.gather(*[tcp_open(host, port) for port, _, _ in PORTS])
for (port, label, kind), elapsed in zip(PORTS, results):
if elapsed is None:
print("{0} {1:<6} closed / filtered {2}".format(NO, port, label))
else:
open_ports.append(port)
print("{0} {1:<6} OPEN ({2:.0f} ms) {3}".format(OK, port, elapsed, label))
if not open_ports:
print("\n" + WARN + " Nothing answered. Check the IP, that the robot has finished")
print(" booting, and that both machines are on the same subnet.")
print(" See docs/NETWORK.md.")
return 1
# ---- 3. AimRT identification ------------------------------------------
head("3. Which ports speak AimRT HTTP JSON-RPC?")
aimrt_ports: List[int] = []
async with httpx.AsyncClient(timeout=httpx.Timeout(5.0, connect=3.0)) as client:
for port in open_ports:
if port == 22:
continue
info = await aimrt_probe(client, host, port)
if not info.get("reachable"):
print("{0} {1:<6} no HTTP response ({2})".format(NO, port, info.get("detail")))
continue
if info.get("aimrt"):
aimrt_ports.append(port)
extra = ""
if info.get("cors"):
extra = " CORS: {0}".format(info["cors"])
print("{0} {1:<6} AimRT RPC server (HTTP {2} for an unknown route){3}".format(
OK, port, info["status"], extra))
else:
print("{0} {1:<6} HTTP {2}, but not an AimRT RPC route".format(
NO, port, info.get("status")))
# ---- 4. summary / speak -------------------------------------------
head("4. Verdict")
if tts_port in aimrt_ports:
print(OK + " Port {0} looks like the A3 TTS RPC service.".format(tts_port))
print("")
print(" Put this in .env:")
print(" ROBOT_MODE=real")
print(" ROBOT_IP={0}".format(host))
print(" ROBOT_PORT={0}".format(tts_port))
print(" A3_TRANSPORT=aimdk")
elif aimrt_ports:
print(WARN + " AimRT RPC found on {0}, but not on the documented TTS port {1}.".format(
", ".join(str(p) for p in aimrt_ports), tts_port))
print(" Try each with --speak, and set ROBOT_PORT to whichever talks.")
else:
print(WARN + " No AimRT RPC server found.")
print(" Most likely the RPC ports are bound only to the robot's internal")
print(" 10.42.10.x network. SSH in and check: ss -ltnp | grep 59301")
print(" See docs/AGIBOT_A3_INTEGRATION.md section 6.")
if text:
head("5. Speech test (the robot should talk now)")
print(' Sending: "{0}"'.format(text))
result = await try_speak(client, host, tts_port, text)
if result.get("ok"):
print(OK + " Accepted in {0} ms.".format(result["ms"]))
print(" is_sucess : {0}".format(result["success_flag"]))
print(" trace_id sent : {0}".format(result["sent_trace"]))
print(" trace_id back : {0}".format(result["trace_id"]))
if result["trace_id"] and result["trace_id"] != result["sent_trace"]:
print(" (differs, as documented - Stop must use the returned id)")
if result["has_header_envelope"]:
print(" header envelope: present")
print("")
print(" If you heard the robot, the integration is confirmed.")
print(" If it was silent, check the robot's volume and whether its")
print(" TTS needs internet access - see docs/AGIBOT_A3_INTEGRATION.md section 7.")
else:
print(WARN + " Speech request failed.")
for key in ("status", "error", "success_flag"):
if result.get(key) is not None:
print(" {0:<14}: {1}".format(key, result[key]))
if result.get("status") == 404:
print(" 404 = the port is right but the route is wrong. Check the")
print(" service/method names against your firmware's documentation.")
if result.get("body"):
print(" body : {0}".format(
json.dumps(result["body"], ensure_ascii=False)[:300]))
return 1
print("\n" + "=" * 72 + "\n")
return 0
def main() -> None:
parser = argparse.ArgumentParser(
description="Probe an AGIBOT A3 for its speech interface.",
epilog="Read-only unless --speak is given.",
)
parser.add_argument("ip", help="The robot's IP address on your network.")
parser.add_argument("--speak", metavar="TEXT", default=None,
help="Also send this text - THE ROBOT WILL TALK.")
parser.add_argument("--port", type=int, default=59301,
help="TTS RPC port (default: 59301, AgiBot's documented port).")
args = parser.parse_args()
raise SystemExit(asyncio.run(run(args.ip, args.speak, args.port)))
if __name__ == "__main__":
main()

194
scripts/fake_a3_server.py Normal file
View File

@ -0,0 +1,194 @@
"""A stand-in for the AGIBOT A3's speech RPC, for testing the REAL adapter path.
The mock robot (ROBOT_MODE=mock) tests the app. This tests the *wire protocol* -
it speaks the interface AgiBot documents for the A3, so you can run the app in
ROBOT_MODE=real against 127.0.0.1 and exercise `aimdk_transport.py` itself:
the URL shape, the JSON body, the trace_id round trip, chunking, Stop, and the
error paths. When the real robot arrives, only ROBOT_IP changes.
It mirrors the documented contract, including its quirks:
* route POST /rpc/<service>/<method>, Content-Type: application/json
* success flag `is_sucess` (one 'c' - as printed in AgiBot's docs)
* trace_id the reply appends a random suffix to the one you sent
* size limit 1024 bytes of UTF-8 on `text`
* unknown route 404; RPC-level failure 500
Docs: https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/07-02-audio_play
Run:
python scripts/fake_a3_server.py # listens on 127.0.0.1:59301
Then in .env:
ROBOT_MODE=real
ROBOT_IP=127.0.0.1
ROBOT_PORT=59301
NOTE: this is a TEST DOUBLE written from public documentation. It is not the
robot, and passing against it proves the client is well-formed - not that the
robot's firmware behaves identically. Verify against the real unit.
"""
from __future__ import annotations
import argparse
import json
import secrets
import sys
import time
from typing import Any, Dict, Optional
try:
import uvicorn
from fastapi import FastAPI, Request, Response
from fastapi.responses import JSONResponse
except ImportError: # pragma: no cover
print("Install dependencies first: pip install -r requirements.txt")
raise SystemExit(2)
def _use_utf8_console() -> None:
"""Windows consoles default to a legacy code page; the robot speaks Chinese."""
for stream in (sys.stdout, sys.stderr):
try:
if stream is not None and hasattr(stream, "reconfigure"):
stream.reconfigure(encoding="utf-8", errors="replace")
except Exception:
pass
_use_utf8_console()
SERVICE = "aimdk.protocol.TTSService"
MAX_BYTES = 1024
app = FastAPI(title="Fake AGIBOT A3 speech RPC", docs_url=None, openapi_url=None)
STATE: Dict[str, Any] = {"utterances": {}, "active": None, "count": 0}
def _log(kind: str, message: str) -> None:
print(" {0} {1:<9} {2}".format(time.strftime("%H:%M:%S"), kind, message), flush=True)
@app.post("/rpc/{service}/{method}")
async def rpc(service: str, method: str, request: Request) -> Response:
if service != SERVICE:
_log("404", "unknown service {0}".format(service))
return JSONResponse({"error": "no such service"}, status_code=404)
# AimRT requires this header and rejects anything else.
content_type = (request.headers.get("content-type") or "").split(";")[0].strip()
if content_type != "application/json":
_log("500", "bad Content-Type: {0!r}".format(content_type))
return JSONResponse({"error": "unsupported content type"}, status_code=500)
try:
body = json.loads(await request.body() or b"{}")
except json.JSONDecodeError:
return JSONResponse({"error": "malformed json"}, status_code=500)
handler = {
"PlayTTS": _play_tts,
"StopTTSTraceId": _stop_tts,
"GetAudioStatus": _get_status,
}.get(method)
if handler is None:
_log("404", "unknown method {0}".format(method))
return JSONResponse({"error": "no such method"}, status_code=404)
return handler(body)
def _play_tts(body: Dict[str, Any]) -> Response:
text = body.get("text") or ""
size = len(text.encode("utf-8"))
if size > MAX_BYTES:
_log("REJECT", "text is {0} bytes (limit {1})".format(size, MAX_BYTES))
return JSONResponse(_reply(body, False, "text exceeds 1024 bytes"), status_code=200)
if not text.strip():
_log("REJECT", "empty text")
return JSONResponse(_reply(body, False, "empty text"), status_code=200)
trace = "{0}_{1}".format(body.get("trace_id") or "trace", secrets.token_urlsafe(16))
STATE["utterances"][trace] = {"text": text, "started": time.time(), "stopped": False}
STATE["active"] = trace
STATE["count"] += 1
_log("SPEAK", '"{0}" [{1} bytes, priority={2}, interrupt={3}]'.format(
text if len(text) <= 70 else text[:67] + "...",
size, body.get("priority_level"), body.get("is_interrupted"),
))
return JSONResponse(_reply(body, True, "", trace), status_code=200)
def _stop_tts(body: Dict[str, Any]) -> Response:
trace = body.get("trace_id")
entry = STATE["utterances"].get(trace)
if entry is None:
_log("STOP", "unknown trace_id {0!r}".format(trace))
return JSONResponse({"is_sucess": False, "error_message": "unknown trace_id"}, status_code=200)
entry["stopped"] = True
if STATE["active"] == trace:
STATE["active"] = None
_log("STOP", "stopped {0}".format(trace))
return JSONResponse({"is_sucess": True, "error_message": "", "trace_id": trace}, status_code=200)
def _get_status(body: Dict[str, Any]) -> Response:
trace = body.get("trace_id")
entry = STATE["utterances"].get(trace)
if entry is None:
status = "TTSStatusType_NOTInQue"
elif entry["stopped"]:
status = "TTSStatusType_Stop"
elif time.time() - entry["started"] < _estimate(entry["text"]):
status = "TTSStatusType_Playing"
else:
status = "TTSStatusType_End"
return JSONResponse({"trace_id": trace, "tts_status": status, "is_sucess": True}, status_code=200)
def _reply(body: Dict[str, Any], ok: bool, error: str, trace: Optional[str] = None) -> Dict[str, Any]:
return {
"text": body.get("text", ""),
"priority_level": body.get("priority_level", ""),
"priority_weight": 0,
"domain": body.get("domain", ""),
"trace_id": trace or body.get("trace_id", ""),
"is_sucess": ok, # the documented spelling - do not "fix" it
"error_message": error,
"estimated_duration": 0,
}
def _estimate(text: str) -> float:
words = max(1, len(text.split()))
return max(1.0, words / 150 * 60)
def main() -> None:
parser = argparse.ArgumentParser(description="Fake AGIBOT A3 speech RPC endpoint.")
parser.add_argument("--host", default="127.0.0.1")
parser.add_argument("--port", type=int, default=59301)
args = parser.parse_args()
print("")
print(" Fake AGIBOT A3 speech RPC")
print(" " + "-" * 52)
print(" Listening : http://{0}:{1}".format(args.host, args.port))
print(" Endpoint : POST /rpc/{0}/PlayTTS".format(SERVICE))
print("")
print(" Point the app at it with:")
print(" ROBOT_MODE=real")
print(" ROBOT_IP={0}".format(args.host))
print(" ROBOT_PORT={0}".format(args.port))
print("")
print(" This is a TEST DOUBLE built from public docs - not the robot.")
print("")
uvicorn.run(app, host=args.host, port=args.port, log_level="warning", access_log=False)
if __name__ == "__main__":
main()

267
scripts/selftest.py Normal file
View File

@ -0,0 +1,267 @@
"""End-to-end self test.
Exercises the whole stack the way the browser does - REST for commands, WebSocket
for the lifecycle - and prints a pass/fail report.
1. start the server: python backend/main.py
2. in another window: python scripts/selftest.py
Works against mock mode out of the box. Against a real robot it also works, but
the robot will actually speak, so only run it when that is fine.
"""
from __future__ import annotations
import argparse
import asyncio
import json
import sys
import time
from typing import Any, Dict, List, Optional
try:
import httpx
except ImportError: # pragma: no cover
print("Install dependencies first: pip install -r requirements.txt")
raise SystemExit(2)
try:
from websockets.asyncio.client import connect as ws_connect
except Exception: # pragma: no cover
from websockets.client import connect as ws_connect # type: ignore
def _use_utf8_console() -> None:
"""Windows consoles default to a legacy code page; the robot speaks Chinese."""
for stream in (sys.stdout, sys.stderr):
try:
if stream is not None and hasattr(stream, "reconfigure"):
stream.reconfigure(encoding="utf-8", errors="replace")
except Exception:
pass
_use_utf8_console()
PASS = " [PASS]"
FAIL = " [FAIL]"
INFO = " [ .. ]"
results: List[tuple] = []
def check(name: str, condition: bool, detail: str = "") -> bool:
results.append((name, condition, detail))
print("{0} {1}{2}".format(PASS if condition else FAIL, name, " - " + detail if detail else ""))
return condition
class Listener:
"""Collects WebSocket events in the background, like the dashboard does."""
def __init__(self, url: str) -> None:
self.url = url
self.events: List[Dict[str, Any]] = []
self.hello: Optional[Dict[str, Any]] = None
self._ws: Any = None
self._task: Optional[asyncio.Task] = None
async def __aenter__(self) -> "Listener":
self._ws = await ws_connect(self.url)
self._task = asyncio.create_task(self._read())
for _ in range(50):
if self.hello is not None:
break
await asyncio.sleep(0.05)
return self
async def __aexit__(self, *exc: Any) -> None:
if self._task:
self._task.cancel()
if self._ws:
await self._ws.close()
async def _read(self) -> None:
async for raw in self._ws:
message = json.loads(raw)
if message.get("type") == "hello":
self.hello = message["data"]
self.events.append(message)
def stages(self, request_id: Optional[str] = None) -> List[str]:
out = []
for event in self.events:
if event.get("type") != "speech.progress":
continue
data = event["data"]
if request_id and data.get("requestId") != request_id:
continue
out.append(data["stage"])
return out
async def wait_for_event(self, event_type: str, timeout: float = 10.0) -> bool:
"""Wait for the next event of a type to arrive (ignoring earlier ones)."""
seen = sum(1 for e in self.events if e.get("type") == event_type)
deadline = time.time() + timeout
while time.time() < deadline:
if sum(1 for e in self.events if e.get("type") == event_type) > seen:
return True
await asyncio.sleep(0.1)
return False
async def wait_for_stage(self, stage: str, request_id: str, timeout: float = 30.0) -> bool:
deadline = time.time() + timeout
while time.time() < deadline:
if stage in self.stages(request_id):
return True
await asyncio.sleep(0.05)
return False
async def run(base: str) -> int:
ws_url = base.replace("http://", "ws://").replace("https://", "wss://") + "/ws"
async with httpx.AsyncClient(base_url=base, timeout=20.0) as client:
print("\n=== 1. Server & configuration " + "=" * 40)
try:
health = (await client.get("/api/health")).json()
except Exception as exc:
print(FAIL + " server unreachable at {0} ({1})".format(base, exc))
print("\n Start it first: python backend/main.py\n")
return 1
check("GET /api/health returns ok", health.get("status") == "ok", "mode=" + str(health.get("mode")))
config = (await client.get("/api/config")).json()
check("GET /api/config returns a mode", config.get("mode") in ("mock", "real"))
check("no blocking config errors",
not [i for i in config.get("issues", []) if i["level"] == "error"],
str(config.get("issues")))
print("\n=== 2. Dashboard is served " + "=" * 43)
page = await client.get("/")
check("GET / serves the dashboard", page.status_code == 200 and "AGIBOT" in page.text)
for asset in ("/styles/main.css", "/js/app.js", "/js/api.js", "/js/socket.js", "/js/ui.js"):
response = await client.get(asset)
check("GET {0}".format(asset), response.status_code == 200)
print("\n=== 3. Robot connection " + "=" * 46)
status = None
for _ in range(40): # allow the supervisor a moment to connect
status = (await client.get("/api/robot/status")).json()
if status.get("connected"):
break
await asyncio.sleep(0.25)
check("robot reports connected", bool(status and status.get("connected")),
"state=" + str(status.get("state") if status else "?"))
async with Listener(ws_url) as listener:
check("WebSocket accepts connection and sends hello", listener.hello is not None)
if listener.hello:
check("hello carries config + status + history",
all(k in listener.hello for k in ("config", "status", "history")))
print("\n=== 4. Speak (the main workflow) " + "=" * 37)
text = "Hello, welcome to our company."
started = time.perf_counter()
response = await client.post("/api/robot/speak", json={"text": text})
http_ms = (time.perf_counter() - started) * 1000
check("POST /api/robot/speak accepted", response.status_code == 200,
"HTTP {0}".format(response.status_code))
if response.status_code != 200:
print(" body:", response.text[:300])
return 1
body = response.json()
request_id = body["requestId"]
check("response reports success", body.get("success") is True, json.dumps(body))
check("HTTP call returns before speech ends (< 3s)", http_ms < 3000,
"{0:.0f} ms".format(http_ms))
check("ack latency reported", isinstance(body.get("ackLatencyMs"), int),
"{0} ms".format(body.get("ackLatencyMs")))
got_speaking = await listener.wait_for_stage("speaking", request_id, timeout=10)
check("lifecycle reaches 'speaking'", got_speaking)
got_completed = await listener.wait_for_stage("completed", request_id, timeout=60)
check("lifecycle reaches 'completed'", got_completed)
stages = listener.stages(request_id)
check("full stage sequence observed",
["sending", "processing", "speaking", "completed"] == [
s for s in stages if s in ("sending", "processing", "speaking", "completed")
],
" -> ".join(stages))
print("\n=== 5. Stop " + "=" * 58)
long_text = "This is a much longer sentence used to verify that the stop button " \
"interrupts an utterance while the robot is still speaking it out loud."
body2 = (await client.post("/api/robot/speak", json={"text": long_text})).json()
rid2 = body2["requestId"]
await listener.wait_for_stage("speaking", rid2, timeout=10)
stop_response = await client.post("/api/robot/stop")
check("POST /api/robot/stop accepted", stop_response.status_code == 200)
cancelled = await listener.wait_for_stage("cancelled", rid2, timeout=10)
check("utterance reports 'cancelled'", cancelled, " -> ".join(listener.stages(rid2)))
print("\n=== 6. Error handling " + "=" * 48)
empty = await client.post("/api/robot/speak", json={"text": " "})
check("empty text rejected with 400", empty.status_code == 400,
"HTTP {0} {1}".format(empty.status_code, empty.text[:120]))
too_long = await client.post(
"/api/robot/speak", json={"text": "x " * (config.get("maxLength", 1000) + 50)}
)
check("over-long text rejected with 400", too_long.status_code == 400,
"HTTP {0}".format(too_long.status_code))
missing = await client.post("/api/robot/speak", json={})
check("malformed body rejected", missing.status_code in (400, 422),
"HTTP {0}".format(missing.status_code))
print("\n=== 7. History " + "=" * 55)
history = (await client.get("/api/speech/history")).json()
check("history contains the utterances", history.get("count", 0) >= 2,
"count={0}".format(history.get("count")))
first = history["items"][0] if history.get("items") else {}
check("history entries carry stage + latency",
"stage" in first and "ackLatencyMs" in first)
cleared = (await client.delete("/api/speech/history")).json()
check("history cleared", cleared.get("success") is True,
"removed={0}".format(cleared.get("removed")))
after = (await client.get("/api/speech/history")).json()
check("history is empty afterwards", after.get("count") == 0)
print("\n=== 8. Diagnostics & recovery " + "=" * 40)
diagnostics = (await client.get("/api/robot/diagnostics")).json()
check("GET /api/robot/diagnostics works", "status" in diagnostics and "adapter" in diagnostics)
reconnect = (await client.post("/api/robot/reconnect")).json()
check("POST /api/robot/reconnect works", reconnect.get("success") is True)
# Status is pushed once per health interval, so wait for one rather
# than assuming the earlier checks took long enough to see it.
budget = float(config.get("healthInterval") or 5.0) + 4.0
saw_status_event = await listener.wait_for_event("robot.status", timeout=budget)
check("robot.status events pushed over WebSocket", saw_status_event,
"waited up to {0:.0f}s".format(budget))
passed = sum(1 for _, ok, _ in results if ok)
total = len(results)
print("\n" + "=" * 72)
print(" {0}/{1} checks passed".format(passed, total))
if passed != total:
print("\n Failures:")
for name, ok, detail in results:
if not ok:
print(" - {0} {1}".format(name, detail))
print("=" * 72 + "\n")
return 0 if passed == total else 1
def main() -> None:
parser = argparse.ArgumentParser(description="End-to-end test for the A3 voice dashboard.")
parser.add_argument("--url", default="http://127.0.0.1:8000", help="Base URL of the running app.")
args = parser.parse_args()
raise SystemExit(asyncio.run(run(args.url.rstrip("/"))))
if __name__ == "__main__":
main()

130
scripts/voices.py Normal file
View File

@ -0,0 +1,130 @@
"""List and audition the PC voices available to the simulator.
python scripts/voices.py # list what is installed
python scripts/voices.py --demo # speak a sample line in each voice
python scripts/voices.py --try Zira # hear one voice, with .env's rate/volume
Whichever you prefer goes in .env:
MOCK_LOCAL_AUDIO=true
MOCK_VOICE=Zira
IMPORTANT - this is the *simulator's* voice, not the robot's. The real AGIBOT A3
synthesises speech on-board; AgiBot does not publish which engine or timbre it
uses, so no PC voice can be claimed to match it. This is a stand-in for
rehearsing a demo. See docs/AGIBOT_A3_INTEGRATION.md.
"""
from __future__ import annotations
import argparse
import platform
import sys
import time
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
def _use_utf8_console() -> None:
for stream in (sys.stdout, sys.stderr):
try:
if stream is not None and hasattr(stream, "reconfigure"):
stream.reconfigure(encoding="utf-8", errors="replace")
except Exception:
pass
_use_utf8_console()
from backend.config.settings import get_settings # noqa: E402
from backend.robot.local_audio import create_local_voice, list_local_voices # noqa: E402
SAMPLE = "Good afternoon, and welcome to our showroom. I am the AGIBOT A3 humanoid robot."
def speak_with(hint, text, rate, volume, pitch=0) -> None:
voice = create_local_voice(voice_hint=hint, rate=rate, volume=volume, pitch=pitch)
if voice is None:
print(" No speech engine available on this PC.")
return
try:
voice.start(text)
while voice.is_speaking():
time.sleep(0.05)
finally:
voice.close()
def main() -> None:
parser = argparse.ArgumentParser(description="List and audition simulator voices.")
parser.add_argument("--demo", action="store_true", help="Speak a sample in every voice.")
parser.add_argument("--try", dest="which", metavar="NAME", help="Speak a sample in one voice.")
parser.add_argument("--text", default=SAMPLE, help="Say this instead of the default sample.")
args = parser.parse_args()
settings = get_settings()
rate, volume = settings.mock.speech_rate, settings.mock.speech_volume
pitch = settings.mock.speech_pitch
if args.which:
print('\n Voice "{0}" (rate {1}, volume {2}, pitch {3})'.format(
args.which, rate, volume, pitch))
print(" " + args.text + "\n")
speak_with(args.which, args.text, rate, volume, pitch)
return
if settings.mock.voice_engine == "gemini":
from backend.robot.gemini_voice import VOICES as GEMINI_VOICES
current = (settings.mock.gemini_voice or "").lower()
print("\n Gemini neural voices (active: MOCK_VOICE_ENGINE=gemini)")
print(" " + "-" * 58)
for name, tone in sorted(GEMINI_VOICES.items()):
mark = "*" if name.lower() == current else " "
note = " <- young + male, closest to the robot's Yunxiao" \
if name in ("Puck", "Fenrir") else ""
print(" {0} {1:<16} {2}{3}".format(mark, name, tone, note))
print("")
print(" * = current GEMINI_VOICE. Change it in .env, then warm your lines:")
print(' python scripts/warm_voice.py "Welcome to our showroom."')
voices = list_local_voices()
print("\n System voices on this PC (used when MOCK_VOICE_ENGINE=system,")
print(" and as the fallback if Gemini fails)")
print(" " + "-" * 58)
if not voices:
if platform.system().lower().startswith("win"):
print(" None found.")
else:
print(" Listing is Windows-only; on macOS the simulator uses `say`,")
print(" on Linux `espeak-ng`.")
return
for description, source in voices:
marker = "*" if settings.mock.voice and settings.mock.voice.lower() in description.lower() else " "
print(" {0} {1:<48} [{2}]".format(marker, description, source))
print("")
print(" * = currently selected by MOCK_VOICE={0}".format(settings.mock.voice or "(unset)"))
print(" 'onecore' voices are Windows' newer, better-sounding set.")
print("")
print(" Hear one: python scripts/voices.py --try Zira")
print(" Hear all: python scripts/voices.py --demo")
print("")
print(" Want more voices (including Mandarin, to match the A3's demo language)?")
print(" Windows Settings > Time & language > Speech > Manage voices > Add voices.")
print("")
if args.demo:
for description, _ in voices:
short = description.replace("Microsoft ", "").split(" - ")[0]
print(" ->", description)
speak_with(description, "{0} speaking. {1}".format(short, args.text),
rate, volume, pitch)
time.sleep(0.4)
print("")
if __name__ == "__main__":
main()

125
scripts/warm_voice.py Normal file
View File

@ -0,0 +1,125 @@
"""Pre-synthesise demo lines so the Gemini voice speaks instantly.
Cloud TTS costs a network round trip - roughly 4 s for a sentence, 8 s for a
paragraph. That is fine while rehearsing and painful in front of an audience.
This warms the on-disk cache ahead of time, so a warmed line plays with **no
delay at all**.
python scripts/warm_voice.py "Good afternoon, welcome to our showroom."
python scripts/warm_voice.py --file demo_lines.txt
python scripts/warm_voice.py --stats
python scripts/warm_voice.py --clear
Audio is saved in audio_library/ as ordinary .wav files with readable names, so
it survives restarts, plays in any media player, and needs no internet once
saved. The server reads the same folder - warm before you present.
Only relevant when MOCK_VOICE_ENGINE=gemini.
"""
from __future__ import annotations
import argparse
import sys
import time
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
def _use_utf8_console() -> None:
for stream in (sys.stdout, sys.stderr):
try:
if stream is not None and hasattr(stream, "reconfigure"):
stream.reconfigure(encoding="utf-8", errors="replace")
except Exception:
pass
_use_utf8_console()
from backend.config.settings import get_settings # noqa: E402
from backend.robot.gemini_voice import GeminiVoice # noqa: E402
def build(settings) -> GeminiVoice:
return GeminiVoice(
api_key=settings.mock.gemini_api_key or "",
model=settings.mock.gemini_model,
voice=settings.mock.gemini_voice,
style=settings.mock.gemini_style,
chunk_chars=settings.mock.gemini_chunk_chars,
fallback=None,
)
def main() -> None:
parser = argparse.ArgumentParser(description="Pre-synthesise lines for the Gemini voice.")
parser.add_argument("lines", nargs="*", help="Lines to warm.")
parser.add_argument("--file", help="Text file, one line per utterance.")
parser.add_argument("--stats", action="store_true", help="Show library size and exit.")
parser.add_argument("--clear", action="store_true", help="Delete all saved audio and exit.")
args = parser.parse_args()
settings = get_settings()
if not settings.mock.gemini_api_key:
print("\n GEMINI_API_KEY is not set in .env - nothing to warm.\n")
raise SystemExit(1)
voice = build(settings)
if args.clear:
voice.clear_cache()
print("\n Cache cleared.\n")
return
if args.stats or not (args.lines or args.file):
stats = voice.cache_stats()
print("\n Saved audio")
print(" " + "-" * 52)
print(" Entries : {0}".format(stats["entries"]))
print(" Size : {0:.1f} MB".format(stats["bytes"] / 1_048_576))
print(" Location: {0}".format(stats["dir"]))
print(" Voice : {0} ({1})".format(settings.mock.gemini_voice, settings.mock.gemini_model))
if not (args.lines or args.file):
print("")
print(' Warm a line: python scripts/warm_voice.py "Welcome to our showroom."')
print(" Warm a file: python scripts/warm_voice.py --file demo_lines.txt")
print("")
return
lines = list(args.lines)
if args.file:
path = Path(args.file)
if not path.exists():
print("\n No such file: {0}\n".format(path))
raise SystemExit(1)
lines += [ln.strip() for ln in path.read_text(encoding="utf-8").splitlines() if ln.strip()]
print("\n Warming {0} line(s) with voice '{1}'...".format(len(lines), settings.mock.gemini_voice))
print(" " + "-" * 60)
total_new = 0
started = time.time()
for index, line in enumerate(lines, 1):
t0 = time.time()
try:
fetched = voice.warm(line)
except Exception as exc:
print(" {0:>3}. FAILED {1}".format(index, exc))
continue
total_new += fetched
state = "cached already" if fetched == 0 else "fetched {0} chunk(s)".format(fetched)
preview = line if len(line) <= 48 else line[:45] + "..."
print(" {0:>3}. {1:<50} {2:>16} {3:5.1f}s".format(
index, preview, state, time.time() - t0))
stats = voice.cache_stats()
print(" " + "-" * 60)
print(" {0} new clip(s) in {1:.1f}s. Library now holds {2} clips ({3:.1f} MB).".format(
total_new, time.time() - started, stats["entries"], stats["bytes"] / 1_048_576))
print(" These lines will now speak instantly.\n")
voice.close()
if __name__ == "__main__":
main()

63
start.bat Normal file
View File

@ -0,0 +1,63 @@
@echo off
REM ===========================================================================
REM AGIBOT A3 - Voice Control | Windows launcher
REM Double-click this file, then open http://localhost:8000
REM ===========================================================================
setlocal
cd /d "%~dp0"
echo.
echo AGIBOT A3 - Voice Control
echo ----------------------------------------------
REM --- locate Python ---------------------------------------------------------
set "PY=python"
where python >nul 2>&1
if errorlevel 1 (
where py >nul 2>&1
if errorlevel 1 (
echo [ERROR] Python was not found on this PC.
echo Install Python 3.9+ from https://python.org and tick
echo "Add python.exe to PATH" during setup.
echo.
pause
exit /b 1
)
set "PY=py -3"
)
REM --- first-run config ------------------------------------------------------
if not exist ".env" (
echo Creating .env from .env.example ...
copy /y ".env.example" ".env" >nul
)
REM --- dependencies ----------------------------------------------------------
%PY% -c "import fastapi, uvicorn, dotenv, httpx" >nul 2>&1
if errorlevel 1 (
echo Installing dependencies ^(first run only^) ...
%PY% -m pip install --disable-pip-version-check -q -r requirements.txt
if errorlevel 1 (
echo [ERROR] Dependency installation failed. Run this by hand:
echo %PY% -m pip install -r requirements.txt
echo.
pause
exit /b 1
)
)
REM --- run -------------------------------------------------------------------
echo Starting server ... http://localhost:8000
echo Press Ctrl+C to stop.
echo.
REM Open the browser a few seconds late, so the page loads against a server that
REM is already listening instead of showing a connection error.
start "" /min cmd /c "timeout /t 4 /nobreak >nul & start "" http://localhost:8000"
%PY% backend\main.py
echo.
echo Server stopped.
pause
endlocal

32
start.sh Normal file
View File

@ -0,0 +1,32 @@
#!/usr/bin/env bash
# ===========================================================================
# AGIBOT A3 - Voice Control | Linux / macOS launcher
# ./start.sh then open http://localhost:8000
# ===========================================================================
set -euo pipefail
cd "$(dirname "$0")"
echo
echo " AGIBOT A3 - Voice Control"
echo " ----------------------------------------------"
PY="${PYTHON:-python3}"
if ! command -v "$PY" >/dev/null 2>&1; then
echo " [ERROR] python3 not found. Install Python 3.9 or newer."
exit 1
fi
if [ ! -f .env ]; then
echo " Creating .env from .env.example ..."
cp .env.example .env
fi
if ! "$PY" -c "import fastapi, uvicorn, dotenv, httpx" >/dev/null 2>&1; then
echo " Installing dependencies (first run only) ..."
"$PY" -m pip install --disable-pip-version-check -q -r requirements.txt
fi
echo " Starting server ... open http://localhost:8000"
echo " Press Ctrl+C to stop."
echo
exec "$PY" backend/main.py