diff --git a/plugins/speech/.sirgent-plugin/plugin.json b/plugins/speech/.sirgent-plugin/plugin.json index 61f53aec4a..64efffd9fa 100644 --- a/plugins/speech/.sirgent-plugin/plugin.json +++ b/plugins/speech/.sirgent-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "speech", "version": "1.0.0", - "description": "Speech capabilities for SirGent AI. Narrates responses aloud through a Stop hook using ElevenLabs text-to-speech, with /speak and /hush slash commands, a voice skill, and playback via native players or afplay/mpv/powershell.", + "description": "Speech capabilities for SirGent AI. Narrates responses aloud through a Stop hook using ElevenLabs text-to-speech (with /speak and /hush slash commands) and takes voice input via /dictate — record from your mic, transcribe with ElevenLabs STT, and send the words as your prompt.", "author": { "name": "SirGent AI", "email": "support@sirgent.ai" diff --git a/plugins/speech/README.md b/plugins/speech/README.md index a2686bc17a..a309ee08b8 100644 --- a/plugins/speech/README.md +++ b/plugins/speech/README.md @@ -1,8 +1,9 @@ # speech Speech capabilities for SirGent AI: voice narration of responses through -ElevenLabs text-to-speech, an on-demand `/speak` command, and a `/hush` -mute toggle. +ElevenLabs text-to-speech, voice dictation into your prompts through +ElevenLabs speech-to-text, an on-demand `/speak` command, a `/hush` mute +toggle, and a `/dictate` command for voice input. ## What it does @@ -11,6 +12,7 @@ mute toggle. | Stop hook narrator | `hooks/speak_response.py` | When SirGent finishes a turn, the final assistant message is distilled (markdown stripped, capped at ~400 chars), converted to MP3 with ElevenLabs TTS, and played through the platform's audio player. | | `/speak` command | `commands/speak.md` | Speak any text the user supplies, on demand. | | `/hush` command | `commands/hush.md` | Mute/unmute the Stop-hook narrator with a flag file (`~/.sirgent/speech-disabled`). | +| `/dictate` command | `commands/dictate.md` + `hooks/dictate_stt.py` | Voice input: record from the default mic, transcribe with ElevenLabs STT, and deliver the words as your prompt to SirGent. | | Voice skill | `skills/voice/SKILL.md` | Guidance for preparing natural spoken text, calling the TTS API, and playing audio on macOS/Linux/Windows. | ## Setup @@ -23,9 +25,14 @@ mute toggle. (default `21m00Tcm4TlvDq8ikWAM`, "Rachel") - `ELEVENLABS_MODEL_ID` — default `eleven_multilingual_v2` - `SPEECH_MAX_CHARS` — narration cap, default `400` + - `SPEECH_DICTATE_SECS` — max dictation length, default `15` + - `SPEECH_STT_MODEL` — STT model, default `scribe_v1` 3. Make sure a player exists for your platform: `afplay` ships with macOS; on Linux install `mpv` or `ffmpeg`; on Windows the PowerShell `Media.SoundPlayer` path needs no extra install. +4. For `/dictate`, install a mic recorder: `sox` (recommended, all + platforms via Homebrew/apt/choco), or use `arecord` (ALSA Linux) / + `ffmpeg` (macOS avfoundation). Without an API key the narrator stays silent (one setup hint on first run) and everything else keeps working. @@ -38,14 +45,15 @@ and everything else keeps working. The hook always exits 0 and never blocks the session: a missing key, an API error, or a missing player degrades to a `systemMessage` note (or silence), -never to a failed turn. +never to a failed turn. `/dictate` fails soft the same way: it prints a +`speech:` reason on stderr and exits 1 rather than inventing content. ## Testing Try it without touching hooks: ```sh -echo '{"transcript_path": ""}' | \ +echo '{}' | \ ELEVENLABS_API_KEY=sk-... python3 hooks/speak_response.py <<< '{}' ``` @@ -63,10 +71,12 @@ speech/ │ └── plugin.json # plugin metadata ├── commands/ │ ├── speak.md # /speak — say arbitrary text aloud -│ └── hush.md # /hush — mute/unmute the narrator +│ ├── hush.md # /hush — mute/unmute the narrator +│ └── dictate.md # /dictate — voice input (mic → STT → prompt) ├── hooks/ │ ├── hooks.json # Stop hook wiring │ ├── speak_response.py # the narrator (TTS + playback) +│ ├── dictate_stt.py # dictation bridge (record → STT → stdout) │ └── tts-python.sh # python3 finder shim (Windows-safe) ├── skills/ │ └── voice/SKILL.md # voice narration guidance diff --git a/plugins/speech/commands/dictate.md b/plugins/speech/commands/dictate.md new file mode 100644 index 0000000000..cb5fd384f4 --- /dev/null +++ b/plugins/speech/commands/dictate.md @@ -0,0 +1,56 @@ +--- +description: Dictate a prompt by voice — record, transcribe with ElevenLabs STT, and send as your message +argument-hint: "[optional context for what to dictate]" +allowed-tools: Bash(python3:*), Bash(python:*), Bash(sox:*), Bash(arecord:*), Bash(ffmpeg:*), Bash(test:*), Bash(rm:*) +--- + +# /dictate — voice input (speech → prompt) + +**Argument:** $ARGUMENTS + +Records your voice from the default microphone, transcribes it with the +ElevenLabs speech-to-text API, and delivers the recognized words as your +prompt to SirGent. This is the input side of the speech plugin: `/speak` +is SirGent's voice, `/dictate` is yours. + +## Steps + +1. Check prerequisites in one command each: + - `test -n "$ELEVENLABS_API_KEY" && echo set || echo missing` — if + missing, tell the user to add it (Settings → Environment or shell + export) and stop. + - `command -v sox || command -v arecord || command -v ffmpeg` — if none + exist, tell the user to install `sox` (recommended, all platforms) and + stop. Do not attempt to record without a recorder. +2. Confirm to the user: "🎙️ Listening for up to 15 seconds — speak now…" + (length is `SPEECH_DICTATE_SECS`, default 15). +3. Run the dictation bridge exactly once: + + ```bash + python3 "${SIRGENT_PLUGIN_ROOT}/hooks/dictate_stt.py" + ``` + + - stdout = recognized words (plain text) + - stderr lines starting with `speech:` = the reason it failed + - exit 1 = no recording or transcription failed +4. On success: take the printed text as the user's message. If `$ARGUMENTS` + was provided, treat it as context and present the dictation as: + the dictated text, framed by the argument's intent (e.g. argument + "refactor plan" + dictation "split the parser module" → the user's + request is to refactor per the dictated detail). Then act on the request + as if the user had typed it. +5. If the bridge printed nothing but exited 0 (empty recording): tell the + user nothing was heard and suggest checking mic input levels. +6. On failure (exit 1): show the `speech:` stderr line verbatim and stop. + Common causes: missing API key, no recorder, no mic audio, HTTP quota. +7. Never read, print, or echo the API key. Never fabricate dictation + content — if transcription failed, say so. + +## Notes + +- Requires `ELEVENLABS_API_KEY` and a recorder: `sox`, `arecord` (ALSA), or + `ffmpeg` (macOS avfoundation). +- Tune `SPEECH_DICTATE_SECS` for longer dictations; `SPEECH_STT_MODEL` + overrides the STT model (default `scribe_v1`). +- Pair with `/voice on` for full two-way voice: SirGent speaks responses, + you dictate replies. diff --git a/plugins/speech/hooks/dictate_stt.py b/plugins/speech/hooks/dictate_stt.py new file mode 100644 index 0000000000..7a4c7034b2 --- /dev/null +++ b/plugins/speech/hooks/dictate_stt.py @@ -0,0 +1,156 @@ +#!/usr/bin/env python3 +""" +Speech plugin for SirGent AI — dictation bridge (speech → text). + +Records a short clip from the default microphone, transcribes it with the +ElevenLabs speech-to-text API, and prints the recognized words to stdout as +plain text. Designed to be called by the /dictate command so spoken words +flow into the prompt like typed text. + +Output contract (for the calling command): +- recognized words -> stdout, plain text, nothing else +- empty recording -> prints nothing (caller falls back to asking the user) +- any error -> a single line starting with "speech:" on stderr, + exit code 1 (caller surfaces it and stops gracefully) + +Configuration (environment variables): +- ELEVENLABS_API_KEY Required. https://elevenlabs.io +- SPEECH_STT_MODEL STT model id. Default: scribe_v1 +- SPEECH_DICTATE_SECS Max recording length in seconds. Default: 15 +- SPEECH_AUDIO_DIR Temp audio directory. Default: system temp +""" + +import json +import mimetypes +import os +import shutil +import subprocess +import sys +import tempfile +import uuid +from pathlib import Path + +STT_URL = "https://api.elevenlabs.io/v1/speech-to-text" +DEFAULT_STT_MODEL = "scribe_v1" +DICTATE_SECS_DEFAULT = 15 + + +def recorder_command(wav_path: Path, max_secs: int) -> list: + """First available recorder, writing 16 kHz mono WAV for STT.""" + for cmd in ( + ["sox", "-d", "-q", "-r", "16000", "-c", "1", "-b", "16", str(wav_path), + "trim", "0", str(max_secs)], + ["arecord", "-q", "-f", "S16_LE", "-r", "16000", "-c", "1", + "-d", str(max_secs), str(wav_path)], + ["ffmpeg", "-y", "-loglevel", "quiet", "-f", "avfoundation", "-i", ":0", + "-t", str(max_secs), "-ar", "16000", "-ac", "1", str(wav_path)], + ): + if shutil.which(cmd[0]): + return cmd + return [] + + +def has_recorder() -> bool: + return bool(shutil.which("sox") or shutil.which("arecord") or shutil.which("ffmpeg")) + + +def transcribe(wav_path: Path, api_key: str) -> str: + """POST a WAV recording to ElevenLabs speech-to-text; return the text.""" + import urllib.error + import urllib.request + + boundary = uuid.uuid4().hex + model = os.environ.get("SPEECH_STT_MODEL", DEFAULT_STT_MODEL) + mime = mimetypes.guess_type(str(wav_path))[0] or "audio/wav" + payload = wav_path.read_bytes() + + parts = [] + for name, value in (("model_id", model), ("diarize", "false")): + parts.append( + f'--{boundary}\r\nContent-Disposition: form-data; name="{name}"' + f"\r\n\r\n{value}\r\n".encode("utf-8") + ) + parts.append( + ( + f"--{boundary}\r\n" + f'Content-Disposition: form-data; name="file"; filename="{wav_path.name}"\r\n' + f"Content-Type: {mime}\r\n\r\n" + ).encode("utf-8") + + payload + + b"\r\n" + ) + parts.append(f"--{boundary}--\r\n".encode("utf-8")) + body = b"".join(parts) + + req = urllib.request.Request(STT_URL, data=body, method="POST", headers={ + "xi-api-key": api_key, + "Content-Type": f"multipart/form-data; boundary={boundary}", + "Accept": "application/json", + }) + try: + with urllib.request.urlopen(req, timeout=60) as resp: + data = json.loads(resp.read().decode("utf-8", "replace")) + return str(data.get("text", "")).strip() + except urllib.error.HTTPError as exc: + detail = exc.read().decode("utf-8", "replace")[:200] + print(f"speech: STT HTTP {exc.code}: {detail}", file=sys.stderr) + return "" + except (urllib.error.URLError, OSError, TimeoutError, ValueError) as exc: + print(f"speech: STT request failed: {exc}", file=sys.stderr) + return "" + + +def main() -> None: + api_key = os.environ.get("ELEVENLABS_API_KEY", "").strip() + if not api_key: + print("speech: set ELEVENLABS_API_KEY to use dictation (https://elevenlabs.io)", + file=sys.stderr) + sys.exit(1) + + if not has_recorder(): + print("speech: no mic recorder found. Install sox (recommended), arecord, " + "or ffmpeg, then retry.", file=sys.stderr) + sys.exit(1) + + max_secs = int(os.environ.get("SPEECH_DICTATE_SECS", DICTATE_SECS_DEFAULT)) + audio_dir = Path(os.environ.get("SPEECH_AUDIO_DIR", tempfile.gettempdir())) / "sirgent-speech" + try: + audio_dir.mkdir(parents=True, exist_ok=True) + except OSError as exc: + print(f"speech: {exc}", file=sys.stderr) + sys.exit(1) + + wav_path = audio_dir / "dictate.wav" + try: + wav_path.unlink(missing_ok=True) + except OSError: + pass + + cmd = recorder_command(wav_path, max_secs) + try: + proc = subprocess.run(cmd, stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, timeout=max_secs + 10) + except (OSError, subprocess.TimeoutExpired): + print("speech: recording failed or timed out.", file=sys.stderr) + sys.exit(1) + + if not wav_path.exists() or wav_path.stat().st_size == 0: + print("speech: recording produced no audio. Check your microphone.", file=sys.stderr) + sys.exit(1) + + text = transcribe(wav_path, api_key) + try: + wav_path.unlink(missing_ok=True) + except OSError: + pass + + if not text: + # transcribe() already reported the reason on stderr + sys.exit(1) + + print(text) + sys.exit(0) + + +if __name__ == "__main__": + main() diff --git a/plugins/speech/hooks/hooks.json b/plugins/speech/hooks/hooks.json index fb192aadf0..91aed3215b 100644 --- a/plugins/speech/hooks/hooks.json +++ b/plugins/speech/hooks/hooks.json @@ -1,5 +1,5 @@ { - "description": "Speech plugin — narrates SirGent's final response aloud via ElevenLabs TTS", + "description": "Speech plugin — narrates SirGent's final response aloud via ElevenLabs TTS; /dictate covers the input side (mic → STT → prompt)", "hooks": { "Stop": [ {