Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 6 additions & 0 deletions gateway/run.py
Original file line number Diff line number Diff line change
Expand Up @@ -2824,6 +2824,7 @@ async def _prepare_inbound_message_text(
image_paths,
)

self._pending_voice_transcript = None
if audio_paths:
message_text = await self._enrich_message_with_transcription(
message_text,
Expand Down Expand Up @@ -3628,6 +3629,10 @@ async def _handle_message_with_agent(self, event, source, _quick_key: str):
)
return None

if response and getattr(self, "_pending_voice_transcript", None):
response = f"πŸŽ™ γ€Œ{self._pending_voice_transcript}」

{response}"
return response

except Exception as e:
Expand Down Expand Up @@ -6751,6 +6756,7 @@ async def _enrich_message_with_transcription(
result = await asyncio.to_thread(transcribe_audio, path)
if result["success"]:
transcript = result["transcript"]
self._pending_voice_transcript = transcript
enriched_parts.append(
f'[The user sent a voice message~ '
f'Here\'s what they said: "{transcript}"]'
Expand Down
35 changes: 35 additions & 0 deletions tools/transcription_tools.py
Original file line number Diff line number Diff line change
Expand Up @@ -88,6 +88,34 @@ def _safe_find_spec(module_name: str) -> bool:
_local_model: Optional[object] = None
_local_model_name: Optional[str] = None

# ---------------------------------------------------------------------------
# Hotwords / initial_prompt helpers
# ---------------------------------------------------------------------------

_HOTWORDS_PATH = Path.home() / ".hermes" / "stt_hotwords.txt"


def _load_initial_prompt() -> Optional[str]:
"""Load hotwords from ~/.hermes/stt_hotwords.txt and return as a comma-separated
string suitable for Whisper's ``initial_prompt`` / ``prompt`` parameter.

Lines starting with ``#`` and blank lines are ignored.
Returns ``None`` if the file is missing or empty.
"""
try:
if not _HOTWORDS_PATH.exists():
return None
words = [
line.strip()
for line in _HOTWORDS_PATH.read_text(encoding="utf-8").splitlines()
if line.strip() and not line.strip().startswith("#")
]
if not words:
return None
return ", ".join(words)
except Exception:
return None

# ---------------------------------------------------------------------------
# Config helpers
# ---------------------------------------------------------------------------
Expand Down Expand Up @@ -334,6 +362,9 @@ def _transcribe_local(file_path: str, model_name: str) -> Dict[str, Any]:
transcribe_kwargs = {"beam_size": 5}
if _forced_lang:
transcribe_kwargs["language"] = _forced_lang
_initial_prompt = _load_initial_prompt()
if _initial_prompt:
transcribe_kwargs["initial_prompt"] = _initial_prompt

segments, info = _local_model.transcribe(file_path, **transcribe_kwargs)
transcript = " ".join(segment.text.strip() for segment in segments)
Expand Down Expand Up @@ -460,11 +491,13 @@ def _transcribe_groq(file_path: str, model_name: str) -> Dict[str, Any]:
from openai import OpenAI, APIError, APIConnectionError, APITimeoutError
client = OpenAI(api_key=api_key, base_url=GROQ_BASE_URL, timeout=30, max_retries=0)
try:
_prompt = _load_initial_prompt()
with open(file_path, "rb") as audio_file:
transcription = client.audio.transcriptions.create(
model=model_name,
file=audio_file,
response_format="text",
**({"prompt": _prompt} if _prompt else {}),
)

transcript_text = str(transcription).strip()
Expand Down Expand Up @@ -517,11 +550,13 @@ def _transcribe_openai(file_path: str, model_name: str) -> Dict[str, Any]:
from openai import OpenAI, APIError, APIConnectionError, APITimeoutError
client = OpenAI(api_key=api_key, base_url=base_url, timeout=30, max_retries=0)
try:
_prompt = _load_initial_prompt()
with open(file_path, "rb") as audio_file:
transcription = client.audio.transcriptions.create(
model=model_name,
file=audio_file,
response_format="text" if model_name == "whisper-1" else "json",
**({"prompt": _prompt} if _prompt else {}),
)

transcript_text = _extract_transcript_text(transcription)
Expand Down