From 3d24260ba2cf26d9eb6734ac8f88fa22ab1341b4 Mon Sep 17 00:00:00 2001 From: aha-lin Date: Sun, 17 May 2026 10:57:54 +0800 Subject: [PATCH 1/2] feat(weixin): prefix transcribed voice messages with [voice:transcribed] Weixin delivers auto-transcribed voice messages as plain text via the callback, indistinguishable from typed input. Prefixing the transcript lets the agent reliably detect voice messages and apply voice-specific reply conventions (e.g. echo the transcript so the user can verify ASR accuracy). --- gateway/platforms/weixin.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/gateway/platforms/weixin.py b/gateway/platforms/weixin.py index 1c9fec0af7fb2..0cf390cc001d7 100644 --- a/gateway/platforms/weixin.py +++ b/gateway/platforms/weixin.py @@ -1001,7 +1001,10 @@ def _extract_text(item_list: List[Dict[str, Any]]) -> str: if item.get("type") == ITEM_VOICE: voice_text = str((item.get("voice_item") or {}).get("text") or "") if voice_text: - return voice_text + # Prefix so the agent can distinguish auto-transcribed voice + # messages from typed text (Weixin's callback delivers them as + # indistinguishable plain strings otherwise). + return f"[voice:transcribed] {voice_text}" return "" From 6eb6c47e63cdfd914f0db96ba9e4e1ab0513d540 Mon Sep 17 00:00:00 2001 From: aha-lin Date: Sun, 17 May 2026 10:57:55 +0800 Subject: [PATCH 2/2] feat(weixin): mechanical voice-echo enforcement on outbound send MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit When the inbound message was an auto-transcribed voice (prefixed with [voice:transcribed]), guarantee the outbound reply starts with the `> 🎤 你说:"…"` echo so the user can verify ASR accuracy. The LLM should add this prefix per the wechat-voice-gateway skill, but enforcing it mechanically removes a recurring drift mode. One-shot: stash on inbound, pop on outbound. Toggleable via WEIXIN_VOICE_ECHO_ENFORCE env or voice_echo_enforce config (default on). --- gateway/platforms/weixin.py | 32 ++++++++++++++++++++++++++++++++ 1 file changed, 32 insertions(+) diff --git a/gateway/platforms/weixin.py b/gateway/platforms/weixin.py index 0cf390cc001d7..2a3b56fc893fc 100644 --- a/gateway/platforms/weixin.py +++ b/gateway/platforms/weixin.py @@ -1196,6 +1196,14 @@ def __init__(self, config: PlatformConfig): self._send_session: Optional[aiohttp.ClientSession] = None self._poll_task: Optional[asyncio.Task] = None self._dedup = MessageDeduplicator(ttl_seconds=MESSAGE_DEDUP_TTL_SECONDS) + # Voice echo enforcement: cache last voice transcript per chat_id so the + # outbound send() can guarantee the `> 🎤 你说:"…"` echo prefix even when + # the LLM forgets to add it. One-shot: consumed on next outbound send. + self._pending_voice_echo: Dict[str, str] = {} + _voice_echo_raw = extra.get("voice_echo_enforce") + if _voice_echo_raw is None: + _voice_echo_raw = os.getenv("WEIXIN_VOICE_ECHO_ENFORCE") + self._voice_echo_enforce: bool = _coerce_bool(_voice_echo_raw, default=True) self._account_id = str(extra.get("account_id") or os.getenv("WEIXIN_ACCOUNT_ID", "")).strip() self._token = str(config.token or extra.get("token") or os.getenv("WEIXIN_TOKEN", "")).strip() @@ -1399,6 +1407,12 @@ async def _process_message(self, message: Dict[str, Any]) -> None: if self._dedup.is_duplicate(content_key): logger.debug("[%s] Content-dedup: skipping duplicate message from %s", self.name, sender_id) return + # Voice echo enforcement: stash transcript keyed on the eventual + # outbound chat_id so send() can guarantee the echo prefix. + if self._voice_echo_enforce and text.startswith("[voice:transcribed] "): + transcript = text[len("[voice:transcribed] "):].strip() + if transcript: + self._pending_voice_echo[sender_id] = transcript chat_type, effective_chat_id = _guess_chat_type(message, self._account_id) if chat_type == "group": @@ -1677,6 +1691,24 @@ async def send( ) -> SendResult: if not self._send_session or not self._token: return SendResult(success=False, error="Not connected") + # Voice echo enforcement (mechanical guarantee, not LLM-dependent): + # if the inbound message for this chat was a voice transcript and the + # outbound reply does NOT already start with the `> 🎤 你说:"…"` echo, + # auto-prepend it. Consumed (popped) so it never bleeds into a later + # turn. Skips media-only sends (empty content after MEDIA: extraction + # would re-trigger on the same transcript next turn — popping here + # is one-shot regardless). + if self._voice_echo_enforce and content and content.strip(): + transcript = self._pending_voice_echo.pop(chat_id, None) + if transcript and not content.lstrip().startswith("> 🎤 你说"): + # Escape any embedded double quotes in the transcript so the + # blockquote parses cleanly. + safe = transcript.replace('"', '\\"') + content = f'> 🎤 你说:"{safe}"\n\n{content}' + logger.info( + "[%s] voice-echo enforcer prepended prefix for %s (LLM forgot)", + self.name, _safe_id(chat_id), + ) context_token = self._token_store.get(self._account_id, chat_id) last_message_id: Optional[str] = None