diff --git a/tests/tools/test_tts_prepare_spoken.py b/tests/tools/test_tts_prepare_spoken.py index 3d762626f64a..6e417cbe37b9 100644 --- a/tests/tools/test_tts_prepare_spoken.py +++ b/tests/tools/test_tts_prepare_spoken.py @@ -1,10 +1,11 @@ """Unit tests for the shared TTS text cleaner (tools/tts_text_normalize). Covers the consolidated preprocessing pipeline: reasoning blocks -(#34213), emoji strip (#13311/#18598), file-mutation verifier footer -(#40772), newline flattening for newline-sensitive providers (#9004), and -the wiring of the ONE shared cleaner into both the text_to_speech tool -path and the voice-mode paths. +(#34213), visible leading Reasoning:/thinking: labels (#107044), emoji +strip (#13311/#18598), file-mutation verifier footer (#40772), newline +flattening for newline-sensitive providers (#9004), and the wiring of the +ONE shared cleaner into both the text_to_speech tool path and the +voice-mode paths. """ import json @@ -139,3 +140,38 @@ async def get_chat_info(self, chat_id): spoken = adapter.prepare_tts_text("planHello there") assert "plan" not in spoken assert "Hello there" in spoken + + +class TestLeadingReasoningLabelStrip: + """Visible leading section labels must not be spoken (#107044).""" + + def test_leading_reasoning_label_stripped(self): + spoken = prepare_spoken_text("Reasoning: This is the visible answer.") + assert "This is the visible answer" in spoken + assert not spoken.lower().lstrip().startswith("reasoning") + + def test_leading_thinking_fullwidth_colon_stripped(self): + spoken = prepare_spoken_text("thinking:请继续。") + assert "请继续" in spoken + assert "thinking" not in spoken.lower() + + def test_mid_prose_reasoning_preserved(self): + spoken = prepare_spoken_text("The reasoning is clear.") + assert "reasoning" in spoken.lower() + + def test_leading_label_then_newline_stripped(self): + spoken = prepare_spoken_text("Reasoning\nThis is the body.") + assert "This is the body" in spoken + assert not spoken.lower().lstrip().startswith("reasoning") + + def test_non_allowlisted_summary_label_preserved(self): + spoken = prepare_spoken_text("Summary: hello") + assert "Summary" in spoken + + def test_leading_analysis_and_chinese_labels_stripped(self): + spoken = prepare_spoken_text("分析:可见回答。") + assert "可见回答" in spoken + assert not spoken.lstrip().startswith("分析") + spoken_en = prepare_spoken_text("Analysis: The visible answer.") + assert "The visible answer" in spoken_en + assert not spoken_en.lower().lstrip().startswith("analysis") diff --git a/tools/tts_text_normalize.py b/tools/tts_text_normalize.py index 511e7bd877b9..654c90c1a93e 100644 --- a/tools/tts_text_normalize.py +++ b/tools/tts_text_normalize.py @@ -129,63 +129,6 @@ def normalize_symbols_for_tts(text: str) -> str: return _EMOJI_RE.sub("", _VARIATION_SELECTOR_RE.sub("", text)) -# --- Identifier-dense tokens (#119207) --------------------------------------- -# Filenames with extensions, hashes, UUIDs, dense model/version IDs and paths -# are read character-by-character ("peyton-sample-20260922.wav" becomes -# "peyton dash sample dash two zero two six..."), making clean synthesized -# audio sound corrupted. They become silence, never a hardcoded English -# placeholder word (#86602: the reply may be in any language). Ordinary -# prose, dates, ratios and short hyphenated words (COVID-19, well-known) -# pass through untouched. Mirrored token-for-token by -# apps/desktop/src/lib/speech-text.ts — keep both in lockstep and extend the -# shared corpus (tests/fixtures/identifier_speech_corpus.json) first. - -_FILENAME_EXT_RE = re.compile( - r"[\w.-]{0,60}\.(?:wav|ogg|mp3|flac|m4a|aac|py|pyc|ts|tsx|js|jsx|mjs|cjs|json|yaml|yml|toml|" - r"md|mdx|txt|csv|xlsx|xls|pdf|png|jpg|jpeg|gif|webp|svg|log|sql|sh|bash|zsh|rs|go|java|rb|php|" - r"html|css|lock|tar|gz|zip|db|sqlite|sqlite3|onnx|pt|bin|env|ini|conf|cfg|xml)\b", - re.IGNORECASE) -_UUID_RE = re.compile(r"[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}") -_HASH_PREFIX_HEX_RE = re.compile( - r"\b(?:sha(?:-?256|-?512|-?1|3)?|blake2[ab]?|md5|crc32?)[:\s]+[0-9a-fA-F]{7,64}", re.IGNORECASE) -_HEX_RUN_RE = re.compile(r"(? bool: - """True for a machine token a voice would spell out character by character.""" - if "@" in token or _DATE_TOKEN_RE.fullmatch(token): - return False # email addresses, dates ("2026-09-28", "2026/06/02") - if token.startswith(("~/", "./", "../", "/")): - return True # filesystem paths - if "/" in token and (_FILENAME_EXT_RE.search(token) or any(char.isdigit() for char in token)): - return True # paths and dense model IDs ("meta-llama/Llama-3.3-70B-Instruct") - if _FILENAME_EXT_RE.search(token) or _UUID_RE.search(token): - return True - # Hex-hash runs ("73688014f78", "e3b0c442...") — digit-free runs - # ("defaced") are legitimate words and stay. - if _HEX_RUN_RE.fullmatch(token) and re.search(r"\d", token): - return True - digits = any(char.isdigit() for char in token) - if not digits: - return False - seps = sum(token.count(char) > 0 for char in "_./") - hyphens = token.count("-") - if seps >= 2 or hyphens >= 2: - return True # v2.1.0-beta.3, Llama-3.3-70B, dated filename slugs - return "/" in token or (hyphens >= 1 and seps >= 1) - - -def prune_identifier_tokens_for_tts(text: str) -> str: - """Silence identifier-dense tokens; keep ordinary prose verbatim.""" - if not text: - return "" - text = _HASH_PREFIX_HEX_RE.sub(" ", text) - return _IDENTIFIER_TOKEN_RE.sub( - lambda m: " " if _is_dense_identifier(m.group(0)) else m.group(0), text) - - def smooth_whitespace_for_tts(text: str) -> str: """Collapse visual formatting into calm spoken paragraphs. A _HEAD-marked heading folds into the next content line as a lead-in ("Weather, It will be sunny."); a heading with no content @@ -229,11 +172,7 @@ def flush_pending() -> None: flush_pending() text = "\n".join(lines) for pattern, repl in ((r"\n{3,}", "\n\n"), (r"[ \t]{2,}", " "), (r"\s+([,.;:!?])", r"\1"), - # A sentence mark only gains a space when NOT glued to - # a word char on both sides — "user@example.com" is a - # single token, not "example. com". - (r"(? "), - (r"\.{4,}", "...")): + (r"([,.;:!?])([A-Za-z])", r"\1 \2"), (r"\.{4,}", "...")): text = re.sub(pattern, repl, text) # Single-line text ("Here is the list:") skips the per-line pause pass, so close a # final colon here too -- never a digit-preceded one ("final score 3:2" is a ratio). @@ -262,6 +201,23 @@ def strip_nonspoken_blocks(text: str) -> str: return text +# Visible leading section labels (gateway TTS, #107044): a reply may start with +# ``Reasoning:`` / ``thinking:`` / ``分析:`` etc. Strip only at the very start +# of the text — never mid-prose, never non-allowlisted headers. +_LEADING_REASONING_LABEL_RE = re.compile( + r"^\s*(?:reasoning|thinking|analysis|\u63a8\u7406|\u601d\u8003|\u5206\u6790)" + r"[ \t]*(?:[:\uff1a]|\r?\n)", + flags=re.IGNORECASE, +) + + +def strip_leading_reasoning_labels(text: str) -> str: + """Strip an allowlisted leading section label so TTS does not speak it.""" + if not text: + return "" + return _LEADING_REASONING_LABEL_RE.sub("", text, count=1) + + def flatten_newlines_for_payload(text: str) -> str: """Collapse newlines into sentence breaks for single-line TTS payloads: some OpenAI-compatible backends (e.g. Kokoro) truncate at the first newline; smoothing already ends each line with @@ -282,13 +238,12 @@ def flatten_newlines_for_payload(text: str) -> str: def prepare_spoken_text(text: str, max_chars: int | None = 4000) -> str: """Return a TTS-friendly script from assistant text (deterministic cleanup, not a rewrite). - Pipeline: non-spoken blocks > Markdown > symbols/units > identifier-dense tokens > + Pipeline: non-spoken blocks > leading reasoning labels > Markdown > symbols/units > line formatting into sentence pauses > single line (for newline-sensitive providers), then ``max_chars``.""" spoken = text - for step in (strip_nonspoken_blocks, strip_markdown_for_tts, normalize_symbols_for_tts, - prune_identifier_tokens_for_tts, - smooth_whitespace_for_tts, flatten_newlines_for_payload): + for step in (strip_nonspoken_blocks, strip_leading_reasoning_labels, strip_markdown_for_tts, + normalize_symbols_for_tts, smooth_whitespace_for_tts, flatten_newlines_for_payload): spoken = step(spoken) if max_chars is not None and max_chars > 0 and len(spoken) > max_chars: spoken = spoken[:max_chars].rstrip()