diff --git a/plugins/platforms/email/adapter.py b/plugins/platforms/email/adapter.py index 8698dc93803ef..133fb8f340e16 100644 --- a/plugins/platforms/email/adapter.py +++ b/plugins/platforms/email/adapter.py @@ -313,9 +313,10 @@ def _strip_html(html: str) -> str: text = re.sub(r"
", "\n", text, flags=re.IGNORECASE) text = re.sub(r"<[^>]+>", "", text) text = re.sub(r" ", " ", text) - text = re.sub(r"&", "&", text) text = re.sub(r"<", "<", text) text = re.sub(r">", ">", text) + # Decode ampersands last so < remains literal <, not markup-like text. + text = re.sub(r"&", "&", text) text = re.sub(r"\n{3,}", "\n\n", text) return text.strip() diff --git a/tests/gateway/test_email.py b/tests/gateway/test_email.py index 8e46600b04344..bf5b49486f5e9 100644 --- a/tests/gateway/test_email.py +++ b/tests/gateway/test_email.py @@ -85,6 +85,14 @@ def test_strip_html_basic(self): self.assertNotIn("", result) self.assertNotIn("", result) + def test_strip_html_does_not_double_decode_entities(self): + from plugins.platforms.email.adapter import _strip_html + + self.assertEqual( + _strip_html("The token is <APIKEY>"), + "The token is <APIKEY>", + ) + class TestExtractTextBody(unittest.TestCase): """Test email body extraction from different message formats."""