Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion gateway/platforms/base.py
Original file line number Diff line number Diff line change
Expand Up @@ -1492,7 +1492,7 @@ def _log_safe_path(path: str) -> str:
r'''[`"']?MEDIA:\s*'''
r'''(?P<path>`[^`\n]+`|"[^"\n]+"|'[^'\n]+'|'''
r'''(?:~/|/|[A-Za-z]:[/\\])\S+(?:[^\S\n]+\S+)*?\.(?:''' + _MEDIA_EXT_ALTERNATION + r'''))'''
r'''(?=[\s`"',;:)\]}]|$)[`"']?''',
r'''(?=[\s`"',;:)\]}\[]|$)[`"']?''',

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Please restrict this new boundary to the literal supported [[as_document]] directive rather than any [. A generic bracket boundary can make MEDIA:/safe/report.xlsx[revision] match and deliver the .xlsx prefix, whereas the current anchored parser leaves that unrecognized path intact.

re.IGNORECASE,
)

Expand Down
61 changes: 61 additions & 0 deletions tests/gateway/test_media_tag_cleanup.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,61 @@
"""Tests for MEDIA_TAG_CLEANUP_RE regex matching behavior (#63632)."""


class TestMediaTagCleanup:
"""Tests for MEDIA_TAG_CLEANUP_RE regex matching behavior."""

def test_media_tag_with_directive_glued_to_extension(self):
"""Regression: MEDIA:<path>[[as_document]] must match when directive is glued
directly to the extension without whitespace (#63632).

The fix adds `\\[` to the lookahead character class in MEDIA_TAG_CLEANUP_RE.
"""
from gateway.platforms.base import MEDIA_TAG_CLEANUP_RE

# Issue case: [[as_document]] glued directly to .xlsx
text = "Готово. MEDIA:/home/hermes/report.xlsx[[as_document]]"
assert MEDIA_TAG_CLEANUP_RE.search(text) is not None
stripped = MEDIA_TAG_CLEANUP_RE.sub("", text)
assert "MEDIA:" not in stripped
assert "/home/hermes/report.xlsx" not in stripped

# Same with whitespace (should still work)
text_with_space = "Готово. MEDIA:/home/hermes/report.xlsx [[as_document]]"
assert MEDIA_TAG_CLEANUP_RE.search(text_with_space) is not None
stripped = MEDIA_TAG_CLEANUP_RE.sub("", text_with_space)
assert "MEDIA:" not in stripped
assert "/home/hermes/report.xlsx" not in stripped

# Other directives ([[as_image]]) should also work

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

[[as_image]] is not a supported directive on current main: cleanup only strips [[audio_as_voice]] and [[as_document]]. Please remove this case or implement that directive separately; add an extract_media() regression for the supported glued [[as_document]] form instead.

text_image = "Done. MEDIA:/tmp/chart.png[[as_image]]"
assert MEDIA_TAG_CLEANUP_RE.search(text_image) is not None
stripped = MEDIA_TAG_CLEANUP_RE.sub("", text_image)
assert "MEDIA:" not in stripped
assert "/tmp/chart.png" not in stripped

def test_media_tag_with_whitespace_still_works(self):
"""Baseline: MEDIA tags with whitespace before/after still match."""
from gateway.platforms.base import MEDIA_TAG_CLEANUP_RE

# Space before closing quote
text = "Here is your report: MEDIA:/tmp/report.md "
stripped = MEDIA_TAG_CLEANUP_RE.sub("", text).strip()
assert "MEDIA:" not in stripped
assert "/tmp/report.md" not in stripped

# Multiple spaces (regex removes tag but preserves surrounding whitespace)
text = "Report at MEDIA:/tmp/data.pdf done"
stripped = MEDIA_TAG_CLEANUP_RE.sub("", text)
assert "MEDIA:" not in stripped
assert "/tmp/data.pdf" not in stripped
assert "Report at" in stripped and "done" in stripped

def test_media_tag_at_end_of_string(self):
"""MEDIA tags at the end of a string should match ($ anchor)."""
from gateway.platforms.base import MEDIA_TAG_CLEANUP_RE

text = "Here is the file: MEDIA:/tmp/file.docx"
stripped = MEDIA_TAG_CLEANUP_RE.sub("", text).strip()
assert "MEDIA:" not in stripped
assert "/tmp/file.docx" not in stripped
assert "Here is the file:" in stripped
Loading