diff --git a/tests/test_noema_document_review_context.py b/tests/test_noema_document_review_context.py
index e6ec2e6d70..063c95d08d 100644
--- a/tests/test_noema_document_review_context.py
+++ b/tests/test_noema_document_review_context.py
@@ -6,6 +6,9 @@
import io
import json
import os
+import re
+import runpy
+import sys
import zipfile
from pathlib import Path
@@ -48,6 +51,7 @@ def _docx_entity_bytes() -> bytes:
def _pr() -> dict[str, object]:
+ """Build a minimal PR payload shared by review-context tests."""
return {
"headRefOid": "head",
"baseRefOid": "base",
@@ -105,6 +109,7 @@ def test_docx_text_reaches_the_actual_reviewer_payload(monkeypatch):
encoded = base64.b64encode(raw).decode("ascii")
def fake_run(args, stdin=None):
+ """Return base64-encoded DOCX bytes for the requested content ref."""
assert "contents/docs/review.docx?ref=head" in args[2]
return encoded
@@ -123,20 +128,26 @@ def fake_run(args, stdin=None):
captured: dict[str, object] = {}
class Response:
+ """Fake urllib response object supporting the context-manager protocol."""
def __enter__(self):
+ """Enter the fake response context manager."""
return self
def __exit__(self, *args):
+ """Exit the fake response context manager."""
return False
def read(self):
+ """Return a fake chat-completion payload as JSON bytes."""
verdict = {"decision": "comment", "summary": "checked", "findings": []}
return json.dumps(
{"choices": [{"message": {"content": json.dumps(verdict)}}]}
).encode()
class Opener:
+ """Fake urllib opener capturing the outgoing request payload."""
def open(self, request):
+ """Capture the request body and return a fake response."""
captured.update(json.loads(request.data.decode()))
return Response()
@@ -200,6 +211,7 @@ def test_hwp_reader_contract_is_local_and_fail_closed(monkeypatch):
raise AssertionError("expected failed local HWP reader to fail closed")
def timed_out(*args, **kwargs):
+ """Simulate a hung reviewed HWP reader subprocess."""
raise document.subprocess.TimeoutExpired(args[0], kwargs["timeout"])
monkeypatch.setattr(document.subprocess, "run", timed_out)
@@ -243,20 +255,26 @@ def test_real_hwp_mcp_fixture_text_reaches_reviewer_payload(
captured: dict[str, object] = {}
class Response:
+ """Fake urllib response object supporting the context-manager protocol."""
def __enter__(self):
+ """Enter the fake response context manager."""
return self
def __exit__(self, *args):
+ """Exit the fake response context manager."""
return False
def read(self):
+ """Return a fake chat-completion payload as JSON bytes."""
verdict = {"decision": "comment", "summary": "checked", "findings": []}
return json.dumps(
{"choices": [{"message": {"content": json.dumps(verdict)}}]}
).encode()
class Opener:
+ """Fake urllib opener capturing the outgoing request payload."""
def open(self, request):
+ """Capture the request body and return a fake response."""
captured.update(json.loads(request.data.decode()))
return Response()
@@ -275,3 +293,228 @@ def open(self, request):
assert expected_text in prompt
if fixture_name == "simple.hwp":
assert "| 이름 | 회사 |" in prompt
+
+
+def test_oversize_raw_bytes_are_rejected_before_any_parsing():
+ """Raw input above the bounded 8 MiB cap fails closed without inspecting the suffix."""
+ raw = b"x" * (document.MAX_DOCUMENT_BYTES + 1)
+ with pytest.raises(
+ document.DocumentReadError,
+ match="document exceeds the bounded 8 MiB review input",
+ ):
+ document.extract_review_document("docs/huge.docx", raw)
+
+
+def test_unsupported_suffix_is_rejected():
+ """A file extension outside the supported set fails closed with the suffix named."""
+ with pytest.raises(
+ document.DocumentReadError,
+ match=r"unsupported review document format: \.txt",
+ ):
+ document.extract_review_document("docs/notes.txt", b"plain text")
+
+
+def _zip_bytes(entries: dict[str, bytes]) -> bytes:
+ """Build an in-memory ZIP archive from an arcname -> content mapping."""
+ output = io.BytesIO()
+ with zipfile.ZipFile(output, "w", zipfile.ZIP_DEFLATED) as archive:
+ for name, content in entries.items():
+ archive.writestr(name, content)
+ return output.getvalue()
+
+
+def test_docx_zip_with_too_many_entries_is_rejected():
+ """A DOCX archive with more than the bounded entry count fails closed."""
+ entries = {
+ f"part-{index}.xml": b""
+ for index in range(document.MAX_DOCUMENT_ZIP_ENTRIES + 1)
+ }
+ with pytest.raises(
+ document.DocumentReadError, match="DOCX archive has too many entries"
+ ):
+ document.extract_review_document("docs/many-entries.docx", _zip_bytes(entries))
+
+
+def test_docx_zip_over_the_bounded_unpacked_size_is_rejected(monkeypatch):
+ """A DOCX archive whose declared unpacked size exceeds the cap fails closed."""
+ monkeypatch.setattr(document, "MAX_DOCUMENT_ZIP_UNCOMPRESSED_BYTES", 5)
+ raw = _zip_bytes({"word/document.xml": b"0123456789"})
+ with pytest.raises(
+ document.DocumentReadError,
+ match="DOCX archive exceeds the bounded unpacked size",
+ ):
+ document.extract_review_document("docs/oversized-unpacked.docx", raw)
+
+
+def test_docx_missing_document_xml_is_rejected():
+ """A DOCX archive without word/document.xml fails closed with a clear message."""
+ raw = _zip_bytes({"word/other.xml": b""})
+ with pytest.raises(
+ document.DocumentReadError, match=re.escape("DOCX archive has no word/document.xml")
+ ):
+ document.extract_review_document("docs/no-document-xml.docx", raw)
+
+
+def test_docx_document_xml_without_body_is_rejected():
+ """A document.xml with no w:body element fails closed."""
+ xml = """
+
+"""
+ raw = _zip_bytes({"word/document.xml": xml.encode("utf-8")})
+ with pytest.raises(
+ document.DocumentReadError, match=re.escape("DOCX document.xml has no document body")
+ ):
+ document.extract_review_document("docs/no-body.docx", raw)
+
+
+def test_docx_with_only_empty_paragraphs_and_rowless_table_has_no_readable_text():
+ """Empty paragraphs and a table with no populated rows leave no readable text."""
+ xml = """
+
+
+
+
+
+"""
+ raw = _zip_bytes({"word/document.xml": xml.encode("utf-8")})
+ with pytest.raises(
+ document.DocumentReadError, match="DOCX contains no readable text"
+ ):
+ document.extract_review_document("docs/empty.docx", raw)
+
+
+def test_docx_paragraph_tabs_breaks_and_table_pipe_escaping():
+ """Tabs, line breaks, and a second table row exercise the paragraph/table branches."""
+ xml = """
+
+
+ Section-MarkerAfter-TabAfter-Break
+
+ a|bc
+
+
+
+
+"""
+ raw = _zip_bytes({"word/document.xml": xml.encode("utf-8")})
+ text = document.extract_review_document("docs/tabs-breaks.docx", raw)
+ assert "Section-Marker\tAfter-Tab\nAfter-Break" in text
+ assert "a\\|b" in text
+
+
+def test_hwp_reader_env_unset_fails_closed(monkeypatch):
+ """HWP/HWPX extraction fails closed when the reviewed reader source is unset."""
+ monkeypatch.delenv(document.HWP_READER_ENV, raising=False)
+ with pytest.raises(
+ document.DocumentReadError,
+ match="reviewed hwp-mcp/rhwp reader is not configured",
+ ):
+ document.extract_review_document("docs/unconfigured.hwp", b"binary")
+
+
+def test_hwp_reader_subprocess_os_error_fails_closed(monkeypatch):
+ """An OSError starting the reviewed reader subprocess fails closed."""
+ monkeypatch.setenv(document.HWP_READER_ENV, "/trusted/hwp-mcp-source")
+
+ def raise_os_error(*_args, **_kwargs):
+ """Simulate a reviewed HWP reader subprocess that cannot start."""
+ raise OSError("node executable not found")
+
+ monkeypatch.setattr(document.subprocess, "run", raise_os_error)
+ with pytest.raises(
+ document.DocumentReadError,
+ match="reviewed hwp-mcp/rhwp reader could not start",
+ ):
+ document.extract_review_document("docs/no-node.hwp", b"binary")
+
+
+def test_hwp_reader_output_over_bounded_size_fails_closed(monkeypatch):
+ """Reader stdout larger than the bounded text cap fails closed."""
+ monkeypatch.setenv(document.HWP_READER_ENV, "/trusted/hwp-mcp-source")
+ completed = document.subprocess.CompletedProcess(
+ ["node"], 0, stdout=b"a" * (document.MAX_DOCUMENT_TEXT_BYTES + 1), stderr=b""
+ )
+ monkeypatch.setattr(document.subprocess, "run", lambda *args, **kwargs: completed)
+ with pytest.raises(
+ document.DocumentReadError,
+ match="reviewed hwp-mcp/rhwp reader exceeded the bounded output",
+ ):
+ document.extract_review_document("docs/too-long.hwp", b"binary")
+
+
+def test_hwp_reader_non_utf8_output_fails_closed(monkeypatch):
+ """Non-UTF-8 reader stdout fails closed instead of producing replacement text."""
+ monkeypatch.setenv(document.HWP_READER_ENV, "/trusted/hwp-mcp-source")
+ completed = document.subprocess.CompletedProcess(
+ ["node"], 0, stdout=b"\xff\xfe\xfa", stderr=b""
+ )
+ monkeypatch.setattr(document.subprocess, "run", lambda *args, **kwargs: completed)
+ with pytest.raises(
+ document.DocumentReadError,
+ match="reviewed hwp-mcp/rhwp reader returned non-UTF-8 text",
+ ):
+ document.extract_review_document("docs/bad-encoding.hwp", b"binary")
+
+
+def test_hwp_reader_empty_output_fails_closed(monkeypatch):
+ """Whitespace-only reader stdout fails closed as empty text."""
+ monkeypatch.setenv(document.HWP_READER_ENV, "/trusted/hwp-mcp-source")
+ completed = document.subprocess.CompletedProcess(
+ ["node"], 0, stdout=b" \n\t ", stderr=b""
+ )
+ monkeypatch.setattr(document.subprocess, "run", lambda *args, **kwargs: completed)
+ with pytest.raises(
+ document.DocumentReadError,
+ match="reviewed hwp-mcp/rhwp reader returned empty text",
+ ):
+ document.extract_review_document("docs/blank.hwp", b"binary")
+
+
+def test_bounded_text_truncates_multibyte_text_cleanly():
+ """Truncation of multibyte text stays valid UTF-8 and reports omitted bytes."""
+ text = "가" * ((document.MAX_DOCUMENT_TEXT_BYTES // 3) + 10)
+ bounded = document._bounded_text(text)
+ assert bounded != text
+ assert "[document text truncated;" in bounded
+ assert "bytes omitted]" in bounded
+ prefix = bounded.split("\n[document text truncated;", 1)[0]
+ assert len(prefix.encode("utf-8")) <= document.MAX_DOCUMENT_TEXT_BYTES
+
+
+def test_main_entrypoint_prints_text_for_a_valid_docx(tmp_path, monkeypatch, capsys):
+ """The __main__ CLI path prints extracted text and exits 0 for a valid DOCX."""
+ docx_path = tmp_path / "good.docx"
+ docx_path.write_bytes(_docx_bytes())
+ monkeypatch.setattr(sys, "argv", ["noema_review_document.py", str(docx_path)])
+
+ with pytest.raises(SystemExit) as exc_info:
+ runpy.run_path(
+ str(Path(__file__).resolve().parents[1] / "scripts/ci/noema_review_document.py"),
+ run_name="__main__",
+ )
+
+ assert exc_info.value.code == 0
+ assert "DOCX-REVIEW-MARKER" in capsys.readouterr().out
+
+
+def test_main_entrypoint_exits_1_for_a_missing_file(tmp_path, monkeypatch, capsys):
+ """The __main__ CLI path exits 1 and reports the error for a missing file."""
+ missing_path = tmp_path / "missing.docx"
+ monkeypatch.setattr(sys, "argv", ["noema_review_document.py", str(missing_path)])
+
+ with pytest.raises(SystemExit) as exc_info:
+ runpy.run_path(
+ str(Path(__file__).resolve().parents[1] / "scripts/ci/noema_review_document.py"),
+ run_name="__main__",
+ )
+
+ assert exc_info.value.code == 1
+ assert capsys.readouterr().err.strip() != ""
+
+
+def test_fetch_file_content_at_ref_rejects_malformed_base64(monkeypatch):
+ """A content response that is not valid base64 fails closed instead of decoding garbage."""
+ monkeypatch.setattr(noema, "run", lambda _args, stdin=None: "not*valid*base64!!")
+
+ with pytest.raises(RuntimeError, match="malformed base64"):
+ noema.fetch_file_content_at_ref("owner/repo", "docs/review.docx", "head")