Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
22 commits
Select commit Hold shift + click to select a range
f6553a3
spec: month-1 mvp — mocks-first real-estate AI agent
choguun Jul 3, 2026
ad52e02
feat(T-001): repo scaffold + FastAPI + Next.js shells + /health + CI
choguun Jul 3, 2026
3e7a226
feat(T-002): mock Supabase adapter + migration runner + factory
choguun Jul 3, 2026
29ff3c0
chore(aidlc): mark T-002 done; next is T-003 (auth)
choguun Jul 3, 2026
edc5444
feat(T-003): auth — signup / login / LIFF / /me + auth pages
choguun Jul 3, 2026
078b61e
feat(T-004): properties CRUD + scoped list page
choguun Jul 3, 2026
de144fd
feat(T-005): mock storage adapter + image upload endpoint + new prope…
choguun Jul 3, 2026
3b2a597
feat(T-006): mock AI adapter + /api/generate-listing + ListingPreview
choguun Jul 3, 2026
1fc1032
feat(T-007): generated listings persistence + property detail page
choguun Jul 3, 2026
3d2580f
feat(T-008): mock LINE adapter + signed webhook (HMAC-SHA256)
choguun Jul 3, 2026
efc5217
feat(T-009): LINE → Lead + Message pipeline (idempotent)
choguun Jul 3, 2026
e613631
feat(T-010): leads list/chat UI + outbound reply via mock LINE
choguun Jul 3, 2026
de764ef
feat(T-011): dashboard endpoint + agent home page
choguun Jul 3, 2026
b09391a
chore(aidlc): mark T-012 done — Month-1 MVP shipped
choguun Jul 3, 2026
a552993
feat(T-012): Playwright E2E + real_swap tests + docs + coverage gate
choguun Jul 3, 2026
eca9ab8
fix(review): Tier-1 fixes from sub-agent codebase review
choguun Jul 3, 2026
7e0beb9
fix(review-t2): Tier-2 review fixes (frontend behavior + missing tests)
choguun Jul 3, 2026
bf6a55b
docs(review-t3): accuracy fixes — docs match shipped code
choguun Jul 3, 2026
7bd407b
chore(state): bump stats after Tier-3 docs cleanup
choguun Jul 3, 2026
93d2607
feat(line): hermes-agent#23197 takeaways — body cap, outbound transfo…
choguun Jul 3, 2026
0e7a943
fix(line): P0 review fixes — protocol bones, mock↔real parity, doc lies
choguun Jul 3, 2026
30f665d
fix(line): deduplicate REPLY_TOKEN_TTL_SECONDS
choguun Jul 3, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
150 changes: 144 additions & 6 deletions backend/app/adapters/line/base.py
Original file line number Diff line number Diff line change
@@ -1,19 +1,33 @@
"""LINE adapter Protocol + HMAC-SHA256 sign/verify helpers.
"""LINE adapter Protocol + HMAC-SHA256 sign/verify helpers + outbound transforms.

LINE's webhook auth scheme: a header `X-Line-Signature` carries a
LINE's webhook auth scheme: a header `X-Line-Signature`` carries a
base64(HMAC-SHA256(channel_secret, raw_request_body)). Verifying the
signature against the raw bytes (BEFORE JSON parsing) is the single
thing that prevents spoofed events.

Outbound transforms (``strip_markdown``, ``split_for_line``) are shared
between mock and real adapters. LINE cannot render Markdown and caps
each bubble at 5000 chars + 5 messages per Reply/Push call.
"""

from __future__ import annotations

import base64
import hashlib
import hmac
import re
from typing import Protocol, runtime_checkable

SIGNATURE_HEADER = "X-Line-Signature"
WEBHOOK_BODY_MAX_BYTES = 1 * 1024 * 1024 # 1 MiB — memory-exhaustion guard

# LINE Messaging API limits (https://developers.line.biz/en/reference/messaging-api/).
LINE_MAX_MESSAGES_PER_CALL = 5
LINE_SAFE_BUBBLE_CHARS = 4500 # leave margin under the 5000 hard cap

# Reply-token lifetime per LINE docs — about 60s. Shared between mock
# and real so the cache TTL is consistent.
REPLY_TOKEN_TTL_SECONDS = 60


def sign_line_webhook(body: bytes, channel_secret: str) -> str:
Expand All @@ -37,13 +51,116 @@ def verify_line_webhook(body: bytes, signature: str | None, channel_secret: str)
return False


# ─── Outbound transforms (used by both mock and real adapters) ─────────
_MD_HEADING = re.compile(r"^#{1,6}\s+", flags=re.MULTILINE)
_MD_BOLD = re.compile(r"\*\*(.+?)\*\*", flags=re.DOTALL)
_MD_ITALIC_STAR = re.compile(r"(?<!\*)\*(?!\*)(.+?)(?<!\*)\*(?!\*)", flags=re.DOTALL)
_MD_BOLD_UNDER = re.compile(r"__(.+?)__", flags=re.DOTALL)
_MD_ITALIC_UNDER = re.compile(r"(?<!_)_(?!_)(.+?)(?<!_)_(?!_)", flags=re.DOTALL)
_MD_CODE_FENCE = re.compile(r"```([\s\S]*?)```")
_MD_CODE_INLINE = re.compile(r"`([^`\n]+)`")
_MD_BULLET = re.compile(r"^\s*[-*+]\s+", flags=re.MULTILINE)
_MD_BLOCKQUOTE = re.compile(r"^\s*>\s?", flags=re.MULTILINE)


def strip_markdown(text: str) -> str:
"""Strip the Markdown marks LINE cannot render. URLs are preserved as-is.

LINE renders a small subset of formatting on iOS/Android clients but
not consistently across the web/desktop/macOS clients. To be safe,
we strip the marks and keep the text. Bare URLs render as tappable
links automatically.

Strips: ATX headings, **bold**, *italic* (and __/underscore__), `code`,
code fences, leading list bullets, leading blockquote markers.
Does not touch: line breaks, tables, links (we leave the URL bare).
"""
text = _MD_CODE_FENCE.sub(lambda m: m.group(1).strip("\n"), text)
text = _MD_CODE_INLINE.sub(r"\1", text)
text = _MD_BOLD.sub(r"\1", text)
text = _MD_BOLD_UNDER.sub(r"\1", text)
text = _MD_ITALIC_STAR.sub(r"\1", text)
text = _MD_ITALIC_UNDER.sub(r"\1", text)
text = _MD_HEADING.sub("", text)
text = _MD_BULLET.sub("", text)
text = _MD_BLOCKQUOTE.sub("", text)
return text.strip()


def split_for_line(text: str, *, max_chars: int = LINE_SAFE_BUBBLE_CHARS) -> list[str]:
"""Split ``text`` into at most ``LINE_MAX_MESSAGES_PER_CALL`` chunks.

Strategy: prefer paragraph boundaries; on overflow within a
paragraph, prefer sentence boundaries (period, Thai ``。``); if a
single sentence is still over ``max_chars``, hard-cut. The caller
should treat the truncated remainder as lost — we never return more
than the LINE API allows.
"""
text = text or ""
if len(text) <= max_chars:
return [text]

# First pass: paragraph boundaries.
paragraphs = text.split("\n\n")
chunks: list[str] = []
current = ""
for p in paragraphs:
candidate = (current + "\n\n" + p).strip() if current else p.strip()
if len(candidate) <= max_chars:
current = candidate
else:
if current:
chunks.append(current)
# Paragraph too big — push it through the sentence splitter.
chunks.extend(_split_long(p.strip(), max_chars))
current = ""
if current:
chunks.append(current)

return chunks[:LINE_MAX_MESSAGES_PER_CALL]


def _split_long(text: str, max_chars: int) -> list[str]:
"""Sentence-or-hard-cut splitter for a single over-long paragraph."""
out: list[str] = []
rest = text
while len(rest) > max_chars and len(out) < LINE_MAX_MESSAGES_PER_CALL:
window = rest[:max_chars]
# Prefer sentence terminators (Western ``. `` + Thai ``。``).
boundary = max(window.rfind(". "), window.rfind("。"))
if boundary == -1 or boundary < max_chars // 2:
# No good sentence boundary — hard cut.
boundary = max_chars - 1
out.append(rest[: boundary + 1].strip())
rest = rest[boundary + 1 :].strip()
if rest and len(out) < LINE_MAX_MESSAGES_PER_CALL:
out.append(rest)
return out


@runtime_checkable
class LineAdapter(Protocol):
"""LINE messaging adapter — mock + real (stub for MVP)."""
"""LINE messaging adapter — mock + real.

Concrete implementations live in ``mock.py`` and ``real.py``. Every
router depends on this Protocol only — never on a concrete class.
"""

@property
def channel_secret(self) -> str: ...

@property
def bot_user_id(self) -> str | None:
"""Own channel's LINE userId.

Used to filter self-echoes: when the bot userId is known,
``send_reply(line_user_id, text)`` short-circuits if
``line_user_id == bot_user_id`` (prevents infinite loops). The
mock returns None (no echo filter); the real adapter populates
this from ``GET /v2/bot/info`` when wiring lands.
"""
...

def sign(self, body: bytes) -> str:
"""Sign `body` with the channel secret. Used by tests/dev tooling."""
...
Expand All @@ -52,11 +169,32 @@ def verify(self, body: bytes, signature: str) -> bool:
"""Verify a request signature. Returns False on any mismatch."""
...

def set_reply_token(
self, chat_id: str, token: str, *, ttl_seconds: int = REPLY_TOKEN_TTL_SECONDS
) -> None:
"""Cache a Reply token off an inbound ``message`` event.

The Reply API is free; Push is metered. Inbound webhook
dispatchers call this when an event has a ``replyToken`` so
``send_reply`` can later try the Reply path first and fall
back to Push if missing/expired.

``chat_id`` is the LINE chat identifier (userId for DMs,
groupId/roomId for groups/rooms). For our MVP single-tenant
setup it's always the userId.
"""
...

def send_reply(self, line_user_id: str, text: str) -> dict[str, object]:
"""Send a reply to a LINE user.

Mock records the call and returns `{id, line_user_id, sent_at}`;
real calls LINE's Reply API and returns the same shape so the
router can stay adapter-agnostic.
Mock records the call (after applying ``strip_markdown`` /
``split_for_line`` for parity with the real adapter's outgoing
shape). Real calls LINE's Reply API (preferring the cached
``replyToken``) and falls back to Push.

Self-message filter: when ``bot_user_id`` is set and
``line_user_id == bot_user_id``, returns
``{"skipped": "self-message", ...}`` instead of sending.
"""
...
111 changes: 99 additions & 12 deletions backend/app/adapters/line/mock.py
Original file line number Diff line number Diff line change
@@ -1,7 +1,12 @@
"""In-memory LINE adapter.

Sign + verify work identically to a real client; the mock also keeps a
sent-replies log so tests can inspect what the agent has sent.
The mock applies the same outbound transforms (``strip_markdown`` +
``split_for_line``) that the real adapter will apply when wired. This
keeps mock ↔ real parity on the recorded text — tests can assert on
``mock.sent_replies[-1].text`` as the "what LINE would receive" value.

Sign + verify work identically to a real client; the mock also keeps
sent-reply + reply-token logs so tests can inspect outbound state.
"""

from __future__ import annotations
Expand All @@ -11,7 +16,10 @@
from datetime import datetime, timezone

from app.adapters.line.base import (
REPLY_TOKEN_TTL_SECONDS,
sign_line_webhook,
split_for_line,
strip_markdown,
verify_line_webhook,
)

Expand All @@ -21,34 +29,113 @@ class _SentReply:
line_user_id: str
text: str
sent_at: str
chunk_index: int # 0-based; -1 for "skipped" replies


class LineMockAdapter:
def __init__(self, channel_secret: str) -> None:
def __init__(self, channel_secret: str, *, bot_user_id: str | None = None) -> None:
self._secret = channel_secret
self.received_events: list[dict[str, object]] = []
self.sent_replies: list[_SentReply] = []
# Reply-token cache — keyed by chat_id, with a TTL. Used by the
# mock to exercise the same Reply-token flow the real adapter
# will use. Tests can inspect via ``cached_reply_tokens``.
self._reply_tokens: dict[str, tuple[str, float]] = {}
self._bot_user_id = bot_user_id

@property
def channel_secret(self) -> str:
return self._secret

@property
def bot_user_id(self) -> str | None:
return self._bot_user_id

def sign(self, body: bytes) -> str:
return sign_line_webhook(body, self._secret)

def verify(self, body: bytes, signature: str) -> bool:
return verify_line_webhook(body, signature, self._secret)

def set_reply_token(
self, chat_id: str, token: str, *, ttl_seconds: int = REPLY_TOKEN_TTL_SECONDS
) -> None:
"""Cache a Reply token off an inbound ``message`` event.

Mirrors ``LineRealAdapter.set_reply_token`` so the webhook
dispatcher can call the method blindly. The cache is in-process
and best-effort — production wiring would persist to DB.
"""
import time

self._reply_tokens[chat_id] = (token, time.time() + ttl_seconds)

def cached_reply_tokens(self) -> dict[str, str]:
"""Snapshot of the current cache, key → token (for tests)."""
return {chat_id: token for chat_id, (token, _exp) in self._reply_tokens.items()}

def send_reply(self, line_user_id: str, text: str) -> dict[str, object]:
reply_id = uuid.uuid4().hex[:12]
sent = _SentReply(
line_user_id=line_user_id,
text=text,
sent_at=datetime.now(timezone.utc).isoformat(),
)
self.sent_replies.append(sent)
"""Record a sent reply after applying the outbound transforms.

Real adapter will do the same ``strip_markdown`` + ``split_for_line``
+ Reply-or-Push flow when wired. For mock↔real parity on the
recorded text we apply the same transforms here.

Self-message filter: when ``bot_user_id`` is set and the reply
is to that userId, records a single ``skipped='self-message'``
entry and returns early (no Reply, no Push). Real adapter
behaves the same way.
"""
sent_at = datetime.now(timezone.utc).isoformat()

# Self-message filter — same path as the real adapter's stub.
if self._bot_user_id is not None and line_user_id == self._bot_user_id:
sent = _SentReply(
line_user_id=line_user_id,
text="",
sent_at=sent_at,
chunk_index=-1,
)
self.sent_replies.append(sent)
return {
"id": f"reply-skipped-{uuid.uuid4().hex[:8]}",
"line_user_id": line_user_id,
"sent_at": sent_at,
"skipped": "self-message",
}

# Apply the same outbound transforms the real adapter will.
cleaned = strip_markdown(text)
chunks = split_for_line(cleaned)

# Reply-token routing: tokens are single-use. The mock consumes
# on use (so the second send falls back to push), matching the
# real adapter's behaviour. The real adapter's send_reply
# will call ``consume_reply_token(chat_id)`` then either Reply
# (if a usable token remains) or fall back to Push.
if self._reply_tokens.get(line_user_id):
# Token present — record as 'reply' and consume (single-use).
self._reply_tokens.pop(line_user_id, None)
mode = "reply"
else:
mode = "push"

sent_ids: list[str] = []
for i, chunk in enumerate(chunks):
reply_id = uuid.uuid4().hex[:12]
self.sent_replies.append(
_SentReply(
line_user_id=line_user_id,
text=chunk,
sent_at=sent_at,
chunk_index=i,
)
)
sent_ids.append(reply_id)
return {
"id": f"reply-{reply_id}",
"id": f"reply-{sent_ids[0]}" if sent_ids else f"reply-empty-{uuid.uuid4().hex[:8]}",
"line_user_id": line_user_id,
"sent_at": sent.sent_at,
"sent_at": sent_at,
"mode": mode,
"chunks": sent_ids,
}
Loading
Loading