Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
36 changes: 36 additions & 0 deletions tests/e2e/helpers.py
Original file line number Diff line number Diff line change
Expand Up @@ -236,8 +236,44 @@
"msg_user": "[data-testid='msg-user']", # user message bubble
"msg_assistant": "[data-testid='msg-assistant']", # assistant message bubble
"msg_system": "[data-testid='msg-system']", # system notice bubble
"message_list_scroll": "[data-testid='message-list-scroll']",
"message_list_content": "[data-testid='message-list-content']",
"message_list_load_older": "[data-testid='message-list-load-older']",
"auth_gate": "[data-testid='auth-gate']",
"auth_gate_for": "[data-testid='auth-gate'][data-auth-challenge='{kind}']",
"auth_token_input": "[data-testid='auth-token-input']",
"auth_oauth_open": "[data-testid='auth-oauth-open']",
"channel_connect_card": "[data-testid='channel-connect-card']",
"channel_connect_card_for": (
"[data-testid='channel-connect-card'][data-channel='{channel}']"
"[data-strategy='{strategy}']"
),
"channel_connect_dismiss": "[data-testid='channel-connect-dismiss']",
"slack_pairing_section": "[data-testid='slack-pairing-section']",
"slack_pairing_code_input": "[data-testid='slack-pairing-code-input']",
"slack_pairing_submit": "[data-testid='slack-pairing-submit']",
"slack_pairing_success": "[data-testid='slack-pairing-success']",
"slack_pairing_error": "[data-testid='slack-pairing-error']",
"approval_card": "[data-testid='approval-card']", # approval gate card
"busy_gate_notice": "[data-testid='busy-gate-notice']", # gate busy notice
"activity_run": "[data-testid='activity-run']",
"activity_run_toggle": "[data-testid='activity-run-toggle']",
"activity_run_items": "[data-testid='activity-run-items']",
"tool_activity_card": "[data-testid='tool-activity-card']",
"tool_activity_card_for": "[data-testid='tool-activity-card'][data-tool-name='{name}']",
"tool_activity_toggle": "[data-testid='tool-activity-toggle']",
"tool_activity_detail": "[data-testid='tool-activity-detail']",
"projects_grid": "[data-testid='projects-grid']",
"projects_search_input": "[data-testid='projects-search-input']",
"project_card": "[data-testid='project-card']",
"project_card_for": "[data-testid='project-card'][data-project-id='{id}']",
"project_open_workspace": "[data-testid='project-open-workspace']",
"project_workspace": "[data-testid='project-workspace']",
"project_workspace_for": "[data-testid='project-workspace'][data-project-id='{id}']",
"project_workspace_title": "[data-testid='project-workspace-title']",
"project_filesystem_entry_for": (
"[data-testid='project-filesystem-entry'][data-entry-path='{path}']"
),
# Download chip for an agent-produced workspace file; `{path}` selects one.
# Clicking a chip opens the shared attachment preview modal, whose footer
# carries the Download action.
Expand Down
129 changes: 111 additions & 18 deletions tests/e2e/mock_llm.py
Original file line number Diff line number Diff line change
Expand Up @@ -38,14 +38,19 @@
(re.compile(r"link test", re.IGNORECASE),
"See [the pull request](https://example.com/pr/1) for details."),
# Reborn v2 download chips: after the agent writes a CSV and a PDF (the
# write_file dispatch lives in TOOL_CALL_PATTERNS), it replies referencing
# their /workspace paths so the WebUI renders downloadable file chips. Fires
# after the tool calls run (match_tool_call dedups the already-run writes).
# builtin__write_file dispatch lives in TOOL_CALL_PATTERNS), it replies
# referencing their /workspace paths so the WebUI renders downloadable file
# chips. Fires after the tool calls run (match_tool_call dedups the
# already-run writes).
(
re.compile(r"produce a downloadable csv and pdf", re.IGNORECASE),
"Done — I saved /workspace/report.csv and /workspace/report.pdf. "
"Both are ready to download.",
),
(
re.compile(r"reborn write approval file (?P<label>[a-z0-9_-]+)", re.IGNORECASE),
"Done - saved the approval test file.",
),
(re.compile(r"\bhello\b|\bhi\b|\bhey\b", re.IGNORECASE), "Hello! How can I help you today?"),
(re.compile(r"2\s*\+\s*2|two plus two", re.IGNORECASE), "The answer is 4."),
(
Expand Down Expand Up @@ -139,6 +144,17 @@
)

TOOL_CALL_PATTERNS = [
# Reborn parallel tool-call port: the Reborn provider-visible builtin tool
# names are namespaced/sanitized, while the legacy engine keeps using the
# unqualified trigger below.
(
re.compile(r"reborn parallel echo and time", re.IGNORECASE),
"builtin__echo",
lambda _: [
{"tool_name": "builtin__echo", "arguments": {"message": "parallel-test"}},
{"tool_name": "builtin__time", "arguments": {"operation": "now"}},
],
),
# Parallel tool calls: return both echo and time in one response
(
re.compile(r"parallel echo and time", re.IGNORECASE),
Expand All @@ -148,24 +164,36 @@
{"tool_name": "time", "arguments": {"operation": "now"}},
],
),
(
re.compile(r"reborn builtin echo (.+)", re.IGNORECASE),
"builtin__echo",
lambda m: {"message": m.group(1)},
),
(
re.compile(r"reborn builtin time", re.IGNORECASE),
"builtin__time",
lambda _: {"operation": "now"},
),
(re.compile(r"echo (.+)", re.IGNORECASE), "echo", lambda m: {"message": m.group(1)}),
# Reborn v2 download chips: one assistant turn writes a CSV and a PDF into
# the project workspace. After both results land, match_tool_call dedups
# write_file and the conversation falls through to the CANNED_RESPONSES
# reply that references the two paths.
# the project workspace. Reborn exposes this first-party tool by capability
# id; the provider-facing tool name sanitizes dots as "__". After both
# results land, match_tool_call dedups builtin__write_file and the
# conversation falls through to the CANNED_RESPONSES reply that
# references the two paths.
(
re.compile(r"produce a downloadable csv and pdf", re.IGNORECASE),
"write_file",
"builtin__write_file",
lambda _: [
{
"tool_name": "write_file",
"tool_name": "builtin__write_file",
"arguments": {
"path": "/workspace/report.csv",
"content": "name,score\nalice,90\nbob,85\n",
},
},
{
"tool_name": "write_file",
"tool_name": "builtin__write_file",
"arguments": {
"path": "/workspace/report.pdf",
"content": (
Expand All @@ -176,6 +204,14 @@
},
],
),
(
re.compile(r"reborn write approval file (?P<label>[a-z0-9_-]+)", re.IGNORECASE),
"builtin__write_file",
lambda m: {
"path": f"/workspace/reborn-approval-{m.group('label')}.txt",
"content": f"approved {m.group('label')}\n",
},
),
(
re.compile(
r"install https://github\.com/Pika-Labs/Pika-Skills/?(?=$|\s)",
Expand Down Expand Up @@ -957,11 +993,28 @@ def _active_skill_names(messages: list[dict]) -> set[str]:
return names


def _typed_user_content_for_skill_detection(messages: list[dict]) -> str:
"""Return user-authored text without generated attachment context.

Reborn appends a model-visible ``<attachments>`` block to user messages so
tools can reason about uploaded files and their /workspace storage paths.
Those paths are not user-typed slash skills, so the mock's missing-skill
heuristic must ignore that generated block.
"""
return re.sub(
r"\n+<attachments>.*?</attachments>\s*$",
"",
_last_user_content(messages),
flags=re.DOTALL,
)


def _missing_explicit_skills(messages: list[dict]) -> list[str]:
active = _active_skill_names(messages)
missing = []
seen = set()
for match in re.finditer(r'(^|[\s"\(])/(?P<name>[A-Za-z0-9._-]+)', _last_user_content(messages)):
content = _typed_user_content_for_skill_detection(messages)
for match in re.finditer(r'(^|[\s"\(])/(?P<name>[A-Za-z0-9._-]+)', content):
name = match.group("name").lower()
if name in active or name in seen:
continue
Expand Down Expand Up @@ -1301,6 +1354,21 @@ def _normalize_tool_calls(tool_name: str, value: object) -> list[dict]:
return [{"tool_name": tool_name, "arguments": value}]


def _advertised_tool_names(tools: object) -> set[str]:
names: set[str] = set()
if not isinstance(tools, list):
return names
for tool in tools:
if not isinstance(tool, dict):
continue
function = tool.get("function")
if isinstance(function, dict) and isinstance(function.get("name"), str):
names.add(function["name"])
elif isinstance(tool.get("name"), str):
names.add(tool["name"])
return names


def match_tool_call(
messages: list[dict],
has_tools: bool,
Expand Down Expand Up @@ -1707,16 +1775,33 @@ async def _send_sse(resp: web.StreamResponse, data: dict):
await resp.write(f"data: {json.dumps(data)}\n\n".encode())


def match_special_response(messages: list[dict], has_tools: bool) -> dict | None:
def _preferred_tool_name(available_tool_names: set[str], legacy: str) -> str:
reborn_name = {
"echo": "builtin__echo",
"time": "builtin__time",
}.get(legacy)
if reborn_name and reborn_name in available_tool_names:
return reborn_name
return legacy


def match_special_response(
messages: list[dict],
has_tools: bool,
available_tool_names: set[str] | None = None,
) -> dict | None:
"""Deterministic issue-specific responses for agent-loop recovery tests."""
last_user = _last_user_content(messages)
available_tool_names = available_tool_names or set()
echo_tool = _preferred_tool_name(available_tool_names, "echo")
time_tool = _preferred_tool_name(available_tool_names, "time")

if _conversation_has_user_trigger(messages, LOOP_FOREVER_TRIGGER):
if has_tools:
return {
"type": "tool_call",
"tool_call": {
"tool_name": "echo",
"tool_name": echo_tool,
"arguments": {"message": "loop-iteration"},
},
}
Expand All @@ -1730,7 +1815,7 @@ def match_special_response(messages: list[dict], has_tools: bool) -> dict | None
return {
"type": "truncated_tool_call",
"tool_call": {
"tool_name": "time",
"tool_name": time_tool,
"arguments": {},
},
"content": "Attempting a tool call but the response was truncated.",
Expand All @@ -1744,7 +1829,7 @@ def match_special_response(messages: list[dict], has_tools: bool) -> dict | None
return {
"type": "tool_call",
"tool_call": {
"tool_name": "time",
"tool_name": time_tool,
"arguments": {"operation": "broken-operation"},
},
}
Expand All @@ -1761,12 +1846,18 @@ def match_special_response(messages: list[dict], has_tools: bool) -> dict | None
if n == 0 and has_tools:
return {
"type": "tool_call",
"tool_call": {"tool_name": "echo", "arguments": {"message": "step-one"}},
"tool_call": {
"tool_name": echo_tool,
"arguments": {"message": "step-one"},
},
}
if n == 1 and has_tools:
return {
"type": "tool_call",
"tool_call": {"tool_name": "time", "arguments": {"operation": "now"}},
"tool_call": {
"tool_name": time_tool,
"arguments": {"operation": "now"},
},
}
return {
"type": "text",
Expand Down Expand Up @@ -2032,7 +2123,9 @@ async def chat_completions(request: web.Request) -> web.StreamResponse:
_last_chat_request = body
messages = body.get("messages", [])
stream = body.get("stream", False)
has_tools = bool(body.get("tools"))
tools = body.get("tools")
has_tools = bool(tools)
available_tool_names = _advertised_tool_names(tools)
cid = f"mock-{uuid.uuid4().hex[:8]}"

slow_response_delay = _conversation_slow_response_delay(messages)
Expand All @@ -2054,7 +2147,7 @@ async def chat_completions(request: web.Request) -> web.StreamResponse:

# Special chat-loop recovery cases that intentionally override the normal
# tool-result summary path (for example, the looping case).
special = match_special_response(messages, has_tools)
special = match_special_response(messages, has_tools, available_tool_names)
if special and _conversation_has_user_trigger(messages, LOOP_FOREVER_TRIGGER):
return await _dispatch_special_response(request, cid, stream, special)
# Multi-step chain: must bypass tool-result-summary to issue second tool call
Expand Down
Loading
Loading