diff --git a/.gitignore b/.gitignore index 03d58da463..f85fb85cd2 100644 --- a/.gitignore +++ b/.gitignore @@ -160,3 +160,11 @@ clients/java/target/ clients/java/build.sbt clients/java/gradle.properties openapitools.json + +# BFCL test data (downloaded from HuggingFace at runtime) +e2e_test/bfcl/data/*.json +!e2e_test/bfcl/data/.gitkeep + +# BFCL test logs (generated per run) +e2e_test/bfcl_logs/* +!e2e_test/bfcl_logs/.gitkeep diff --git a/e2e_test/bfcl/__init__.py b/e2e_test/bfcl/__init__.py new file mode 100644 index 0000000000..3adcb78394 --- /dev/null +++ b/e2e_test/bfcl/__init__.py @@ -0,0 +1,25 @@ +"""BFCL (Berkeley Function Calling Leaderboard) test infrastructure. + +Provides data loading, evaluation, and per-test logging for open-source +BFCL v3 test cases run against the SMG gateway. +""" + +from .converter import bfcl_to_openai_tools +from .evaluator import BFCLEvaluator, extract_tool_calls, log_file_for_summary +from .loader import ( + BFCL_CATEGORIES, + BFCLCase, + MissingBFCLAnswerFileError, + load_bfcl_category, +) + +__all__ = [ + "BFCL_CATEGORIES", + "BFCLCase", + "BFCLEvaluator", + "MissingBFCLAnswerFileError", + "bfcl_to_openai_tools", + "extract_tool_calls", + "load_bfcl_category", + "log_file_for_summary", +] diff --git a/e2e_test/bfcl/converter.py b/e2e_test/bfcl/converter.py new file mode 100644 index 0000000000..a18f4b92d1 --- /dev/null +++ b/e2e_test/bfcl/converter.py @@ -0,0 +1,47 @@ +"""BFCL-to-OpenAI format converter. + +Converts BFCL function definitions to OpenAI-compatible tool calling format, +handling BFCL-specific schema quirks (non-standard JSON Schema types). +""" + +from __future__ import annotations + + +def _fix_parameter_type(params: dict) -> dict: + """Convert BFCL's non-standard types to valid JSON Schema types recursively.""" + result = dict(params) + ptype = result.get("type") + if ptype == "dict": + result["type"] = "object" + elif ptype == "float": + result["type"] = "number" + elif ptype == "int": + result["type"] = "integer" + elif ptype in ("list", "tuple"): + result["type"] = "array" + props = result.get("properties") + if isinstance(props, dict): + result["properties"] = { + k: _fix_parameter_type(v) if isinstance(v, dict) else v for k, v in props.items() + } + items = result.get("items") + if isinstance(items, dict): + result["items"] = _fix_parameter_type(items) + return result + + +def bfcl_to_openai_tools(bfcl_functions: list[dict]) -> list[dict]: + """Convert BFCL function definitions to OpenAI tools format. + + Handles the BFCL-specific quirks: + - parameters.type "dict" → "object" + - parameters.type "float" → "number" + - Wraps in {"type": "function", "function": ...} + """ + tools = [] + for fn in bfcl_functions: + fixed_fn = dict(fn) + if "parameters" in fixed_fn: + fixed_fn["parameters"] = _fix_parameter_type(fixed_fn["parameters"]) + tools.append({"type": "function", "function": fixed_fn}) + return tools diff --git a/e2e_test/bfcl/data/.gitkeep b/e2e_test/bfcl/data/.gitkeep new file mode 100644 index 0000000000..e69de29bb2 diff --git a/e2e_test/bfcl/download_data.py b/e2e_test/bfcl/download_data.py new file mode 100755 index 0000000000..ad566e702b --- /dev/null +++ b/e2e_test/bfcl/download_data.py @@ -0,0 +1,83 @@ +#!/usr/bin/env python3 +"""Download BFCL v3 open-source test data from HuggingFace. + +Usage: + python e2e_test/bfcl/download_data.py +""" + +from __future__ import annotations + +import os +import shutil +import urllib.request +from pathlib import Path + +HF_BASE = ( + "https://huggingface.co/datasets/gorilla-llm/Berkeley-Function-Calling-Leaderboard/resolve/main" +) + +DATA_DIR = Path(__file__).parent / "data" +TIMEOUT_SECONDS = 60 + +FILES = [ + "BFCL_v3_simple.json", + "BFCL_v3_multiple.json", + "BFCL_v3_parallel.json", + "BFCL_v3_parallel_multiple.json", + "BFCL_v3_irrelevance.json", +] + +ANSWER_FILES = [ + ("possible_answer/BFCL_v3_simple.json", "BFCL_v3_simple_answer.json"), + ("possible_answer/BFCL_v3_multiple.json", "BFCL_v3_multiple_answer.json"), + ("possible_answer/BFCL_v3_parallel.json", "BFCL_v3_parallel_answer.json"), + ( + "possible_answer/BFCL_v3_parallel_multiple.json", + "BFCL_v3_parallel_multiple_answer.json", + ), +] + + +def _count_lines(path: Path) -> int: + with path.open(encoding="utf-8") as handle: + return sum(1 for _ in handle) + + +def _download_one(remote_path: str, local_name: str, *, force: bool = False) -> None: + url = f"{HF_BASE}/{remote_path}" + dest = DATA_DIR / local_name + if dest.exists() and not force: + print(f"Skipping {local_name} (already downloaded)") + print(f" -> {dest.name} ({_count_lines(dest)} entries)") + return + + print(f"Downloading {local_name}...") + tmp = dest.with_suffix(dest.suffix + ".tmp") + try: + with urllib.request.urlopen(url, timeout=TIMEOUT_SECONDS) as resp, tmp.open("wb") as out: + shutil.copyfileobj(resp, out) + out.flush() + os.fsync(out.fileno()) + tmp.replace(dest) + finally: + if tmp.exists(): + tmp.unlink() + + lines = _count_lines(dest) + print(f" → {dest.name} ({lines} entries)") + + +def download(*, force_redownload: bool = False) -> None: + DATA_DIR.mkdir(parents=True, exist_ok=True) + + for fname in FILES: + _download_one(fname, fname, force=force_redownload) + + for remote_path, local_name in ANSWER_FILES: + _download_one(remote_path, local_name, force=force_redownload) + + print("\nDone. All BFCL v3 data downloaded.") + + +if __name__ == "__main__": + download() diff --git a/e2e_test/bfcl/evaluator.py b/e2e_test/bfcl/evaluator.py new file mode 100644 index 0000000000..d6b9349a18 --- /dev/null +++ b/e2e_test/bfcl/evaluator.py @@ -0,0 +1,360 @@ +"""BFCL test evaluator and per-test log writer. + +Validates model tool call output against BFCL structured ground truth +and writes detailed JSON logs for every test case (pass or fail) + +Log directory layout: + bfcl_logs/ + └── / + ├── summary.json + ├── simple/ + │ ├── simple_0_PASS.json + │ └── simple_1_FAIL.json + ├── multiple/ + ├── parallel/ + └── irrelevance/ +""" + +from __future__ import annotations + +import json +import logging +from datetime import UTC, datetime +from pathlib import Path +from typing import Any + +logger = logging.getLogger(__name__) + +BFCL_LOGS_DIR = Path(__file__).parent.parent / "bfcl_logs" + + +def extract_tool_calls(response: Any) -> list[dict[str, Any]]: + """Pull structured tool calls out of an OpenAI ChatCompletion response. + + Returns a list of ``{"name": str, "arguments": dict}`` entries. + Malformed tool calls are included with ``{"_parse_error": ...}`` arguments + so downstream evaluation can distinguish parse failures from missing calls. + """ + choices = getattr(response, "choices", None) or [] + if not choices: + return [] + message = getattr(choices[0], "message", None) + if message is None: + return [] + tool_calls = getattr(message, "tool_calls", None) or [] + result = [] + for tc in tool_calls: + function = getattr(tc, "function", None) + if function is None: + logger.warning("Skipping malformed tool call with no function payload: %r", tc) + continue + try: + args = json.loads(function.arguments) + except (json.JSONDecodeError, TypeError) as exc: + logger.warning( + "Malformed tool call arguments for %r — %s | raw: %r", + function.name, + exc, + function.arguments, + ) + args = {"_parse_error": str(exc)} + result.append({"name": function.name, "arguments": args}) + return result + + +def log_file_for_summary(run_dir: Path, log_path: Path) -> str: + """Return a log path relative to the BFCL run directory for summary entries.""" + return str(log_path.relative_to(run_dir)) + + +class BFCLEvaluator: + """Evaluates BFCL tool call results and manages per-test logging. + + Encapsulates run directory management, tool call comparison logic, + and log/summary writing in a single extensible class. + """ + + def __init__(self, logs_dir: Path = BFCL_LOGS_DIR) -> None: + self.logs_dir = logs_dir + self._run_dir: Path | None = None + + def get_run_dir(self) -> Path: + """Create and return a timestamped run directory for this test session. + + The directory is created once and reused for subsequent calls. + """ + if self._run_dir is None: + ts = datetime.now(UTC).strftime("%Y-%m-%dT%H-%M-%S") + self._run_dir = self.logs_dir / ts + self._run_dir.mkdir(parents=True, exist_ok=True) + return self._run_dir + + def get_existing_run_dir(self) -> Path | None: + """Return the current run directory if one has already been created.""" + return self._run_dir + + def evaluate_tool_calls( + self, + actual_tool_calls: list[dict[str, Any]], + ground_truth: list[dict[str, Any]], + *, + category: str = "", + ) -> tuple[bool, list[str]]: + """Compare actual tool calls against BFCL structured ground truth. + + For irrelevance tests, expects zero tool calls. + For all others, validates function names and arguments against + the ground truth's list-of-possible-values format. + + Returns (passed, list_of_error_messages). + """ + if category == "irrelevance": + if actual_tool_calls: + return False, [ + f"Irrelevance test: expected 0 tool calls, got {len(actual_tool_calls)}" + ] + return True, [] + + if not ground_truth: + if not actual_tool_calls: + return True, [] + return False, ["No ground truth available but model produced tool calls"] + + errors: list[str] = [] + + if len(actual_tool_calls) != len(ground_truth): + errors.append( + f"Expected {len(ground_truth)} tool call(s), got {len(actual_tool_calls)}" + ) + + gt_for_actual = self._match_tool_calls(actual_tool_calls, ground_truth) + matched_gt_indices = {gt_idx for gt_idx in gt_for_actual if gt_idx is not None} + + for actual, gt_idx in zip(actual_tool_calls, gt_for_actual): + if gt_idx is None: + errors.append(f"Unexpected tool call: '{actual.get('name', '?')}'") + + for gt_idx, gt_entry in enumerate(ground_truth): + if gt_idx not in matched_gt_indices: + gt_name = next(iter(gt_entry.keys()), "?") + errors.append(f"Unmatched expected call: '{gt_name}'") + + return len(errors) == 0, errors + + def save_test_log( + self, + run_dir: Path, + *, + test_id: str, + category: str, + model: str, + parser: str, + backend: str, + request_payload: dict[str, Any], + response_payload: dict[str, Any] | None, + ground_truth: list[Any], + actual_tool_calls: list[dict[str, Any]], + passed: bool, + errors: list[str], + latency_ms: float, + ) -> Path: + """Write a detailed JSON log file for a single BFCL test case.""" + cat_dir = run_dir / category + cat_dir.mkdir(parents=True, exist_ok=True) + + safe_id = test_id.replace("/", "_").replace(" ", "_") + safe_model = model.replace("/", "_").replace(" ", "_") + safe_backend = backend.replace("/", "_").replace(" ", "_") + safe_parser = parser.replace("/", "_").replace(" ", "_") + status = "PASS" if passed else "FAIL" + filename = f"{safe_id}_{safe_model}_{safe_backend}_{safe_parser}_{status}.json" + + log_entry = { + "test_id": test_id, + "category": category, + "model": model, + "parser": parser, + "backend": backend, + "timestamp": datetime.now(UTC).isoformat(), + "passed": passed, + "request": request_payload, + "response": response_payload, + "ground_truth": ground_truth, + "actual_tool_calls": actual_tool_calls, + "errors": errors, + "latency_ms": round(latency_ms, 2), + } + + out_path = cat_dir / filename + out_path.write_text(json.dumps(log_entry, indent=2, default=str), encoding="utf-8") + return out_path + + def save_summary(self, run_dir: Path, results: list[dict[str, Any]]) -> dict[str, Any]: + """Write a rich summary.json with per-category stats, failure diagnostics, + and latency distribution — everything you need at a glance.""" + by_cat: dict[str, dict[str, Any]] = {} + all_latencies: list[float] = [] + failures: list[dict[str, Any]] = [] + + for r in results: + cat = r["category"] + if cat not in by_cat: + by_cat[cat] = {"total": 0, "passed": 0, "failed": 0, "latencies_ms": []} + by_cat[cat]["total"] += 1 + lat = r.get("latency_ms", 0.0) + by_cat[cat]["latencies_ms"].append(lat) + all_latencies.append(lat) + + if r["passed"]: + by_cat[cat]["passed"] += 1 + else: + by_cat[cat]["failed"] += 1 + failures.append( + { + "test_id": r.get("test_id", "?"), + "category": cat, + "errors": r.get("errors", []), + "latency_ms": round(lat, 1), + "finish_reason": r.get("finish_reason"), + "completion_tokens": r.get("completion_tokens"), + "had_reasoning": r.get("had_reasoning", False), + "log_file": r.get("log_file", ""), + } + ) + + for cat_stats in by_cat.values(): + lats = cat_stats.pop("latencies_ms") + if lats: + lats_sorted = sorted(lats) + n = len(lats_sorted) + cat_stats["latency_ms"] = { + "min": round(lats_sorted[0], 1), + "median": round(lats_sorted[n // 2], 1), + "p95": round(lats_sorted[min(int(n * 0.95), n - 1)], 1), + "max": round(lats_sorted[-1], 1), + "mean": round(sum(lats) / n, 1), + } + + total = sum(c["total"] for c in by_cat.values()) + passed = sum(c["passed"] for c in by_cat.values()) + + latency_summary = {} + if all_latencies: + s = sorted(all_latencies) + n = len(s) + latency_summary = { + "min": round(s[0], 1), + "median": round(s[n // 2], 1), + "p95": round(s[min(int(n * 0.95), n - 1)], 1), + "max": round(s[-1], 1), + "mean": round(sum(s) / n, 1), + } + + summary = { + "timestamp": datetime.now(UTC).isoformat(), + "total": total, + "passed": passed, + "failed": total - passed, + "accuracy_pct": round(passed / total * 100, 2) if total else 0.0, + "latency_ms": latency_summary, + "by_category": by_cat, + "failures": failures, + } + + (run_dir / "summary.json").write_text(json.dumps(summary, indent=2), encoding="utf-8") + return summary + + # ------------------------------------------------------------------ + # Private helpers + # ------------------------------------------------------------------ + + def _call_matches(self, actual: dict[str, Any], gt_entry: dict[str, Any]) -> bool: + """Check whether a single actual tool call satisfies a ground truth entry.""" + expected_name = next(iter(gt_entry.keys()), None) + if expected_name is None or actual.get("name") != expected_name: + return False + expected_args = gt_entry.get(expected_name, {}) + if not isinstance(expected_args, dict): + return False + actual_args = actual.get("arguments") + if not isinstance(actual_args, dict): + return False + if set(actual_args) != set(expected_args): + return False + for param_name, possible_values in expected_args.items(): + actual_val = actual_args.get(param_name) + if not isinstance(possible_values, list): + possible_values = [possible_values] + if not any(self._values_match(actual_val, pv) for pv in possible_values): + return False + return True + + def _match_tool_calls( + self, + actual_tool_calls: list[dict[str, Any]], + ground_truth: list[dict[str, Any]], + ) -> list[int | None]: + """Return a maximum matching from actual calls to ground-truth entries.""" + candidate_gt_indices = [ + [ + gt_idx + for gt_idx, gt_entry in enumerate(ground_truth) + if self._call_matches(actual, gt_entry) + ] + for actual in actual_tool_calls + ] + actual_for_gt: list[int | None] = [None] * len(ground_truth) + + def _assign(actual_idx: int, seen_gt_indices: set[int]) -> bool: + for gt_idx in candidate_gt_indices[actual_idx]: + if gt_idx in seen_gt_indices: + continue + seen_gt_indices.add(gt_idx) + prev_actual_idx = actual_for_gt[gt_idx] + if prev_actual_idx is None or _assign(prev_actual_idx, seen_gt_indices): + actual_for_gt[gt_idx] = actual_idx + return True + return False + + for actual_idx in sorted( + range(len(actual_tool_calls)), + key=lambda idx: len(candidate_gt_indices[idx]), + ): + _assign(actual_idx, set()) + + gt_for_actual: list[int | None] = [None] * len(actual_tool_calls) + for gt_idx, matched_actual_idx in enumerate(actual_for_gt): + if matched_actual_idx is not None: + gt_for_actual[matched_actual_idx] = gt_idx + return gt_for_actual + + @staticmethod + def _values_match(actual: Any, expected: Any) -> bool: + """Flexible comparison handling type coercion and empty-string-means-absent.""" + if isinstance(expected, dict) and isinstance(actual, dict): + if len(expected) != len(actual): + return False + return all( + k in actual and BFCLEvaluator._values_match(actual[k], v) + for k, v in expected.items() + ) + if isinstance(expected, list) and isinstance(actual, list): + if len(expected) != len(actual): + return False + return all(BFCLEvaluator._values_match(a, e) for a, e in zip(actual, expected)) + if actual == expected: + if isinstance(actual, bool) != isinstance(expected, bool): + return False + return True + if expected == "" and actual is None: + return True + try: + if not isinstance(actual, bool) and not isinstance(expected, bool): + if float(actual) == float(expected): + return True + except (TypeError, ValueError): + pass + if isinstance(expected, str) and isinstance(actual, str): + if actual.strip().lower() == expected.strip().lower(): + return True + return False diff --git a/e2e_test/bfcl/loader.py b/e2e_test/bfcl/loader.py new file mode 100644 index 0000000000..b083e3499d --- /dev/null +++ b/e2e_test/bfcl/loader.py @@ -0,0 +1,124 @@ +"""BFCL v3 data loader. + +Loads open-source BFCL v3 test data from the Gorilla project +(gorilla-llm/Berkeley-Function-Calling-Leaderboard on HuggingFace) +and converts it to OpenAI-compatible tool calling format. + +""" + +from __future__ import annotations + +import json +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any + +DATA_DIR = Path(__file__).parent / "data" + +BFCL_CATEGORIES = [ + "simple", + "multiple", + "parallel", + "parallel_multiple", + "irrelevance", +] + +_FILE_MAP = { + "simple": "BFCL_v3_simple.json", + "multiple": "BFCL_v3_multiple.json", + "parallel": "BFCL_v3_parallel.json", + "parallel_multiple": "BFCL_v3_parallel_multiple.json", + "irrelevance": "BFCL_v3_irrelevance.json", +} + +_ANSWER_FILE_MAP = { + "simple": "BFCL_v3_simple_answer.json", + "multiple": "BFCL_v3_multiple_answer.json", + "parallel": "BFCL_v3_parallel_answer.json", + "parallel_multiple": "BFCL_v3_parallel_multiple_answer.json", +} + +DOWNLOAD_HINT = "Run: python e2e_test/bfcl/download_data.py" + + +class MissingBFCLAnswerFileError(FileNotFoundError): + """Raised when BFCL question data exists but the mapped answer file is missing.""" + + +@dataclass +class BFCLCase: + """A single BFCL test case.""" + + id: str + question: list[dict[str, Any]] + function: list[dict[str, Any]] + ground_truth: list[dict[str, Any]] = field(default_factory=list) + category: str = "" + + +def load_bfcl_category( + category: str, + *, + limit: int | None = None, +) -> list[BFCLCase]: + """Load BFCL test cases for a category, merged with ground truth answers.""" + if category not in _FILE_MAP: + raise ValueError(f"Unknown category: {category}. Choose from {BFCL_CATEGORIES}") + + question_path = DATA_DIR / _FILE_MAP[category] + if not question_path.exists(): + raise FileNotFoundError(f"BFCL data not found: {question_path}. {DOWNLOAD_HINT}") + + questions_by_id: dict[str, dict] = {} + with question_path.open(encoding="utf-8") as f: + for line in f: + line = line.strip() + if not line: + continue + entry = json.loads(line) + questions_by_id[entry["id"]] = entry + + answer_filename = _ANSWER_FILE_MAP.get(category) + answers_by_id: dict[str, list] = {} + if answer_filename: + answer_path = DATA_DIR / answer_filename + if not answer_path.exists(): + raise MissingBFCLAnswerFileError( + f"BFCL answer file not found: {answer_path}. {DOWNLOAD_HINT}" + ) + with answer_path.open(encoding="utf-8") as f: + for line in f: + line = line.strip() + if not line: + continue + entry = json.loads(line) + answers_by_id[entry["id"]] = entry.get("ground_truth", []) + + results: list[BFCLCase] = [] + for test_id, entry in questions_by_id.items(): + if answer_filename and test_id not in answers_by_id: + raise MissingBFCLAnswerFileError( + f"BFCL ground truth missing for category={category!r}, " + f"id={test_id!r} in {answer_filename}" + ) + raw_question = entry.get("question", []) + messages = ( + raw_question[0] if raw_question and isinstance(raw_question[0], list) else raw_question + ) + + ground_truth = answers_by_id.get(test_id, []) + + results.append( + BFCLCase( + id=test_id, + question=messages, + function=entry.get("function", []), + ground_truth=ground_truth, + category=category, + ) + ) + + if limit is not None: + results = results[:limit] + + return results diff --git a/e2e_test/bfcl/session_state.py b/e2e_test/bfcl/session_state.py new file mode 100644 index 0000000000..5aa97a8b2c --- /dev/null +++ b/e2e_test/bfcl/session_state.py @@ -0,0 +1,55 @@ +"""Thread-safe session-level result collector for BFCL tests. + +Isolated here so both test_bfcl.py and fixtures/hooks.py import the same +module object — giving them a shared view of accumulated results, and routing +session lifecycle. +""" + +from __future__ import annotations + +import logging +import threading +from pathlib import Path +from typing import Any + +from .evaluator import BFCLEvaluator + +logger = logging.getLogger(__name__) + +_results_lock = threading.Lock() +_all_results: list[dict[str, Any]] = [] +_evaluator = BFCLEvaluator() + + +def get_evaluator() -> BFCLEvaluator: + """Return the shared session evaluator instance.""" + return _evaluator + + +def append_result(result: dict[str, Any]) -> None: + """Record a single BFCL case result for session summary generation.""" + with _results_lock: + _all_results.append(result) + + +def get_or_create_run_dir() -> Path: + """Return the session's log directory, creating it exactly once.""" + with _results_lock: + return _evaluator.get_run_dir() + + +def write_summary_if_needed() -> None: + """Write summary.json — called from fixtures/hooks.py at session end.""" + with _results_lock: + results = list(_all_results) + run_dir = _evaluator.get_existing_run_dir() + if not results or run_dir is None: + return + summary = _evaluator.save_summary(run_dir, results) + logger.info( + "BFCL summary: %d/%d passed (%.1f%%) — %s/summary.json", + summary["passed"], + summary["total"], + summary["accuracy_pct"], + run_dir, + ) diff --git a/e2e_test/bfcl_logs/.gitkeep b/e2e_test/bfcl_logs/.gitkeep new file mode 100644 index 0000000000..e69de29bb2 diff --git a/e2e_test/chat_completions/test_bfcl.py b/e2e_test/chat_completions/test_bfcl.py new file mode 100644 index 0000000000..1ebc04bdcb --- /dev/null +++ b/e2e_test/chat_completions/test_bfcl.py @@ -0,0 +1,330 @@ +"""BFCL (Berkeley Function Calling Leaderboard) E2E Tests. + +Runs open-source BFCL v3 test cases against the SMG gateway with +per-test JSON logging for full observability. + +Logs are written to e2e_test/bfcl_logs///_PASS|FAIL.json +with a summary.json at the run root. + +Test categories: + - simple: single function call (400 cases) + - multiple: pick correct function from several (200 cases) + - parallel: make parallel function calls (200 cases) + - parallel_multiple: parallel + multiple (200 cases) + - irrelevance: model should NOT call any function (240 cases) + +Usage: + # Run all BFCL tests (requires GPU worker + gateway via setup_backend fixture) + pytest e2e_test/chat_completions/test_bfcl.py -v + + # Run only simple category + pytest e2e_test/chat_completions/test_bfcl.py -k "simple_" -v + + # Run with a subset (first 20 per category) + BFCL_LIMIT=20 pytest e2e_test/chat_completions/test_bfcl.py -v + +""" + +from __future__ import annotations + +import logging +import os +import re +import time +from pathlib import Path +from typing import Any + +import openai +import pytest +from bfcl import ( + BFCLCase, + MissingBFCLAnswerFileError, + bfcl_to_openai_tools, + extract_tool_calls, + load_bfcl_category, + log_file_for_summary, +) +from bfcl.session_state import append_result, get_evaluator, get_or_create_run_dir + +logger = logging.getLogger(__name__) + +BFCL_LIMIT = int(os.environ.get("BFCL_LIMIT", "0")) or None +BFCL_CATEGORIES = ("simple", "multiple", "parallel", "parallel_multiple", "irrelevance") + + +# --------------------------------------------------------------------------- +# Data loading helpers (pytest-specific, stay in the test module) +# --------------------------------------------------------------------------- + + +def _load(category: str) -> list[BFCLCase]: + try: + return load_bfcl_category(category, limit=BFCL_LIMIT) + except MissingBFCLAnswerFileError: + raise + except FileNotFoundError: + logger.warning("BFCL data not found for category %r — run download_data.py", category) + return [] + + +# Safe without a lock: pytest_generate_tests runs during collection, +# which is single-threaded even under pytest-parallel (--tests-per-worker N). +_cases_cache: dict[str, list[BFCLCase]] = {} + + +def _selected_categories(keyword: str | None) -> list[str]: + """Infer which BFCL categories need loading from the pytest -k filter.""" + if not keyword: + return list(BFCL_CATEGORIES) + + matched = [] + for category in BFCL_CATEGORIES: + patterns = { + category, + f"{category}_", + category.replace("_", "-"), + category.replace("_", " "), + } + if any(re.search(rf"(? list[BFCLCase]: + """Load and cache BFCL cases only for categories needed by this run.""" + if category not in _cases_cache: + _cases_cache[category] = _load(category) + return _cases_cache[category] + + +def pytest_generate_tests(metafunc: pytest.Metafunc) -> None: + """Parametrize 'case' lazily — data is loaded only when BFCL tests are collected.""" + if "case" not in metafunc.fixturenames: + return + categories = _selected_categories(getattr(metafunc.config.option, "keyword", None)) + cases = [case for category in categories for case in _get_cases_for_category(category)] + if not cases: + metafunc.parametrize("case", [], ids=[]) + return + metafunc.parametrize("case", cases, ids=[c.id for c in cases]) + + +# --------------------------------------------------------------------------- +# Session-scoped fixture for the log directory +# --------------------------------------------------------------------------- + + +@pytest.fixture(scope="session") +def bfcl_run_dir() -> Path: + """Single timestamped directory for all BFCL logs in this test session. + + Safe under pytest-parallel: get_or_create_run_dir() creates the directory + exactly once under a lock; subsequent calls return the same path. + """ + return get_or_create_run_dir() + + +# --------------------------------------------------------------------------- +# Core runner shared by all test classes +# --------------------------------------------------------------------------- + + +def _run_bfcl_case( + *, + case: BFCLCase, + model: str, + parser: str, + backend: str, + client: openai.OpenAI, + run_dir: Path, +) -> None: + """Execute a single BFCL test case, log the result, assert on failure.""" + evaluator = get_evaluator() + category = case.category + test_id = case.id + messages = case.question + tools = bfcl_to_openai_tools(case.function) + + request_payload = { + "model": model, + "messages": messages, + "tools": tools, + "tool_choice": "auto", + "temperature": 0.01, + "max_tokens": 1024, + } + + start = time.monotonic() + response_payload = None + actual: list[dict[str, Any]] = [] + + try: + response = client.chat.completions.create(**request_payload) + response_payload = response.model_dump() + actual = extract_tool_calls(response) + except Exception as exc: + latency = (time.monotonic() - start) * 1000 + log_path = evaluator.save_test_log( + run_dir, + test_id=test_id, + category=category, + model=model, + parser=parser, + backend=backend, + request_payload=request_payload, + response_payload=None, + ground_truth=case.ground_truth, + actual_tool_calls=[], + passed=False, + errors=[f"API error: {exc}"], + latency_ms=latency, + ) + append_result( + { + "test_id": test_id, + "category": category, + "passed": False, + "errors": [f"API error: {exc}"], + "latency_ms": latency, + "finish_reason": None, + "completion_tokens": None, + "had_reasoning": False, + "log_file": log_file_for_summary(run_dir, log_path), + "model": model, + "backend": backend, + } + ) + pytest.fail(f"BFCL {test_id}: API call failed — {exc}") + + latency = (time.monotonic() - start) * 1000 + passed, errors = evaluator.evaluate_tool_calls( + actual, + case.ground_truth, + category=category, + ) + + log_path = evaluator.save_test_log( + run_dir, + test_id=test_id, + category=category, + model=model, + parser=parser, + backend=backend, + request_payload=request_payload, + response_payload=response_payload, + ground_truth=case.ground_truth, + actual_tool_calls=actual, + passed=passed, + errors=errors, + latency_ms=latency, + ) + + finish_reason = None + completion_tokens = None + had_reasoning = False + if response_payload: + choices = response_payload.get("choices") or [] + if choices: + finish_reason = choices[0].get("finish_reason") + msg = choices[0].get("message") or {} + had_reasoning = bool(msg.get("reasoning_content")) + usage = response_payload.get("usage") or {} + completion_tokens = usage.get("completion_tokens") + + append_result( + { + "test_id": test_id, + "category": category, + "passed": passed, + "errors": errors, + "latency_ms": latency, + "finish_reason": finish_reason, + "completion_tokens": completion_tokens, + "had_reasoning": had_reasoning, + "log_file": log_file_for_summary(run_dir, log_path), + "model": model, + "backend": backend, + } + ) + + status = "PASS" if passed else "FAIL" + logger.info( + "BFCL %s [%s] %.0fms → %s", + test_id, + status, + latency, + log_file_for_summary(run_dir, log_path), + ) + + if not passed: + pytest.fail(f"BFCL {test_id}: {'; '.join(errors)}") + + +# ============================================================================ +# Fixture-based tests (use setup_backend → launches GPU worker + gateway) +# ============================================================================ + + +@pytest.mark.e2e +@pytest.mark.skip_for_runtime( + "trtllm", reason="TRT-LLM does not support guided decoding (json_schema)" +) +@pytest.mark.model("Qwen/Qwen2.5-7B-Instruct") +@pytest.mark.gateway(extra_args=["--tool-call-parser", "qwen", "--history-backend", "memory"]) +@pytest.mark.parametrize("setup_backend", ["grpc"], indirect=True) +class TestBFCLQwen: + """BFCL v3 accuracy — Qwen 2.5 7B with qwen parser (all categories).""" + + def test_case(self, setup_backend, bfcl_run_dir, case): + backend_name, model, client, _ = setup_backend + _run_bfcl_case( + case=case, + model=model, + parser="qwen", + backend=backend_name, + client=client, + run_dir=bfcl_run_dir, + ) + + +# ============================================================================ +# Standalone mode: run against an already-running gateway +# ============================================================================ + + +@pytest.mark.e2e +@pytest.mark.skipif( + not os.environ.get("BFCL_BASE_URL"), + reason="Set BFCL_BASE_URL to run standalone BFCL tests", +) +class TestBFCLStandalone: + """Run BFCL tests against an externally managed gateway. + + Set environment variables: + BFCL_BASE_URL=http://localhost:30000 + BFCL_MODEL=Qwen/Qwen2.5-7B-Instruct + BFCL_PARSER=qwen (for logging only) + BFCL_LIMIT=20 (optional: limit cases per category) + """ + + @pytest.fixture(autouse=True) + def _setup_client(self, bfcl_run_dir): + base_url = os.environ["BFCL_BASE_URL"] + self.model = os.environ.get("BFCL_MODEL", "default") + self.parser = os.environ.get("BFCL_PARSER", "unknown") + self.client = openai.OpenAI( + base_url=f"{base_url.rstrip('/')}/v1", + api_key=os.environ.get("BFCL_API_KEY", "not-used"), + ) + self.run_dir = bfcl_run_dir + + def test_case(self, case): + _run_bfcl_case( + case=case, + model=self.model, + parser=self.parser, + backend="standalone", + client=self.client, + run_dir=self.run_dir, + ) diff --git a/e2e_test/conftest.py b/e2e_test/conftest.py index 4a9d33fa80..b76aa35b93 100644 --- a/e2e_test/conftest.py +++ b/e2e_test/conftest.py @@ -37,18 +37,25 @@ _ROOT = Path(__file__).resolve().parents[1] # smg/ _E2E_TEST = Path(__file__).resolve().parent # e2e_test/ -_SRC = _ROOT / "bindings" / "python" +_SRC = _ROOT / "bindings" / "python" / "src" +_CLIENTS_PYTHON = _ROOT / "clients" / "python" # Add e2e_test to path so "from infra import ..." works if str(_E2E_TEST) not in sys.path: sys.path.insert(0, str(_E2E_TEST)) -# Add bindings/python to path if the wheel is not installed (for local development) -_wheel_installed = find_spec("smg.smg_rs") is not None +# Add bindings/python/src to path if the wheel is not installed (for local development) +try: + _wheel_installed = find_spec("smg.smg_rs") is not None +except ModuleNotFoundError: + _wheel_installed = False if not _wheel_installed and str(_SRC) not in sys.path: sys.path.insert(0, str(_SRC)) +if str(_CLIENTS_PYTHON) not in sys.path: + sys.path.insert(0, str(_CLIENTS_PYTHON)) + # --------------------------------------------------------------------------- # Logging setup (clean output without pytest's "---- live log ----" dividers) @@ -108,9 +115,9 @@ def pytest_runtest_logstart(nodeid: str, location: tuple) -> None: pytest_collection_modifyitems, pytest_configure, pytest_runtest_setup, + pytest_sessionfinish, setup_backend, ) -from smg_client import SmgClient # --------------------------------------------------------------------------- # Fixtures @@ -137,6 +144,8 @@ def api_client(request, setup_backend): if param == "openai": yield openai_client elif param == "smg": + from smg_client import SmgClient + client = SmgClient(base_url=gateway.base_url, max_retries=0) yield client client.close() @@ -151,6 +160,7 @@ def api_client(request, setup_backend): "pytest_runtest_setup", "pytest_collection_modifyitems", "pytest_configure", + "pytest_sessionfinish", # Fixtures "setup_backend", "backend_router", diff --git a/e2e_test/fixtures/__init__.py b/e2e_test/fixtures/__init__.py index 291915a2a8..3cfc8ae102 100644 --- a/e2e_test/fixtures/__init__.py +++ b/e2e_test/fixtures/__init__.py @@ -12,6 +12,7 @@ pytest_collection_modifyitems, pytest_configure, pytest_runtest_setup, + pytest_sessionfinish, ) # Marker helpers @@ -25,6 +26,7 @@ "pytest_collection_modifyitems", "pytest_configure", "pytest_runtest_setup", + "pytest_sessionfinish", # Backend fixtures "setup_backend", "backend_router", diff --git a/e2e_test/fixtures/hooks.py b/e2e_test/fixtures/hooks.py index ca15e00276..88d3a299f7 100644 --- a/e2e_test/fixtures/hooks.py +++ b/e2e_test/fixtures/hooks.py @@ -7,6 +7,7 @@ from __future__ import annotations +import logging import os import pytest @@ -14,6 +15,8 @@ from .markers import resolve_class_marker +logger = logging.getLogger(__name__) + # --------------------------------------------------------------------------- # Marker registration # --------------------------------------------------------------------------- @@ -81,6 +84,21 @@ def pytest_configure(config: pytest.Config) -> None: # --------------------------------------------------------------------------- +def pytest_sessionfinish(session: pytest.Session, exitstatus: int) -> None: + """Write BFCL summary at session end. + + Backend teardown is handled directly by the current setup fixtures, so the + session hook only needs to emit the BFCL summary. + """ + + from bfcl.session_state import write_summary_if_needed + + try: + write_summary_if_needed() + except Exception: + logger.warning("Failed to write BFCL summary", exc_info=True) + + def pytest_runtest_setup(item: pytest.Item) -> None: """Skip tests marked with ``@pytest.mark.skip_for_runtime``.""" marker = item.get_closest_marker("skip_for_runtime")