feat: enable custom LLM support for Crew.test()

- Added new llm parameter to Crew.test() that accepts string or LLM instance - Maintained backward compatibility with openai_model_name parameter - Updated CrewEvaluator to handle any LLM implementation - Added comprehensive test coverage Fixes #2081 Co-Authored-By: Joe Moura <joao@crewai.com>
2026-01-10 16:48:30 +00:00 · 2025-02-09 23:25:02 +00:00
parent d6d98ee969
commit f838909220
3 changed files with 100 additions and 9 deletions
--- a/src/crewai/crew.py
+++ b/src/crewai/crew.py
@@ -1148,19 +1148,31 @@ class Crew(BaseModel):
    def test(
        self,
        n_iterations: int,
+        llm: Optional[Union[str, LLM]] = None,
        openai_model_name: Optional[str] = None,
        inputs: Optional[Dict[str, Any]] = None,
    ) -> None:
-        """Test and evaluate the Crew with the given inputs for n iterations concurrently using concurrent.futures."""
+        """Test and evaluate the Crew with the given inputs for n iterations concurrently.
+        
+        Args:
+            n_iterations: Number of test iterations to run
+            llm: LLM instance or model name string to use for evaluation
+            openai_model_name: (Deprecated) OpenAI model name string (kept for backward compatibility)
+            inputs: Optional dictionary of inputs to pass to each test iteration
+        """
        test_crew = self.copy()
+        model = llm or openai_model_name
+
+        if model is None:
+            raise ValueError("Either llm or openai_model_name must be provided")

        self._test_execution_span = test_crew._telemetry.test_execution_span(
            test_crew,
            n_iterations,
            inputs,
-            openai_model_name,  # type: ignore[arg-type]
-        )  # type: ignore[arg-type]
-        evaluator = CrewEvaluator(test_crew, openai_model_name)  # type: ignore[arg-type]
+            str(model) if isinstance(model, LLM) else model,
+        )
+        evaluator = CrewEvaluator(test_crew, model)

        for i in range(1, n_iterations + 1):
            evaluator.set_iteration(i)