From 8640cda0424e9bea97f48379b16145846d011bc6 Mon Sep 17 00:00:00 2001 From: Joao Moura Date: Sun, 20 Sep 2026 00:53:52 -0700 Subject: [PATCH] feat(cli): crewai eval evaluates the last traced run through AMP MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `crewai eval` reads `.crewai/last_run.json` — the record crewAI writes when a traced run's spans reach Wharf — and asks AMP to evaluate that run: POST /crewai_plus/api/v1/tracing/evaluations with the execution id, sending the saved `crewai login` when there is one and nothing otherwise. AMP answers with an evaluation id and a URL; the command prints the URL, opens it, waits for the verdict and prints it (goal gate and the four grades), exit 1 only when the evaluation itself failed. `--run EXECUTION_ID` evaluates another run. With no traced run recorded it offers to turn tracing on for the project (`CREWAI_TRACING_ENABLED=true` in .env, set_key so nothing else in the file moves) and run the crew now with `crewai run`; without a terminal, or declined, it prints the three steps instead. A run that leaves no record behind is explained, never guessed at. AMP's refusals are printed in its own words: a run that needs an account (401 account_required), a refused credential (then `crewai login`), a run AMP does not hold (404), rate limiting (429 with Retry-After), and any other status with AMP's message. The two AMP calls live on the CLI's PlusAPI subclass, so no crewai-core release is needed. Who may evaluate what is AMP's decision, not the command's: an anonymous run once without an account, then it needs one; a run traced while logged in for that organization's members; a deployment execution for members who may see its traces. Co-Authored-By: Claude Fable 5.1 --- lib/cli/src/crewai_cli/cli.py | 23 +++ lib/cli/src/crewai_cli/eval_crew.py | 218 ++++++++++++++++++++++++++++ lib/cli/src/crewai_cli/plus_api.py | 15 +- lib/cli/tests/test_eval_crew.py | 216 +++++++++++++++++++++++++++ lib/cli/tests/test_plus_api.py | 26 ++++ 5 files changed, 497 insertions(+), 1 deletion(-) create mode 100644 lib/cli/src/crewai_cli/eval_crew.py create mode 100644 lib/cli/tests/test_eval_crew.py diff --git a/lib/cli/src/crewai_cli/cli.py b/lib/cli/src/crewai_cli/cli.py index 4c922f84c..0250289d3 100644 --- a/lib/cli/src/crewai_cli/cli.py +++ b/lib/cli/src/crewai_cli/cli.py @@ -48,6 +48,12 @@ def run_crew(*args: Any, **kwargs: Any) -> Any: return _run_crew(*args, **kwargs) +def eval_crew(*args: Any, **kwargs: Any) -> Any: + from crewai_cli.eval_crew import eval_crew as _eval_crew + + return _eval_crew(*args, **kwargs) + + if TYPE_CHECKING: # mypy sees the real classes; at runtime the shims below defer the # heavy imports until a command actually instantiates them. @@ -674,6 +680,23 @@ def run( ) +@crewai.command(name="eval") +@click.option( + "--run", + "run_id", + type=str, + default=None, + metavar="EXECUTION_ID", + help=( + "Evaluate this traced run instead of the last one. The execution id " + "crewAI recorded for the run." + ), +) +def eval_command(run_id: str | None) -> None: + """Evaluate the last traced run through CrewAI AMP.""" + eval_crew(run_id=run_id) + + @crewai.command() def update() -> None: """Update the pyproject.toml of the Crew project to use uv.""" diff --git a/lib/cli/src/crewai_cli/eval_crew.py b/lib/cli/src/crewai_cli/eval_crew.py new file mode 100644 index 000000000..56ec9e8e8 --- /dev/null +++ b/lib/cli/src/crewai_cli/eval_crew.py @@ -0,0 +1,218 @@ +"""`crewai eval`: evaluate the last traced run through CrewAI AMP. + +crewAI records a traced run in `.crewai/last_run.json` when the run's spans +reach Wharf. This command reads that record (or takes `--run EXECUTION_ID`), +asks AMP to evaluate the run, prints and opens the URL AMP answers with, +waits for the verdict and prints it. With no traced run recorded it offers +to turn tracing on for the project and run the crew now. + +Who may evaluate what is AMP's decision: an anonymous run once without an +account, then it needs one; a run traced while logged in for that +organization's members; a deployment execution for members who may see its +traces. The command sends the saved `crewai login` when there is one. +""" + +from __future__ import annotations + +import contextlib +import json +import os +from pathlib import Path +import sys +import time +from typing import Any +import webbrowser + +import click +from dotenv import set_key +import httpx +from rich.console import Console + +from crewai_cli.authentication.token import get_auth_token +from crewai_cli.plus_api import PlusAPI +from crewai_cli.utils import get_or_create_project_id, is_dmn_mode_enabled + + +console = Console() + +LAST_RUN_FILE = Path(".crewai") / "last_run.json" +TRACING_ENV_VAR = "CREWAI_TRACING_ENABLED" +POLL_SECONDS = 3.0 +FINISHED = {"done", "failed"} + + +def eval_crew(run_id: str | None = None) -> None: + """Evaluate the last traced run of this project, or the run RUN_ID.""" + get_or_create_project_id() + execution_id = run_id or last_run_id() + if execution_id is None: + execution_id = _run_now_or_explain() + + client = PlusAPI(api_key=saved_login()) + started = _start_evaluation(client, execution_id) + url = started.get("url") + console.print(f"Evaluating run [bold]{execution_id}[/bold]") + if url: + console.print(f"Follow it at [cyan underline]{url}[/cyan underline]") + _open(url) + + finished = _wait(client, str(started["id"]), url) + _print_verdict(finished, url) + if finished.get("status") != "done": + raise SystemExit(1) + + +def last_run_id(directory: Path | None = None) -> str | None: + """The execution id crewAI recorded for the project's last traced run.""" + path = (directory or Path.cwd()) / LAST_RUN_FILE + try: + loaded = json.loads(path.read_text(encoding="utf-8")) + except (OSError, ValueError): + return None + if not isinstance(loaded, dict): + return None + execution_id = loaded.get("execution_id") + return str(execution_id) if execution_id else None + + +def saved_login() -> str | None: + """The `crewai login` token, or None: AMP then treats the caller as anonymous.""" + try: + return get_auth_token() + except Exception: + return None + + +def _run_now_or_explain() -> str: + """No traced run recorded here: offer to turn tracing on and run the crew now.""" + steps = ( + "No traced run is recorded in this project. Turn tracing on and run the crew, " + f"then come back:\n 1. add {TRACING_ENV_VAR}=true to .env\n 2. crewai run\n 3. crewai eval" + ) + if is_dmn_mode_enabled() or not sys.stdin.isatty(): + console.print(steps, style="yellow") + raise SystemExit(1) + if not click.confirm( + "No traced run is recorded in this project. Turn tracing on and run the crew now?", + default=True, + ): + console.print(steps, style="yellow") + raise SystemExit(0) + + _enable_tracing() + from crewai_cli.run_crew import run_crew + + run_crew() + execution_id = last_run_id() + if execution_id is None: + console.print( + "The run finished but no trace was recorded: the run may have failed, or sharing " + "the trace was declined. Run the crew again and accept when asked, then `crewai eval`.", + style="bold red", + ) + raise SystemExit(1) + return execution_id + + +def _enable_tracing() -> None: + """`CREWAI_TRACING_ENABLED=true` in the project's .env, and in this process for the run about to start.""" + env_file = Path.cwd() / ".env" + env_file.touch(exist_ok=True) + set_key(str(env_file), TRACING_ENV_VAR, "true", quote_mode="never") + os.environ[TRACING_ENV_VAR] = "true" + console.print( + f"Tracing is on for this project ({TRACING_ENV_VAR}=true in .env).", + style="green", + ) + + +def _start_evaluation(client: PlusAPI, execution_id: str) -> dict[str, Any]: + response = client.create_evaluation(execution_id) + if response.status_code in (200, 202): + payload = _payload(response) + if payload and payload.get("id"): + return payload + _fail(f"AMP answered without an evaluation id ({response.status_code}).") + _refused(response, execution_id) + raise AssertionError("unreachable") + + +def _wait(client: PlusAPI, evaluation_id: str, url: str | None) -> dict[str, Any]: + """Poll until the evaluation is done or failed; Ctrl-C leaves it running.""" + console.print("Waiting for the verdict…", style="dim") + try: + while True: + response = client.get_evaluation(evaluation_id) + if response.status_code != 200: + _refused(response, evaluation_id) + payload = _payload(response) or {} + if payload.get("status") in FINISHED: + return payload + time.sleep(POLL_SECONDS) + except KeyboardInterrupt: + where = f" at {url}" if url else "" + console.print(f"\nStill running{where}.", style="yellow") + raise SystemExit(130) from None + + +def _print_verdict(finished: dict[str, Any], url: str | None) -> None: + if finished.get("status") != "done": + console.print( + f"Evaluation failed: {finished.get('error') or 'no reason given'}", + style="bold red", + ) + return + verdict = finished.get("verdict") or {} + gate = str(verdict.get("gate") or "inconclusive").upper() + style = {"PASSED": "bold green", "FAILED": "bold red"}.get(gate, "bold yellow") + grades = verdict.get("grades") or {} + parts = [ + f"{area} {grades[area]}/5" + if grades.get(area) is not None + else f"{area} not measured" + for area in ("goal", "quality", "process", "cost") + ] + console.print(f"Goal gate: [{style}]{gate}[/{style}] · " + " · ".join(parts)) + if url: + console.print(f"Full report: {url}") + + +def _open(url: str) -> None: + if is_dmn_mode_enabled(): + return + with contextlib.suppress(Exception): # no browser is not an error + webbrowser.open(url) + + +def _payload(response: httpx.Response) -> dict[str, Any] | None: + try: + loaded = response.json() + except ValueError: + return None + return loaded if isinstance(loaded, dict) else None + + +def _refused(response: httpx.Response, subject: str) -> None: + """AMP's own words when it sent them, then exit 1.""" + payload = _payload(response) or {} + message = str(payload.get("message") or "").strip() + error = str(payload.get("error") or "") + if response.status_code in (401, 403): + if error == "account_required" and message: + _fail(message) + _fail( + f"{message or 'AMP refused the credential'}. Log in with `crewai login` and try again." + ) + if response.status_code == 404: + _fail(message or f"AMP holds no run {subject}.") + if response.status_code == 429: + retry = response.headers.get("Retry-After") + _fail( + f"{message or 'AMP is rate limiting this request'}{f' — retry after {retry}s' if retry else ''}." + ) + _fail(f"AMP answered {response.status_code}{': ' + message if message else ''}.") + + +def _fail(message: str) -> None: + console.print(message, style="bold red") + raise SystemExit(1) diff --git a/lib/cli/src/crewai_cli/plus_api.py b/lib/cli/src/crewai_cli/plus_api.py index 6f94d96d3..ea3733d8b 100644 --- a/lib/cli/src/crewai_cli/plus_api.py +++ b/lib/cli/src/crewai_cli/plus_api.py @@ -18,9 +18,22 @@ class PlusAPI(_CorePlusAPI): The ZIP deployment methods live here as well as in newer crewai-core versions so editable CLI installs still work when an older crewai-core is - present in the runtime environment. + present in the runtime environment. The evaluation methods live here + because only the CLI calls them. """ + EVALUATIONS_RESOURCE = f"{_CorePlusAPI.TRACING_RESOURCE}/evaluations" + + def create_evaluation(self, execution_id: str) -> httpx.Response: + """Ask AMP to evaluate the traced run EXECUTION_ID (crewai eval).""" + return self._make_request( + "POST", self.EVALUATIONS_RESOURCE, json={"execution_id": execution_id} + ) + + def get_evaluation(self, evaluation_id: str) -> httpx.Response: + """The evaluation's status and, once done, its verdict.""" + return self._make_request("GET", f"{self.EVALUATIONS_RESOURCE}/{evaluation_id}") + def _make_multipart_request( self, method: HttpMethod, diff --git a/lib/cli/tests/test_eval_crew.py b/lib/cli/tests/test_eval_crew.py new file mode 100644 index 000000000..b0b50109b --- /dev/null +++ b/lib/cli/tests/test_eval_crew.py @@ -0,0 +1,216 @@ +"""`crewai eval`: the last traced run, evaluated through AMP.""" + +from __future__ import annotations + +import json +from pathlib import Path + +from click.testing import CliRunner +import httpx +import pytest + +from crewai_cli import eval_crew as eval_module +from crewai_cli.cli import eval_command + + +EXECUTION_ID = "6f31fe1a-20bd-4bfe-a011-25d6b9341f62" +URL = "https://evolve.crewai.test/e/ev-1" + + +class FakeAMP: + """A PlusAPI double: scripted answers, calls recorded.""" + + def __init__(self, create=None, statuses=None): + self.create = create if create is not None else httpx.Response( + 202, json={"id": "ev-1", "url": URL, "status": "queued"} + ) + self.statuses = list(statuses or []) + self.calls: list[tuple] = [] + self.api_key = None + + def create_evaluation(self, execution_id): + self.calls.append(("create", execution_id)) + return self.create + + def get_evaluation(self, evaluation_id): + self.calls.append(("get", evaluation_id)) + return self.statuses.pop(0) if self.statuses else httpx.Response(200, json={"id": evaluation_id, "status": "running"}) + + +def done(gate="passed", grades=None): + return httpx.Response(200, json={ + "id": "ev-1", "status": "done", "url": URL, + "verdict": {"gate": gate, "grades": grades if grades is not None else {"goal": 5, "quality": 4, "process": 5, "cost": None}}, + }) + + +@pytest.fixture +def project(tmp_path, monkeypatch): + monkeypatch.chdir(tmp_path) + monkeypatch.setattr(eval_module, "get_or_create_project_id", lambda: None) + monkeypatch.setattr(eval_module, "saved_login", lambda: "login-token") + monkeypatch.setattr(eval_module.time, "sleep", lambda seconds: None) + monkeypatch.setattr(eval_module, "is_dmn_mode_enabled", lambda: False) + opened: list[str] = [] + monkeypatch.setattr(eval_module.webbrowser, "open", lambda url: opened.append(url) or True) + return tmp_path, opened + + +def record_last_run(directory: Path, execution_id: str = EXECUTION_ID) -> None: + (directory / ".crewai").mkdir(exist_ok=True) + (directory / ".crewai" / "last_run.json").write_text(json.dumps({"execution_id": execution_id, "tier": "ephemeral"})) + + +def install(monkeypatch, amp: FakeAMP) -> FakeAMP: + monkeypatch.setattr(eval_module, "PlusAPI", lambda api_key=None: (setattr(amp, "api_key", api_key), amp)[1]) + return amp + + +def test_the_last_run_is_evaluated_the_url_opened_and_the_verdict_printed(project, monkeypatch, capsys): + directory, opened = project + record_last_run(directory) + amp = install(monkeypatch, FakeAMP(statuses=[httpx.Response(200, json={"id": "ev-1", "status": "running"}), done()])) + + eval_module.eval_crew() + + out = capsys.readouterr().out + assert amp.api_key == "login-token" + assert amp.calls == [("create", EXECUTION_ID), ("get", "ev-1"), ("get", "ev-1")] + assert opened == [URL] + assert EXECUTION_ID in out and URL in out + assert "Goal gate: PASSED" in out and "goal 5/5" in out and "cost not measured" in out + + +def test_run_names_another_execution_and_an_anonymous_caller_sends_no_token(project, monkeypatch, capsys): + directory, _ = project + record_last_run(directory) + monkeypatch.setattr(eval_module, "saved_login", lambda: None) + amp = install(monkeypatch, FakeAMP(statuses=[done("failed")])) + + eval_module.eval_crew(run_id="other-run") + + assert amp.api_key is None + assert amp.calls[0] == ("create", "other-run") + assert "Goal gate: FAILED" in capsys.readouterr().out + + +def test_a_failed_evaluation_exits_one_with_amps_reason(project, monkeypatch, capsys): + directory, _ = project + record_last_run(directory) + install(monkeypatch, FakeAMP(statuses=[httpx.Response(200, json={"id": "ev-1", "status": "failed", "error": "the judge was unreachable"})])) + + with pytest.raises(SystemExit) as exit_: + eval_module.eval_crew() + + assert exit_.value.code == 1 + assert "Evaluation failed: the judge was unreachable" in capsys.readouterr().out + + +@pytest.mark.parametrize( + ("response", "expected"), + [ + (httpx.Response(401, json={"error": "account_required", "message": "Execution x was already read once without an account. Log in with `crewai login`, or create an account, to read it again."}), + "already read once without an account"), + (httpx.Response(401, json={"error": "bad_credentials", "message": "Bad credentials"}), "Bad credentials. Log in with `crewai login`"), + (httpx.Response(404, json={"error": "trace_not_found", "message": "No spans recorded for execution 6f31"}), "No spans recorded for execution 6f31"), + (httpx.Response(429, json={"error": "rate_limit_exceeded", "message": "Too many requests"}, headers={"Retry-After": "60"}), "Too many requests — retry after 60s"), + (httpx.Response(503, json={"error": "service_unavailable", "message": "Wharf could not list the spans"}), "AMP answered 503: Wharf could not list the spans"), + (httpx.Response(500, text="boom"), "AMP answered 500."), + ], +) +def test_amps_refusals_are_printed_in_its_words_and_exit_one(project, monkeypatch, capsys, response, expected): + directory, opened = project + record_last_run(directory) + install(monkeypatch, FakeAMP(create=response)) + + with pytest.raises(SystemExit) as exit_: + eval_module.eval_crew() + + assert exit_.value.code == 1 + assert expected in capsys.readouterr().out + assert opened == [] + + +def test_without_a_traced_run_and_no_terminal_it_explains_and_exits(project, monkeypatch, capsys): + monkeypatch.setattr(eval_module, "is_dmn_mode_enabled", lambda: True) + amp = install(monkeypatch, FakeAMP()) + + with pytest.raises(SystemExit) as exit_: + eval_module.eval_crew() + + assert exit_.value.code == 1 + out = capsys.readouterr().out + assert "No traced run is recorded" in out and "CREWAI_TRACING_ENABLED=true" in out and "crewai run" in out + assert amp.calls == [] + + +def test_without_a_traced_run_it_offers_to_turn_tracing_on_and_run_the_crew(project, monkeypatch, capsys): + directory, _ = project + monkeypatch.setattr(eval_module.sys.stdin, "isatty", lambda: True) + monkeypatch.setattr(eval_module.click, "confirm", lambda *args, **kwargs: True) + ran: list[str] = [] + + def fake_run_crew() -> None: + ran.append("run") + assert eval_module.os.environ.get("CREWAI_TRACING_ENABLED") == "true" + record_last_run(directory, "fresh-run") + + import crewai_cli.run_crew as run_crew_module + + monkeypatch.setattr(run_crew_module, "run_crew", fake_run_crew) + monkeypatch.delenv("CREWAI_TRACING_ENABLED", raising=False) + amp = install(monkeypatch, FakeAMP(statuses=[done()])) + + eval_module.eval_crew() + + assert ran == ["run"] + assert "CREWAI_TRACING_ENABLED=true" in (directory / ".env").read_text() + assert amp.calls[0] == ("create", "fresh-run") + assert "Tracing is on for this project" in capsys.readouterr().out + + +def test_declining_the_offer_exits_cleanly_with_the_steps(project, monkeypatch, capsys): + monkeypatch.setattr(eval_module.sys.stdin, "isatty", lambda: True) + monkeypatch.setattr(eval_module.click, "confirm", lambda *args, **kwargs: False) + amp = install(monkeypatch, FakeAMP()) + + with pytest.raises(SystemExit) as exit_: + eval_module.eval_crew() + + assert exit_.value.code == 0 + assert "crewai eval" in capsys.readouterr().out and amp.calls == [] + + +def test_a_run_that_leaves_no_trace_behind_is_explained(project, monkeypatch, capsys): + monkeypatch.setattr(eval_module.sys.stdin, "isatty", lambda: True) + monkeypatch.setattr(eval_module.click, "confirm", lambda *args, **kwargs: True) + import crewai_cli.run_crew as run_crew_module + + monkeypatch.setattr(run_crew_module, "run_crew", lambda: None) + install(monkeypatch, FakeAMP()) + + with pytest.raises(SystemExit) as exit_: + eval_module.eval_crew() + + assert exit_.value.code == 1 + assert "no trace was recorded" in capsys.readouterr().out + + +def test_the_cli_command_maps_to_the_implementation(monkeypatch): + calls = [] + monkeypatch.setattr("crewai_cli.cli.eval_crew", lambda **kwargs: calls.append(kwargs)) + runner = CliRunner() + + assert runner.invoke(eval_command, []).exit_code == 0 + assert runner.invoke(eval_command, ["--run", EXECUTION_ID]).exit_code == 0 + assert calls == [{"run_id": None}, {"run_id": EXECUTION_ID}] + assert "Evaluate the last traced run" in runner.invoke(eval_command, ["--help"]).output + + +def test_last_run_id_reads_the_record_crewai_writes(tmp_path): + assert eval_module.last_run_id(tmp_path) is None + (tmp_path / ".crewai").mkdir() + (tmp_path / ".crewai" / "last_run.json").write_text("not json") + assert eval_module.last_run_id(tmp_path) is None + record_last_run(tmp_path) + assert eval_module.last_run_id(tmp_path) == EXECUTION_ID diff --git a/lib/cli/tests/test_plus_api.py b/lib/cli/tests/test_plus_api.py index 16cf684d5..8f683a80f 100644 --- a/lib/cli/tests/test_plus_api.py +++ b/lib/cli/tests/test_plus_api.py @@ -18,6 +18,32 @@ class TestPlusAPI(unittest.TestCase): self.assertIn("CrewAI-CLI/", self.api.headers["User-Agent"]) self.assertTrue(self.api.headers["X-Crewai-Version"]) + @patch("crewai_core.plus_api.PlusAPI._make_request") + def test_create_evaluation(self, mock_make_request): + mock_response = MagicMock() + mock_make_request.return_value = mock_response + + response = self.api.create_evaluation("6f31fe1a-20bd-4bfe-a011-25d6b9341f62") + + mock_make_request.assert_called_once_with( + "POST", + "/crewai_plus/api/v1/tracing/evaluations", + json={"execution_id": "6f31fe1a-20bd-4bfe-a011-25d6b9341f62"}, + ) + self.assertEqual(response, mock_response) + + @patch("crewai_core.plus_api.PlusAPI._make_request") + def test_get_evaluation(self, mock_make_request): + mock_response = MagicMock() + mock_make_request.return_value = mock_response + + response = self.api.get_evaluation("ev-1") + + mock_make_request.assert_called_once_with( + "GET", "/crewai_plus/api/v1/tracing/evaluations/ev-1" + ) + self.assertEqual(response, mock_response) + @patch("crewai_core.plus_api.PlusAPI._make_request") def test_login_to_tool_repository(self, mock_make_request): mock_response = MagicMock()