feat(cli): crewai eval evaluates the last traced run through AMP

`crewai eval` reads `.crewai/last_run.json` — the record crewAI writes
when a traced run's spans reach Wharf — and asks AMP to evaluate that
run: POST /crewai_plus/api/v1/tracing/evaluations with the execution id,
sending the saved `crewai login` when there is one and nothing otherwise.
AMP answers with an evaluation id and a URL; the command prints the URL,
opens it, waits for the verdict and prints it (goal gate and the four
grades), exit 1 only when the evaluation itself failed. `--run
EXECUTION_ID` evaluates another run.

With no traced run recorded it offers to turn tracing on for the project
(`CREWAI_TRACING_ENABLED=true` in .env, set_key so nothing else in the
file moves) and run the crew now with `crewai run`; without a terminal, or
declined, it prints the three steps instead. A run that leaves no record
behind is explained, never guessed at.

AMP's refusals are printed in its own words: a run that needs an account
(401 account_required), a refused credential (then `crewai login`), a run
AMP does not hold (404), rate limiting (429 with Retry-After), and any
other status with AMP's message. The two AMP calls live on the CLI's
PlusAPI subclass, so no crewai-core release is needed.

Who may evaluate what is AMP's decision, not the command's: an anonymous
run once without an account, then it needs one; a run traced while logged
in for that organization's members; a deployment execution for members
who may see its traces.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
Joao Moura
2026-09-20 00:53:52 -07:00
parent 0374c63129
commit 8640cda042
5 changed files with 497 additions and 1 deletions

View File

@@ -48,6 +48,12 @@ def run_crew(*args: Any, **kwargs: Any) -> Any:
return _run_crew(*args, **kwargs)
def eval_crew(*args: Any, **kwargs: Any) -> Any:
from crewai_cli.eval_crew import eval_crew as _eval_crew
return _eval_crew(*args, **kwargs)
if TYPE_CHECKING:
# mypy sees the real classes; at runtime the shims below defer the
# heavy imports until a command actually instantiates them.
@@ -674,6 +680,23 @@ def run(
)
@crewai.command(name="eval")
@click.option(
"--run",
"run_id",
type=str,
default=None,
metavar="EXECUTION_ID",
help=(
"Evaluate this traced run instead of the last one. The execution id "
"crewAI recorded for the run."
),
)
def eval_command(run_id: str | None) -> None:
"""Evaluate the last traced run through CrewAI AMP."""
eval_crew(run_id=run_id)
@crewai.command()
def update() -> None:
"""Update the pyproject.toml of the Crew project to use uv."""

View File

@@ -0,0 +1,218 @@
"""`crewai eval`: evaluate the last traced run through CrewAI AMP.
crewAI records a traced run in `.crewai/last_run.json` when the run's spans
reach Wharf. This command reads that record (or takes `--run EXECUTION_ID`),
asks AMP to evaluate the run, prints and opens the URL AMP answers with,
waits for the verdict and prints it. With no traced run recorded it offers
to turn tracing on for the project and run the crew now.
Who may evaluate what is AMP's decision: an anonymous run once without an
account, then it needs one; a run traced while logged in for that
organization's members; a deployment execution for members who may see its
traces. The command sends the saved `crewai login` when there is one.
"""
from __future__ import annotations
import contextlib
import json
import os
from pathlib import Path
import sys
import time
from typing import Any
import webbrowser
import click
from dotenv import set_key
import httpx
from rich.console import Console
from crewai_cli.authentication.token import get_auth_token
from crewai_cli.plus_api import PlusAPI
from crewai_cli.utils import get_or_create_project_id, is_dmn_mode_enabled
console = Console()
LAST_RUN_FILE = Path(".crewai") / "last_run.json"
TRACING_ENV_VAR = "CREWAI_TRACING_ENABLED"
POLL_SECONDS = 3.0
FINISHED = {"done", "failed"}
def eval_crew(run_id: str | None = None) -> None:
"""Evaluate the last traced run of this project, or the run RUN_ID."""
get_or_create_project_id()
execution_id = run_id or last_run_id()
if execution_id is None:
execution_id = _run_now_or_explain()
client = PlusAPI(api_key=saved_login())
started = _start_evaluation(client, execution_id)
url = started.get("url")
console.print(f"Evaluating run [bold]{execution_id}[/bold]")
if url:
console.print(f"Follow it at [cyan underline]{url}[/cyan underline]")
_open(url)
finished = _wait(client, str(started["id"]), url)
_print_verdict(finished, url)
if finished.get("status") != "done":
raise SystemExit(1)
def last_run_id(directory: Path | None = None) -> str | None:
"""The execution id crewAI recorded for the project's last traced run."""
path = (directory or Path.cwd()) / LAST_RUN_FILE
try:
loaded = json.loads(path.read_text(encoding="utf-8"))
except (OSError, ValueError):
return None
if not isinstance(loaded, dict):
return None
execution_id = loaded.get("execution_id")
return str(execution_id) if execution_id else None
def saved_login() -> str | None:
"""The `crewai login` token, or None: AMP then treats the caller as anonymous."""
try:
return get_auth_token()
except Exception:
return None
def _run_now_or_explain() -> str:
"""No traced run recorded here: offer to turn tracing on and run the crew now."""
steps = (
"No traced run is recorded in this project. Turn tracing on and run the crew, "
f"then come back:\n 1. add {TRACING_ENV_VAR}=true to .env\n 2. crewai run\n 3. crewai eval"
)
if is_dmn_mode_enabled() or not sys.stdin.isatty():
console.print(steps, style="yellow")
raise SystemExit(1)
if not click.confirm(
"No traced run is recorded in this project. Turn tracing on and run the crew now?",
default=True,
):
console.print(steps, style="yellow")
raise SystemExit(0)
_enable_tracing()
from crewai_cli.run_crew import run_crew
run_crew()
execution_id = last_run_id()
if execution_id is None:
console.print(
"The run finished but no trace was recorded: the run may have failed, or sharing "
"the trace was declined. Run the crew again and accept when asked, then `crewai eval`.",
style="bold red",
)
raise SystemExit(1)
return execution_id
def _enable_tracing() -> None:
"""`CREWAI_TRACING_ENABLED=true` in the project's .env, and in this process for the run about to start."""
env_file = Path.cwd() / ".env"
env_file.touch(exist_ok=True)
set_key(str(env_file), TRACING_ENV_VAR, "true", quote_mode="never")
os.environ[TRACING_ENV_VAR] = "true"
console.print(
f"Tracing is on for this project ({TRACING_ENV_VAR}=true in .env).",
style="green",
)
def _start_evaluation(client: PlusAPI, execution_id: str) -> dict[str, Any]:
response = client.create_evaluation(execution_id)
if response.status_code in (200, 202):
payload = _payload(response)
if payload and payload.get("id"):
return payload
_fail(f"AMP answered without an evaluation id ({response.status_code}).")
_refused(response, execution_id)
raise AssertionError("unreachable")
def _wait(client: PlusAPI, evaluation_id: str, url: str | None) -> dict[str, Any]:
"""Poll until the evaluation is done or failed; Ctrl-C leaves it running."""
console.print("Waiting for the verdict…", style="dim")
try:
while True:
response = client.get_evaluation(evaluation_id)
if response.status_code != 200:
_refused(response, evaluation_id)
payload = _payload(response) or {}
if payload.get("status") in FINISHED:
return payload
time.sleep(POLL_SECONDS)
except KeyboardInterrupt:
where = f" at {url}" if url else ""
console.print(f"\nStill running{where}.", style="yellow")
raise SystemExit(130) from None
def _print_verdict(finished: dict[str, Any], url: str | None) -> None:
if finished.get("status") != "done":
console.print(
f"Evaluation failed: {finished.get('error') or 'no reason given'}",
style="bold red",
)
return
verdict = finished.get("verdict") or {}
gate = str(verdict.get("gate") or "inconclusive").upper()
style = {"PASSED": "bold green", "FAILED": "bold red"}.get(gate, "bold yellow")
grades = verdict.get("grades") or {}
parts = [
f"{area} {grades[area]}/5"
if grades.get(area) is not None
else f"{area} not measured"
for area in ("goal", "quality", "process", "cost")
]
console.print(f"Goal gate: [{style}]{gate}[/{style}] · " + " · ".join(parts))
if url:
console.print(f"Full report: {url}")
def _open(url: str) -> None:
if is_dmn_mode_enabled():
return
with contextlib.suppress(Exception): # no browser is not an error
webbrowser.open(url)
def _payload(response: httpx.Response) -> dict[str, Any] | None:
try:
loaded = response.json()
except ValueError:
return None
return loaded if isinstance(loaded, dict) else None
def _refused(response: httpx.Response, subject: str) -> None:
"""AMP's own words when it sent them, then exit 1."""
payload = _payload(response) or {}
message = str(payload.get("message") or "").strip()
error = str(payload.get("error") or "")
if response.status_code in (401, 403):
if error == "account_required" and message:
_fail(message)
_fail(
f"{message or 'AMP refused the credential'}. Log in with `crewai login` and try again."
)
if response.status_code == 404:
_fail(message or f"AMP holds no run {subject}.")
if response.status_code == 429:
retry = response.headers.get("Retry-After")
_fail(
f"{message or 'AMP is rate limiting this request'}{f' — retry after {retry}s' if retry else ''}."
)
_fail(f"AMP answered {response.status_code}{': ' + message if message else ''}.")
def _fail(message: str) -> None:
console.print(message, style="bold red")
raise SystemExit(1)

View File

@@ -18,9 +18,22 @@ class PlusAPI(_CorePlusAPI):
The ZIP deployment methods live here as well as in newer crewai-core
versions so editable CLI installs still work when an older crewai-core is
present in the runtime environment.
present in the runtime environment. The evaluation methods live here
because only the CLI calls them.
"""
EVALUATIONS_RESOURCE = f"{_CorePlusAPI.TRACING_RESOURCE}/evaluations"
def create_evaluation(self, execution_id: str) -> httpx.Response:
"""Ask AMP to evaluate the traced run EXECUTION_ID (crewai eval)."""
return self._make_request(
"POST", self.EVALUATIONS_RESOURCE, json={"execution_id": execution_id}
)
def get_evaluation(self, evaluation_id: str) -> httpx.Response:
"""The evaluation's status and, once done, its verdict."""
return self._make_request("GET", f"{self.EVALUATIONS_RESOURCE}/{evaluation_id}")
def _make_multipart_request(
self,
method: HttpMethod,

View File

@@ -0,0 +1,216 @@
"""`crewai eval`: the last traced run, evaluated through AMP."""
from __future__ import annotations
import json
from pathlib import Path
from click.testing import CliRunner
import httpx
import pytest
from crewai_cli import eval_crew as eval_module
from crewai_cli.cli import eval_command
EXECUTION_ID = "6f31fe1a-20bd-4bfe-a011-25d6b9341f62"
URL = "https://evolve.crewai.test/e/ev-1"
class FakeAMP:
"""A PlusAPI double: scripted answers, calls recorded."""
def __init__(self, create=None, statuses=None):
self.create = create if create is not None else httpx.Response(
202, json={"id": "ev-1", "url": URL, "status": "queued"}
)
self.statuses = list(statuses or [])
self.calls: list[tuple] = []
self.api_key = None
def create_evaluation(self, execution_id):
self.calls.append(("create", execution_id))
return self.create
def get_evaluation(self, evaluation_id):
self.calls.append(("get", evaluation_id))
return self.statuses.pop(0) if self.statuses else httpx.Response(200, json={"id": evaluation_id, "status": "running"})
def done(gate="passed", grades=None):
return httpx.Response(200, json={
"id": "ev-1", "status": "done", "url": URL,
"verdict": {"gate": gate, "grades": grades if grades is not None else {"goal": 5, "quality": 4, "process": 5, "cost": None}},
})
@pytest.fixture
def project(tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
monkeypatch.setattr(eval_module, "get_or_create_project_id", lambda: None)
monkeypatch.setattr(eval_module, "saved_login", lambda: "login-token")
monkeypatch.setattr(eval_module.time, "sleep", lambda seconds: None)
monkeypatch.setattr(eval_module, "is_dmn_mode_enabled", lambda: False)
opened: list[str] = []
monkeypatch.setattr(eval_module.webbrowser, "open", lambda url: opened.append(url) or True)
return tmp_path, opened
def record_last_run(directory: Path, execution_id: str = EXECUTION_ID) -> None:
(directory / ".crewai").mkdir(exist_ok=True)
(directory / ".crewai" / "last_run.json").write_text(json.dumps({"execution_id": execution_id, "tier": "ephemeral"}))
def install(monkeypatch, amp: FakeAMP) -> FakeAMP:
monkeypatch.setattr(eval_module, "PlusAPI", lambda api_key=None: (setattr(amp, "api_key", api_key), amp)[1])
return amp
def test_the_last_run_is_evaluated_the_url_opened_and_the_verdict_printed(project, monkeypatch, capsys):
directory, opened = project
record_last_run(directory)
amp = install(monkeypatch, FakeAMP(statuses=[httpx.Response(200, json={"id": "ev-1", "status": "running"}), done()]))
eval_module.eval_crew()
out = capsys.readouterr().out
assert amp.api_key == "login-token"
assert amp.calls == [("create", EXECUTION_ID), ("get", "ev-1"), ("get", "ev-1")]
assert opened == [URL]
assert EXECUTION_ID in out and URL in out
assert "Goal gate: PASSED" in out and "goal 5/5" in out and "cost not measured" in out
def test_run_names_another_execution_and_an_anonymous_caller_sends_no_token(project, monkeypatch, capsys):
directory, _ = project
record_last_run(directory)
monkeypatch.setattr(eval_module, "saved_login", lambda: None)
amp = install(monkeypatch, FakeAMP(statuses=[done("failed")]))
eval_module.eval_crew(run_id="other-run")
assert amp.api_key is None
assert amp.calls[0] == ("create", "other-run")
assert "Goal gate: FAILED" in capsys.readouterr().out
def test_a_failed_evaluation_exits_one_with_amps_reason(project, monkeypatch, capsys):
directory, _ = project
record_last_run(directory)
install(monkeypatch, FakeAMP(statuses=[httpx.Response(200, json={"id": "ev-1", "status": "failed", "error": "the judge was unreachable"})]))
with pytest.raises(SystemExit) as exit_:
eval_module.eval_crew()
assert exit_.value.code == 1
assert "Evaluation failed: the judge was unreachable" in capsys.readouterr().out
@pytest.mark.parametrize(
("response", "expected"),
[
(httpx.Response(401, json={"error": "account_required", "message": "Execution x was already read once without an account. Log in with `crewai login`, or create an account, to read it again."}),
"already read once without an account"),
(httpx.Response(401, json={"error": "bad_credentials", "message": "Bad credentials"}), "Bad credentials. Log in with `crewai login`"),
(httpx.Response(404, json={"error": "trace_not_found", "message": "No spans recorded for execution 6f31"}), "No spans recorded for execution 6f31"),
(httpx.Response(429, json={"error": "rate_limit_exceeded", "message": "Too many requests"}, headers={"Retry-After": "60"}), "Too many requests — retry after 60s"),
(httpx.Response(503, json={"error": "service_unavailable", "message": "Wharf could not list the spans"}), "AMP answered 503: Wharf could not list the spans"),
(httpx.Response(500, text="boom"), "AMP answered 500."),
],
)
def test_amps_refusals_are_printed_in_its_words_and_exit_one(project, monkeypatch, capsys, response, expected):
directory, opened = project
record_last_run(directory)
install(monkeypatch, FakeAMP(create=response))
with pytest.raises(SystemExit) as exit_:
eval_module.eval_crew()
assert exit_.value.code == 1
assert expected in capsys.readouterr().out
assert opened == []
def test_without_a_traced_run_and_no_terminal_it_explains_and_exits(project, monkeypatch, capsys):
monkeypatch.setattr(eval_module, "is_dmn_mode_enabled", lambda: True)
amp = install(monkeypatch, FakeAMP())
with pytest.raises(SystemExit) as exit_:
eval_module.eval_crew()
assert exit_.value.code == 1
out = capsys.readouterr().out
assert "No traced run is recorded" in out and "CREWAI_TRACING_ENABLED=true" in out and "crewai run" in out
assert amp.calls == []
def test_without_a_traced_run_it_offers_to_turn_tracing_on_and_run_the_crew(project, monkeypatch, capsys):
directory, _ = project
monkeypatch.setattr(eval_module.sys.stdin, "isatty", lambda: True)
monkeypatch.setattr(eval_module.click, "confirm", lambda *args, **kwargs: True)
ran: list[str] = []
def fake_run_crew() -> None:
ran.append("run")
assert eval_module.os.environ.get("CREWAI_TRACING_ENABLED") == "true"
record_last_run(directory, "fresh-run")
import crewai_cli.run_crew as run_crew_module
monkeypatch.setattr(run_crew_module, "run_crew", fake_run_crew)
monkeypatch.delenv("CREWAI_TRACING_ENABLED", raising=False)
amp = install(monkeypatch, FakeAMP(statuses=[done()]))
eval_module.eval_crew()
assert ran == ["run"]
assert "CREWAI_TRACING_ENABLED=true" in (directory / ".env").read_text()
assert amp.calls[0] == ("create", "fresh-run")
assert "Tracing is on for this project" in capsys.readouterr().out
def test_declining_the_offer_exits_cleanly_with_the_steps(project, monkeypatch, capsys):
monkeypatch.setattr(eval_module.sys.stdin, "isatty", lambda: True)
monkeypatch.setattr(eval_module.click, "confirm", lambda *args, **kwargs: False)
amp = install(monkeypatch, FakeAMP())
with pytest.raises(SystemExit) as exit_:
eval_module.eval_crew()
assert exit_.value.code == 0
assert "crewai eval" in capsys.readouterr().out and amp.calls == []
def test_a_run_that_leaves_no_trace_behind_is_explained(project, monkeypatch, capsys):
monkeypatch.setattr(eval_module.sys.stdin, "isatty", lambda: True)
monkeypatch.setattr(eval_module.click, "confirm", lambda *args, **kwargs: True)
import crewai_cli.run_crew as run_crew_module
monkeypatch.setattr(run_crew_module, "run_crew", lambda: None)
install(monkeypatch, FakeAMP())
with pytest.raises(SystemExit) as exit_:
eval_module.eval_crew()
assert exit_.value.code == 1
assert "no trace was recorded" in capsys.readouterr().out
def test_the_cli_command_maps_to_the_implementation(monkeypatch):
calls = []
monkeypatch.setattr("crewai_cli.cli.eval_crew", lambda **kwargs: calls.append(kwargs))
runner = CliRunner()
assert runner.invoke(eval_command, []).exit_code == 0
assert runner.invoke(eval_command, ["--run", EXECUTION_ID]).exit_code == 0
assert calls == [{"run_id": None}, {"run_id": EXECUTION_ID}]
assert "Evaluate the last traced run" in runner.invoke(eval_command, ["--help"]).output
def test_last_run_id_reads_the_record_crewai_writes(tmp_path):
assert eval_module.last_run_id(tmp_path) is None
(tmp_path / ".crewai").mkdir()
(tmp_path / ".crewai" / "last_run.json").write_text("not json")
assert eval_module.last_run_id(tmp_path) is None
record_last_run(tmp_path)
assert eval_module.last_run_id(tmp_path) == EXECUTION_ID

View File

@@ -18,6 +18,32 @@ class TestPlusAPI(unittest.TestCase):
self.assertIn("CrewAI-CLI/", self.api.headers["User-Agent"])
self.assertTrue(self.api.headers["X-Crewai-Version"])
@patch("crewai_core.plus_api.PlusAPI._make_request")
def test_create_evaluation(self, mock_make_request):
mock_response = MagicMock()
mock_make_request.return_value = mock_response
response = self.api.create_evaluation("6f31fe1a-20bd-4bfe-a011-25d6b9341f62")
mock_make_request.assert_called_once_with(
"POST",
"/crewai_plus/api/v1/tracing/evaluations",
json={"execution_id": "6f31fe1a-20bd-4bfe-a011-25d6b9341f62"},
)
self.assertEqual(response, mock_response)
@patch("crewai_core.plus_api.PlusAPI._make_request")
def test_get_evaluation(self, mock_make_request):
mock_response = MagicMock()
mock_make_request.return_value = mock_response
response = self.api.get_evaluation("ev-1")
mock_make_request.assert_called_once_with(
"GET", "/crewai_plus/api/v1/tracing/evaluations/ev-1"
)
self.assertEqual(response, mock_response)
@patch("crewai_core.plus_api.PlusAPI._make_request")
def test_login_to_tool_repository(self, mock_make_request):
mock_response = MagicMock()