Skip to content

Commit aa190ef

Browse files
starfolkbotclaude
andcommitted
fix(crewai): add provider metadata, route tools to metadata, drop eager serialization
Align the CrewAI integration with .agents/skills/sdk-integrations/SKILL.md. - Add metadata.provider on crewai.llm spans, read from LLM.provider on the event source. Every llm span must carry both metadata.model AND metadata.provider per the instrumentation spec. - Route tool definitions into metadata.tools (both LLMCallStartedEvent tools and AgentExecutionStartedEvent tools). Tools no longer leak into input. - Drop eager _try_to_dict / _normalize_output / _normalize_tools passes in kickoff/task/agent/llm/tool output paths. bt_json at log time already handles Pydantic v2/v1 and dataclasses, so the eager pass is wasted work. - Narrow _agent_metadata's agent_llm read: previously ``str(llm)`` was a fallback, which dumps the pydantic repr including api_key / api_base / client_params. Now the allowlist reads only the model name. - Extend the existing direct-event tests: assert metadata.provider on the kickoff/llm tree and add a positive+negative check that tools route to metadata (not input). No new cassettes — the file docstring documents why VCR is impractical for CrewAI. Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
1 parent a2989d3 commit aa190ef

2 files changed

Lines changed: 87 additions & 39 deletions

File tree

‎py/src/braintrust/integrations/crewai/test_crewai.py‎

Lines changed: 38 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -250,8 +250,23 @@ def test_kickoff_llm_event_tree_parents_and_shape(memory_logger):
250250

251251
kickoff = _build_kickoff_started()
252252
_emit(kickoff)
253+
254+
# Emit with a real CrewAI LLM as source so ``metadata.provider`` gets
255+
# populated from ``LLM.provider`` — the spec requires every llm span
256+
# to carry provider metadata.
257+
from crewai import LLM
258+
from crewai.events.event_bus import crewai_event_bus
259+
260+
llm_source = LLM(model="gpt-4o-mini")
253261
llm_started = _build_llm_started(parent_event_id=kickoff.event_id)
254-
_emit(llm_started)
262+
future = crewai_event_bus.emit(llm_source, llm_started)
263+
if future is not None:
264+
try:
265+
future.result(timeout=5.0)
266+
except Exception:
267+
pass
268+
_flush_event_bus(crewai_event_bus, timeout=5.0)
269+
255270
_emit(_build_llm_completed(llm_started, usage={"prompt_tokens": 3, "completion_tokens": 5, "total_tokens": 8}))
256271
_emit(_build_kickoff_completed(kickoff))
257272

@@ -277,12 +292,34 @@ def test_kickoff_llm_event_tree_parents_and_shape(memory_logger):
277292
# Shape assertions.
278293
assert llm_span["input"]["messages"] == llm_started.messages
279294
assert llm_span["metadata"]["model"] == "gpt-4o-mini"
295+
# Every llm span must carry ``metadata.provider`` per the instrumentation spec.
296+
assert llm_span["metadata"]["provider"] == "openai"
280297
assert llm_span["metadata"]["call_id"] == "call-1"
281298
assert llm_span["output"] == "4"
282299
assert kickoff_span["output"] == "final answer"
283300
assert kickoff_span["input"] == kickoff.inputs
284301

285302

303+
def test_llm_tools_route_to_metadata_not_input(memory_logger):
304+
"""Tool definitions belong in ``metadata.tools`` per the spec, not in ``input``."""
305+
patch_crewai()
306+
307+
tools = [
308+
{"type": "function", "function": {"name": "search", "description": "search the web"}},
309+
{"type": "function", "function": {"name": "sum", "description": "add two numbers"}},
310+
]
311+
started = _build_llm_started(tools=tools)
312+
_emit(started)
313+
_emit(_build_llm_completed(started))
314+
315+
span = memory_logger.pop()[0]
316+
assert span["span_attributes"]["name"] == "crewai.llm"
317+
# Positive: tools land in metadata untouched (no eager serialization).
318+
assert span["metadata"]["tools"] == tools
319+
# Negative: tools MUST NOT leak into input.
320+
assert "tools" not in span["input"], span["input"]
321+
322+
286323
def test_llm_tokens_skipped_when_litellm_patched(memory_logger, monkeypatch):
287324
"""Leaf-only rule: LiteLLM patched -> no token metrics on crewai.llm."""
288325
# Pretend LiteLLM is patched without actually wrapping the module, so

‎py/src/braintrust/integrations/crewai/tracing.py‎

Lines changed: 49 additions & 38 deletions
Original file line numberDiff line numberDiff line change
@@ -45,7 +45,6 @@
4545
from braintrust.integrations.utils import (
4646
_normalize_chat_messages,
4747
_parse_openai_usage_metrics,
48-
_try_to_dict,
4948
)
5049
from braintrust.logger import NOOP_SPAN, Span, current_span
5150
from braintrust.logger import start_span as _bt_start_span
@@ -127,7 +126,12 @@ def _litellm_owns_leaf_span() -> bool:
127126

128127

129128
def _agent_metadata(agent: Any) -> dict[str, Any]:
130-
"""Extract identity + configuration metadata from a CrewAI agent object."""
129+
"""Extract identity + configuration metadata from a CrewAI agent object.
130+
131+
Only attributes on the explicit allowlist below are read, so an SDK-level
132+
change that adds new fields (potentially containing API keys or other
133+
secrets) does not silently leak into span metadata.
134+
"""
131135
if agent is None:
132136
return {}
133137
meta: dict[str, Any] = {}
@@ -137,7 +141,12 @@ def _agent_metadata(agent: Any) -> dict[str, Any]:
137141
meta[f"agent_{attr}"] = str(value) if attr == "id" else value
138142
llm = getattr(agent, "llm", None)
139143
if llm is not None:
140-
meta["agent_llm"] = getattr(llm, "model", None) or str(llm)
144+
# Only read the model name off the LLM object. Falling back to
145+
# ``str(llm)`` would dump the full pydantic repr, which on CrewAI's
146+
# ``LLM`` includes ``api_key`` / ``api_base`` / ``client_params``.
147+
model_name = getattr(llm, "model", None)
148+
if model_name:
149+
meta["agent_llm"] = model_name
141150
return meta
142151

143152

@@ -198,32 +207,20 @@ def _llm_config_metadata(source: Any) -> dict[str, Any]:
198207
return meta
199208

200209

201-
def _normalize_tools(tools: Any) -> Any:
202-
"""Normalize a list of CrewAI tool descriptors for logging."""
203-
if tools is None:
204-
return None
205-
if isinstance(tools, (list, tuple)):
206-
out = []
207-
for tool in tools:
208-
if isinstance(tool, dict):
209-
out.append(tool)
210-
else:
211-
out.append(_try_to_dict(tool))
212-
return out
213-
return _try_to_dict(tools)
214-
215-
216-
def _normalize_output(value: Any) -> Any:
217-
"""Coerce provider-owned output objects into plain dicts/strings for logging."""
218-
if value is None:
210+
def _provider_from_source(source: Any) -> str | None:
211+
"""Best-effort extraction of the underlying provider for a CrewAI LLM call.
212+
213+
CrewAI's :class:`LLM` object exposes a ``provider`` string
214+
(``"openai"``, ``"anthropic"``, ``"bedrock"``, ...). Every ``llm`` span
215+
must carry ``metadata.provider`` per the instrumentation spec, so we
216+
pull it directly from the emitting source when possible.
217+
"""
218+
if source is None:
219219
return None
220-
if isinstance(value, (str, int, float, bool, dict, list)):
221-
return value
222-
coerced = _try_to_dict(value)
223-
if coerced is value and not isinstance(value, (str, int, float, bool, dict, list)):
224-
# Last-resort: render as a string rather than log a raw SDK object.
225-
return str(value)
226-
return coerced
220+
provider = getattr(source, "provider", None)
221+
if isinstance(provider, str) and provider:
222+
return provider
223+
return None
227224

228225

229226
# ---------------------------------------------------------------------------
@@ -366,7 +363,10 @@ def on_crew_kickoff_started(source: Any, event: CrewKickoffStartedEvent) -> None
366363

367364
@crewai_event_bus.on(CrewKickoffCompletedEvent)
368365
def on_crew_kickoff_completed(_source: Any, event: CrewKickoffCompletedEvent) -> None:
369-
self._end_span(event, output=_normalize_output(getattr(event, "output", None)))
366+
# Pass provider objects through untouched — ``bt_json`` at log time
367+
# handles Pydantic models, dataclasses, and stringly fallbacks, so
368+
# eagerly ``model_dump``-ing here would just be wasted work.
369+
self._end_span(event, output=getattr(event, "output", None))
370370
# Kickoff is the outermost scope; drop any orphan entries left over
371371
# when an inner end-event was never delivered so state does not grow
372372
# unbounded in long-running services.
@@ -399,7 +399,7 @@ def on_task_started(source: Any, event: TaskStartedEvent) -> None:
399399

400400
@crewai_event_bus.on(TaskCompletedEvent)
401401
def on_task_completed(_source: Any, event: TaskCompletedEvent) -> None:
402-
self._end_span(event, output=_normalize_output(getattr(event, "output", None)))
402+
self._end_span(event, output=getattr(event, "output", None))
403403

404404
@crewai_event_bus.on(TaskFailedEvent)
405405
def on_task_failed(_source: Any, event: TaskFailedEvent) -> None:
@@ -418,7 +418,9 @@ def on_agent_started(_source: Any, event: AgentExecutionStartedEvent) -> None:
418418
**_task_metadata(task),
419419
**_causal_metadata(event),
420420
}
421-
metadata["tools"] = _normalize_tools(getattr(event, "tools", None))
421+
agent_tools = getattr(event, "tools", None)
422+
if agent_tools:
423+
metadata["tools"] = agent_tools
422424
self._open_span(
423425
event,
424426
name="crewai.agent",
@@ -447,18 +449,27 @@ def on_llm_started(source: Any, event: LLMCallStartedEvent) -> None:
447449
}
448450
if getattr(event, "model", None):
449451
metadata["model"] = event.model
452+
# Every ``llm`` span must carry ``metadata.provider`` per the
453+
# instrumentation spec. CrewAI ``LLM`` objects expose ``.provider``
454+
# directly (``openai`` / ``anthropic`` / ``bedrock`` / ...).
455+
provider = _provider_from_source(source)
456+
if provider:
457+
metadata["provider"] = provider
450458
if getattr(event, "call_id", None):
451459
metadata["call_id"] = event.call_id
452460
available_functions = getattr(event, "available_functions", None)
453461
if available_functions:
462+
# Only the callable *names* — never the callables themselves.
454463
metadata["available_functions"] = list(available_functions)
455-
span_input: dict[str, Any] = {
456-
"messages": getattr(event, "messages", None),
457-
}
464+
# Tool definitions belong in ``metadata.tools``, not ``input``,
465+
# per the spec. ``bt_json`` at log time handles Pydantic models
466+
# and dataclasses, so pass through without eager serialization.
458467
tools = getattr(event, "tools", None)
459468
if tools:
460-
span_input["tools"] = _normalize_tools(tools)
461-
span_input["messages"] = _normalize_chat_messages(span_input["messages"])
469+
metadata["tools"] = tools
470+
span_input: dict[str, Any] = {
471+
"messages": _normalize_chat_messages(getattr(event, "messages", None)),
472+
}
462473
self._open_span(
463474
event,
464475
name="crewai.llm",
@@ -497,7 +508,7 @@ def on_llm_completed(_source: Any, event: LLMCallCompletedEvent) -> None:
497508

498509
self._end_span(
499510
event,
500-
output=_normalize_output(getattr(event, "response", None)),
511+
output=getattr(event, "response", None),
501512
metadata=metadata or None,
502513
extra_metrics=extra_metrics,
503514
)
@@ -540,7 +551,7 @@ def on_tool_finished(_source: Any, event: ToolUsageFinishedEvent) -> None:
540551
extra_metadata["run_attempts"] = run_attempts
541552
self._end_span(
542553
event,
543-
output=_normalize_output(getattr(event, "output", None)),
554+
output=getattr(event, "output", None),
544555
metadata=extra_metadata or None,
545556
)
546557

0 commit comments

Comments
 (0)