Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

33 changes: 33 additions & 0 deletions py/src/braintrust/integrations/openai/test_openai.py
Original file line number Diff line number Diff line change
Expand Up @@ -93,6 +93,15 @@ def _supports_agents_api() -> bool:
return True


def _supports_responses_access_programs() -> bool:
try:
from openai.resources.responses import Responses

return "access_programs" in inspect.signature(Responses.create).parameters
except (ImportError, AttributeError, ValueError):
return False


@pytest.mark.vcr
def test_openai_chat_metrics(memory_logger):
assert not memory_logger.pop()
Expand Down Expand Up @@ -189,6 +198,11 @@ def test_openai_responses_metrics(memory_logger):
assert TEST_MODEL in span["metadata"]["model"]
assert span["metadata"]["provider"] == "openai"
assert span["metadata"]["instructions"] == "Just the number please"
if hasattr(response, "access_programs"):
# Current OpenAI cassettes include the field with a null value. Older
# provider pins do not expose it; neither shape should add null metadata.
assert response.access_programs is None
assert "access_programs" not in span["metadata"]
assert TEST_PROMPT in str(span["input"])
assert len(span["output"]) > 0
span_output_text = span["output"][0]["content"][0]["text"]
Expand Down Expand Up @@ -239,6 +253,25 @@ class NumberAnswer(BaseModel):
assert span["output"][0]["content"][0]["parsed"]["reasoning"] == parse_response.output_parsed.reasoning


@pytest.mark.vcr
def test_openai_responses_access_programs(memory_logger):
if not _supports_responses_access_programs():
pytest.skip("Responses access_programs is not available in this SDK version")

access_programs = {"cyber": "standard"}
client = wrap_openai(openai.OpenAI())
response = client.responses.create(
model=TEST_MODEL,
input="Say hello in one word.",
access_programs=access_programs,
)

spans = memory_logger.pop()
assert len(spans) == 1
assert _try_to_dict(response.access_programs) == access_programs
assert _try_to_dict(spans[0]["metadata"]["access_programs"]) == access_programs


@pytest.mark.vcr
def test_openai_chat_completion_inline_moderation_metadata(memory_logger):
if not _supports_inline_moderation():
Expand Down
42 changes: 38 additions & 4 deletions py/src/braintrust/integrations/openai/tracing.py
Original file line number Diff line number Diff line change
Expand Up @@ -105,6 +105,7 @@ def start_span(*args, **kwargs):
"conversation",
"verbosity",
"moderation",
"access_programs",
)

_RESPONSES_RESULT_METADATA_KEYS = (
Expand All @@ -114,6 +115,7 @@ def start_span(*args, **kwargs):
"created_at",
"moderation",
"service_tier",
"access_programs",
"object",
"background",
"incomplete_details",
Expand Down Expand Up @@ -1183,6 +1185,10 @@ def _log_response_tool_spans(output: Any, *, parent_export: str | None) -> None:
"command_execution": ("command", "cwd"),
"mcp_call": ("arguments",),
"web_search_call": ("action",),
"computer_use_call": ("title",),
"browser_authentication_request": ("request_id",),
"computer_use_approval_request": ("request_id", "request"),
"computer_use_approval_request_result": ("request_id",),
Comment on lines +1189 to +1191

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P2 Badge Preserve the approval payloads in tool spans

When computer use requests browser authentication, the actual computer_use_approval_request item carries a request object and the corresponding result carries a credential-free response describing submit/cancel and the selected option. These allowlists retain only request_id, while _agent_tool_span_output has no output keys for either type, so the emitted spans discard both the form being requested and the user's admitted response. Include request/response in the appropriate span data and cover the real item shapes with a cassette-backed test.

AGENTS.md reference: AGENTS.md:L176-L182

Useful? React with 👍 / 👎.

"create_subagent_call": ("content", "model", "reasoning_effort"),
"send_subagent_input_call": ("content", "recipient_agent_id"),
"resume_subagent_call": ("recipient_agent_id",),
Expand All @@ -1194,6 +1200,7 @@ def _log_response_tool_spans(output: Any, *, parent_export: str | None) -> None:
_AGENT_TOOL_ITEM_OUTPUT_KEYS = {
"command_execution": ("output", "exit_code", "duration_ms"),
"mcp_call": ("output",),
"computer_use_approval_request_result": ("response",),
}

_AGENT_TURN_EVENTS = {
Expand Down Expand Up @@ -1272,7 +1279,13 @@ def _agent_tool_span_name(item: Any) -> str:


def _agent_tool_span_data(item: Any, keys: tuple[str, ...]) -> Any:
values = clean_nones({key: getattr(item, key, None) for key in keys})
values = clean_nones(
{
key: _try_to_dict(value) if key in {"request", "response"} else value
for key in keys
if (value := getattr(item, key, None)) is not None
}
)
if not values:
return None
if keys == ("arguments",):
Expand All @@ -1286,7 +1299,7 @@ def _agent_tool_span_metadata(item: Any) -> dict[str, Any]:
"tool_type": item.type,
"tool_id": item.id,
"call_id": getattr(item, "call_id", None),
"status": item.status,
"status": getattr(item, "status", None),
"turn_id": item.turn_id,
"server_label": getattr(item, "server_label", None),
"agent_id": getattr(item, "agent_id", None),
Expand All @@ -1299,7 +1312,7 @@ def _agent_tool_span_error(item: Any) -> Any:
error = getattr(item, "error", None)
if error is not None:
return error
if item.status == "failed":
if getattr(item, "status", None) == "failed":
return "Agent tool call failed"
return None

Expand Down Expand Up @@ -1383,7 +1396,7 @@ def _observe_completed_item(self, item: Any) -> None:
error = _agent_tool_span_error(item)
if error is not None:
tool_span.log(error=error)
output = _agent_tool_span_data(item, _AGENT_TOOL_ITEM_OUTPUT_KEYS.get(item.type, ()))
output = _agent_tool_span_output(item)
if output is not None:
tool_span.log(output=output)

Expand Down Expand Up @@ -1412,6 +1425,27 @@ def finish(self) -> None:
self.span.end()


def _agent_tool_span_output(item: Any) -> Any:
if item.type == "computer_use_call":
output = getattr(item, "output", None)
screenshot = getattr(output, "image_url", None)
if screenshot is None and isinstance(output, Mapping):
screenshot = output.get("image_url")
if screenshot is not None:
resolved = _materialize_attachment(screenshot, prefix="computer_use_screenshot")
if resolved is not None:
output_data = _try_to_dict(output)
if isinstance(output_data, dict):
return {
**output_data,
"image_url": resolved.multimodal_part_payload["image_url"],
}
return resolved.multimodal_part_payload
return output

return _agent_tool_span_data(item, _AGENT_TOOL_ITEM_OUTPUT_KEYS.get(item.type, ()))


class AgentSessionWrapper:
def __init__(self, create_fn: Callable[..., Any] | None, acreate_fn: Callable[..., Any] | None) -> None:
self.create_fn = create_fn
Expand Down
Loading