diff --git a/README.md b/README.md index b2b75b5eb..e4f315564 100644 --- a/README.md +++ b/README.md @@ -26,12 +26,12 @@ All instrumentations use [opentelemetry-util-genai](./util/opentelemetry-util-ge | Instrumentation | Supported Package | Version | Status | | --------------- | ----------------- | ------- | ------ | -| [opentelemetry-instrumentation-genai-bedrock](./instrumentation/opentelemetry-instrumentation-genai-bedrock) | boto3 >= 1.40.46, < 2 | 1.2b0.dev | to be released | +| [opentelemetry-instrumentation-genai-bedrock](./instrumentation/opentelemetry-instrumentation-genai-bedrock) | boto3 >= 1.40.46, < 2 | 1.2b0 | to be released | | [opentelemetry-instrumentation-genai-claude-agent-sdk](./instrumentation/opentelemetry-instrumentation-genai-claude-agent-sdk) | claude-agent-sdk >= 0.1.14, < 1 | 1.2b0.dev | skeleton | | [opentelemetry-instrumentation-genai-crewai](./instrumentation/opentelemetry-instrumentation-genai-crewai) | crewai >= 1.10.1, < 2 | 1.2b0.dev | skeleton | -| [opentelemetry-instrumentation-genai-dspy](./instrumentation/opentelemetry-instrumentation-genai-dspy) | dspy >= 3.3.0, < 4 | 1.2b0.dev | to be released | -| [opentelemetry-instrumentation-genai-llama-index](./instrumentation/opentelemetry-instrumentation-genai-llama-index) | llama-index-core >= 0.14.19, < 1, llama-index-instrumentation >= 0.4.3, < 1, llama-index-workflows >= 2.17.1, != 2.24.0, < 3 | 1.2b0.dev | to be released | -| [opentelemetry-instrumentation-genai-portkey](./instrumentation/opentelemetry-instrumentation-genai-portkey) | portkey-ai >= 1.0.0, < 3 | 1.2b0.dev | to be released | +| [opentelemetry-instrumentation-genai-dspy](./instrumentation/opentelemetry-instrumentation-genai-dspy) | dspy >= 3.3.0, < 4 | 1.2b0 | to be released | +| [opentelemetry-instrumentation-genai-llama-index](./instrumentation/opentelemetry-instrumentation-genai-llama-index) | llama-index-core >= 0.14.19, < 1, llama-index-instrumentation >= 0.4.3, < 1, llama-index-workflows >= 2.17.1, != 2.24.0, < 3 | 1.2b0 | to be released | +| [opentelemetry-instrumentation-genai-portkey](./instrumentation/opentelemetry-instrumentation-genai-portkey) | portkey-ai >= 1.0.0, < 3 | 1.2b0 | to be released | | [opentelemetry-instrumentation-genai-weaviate-client](./instrumentation/opentelemetry-instrumentation-genai-weaviate-client) | weaviate-client >= 3.0.0, <5.0.0 | 1.2b0.dev | skeleton | diff --git a/instrumentation/opentelemetry-instrumentation-genai-openai/.changelog/786.added b/instrumentation/opentelemetry-instrumentation-genai-openai/.changelog/786.added new file mode 100644 index 000000000..a8ed66d81 --- /dev/null +++ b/instrumentation/opentelemetry-instrumentation-genai-openai/.changelog/786.added @@ -0,0 +1 @@ +Instrument `Responses.parse` and `AsyncResponses.parse`, so structured-output calls emit GenAI spans like `Responses.create`, reporting `gen_ai.output.type` as `json` for the `text_format` argument. diff --git a/instrumentation/opentelemetry-instrumentation-genai-openai/src/opentelemetry/instrumentation/genai/openai/__init__.py b/instrumentation/opentelemetry-instrumentation-genai-openai/src/opentelemetry/instrumentation/genai/openai/__init__.py index 7fc381429..dda69231a 100644 --- a/instrumentation/opentelemetry-instrumentation-genai-openai/src/opentelemetry/instrumentation/genai/openai/__init__.py +++ b/instrumentation/opentelemetry-instrumentation-genai-openai/src/opentelemetry/instrumentation/genai/openai/__init__.py @@ -103,9 +103,27 @@ def _is_parse_supported(): return False +def _is_responses_parse_supported(): + """Check if parse() is available on the Responses class. + + The Responses API structured-output helper ``parse()`` calls the SDK's + request path directly rather than delegating to the instrumented + ``Responses.create()``, so it must be wrapped separately (issue #659). + """ + try: + from openai.resources.responses.responses import ( # pylint: disable=import-outside-toplevel + Responses, + ) + + return hasattr(Responses, "parse") + except ImportError: + return False + + class OpenAIInstrumentor(BaseInstrumentor): def __init__(self): self._parse_supported = False + self._responses_parse_supported = False def instrumentation_dependencies(self) -> Collection[str]: return _instruments @@ -207,6 +225,25 @@ def _instrument(self, **kwargs): async_responses_retrieve(handler), ) + # parse() is the Responses API structured-output helper. Like + # chat.completions.parse it maps to the same inference operation + # as create() -- the telemetry-relevant request/response fields + # are identical and its ParsedResponse result is already handled + # by the create wrappers -- but it does not delegate to the + # instrumented create(), so it must be wrapped separately (#659). + self._responses_parse_supported = _is_responses_parse_supported() + if self._responses_parse_supported: + wrap_function_wrapper( + "openai.resources.responses.responses", + "Responses.parse", + responses_create(handler), + ) + wrap_function_wrapper( + "openai.resources.responses.responses", + "AsyncResponses.parse", + async_responses_create(handler), + ) + def _uninstrument(self, **kwargs): import openai # pylint: disable=import-outside-toplevel @@ -225,6 +262,9 @@ def _uninstrument(self, **kwargs): unwrap(responses_module.AsyncResponses, "stream") unwrap(responses_module.Responses, "retrieve") unwrap(responses_module.AsyncResponses, "retrieve") + if self._responses_parse_supported: + unwrap(responses_module.Responses, "parse") + unwrap(responses_module.AsyncResponses, "parse") def _get_responses_module(): diff --git a/instrumentation/opentelemetry-instrumentation-genai-openai/src/opentelemetry/instrumentation/genai/openai/response_extractors.py b/instrumentation/opentelemetry-instrumentation-genai-openai/src/opentelemetry/instrumentation/genai/openai/response_extractors.py index 8c0ff3793..f30e9b931 100644 --- a/instrumentation/opentelemetry-instrumentation-genai-openai/src/opentelemetry/instrumentation/genai/openai/response_extractors.py +++ b/instrumentation/opentelemetry-instrumentation-genai-openai/src/opentelemetry/instrumentation/genai/openai/response_extractors.py @@ -182,6 +182,21 @@ def _extract_output_type_from_value(text_config: object) -> str | None: return None +def _extract_output_type_from_text_format(text_format: object) -> str | None: + """Map ``Responses.parse(text_format=...)`` onto an output type. + + ``parse()`` takes the caller's Pydantic model (or dataclass) as + ``text_format`` and only turns it into a JSON-schema ``text.format`` + inside the SDK, after this wrapper has read the request kwargs, so the + ``text`` mapping alone misses it (issue #659). Mirror the + ``chat.completions.parse`` handling: a structured-output type means JSON. + Only the format metadata is recorded -- never the caller's schema. + """ + if isinstance(text_format, type): + return GenAIAttributes.GenAiOutputTypeValues.JSON.value + return None + + def _extract_conversation_id(conversation: object) -> str | None: """Return the conversation id the ``conversation`` parameter names.""" if isinstance(conversation, str): @@ -203,6 +218,7 @@ def extract_params( service_tier: str | None = None, temperature: float | None = None, text: object | None = None, + text_format: object | None = None, tools: Iterable[ToolParam] | None = None, top_p: float | None = None, **_kwargs: object, @@ -230,7 +246,10 @@ def extract_params( else None ), temperature=_get_float(temperature), - output_type=_extract_output_type_from_value(text), + output_type=( + _extract_output_type_from_value(text) + or _extract_output_type_from_text_format(text_format) + ), tools=_get_tools(tools), top_p=_get_float(top_p), ) diff --git a/instrumentation/opentelemetry-instrumentation-genai-openai/tests/cassettes/test_async_responses_parse_basic[content_mode0].yaml b/instrumentation/opentelemetry-instrumentation-genai-openai/tests/cassettes/test_async_responses_parse_basic[content_mode0].yaml new file mode 100644 index 000000000..d9040c980 --- /dev/null +++ b/instrumentation/opentelemetry-instrumentation-genai-openai/tests/cassettes/test_async_responses_parse_basic[content_mode0].yaml @@ -0,0 +1,143 @@ +interactions: +- request: + body: |- + { + "input": "Say this is a test", + "instructions": "You are a helpful assistant.", + "model": "gpt-4o-mini", + "stream": false, + "text": { + "format": { + "type": "json_schema", + "name": "_AsyncParseCalendarEvent", + "strict": true, + "schema": { + "properties": { + "name": { + "title": "Name", + "type": "string" + }, + "date": { + "title": "Date", + "type": "string" + }, + "participants": { + "items": { + "type": "string" + }, + "title": "Participants", + "type": "array" + } + }, + "required": [ + "name", + "date", + "participants" + ], + "title": "_AsyncParseCalendarEvent", + "type": "object", + "additionalProperties": false + } + } + } + } + headers: + Accept: + - application/json + Accept-Encoding: + - gzip, deflate + Connection: + - keep-alive + Content-Type: + - application/json + Host: + - api.openai.com + User-Agent: + - OpenAI/Python 1.109.1 + authorization: + - Bearer test_openai_api_key + method: POST + uri: https://api.openai.com/v1/responses + response: + body: + string: |- + { + "id": "resp_0f4faba17dcd0f1e0069e2f3e4907881909179832ba1237100", + "object": "response", + "created_at": 1776481253, + "status": "completed", + "background": false, + "error": null, + "frequency_penalty": 0.0, + "incomplete_details": null, + "instructions": "You are a helpful assistant.", + "max_output_tokens": null, + "max_tool_calls": null, + "model": "gpt-4o-mini-2024-07-18", + "output": [ + { + "id": "msg_0f4faba17dcd0f1e0069e2f3e7b2b88190bff23981628ac400", + "type": "message", + "status": "completed", + "content": [ + { + "type": "output_text", + "annotations": [], + "logprobs": [], + "text": "{\"name\":\"science fair\",\"date\":\"Friday\",\"participants\":[\"Alice\",\"Bob\"]}" + } + ], + "role": "assistant" + } + ], + "parallel_tool_calls": true, + "presence_penalty": 0.0, + "previous_response_id": null, + "reasoning": { + "effort": null, + "summary": null + }, + "safety_identifier": null, + "service_tier": "default", + "store": true, + "temperature": 1.0, + "text": { + "format": { + "type": "json_schema", + "name": "_AsyncParseCalendarEvent" + }, + "verbosity": "medium" + }, + "tool_choice": "auto", + "tools": [], + "top_logprobs": 0, + "top_p": 1.0, + "truncation": "disabled", + "usage": { + "input_tokens": 22, + "input_tokens_details": { + "cached_tokens": 0 + }, + "output_tokens": 6, + "output_tokens_details": { + "reasoning_tokens": 0 + }, + "total_tokens": 28 + }, + "user": null, + "metadata": {} + } + headers: + Content-Type: + - application/json + openai-organization: test_openai_org_id + openai-project: + - test_openai_project_id + openai-version: + - '2020-10-01' + set-cookie: + - test_set_cookie + status: + code: 200 + message: OK +version: 1 diff --git a/instrumentation/opentelemetry-instrumentation-genai-openai/tests/cassettes/test_responses_parse_basic[content_mode0].yaml b/instrumentation/opentelemetry-instrumentation-genai-openai/tests/cassettes/test_responses_parse_basic[content_mode0].yaml new file mode 100644 index 000000000..2e1fff217 --- /dev/null +++ b/instrumentation/opentelemetry-instrumentation-genai-openai/tests/cassettes/test_responses_parse_basic[content_mode0].yaml @@ -0,0 +1,143 @@ +interactions: +- request: + body: |- + { + "input": "Say this is a test", + "instructions": "You are a helpful assistant.", + "model": "gpt-4o-mini", + "stream": false, + "text": { + "format": { + "type": "json_schema", + "name": "_ParseCalendarEvent", + "strict": true, + "schema": { + "properties": { + "name": { + "title": "Name", + "type": "string" + }, + "date": { + "title": "Date", + "type": "string" + }, + "participants": { + "items": { + "type": "string" + }, + "title": "Participants", + "type": "array" + } + }, + "required": [ + "name", + "date", + "participants" + ], + "title": "_ParseCalendarEvent", + "type": "object", + "additionalProperties": false + } + } + } + } + headers: + Accept: + - application/json + Accept-Encoding: + - gzip, deflate + Connection: + - keep-alive + Content-Type: + - application/json + Host: + - api.openai.com + User-Agent: + - OpenAI/Python 1.109.1 + authorization: + - Bearer test_openai_api_key + method: POST + uri: https://api.openai.com/v1/responses + response: + body: + string: |- + { + "id": "resp_0f4faba17dcd0f1e0069e2f3e4907881909179832ba1237099", + "object": "response", + "created_at": 1776481253, + "status": "completed", + "background": false, + "error": null, + "frequency_penalty": 0.0, + "incomplete_details": null, + "instructions": "You are a helpful assistant.", + "max_output_tokens": null, + "max_tool_calls": null, + "model": "gpt-4o-mini-2024-07-18", + "output": [ + { + "id": "msg_0f4faba17dcd0f1e0069e2f3e7b2b88190bff23981628ac399", + "type": "message", + "status": "completed", + "content": [ + { + "type": "output_text", + "annotations": [], + "logprobs": [], + "text": "{\"name\":\"science fair\",\"date\":\"Friday\",\"participants\":[\"Alice\",\"Bob\"]}" + } + ], + "role": "assistant" + } + ], + "parallel_tool_calls": true, + "presence_penalty": 0.0, + "previous_response_id": null, + "reasoning": { + "effort": null, + "summary": null + }, + "safety_identifier": null, + "service_tier": "default", + "store": true, + "temperature": 1.0, + "text": { + "format": { + "type": "json_schema", + "name": "_ParseCalendarEvent" + }, + "verbosity": "medium" + }, + "tool_choice": "auto", + "tools": [], + "top_logprobs": 0, + "top_p": 1.0, + "truncation": "disabled", + "usage": { + "input_tokens": 22, + "input_tokens_details": { + "cached_tokens": 0 + }, + "output_tokens": 6, + "output_tokens_details": { + "reasoning_tokens": 0 + }, + "total_tokens": 28 + }, + "user": null, + "metadata": {} + } + headers: + Content-Type: + - application/json + openai-organization: test_openai_org_id + openai-project: + - test_openai_project_id + openai-version: + - '2020-10-01' + set-cookie: + - test_set_cookie + status: + code: 200 + message: OK +version: 1 diff --git a/instrumentation/opentelemetry-instrumentation-genai-openai/tests/test_async_responses.py b/instrumentation/opentelemetry-instrumentation-genai-openai/tests/test_async_responses.py index 9380b9129..863faede9 100644 --- a/instrumentation/opentelemetry-instrumentation-genai-openai/tests/test_async_responses.py +++ b/instrumentation/opentelemetry-instrumentation-genai-openai/tests/test_async_responses.py @@ -91,6 +91,10 @@ not HAS_RESPONSES_API, reason="Responses API requires a newer openai SDK" ) +_HAS_RESPONSES_PARSE = HAS_RESPONSES_API and hasattr( + _responses_module.AsyncResponses, "parse" +) + SYSTEM_INSTRUCTIONS = "You are a helpful assistant." EXPECTED_SYSTEM_INSTRUCTIONS = [ { @@ -1598,3 +1602,53 @@ async def test_async_responses_create_event_only_no_content_in_span( logs[0].log_record.event_name == "gen_ai.client.inference.operation.details" ) + + +class _AsyncParseCalendarEvent(BaseModel): + name: str + date: str + participants: list[str] + + +@pytest.mark.skipif( + not _HAS_RESPONSES_PARSE, + reason="AsyncResponses.parse requires a newer openai SDK", +) +@pytest.mark.asyncio() +async def test_async_responses_parse_basic( + span_exporter, async_openai_client, instrument_no_content, vcr +): + """AsyncResponses.parse() emits a GenAI span like create() (#659).""" + _skip_if_not_latest() + + with vcr.use_cassette( + "test_async_responses_parse_basic[content_mode0].yaml" + ): + response = await async_openai_client.responses.parse( + model=DEFAULT_MODEL, + instructions=SYSTEM_INSTRUCTIONS, + input=USER_ONLY_PROMPT[0]["content"], + text_format=_AsyncParseCalendarEvent, + stream=False, + ) + + (span,) = span_exporter.get_finished_spans() + assert_all_attributes( + span, + DEFAULT_MODEL, + True, + response.id, + response.model, + response.usage.input_tokens, + response.usage.output_tokens, + response_service_tier=getattr(response, "service_tier", None), + ) + assert ( + span.attributes[OpenAIAttributes.OPENAI_API_TYPE] + == OpenAIAttributes.OpenaiApiTypeValues.RESPONSES.value + ) + # parse(text_format=...) is a structured-output call, so the span records + # the JSON output type -- but only the format metadata, never the caller's + # Pydantic schema (issue #659). + _assert_request_attrs(span, output_type="json") + assert "_AsyncParseCalendarEvent" not in str(span.attributes) diff --git a/instrumentation/opentelemetry-instrumentation-genai-openai/tests/test_responses.py b/instrumentation/opentelemetry-instrumentation-genai-openai/tests/test_responses.py index 42ba6e3a2..da519bb70 100644 --- a/instrumentation/opentelemetry-instrumentation-genai-openai/tests/test_responses.py +++ b/instrumentation/opentelemetry-instrumentation-genai-openai/tests/test_responses.py @@ -90,6 +90,8 @@ not HAS_RESPONSES_API, reason="Responses API requires a newer openai SDK" ) +_HAS_RESPONSES_PARSE = HAS_RESPONSES_API and hasattr(_Responses, "parse") + SYSTEM_INSTRUCTIONS = "You are a helpful assistant." EXPECTED_SYSTEM_INSTRUCTIONS = [ { @@ -1595,3 +1597,127 @@ def test_responses_create_event_only_no_content_in_span( logs[0].log_record.event_name == "gen_ai.client.inference.operation.details" ) + + +class _ParseCalendarEvent(BaseModel): + name: str + date: str + participants: list[str] + + +@pytest.mark.skipif( + not _HAS_RESPONSES_PARSE, + reason="Responses.parse requires a newer openai SDK", +) +def test_responses_parse_basic( + span_exporter, openai_client, instrument_no_content, vcr +): + """Responses.parse() emits a GenAI span like create(). + + parse() is the structured-output helper. It does not delegate to the + instrumented create(), so it is wrapped separately (#659), but it maps to + the same inference operation as create() -- the request/response fields + are identical -- exactly as chat.completions.parse reuses the completions + create wrapper. The recorded response body is valid structured JSON so + the SDK can materialize the ``text_format`` model. + """ + _skip_if_not_latest() + + with vcr.use_cassette("test_responses_parse_basic[content_mode0].yaml"): + response = openai_client.responses.parse( + model=DEFAULT_MODEL, + instructions=SYSTEM_INSTRUCTIONS, + input=USER_ONLY_PROMPT[0]["content"], + text_format=_ParseCalendarEvent, + stream=False, + ) + + (span,) = span_exporter.get_finished_spans() + assert_all_attributes( + span, + DEFAULT_MODEL, + True, + response.id, + response.model, + response.usage.input_tokens, + response.usage.output_tokens, + response_service_tier=getattr(response, "service_tier", None), + ) + assert ( + span.attributes[OpenAIAttributes.OPENAI_API_TYPE] + == OpenAIAttributes.OpenaiApiTypeValues.RESPONSES.value + ) + # parse(text_format=...) is a structured-output call, so the span records + # the JSON output type -- but only the format metadata, never the caller's + # Pydantic schema (issue #659). + _assert_request_attrs(span, output_type="json") + assert "_ParseCalendarEvent" not in str(span.attributes) + + +@pytest.mark.skipif( + not _HAS_RESPONSES_PARSE, + reason="Responses.parse requires a newer openai SDK", +) +def test_responses_parse_wrapping_lifecycle( + tracer_provider, logger_provider, meter_provider +): + """instrument() wraps Responses.parse / AsyncResponses.parse and + uninstrument() restores them.""" + from openai.resources.responses.responses import ( # pylint: disable=no-name-in-module + AsyncResponses, + Responses, + ) + + before_sync = Responses.parse + before_async = AsyncResponses.parse + + instrumentor = OpenAIInstrumentor() + instrumentor.instrument( + tracer_provider=tracer_provider, + logger_provider=logger_provider, + meter_provider=meter_provider, + ) + assert hasattr(Responses.parse, "__wrapped__") + assert hasattr(AsyncResponses.parse, "__wrapped__") + + instrumentor.uninstrument() + assert Responses.parse is before_sync + assert AsyncResponses.parse is before_async + + +@pytest.mark.skipif( + not _HAS_RESPONSES_PARSE, + reason="Responses.parse requires a newer openai SDK", +) +def test_responses_create_output_type_unchanged_by_parse( + span_exporter, openai_client, instrument_no_content, vcr +): + """Responses.create() behaviour is unchanged by the parse() wrapper. + + ``create`` carries no ``text_format``, so reusing the ``responses_create`` + wrapper for ``parse`` must not start reporting ``gen_ai.output.type`` for a + plain text call (regression guard for issue #659). Reuses the existing + create cassette -- VCR does not match on the request body. + """ + _skip_if_not_latest() + + with vcr.use_cassette("test_responses_create_basic[content_mode0].yaml"): + response = openai_client.responses.create( + model=DEFAULT_MODEL, + instructions=SYSTEM_INSTRUCTIONS, + input=USER_ONLY_PROMPT[0]["content"], + stream=False, + ) + + (span,) = span_exporter.get_finished_spans() + assert_all_attributes( + span, + DEFAULT_MODEL, + True, + response.id, + response.model, + response.usage.input_tokens, + response.usage.output_tokens, + response_service_tier=getattr(response, "service_tier", None), + ) + assert GenAIAttributes.GEN_AI_OUTPUT_TYPE not in span.attributes