diff --git a/docs-website/docs/pipeline-components/generators/openairesponseschatgenerator.mdx b/docs-website/docs/pipeline-components/generators/openairesponseschatgenerator.mdx index 1c90797cbd0..fa7a1d9d3a7 100644 --- a/docs-website/docs/pipeline-components/generators/openairesponseschatgenerator.mdx +++ b/docs-website/docs/pipeline-components/generators/openairesponseschatgenerator.mdx @@ -75,6 +75,8 @@ The `reasoning` parameter accepts: OpenAI does not return the actual reasoning tokens, but you can view the summary if enabled. For more details, see the [OpenAI Reasoning documentation](https://platform.openai.com/docs/guides/reasoning). ::: +When streaming from a compatible provider that returns `reasoning_text` entries in a completed reasoning item's `content`, Haystack exposes that text in the streaming callback and final message's `reasoning.reasoning_text`. If the same text was already streamed for that output item, Haystack does not append it again. If the completed content differs from text streamed earlier for that item, Haystack adds it on a new line. The original content remains available in `reasoning.extra["content"]`. + ### Multi-turn Conversations The Responses API supports multi-turn conversations using `previous_response_id`. You can pass the response ID from a previous turn to maintain conversation context: diff --git a/docs-website/versioned_docs/version-3.3/pipeline-components/generators/openairesponseschatgenerator.mdx b/docs-website/versioned_docs/version-3.3/pipeline-components/generators/openairesponseschatgenerator.mdx index 1c90797cbd0..fa7a1d9d3a7 100644 --- a/docs-website/versioned_docs/version-3.3/pipeline-components/generators/openairesponseschatgenerator.mdx +++ b/docs-website/versioned_docs/version-3.3/pipeline-components/generators/openairesponseschatgenerator.mdx @@ -75,6 +75,8 @@ The `reasoning` parameter accepts: OpenAI does not return the actual reasoning tokens, but you can view the summary if enabled. For more details, see the [OpenAI Reasoning documentation](https://platform.openai.com/docs/guides/reasoning). ::: +When streaming from a compatible provider that returns `reasoning_text` entries in a completed reasoning item's `content`, Haystack exposes that text in the streaming callback and final message's `reasoning.reasoning_text`. If the same text was already streamed for that output item, Haystack does not append it again. If the completed content differs from text streamed earlier for that item, Haystack adds it on a new line. The original content remains available in `reasoning.extra["content"]`. + ### Multi-turn Conversations The Responses API supports multi-turn conversations using `previous_response_id`. You can pass the response ID from a previous turn to maintain conversation context: diff --git a/haystack/components/generators/chat/openai_responses.py b/haystack/components/generators/chat/openai_responses.py index 94b91538326..e8733271acd 100644 --- a/haystack/components/generators/chat/openai_responses.py +++ b/haystack/components/generators/chat/openai_responses.py @@ -773,16 +773,21 @@ def _convert_response_chunk_to_streaming_chunk( # noqa: PLR0911 # event falls through to the generic default and reasoning=None, so encrypted_content # is never available for multi-turn conversations. if chunk.item.type == "reasoning": + streamed_reasoning_text = "".join( + previous.reasoning.reasoning_text + for previous in previous_chunks + if previous.index == chunk.output_index and previous.reasoning + ) + reasoning_text = "" if chunk.item.content: - logger.warning( - "OpenAI returned a non-empty 'content' field on a reasoning item ({_id}). " - "This field is currently undocumented and was never observed in practice. " - "The content is preserved in ReasoningContent.extra['content'] but is NOT " - "reflected in ReasoningContent.reasoning_text. Please report this at " - "https://github.com/deepset-ai/haystack/issues so we can update the mapping.", - _id=chunk.item.id, + completed_reasoning_text = "\n".join( + part.text for part in chunk.item.content if part.type == "reasoning_text" ) - reasoning = ReasoningContent(reasoning_text="", extra=chunk.item.to_dict()) + if completed_reasoning_text != streamed_reasoning_text: + reasoning_text = completed_reasoning_text + if streamed_reasoning_text: + reasoning_text = f"\n{reasoning_text}" + reasoning = ReasoningContent(reasoning_text=reasoning_text, extra=chunk.item.to_dict()) return StreamingChunk( content="", component_info=component_info, diff --git a/releasenotes/notes/streaming-reasoning-content-dc7520aa9f334bb3.yaml b/releasenotes/notes/streaming-reasoning-content-dc7520aa9f334bb3.yaml new file mode 100644 index 00000000000..791206381a5 --- /dev/null +++ b/releasenotes/notes/streaming-reasoning-content-dc7520aa9f334bb3.yaml @@ -0,0 +1,4 @@ +--- +fixes: + - |- + `OpenAIResponsesChatGenerator` now maps `reasoning_text` entries from a completed reasoning item's `content` into `reasoning.reasoning_text` when streaming. It skips text already streamed for the same output item and keeps distinct content on a new line. diff --git a/test/components/generators/chat/test_openai_responses_conversion.py b/test/components/generators/chat/test_openai_responses_conversion.py index 4e5e3cb192c..fd8a82d16ad 100644 --- a/test/components/generators/chat/test_openai_responses_conversion.py +++ b/test/components/generators/chat/test_openai_responses_conversion.py @@ -30,6 +30,7 @@ ResponseUsage, ) from openai.types.responses.response import IncompleteDetails +from openai.types.responses.response_reasoning_item import Content from openai.types.responses.response_usage import InputTokensDetails, OutputTokensDetails from haystack.components.generators.chat.openai_responses import ( @@ -52,6 +53,50 @@ ) +@pytest.mark.parametrize( + "content_texts,previous_text,previous_index,expected_text", + [ + (["First step.", "Second step."], None, None, "First step.\nSecond step."), + (["First step.", "Second step."], "First step.\nSecond step.", 0, ""), + (["First step.", "Second step."], "Summary.", 0, "\nFirst step.\nSecond step."), + (["First step.", "Second step."], "First step.\nSecond step.", 1, "First step.\nSecond step."), + ([], None, None, ""), + (None, None, None, ""), + ], +) +def test_streaming_reasoning_content_fallback( + content_texts: list[str] | None, previous_text: str | None, previous_index: int | None, expected_text: str +) -> None: + previous_chunks = [] + if previous_index is not None: + previous_chunks.append( + StreamingChunk( + content="", + index=previous_index, + reasoning=ReasoningContent(reasoning_text=previous_text or "", extra={"item_id": "rs_summary"}), + ) + ) + item = ResponseReasoningItem( + id="rs_content", + type="reasoning", + summary=[], + content=[Content(text=text, type="reasoning_text") for text in content_texts] + if content_texts is not None + else None, + encrypted_content="encrypted", + status="completed", + ) + event = ResponseOutputItemDoneEvent(item=item, output_index=0, sequence_number=1, type="response.output_item.done") + chunk = _convert_response_chunk_to_streaming_chunk(chunk=event, previous_chunks=previous_chunks) + assert chunk.reasoning == ReasoningContent(reasoning_text=expected_text, extra=item.to_dict()) + + message = _convert_streaming_chunks_to_chat_message(chunks=[*previous_chunks, chunk]) + assert message.reasoning == ReasoningContent( + reasoning_text=(previous_text or "") + expected_text, + extra={**({"item_id": "rs_summary"} if previous_index is not None else {}), **item.to_dict()}, + ) + + @pytest.fixture def openai_responses_streaming_chunks_with_tool_call(): return [