From a843d4fbaf37b40ccf19ed8d8bcdffae0f79db8c Mon Sep 17 00:00:00 2001 From: Ryan-amazinghorse <2194553866@qq.com> Date: Fri, 2 Oct 2026 19:13:21 +0800 Subject: [PATCH 1/2] fix: map streamed reasoning item content --- .../openairesponseschatgenerator.mdx | 2 + .../openairesponseschatgenerator.mdx | 2 + .../generators/chat/openai_responses.py | 19 ++++---- ...ng-reasoning-content-dc7520aa9f334bb3.yaml | 4 ++ .../chat/test_openai_responses_conversion.py | 44 +++++++++++++++++++ 5 files changed, 61 insertions(+), 10 deletions(-) create mode 100644 releasenotes/notes/streaming-reasoning-content-dc7520aa9f334bb3.yaml diff --git a/docs-website/docs/pipeline-components/generators/openairesponseschatgenerator.mdx b/docs-website/docs/pipeline-components/generators/openairesponseschatgenerator.mdx index 1c90797cbd0..31b3ed979de 100644 --- a/docs-website/docs/pipeline-components/generators/openairesponseschatgenerator.mdx +++ b/docs-website/docs/pipeline-components/generators/openairesponseschatgenerator.mdx @@ -75,6 +75,8 @@ The `reasoning` parameter accepts: OpenAI does not return the actual reasoning tokens, but you can view the summary if enabled. For more details, see the [OpenAI Reasoning documentation](https://platform.openai.com/docs/guides/reasoning). ::: +When streaming from a compatible provider that returns `reasoning_text` entries in a completed reasoning item's `content`, Haystack exposes that text in the streaming callback and final message's `reasoning.reasoning_text`. If text has already been streamed for the same output item, Haystack preserves that text instead of appending the completed content again. The original content remains available in `reasoning.extra["content"]`. + ### Multi-turn Conversations The Responses API supports multi-turn conversations using `previous_response_id`. You can pass the response ID from a previous turn to maintain conversation context: diff --git a/docs-website/versioned_docs/version-3.3/pipeline-components/generators/openairesponseschatgenerator.mdx b/docs-website/versioned_docs/version-3.3/pipeline-components/generators/openairesponseschatgenerator.mdx index 1c90797cbd0..31b3ed979de 100644 --- a/docs-website/versioned_docs/version-3.3/pipeline-components/generators/openairesponseschatgenerator.mdx +++ b/docs-website/versioned_docs/version-3.3/pipeline-components/generators/openairesponseschatgenerator.mdx @@ -75,6 +75,8 @@ The `reasoning` parameter accepts: OpenAI does not return the actual reasoning tokens, but you can view the summary if enabled. For more details, see the [OpenAI Reasoning documentation](https://platform.openai.com/docs/guides/reasoning). ::: +When streaming from a compatible provider that returns `reasoning_text` entries in a completed reasoning item's `content`, Haystack exposes that text in the streaming callback and final message's `reasoning.reasoning_text`. If text has already been streamed for the same output item, Haystack preserves that text instead of appending the completed content again. The original content remains available in `reasoning.extra["content"]`. + ### Multi-turn Conversations The Responses API supports multi-turn conversations using `previous_response_id`. You can pass the response ID from a previous turn to maintain conversation context: diff --git a/haystack/components/generators/chat/openai_responses.py b/haystack/components/generators/chat/openai_responses.py index 94b91538326..a73ec78956c 100644 --- a/haystack/components/generators/chat/openai_responses.py +++ b/haystack/components/generators/chat/openai_responses.py @@ -773,16 +773,15 @@ def _convert_response_chunk_to_streaming_chunk( # noqa: PLR0911 # event falls through to the generic default and reasoning=None, so encrypted_content # is never available for multi-turn conversations. if chunk.item.type == "reasoning": - if chunk.item.content: - logger.warning( - "OpenAI returned a non-empty 'content' field on a reasoning item ({_id}). " - "This field is currently undocumented and was never observed in practice. " - "The content is preserved in ReasoningContent.extra['content'] but is NOT " - "reflected in ReasoningContent.reasoning_text. Please report this at " - "https://github.com/deepset-ai/haystack/issues so we can update the mapping.", - _id=chunk.item.id, - ) - reasoning = ReasoningContent(reasoning_text="", extra=chunk.item.to_dict()) + # Prefer text already streamed for this item so the completed content is not appended twice. + has_streamed_reasoning = any( + previous.index == chunk.output_index and previous.reasoning and previous.reasoning.reasoning_text + for previous in previous_chunks + ) + reasoning_text = "" + if not has_streamed_reasoning and chunk.item.content: + reasoning_text = "\n".join(part.text for part in chunk.item.content if part.type == "reasoning_text") + reasoning = ReasoningContent(reasoning_text=reasoning_text, extra=chunk.item.to_dict()) return StreamingChunk( content="", component_info=component_info, diff --git a/releasenotes/notes/streaming-reasoning-content-dc7520aa9f334bb3.yaml b/releasenotes/notes/streaming-reasoning-content-dc7520aa9f334bb3.yaml new file mode 100644 index 00000000000..878a71f9ef9 --- /dev/null +++ b/releasenotes/notes/streaming-reasoning-content-dc7520aa9f334bb3.yaml @@ -0,0 +1,4 @@ +--- +fixes: + - |- + `OpenAIResponsesChatGenerator` now maps `reasoning_text` entries from a completed reasoning item's `content` into `reasoning.reasoning_text` when streaming, unless text was already streamed for that output item. diff --git a/test/components/generators/chat/test_openai_responses_conversion.py b/test/components/generators/chat/test_openai_responses_conversion.py index 4e5e3cb192c..9a11f57c131 100644 --- a/test/components/generators/chat/test_openai_responses_conversion.py +++ b/test/components/generators/chat/test_openai_responses_conversion.py @@ -30,6 +30,7 @@ ResponseUsage, ) from openai.types.responses.response import IncompleteDetails +from openai.types.responses.response_reasoning_item import Content from openai.types.responses.response_usage import InputTokensDetails, OutputTokensDetails from haystack.components.generators.chat.openai_responses import ( @@ -52,6 +53,49 @@ ) +@pytest.mark.parametrize( + "content_texts,previous_index,expected_text", + [ + (["First step.", "Second step."], None, "First step.\nSecond step."), + (["First step.", "Second step."], 0, ""), + (["First step.", "Second step."], 1, "First step.\nSecond step."), + ([], None, ""), + (None, None, ""), + ], +) +def test_streaming_reasoning_content_fallback( + content_texts: list[str] | None, previous_index: int | None, expected_text: str +) -> None: + previous_chunks = [] + if previous_index is not None: + previous_chunks.append( + StreamingChunk( + content="", + index=previous_index, + reasoning=ReasoningContent(reasoning_text="Summary.", extra={"item_id": "rs_summary"}), + ) + ) + item = ResponseReasoningItem( + id="rs_content", + type="reasoning", + summary=[], + content=[Content(text=text, type="reasoning_text") for text in content_texts] + if content_texts is not None + else None, + encrypted_content="encrypted", + status="completed", + ) + event = ResponseOutputItemDoneEvent(item=item, output_index=0, sequence_number=1, type="response.output_item.done") + chunk = _convert_response_chunk_to_streaming_chunk(chunk=event, previous_chunks=previous_chunks) + assert chunk.reasoning == ReasoningContent(reasoning_text=expected_text, extra=item.to_dict()) + + message = _convert_streaming_chunks_to_chat_message(chunks=[*previous_chunks, chunk]) + assert message.reasoning == ReasoningContent( + reasoning_text=("Summary." if previous_index is not None else "") + expected_text, + extra={**({"item_id": "rs_summary"} if previous_index is not None else {}), **item.to_dict()}, + ) + + @pytest.fixture def openai_responses_streaming_chunks_with_tool_call(): return [ From dcbaca960341ce23e531a268469eee4ff5ee9c8f Mon Sep 17 00:00:00 2001 From: Ryan-amazinghorse <2194553866@qq.com> Date: Fri, 2 Oct 2026 19:26:50 +0800 Subject: [PATCH 2/2] fix: deduplicate streamed reasoning text precisely --- .../openairesponseschatgenerator.mdx | 2 +- .../openairesponseschatgenerator.mdx | 2 +- .../generators/chat/openai_responses.py | 16 +++++++++++----- ...ng-reasoning-content-dc7520aa9f334bb3.yaml | 2 +- .../chat/test_openai_responses_conversion.py | 19 ++++++++++--------- 5 files changed, 24 insertions(+), 17 deletions(-) diff --git a/docs-website/docs/pipeline-components/generators/openairesponseschatgenerator.mdx b/docs-website/docs/pipeline-components/generators/openairesponseschatgenerator.mdx index 31b3ed979de..fa7a1d9d3a7 100644 --- a/docs-website/docs/pipeline-components/generators/openairesponseschatgenerator.mdx +++ b/docs-website/docs/pipeline-components/generators/openairesponseschatgenerator.mdx @@ -75,7 +75,7 @@ The `reasoning` parameter accepts: OpenAI does not return the actual reasoning tokens, but you can view the summary if enabled. For more details, see the [OpenAI Reasoning documentation](https://platform.openai.com/docs/guides/reasoning). ::: -When streaming from a compatible provider that returns `reasoning_text` entries in a completed reasoning item's `content`, Haystack exposes that text in the streaming callback and final message's `reasoning.reasoning_text`. If text has already been streamed for the same output item, Haystack preserves that text instead of appending the completed content again. The original content remains available in `reasoning.extra["content"]`. +When streaming from a compatible provider that returns `reasoning_text` entries in a completed reasoning item's `content`, Haystack exposes that text in the streaming callback and final message's `reasoning.reasoning_text`. If the same text was already streamed for that output item, Haystack does not append it again. If the completed content differs from text streamed earlier for that item, Haystack adds it on a new line. The original content remains available in `reasoning.extra["content"]`. ### Multi-turn Conversations diff --git a/docs-website/versioned_docs/version-3.3/pipeline-components/generators/openairesponseschatgenerator.mdx b/docs-website/versioned_docs/version-3.3/pipeline-components/generators/openairesponseschatgenerator.mdx index 31b3ed979de..fa7a1d9d3a7 100644 --- a/docs-website/versioned_docs/version-3.3/pipeline-components/generators/openairesponseschatgenerator.mdx +++ b/docs-website/versioned_docs/version-3.3/pipeline-components/generators/openairesponseschatgenerator.mdx @@ -75,7 +75,7 @@ The `reasoning` parameter accepts: OpenAI does not return the actual reasoning tokens, but you can view the summary if enabled. For more details, see the [OpenAI Reasoning documentation](https://platform.openai.com/docs/guides/reasoning). ::: -When streaming from a compatible provider that returns `reasoning_text` entries in a completed reasoning item's `content`, Haystack exposes that text in the streaming callback and final message's `reasoning.reasoning_text`. If text has already been streamed for the same output item, Haystack preserves that text instead of appending the completed content again. The original content remains available in `reasoning.extra["content"]`. +When streaming from a compatible provider that returns `reasoning_text` entries in a completed reasoning item's `content`, Haystack exposes that text in the streaming callback and final message's `reasoning.reasoning_text`. If the same text was already streamed for that output item, Haystack does not append it again. If the completed content differs from text streamed earlier for that item, Haystack adds it on a new line. The original content remains available in `reasoning.extra["content"]`. ### Multi-turn Conversations diff --git a/haystack/components/generators/chat/openai_responses.py b/haystack/components/generators/chat/openai_responses.py index a73ec78956c..e8733271acd 100644 --- a/haystack/components/generators/chat/openai_responses.py +++ b/haystack/components/generators/chat/openai_responses.py @@ -773,14 +773,20 @@ def _convert_response_chunk_to_streaming_chunk( # noqa: PLR0911 # event falls through to the generic default and reasoning=None, so encrypted_content # is never available for multi-turn conversations. if chunk.item.type == "reasoning": - # Prefer text already streamed for this item so the completed content is not appended twice. - has_streamed_reasoning = any( - previous.index == chunk.output_index and previous.reasoning and previous.reasoning.reasoning_text + streamed_reasoning_text = "".join( + previous.reasoning.reasoning_text for previous in previous_chunks + if previous.index == chunk.output_index and previous.reasoning ) reasoning_text = "" - if not has_streamed_reasoning and chunk.item.content: - reasoning_text = "\n".join(part.text for part in chunk.item.content if part.type == "reasoning_text") + if chunk.item.content: + completed_reasoning_text = "\n".join( + part.text for part in chunk.item.content if part.type == "reasoning_text" + ) + if completed_reasoning_text != streamed_reasoning_text: + reasoning_text = completed_reasoning_text + if streamed_reasoning_text: + reasoning_text = f"\n{reasoning_text}" reasoning = ReasoningContent(reasoning_text=reasoning_text, extra=chunk.item.to_dict()) return StreamingChunk( content="", diff --git a/releasenotes/notes/streaming-reasoning-content-dc7520aa9f334bb3.yaml b/releasenotes/notes/streaming-reasoning-content-dc7520aa9f334bb3.yaml index 878a71f9ef9..791206381a5 100644 --- a/releasenotes/notes/streaming-reasoning-content-dc7520aa9f334bb3.yaml +++ b/releasenotes/notes/streaming-reasoning-content-dc7520aa9f334bb3.yaml @@ -1,4 +1,4 @@ --- fixes: - |- - `OpenAIResponsesChatGenerator` now maps `reasoning_text` entries from a completed reasoning item's `content` into `reasoning.reasoning_text` when streaming, unless text was already streamed for that output item. + `OpenAIResponsesChatGenerator` now maps `reasoning_text` entries from a completed reasoning item's `content` into `reasoning.reasoning_text` when streaming. It skips text already streamed for the same output item and keeps distinct content on a new line. diff --git a/test/components/generators/chat/test_openai_responses_conversion.py b/test/components/generators/chat/test_openai_responses_conversion.py index 9a11f57c131..fd8a82d16ad 100644 --- a/test/components/generators/chat/test_openai_responses_conversion.py +++ b/test/components/generators/chat/test_openai_responses_conversion.py @@ -54,17 +54,18 @@ @pytest.mark.parametrize( - "content_texts,previous_index,expected_text", + "content_texts,previous_text,previous_index,expected_text", [ - (["First step.", "Second step."], None, "First step.\nSecond step."), - (["First step.", "Second step."], 0, ""), - (["First step.", "Second step."], 1, "First step.\nSecond step."), - ([], None, ""), - (None, None, ""), + (["First step.", "Second step."], None, None, "First step.\nSecond step."), + (["First step.", "Second step."], "First step.\nSecond step.", 0, ""), + (["First step.", "Second step."], "Summary.", 0, "\nFirst step.\nSecond step."), + (["First step.", "Second step."], "First step.\nSecond step.", 1, "First step.\nSecond step."), + ([], None, None, ""), + (None, None, None, ""), ], ) def test_streaming_reasoning_content_fallback( - content_texts: list[str] | None, previous_index: int | None, expected_text: str + content_texts: list[str] | None, previous_text: str | None, previous_index: int | None, expected_text: str ) -> None: previous_chunks = [] if previous_index is not None: @@ -72,7 +73,7 @@ def test_streaming_reasoning_content_fallback( StreamingChunk( content="", index=previous_index, - reasoning=ReasoningContent(reasoning_text="Summary.", extra={"item_id": "rs_summary"}), + reasoning=ReasoningContent(reasoning_text=previous_text or "", extra={"item_id": "rs_summary"}), ) ) item = ResponseReasoningItem( @@ -91,7 +92,7 @@ def test_streaming_reasoning_content_fallback( message = _convert_streaming_chunks_to_chat_message(chunks=[*previous_chunks, chunk]) assert message.reasoning == ReasoningContent( - reasoning_text=("Summary." if previous_index is not None else "") + expected_text, + reasoning_text=(previous_text or "") + expected_text, extra={**({"item_id": "rs_summary"} if previous_index is not None else {}), **item.to_dict()}, )