From b76bddbbe138860cd7d7eff6d284b802c473726b Mon Sep 17 00:00:00 2001 From: fei <204683769+feiiiiii5@users.noreply.github.com> Date: Tue, 29 Sep 2026 19:26:37 +0800 Subject: [PATCH 1/8] FIX: raise when a completed Responses output has nothing readable _construct_message_from_response_async tracks has_visible_response but only consulted it on the truncated path, so a completed response whose output PyRIT cannot read came back as a successful Message holding the reasoning dump. With a built-in tool enabled (image_generation, code_interpreter, file_search, ...) the Responses API returns sections such as image_generation_call, which _parse_response_output_section skips with `return None`; the run then scored the reasoning JSON as the model's answer with response_error="none". OpenAIChatTarget already raises EmptyResponseException when a response that is not truncated yields no content. Do the same here, and keep returning the message when a readable section is present next to an unmodelled one. --- .../openai/openai_response_target.py | 17 ++++++-- .../target/test_openai_response_target.py | 41 +++++++++++++++++++ 2 files changed, 55 insertions(+), 3 deletions(-) diff --git a/pyrit/prompt_target/openai/openai_response_target.py b/pyrit/prompt_target/openai/openai_response_target.py index 6403e20649..20d0adc1a2 100644 --- a/pyrit/prompt_target/openai/openai_response_target.py +++ b/pyrit/prompt_target/openai/openai_response_target.py @@ -524,6 +524,11 @@ async def _construct_message_from_response_async(self, response: Response, reque counts from ``response.usage`` are recorded in the first piece's ``prompt_metadata``. Truncated responses are flagged via ``MessagePiece.mark_as_truncated`` on the first piece. + + Raises: + EmptyResponseException: If a response that is not truncated carries no section PyRIT can + read. Reasoning and section types PyRIT does not model (e.g. ``image_generation_call``) + are not answers, so reporting them as the model's response would score them as one. """ truncated = self._is_truncated_response(response) @@ -552,9 +557,15 @@ async def _construct_message_from_response_async(self, response: Response, reque if piece.original_value and piece.original_value_data_type != "reasoning": has_visible_response = True - if truncated and not has_visible_response: - empty_piece = build_empty_truncated_response(request=request).message_pieces[0] - extracted_response_pieces.append(empty_piece) + if not has_visible_response: + if truncated: + empty_piece = build_empty_truncated_response(request=request).message_pieces[0] + extracted_response_pieces.append(empty_piece) + else: + # A response that completed without a readable section is a failure, not an answer. + # The chat target raises in the same situation; reporting reasoning or a section + # type PyRIT does not model as the model's response would let it be scored as one. + raise EmptyResponseException(message="Failed to extract any response content.") # Consumers use the first piece as the semantic response. Responses API # reasoning commonly precedes the actual message in provider output, so diff --git a/tests/unit/prompt_target/target/test_openai_response_target.py b/tests/unit/prompt_target/target/test_openai_response_target.py index 9964216b82..19eaed0ac8 100644 --- a/tests/unit/prompt_target/target/test_openai_response_target.py +++ b/tests/unit/prompt_target/target/test_openai_response_target.py @@ -1655,6 +1655,47 @@ async def test_construct_message_truncated_skips_partial_tool_call( assert any(p.original_value_data_type == "text" and p.response_error == "empty" for p in result.message_pieces) +def _make_unreadable_section() -> MagicMock: + """A completed section PyRIT does not model, e.g. the ``image_generation_call`` item the + Responses API returns once a run enables the built-in image_generation tool.""" + section = MagicMock() + section.type = "image_generation_call" + return section + + +def _make_completed_response(output: list | None) -> MagicMock: + response = MagicMock() + response.error = None + response.status = "completed" + response.incomplete_details = None + response.output = output + return response + + +async def test_construct_message_completed_without_readable_output_raises( + target: OpenAIResponseTarget, dummy_text_message_piece: MessagePiece +): + """A completed response with nothing PyRIT can read is an error, not a reasoning-only answer.""" + response = _make_completed_response(output=[_make_reasoning_section(), _make_unreadable_section()]) + + with pytest.raises(EmptyResponseException): + await target._construct_message_from_response_async(response, dummy_text_message_piece) + + +async def test_construct_message_completed_keeps_readable_output_next_to_unreadable( + target: OpenAIResponseTarget, dummy_text_message_piece: MessagePiece +): + """A readable section alongside an unmodelled one is still returned.""" + response = _make_completed_response( + output=[_make_reasoning_section(), _make_unreadable_section(), _make_message_section("An answer")] + ) + + result = await target._construct_message_from_response_async(response, dummy_text_message_piece) + + text_pieces = [p for p in result.message_pieces if p.original_value_data_type == "text"] + assert [p.original_value for p in text_pieces] == ["An answer"] + + async def test_construct_message_from_response(target: OpenAIResponseTarget, dummy_text_message_piece: MessagePiece): """Test _construct_message_from_response parses output sections.""" mock_response = MagicMock() From 1de72b6c6e2136fb867d8cb7c3fadebf5782fc10 Mon Sep 17 00:00:00 2001 From: feiiiiii5 <204683769+feiiiiii5@users.noreply.github.com> Date: Fri, 2 Oct 2026 09:00:47 +0800 Subject: [PATCH 2/8] FIX: do not retry an unreadable Responses output `_send_model_request_async` is wrapped in `@pyrit_target_retry`, which retries `RateLimitError | EmptyResponseException | RateLimitException`. The check added in this branch raised `EmptyResponseException` for a response that *completed* with no section PyRIT models, so every retry reproduced the same shape: ten billed calls plus backoff before the agentic loop gave up, and the tool messages it had collected were dropped with the exception. `doc/contributing/9_exception.md` scopes retry to rate limits and parse failures, so raise `PyritException` instead. The docstring now says so rather than naming an exception that is no longer raised. The test asserts `type(excinfo.value) is PyritException` and that it is not an `EmptyResponseException`. Asserting only `PyritException` would not have caught this, since `EmptyResponseException` subclasses it -- the weaker assertion passes on the old code too. Reported by @hannahwestra25. --- pyrit/prompt_target/openai/openai_response_target.py | 10 ++++++++-- .../target/test_openai_response_target.py | 8 +++++++- 2 files changed, 15 insertions(+), 3 deletions(-) diff --git a/pyrit/prompt_target/openai/openai_response_target.py b/pyrit/prompt_target/openai/openai_response_target.py index 6244679008..be72c8e701 100644 --- a/pyrit/prompt_target/openai/openai_response_target.py +++ b/pyrit/prompt_target/openai/openai_response_target.py @@ -559,9 +559,11 @@ async def _construct_message_from_response_async(self, response: Response, reque piece. Raises: - EmptyResponseException: If a response that is not truncated carries no section PyRIT can + PyritException: If a response that is not truncated carries no section PyRIT can read. Reasoning and section types PyRIT does not model (e.g. ``image_generation_call``) are not answers, so reporting them as the model's response would score them as one. + This is raised rather than ``EmptyResponseException`` so that + ``@pyrit_target_retry`` does not re-send a request whose outcome is deterministic. """ truncated = self._is_truncated_response(response) @@ -598,7 +600,11 @@ async def _construct_message_from_response_async(self, response: Response, reque # A response that completed without a readable section is a failure, not an answer. # The chat target raises in the same situation; reporting reasoning or a section # type PyRIT does not model as the model's response would let it be scored as one. - raise EmptyResponseException(message="Failed to extract any response content.") + # Not retried: the response completed, so an unmodelled section type is + # deterministic and re-sending the same request bills the same outcome again. + # doc/contributing/9_exception.md scopes @pyrit_target_retry to rate limits + # and parse failures; this is neither. + raise PyritException(message="Failed to extract any response content.") # Consumers use the first piece as the semantic response. Responses API # reasoning commonly precedes the actual message in provider output, so diff --git a/tests/unit/prompt_target/target/test_openai_response_target.py b/tests/unit/prompt_target/target/test_openai_response_target.py index 320e6bf2b6..255a591c4d 100644 --- a/tests/unit/prompt_target/target/test_openai_response_target.py +++ b/tests/unit/prompt_target/target/test_openai_response_target.py @@ -1675,9 +1675,15 @@ async def test_construct_message_completed_without_readable_output_raises( """A completed response with nothing PyRIT can read is an error, not a reasoning-only answer.""" response = _make_completed_response(output=[_make_reasoning_section(), _make_unreadable_section()]) - with pytest.raises(EmptyResponseException): + with pytest.raises(PyritException) as excinfo: await target._construct_message_from_response_async(response, dummy_text_message_piece) + # Deliberately not an EmptyResponseException: @pyrit_target_retry retries that + # type, and a section type PyRIT does not model comes back identically on every + # attempt, so retrying only bills the same outcome again. + assert not isinstance(excinfo.value, EmptyResponseException) + assert type(excinfo.value) is PyritException + async def test_construct_message_completed_keeps_readable_output_next_to_unreadable( target: OpenAIResponseTarget, dummy_text_message_piece: MessagePiece From b79805a18646792c84bf5b0b8c05e16764813902 Mon Sep 17 00:00:00 2001 From: fei <204683769+feiiiiii5@users.noreply.github.com> Date: Wed, 7 Oct 2026 09:02:46 +0800 Subject: [PATCH 3/8] Degrade unreadable Responses output to an empty marker piece Raising made the batch fail and skipped metadata capture, and the proposed PyritException was retried by pyrit_target_retry on a deterministic outcome. Append the graceful empty marker instead, keep reasoning pieces, and pin the shape with a reasoning-only regression test. --- .../openai/openai_response_target.py | 25 ++++---------- .../target/test_openai_response_target.py | 33 ++++++++++++++----- 2 files changed, 30 insertions(+), 28 deletions(-) diff --git a/pyrit/prompt_target/openai/openai_response_target.py b/pyrit/prompt_target/openai/openai_response_target.py index be72c8e701..994d0e7c9f 100644 --- a/pyrit/prompt_target/openai/openai_response_target.py +++ b/pyrit/prompt_target/openai/openai_response_target.py @@ -557,13 +557,6 @@ async def _construct_message_from_response_async(self, response: Response, reque counts from ``response.usage`` are recorded in the first piece's ``prompt_metadata``. Truncated responses are flagged via ``MessagePiece.mark_as_truncated`` on the first piece. - - Raises: - PyritException: If a response that is not truncated carries no section PyRIT can - read. Reasoning and section types PyRIT does not model (e.g. ``image_generation_call``) - are not answers, so reporting them as the model's response would score them as one. - This is raised rather than ``EmptyResponseException`` so that - ``@pyrit_target_retry`` does not re-send a request whose outcome is deterministic. """ truncated = self._is_truncated_response(response) @@ -593,18 +586,12 @@ async def _construct_message_from_response_async(self, response: Response, reque has_visible_response = True if not has_visible_response: - if truncated: - empty_piece = build_empty_truncated_response(request=request).message_pieces[0] - extracted_response_pieces.append(empty_piece) - else: - # A response that completed without a readable section is a failure, not an answer. - # The chat target raises in the same situation; reporting reasoning or a section - # type PyRIT does not model as the model's response would let it be scored as one. - # Not retried: the response completed, so an unmodelled section type is - # deterministic and re-sending the same request bills the same outcome again. - # doc/contributing/9_exception.md scopes @pyrit_target_retry to rate limits - # and parse failures; this is neither. - raise PyritException(message="Failed to extract any response content.") + # A response with no readable section is not an exception: EmptyResponseException + # would be retried by @pyrit_target_retry even though the outcome is deterministic, + # and raising would skip metadata capture below. Append a graceful empty marker + # piece instead and keep any reasoning pieces; nothing raises, so nothing retries. + empty_piece = build_empty_truncated_response(request=request).message_pieces[0] + extracted_response_pieces.append(empty_piece) # Consumers use the first piece as the semantic response. Responses API # reasoning commonly precedes the actual message in provider output, so diff --git a/tests/unit/prompt_target/target/test_openai_response_target.py b/tests/unit/prompt_target/target/test_openai_response_target.py index 255a591c4d..4536b1c28b 100644 --- a/tests/unit/prompt_target/target/test_openai_response_target.py +++ b/tests/unit/prompt_target/target/test_openai_response_target.py @@ -1669,20 +1669,35 @@ def _make_completed_response(output: list | None) -> MagicMock: return response -async def test_construct_message_completed_without_readable_output_raises( +async def test_construct_message_completed_without_readable_output_returns_empty_marker( target: OpenAIResponseTarget, dummy_text_message_piece: MessagePiece ): - """A completed response with nothing PyRIT can read is an error, not a reasoning-only answer.""" + """A completed response with nothing PyRIT can read degrades to an empty marker piece.""" response = _make_completed_response(output=[_make_reasoning_section(), _make_unreadable_section()]) - with pytest.raises(PyritException) as excinfo: - await target._construct_message_from_response_async(response, dummy_text_message_piece) + result = await target._construct_message_from_response_async(response, dummy_text_message_piece) + + # Nothing raises here, so @pyrit_target_retry does not re-send a deterministic + # outcome; the empty marker is first and the reasoning piece is retained. + assert result.message_pieces[0].original_value == "" + assert result.message_pieces[0].response_error == "empty" + assert result.message_pieces[0].original_value_data_type == "text" + reasoning_pieces = [p for p in result.message_pieces if p.original_value_data_type == "reasoning"] + assert len(reasoning_pieces) == 1 + + +async def test_construct_message_completed_reasoning_only_returns_empty_marker( + target: OpenAIResponseTarget, dummy_text_message_piece: MessagePiece +): + """Real regression shape: the model answered with reasoning only, no visible text.""" + response = _make_completed_response(output=[_make_reasoning_section()]) + + result = await target._construct_message_from_response_async(response, dummy_text_message_piece) - # Deliberately not an EmptyResponseException: @pyrit_target_retry retries that - # type, and a section type PyRIT does not model comes back identically on every - # attempt, so retrying only bills the same outcome again. - assert not isinstance(excinfo.value, EmptyResponseException) - assert type(excinfo.value) is PyritException + assert result.message_pieces[0].original_value == "" + assert result.message_pieces[0].response_error == "empty" + reasoning_pieces = [p for p in result.message_pieces if p.original_value_data_type == "reasoning"] + assert len(reasoning_pieces) == 1 async def test_construct_message_completed_keeps_readable_output_next_to_unreadable( From d15ae9fb15237918a4e1a83d05b37ef48683f7d1 Mon Sep 17 00:00:00 2001 From: fei <204683769+feiiiiii5@users.noreply.github.com> Date: Thu, 8 Oct 2026 08:50:16 +0800 Subject: [PATCH 4/8] Log a warning when a completed Responses output has no readable section --- pyrit/prompt_target/openai/openai_response_target.py | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/pyrit/prompt_target/openai/openai_response_target.py b/pyrit/prompt_target/openai/openai_response_target.py index 52d3398bfc..50e9ad40ef 100644 --- a/pyrit/prompt_target/openai/openai_response_target.py +++ b/pyrit/prompt_target/openai/openai_response_target.py @@ -597,6 +597,13 @@ async def _construct_message_from_response_async(self, response: Response, reque # would be retried by @pyrit_target_retry even though the outcome is deterministic, # and raising would skip metadata capture below. Append a graceful empty marker # piece instead and keep any reasoning pieces; nothing raises, so nothing retries. + if not truncated: + logger.warning( + "Responses output for conversation %s completed with no readable section; " + "returning an empty response marker. Reasoning-only output or a section type " + "PyRIT does not model can cause this.", + request.conversation_id, + ) empty_piece = build_empty_truncated_response(request=request).message_pieces[0] extracted_response_pieces.append(empty_piece) From 19c660c8c927b5f0eeff780aad7aa08a2ed4173a Mon Sep 17 00:00:00 2001 From: fei <204683769+feiiiiii5@users.noreply.github.com> Date: Thu, 8 Oct 2026 08:50:45 +0800 Subject: [PATCH 5/8] Align the construction docstring with the empty-marker behavior --- pyrit/prompt_target/openai/openai_response_target.py | 9 +++++---- 1 file changed, 5 insertions(+), 4 deletions(-) diff --git a/pyrit/prompt_target/openai/openai_response_target.py b/pyrit/prompt_target/openai/openai_response_target.py index 50e9ad40ef..7c45247021 100644 --- a/pyrit/prompt_target/openai/openai_response_target.py +++ b/pyrit/prompt_target/openai/openai_response_target.py @@ -550,10 +550,11 @@ async def _construct_message_from_response_async(self, response: Response, reque """ Construct a Message from a Response API response. - For a truncated response (see ``_is_truncated_response``), empty output sections are - tolerated, partial tool/function calls are skipped so an incomplete call cannot re-enter the - agentic loop, and a graceful empty text piece is appended when no visible response was - produced. Reasoning, any partial text, and structured refusals are always preserved. + Empty output sections are tolerated on a truncated response (see + ``_is_truncated_response``), where partial tool/function calls are also skipped so an + incomplete call cannot re-enter the agentic loop. Whenever no visible response was + produced — truncated or completed — a graceful empty text piece is appended. Reasoning, + any partial text, and structured refusals are always preserved. Args: response: The Response object from OpenAI SDK. From 68be2330f2e759c18301576d0dfd28222a27f86d Mon Sep 17 00:00:00 2001 From: Copilot App <223556219+Copilot@users.noreply.github.com> Date: Thu, 8 Oct 2026 11:41:04 -0400 Subject: [PATCH 6/8] Cover the empty-marker degradation warning on both paths The completed path warns and the truncated path stays quiet, but neither outcome was asserted, so dropping the warning or the truncation gate would both regress silently. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- .../target/test_openai_response_target.py | 24 +++++++++++++++++++ 1 file changed, 24 insertions(+) diff --git a/tests/unit/prompt_target/target/test_openai_response_target.py b/tests/unit/prompt_target/target/test_openai_response_target.py index 4802e4ed22..03c68aa5ac 100644 --- a/tests/unit/prompt_target/target/test_openai_response_target.py +++ b/tests/unit/prompt_target/target/test_openai_response_target.py @@ -1738,6 +1738,30 @@ async def test_construct_message_completed_keeps_readable_output_next_to_unreada assert [p.original_value for p in text_pieces] == ["An answer"] +async def test_construct_message_completed_without_readable_output_warns( + target: OpenAIResponseTarget, dummy_text_message_piece: MessagePiece, caplog: pytest.LogCaptureFixture +): + """A completed response degrades silently, so the warning is the operator's only signal.""" + response = _make_completed_response(output=[_make_reasoning_section()]) + + with caplog.at_level(logging.WARNING): + await target._construct_message_from_response_async(response, dummy_text_message_piece) + + assert "completed with no readable section" in caplog.text + + +async def test_construct_message_truncated_without_readable_output_does_not_warn( + target: OpenAIResponseTarget, dummy_text_message_piece: MessagePiece, caplog: pytest.LogCaptureFixture +): + """Hitting the token cap is an expected outcome, so the same fallback stays quiet.""" + response = _make_truncated_response(output=[_make_reasoning_section()]) + + with caplog.at_level(logging.WARNING): + await target._construct_message_from_response_async(response, dummy_text_message_piece) + + assert "no readable section" not in caplog.text + + async def test_construct_message_from_response(target: OpenAIResponseTarget, dummy_text_message_piece: MessagePiece): """Test _construct_message_from_response parses output sections.""" mock_response = MagicMock() From 3f969e7345587f546fa1000b751e6fa44f785c23 Mon Sep 17 00:00:00 2001 From: hannahwestra25 Date: Thu, 8 Oct 2026 11:45:43 -0400 Subject: [PATCH 7/8] Apply suggestion from @hannahwestra25 --- pyrit/prompt_target/openai/openai_response_target.py | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/pyrit/prompt_target/openai/openai_response_target.py b/pyrit/prompt_target/openai/openai_response_target.py index 7c45247021..31df27d79c 100644 --- a/pyrit/prompt_target/openai/openai_response_target.py +++ b/pyrit/prompt_target/openai/openai_response_target.py @@ -594,9 +594,8 @@ async def _construct_message_from_response_async(self, response: Response, reque has_visible_response = True if not has_visible_response: - # A response with no readable section is not an exception: EmptyResponseException - # would be retried by @pyrit_target_retry even though the outcome is deterministic, - # and raising would skip metadata capture below. Append a graceful empty marker + # Append a graceful empty marker piece and keep any reasoning pieces when a + # response with no readable section is found # piece instead and keep any reasoning pieces; nothing raises, so nothing retries. if not truncated: logger.warning( From 89df3c3f3f9be4246f27dd5583be1fcc70cc6b07 Mon Sep 17 00:00:00 2001 From: hannahwestra25 Date: Thu, 8 Oct 2026 11:46:14 -0400 Subject: [PATCH 8/8] Apply suggestion from @hannahwestra25 --- pyrit/prompt_target/openai/openai_response_target.py | 1 - 1 file changed, 1 deletion(-) diff --git a/pyrit/prompt_target/openai/openai_response_target.py b/pyrit/prompt_target/openai/openai_response_target.py index 31df27d79c..3c56aedcfd 100644 --- a/pyrit/prompt_target/openai/openai_response_target.py +++ b/pyrit/prompt_target/openai/openai_response_target.py @@ -596,7 +596,6 @@ async def _construct_message_from_response_async(self, response: Response, reque if not has_visible_response: # Append a graceful empty marker piece and keep any reasoning pieces when a # response with no readable section is found - # piece instead and keep any reasoning pieces; nothing raises, so nothing retries. if not truncated: logger.warning( "Responses output for conversation %s completed with no readable section; "