diff --git a/cecli/coders/base_coder.py b/cecli/coders/base_coder.py index 126781384fd..57c7ea15784 100755 --- a/cecli/coders/base_coder.py +++ b/cecli/coders/base_coder.py @@ -3984,7 +3984,7 @@ async def show_send_output(self, completion): if ( not len(self.partial_response_content) and not len(self.partial_response_tool_calls) - and not len(self.partial_response_reasoning_content) + and not _is_meaningful_reasoning(self.partial_response_reasoning_content) ): self.empty_response = True return @@ -4114,7 +4114,8 @@ async def show_send_output_stream(self, completion): text += reasoning_content self.got_reasoning_content = True - received_content = True + if _is_meaningful_reasoning(reasoning_content): + received_content = True self.token_profiler.on_token() self.io.update_spinner_suffix(reasoning_content) @@ -4190,7 +4191,14 @@ async def show_send_output_stream(self, completion): self.io.tool_warning("Execution stopped by on message hook") return - if not received_content and len(self.partial_response_tool_calls) == 0: + # Treat the response as empty when nothing was received, or when the + # only thing received was reasoning made entirely of non-alphanumeric + # characters (e.g. moonshotai/kimi-k3 returning "!!!!"). + if ( + not received_content + and len(self.partial_response_tool_calls) == 0 + and not _is_meaningful_reasoning(self.partial_response_reasoning_content) + ): self.empty_response = True return @@ -5379,3 +5387,15 @@ def _first_usage_tokens(usage: object, paths: list[str], default: int = 0) -> in if value is not None: return value return default + + +def _is_meaningful_reasoning(text): + """Return True if reasoning text contains at least one alphanumeric character. + + Some providers (e.g. moonshotai/kimi-k3) occasionally return completions + with empty ``content`` and a ``reasoning_content`` made entirely of + punctuation (e.g. ``"!!!!"``). Those responses are effectively empty, so + the empty-response detector only lets reasoning count as response + content when it holds at least one alphanumeric character. + """ + return bool(text) and any(ch.isalnum() for ch in text) diff --git a/cecli/website/docs/config/retries.md b/cecli/website/docs/config/retries.md index 3140c2de6ed..0aa01c39f85 100644 --- a/cecli/website/docs/config/retries.md +++ b/cecli/website/docs/config/retries.md @@ -11,6 +11,7 @@ Cecli can be configured to retry failed API calls. This is useful for handling i - `retry-timeout`: The timeout in seconds for each retry. - `retry-backoff-factor`: The backoff factor to use between retries. - `retry-on-unavailable`: Whether to retry on 503 Service Unavailable errors. +- `retry-on-empty`: Whether to retry when the model returns an empty response. A response is considered empty when it has no content, no tool calls, and no *meaningful* reasoning. Reasoning that contains no alphanumeric characters (for example a `reasoning_content` of `"!!!!"`) does not count as a response, so it is retried like any other empty response. - `retry-on-unauthorized`: Whether to retry on 401 Unauthorized (and 403 Forbidden) errors. Default: false. Example usage in `.cecli.conf.yml`: diff --git a/tests/basic/test_empty_response.py b/tests/basic/test_empty_response.py new file mode 100644 index 00000000000..4b562fc7222 --- /dev/null +++ b/tests/basic/test_empty_response.py @@ -0,0 +1,223 @@ +"""Tests for empty-response detection with non-meaningful reasoning content. + +Regression coverage for the ``retry-on-empty`` fix: a completion with empty +``content``, no tool calls, and a ``reasoning_content`` made entirely of +non-alphanumeric characters (e.g. moonshotai/kimi-k3 returning ``"!!!!"``) must +be classified as an empty response so the existing retry loop engages. +""" + +from types import SimpleNamespace + +import pytest + +from cecli.coders.base_coder import Coder, _is_meaningful_reasoning +from cecli.helpers.threading import ThreadSafeEvent +from cecli.llm import litellm + + +# --------------------------------------------------------------------------- # +# Test doubles +# --------------------------------------------------------------------------- # +class _AlwaysSetEvent: + def is_set(self): + return True + + +class _FakeIO: + def __init__(self): + self.confirmation_in_progress_event = _AlwaysSetEvent() + self.assistant_outputs = [] + + def tool_error(self, *a, **k): + pass + + def tool_warning(self, *a, **k): + pass + + def update_spinner_suffix(self, *a, **k): + pass + + def reset_streaming_response(self): + pass + + def stream_output(self, *a, **k): + pass + + def ai_output(self, *a, **k): + pass + + def tool_output(self, *a, **k): + pass + + def assistant_output(self, *a, **k): + self.assistant_outputs.append(a) + + +class _FakeTokenProfiler: + def start(self): + pass + + def on_token(self): + pass + + def on_error(self): + pass + + def add_to_usage_report(self, *a, **k): + return a[0] if a else "" + + +def _make_coder(stream=False): + coder = Coder.__new__(Coder) + coder.stream = stream + coder.verbose = False + coder.args = SimpleNamespace(debug=False, show_thinking=False) + coder.io = _FakeIO() + coder.interrupt_event = ThreadSafeEvent() + coder.pretty = False + coder.reasoning_tag_name = "THINKING" + coder.got_reasoning_content = False + coder.ended_reasoning_content = False + coder.empty_response = False + coder.tool_reflection = False + coder.partial_response_content = "" + coder.partial_response_reasoning_content = "" + coder.partial_response_chunks = [] + coder.partial_response_tool_calls = [] + coder.partial_response_function_call = dict() + coder.partial_response_consolidated = None + coder.multi_response_content = "" + coder.chat_completion_response_hashes = [] + coder._streaming_buffer_length = 0 + coder.token_profiler = _FakeTokenProfiler() + coder._output_loop_detected = False + coder._output_loop_message = "" + coder._has_empty_reflected = False + coder.edit_format = "code" + return coder + + +def _tc(index, call_id, name, arguments): + return litellm.ChatCompletionMessageToolCall( + id=call_id, + function=litellm.Function(arguments=arguments or "", name=name), + type="function", + index=index, + ) + + +def _non_streaming_completion(content="", reasoning=None, tool_calls=None): + message = {"role": "assistant", "content": content} + if reasoning is not None: + message["reasoning_content"] = reasoning + if tool_calls is not None: + message["tool_calls"] = tool_calls + return litellm.ModelResponse( + id="test-completion", + created=0, + model="gpt-test", + object="chat.completion", + choices=[{"finish_reason": "stop", "index": 0, "message": message}], + usage={"completion_tokens": 1, "prompt_tokens": 1, "total_tokens": 2}, + ) + + +def _reasoning_chunk(text): + delta = litellm.Delta(role="assistant", content=None, reasoning_content=text) + choice = litellm.StreamChoice(finish_reason=None, index=0, delta=delta) + return litellm.StreamChunk( + id="cmpl-test", created=1000, model="gpt-test", choices=[choice], usage=None + ) + + +async def _agen(chunks): + for chunk in chunks: + yield chunk + + +# --------------------------------------------------------------------------- # +# _is_meaningful_reasoning unit tests +# --------------------------------------------------------------------------- # +def test_is_meaningful_reasoning_empty_and_none(): + assert _is_meaningful_reasoning("") is False + assert _is_meaningful_reasoning(None) is False + + +def test_is_meaningful_reasoning_punctuation_only(): + assert _is_meaningful_reasoning("!!!!") is False + assert _is_meaningful_reasoning("...?!,;:") is False + + +def test_is_meaningful_reasoning_whitespace_only(): + assert _is_meaningful_reasoning(" \n\t ") is False + + +def test_is_meaningful_reasoning_normal_text(): + assert _is_meaningful_reasoning("Let me think about this") is True + + +def test_is_meaningful_reasoning_mixed_punctuation_and_alnum(): + assert _is_meaningful_reasoning("42?") is True + assert _is_meaningful_reasoning("!!!a!!!") is True + + +# --------------------------------------------------------------------------- # +# Non-streaming path (show_send_output) +# --------------------------------------------------------------------------- # +@pytest.mark.asyncio +async def test_junk_reasoning_is_empty_non_streaming(): + coder = _make_coder(stream=False) + await coder.show_send_output(_non_streaming_completion(content="", reasoning="!" * 40)) + assert coder.empty_response is True + + +@pytest.mark.asyncio +async def test_meaningful_reasoning_is_not_empty_non_streaming(): + coder = _make_coder(stream=False) + await coder.show_send_output( + _non_streaming_completion(content="", reasoning="Let me think about this") + ) + assert coder.empty_response is False + + +@pytest.mark.asyncio +async def test_tool_calls_are_not_empty_non_streaming(): + coder = _make_coder(stream=False) + await coder.show_send_output( + _non_streaming_completion( + content="", + reasoning="", + tool_calls=[_tc(0, "call_1", "Local--ls", "{}")], + ) + ) + assert coder.empty_response is False + + +@pytest.mark.asyncio +async def test_mixed_reasoning_is_not_empty_non_streaming(): + coder = _make_coder(stream=False) + await coder.show_send_output(_non_streaming_completion(content="", reasoning="42?")) + assert coder.empty_response is False + + +# --------------------------------------------------------------------------- # +# Streaming path (show_send_output_stream) +# --------------------------------------------------------------------------- # +@pytest.mark.asyncio +async def test_junk_reasoning_is_empty_streaming(): + coder = _make_coder(stream=True) + coder.args.show_thinking = True + async for _ in coder.show_send_output_stream(_agen([_reasoning_chunk("!" * 40)])): + pass + assert coder.empty_response is True + + +@pytest.mark.asyncio +async def test_meaningful_reasoning_is_not_empty_streaming(): + coder = _make_coder(stream=True) + coder.args.show_thinking = True + async for _ in coder.show_send_output_stream( + _agen([_reasoning_chunk("Let me think about this")]) + ): + pass + assert coder.empty_response is False