diff --git a/openevolve/llm/claude_code.py b/openevolve/llm/claude_code.py index ddbe2e3b7..5bd3c9fb3 100644 --- a/openevolve/llm/claude_code.py +++ b/openevolve/llm/claude_code.py @@ -79,7 +79,8 @@ async def generate_with_context( budget = kwargs.get("max_budget_usd", self.max_budget_usd) cmd.extend(["--max-budget-usd", str(budget)]) - cmd.append(user_content) + # The prompt goes through stdin: as one argv string it hits Linux's 128 KiB + # per-argument limit (MAX_ARG_STRLEN) once a few programs are in the prompt. timeout = kwargs.get("timeout", self.timeout) retries = kwargs.get("retries", self.retries) @@ -89,7 +90,7 @@ async def generate_with_context( for attempt in range(retries + 1): try: result = await asyncio.wait_for( - loop.run_in_executor(None, lambda: self._run_cli(cmd, timeout)), + loop.run_in_executor(None, lambda: self._run_cli(cmd, timeout, user_content)), timeout=timeout + 30, ) return result @@ -112,10 +113,11 @@ async def generate_with_context( logger.error(f"All {retries + 1} attempts failed with error: {e}") raise - def _run_cli(self, cmd: list, timeout: int) -> str: + def _run_cli(self, cmd: list, timeout: int, prompt: Optional[str] = None) -> str: try: result = subprocess.run( cmd, + input=prompt, capture_output=True, text=True, timeout=timeout, diff --git a/tests/test_claude_code_llm.py b/tests/test_claude_code_llm.py index b33a9d62e..d803aa4a6 100644 --- a/tests/test_claude_code_llm.py +++ b/tests/test_claude_code_llm.py @@ -50,7 +50,20 @@ def test_generate_calls_cli(self, mock_run): self.assertIn("-p", cmd) self.assertIn("--model", cmd) self.assertIn("sonnet", cmd) - self.assertIn("test prompt", cmd) + self.assertNotIn("test prompt", cmd) + self.assertEqual(mock_run.call_args.kwargs["input"], "test prompt") + + @patch("openevolve.llm.claude_code.subprocess.run") + def test_large_prompt_not_passed_as_argument(self, mock_run): + # Linux caps a single argv string at 128 KiB (MAX_ARG_STRLEN); a prompt that + # carries a few parent programs easily exceeds it, so it must go through stdin. + mock_run.return_value = MagicMock(returncode=0, stdout="response", stderr="") + llm = ClaudeCodeLLM() + prompt = "x" * 200_000 + asyncio.run(llm.generate(prompt)) + cmd = mock_run.call_args[0][0] + self.assertTrue(all(len(arg) < 128 * 1024 for arg in cmd)) + self.assertEqual(mock_run.call_args.kwargs["input"], prompt) @patch("openevolve.llm.claude_code.subprocess.run") def test_system_message_passed(self, mock_run): @@ -102,8 +115,7 @@ def test_generate_with_context(self, mock_run): ) ) self.assertEqual(result, "ctx response") - cmd = mock_run.call_args[0][0] - self.assertIn("first\n\nsecond", cmd[-1]) + self.assertIn("first\n\nsecond", mock_run.call_args.kwargs["input"]) class TestMaxBudgetConfig(unittest.TestCase):