Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 5 additions & 3 deletions openevolve/llm/claude_code.py
Original file line number Diff line number Diff line change
Expand Up @@ -79,7 +79,8 @@ async def generate_with_context(
budget = kwargs.get("max_budget_usd", self.max_budget_usd)
cmd.extend(["--max-budget-usd", str(budget)])

cmd.append(user_content)
# The prompt goes through stdin: as one argv string it hits Linux's 128 KiB
# per-argument limit (MAX_ARG_STRLEN) once a few programs are in the prompt.

timeout = kwargs.get("timeout", self.timeout)
retries = kwargs.get("retries", self.retries)
Expand All @@ -89,7 +90,7 @@ async def generate_with_context(
for attempt in range(retries + 1):
try:
result = await asyncio.wait_for(
loop.run_in_executor(None, lambda: self._run_cli(cmd, timeout)),
loop.run_in_executor(None, lambda: self._run_cli(cmd, timeout, user_content)),
timeout=timeout + 30,
)
return result
Expand All @@ -112,10 +113,11 @@ async def generate_with_context(
logger.error(f"All {retries + 1} attempts failed with error: {e}")
raise

def _run_cli(self, cmd: list, timeout: int) -> str:
def _run_cli(self, cmd: list, timeout: int, prompt: Optional[str] = None) -> str:
try:
result = subprocess.run(
cmd,
input=prompt,
capture_output=True,
text=True,
timeout=timeout,
Expand Down
18 changes: 15 additions & 3 deletions tests/test_claude_code_llm.py
Original file line number Diff line number Diff line change
Expand Up @@ -50,7 +50,20 @@ def test_generate_calls_cli(self, mock_run):
self.assertIn("-p", cmd)
self.assertIn("--model", cmd)
self.assertIn("sonnet", cmd)
self.assertIn("test prompt", cmd)
self.assertNotIn("test prompt", cmd)
self.assertEqual(mock_run.call_args.kwargs["input"], "test prompt")

@patch("openevolve.llm.claude_code.subprocess.run")
def test_large_prompt_not_passed_as_argument(self, mock_run):
# Linux caps a single argv string at 128 KiB (MAX_ARG_STRLEN); a prompt that
# carries a few parent programs easily exceeds it, so it must go through stdin.
mock_run.return_value = MagicMock(returncode=0, stdout="response", stderr="")
llm = ClaudeCodeLLM()
prompt = "x" * 200_000
asyncio.run(llm.generate(prompt))
cmd = mock_run.call_args[0][0]
self.assertTrue(all(len(arg) < 128 * 1024 for arg in cmd))
self.assertEqual(mock_run.call_args.kwargs["input"], prompt)

@patch("openevolve.llm.claude_code.subprocess.run")
def test_system_message_passed(self, mock_run):
Expand Down Expand Up @@ -102,8 +115,7 @@ def test_generate_with_context(self, mock_run):
)
)
self.assertEqual(result, "ctx response")
cmd = mock_run.call_args[0][0]
self.assertIn("first\n\nsecond", cmd[-1])
self.assertIn("first\n\nsecond", mock_run.call_args.kwargs["input"])


class TestMaxBudgetConfig(unittest.TestCase):
Expand Down
Loading