diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml index e9b3a846..67698a21 100644 --- a/.github/workflows/publish.yml +++ b/.github/workflows/publish.yml @@ -43,8 +43,8 @@ jobs: - name: Install build dependencies run: | python -m pip install --upgrade pip - pip install build twine pip install -e . + pip install build twine - name: Build package run: python -m build diff --git a/requirements.txt b/requirements.txt index c408a65e..709eee69 100644 --- a/requirements.txt +++ b/requirements.txt @@ -18,14 +18,15 @@ frozenlist==1.7.0 h11==0.16.0 httpcore==1.0.9 httpx==0.28.1 -idna==3.10 +httpx2==2.13.0 +idna==3.20 iniconfig==2.1.0 -jiter==0.11.0 +jiter==0.17.0 markdown-it-py==4.0.0 mdurl==0.1.2 multidict==6.6.4 -openai==1.107.3 -packaging==25.0 +openai==3.17.0 +packaging==26.3 pexpect==4.9.0 pluggy==1.6.0 propcache==0.3.2 @@ -40,7 +41,7 @@ python-dotenv==1.1.1 python-multipart==0.0.20 PyYAML==6.0.2 requests==2.32.5 -rich==14.1.0 +rich==15.0.0 shellingham==1.5.4 six==1.17.0 sniffio==1.3.1 diff --git a/src/microbots/auto_memory/cli.py b/src/microbots/auto_memory/cli.py index a362a421..bd10f27d 100644 --- a/src/microbots/auto_memory/cli.py +++ b/src/microbots/auto_memory/cli.py @@ -55,6 +55,12 @@ def parse_args(argv: list[str] | None = None) -> argparse.Namespace: "--config-file", type=Path, help="Path to the task configuration file.", ) parser.add_argument("--max-rounds", type=int, default=5) + parser.add_argument( + "--debug-http", + action="store_true", + help="Log raw HTTP request/response details (e.g. rate-limit headers) " + "from the LLM client at DEBUG level to stderr.", + ) return parser.parse_args(argv) @@ -68,6 +74,11 @@ def main(argv: list[str] | None = None) -> None: """ args = parse_args(argv) + if args.debug_http: + http_logger = logging.getLogger("httpx2") + http_logger.setLevel(logging.DEBUG) + http_logger.addHandler(logging.StreamHandler()) + # The user can pass either an existing workdir containing a # task_config.yml file or a task_config.yml file using --config # option. In the later case, the workdir will be created in diff --git a/src/microbots/auto_memory/eval/swebenchverified.py b/src/microbots/auto_memory/eval/swebenchverified.py index 7974f46f..368de1d7 100644 --- a/src/microbots/auto_memory/eval/swebenchverified.py +++ b/src/microbots/auto_memory/eval/swebenchverified.py @@ -28,6 +28,7 @@ from microbots.tools.tool_definitions.memory_tool import MemoryTool logger = getLogger(__name__) +feedback_logger = getLogger("microbots.feedback") SWE_BENCH_VERIFIED = "SWE-bench/SWE-bench_Verified" EVAL_AGENT_MODEL_NAME = "microbots-eval-agent" @@ -501,7 +502,8 @@ def eval(self, memory_dir: str, model: str, eval_dir: str, training_repo_dir: st # root, leaving it unresettable for any instance that came after. eval_repos_path = eval_path / "eval_repo" log_dir = eval_log_dir(eval_path) - results = [] + evaluation_results = [] + instance_feedback = [] for instance in self.dataset: inst_log_path = log_dir / f"{instance.instance_id}_log.txt" @@ -509,17 +511,27 @@ def eval(self, memory_dir: str, model: str, eval_dir: str, training_repo_dir: st task = SweBenchVerifiedTask_one(instance) with log_to_file(inst_log_path): - res = task.eval(str(inst_repo_path), memory_dir, model, str(inst_log_path)) + agent_result = task.eval(str(inst_repo_path), memory_dir, model, str(inst_log_path)) - if not res.status: - logger.info(f"Evaluation failed for instance {instance.instance_id}: {res.error if res.error else 'Unknown error'}") - results.append(res) + if not agent_result.status: + harness_result = None + evaluation_result = agent_result + logger.info(f"Evaluation failed for instance {instance.instance_id}: {agent_result.error if agent_result.error else 'Unknown error'}") else: - res = task.check(str(inst_repo_path), "", str(inst_log_path)) - results.append(res) + harness_result = task.check(str(inst_repo_path), agent_result.result or "", str(inst_log_path)) + evaluation_result = harness_result + + evaluation_results.append(evaluation_result) + if not evaluation_result.status: + feedback = self._generate_instance_feedback( + instance, agent_result, harness_result, model, str(eval_path) + ) + if feedback: + instance_feedback.append(feedback) + feedback_logger.info("Generated feedback:\n%s", feedback) score = 0 - for result in results: + for result in evaluation_results: if result.status: score += 1 @@ -530,7 +542,8 @@ def eval(self, memory_dir: str, model: str, eval_dir: str, training_repo_dir: st else: combine_log_path = log_dir / "combine_result_feedback_log.txt" with log_to_file(combine_log_path): - feedback = self._combine_result_feedback(results, model, training_repo_dir) + feedback = self._combine_result_feedback(instance_feedback, model, training_repo_dir) + feedback_logger.info("Combined feedback:\n%s", feedback) # NOTE: Let's not teardown the repository as it will be useful for debugging @@ -540,13 +553,91 @@ def eval(self, memory_dir: str, model: str, eval_dir: str, training_repo_dir: st feedback = feedback ) - def _combine_result_feedback(self, results: list[BotRunResult], model: str, training_repo_dir: str) -> str: + def _generate_instance_feedback( + self, + instance: SweBenchInstance, + agent_result: BotRunResult, + harness_result: BotRunResult | None, + model: str, + eval_dir: str, + ) -> str: + """Analyze one failed instance and produce generalized feedback for it. + + Parameters + ---------- + instance : SweBenchInstance + The failed instance being analyzed. + agent_result : BotRunResult + The agent's own run result for this instance. + harness_result : BotRunResult | None + The independent harness verdict, or None if the agent itself + did not complete. + model : str + The model to use, in the format ``/``. + eval_dir : str + Directory containing this instance's checkout and log file. + + Returns + ------- + str + Generalized feedback text, or an empty string if generation + failed. + """ + task_results = json.dumps({ + "instance_id": instance.instance_id, + "task": instance.problem_statement, + "agent_result": { + "status": agent_result.status, + "result": agent_result.result, + "error": agent_result.error, + }, + "harness_result": None if harness_result is None else { + "status": harness_result.status, + "result": harness_result.result, + "error": harness_result.error, + }, + }) + + try: + bot = ReadingBot(model=model, folder_to_mount=eval_dir) + bot_result = bot.run(task=f""" + Analyze failed SWE-bench instance {instance.instance_id}. Inspect its checkout at + eval_repo/{instance.instance_id}, its log at logs/{instance.instance_id}_log.txt, and any + relevant repository context. Analyze why the attempt failed and then generate a feedback + that will be used to create context for FUTURE agents, solving DIFFERENT future tasks more effectively. + + The agent result records whether the agent completed and what it reported. The harness result + records the independent test verdict, or is null when the agent did not complete: + + {task_results} + + Write concise feedback for a separate agent that will investigate the repository and update + memory. Do not write the memory or prescribe its contents to any specificity. + Consider Feedback on these: + - Mistakes to avoid: what went wrong and how a future agent should approach it better. + - Useful repository context: what the agent needed to know or find about the relevant files, + functions, behavior, environment, or development workflow. + + The feedback must NOT reveal or reconstruct this evaluation instance. Do not include its ID, task, + result, tests, inputs, attempted patch, or specific fix. Keep only reusable lessons and + repository context that are generalized, and would be helpful even for future, different tasks. + """) + except Exception as exc: + logger.warning("Generating instance feedback failed with exception: %s", exc) + return "" + + if not bot_result.status or not bot_result.result: + logger.warning("Generating instance feedback failed: %s", bot_result.error) + return "" + return bot_result.result + + def _combine_result_feedback(self, feedback_items: list[str], model: str, training_repo_dir: str) -> str: """Summarize every instance's result into one feedback string. Parameters ---------- - results : list[BotRunResult] - One result per attempted instance. + feedback_items : list[str] + Generic diagnostic feedback from each failed instance. model : str The model to use, in the format ``/``. training_repo_dir : str @@ -560,12 +651,9 @@ def _combine_result_feedback(self, results: list[BotRunResult], model: str, trai results if the bot is unavailable or fails. """ - serialized_str = f"Total {len(results)} tests ran and their result and feedback:\n" - - for res in results: - serialized_str += f"\nResult: {'Passed' if res.status else 'Failed'}\n" - serialized_str += f"Optional Feedback: {res.result if res.result else 'None'}\n" - serialized_str += f"Error if there are any: {res.error if res.error else 'None'}\n" + serialized_str = "\n\n".join(feedback_items) + if not serialized_str: + return "" try: bot = ReadingBot( @@ -573,45 +661,39 @@ def _combine_result_feedback(self, results: list[BotRunResult], model: str, trai folder_to_mount=training_repo_dir ) task = f""" - You are combining results from {len(results)} SWE-bench evaluation - runs into ONE feedback report for the next training iteration. The - training agent will read your report to decide what to add or fix - in its memory notes. - - For each result below, note whether it passed or failed. For each - failure, briefly identify the underlying cause (e.g. wrong - file/line targeted, incorrect patch logic, response format error, - timeout) rather than only quoting the raw error. You may open - files under the mounted repo if you need to confirm a root - cause, but do not turn this into a debugging session. - Do not refer to any specific instance or test case by name/ID — - describe causes and guidance in general terms only. + You are combining {len(feedback_items)} pieces of feedback from failed SWE-bench + evaluation runs into ONE feedback report for the next training iteration. The + training agent will read your report to decide what to add or fix in its memory + notes. + + Each item below is already a generalized analysis of one failed run (mistakes to + avoid, useful repository context). Deduplicate and group items that share the same + root cause or lesson rather than repeating yourself. Then write a report with: - 1. A one-line summary: how many passed vs failed. - 2. Grouped failure patterns: if multiple failures share the same - root cause, describe that cause once rather than repeating - yourself. - 3. Concrete, actionable guidance for the training agent — say - what to change in the memory notes to avoid each failure - pattern next time. Be specific and imperative - (e.g. "Record that config paths must be normalized before + 1. A one-line summary: how many failures were analyzed. + 2. Grouped failure patterns: describe each shared root cause once rather than + repeating yourself. + 3. Concrete, actionable guidance for the training agent — say what to change in + the memory notes to avoid each failure pattern next time. Be specific and + imperative (e.g. "Record that config paths must be normalized before comparison", not "there was a path issue"). - 4. Skip anything about passed cases beyond the summary count; - don't restate their feedback. - Keep the report tight and skimmable — short paragraphs or bullet - points, no code dumps. Put the final report in the `result` - field once you set task_done=true. + Do not reveal or reconstruct any specific evaluated task, test, input, attempted + patch, or fix. Do not build memory yourself; only provide the feedback that will + guide the next agent. + + Keep the report tight and skimmable — short paragraphs or bullet points, no code + dumps. Put the final report in the `result` field once you set task_done=true. {serialized_str} """ bot_result = bot.run(task=task) except Exception as e: logger.warning(f"Combining results failed with exception: {e}") - return f"Combining results failed. raw combined output:\n\n{serialized_str}" + return serialized_str if bot_result.status: - return bot_result.result if bot_result.result else 'No feedback provided' + return bot_result.result if bot_result.result else serialized_str else: - return f"Combining results failed. raw combined output:\n\n{serialized_str}" + return serialized_str diff --git a/src/microbots/llm/azure_openai_api.py b/src/microbots/llm/azure_openai_api.py index 7424df85..9f12905a 100644 --- a/src/microbots/llm/azure_openai_api.py +++ b/src/microbots/llm/azure_openai_api.py @@ -1,7 +1,9 @@ +"""Azure OpenAI Responses API client implementing the LLMInterface.""" import json import os from collections.abc import Callable from dataclasses import asdict +from logging import getLogger from dotenv import load_dotenv from openai import AzureOpenAI @@ -9,16 +11,63 @@ load_dotenv() +logger = getLogger(__name__) + endpoint = os.getenv("AZURE_OPENAI_ENDPOINT") api_version = os.getenv("AZURE_OPENAI_API_VERSION") deployment_name = os.getenv("AZURE_OPENAI_DEPLOYMENT_NAME") api_key = os.getenv("AZURE_OPENAI_API_KEY") +DEFAULT_COMPACT_THRESHOLD = 200_000 + class AzureOpenAIApi(LLMInterface): + """ + LLM client backed by the Azure OpenAI Responses API. + + Parameters + ---------- + system_prompt : str + System prompt to seed the conversation. + deployment_name : str + Azure OpenAI deployment name. + max_retries : int + Max retries on invalid LLM responses. + token_provider : Callable[[], str] | None + Optional callable returning an Azure AD bearer token, used + instead of AZURE_OPENAI_API_KEY when provided. + compact_threshold : int | None + Token threshold that triggers server-side compaction, or None + to disable it. + """ def __init__(self, system_prompt, deployment_name=deployment_name, max_retries=3, - token_provider: Callable[[], str] | None = None): + token_provider: Callable[[], str] | None = None, + compact_threshold: int | None = DEFAULT_COMPACT_THRESHOLD): + """ + Create the client and seed the conversation. + + Parameters + ---------- + system_prompt : str + System prompt to seed the conversation. + deployment_name : str + Azure OpenAI deployment name. + max_retries : int + Max retries on invalid LLM responses. + token_provider : Callable[[], str] | None + Optional callable returning an Azure AD bearer token, used + instead of AZURE_OPENAI_API_KEY when provided. + compact_threshold : int | None + Token threshold that triggers server-side compaction, or + None to disable it. + + Raises + ------ + ValueError + If required Azure OpenAI configuration or authentication + is missing or invalid. + """ self.token_provider = token_provider if not endpoint: @@ -63,32 +112,75 @@ def __init__(self, system_prompt, deployment_name=deployment_name, max_retries=3 self.deployment_name = deployment_name self.system_prompt = system_prompt self.messages = [{"role": "system", "content": system_prompt}] + self.compact_threshold = compact_threshold # Set these values here. This logic will be handled in the parent class. self.max_retries = max_retries self.retries = 0 def ask(self, message) -> LLMAskResponse: + """ + Send a message to the LLM and return its parsed response. + + Parameters + ---------- + message : str + The message/prompt to send to the LLM. + + Returns + ------- + LLMAskResponse + The parsed LLM response. + """ self.retries = 0 # reset retries for each ask. Handled in parent class. self.messages.append({"role": "user", "content": message}) valid = False while not valid: - response = self.ai_client.responses.create( - model=self.deployment_name, - input=self.messages, - ) + create_kwargs = { + "model": self.deployment_name, + "input": self.messages, + } + if self.compact_threshold is not None: + create_kwargs["context_management"] = [ + {"type": "compaction", "compact_threshold": self.compact_threshold} + ] + + response = self.ai_client.responses.create(**create_kwargs) + self._log_token_usage(response) self.messages.append({"role": "assistant", "content": response.output_text}) valid, askResponse = self._validate_llm_response(response=response.output_text) # Remove last assistant message and replace with structured response self.messages.pop() - self.messages.append({"role": "assistant", "content": json.dumps(asdict(askResponse))}) + assistant_message = {"role": "assistant", "content": json.dumps(asdict(askResponse))} + self.messages.append(assistant_message) + + compaction_item = self._extract_compaction_item(response) + if compaction_item is not None: + logger.info( + "Azure OpenAI compaction occurred: id=%s, self.messages reset to 4 items", + compaction_item.get("id"), + ) + self.messages = [ + {"role": "system", "content": self.system_prompt}, + compaction_item, + {"role": "user", "content": message}, + assistant_message, + ] return askResponse def clear_history(self): + """ + Clear the LLM's conversation history. + + Returns + ------- + bool + True if the history was cleared successfully. + """ self.messages = [ { "role": "system", @@ -97,3 +189,51 @@ def clear_history(self): ] return True + def _log_token_usage(self, response) -> None: + """ + Log token usage reported by the Responses API for this call. + + Parameters + ---------- + response : openai.types.responses.Response + The response object returned by ``responses.create``. + """ + usage = getattr(response, "usage", None) + if usage is None: + logger.warning("Azure OpenAI response did not include token usage information.") + return + + logger.info( + "Azure OpenAI token usage: input=%s output=%s total=%s", + getattr(usage, "input_tokens", None), + getattr(usage, "output_tokens", None), + getattr(usage, "total_tokens", None), + ) + + def _extract_compaction_item(self, response) -> dict | None: + """ + Find the compaction item in a response, if the server ran one. + + Parameters + ---------- + response : openai.types.responses.Response + The response object returned by ``responses.create``. + + Returns + ------- + dict | None + An input-ready compaction item dict, or None if the + response did not include one. + """ + output = getattr(response, "output", None) + if not isinstance(output, list): + return None + for item in output: + if getattr(item, "type", None) == "compaction": + return { + "type": "compaction", + "id": getattr(item, "id", None), + "encrypted_content": getattr(item, "encrypted_content", None), + } + return None + diff --git a/src/microbots/llm/openai_api.py b/src/microbots/llm/openai_api.py index 258a0055..d6518c07 100644 --- a/src/microbots/llm/openai_api.py +++ b/src/microbots/llm/openai_api.py @@ -1,6 +1,8 @@ +"""OpenAI Responses API client implementing the LLMInterface.""" import json import os from dataclasses import asdict +from logging import getLogger from dotenv import load_dotenv from openai import OpenAI @@ -8,13 +10,53 @@ load_dotenv() +logger = getLogger(__name__) + endpoint = os.getenv("OPENAI_ENDPOINT", "https://api.openai.com/v1") api_key = os.getenv("OPENAI_API_KEY") +DEFAULT_COMPACT_THRESHOLD = 200_000 + class OpenAIApi(LLMInterface): + """ + LLM client backed by the OpenAI Responses API. + + Parameters + ---------- + system_prompt : str + System prompt to seed the conversation. + deployment_name : str + OpenAI model name (e.g. 'gpt-4'). + max_retries : int + Max retries on invalid LLM responses. + compact_threshold : int | None + Token threshold that triggers server-side compaction, or None + to disable it. + """ + + def __init__(self, system_prompt, deployment_name, max_retries=3, + compact_threshold: int | None = DEFAULT_COMPACT_THRESHOLD): + """ + Create the client and seed the conversation. + + Parameters + ---------- + system_prompt : str + System prompt to seed the conversation. + deployment_name : str + OpenAI model name (e.g. 'gpt-4'). + max_retries : int + Max retries on invalid LLM responses. + compact_threshold : int | None + Token threshold that triggers server-side compaction, or + None to disable it. - def __init__(self, system_prompt, deployment_name, max_retries=3): + Raises + ------ + ValueError + If OPENAI_API_KEY is not set. + """ if not api_key: raise ValueError( "No authentication configured for OpenAI. " @@ -28,31 +70,74 @@ def __init__(self, system_prompt, deployment_name, max_retries=3): self.deployment_name = deployment_name self.system_prompt = system_prompt self.messages = [{"role": "system", "content": system_prompt}] + self.compact_threshold = compact_threshold self.max_retries = max_retries self.retries = 0 def ask(self, message) -> LLMAskResponse: + """ + Send a message to the LLM and return its parsed response. + + Parameters + ---------- + message : str + The message/prompt to send to the LLM. + + Returns + ------- + LLMAskResponse + The parsed LLM response. + """ self.retries = 0 self.messages.append({"role": "user", "content": message}) valid = False while not valid: - response = self.ai_client.responses.create( - model=self.deployment_name, - input=self.messages, - ) + create_kwargs = { + "model": self.deployment_name, + "input": self.messages, + } + if self.compact_threshold is not None: + create_kwargs["context_management"] = [ + {"type": "compaction", "compact_threshold": self.compact_threshold} + ] + + response = self.ai_client.responses.create(**create_kwargs) + self._log_token_usage(response) self.messages.append({"role": "assistant", "content": response.output_text}) valid, askResponse = self._validate_llm_response(response=response.output_text) # Remove last assistant message and replace with structured response self.messages.pop() - self.messages.append({"role": "assistant", "content": json.dumps(asdict(askResponse))}) + assistant_message = {"role": "assistant", "content": json.dumps(asdict(askResponse))} + self.messages.append(assistant_message) + + compaction_item = self._extract_compaction_item(response) + if compaction_item is not None: + logger.info( + "OpenAI compaction occurred: id=%s, self.messages reset to 4 items", + compaction_item.get("id"), + ) + self.messages = [ + {"role": "system", "content": self.system_prompt}, + compaction_item, + {"role": "user", "content": message}, + assistant_message, + ] return askResponse def clear_history(self): + """ + Clear the LLM's conversation history. + + Returns + ------- + bool + True if the history was cleared successfully. + """ self.messages = [ { "role": "system", @@ -60,3 +145,51 @@ def clear_history(self): } ] return True + + def _log_token_usage(self, response) -> None: + """ + Log token usage reported by the Responses API for this call. + + Parameters + ---------- + response : openai.types.responses.Response + The response object returned by ``responses.create``. + """ + usage = getattr(response, "usage", None) + if usage is None: + logger.warning("OpenAI response did not include token usage information.") + return + + logger.info( + "OpenAI token usage: input=%s output=%s total=%s", + getattr(usage, "input_tokens", None), + getattr(usage, "output_tokens", None), + getattr(usage, "total_tokens", None), + ) + + def _extract_compaction_item(self, response) -> dict | None: + """ + Find the compaction item in a response, if the server ran one. + + Parameters + ---------- + response : openai.types.responses.Response + The response object returned by ``responses.create``. + + Returns + ------- + dict | None + An input-ready compaction item dict, or None if the + response did not include one. + """ + output = getattr(response, "output", None) + if not isinstance(output, list): + return None + for item in output: + if getattr(item, "type", None) == "compaction": + return { + "type": "compaction", + "id": getattr(item, "id", None), + "encrypted_content": getattr(item, "encrypted_content", None), + } + return None diff --git a/test/auto_memory/eval/test_swebenchverified.py b/test/auto_memory/eval/test_swebenchverified.py index d07f569e..d8b34db8 100644 --- a/test/auto_memory/eval/test_swebenchverified.py +++ b/test/auto_memory/eval/test_swebenchverified.py @@ -206,9 +206,85 @@ def test_eval_skips_the_harness_when_the_agent_itself_failed(tmp_path): @pytest.mark.unit def test_combined_feedback_falls_back_to_raw_results_when_the_bot_fails(tmp_path): task = _task(tmp_path, instance_id_list=["django__django-11099"]) - results = [BotRunResult(status=False, result="not resolved", error="assertion failed")] + feedback_items = ["Instance failed: assertion failed"] with patch(f"{MODULE}.ReadingBot", side_effect=RuntimeError("no model configured")): - feedback = task._combine_result_feedback(results, "azure-openai/gpt-4o", "/repo") + feedback = task._combine_result_feedback(feedback_items, "azure-openai/gpt-4o", "/repo") assert "assertion failed" in feedback + + +@pytest.mark.unit +def test_combined_feedback_is_empty_when_there_is_nothing_to_combine(tmp_path): + task = _task(tmp_path, instance_id_list=["django__django-11099"]) + + with patch(f"{MODULE}.ReadingBot") as reading_bot: + feedback = task._combine_result_feedback([], "azure-openai/gpt-4o", "/repo") + + assert feedback == "" + reading_bot.assert_not_called() + + +@pytest.mark.unit +def test_combined_feedback_falls_back_to_raw_results_when_the_bot_returns_a_failure(tmp_path): + task = _task(tmp_path, instance_id_list=["django__django-11099"]) + feedback_items = ["Instance failed: assertion failed"] + + with patch(f"{MODULE}.ReadingBot") as reading_bot: + reading_bot.return_value.run.return_value = BotRunResult( + status=False, result=None, error="bot gave up" + ) + feedback = task._combine_result_feedback(feedback_items, "azure-openai/gpt-4o", "/repo") + + assert feedback == "Instance failed: assertion failed" + + +# --------------------------------------------------------------------------- +# _generate_instance_feedback +# --------------------------------------------------------------------------- + +@pytest.mark.unit +def test_instance_feedback_returns_the_bot_report(tmp_path): + task = _task(tmp_path, instance_id_list=["django__django-11099"]) + agent_result = BotRunResult(status=True, result="patched", error=None) + harness_result = BotRunResult(status=False, result="not resolved", error="tests failed") + + with patch(f"{MODULE}.ReadingBot") as reading_bot: + reading_bot.return_value.run.return_value = BotRunResult( + status=True, result="check the settings module first", error=None + ) + feedback = task._generate_instance_feedback( + INSTANCE, agent_result, harness_result, "azure-openai/gpt-4o", str(tmp_path) + ) + + assert feedback == "check the settings module first" + + +@pytest.mark.unit +def test_instance_feedback_is_empty_when_the_bot_is_unavailable(tmp_path): + task = _task(tmp_path, instance_id_list=["django__django-11099"]) + agent_result = BotRunResult(status=False, result=None, error="agent timed out") + + with patch(f"{MODULE}.ReadingBot", side_effect=RuntimeError("no model configured")): + feedback = task._generate_instance_feedback( + INSTANCE, agent_result, None, "azure-openai/gpt-4o", str(tmp_path) + ) + + assert feedback == "" + + +@pytest.mark.unit +def test_instance_feedback_is_empty_when_the_bot_fails(tmp_path): + task = _task(tmp_path, instance_id_list=["django__django-11099"]) + agent_result = BotRunResult(status=True, result="patched", error=None) + harness_result = BotRunResult(status=False, result="not resolved", error="tests failed") + + with patch(f"{MODULE}.ReadingBot") as reading_bot: + reading_bot.return_value.run.return_value = BotRunResult( + status=False, result=None, error="bot crashed" + ) + feedback = task._generate_instance_feedback( + INSTANCE, agent_result, harness_result, "azure-openai/gpt-4o", str(tmp_path) + ) + + assert feedback == "" diff --git a/test/auto_memory/test_cli.py b/test/auto_memory/test_cli.py index be912cff..dd085a40 100644 --- a/test/auto_memory/test_cli.py +++ b/test/auto_memory/test_cli.py @@ -1,5 +1,6 @@ """Unit tests for microbots.auto_memory.cli.""" +import logging from pathlib import Path from unittest.mock import MagicMock, patch @@ -76,3 +77,37 @@ def test_main_fails_fast_when_no_config_file_exists(tmp_path): "--task", "swebenchverified", "--workdir", str(workdir), ]) + + +@pytest.mark.unit +def test_parse_args_debug_http_defaults_to_false(): + args = parse_args(["--model", "azure-openai/gpt-4o", "--task", "swebenchverified"]) + + assert args.debug_http is False + + +@pytest.mark.unit +def test_main_enables_http_debug_logging_when_requested(tmp_path): + workdir = tmp_path / "workdir" + workdir.mkdir() + (workdir / "task_config.yaml").write_text("repo: django/django") + + http_logger = logging.getLogger("httpx2") + original_level = http_logger.level + original_handlers = list(http_logger.handlers) + + try: + with patch.dict(f"{MODULE}.TASK_REGISTRY", {"swebenchverified": MagicMock()}, clear=True), \ + patch(f"{MODULE}.run"): + main([ + "--model", "azure-openai/gpt-4o", + "--task", "swebenchverified", + "--workdir", str(workdir), + "--debug-http", + ]) + + assert http_logger.level == logging.DEBUG + assert len(http_logger.handlers) == len(original_handlers) + 1 + finally: + http_logger.setLevel(original_level) + http_logger.handlers = original_handlers diff --git a/test/llm/test_azure_openai_api.py b/test/llm/test_azure_openai_api.py index 69300c93..2ec6c326 100644 --- a/test/llm/test_azure_openai_api.py +++ b/test/llm/test_azure_openai_api.py @@ -452,3 +452,139 @@ def test_multiple_ask_calls_append_messages(self): assert user_messages[0]["content"] == "First question" assert user_messages[1]["content"] == "Second question" assert user_messages[2]["content"] == "Third question" + + +@pytest.mark.unit +class TestAzureOpenAIApiTokenUsageLogging: + """Tests for token usage logging""" + + def test_ask_logs_token_usage(self, caplog): + """Token usage from response.usage is logged after each call""" + api = AzureOpenAIApi(system_prompt="test") + + mock_response = Mock() + mock_response.output_text = json.dumps({ + "task_done": False, "command": "cmd", "thoughts": None + }) + mock_response.usage = Mock(input_tokens=12, output_tokens=34, total_tokens=46) + api.ai_client.responses.create = Mock(return_value=mock_response) + + with caplog.at_level("INFO"): + api.ask("hello") + + assert any("input=12" in r.message and "output=34" in r.message and "total=46" in r.message + for r in caplog.records) + + def test_ask_warns_when_token_usage_is_missing(self, caplog): + """A warning is logged when the response carries no usage information""" + api = AzureOpenAIApi(system_prompt="test") + + mock_response = Mock() + mock_response.output_text = json.dumps({ + "task_done": False, "command": "cmd", "thoughts": None + }) + mock_response.usage = None + api.ai_client.responses.create = Mock(return_value=mock_response) + + with caplog.at_level("WARNING"): + api.ask("hello") + + assert any("did not include token usage" in r.message for r in caplog.records) + + +@pytest.mark.unit +class TestAzureOpenAIApiCompaction: + """Tests for compaction request wiring and history reset""" + + def _mock_response(self, output=None): + mock_response = Mock() + mock_response.output_text = json.dumps({ + "task_done": False, "command": "cmd", "thoughts": None + }) + mock_response.usage = Mock(input_tokens=1, output_tokens=1, total_tokens=2) + mock_response.output = output if output is not None else [] + return mock_response + + def test_context_management_sent_with_default_threshold(self): + """context_management is sent using DEFAULT_COMPACT_THRESHOLD by default""" + from microbots.llm.azure_openai_api import DEFAULT_COMPACT_THRESHOLD + + api = AzureOpenAIApi(system_prompt="test") + api.ai_client.responses.create = Mock(return_value=self._mock_response()) + + api.ask("hello") + + _, kwargs = api.ai_client.responses.create.call_args + assert kwargs["context_management"] == [ + {"type": "compaction", "compact_threshold": DEFAULT_COMPACT_THRESHOLD} + ] + + def test_context_management_sent_with_custom_threshold(self): + """context_management uses a custom compact_threshold when provided""" + api = AzureOpenAIApi(system_prompt="test", compact_threshold=500) + api.ai_client.responses.create = Mock(return_value=self._mock_response()) + + api.ask("hello") + + _, kwargs = api.ai_client.responses.create.call_args + assert kwargs["context_management"] == [ + {"type": "compaction", "compact_threshold": 500} + ] + + def test_context_management_omitted_when_disabled(self): + """context_management is not sent when compact_threshold is None""" + api = AzureOpenAIApi(system_prompt="test", compact_threshold=None) + api.ai_client.responses.create = Mock(return_value=self._mock_response()) + + api.ask("hello") + + _, kwargs = api.ai_client.responses.create.call_args + assert "context_management" not in kwargs + + def test_extract_compaction_item_returns_none_without_compaction(self): + """_extract_compaction_item returns None when no compaction item is present""" + api = AzureOpenAIApi(system_prompt="test") + + other_item = Mock(type="message") + assert api._extract_compaction_item(self._mock_response(output=[other_item])) is None + + def test_extract_compaction_item_parses_compaction(self): + """_extract_compaction_item returns a plain input-ready dict""" + api = AzureOpenAIApi(system_prompt="test") + + compaction_output_item = Mock(type="compaction", id="comp_123", encrypted_content="abc") + result = api._extract_compaction_item(self._mock_response(output=[compaction_output_item])) + + assert result == {"type": "compaction", "id": "comp_123", "encrypted_content": "abc"} + + def test_ask_does_not_reset_messages_without_compaction(self): + """self.messages keeps growing normally when no compaction occurs""" + api = AzureOpenAIApi(system_prompt="test") + api.ai_client.responses.create = Mock(return_value=self._mock_response()) + + api.ask("hello") + api.ask("hello again") + + # system + 2x(user + assistant) + assert len(api.messages) == 5 + + def test_ask_resets_messages_when_compaction_present(self): + """self.messages is reset to [system, compaction_item, user, assistant] after compaction""" + api = AzureOpenAIApi(system_prompt="test") + + compaction_output_item = Mock(type="compaction", id="comp_123", encrypted_content="abc") + api.ai_client.responses.create = Mock( + return_value=self._mock_response(output=[compaction_output_item]) + ) + + api.ask("first message") + api.ask("second message") + + assert len(api.messages) == 4 + assert api.messages[0] == {"role": "system", "content": "test"} + assert api.messages[1] == { + "type": "compaction", "id": "comp_123", "encrypted_content": "abc" + } + assert api.messages[2]["role"] == "user" + assert api.messages[2]["content"] == "second message" + assert api.messages[3]["role"] == "assistant" diff --git a/test/llm/test_openai_api.py b/test/llm/test_openai_api.py index 442471a8..86da6177 100644 --- a/test/llm/test_openai_api.py +++ b/test/llm/test_openai_api.py @@ -291,3 +291,154 @@ def test_multiple_ask_calls_append_messages(self): # system + (user + assistant) * 2 = 5 assert len(api.messages) == 5 + + +@pytest.mark.unit +class TestOpenAIApiTokenUsageLogging: + """Tests for token usage logging""" + + def test_ask_logs_token_usage(self, caplog): + """Token usage from response.usage is logged after each call""" + api = OpenAIApi(system_prompt="test", deployment_name="gpt-4") + + mock_response = Mock() + mock_response.output_text = json.dumps({ + "task_done": False, "command": "cmd", "thoughts": None + }) + mock_response.usage = Mock(input_tokens=12, output_tokens=34, total_tokens=46) + api.ai_client.responses.create = Mock(return_value=mock_response) + + with caplog.at_level("INFO"): + api.ask("hello") + + assert any("input=12" in r.message and "output=34" in r.message and "total=46" in r.message + for r in caplog.records) + + def test_ask_warns_when_token_usage_is_missing(self, caplog): + """A warning is logged when the response carries no usage information""" + api = OpenAIApi(system_prompt="test", deployment_name="gpt-4") + + mock_response = Mock() + mock_response.output_text = json.dumps({ + "task_done": False, "command": "cmd", "thoughts": None + }) + mock_response.usage = None + api.ai_client.responses.create = Mock(return_value=mock_response) + + with caplog.at_level("WARNING"): + api.ask("hello") + + assert any("did not include token usage" in r.message for r in caplog.records) + + +@pytest.mark.unit +class TestOpenAIApiCompaction: + """Tests for compaction request wiring and history reset""" + + def _mock_response(self, output=None): + mock_response = Mock() + mock_response.output_text = json.dumps({ + "task_done": False, "command": "cmd", "thoughts": None + }) + mock_response.usage = Mock(input_tokens=1, output_tokens=1, total_tokens=2) + mock_response.output = output if output is not None else [] + return mock_response + + def test_context_management_sent_with_default_threshold(self): + """context_management is sent using DEFAULT_COMPACT_THRESHOLD by default""" + from microbots.llm.openai_api import DEFAULT_COMPACT_THRESHOLD + + api = OpenAIApi(system_prompt="test", deployment_name="gpt-4") + api.ai_client.responses.create = Mock(return_value=self._mock_response()) + + api.ask("hello") + + _, kwargs = api.ai_client.responses.create.call_args + assert kwargs["context_management"] == [ + {"type": "compaction", "compact_threshold": DEFAULT_COMPACT_THRESHOLD} + ] + + def test_context_management_sent_with_custom_threshold(self): + """context_management uses a custom compact_threshold when provided""" + api = OpenAIApi(system_prompt="test", deployment_name="gpt-4", compact_threshold=500) + api.ai_client.responses.create = Mock(return_value=self._mock_response()) + + api.ask("hello") + + _, kwargs = api.ai_client.responses.create.call_args + assert kwargs["context_management"] == [ + {"type": "compaction", "compact_threshold": 500} + ] + + def test_context_management_omitted_when_disabled(self): + """context_management is not sent when compact_threshold is None""" + api = OpenAIApi(system_prompt="test", deployment_name="gpt-4", compact_threshold=None) + api.ai_client.responses.create = Mock(return_value=self._mock_response()) + + api.ask("hello") + + _, kwargs = api.ai_client.responses.create.call_args + assert "context_management" not in kwargs + + def test_extract_compaction_item_returns_none_without_compaction(self): + """_extract_compaction_item returns None when no compaction item is present""" + api = OpenAIApi(system_prompt="test", deployment_name="gpt-4") + + other_item = Mock(type="message") + assert api._extract_compaction_item(self._mock_response(output=[other_item])) is None + + def test_extract_compaction_item_parses_compaction(self): + """_extract_compaction_item returns a plain input-ready dict""" + api = OpenAIApi(system_prompt="test", deployment_name="gpt-4") + + compaction_output_item = Mock(type="compaction", id="comp_123", encrypted_content="abc") + result = api._extract_compaction_item(self._mock_response(output=[compaction_output_item])) + + assert result == {"type": "compaction", "id": "comp_123", "encrypted_content": "abc"} + + def test_ask_does_not_reset_messages_without_compaction(self): + """self.messages keeps growing normally when no compaction occurs""" + api = OpenAIApi(system_prompt="test", deployment_name="gpt-4") + api.ai_client.responses.create = Mock(return_value=self._mock_response()) + + api.ask("hello") + api.ask("hello again") + + # system + 2x(user + assistant) + assert len(api.messages) == 5 + + def test_ask_resets_messages_when_compaction_present(self): + """self.messages is reset to [system, compaction_item, user, assistant] after compaction""" + api = OpenAIApi(system_prompt="test", deployment_name="gpt-4") + + compaction_output_item = Mock(type="compaction", id="comp_123", encrypted_content="abc") + api.ai_client.responses.create = Mock( + return_value=self._mock_response(output=[compaction_output_item]) + ) + + api.ask("first message") + api.ask("second message") + + assert len(api.messages) == 4 + assert api.messages[0] == {"role": "system", "content": "test"} + assert api.messages[1] == { + "type": "compaction", "id": "comp_123", "encrypted_content": "abc" + } + assert api.messages[2]["role"] == "user" + assert api.messages[2]["content"] == "second message" + assert api.messages[3]["role"] == "assistant" + + def test_ask_logs_when_compaction_occurs(self, caplog): + """A log line identifies when compaction occurred and its item id""" + api = OpenAIApi(system_prompt="test", deployment_name="gpt-4") + + compaction_output_item = Mock(type="compaction", id="comp_123", encrypted_content="abc") + api.ai_client.responses.create = Mock( + return_value=self._mock_response(output=[compaction_output_item]) + ) + + with caplog.at_level("INFO"): + api.ask("hello") + + assert any("compaction occurred" in r.message and "comp_123" in r.message + for r in caplog.records)