Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/workflows/publish.yml
Original file line number Diff line number Diff line change
Expand Up @@ -43,8 +43,8 @@ jobs:
- name: Install build dependencies
run: |
python -m pip install --upgrade pip
pip install build twine
pip install -e .
pip install build twine

- name: Build package
run: python -m build
Expand Down
11 changes: 6 additions & 5 deletions requirements.txt
Original file line number Diff line number Diff line change
Expand Up @@ -18,14 +18,15 @@ frozenlist==1.7.0
h11==0.16.0
httpcore==1.0.9
httpx==0.28.1
idna==3.10
httpx2==2.13.0
idna==3.20
iniconfig==2.1.0
jiter==0.11.0
jiter==0.17.0
markdown-it-py==4.0.0
mdurl==0.1.2
multidict==6.6.4
openai==1.107.3
packaging==25.0
openai==3.17.0
packaging==26.3
pexpect==4.9.0
pluggy==1.6.0
propcache==0.3.2
Expand All @@ -40,7 +41,7 @@ python-dotenv==1.1.1
python-multipart==0.0.20
PyYAML==6.0.2
requests==2.32.5
rich==14.1.0
rich==15.0.0
shellingham==1.5.4
six==1.17.0
sniffio==1.3.1
Expand Down
11 changes: 11 additions & 0 deletions src/microbots/auto_memory/cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -55,6 +55,12 @@ def parse_args(argv: list[str] | None = None) -> argparse.Namespace:
"--config-file", type=Path, help="Path to the task configuration file.",
)
parser.add_argument("--max-rounds", type=int, default=5)
parser.add_argument(
"--debug-http",
action="store_true",
help="Log raw HTTP request/response details (e.g. rate-limit headers) "
"from the LLM client at DEBUG level to stderr.",
)

return parser.parse_args(argv)

Expand All @@ -68,6 +74,11 @@ def main(argv: list[str] | None = None) -> None:
"""
args = parse_args(argv)

if args.debug_http:
http_logger = logging.getLogger("httpx2")
http_logger.setLevel(logging.DEBUG)
http_logger.addHandler(logging.StreamHandler())

# The user can pass either an existing workdir containing a
# task_config.yml file or a task_config.yml file using --config
# option. In the later case, the workdir will be created in
Expand Down
176 changes: 129 additions & 47 deletions src/microbots/auto_memory/eval/swebenchverified.py
Original file line number Diff line number Diff line change
Expand Up @@ -28,6 +28,7 @@
from microbots.tools.tool_definitions.memory_tool import MemoryTool

logger = getLogger(__name__)
feedback_logger = getLogger("microbots.feedback")

SWE_BENCH_VERIFIED = "SWE-bench/SWE-bench_Verified"
EVAL_AGENT_MODEL_NAME = "microbots-eval-agent"
Expand Down Expand Up @@ -501,25 +502,36 @@ def eval(self, memory_dir: str, model: str, eval_dir: str, training_repo_dir: st
# root, leaving it unresettable for any instance that came after.
eval_repos_path = eval_path / "eval_repo"
log_dir = eval_log_dir(eval_path)
results = []
evaluation_results = []
instance_feedback = []

for instance in self.dataset:
inst_log_path = log_dir / f"{instance.instance_id}_log.txt"
inst_repo_path = eval_repos_path / instance.instance_id
task = SweBenchVerifiedTask_one(instance)

with log_to_file(inst_log_path):
res = task.eval(str(inst_repo_path), memory_dir, model, str(inst_log_path))
agent_result = task.eval(str(inst_repo_path), memory_dir, model, str(inst_log_path))

if not res.status:
logger.info(f"Evaluation failed for instance {instance.instance_id}: {res.error if res.error else 'Unknown error'}")
results.append(res)
if not agent_result.status:
harness_result = None
evaluation_result = agent_result
logger.info(f"Evaluation failed for instance {instance.instance_id}: {agent_result.error if agent_result.error else 'Unknown error'}")
else:
res = task.check(str(inst_repo_path), "", str(inst_log_path))
results.append(res)
harness_result = task.check(str(inst_repo_path), agent_result.result or "", str(inst_log_path))
evaluation_result = harness_result

evaluation_results.append(evaluation_result)
if not evaluation_result.status:
feedback = self._generate_instance_feedback(
instance, agent_result, harness_result, model, str(eval_path)
)
if feedback:
instance_feedback.append(feedback)
feedback_logger.info("Generated feedback:\n%s", feedback)

score = 0
for result in results:
for result in evaluation_results:
if result.status:
score += 1

Expand All @@ -530,7 +542,8 @@ def eval(self, memory_dir: str, model: str, eval_dir: str, training_repo_dir: st
else:
combine_log_path = log_dir / "combine_result_feedback_log.txt"
with log_to_file(combine_log_path):
feedback = self._combine_result_feedback(results, model, training_repo_dir)
feedback = self._combine_result_feedback(instance_feedback, model, training_repo_dir)
feedback_logger.info("Combined feedback:\n%s", feedback)

# NOTE: Let's not teardown the repository as it will be useful for debugging

Expand All @@ -540,13 +553,91 @@ def eval(self, memory_dir: str, model: str, eval_dir: str, training_repo_dir: st
feedback = feedback
)

def _combine_result_feedback(self, results: list[BotRunResult], model: str, training_repo_dir: str) -> str:
def _generate_instance_feedback(
self,
instance: SweBenchInstance,
agent_result: BotRunResult,
harness_result: BotRunResult | None,
model: str,
eval_dir: str,
) -> str:
"""Analyze one failed instance and produce generalized feedback for it.

Parameters
----------
instance : SweBenchInstance
The failed instance being analyzed.
agent_result : BotRunResult
The agent's own run result for this instance.
harness_result : BotRunResult | None
The independent harness verdict, or None if the agent itself
did not complete.
model : str
The model to use, in the format ``<provider>/<model_name>``.
eval_dir : str
Directory containing this instance's checkout and log file.

Returns
-------
str
Generalized feedback text, or an empty string if generation
failed.
"""
task_results = json.dumps({
"instance_id": instance.instance_id,
"task": instance.problem_statement,
"agent_result": {
"status": agent_result.status,
"result": agent_result.result,
"error": agent_result.error,
},
"harness_result": None if harness_result is None else {
"status": harness_result.status,
"result": harness_result.result,
"error": harness_result.error,
},
})

try:
bot = ReadingBot(model=model, folder_to_mount=eval_dir)
bot_result = bot.run(task=f"""
Analyze failed SWE-bench instance {instance.instance_id}. Inspect its checkout at
eval_repo/{instance.instance_id}, its log at logs/{instance.instance_id}_log.txt, and any
relevant repository context. Analyze why the attempt failed and then generate a feedback
that will be used to create context for FUTURE agents, solving DIFFERENT future tasks more effectively.

The agent result records whether the agent completed and what it reported. The harness result
records the independent test verdict, or is null when the agent did not complete:

{task_results}

Write concise feedback for a separate agent that will investigate the repository and update
memory. Do not write the memory or prescribe its contents to any specificity.
Consider Feedback on these:
- Mistakes to avoid: what went wrong and how a future agent should approach it better.
- Useful repository context: what the agent needed to know or find about the relevant files,
functions, behavior, environment, or development workflow.

The feedback must NOT reveal or reconstruct this evaluation instance. Do not include its ID, task,
result, tests, inputs, attempted patch, or specific fix. Keep only reusable lessons and
repository context that are generalized, and would be helpful even for future, different tasks.
""")
except Exception as exc:
logger.warning("Generating instance feedback failed with exception: %s", exc)
return ""

if not bot_result.status or not bot_result.result:
logger.warning("Generating instance feedback failed: %s", bot_result.error)
return ""
return bot_result.result

def _combine_result_feedback(self, feedback_items: list[str], model: str, training_repo_dir: str) -> str:
"""Summarize every instance's result into one feedback string.

Parameters
----------
results : list[BotRunResult]
One result per attempted instance.
feedback_items : list[str]
Generic diagnostic feedback from each failed instance.
model : str
The model to use, in the format ``<provider>/<model_name>``.
training_repo_dir : str
Expand All @@ -560,58 +651,49 @@ def _combine_result_feedback(self, results: list[BotRunResult], model: str, trai
results if the bot is unavailable or fails.
"""

serialized_str = f"Total {len(results)} tests ran and their result and feedback:\n"

for res in results:
serialized_str += f"\nResult: {'Passed' if res.status else 'Failed'}\n"
serialized_str += f"Optional Feedback: {res.result if res.result else 'None'}\n"
serialized_str += f"Error if there are any: {res.error if res.error else 'None'}\n"
serialized_str = "\n\n".join(feedback_items)
if not serialized_str:
return ""

try:
bot = ReadingBot(
model = model,
folder_to_mount=training_repo_dir
)
task = f"""
You are combining results from {len(results)} SWE-bench evaluation
runs into ONE feedback report for the next training iteration. The
training agent will read your report to decide what to add or fix
in its memory notes.

For each result below, note whether it passed or failed. For each
failure, briefly identify the underlying cause (e.g. wrong
file/line targeted, incorrect patch logic, response format error,
timeout) rather than only quoting the raw error. You may open
files under the mounted repo if you need to confirm a root
cause, but do not turn this into a debugging session.
Do not refer to any specific instance or test case by name/ID —
describe causes and guidance in general terms only.
You are combining {len(feedback_items)} pieces of feedback from failed SWE-bench
evaluation runs into ONE feedback report for the next training iteration. The
training agent will read your report to decide what to add or fix in its memory
notes.

Each item below is already a generalized analysis of one failed run (mistakes to
avoid, useful repository context). Deduplicate and group items that share the same
root cause or lesson rather than repeating yourself.

Then write a report with:
1. A one-line summary: how many passed vs failed.
2. Grouped failure patterns: if multiple failures share the same
root cause, describe that cause once rather than repeating
yourself.
3. Concrete, actionable guidance for the training agent — say
what to change in the memory notes to avoid each failure
pattern next time. Be specific and imperative
(e.g. "Record that config paths must be normalized before
1. A one-line summary: how many failures were analyzed.
2. Grouped failure patterns: describe each shared root cause once rather than
repeating yourself.
3. Concrete, actionable guidance for the training agent — say what to change in
the memory notes to avoid each failure pattern next time. Be specific and
imperative (e.g. "Record that config paths must be normalized before
comparison", not "there was a path issue").
4. Skip anything about passed cases beyond the summary count;
don't restate their feedback.

Keep the report tight and skimmable — short paragraphs or bullet
points, no code dumps. Put the final report in the `result`
field once you set task_done=true.
Do not reveal or reconstruct any specific evaluated task, test, input, attempted
patch, or fix. Do not build memory yourself; only provide the feedback that will
guide the next agent.

Keep the report tight and skimmable — short paragraphs or bullet points, no code
dumps. Put the final report in the `result` field once you set task_done=true.

{serialized_str}
"""
bot_result = bot.run(task=task)
except Exception as e:
logger.warning(f"Combining results failed with exception: {e}")
return f"Combining results failed. raw combined output:\n\n{serialized_str}"
return serialized_str

if bot_result.status:
return bot_result.result if bot_result.result else 'No feedback provided'
return bot_result.result if bot_result.result else serialized_str
else:
return f"Combining results failed. raw combined output:\n\n{serialized_str}"
return serialized_str
Loading
Loading