Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion src/bcbench/agent/shared/mcp_gateway.py
Original file line number Diff line number Diff line change
Expand Up @@ -19,7 +19,7 @@
import time
from http.client import HTTPConnection
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from typing import cast
from typing import cast, override
from urllib.parse import urlsplit

from bcbench_core.container import ContainerConfig
Expand Down Expand Up @@ -236,6 +236,7 @@ def _build_handler(gateway: BcMcpGateway) -> type[BaseHTTPRequestHandler]:
class _ProxyHandler(BaseHTTPRequestHandler):
protocol_version = "HTTP/1.1"

@override
def log_message(self, format: str, *args: object) -> None: # match stdlib signature; silence access log
pass

Expand Down
6 changes: 5 additions & 1 deletion src/bcbench/commands/evaluate.py
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
import logging
import random
from pathlib import Path
from typing import Annotated, cast
from typing import Annotated, cast, override

import typer
from bcbench_core.filesystem import prepare_run_dir
Expand Down Expand Up @@ -346,12 +346,15 @@ class MockEvaluationPipeline(EvaluationPipeline[BaseDatasetEntry]):
It randomly generates different scenarios to test result handling and serialization.
"""

@override
def setup_workspace(self, entry: BaseDatasetEntry, repo_path: Path) -> None:
logger.info("Mock pipeline: Skipping workspace setup")

@override
def setup(self, context: EvaluationContext[BaseDatasetEntry]) -> None:
logger.info("Mock pipeline: Skipping setup")

@override
def run_agent(self, context: EvaluationContext[BaseDatasetEntry], agent_runner: AgentRunner[BaseDatasetEntry]) -> None:
"""Generate random agent metrics and experiment configuration."""
logger.info("Mock pipeline: Generating random metrics and experiment configuration")
Expand Down Expand Up @@ -382,6 +385,7 @@ def run_agent(self, context: EvaluationContext[BaseDatasetEntry], agent_runner:
logger.info(f"Using agent metrics: {context.metrics}")
logger.info(f"Using experiment configuration: {context.experiment}")

@override
def evaluate(self, context: EvaluationContext[BaseDatasetEntry]) -> None:
"""Create random evaluation result to test different outcome scenarios."""
logger.info("Mock pipeline: Generating random evaluation result")
Expand Down
5 changes: 4 additions & 1 deletion src/bcbench/dataset/codereview.py
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
from __future__ import annotations

from enum import StrEnum
from typing import Annotated, Self
from typing import Annotated, Self, override

from pydantic import BaseModel, ConfigDict, Field, field_validator, model_validator

Expand Down Expand Up @@ -81,6 +81,7 @@ def _coerce_severity(cls, value: object) -> Severity | None:
def severity_label(self) -> str:
return self.severity.value if self.severity is not None else "unspecified"

@override
def __str__(self) -> str:
loc = f"{self.file}:{self.line_start}"
if self.line_end and self.line_end != self.line_start:
Expand Down Expand Up @@ -129,9 +130,11 @@ def _validate_article_annotations(self) -> Self:
)
return self

@override
def get_task(self) -> str:
return self.patch

@override
def get_expected_output(self) -> str:
return "\n".join(str(c) for c in self.expected_comments)

Expand Down
12 changes: 11 additions & 1 deletion src/bcbench/dataset/dataset_entry.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,7 @@
import re
from abc import abstractmethod
from pathlib import Path
from typing import Annotated, Literal, Self
from typing import Annotated, Literal, Self, override

from bcbench_core.dataset import TestEntry
from pydantic import BaseModel, ConfigDict, Field, model_validator
Expand Down Expand Up @@ -118,13 +118,15 @@ class RepoGroundedEntry(BaseDatasetEntry):
patch: Annotated[str, Field(min_length=1, pattern=r"^[^\x00]*$")]

@property
@override
def customization_profile(self) -> str:
return self.repo.replace("/", "-")

@property
def problem_statement_dir(self) -> Path:
return _config.paths.problem_statement_dir / self.instance_id

@override
def get_task(self) -> str:
readme_path = self.problem_statement_dir / _config.file_patterns.problem_statement_readme
return readme_path.read_text(encoding="utf-8")
Expand Down Expand Up @@ -158,13 +160,15 @@ def validate_baseapp_patches_are_w1_only(self) -> Self:
class BugFixEntry(_BugFixTestGenBase):
"""Dataset entry for the bug-fix category."""

@override
def get_expected_output(self) -> str:
return self.patch


class TestGenEntry(_BugFixTestGenBase):
"""Dataset entry for the test-generation category."""

@override
def get_expected_output(self) -> str:
return self.test_patch

Expand All @@ -178,12 +182,15 @@ class NL2ALEntry(BaseDatasetEntry):
audience: Literal["Business", "Technical", "Both"]

@property
@override
def customization_profile(self) -> str:
return "nl2al"

@override
def get_task(self) -> str:
return self.nl_prompt

@override
def get_expected_output(self) -> Checklist:
return {"assertions": self.expected}

Expand All @@ -204,11 +211,14 @@ class DataQueryEntry(BaseDatasetEntry):
ordered: bool = False

@property
@override
def customization_profile(self) -> str:
return "dataquery"

@override
def get_task(self) -> str:
return self.nl_prompt

@override
def get_expected_output(self) -> str:
return self.gold_query
7 changes: 6 additions & 1 deletion src/bcbench/dataset/extensibility_request.py
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
from __future__ import annotations

from typing import Annotated, Literal
from typing import Annotated, Literal, override

from pydantic import Field

Expand All @@ -20,12 +20,14 @@ class ExtRequestAdvisorEntry(RepoGroundedEntry):
comments: str = ""
expected: Annotated[list[ChecklistAssertion], Field(min_length=1)]

@override
def get_task(self) -> str:
sections = [f"# {self.title}", "", self.description.rstrip()]
if self.comments.strip():
sections += ["", "## Additional requester context", "", self.comments.rstrip()]
return "\n".join(sections)

@override
def get_expected_output(self) -> Checklist:
return {"assertions": self.expected}

Expand All @@ -43,6 +45,7 @@ class ExtRequestImplementEntry(RepoGroundedEntry):
# LLM-judge checklist: expected event/signature/placement and expected layer propagation.
expected: Annotated[list[ChecklistAssertion], Field(min_length=1)]

@override
def get_expected_output(self) -> Checklist:
return {"assertions": self.expected}

Expand Down Expand Up @@ -83,6 +86,7 @@ class ExtRequestTriageEntry(RepoGroundedEntry):
# LLM-judge checklist: expected labels_to_set, issue_state and advisory-comment substance.
expected: Annotated[list[ChecklistAssertion], Field(min_length=1)]

@override
def get_task(self) -> str:
sections = [f"# {self.title}", "", self.description.rstrip()]
if self.current_labels:
Expand All @@ -91,5 +95,6 @@ def get_task(self) -> str:
sections += ["", "## Follow-up conversation", "", self.comments.rstrip()]
return "\n".join(sections)

@override
def get_expected_output(self) -> Checklist:
return {"assertions": self.expected}
5 changes: 5 additions & 0 deletions src/bcbench/evaluate/bugfix.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
import logging
from pathlib import Path
from typing import override

from bcbench_core.bc import build_and_publish_projects
from bcbench_core.exceptions import BuildError, TestExecutionError
Expand All @@ -21,11 +22,13 @@
class BugFixPipeline(EvaluationPipeline[BugFixEntry]):
"""Pipeline for bug-fix evaluation category."""

@override
def setup_workspace(self, entry: BugFixEntry, repo_path: Path) -> None:
setup_repo_prebuild(entry, repo_path)
copy_problem_statement_folder(entry, repo_path)
set_runtime_version(repo_path, entry.project_paths)

@override
def setup(self, context: EvaluationContext[BugFixEntry]) -> None:
setup_repo_prebuild(context.entry, context.repo_path)

Expand All @@ -39,10 +42,12 @@ def setup(self, context: EvaluationContext[BugFixEntry]) -> None:
copy_problem_statement_folder(context.entry, context.repo_path)
set_runtime_version(context.repo_path, context.entry.project_paths)

@override
def run_agent(self, context: EvaluationContext[BugFixEntry], agent_runner: AgentRunner[BugFixEntry]) -> None:
with github_log_group(f"{context.agent_name} -- Entry: {context.entry.instance_id}"):
context.metrics, context.experiment = agent_runner(context)

@override
def evaluate(self, context: EvaluationContext[BugFixEntry]) -> None:
container = context.get_container()
test_projects, _app_projects = categorize_projects(context.entry.project_paths)
Expand Down
5 changes: 5 additions & 0 deletions src/bcbench/evaluate/codereview.py
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
import logging
import subprocess
from pathlib import Path
from typing import override

from bcbench_core.git import apply_patch, fetch_commit_if_missing

Expand Down Expand Up @@ -31,6 +32,7 @@ class CodeReviewPipeline(EvaluationPipeline[CodeReviewEntry]):
as local git changes so the agent can review the branch diff directly.
"""

@override
def setup_workspace(self, entry: CodeReviewEntry, repo_path: Path) -> None:
"""Setup workspace for code review by applying the entry patch as local changes."""
# Code-review base commits are pre-squash PR commits, so they might be missing from local dev setups.
Expand All @@ -48,13 +50,16 @@ def setup_workspace(self, entry: CodeReviewEntry, repo_path: Path) -> None:
check=True,
)

@override
def setup(self, context: EvaluationContext[CodeReviewEntry]) -> None:
self.setup_workspace(context.entry, context.repo_path)

@override
def run_agent(self, context: EvaluationContext[CodeReviewEntry], agent_runner: AgentRunner[CodeReviewEntry]) -> None:
with github_log_group(f"{context.agent_name} -- Entry: {context.entry.instance_id}"):
context.metrics, context.experiment = agent_runner(context)

@override
def evaluate(self, context: EvaluationContext[CodeReviewEntry]) -> None:
review_output_file: Path = context.repo_path / REVIEW_OUTPUT_FILE

Expand Down
5 changes: 5 additions & 0 deletions src/bcbench/evaluate/dataquery.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@
from collections.abc import Callable, Mapping, Sequence
from decimal import Decimal, InvalidOperation
from pathlib import Path
from typing import override

from bcbench_core.filesystem import clear_directory

Expand Down Expand Up @@ -94,17 +95,21 @@ class DataQueryPipeline(EvaluationPipeline[DataQueryEntry]):
genuinely querying the environment.
"""

@override
def setup_workspace(self, entry: DataQueryEntry, repo_path: Path) -> None:
# The workspace is shared into the running container, so its contents are cleared in place.
clear_directory(repo_path)

@override
def setup(self, context: EvaluationContext[DataQueryEntry]) -> None:
self.setup_workspace(context.entry, context.repo_path)

@override
def run_agent(self, context: EvaluationContext[DataQueryEntry], agent_runner: Callable) -> None:
with github_log_group(f"{context.agent_name} -- Entry: {context.entry.instance_id}"):
context.metrics, context.experiment = agent_runner(context)

@override
def evaluate(self, context: EvaluationContext[DataQueryEntry]) -> None:
query_file = context.repo_path / GENERATED_QUERY_FILE
# query.al is only an inspection artifact now; scoring is on the data the agent retrieved.
Expand Down
5 changes: 5 additions & 0 deletions src/bcbench/evaluate/ext_request_advisor.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
import logging
from pathlib import Path
from typing import override

from bcbench.dataset import ExtRequestAdvisorEntry
from bcbench.evaluate.base import AgentRunner, EvaluationPipeline
Expand All @@ -18,17 +19,21 @@
class ExtRequestAdvisorPipeline(EvaluationPipeline[ExtRequestAdvisorEntry]):
"""Offline single-shot proxy for the interactive extensibility advisor."""

@override
def setup_workspace(self, entry: ExtRequestAdvisorEntry, repo_path: Path) -> None:
setup_repo_prebuild(entry, repo_path)
(repo_path / ADVISOR_RESULT_FILE).unlink(missing_ok=True)

@override
def setup(self, context: EvaluationContext[ExtRequestAdvisorEntry]) -> None:
self.setup_workspace(context.entry, context.repo_path)

@override
def run_agent(self, context: EvaluationContext[ExtRequestAdvisorEntry], agent_runner: AgentRunner[ExtRequestAdvisorEntry]) -> None:
with github_log_group(f"{context.agent_name} -- Entry: {context.entry.instance_id}"):
context.metrics, context.experiment = agent_runner(context)

@override
def evaluate(self, context: EvaluationContext[ExtRequestAdvisorEntry]) -> None:
result_path = context.repo_path / ADVISOR_RESULT_FILE
raw = result_path.read_text(encoding="utf-8").strip() if result_path.exists() else ""
Expand Down
5 changes: 5 additions & 0 deletions src/bcbench/evaluate/ext_request_implement.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
import logging
from pathlib import Path
from typing import override

from bcbench_core.exceptions import EmptyDiffError
from bcbench_core.git import stage_and_get_diff
Expand All @@ -24,18 +25,22 @@ class ExtRequestImplementPipeline(EvaluationPipeline[ExtRequestImplementEntry]):
by an LLM judge against the entry checklist.
"""

@override
def setup_workspace(self, entry: ExtRequestImplementEntry, repo_path: Path) -> None:
setup_repo_prebuild(entry, repo_path)
copy_problem_statement_folder(entry, repo_path)
set_runtime_version(repo_path, entry.project_paths)

@override
def setup(self, context: EvaluationContext[ExtRequestImplementEntry]) -> None:
self.setup_workspace(context.entry, context.repo_path)

@override
def run_agent(self, context: EvaluationContext[ExtRequestImplementEntry], agent_runner: AgentRunner[ExtRequestImplementEntry]) -> None:
with github_log_group(f"{context.agent_name} -- Entry: {context.entry.instance_id}"):
context.metrics, context.experiment = agent_runner(context)

@override
def evaluate(self, context: EvaluationContext[ExtRequestImplementEntry]) -> None:
try:
generated_patch = stage_and_get_diff(context.repo_path, exclude=("**/app.json", "*.docx", "*.md"))
Expand Down
5 changes: 5 additions & 0 deletions src/bcbench/evaluate/ext_request_triage.py
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,7 @@

import logging
from pathlib import Path
from typing import override

from bcbench.dataset import ExtRequestTriageEntry
from bcbench.evaluate.base import AgentRunner, EvaluationPipeline
Expand All @@ -28,17 +29,21 @@
class ExtRequestTriagePipeline(EvaluationPipeline[ExtRequestTriageEntry]):
"""Pipeline for the extensibility-request-triage category — no BC container, no build, no tests."""

@override
def setup_workspace(self, entry: ExtRequestTriageEntry, repo_path: Path) -> None:
setup_repo_prebuild(entry, repo_path)
(repo_path / TRIAGE_RESULT_FILE).unlink(missing_ok=True)

@override
def setup(self, context: EvaluationContext[ExtRequestTriageEntry]) -> None:
self.setup_workspace(context.entry, context.repo_path)

@override
def run_agent(self, context: EvaluationContext[ExtRequestTriageEntry], agent_runner: AgentRunner[ExtRequestTriageEntry]) -> None:
with github_log_group(f"{context.agent_name} -- Entry: {context.entry.instance_id}"):
context.metrics, context.experiment = agent_runner(context)

@override
def evaluate(self, context: EvaluationContext[ExtRequestTriageEntry]) -> None:
result_path = context.repo_path / TRIAGE_RESULT_FILE
raw = result_path.read_text(encoding="utf-8").strip() if result_path.exists() else ""
Expand Down
Loading
Loading