Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/workflows/CI.yml
Original file line number Diff line number Diff line change
Expand Up @@ -37,7 +37,7 @@ jobs:
RUFF_OUTPUT_FORMAT: github

- name: Run tests with coverage
run: uv run pytest --cov=src/bcbench --cov-report=term-missing
run: uv run pytest --cov=src/bcbench --cov=packages/bcbench-core/src/bcbench_core --cov-report=term-missing

e2e:
strategy:
Expand Down
4 changes: 2 additions & 2 deletions CONTRIBUTING.md
Original file line number Diff line number Diff line change
Expand Up @@ -61,8 +61,8 @@ uv run bcbench --help
# This is very fast, give it a go and see it live!
uv run bcbench run copilot microsoft__BCApps-5633 --category bug-fix --repo-path /path/to/BCApps

# Run tests
uv run pytest --cov=src/bcbench --cov-report=term-missing
# Run tests (bcbench and bcbench-core)
uv run pytest --cov=src/bcbench --cov=packages/bcbench-core/src/bcbench_core --cov-report=term-missing

# Lint and format
uv run pre-commit run --all-files
Expand Down
3 changes: 1 addition & 2 deletions notebooks/archive/multi-run.ipynb
Original file line number Diff line number Diff line change
Expand Up @@ -1126,10 +1126,9 @@
],
"source": [
"import plotly.graph_objects as go\n",
"from bcbench_core.scoring import pass_at_k, pass_hat_k\n",
"from plotly.subplots import make_subplots\n",
"\n",
"from bcbench.results.metrics import pass_at_k, pass_hat_k\n",
"\n",
"# Visualize: pass^k vs pass@k comparison using correct formulas\n",
"run_ids_sorted = sorted(analysis_model_df[\"run_id\"].unique())\n",
"pivot = analysis_model_df.pivot_table(index=\"instance_id\", columns=\"run_id\", values=\"resolved\", aggfunc=lambda x: x.iloc[0])\n",
Expand Down
4 changes: 2 additions & 2 deletions notebooks/bug-fix/altool-comparison.ipynb
Original file line number Diff line number Diff line change
Expand Up @@ -46,8 +46,8 @@
"source": [
"import numpy as np\n",
"import plotly.graph_objects as go\n",
"\n",
"from bcbench.results.metrics import bootstrap_ci, pass_at_k, pass_hat_k"
"from bcbench_core.scoring import pass_at_k, pass_hat_k\n",
"from bcbench_core.stats import bootstrap_ci"
]
},
{
Expand Down
3 changes: 1 addition & 2 deletions notebooks/bug-fix/claude-vs-copilot.ipynb
Original file line number Diff line number Diff line change
Expand Up @@ -55,10 +55,9 @@
"outputs": [],
"source": [
"import plotly.graph_objects as go\n",
"from bcbench_core.stats import bootstrap_ci\n",
"from IPython.display import Markdown, display\n",
"\n",
"from bcbench.results.metrics import bootstrap_ci\n",
"\n",
"\n",
"def build_mean_ci_table(model: str) -> pd.DataFrame:\n",
" rows = []\n",
Expand Down
4 changes: 2 additions & 2 deletions notebooks/test-generation/altest-comparison.ipynb
Original file line number Diff line number Diff line change
Expand Up @@ -37,8 +37,8 @@
"source": [
"import numpy as np\n",
"import plotly.graph_objects as go\n",
"\n",
"from bcbench.results.metrics import bootstrap_ci, pass_at_k, pass_hat_k"
"from bcbench_core.scoring import pass_at_k, pass_hat_k\n",
"from bcbench_core.stats import bootstrap_ci"
]
},
{
Expand Down
3 changes: 1 addition & 2 deletions notebooks/utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,10 +3,9 @@
from typing import Literal, TypedDict

import pandas as pd
from bcbench_core.scoring import pass_at_k, pass_hat_k
from unidiff import PatchSet

from bcbench.results.metrics import pass_at_k, pass_hat_k

# Root paths - all notebooks should use these
NOTEBOOKS_ROOT = Path(__file__).parent
DATASET_PATH = NOTEBOOKS_ROOT.parent / "dataset" / "bcbench.jsonl"
Expand Down
1 change: 1 addition & 0 deletions packages/bcbench-core/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -23,5 +23,6 @@ The package is a [uv workspace](https://docs.astral.sh/uv/concepts/projects/work
uv sync --all-groups
uv run ruff check packages/bcbench-core
uv check --package bcbench-core --preview-features check-command
uv run pytest packages/bcbench-core
uv build --package bcbench-core
```
11 changes: 10 additions & 1 deletion packages/bcbench-core/pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -18,7 +18,13 @@ classifiers = [
"Programming Language :: Python :: 3.13",
"Typing :: Typed",
]
dependencies = []
dependencies = [
"numpy>=2.3.5",
"scipy>=1.16.3",
]

[dependency-groups]
dev = ["pytest>=9.0.3"]

[tool.ruff]
target-version = "py313"
Expand Down Expand Up @@ -73,6 +79,9 @@ ignore = [
[tool.ruff.lint.flake8-tidy-imports]
ban-relative-imports = "all"

[tool.ruff.lint.per-file-ignores]
"tests/**" = ["ANN"]

[tool.ruff.lint.flake8-tidy-imports.banned-api]
"bcbench".msg = "bcbench-core must not depend on the BC-Bench application."
"os.environ".msg = "bcbench-core must not read the environment; accept values as parameters."
Expand Down
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
"""Filesystem operations."""
"""Filesystem helpers that tolerate read-only files on Windows."""

import shutil
import stat
Expand Down
Original file line number Diff line number Diff line change
@@ -1,44 +1,41 @@
"""Project path categorization and management operations."""
"""Project path categorization for AL repositories."""

from bcbench.config import get_config
from bcbench.logger import get_logger
import logging

logger = get_logger(__name__)
_config = get_config()
logger = logging.getLogger(__name__)


def _is_test_project(project_path: str, test_identifiers: tuple[str, ...]) -> bool:
r"""Check if a project path is a test project based on configured identifiers.
def is_test_project(project_path: str, test_identifiers: tuple[str, ...] = ("test", "tests")) -> bool:
r"""Check if a project path is a test project.

The function checks if any test identifier appears as a complete path component
by looking for the identifier preceded by a path separator (/ or \).
This ensures that 'src/contest' does not match 'test', but 'src/test' does.
A path is a test project when one of its components, after a path separator (/ or \),
starts with a test identifier: 'src/test' and 'src/test1' match 'test', 'src/contest' does not.

Args:
project_path: The project path to check
test_identifiers: Tuple of test identifier strings (e.g., 'test', 'tests')
test_identifiers: Case-insensitive prefixes that mark a test project component

Returns:
True if the project path contains a test identifier as a path component
True if a path component starts with a test identifier
"""
project_lower = project_path.lower()
return any(f"/{identifier}" in project_lower or f"\\{identifier}" in project_lower for identifier in test_identifiers)


def categorize_projects(project_paths: list[str]) -> tuple[list[str], list[str]]:
def categorize_projects(project_paths: list[str], test_identifiers: tuple[str, ...] = ("test", "tests")) -> tuple[list[str], list[str]]:
"""Categorize project paths into test projects and application projects.

Args:
project_paths: List of project paths to categorize
test_identifiers: Case-insensitive prefixes that mark a test project component

Returns:
Tuple of (test_projects, app_projects)

Raises:
RuntimeError: If project categorization fails (no test or app projects found)
"""
test_identifiers = _config.file_patterns.test_project_identifiers
test_projects: list[str] = [project for project in project_paths if _is_test_project(project, test_identifiers)]
test_projects: list[str] = [project for project in project_paths if is_test_project(project, test_identifiers)]
app_projects: list[str] = [project for project in project_paths if project not in test_projects]

if not test_projects or not app_projects:
Expand Down
Original file line number Diff line number Diff line change
@@ -1,30 +1,6 @@
import math
"""Scoring functions for evaluation results."""

import numpy as np
from scipy.stats import bootstrap as scipy_bootstrap


def bootstrap_ci(values: list[float] | np.ndarray, n_bootstrap: int = 10000, ci_level: float = 0.95) -> dict[str, float | None]:
data = np.asarray(values, dtype=float)
mean = float(data.mean()) if len(data) >= 1 else 0.0
if len(data) < 2:
return {"mean": mean, "ci_low": None, "ci_high": None}
# BCa is undefined for zero-variance data (jackknife acceleration divides by zero)
if np.all(data == data[0]):
return {"mean": mean, "ci_low": None, "ci_high": None}
result = scipy_bootstrap(
(data,),
statistic=np.mean,
n_resamples=n_bootstrap,
confidence_level=ci_level,
method="BCa",
rng=np.random.default_rng(42),
)
return {
"mean": mean,
"ci_low": float(result.confidence_interval.low),
"ci_high": float(result.confidence_interval.high),
}
import math


def precision_recall(matched_count: int, generated_count: int, expected_count: int) -> tuple[float, float]:
Expand Down
27 changes: 27 additions & 0 deletions packages/bcbench-core/src/bcbench_core/stats.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,27 @@
"""Statistics for aggregating evaluation results across runs."""

import numpy as np
from scipy.stats import bootstrap as scipy_bootstrap


def bootstrap_ci(values: list[float] | np.ndarray, n_bootstrap: int = 10000, ci_level: float = 0.95) -> dict[str, float | None]:
data = np.asarray(values, dtype=float)
mean = float(data.mean()) if len(data) >= 1 else 0.0
if len(data) < 2:
return {"mean": mean, "ci_low": None, "ci_high": None}
# BCa is undefined for zero-variance data (jackknife acceleration divides by zero)
if np.all(data == data[0]):
return {"mean": mean, "ci_low": None, "ci_high": None}
result = scipy_bootstrap(
(data,),
statistic=np.mean,
n_resamples=n_bootstrap,
confidence_level=ci_level,
method="BCa",
rng=np.random.default_rng(42),
)
return {
"mean": mean,
"ci_low": float(result.confidence_interval.low),
"ci_high": float(result.confidence_interval.high),
}
55 changes: 55 additions & 0 deletions packages/bcbench-core/tests/test_filesystem.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,55 @@
import stat

from bcbench_core.filesystem import clear_directory, prepare_run_dir, remove_tree


def test_remove_tree_removes_read_only_files(tmp_path):
tree = tmp_path / "tree"
(tree / "nested").mkdir(parents=True)
read_only_file = tree / "nested" / "read-only.txt"
read_only_file.write_text("content", encoding="utf-8")
read_only_file.chmod(stat.S_IREAD)

remove_tree(tree)

assert not tree.exists()


def test_prepare_run_dir_replaces_existing_run(tmp_path):
stale_file = tmp_path / "run-1" / "stale.jsonl"
stale_file.parent.mkdir()
stale_file.write_text("{}", encoding="utf-8")

run_dir = prepare_run_dir(tmp_path, "run-1")

assert run_dir == tmp_path / "run-1"
assert run_dir.is_dir()
assert list(run_dir.iterdir()) == []


def test_prepare_run_dir_creates_missing_parents(tmp_path):
run_dir = prepare_run_dir(tmp_path / "results", "run-2")

assert run_dir.is_dir()


def test_clear_directory_removes_contents_and_preserves_directory(tmp_path):
nested_directory = tmp_path / "nested"
nested_directory.mkdir()
(nested_directory / "nested.txt").write_text("nested", encoding="utf-8")
read_only_file = tmp_path / "read-only.txt"
read_only_file.write_text("content", encoding="utf-8")
read_only_file.chmod(stat.S_IREAD)

clear_directory(tmp_path)

assert tmp_path.is_dir()
assert list(tmp_path.iterdir()) == []


def test_clear_directory_creates_missing_directory(tmp_path):
missing_directory = tmp_path / "missing"

clear_directory(missing_directory)

assert missing_directory.is_dir()
Original file line number Diff line number Diff line change
Expand Up @@ -2,29 +2,29 @@

import pytest

from bcbench.operations.project_operations import _is_test_project, categorize_projects
from bcbench_core.projects import categorize_projects, is_test_project


class TestIsTestProject:
"""Test suite for _is_test_project helper function."""
"""Test suite for is_test_project."""

def test_is_test_project_with_test_identifier(self):
assert _is_test_project("src/test", ("test", "tests")) is True
assert is_test_project("src/test", ("test", "tests")) is True

def test_is_test_project_with_tests_identifier(self):
assert _is_test_project("src/tests", ("test", "tests")) is True
assert is_test_project("src/tests", ("test", "tests")) is True

def test_is_test_project_with_windows_separator(self):
assert _is_test_project("src\\test", ("test", "tests")) is True
assert is_test_project("src\\test", ("test", "tests")) is True

def test_is_test_project_case_insensitive(self):
assert _is_test_project("src/Test", ("test", "tests")) is True
assert is_test_project("src/Test", ("test", "tests")) is True

def test_is_test_project_substring_not_path_component(self):
assert _is_test_project("src/contest", ("test", "tests")) is False
assert is_test_project("src/contest", ("test", "tests")) is False

def test_is_test_project_without_identifier(self):
assert _is_test_project("src/app", ("test", "tests")) is False
assert is_test_project("src/app", ("test", "tests")) is False


class TestCategorizeProjects:
Expand Down
Loading
Loading