From 16209b3ed2b2d714d9dca173565cdf23262eb388 Mon Sep 17 00:00:00 2001 From: "Haoran Sun (Business Central)" Date: Fri, 2 Oct 2026 12:35:31 +0200 Subject: [PATCH] Initialize bcbench-core workspace package Add packages/bcbench-core as a uv workspace member consumed by the bcbench application. The package is empty: it establishes packaging and boundary enforcement before code moves in later stacked PRs. - Both projects build with uv_build (src layout, PEP 639 license metadata); the app no longer needs setuptools package-data rules - Core ships a PEP 561 py.typed marker - Core owns the ruff baseline and bans bcbench imports and environment reads; the app extends it with application-only settings - ty checks core with all rules as errors through `uv check`, which enables the uv integration that missing-direct-dependency requires Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: 3fa47b3c-f5d6-4c36-9be3-c4fdf2bebe47 --- .github/copilot-instructions.md | 3 +- .pre-commit-config.yaml | 8 +- CONTRIBUTING.md | 15 ++-- packages/bcbench-core/LICENSE | 21 +++++ packages/bcbench-core/README.md | 27 ++++++ packages/bcbench-core/pyproject.toml | 84 +++++++++++++++++++ .../bcbench-core/src/bcbench_core/__init__.py | 0 .../bcbench-core/src/bcbench_core/py.typed | 0 pyproject.toml | 75 ++++------------- uv.lock | 13 +++ 10 files changed, 177 insertions(+), 69 deletions(-) create mode 100644 packages/bcbench-core/LICENSE create mode 100644 packages/bcbench-core/README.md create mode 100644 packages/bcbench-core/pyproject.toml create mode 100644 packages/bcbench-core/src/bcbench_core/__init__.py create mode 100644 packages/bcbench-core/src/bcbench_core/py.typed diff --git a/.github/copilot-instructions.md b/.github/copilot-instructions.md index 8e2ae8239..54e207756 100644 --- a/.github/copilot-instructions.md +++ b/.github/copilot-instructions.md @@ -4,6 +4,7 @@ This is a benchmark for evaluating coding agents on real-world Business Central - **Dataset**: Benchmark entries following SWE-Bench schema with BC-specific adjustments - **Python Package** (`src/bcbench/`): CLI tools, agent implementations, and validation utilities +- **Core Library** (`packages/bcbench-core/`): Reusable, distributable evaluation library consumed by `bcbench`; must never import `bcbench`, read environment variables, or contain BC-Bench policy - **PowerShell Scripts** (`scripts/`): Environment setup and dataset verification using AL-GO/BCContainerHelper - **Tools** (`tools/`): Ad-hoc scripts for GitHub Artifacts download, etc - **Agent Evaluations**: Focuses on GitHub Copilot CLI and Claude Code @@ -52,7 +53,7 @@ def test_full_metrics_flow_to_success_result(self, sample_context): ``` ### Linting and formatting -Ruff is the single source of truth (`uv run ruff check --fix`, `uv run ruff format`); config lives in `pyproject.toml`. +Ruff is the single source of truth (`uv run ruff check --fix`, `uv run ruff format`); the baseline lives in `packages/bcbench-core/pyproject.toml` and the root `pyproject.toml` extends it. Lean on ruff's default rule set rather than growing `extend-select`, and prefer fixing violations over suppressing them. If a violation is genuinely intentional, use a targeted `# noqa: RULE - rationale` at that line instead of a repo-wide `ignore` entry. ## No Backward compatibility diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml index da3cc93fd..84c3f9b26 100644 --- a/.pre-commit-config.yaml +++ b/.pre-commit-config.yaml @@ -30,7 +30,13 @@ repos: - repo: local hooks: - id: ty - name: ty check + name: ty check (bcbench) entry: uv run ty check language: system pass_filenames: false + - id: ty-bcbench-core + name: ty check (bcbench-core) + # `uv check` enables ty's uv integration, which missing-direct-dependency requires + entry: uv check --package bcbench-core --locked --no-sync --preview-features check-command + language: system + pass_filenames: false diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 19dd4fa9b..5507736d9 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -19,14 +19,17 @@ A very high-level overview of the repository structure: ``` BC-Bench/ -├── src/bcbench/ # Evaluation harness — agent orchestration, build/test pipeline, results -├── dataset/ # Benchmark dataset tasks -├── scripts/ # Scripts for container setup & test execution; not needed for local development -├── notebooks/ # Analysis and visualization of results -├── evaluator/ # Braintrust scorer integration, used only when uploading result to Braintrust -└── docs/ # GitHub Page for the leaderboard site +├── src/bcbench/ # Evaluation harness — agent orchestration, build/test pipeline, results +├── packages/bcbench-core/ # Reusable evaluation library consumed by src/bcbench (uv workspace member) +├── dataset/ # Benchmark dataset tasks +├── scripts/ # Scripts for container setup & test execution; not needed for local development +├── notebooks/ # Analysis and visualization of results +├── evaluator/ # Braintrust scorer integration, used only when uploading result to Braintrust +└── docs/ # GitHub Page for the leaderboard site ``` +The repository is a [uv workspace](https://docs.astral.sh/uv/concepts/projects/workspaces/) with two Python projects: the `bcbench` application at the root and the `bcbench-core` library. `bcbench` depends on `bcbench-core`; never the reverse. The ruff baseline lives in `packages/bcbench-core/pyproject.toml` and the root config extends it with application-only settings. See [packages/bcbench-core/README.md](packages/bcbench-core/README.md) for the library boundary. + ## Setup Prerequisites: diff --git a/packages/bcbench-core/LICENSE b/packages/bcbench-core/LICENSE new file mode 100644 index 000000000..9e841e7a2 --- /dev/null +++ b/packages/bcbench-core/LICENSE @@ -0,0 +1,21 @@ + MIT License + + Copyright (c) Microsoft Corporation. + + Permission is hereby granted, free of charge, to any person obtaining a copy + of this software and associated documentation files (the "Software"), to deal + in the Software without restriction, including without limitation the rights + to use, copy, modify, merge, publish, distribute, sublicense, and/or sell + copies of the Software, and to permit persons to whom the Software is + furnished to do so, subject to the following conditions: + + The above copyright notice and this permission notice shall be included in all + copies or substantial portions of the Software. + + THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + SOFTWARE diff --git a/packages/bcbench-core/README.md b/packages/bcbench-core/README.md new file mode 100644 index 000000000..770ec7672 --- /dev/null +++ b/packages/bcbench-core/README.md @@ -0,0 +1,27 @@ +# bcbench-core + +Reusable, strongly typed building blocks for evaluating coding agents on Business Central (AL) tasks. [BC-Bench](https://github.com/microsoft/BC-Bench) is the first consumer; other repositories can build their own evaluation applications on top of it. + +> Pre-release: not yet published. The API is being extracted from BC-Bench incrementally and may change without notice. + +## Boundary + +`bcbench-core` contains only genuinely reusable contracts and operations. It must not contain: + +- Benchmark policy: categories, datasets, prompts, scoring thresholds, or workflows +- Repository-specific configuration, integrations, credentials, or internal material +- Reads of global configuration or environment variables; callers pass values explicitly +- Imports of the `bcbench` application + +The import and environment rules are enforced by ruff (`banned-api` in [`pyproject.toml`](pyproject.toml)); imports of undeclared dependencies are rejected by ty's `missing-direct-dependency` rule. + +## Development + +The package is a [uv workspace](https://docs.astral.sh/uv/concepts/projects/workspaces/) member of the BC-Bench repository. From the repository root: + +```bash +uv sync --all-groups +uv run ruff check packages/bcbench-core +uv check --package bcbench-core --preview-features check-command +uv build --package bcbench-core +``` diff --git a/packages/bcbench-core/pyproject.toml b/packages/bcbench-core/pyproject.toml new file mode 100644 index 000000000..6b60943aa --- /dev/null +++ b/packages/bcbench-core/pyproject.toml @@ -0,0 +1,84 @@ +[build-system] +requires = ["uv_build>=0.12.19,<0.13"] +build-backend = "uv_build" + +[project] +name = "bcbench-core" +version = "0.1.0" +description = "Reusable, strongly typed building blocks for evaluating coding agents on Business Central (AL) tasks" +readme = "README.md" +requires-python = ">=3.13" +license = "MIT" +license-files = ["LICENSE"] +authors = [{ name = "Microsoft Corporation" }] +classifiers = [ + # Remove once publishing is decided; PyPI rejects uploads carrying this classifier. + "Private :: Do Not Upload", + "Programming Language :: Python :: 3", + "Programming Language :: Python :: 3.13", + "Typing :: Typed", +] +dependencies = [] + +[tool.ruff] +target-version = "py313" +line-length = 200 + +[tool.ruff.lint] +extend-select = [ + # Restore the pyflakes/pycodestyle correctness rules that ruff 0.16 dropped from + # its default set (E711/E712/E713/E714/E721/E731/E741, F403/F405/F406/F722, etc.). + "E4", # pycodestyle: import placement (E401/E402) + "E7", # pycodestyle: statement/comparison lints (== None, lambda assignment, ...) + "F", # pyflakes: undefined names, star-import hazards, unused imports + "PLE", # Pylint errors (includes __all__ validation) + "PLW", # Pylint warnings (includes subprocess.run check requirement) + "I", # isort (import sorting) + "UP", # pyupgrade (modernize Python code) + "B", # flake8-bugbear (find likely bugs) + "SIM", # flake8-simplify (simplify code) + "C4", # flake8-comprehensions (better list/dict comprehensions) + "RET", # flake8-return (simplify return statements) + "PTH", # flake8-use-pathlib (prefer pathlib over os.path) + "RUF", # Ruff-specific rules + "TID", # flake8-tidy-imports (ban relative imports + extensible banned-API list) + "LOG", # flake8-logging: catch deprecated logging.warn, misuse of exception(), etc. + "G", # flake8-logging-format: catch string concat / % formatting in log calls + "TRY", # tryceratops: better exception handling + "PT", # flake8-pytest-style: consistent pytest patterns + "ANN", # flake8-annotations: enforce type hints (we prefer strong typing) + "T20", # flake8-print: prefer logger over print (we have a proper logger) + "FURB", # refurb: modern Python idioms (e.g. x or y over x if x else y) + "ISC", # implicit string concat (catches missing comma in lists of strings) + "ICN", # import conventions (np, pd, plt aliases) + "PYI", # type stubs (no-op now, guard for future) + "SLOT", # require __slots__ on str/tuple/namedtuple subclasses + "ASYNC", # async best practices (no-op now, guard for future) + "PERF", # perflint: avoid needless per-iteration overhead (prefer extend/comprehensions) + "RSE", # flake8-raise: drop redundant parentheses on bare exception raises + "PGH", # pygrep-hooks: require codes on noqa/type-ignore comments (PGH003/PGH004) + "S307", # flake8-bandit: ban eval (was PGH001, which ruff removed in favour of this rule) + "DTZ", # flake8-datetimez: require timezone-aware datetime usage + "NPY", # NumPy-specific correctness and modernization checks + "PIE", # flake8-pie: miscellaneous correctness and simplification checks +] + +ignore = [ + # Deliberate style choices with many intentional violations; everything else is fixed in-tree + # (targeted `# noqa` with a rationale is preferred over adding entries here). + "G004", # logging-f-string: f-strings are more readable; perf cost is negligible + "TRY003", # raise-vanilla-args: forces a custom exception class for every error message +] + +[tool.ruff.lint.flake8-tidy-imports] +ban-relative-imports = "all" + +[tool.ruff.lint.flake8-tidy-imports.banned-api] +"bcbench".msg = "bcbench-core must not depend on the BC-Bench application." +"os.environ".msg = "bcbench-core must not read the environment; accept values as parameters." +"os.getenv".msg = "bcbench-core must not read the environment; accept values as parameters." +"dotenv".msg = "bcbench-core must not load configuration; accept values as parameters." + +[tool.ty.rules] +# Strictest baseline for the public library; relax a specific rule only with a rationale. +all = "error" diff --git a/packages/bcbench-core/src/bcbench_core/__init__.py b/packages/bcbench-core/src/bcbench_core/__init__.py new file mode 100644 index 000000000..e69de29bb diff --git a/packages/bcbench-core/src/bcbench_core/py.typed b/packages/bcbench-core/src/bcbench_core/py.typed new file mode 100644 index 000000000..e69de29bb diff --git a/pyproject.toml b/pyproject.toml index d996542e3..288d9ed8c 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [build-system] -requires = ["setuptools>=61.0"] -build-backend = "setuptools.build_meta" +requires = ["uv_build>=0.12.19,<0.13"] +build-backend = "uv_build" [project] name = "bcbench" @@ -8,12 +8,14 @@ version = "0.13.0" description = "Benchmarking tool for Business Central (AL) ecosystem, inspired by SWE-Bench" readme = "README.md" requires-python = ">=3.13,<3.14" -license = {text = "MIT"} +license = "MIT" +license-files = ["LICENSE"] authors = [ {name = "Microsoft Corporation"} ] classifiers = ["Private :: Do Not Upload"] dependencies = [ + "bcbench-core", "jsonschema>=4.0", "python-dotenv>=1.2.2", "requests>=2.0", @@ -35,17 +37,17 @@ bcbench = "bcbench.cli:app" # ty's uv integration (TY_UV=1, used for missing-direct-dependency) needs uv >= 0.12.3 required-version = ">=0.12.19" +[tool.uv.workspace] +members = ["packages/bcbench-core"] + +[tool.uv.sources] +bcbench-core = { workspace = true } + [[tool.uv.index]] name = "microsoft-cfs" url = "https://packagefeedproxy.microsoft.io/pypi/simple" default = true -[tool.setuptools.packages.find] -where = ["src"] - -[tool.setuptools.package-data] -bcbench = ["agent/*.yaml", "agent/pr_review/scripts/*.ps1"] - [tool.pytest.ini_options] testpaths = ["tests"] addopts = ["-v", "--strict-markers", "-m", "not e2e"] @@ -56,57 +58,8 @@ markers = [ collect_imported_tests = false [tool.ruff] -target-version = "py313" -line-length = 200 - -[tool.ruff.lint] -extend-select = [ - # Restore the pyflakes/pycodestyle correctness rules that ruff 0.16 dropped from - # its default set (E711/E712/E713/E714/E721/E731/E741, F403/F405/F406/F722, etc.). - "E4", # pycodestyle: import placement (E401/E402) - "E7", # pycodestyle: statement/comparison lints (== None, lambda assignment, ...) - "F", # pyflakes: undefined names, star-import hazards, unused imports - "PLE", # Pylint errors (includes __all__ validation) - "PLW", # Pylint warnings (includes subprocess.run check requirement) - "I", # isort (import sorting) - "UP", # pyupgrade (modernize Python code) - "B", # flake8-bugbear (find likely bugs) - "SIM", # flake8-simplify (simplify code) - "C4", # flake8-comprehensions (better list/dict comprehensions) - "RET", # flake8-return (simplify return statements) - "PTH", # flake8-use-pathlib (prefer pathlib over os.path) - "RUF", # Ruff-specific rules - "TID", # flake8-tidy-imports (ban relative imports + extensible banned-API list) - "LOG", # flake8-logging: catch deprecated logging.warn, misuse of exception(), etc. - "G", # flake8-logging-format: catch string concat / % formatting in log calls - "TRY", # tryceratops: better exception handling - "PT", # flake8-pytest-style: consistent pytest patterns (~30 test files) - "ANN", # flake8-annotations: enforce type hints (we prefer strong typing) - "T20", # flake8-print: prefer logger over print (we have a proper logger) - "FURB", # refurb: modern Python idioms (e.g. x or y over x if x else y) - "ISC", # implicit string concat (catches missing comma in lists of strings) - "ICN", # import conventions (np, pd, plt aliases) - "PYI", # type stubs (no-op now, guard for future) - "SLOT", # require __slots__ on str/tuple/namedtuple subclasses - "ASYNC", # async best practices (no-op now, guard for future) - "PERF", # perflint: avoid needless per-iteration overhead (prefer extend/comprehensions) - "RSE", # flake8-raise: drop redundant parentheses on bare exception raises - "PGH", # pygrep-hooks: require codes on noqa/type-ignore comments (PGH003/PGH004) - "S307", # flake8-bandit: ban eval (was PGH001, which ruff removed in favour of this rule) - "DTZ", # flake8-datetimez: require timezone-aware datetime usage - "NPY", # NumPy-specific correctness and modernization checks - "PIE", # flake8-pie: miscellaneous correctness and simplification checks -] - -ignore = [ - # Deliberate style choices with many intentional violations; everything else is fixed in-tree - # (targeted `# noqa` with a rationale is preferred over adding entries here). - "G004", # logging-f-string: f-strings are more readable; perf cost is negligible (143 sites) - "TRY003", # raise-vanilla-args: forces a custom exception class for every error message (96 sites) -] - -[tool.ruff.lint.flake8-tidy-imports] -ban-relative-imports = "all" +# The app inherits the bcbench-core lint baseline; tables redefined below replace the inherited ones. +extend = "packages/bcbench-core/pyproject.toml" [tool.ruff.lint.flake8-tidy-imports.banned-api] # Force sandboxed rendering: @@ -120,7 +73,7 @@ ban-relative-imports = "all" "tools/**" = ["T20"] # standalone CLI scripts [tool.ty.src] -exclude = ["notebooks/"] +exclude = ["notebooks/", "packages/"] [tool.ty.analysis] # bc-eval[capi] is internal and installed only into a separate venv by bcal-evaluation.yml diff --git a/uv.lock b/uv.lock index 0098f24b6..616874c7e 100644 --- a/uv.lock +++ b/uv.lock @@ -2,6 +2,12 @@ version = 1 revision = 3 requires-python = "==3.13.*" +[manifest] +members = [ + "bcbench", + "bcbench-core", +] + [[package]] name = "aiofiles" version = "24.1.0" @@ -248,6 +254,7 @@ name = "bcbench" version = "0.13.0" source = { editable = "." } dependencies = [ + { name = "bcbench-core" }, { name = "jinja2" }, { name = "jsonschema" }, { name = "numpy" }, @@ -283,6 +290,7 @@ redteam = [ [package.metadata] requires-dist = [ + { name = "bcbench-core", editable = "packages/bcbench-core" }, { name = "jinja2", specifier = ">=3.1.6" }, { name = "jsonschema", specifier = ">=4.0" }, { name = "numpy", specifier = ">=2.3.5" }, @@ -316,6 +324,11 @@ redteam = [ { name = "azure-identity", specifier = ">=1.25.3" }, ] +[[package]] +name = "bcbench-core" +version = "0.1.0" +source = { editable = "packages/bcbench-core" } + [[package]] name = "certifi" version = "2025.10.5"