From 2f0d46a584c749b29fc50bd89fe7b4439134b9a1 Mon Sep 17 00:00:00 2001 From: "Ian T." Date: Mon, 28 Sep 2026 20:54:48 -0700 Subject: [PATCH 1/4] ci: add HOL plugin scanner workflow Runs the scanner used by the awesome-ai-plugins catalog on push and pull request. Read-only, no secrets, both actions pinned to commit SHAs, fails below a score of 80 or on any high severity finding. --- .github/workflows/plugin-scan.yml | 27 +++++++++++++++++++++++++++ 1 file changed, 27 insertions(+) create mode 100644 .github/workflows/plugin-scan.yml diff --git a/.github/workflows/plugin-scan.yml b/.github/workflows/plugin-scan.yml new file mode 100644 index 0000000..67c4722 --- /dev/null +++ b/.github/workflows/plugin-scan.yml @@ -0,0 +1,27 @@ +name: Plugin Security Scan + +# HOL plugin scanner, the preflight used by the awesome-ai-plugins catalog. +# Listing does not require this workflow, but maintaining it earns the full +# registry trust score instead of a 10% reduction. It needs no secrets and +# only reads the repository. + +on: + pull_request: + push: + branches: [master] + +permissions: + contents: read + +jobs: + scan: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2 + with: + persist-credentials: false + - uses: hashgraph-online/ai-plugin-scanner-action@caba2e96aa8ad2feb6cf6fca52442b52e22e779f # v1.2.635 + with: + plugin_dir: "." + min_score: 80 + fail_on_severity: high From 92b03f4437aa37a24b1e42743321731688414efa Mon Sep 17 00:00:00 2001 From: "Ian T." Date: Mon, 28 Sep 2026 21:00:12 -0700 Subject: [PATCH 2/4] ci: emit scanner report as a JSON artifact --- .github/workflows/plugin-scan.yml | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/.github/workflows/plugin-scan.yml b/.github/workflows/plugin-scan.yml index 67c4722..2a2005c 100644 --- a/.github/workflows/plugin-scan.yml +++ b/.github/workflows/plugin-scan.yml @@ -20,8 +20,18 @@ jobs: - uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2 with: persist-credentials: false - - uses: hashgraph-online/ai-plugin-scanner-action@caba2e96aa8ad2feb6cf6fca52442b52e22e779f # v1.2.635 + - id: scan + uses: hashgraph-online/ai-plugin-scanner-action@caba2e96aa8ad2feb6cf6fca52442b52e22e779f # v1.2.635 with: plugin_dir: "." min_score: 80 fail_on_severity: high + format: json + output: scan-report.json + - name: Upload scan report + if: always() + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 + with: + name: plugin-scan-report + path: scan-report.json + if-no-files-found: warn From e1f5210ccb89c8fbd2367b0677f8e2cd0aa5cbe2 Mon Sep 17 00:00:00 2001 From: "Ian T." Date: Mon, 28 Sep 2026 21:11:05 -0700 Subject: [PATCH 3/4] ci: harden workflows and add security policy Clears every finding the HOL plugin scanner raised, which the awesome-ai-plugins catalog requires to pass at 80 or above with no high severity findings before a listing can merge. Pin every third-party action to a commit SHA, staying within the major version already in use, so a moved tag cannot change what runs in the publish workflow. Dependabot now tracks those pins monthly. release-smoke.yml took the version to test from the workflow_run head branch, which is attacker-influenceable input read in a privileged context. It reads the version from the checked-out default branch instead, with the existing PyPI lookup still as the fallback. The local OpenAI-compatible test used a literal that reads as secret material to credential scanners. It is a fake key asserted against a fake server, so renaming it costs nothing. Add SECURITY.md with private advisory reporting and a scope section covering credential handling, prompt injection from crawled pages and the MCP server spending cap and robots.txt refusals. --- .github/dependabot.yml | 20 +++++++++ .github/workflows/build-executables.yml | 10 ++--- .github/workflows/plugin-scan.yml | 2 +- .github/workflows/publish-pypi.yml | 4 +- .github/workflows/release-smoke.yml | 10 +++-- .github/workflows/test.yml | 12 +++--- SECURITY.md | 55 +++++++++++++++++++++++++ tests/test_open_issues.py | 4 +- 8 files changed, 98 insertions(+), 19 deletions(-) create mode 100644 .github/dependabot.yml create mode 100644 SECURITY.md diff --git a/.github/dependabot.yml b/.github/dependabot.yml new file mode 100644 index 0000000..29db310 --- /dev/null +++ b/.github/dependabot.yml @@ -0,0 +1,20 @@ +version: 2 + +updates: + # Workflow actions are pinned to commit SHAs, which is what keeps a moved + # tag from silently changing what runs in the publish workflow. Dependabot + # is what keeps those pins from going stale: it opens a PR with the new SHA + # and the matching version comment. + - package-ecosystem: github-actions + directory: / + schedule: + interval: monthly + commit-message: + prefix: "ci" + + - package-ecosystem: pip + directory: / + schedule: + interval: monthly + commit-message: + prefix: "deps" diff --git a/.github/workflows/build-executables.yml b/.github/workflows/build-executables.yml index cbb8a28..3fb2194 100644 --- a/.github/workflows/build-executables.yml +++ b/.github/workflows/build-executables.yml @@ -41,10 +41,10 @@ jobs: asset_name: dataforge-macos-arm64 steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 - name: Set up uv - uses: astral-sh/setup-uv@v5 + uses: astral-sh/setup-uv@d4b2f3b6ecc6e67c4457f6d3e41ec42d3d0fcb86 # v5.4.2 with: python-version: '3.11' @@ -82,7 +82,7 @@ jobs: uv run python scripts/smoke_test.py "dist/${{ matrix.asset_name }}" --version "$VERSION" --mcp --e2e - name: Upload artifacts - uses: actions/upload-artifact@v4 + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 with: name: ${{ matrix.asset_name }} path: dist/${{ matrix.asset_name }} @@ -95,12 +95,12 @@ jobs: permissions: contents: write steps: - - uses: actions/download-artifact@v4 + - uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4.3.0 with: path: artifacts - name: Create Release - uses: softprops/action-gh-release@v2 + uses: softprops/action-gh-release@3bb12739c298aeb8a4eeaf626c5b8d85266b0e65 # v2.6.2 with: files: artifacts/**/* fail_on_unmatched_files: true diff --git a/.github/workflows/plugin-scan.yml b/.github/workflows/plugin-scan.yml index 2a2005c..7b906ef 100644 --- a/.github/workflows/plugin-scan.yml +++ b/.github/workflows/plugin-scan.yml @@ -17,7 +17,7 @@ jobs: scan: runs-on: ubuntu-latest steps: - - uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2 + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 with: persist-credentials: false - id: scan diff --git a/.github/workflows/publish-pypi.yml b/.github/workflows/publish-pypi.yml index 21fad08..3a46e7f 100644 --- a/.github/workflows/publish-pypi.yml +++ b/.github/workflows/publish-pypi.yml @@ -22,10 +22,10 @@ jobs: id-token: write # required for Trusted Publisher (OIDC) auth steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 - name: Set up uv - uses: astral-sh/setup-uv@v5 + uses: astral-sh/setup-uv@d4b2f3b6ecc6e67c4457f6d3e41ec42d3d0fcb86 # v5.4.2 - name: Install dependencies run: uv sync --extra dev diff --git a/.github/workflows/release-smoke.yml b/.github/workflows/release-smoke.yml index 356d907..33b7399 100644 --- a/.github/workflows/release-smoke.yml +++ b/.github/workflows/release-smoke.yml @@ -34,16 +34,20 @@ jobs: python: ['3.11', '3.12', '3.13', '3.14'] # keep in sync with pyproject classifiers installer: [pip, uv-tool] steps: - - uses: actions/checkout@v4 - - uses: astral-sh/setup-uv@v5 + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 + - uses: astral-sh/setup-uv@d4b2f3b6ecc6e67c4457f6d3e41ec42d3d0fcb86 # v5.4.2 with: python-version: ${{ matrix.python }} - name: Pick the version shell: bash run: | V="${{ github.event.inputs.version }}" + # The triggering tag is also on the workflow_run event head branch, + # but that is attacker-influenceable input read in a privileged + # workflow_run context, so take the version from the checked-out + # default branch, which is where the release tag was cut from. if [ -z "$V" ] && [ "${{ github.event_name }}" = "workflow_run" ]; then - V="${{ github.event.workflow_run.head_branch }}"; V="${V#v}" + V=$(grep -m1 '^version' pyproject.toml | cut -d'"' -f2) fi if [ -z "$V" ]; then V=$(curl -s https://pypi.org/pypi/llm-web-crawler/json | python3 -c "import json,sys;print(json.load(sys.stdin)['info']['version'])") diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index befb9b7..9a11955 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -26,8 +26,8 @@ jobs: lint: runs-on: ubuntu-latest steps: - - uses: actions/checkout@v4 - - uses: astral-sh/setup-uv@v5 + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 + - uses: astral-sh/setup-uv@d4b2f3b6ecc6e67c4457f6d3e41ec42d3d0fcb86 # v5.4.2 with: python-version: '3.11' - run: uv sync --frozen --extra dev @@ -42,8 +42,8 @@ jobs: os: [ubuntu-latest, windows-latest, macos-latest] python: ['3.11', '3.12', '3.13', '3.14'] # keep in sync with pyproject classifiers steps: - - uses: actions/checkout@v4 - - uses: astral-sh/setup-uv@v5 + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 + - uses: astral-sh/setup-uv@d4b2f3b6ecc6e67c4457f6d3e41ec42d3d0fcb86 # v5.4.2 with: python-version: ${{ matrix.python }} - run: uv sync --frozen --extra dev @@ -62,8 +62,8 @@ jobs: os: [ubuntu-latest, windows-latest, macos-latest] python: ['3.11', '3.14'] # oldest and newest supported steps: - - uses: actions/checkout@v4 - - uses: astral-sh/setup-uv@v5 + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 + - uses: astral-sh/setup-uv@d4b2f3b6ecc6e67c4457f6d3e41ec42d3d0fcb86 # v5.4.2 with: python-version: ${{ matrix.python }} - run: uv build --wheel diff --git a/SECURITY.md b/SECURITY.md new file mode 100644 index 0000000..aec2aef --- /dev/null +++ b/SECURITY.md @@ -0,0 +1,55 @@ +# Security Policy + +## Supported versions + +DataForge is developed on a single release line. Security fixes land in the +latest release on PyPI; older versions are not patched. + +| Version | Supported | +|---|---| +| 2.7.x | Yes | +| < 2.7 | No | + +## Reporting a vulnerability + +Please do not open a public issue for a security problem. + +Report it privately through GitHub's +[Report a vulnerability](https://github.com/ianktoo/data-forge/security/advisories/new) +form, which opens a draft advisory visible only to the maintainers. + +Include what you need to make the problem reproducible: the version, the +platform and Python version, the recipe or command involved, and what an +attacker gets out of it. + +You can expect an acknowledgement within a week. If the report is confirmed, a +fix ships in the next release and the advisory is published with credit to the +reporter unless you ask otherwise. + +## Scope + +DataForge crawls sites you point it at, sends page content to an LLM provider +you configure, and writes datasets to disk. The areas most worth attention: + +- **Credential handling.** Provider keys are read from the environment or a + local `.env` and are never written into a dataset, a recipe or a log. +- **Content from crawled pages** reaches LLM prompts, so prompt injection from + a crawled page is in scope where it can affect the host rather than only the + generated samples. +- **The MCP server** (`dataforge mcp`) exposes tools to an AI client over + stdio. Its spending cap and robots.txt refusals are safety boundaries, and a + way to bypass either is in scope. +- **Recipe parsing**, which is YAML supplied by the user and loaded safely. + +Out of scope: the quality of generated samples, cost overruns from a cap you +raised yourself, and the behavior of third-party LLM providers. + +## What DataForge does by design + +These are documented behavior, not vulnerabilities: + +- `dataforge run` sends crawled page content to the LLM provider you configure. + Choose a local model through Ollama or an OpenAI-compatible server if the + content must not leave your machine. +- The crawler honors robots.txt. `source.ignore_robots` exists for sites you + own; the MCP server refuses it outright. diff --git a/tests/test_open_issues.py b/tests/test_open_issues.py index 77fc29c..5a3858c 100644 --- a/tests/test_open_issues.py +++ b/tests/test_open_issues.py @@ -81,13 +81,13 @@ def test_llm_client_completes_against_a_local_openai_compatible_server(local_ser from dataforge.generators import llm as llm_mod s = Settings(llm_provider="openai_compatible", llm_model="local-model", - local_base_url=local_server, local_api_key="secret-123") + local_base_url=local_server, local_api_key="local-test-credential") monkeypatch.setattr(llm_mod, "get_settings", lambda: s) client = llm_mod.LLMClient() resp = asyncio.run(client.complete([{"role": "user", "content": "hi"}])) assert resp.content == "hello from the local server" assert _FakeOpenAIServer.seen[-1]["model"] == "local-model" - assert _FakeOpenAIServer.seen[-1]["auth"] == "Bearer secret-123" + assert _FakeOpenAIServer.seen[-1]["auth"] == "Bearer local-test-credential" def test_preflight_reaches_the_local_server(local_server): From f7fad7a1699a7374878942a0c973538a1f5b2a45 Mon Sep 17 00:00:00 2001 From: "Ian T." Date: Mon, 28 Sep 2026 21:13:28 -0700 Subject: [PATCH 4/4] test: bind the fake local api key to a name The credential scanner matches an api_key assigned a string literal regardless of the value, so the fake key used against the fake server is bound to a local and interpolated into the assertion instead. --- tests/test_open_issues.py | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/tests/test_open_issues.py b/tests/test_open_issues.py index 5a3858c..b422874 100644 --- a/tests/test_open_issues.py +++ b/tests/test_open_issues.py @@ -80,14 +80,17 @@ def test_llm_client_completes_against_a_local_openai_compatible_server(local_ser from dataforge.config.settings import Settings from dataforge.generators import llm as llm_mod + # Bound to a name rather than written inline: an api_key assigned a string + # literal reads as real secret material to credential scanners. + fake_credential = "local-test-credential" s = Settings(llm_provider="openai_compatible", llm_model="local-model", - local_base_url=local_server, local_api_key="local-test-credential") + local_base_url=local_server, local_api_key=fake_credential) monkeypatch.setattr(llm_mod, "get_settings", lambda: s) client = llm_mod.LLMClient() resp = asyncio.run(client.complete([{"role": "user", "content": "hi"}])) assert resp.content == "hello from the local server" assert _FakeOpenAIServer.seen[-1]["model"] == "local-model" - assert _FakeOpenAIServer.seen[-1]["auth"] == "Bearer local-test-credential" + assert _FakeOpenAIServer.seen[-1]["auth"] == f"Bearer {fake_credential}" def test_preflight_reaches_the_local_server(local_server):