diff --git a/.github/workflows/system-one-code-locator.yml b/.github/workflows/system-one-code-locator.yml new file mode 100644 index 00000000..c6c00577 --- /dev/null +++ b/.github/workflows/system-one-code-locator.yml @@ -0,0 +1,435 @@ +name: system-one-code-locator-experiment + +on: + push: + paths: + - "research/code-locator/src/**" + - "research/code-locator/tests/**" + - "research/code-locator/fixtures/**" + - "research/code-locator/docs/**" + - "research/code-locator/README.md" + - ".github/workflows/system-one-code-locator.yml" + pull_request: + paths: + - "research/code-locator/src/**" + - "research/code-locator/tests/**" + - "research/code-locator/fixtures/**" + - "research/code-locator/docs/**" + - "research/code-locator/README.md" + - ".github/workflows/system-one-code-locator.yml" + workflow_dispatch: + inputs: + subject_repository: + description: Repository used by the code-localization experiment + required: true + default: BestNathan/nession + type: string + subject_ref: + description: Git ref of the subject repository + required: true + default: staging + type: string + query: + description: Code-localization task + required: true + default: Help me optimize the websocket connection implementation + type: string + directory_threshold: + description: Directory Noul threshold + required: true + default: "0.35" + type: string + file_threshold: + description: File Noul threshold + required: true + default: "0.50" + type: string + line_threshold: + description: Source-line Noul threshold + required: true + default: "0.70" + type: string + batch_size: + description: Noul questions per TypeSafe request + required: true + default: "48" + type: string + model: + description: TypeSafe model alias + required: true + default: jev-latest + type: string + run_typesafe: + description: Run the real TypeSafe System One experiment + required: true + default: false + type: boolean + +permissions: + contents: read + +concurrency: + group: system-one-code-locator-${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: false + +env: + QUERY: ${{ inputs.query || 'Help me optimize the websocket connection implementation' }} + DIRECTORY_THRESHOLD: ${{ inputs.directory_threshold || '0.35' }} + FILE_THRESHOLD: ${{ inputs.file_threshold || '0.50' }} + LINE_THRESHOLD: ${{ inputs.line_threshold || '0.70' }} + BATCH_SIZE: ${{ inputs.batch_size || '48' }} + TYPESAFE_MODEL: ${{ inputs.model || vars.TYPESAFE_MODEL || 'jev-latest' }} + +jobs: + offline: + name: Offline code-locator validation + runs-on: ubuntu-24.04 + timeout-minutes: 10 + + steps: + - uses: actions/checkout@v4 + + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Prepare artifact directory + run: mkdir -p artifacts/system-one-code-locator/offline + + - name: Run unit tests + shell: bash + run: | + set -o pipefail + python3 -m unittest discover -s research/code-locator/tests -p 'test_*.py' -v 2>&1 | tee artifacts/system-one-code-locator/offline/tests.log + + - name: Run deterministic fixture experiment + shell: bash + run: | + set -o pipefail + python3 research/code-locator/src/system_one_code_locator.py research/code-locator/fixtures/repository "$QUERY" --offline-decider --directory-threshold "$DIRECTORY_THRESHOLD" --file-threshold "$FILE_THRESHOLD" --line-threshold 0.60 --batch-size "$BATCH_SIZE" --trace-file artifacts/system-one-code-locator/offline/trace.jsonl --output-json artifacts/system-one-code-locator/offline/result.json 2>&1 | tee artifacts/system-one-code-locator/offline/run.log + + - name: Save run metadata and summary + env: + RUN_ID: ${{ github.run_id }} + RUN_ATTEMPT: ${{ github.run_attempt }} + COMMIT_SHA: ${{ github.sha }} + REF_NAME: ${{ github.ref_name }} + run: | + python3 - <<'PY' + import json + import os + from pathlib import Path + + root = Path("artifacts/system-one-code-locator/offline") + result = json.loads((root / "result.json").read_text()) + + metadata = { + "schema_version": 1, + "mode": "offline", + "query": os.environ["QUERY"], + "model": "offline-lexical-fixture", + "thresholds": { + "directory": float(os.environ["DIRECTORY_THRESHOLD"]), + "file": float(os.environ["FILE_THRESHOLD"]), + "line": 0.60, + }, + "batch_size": int(os.environ["BATCH_SIZE"]), + "github_run_id": os.environ["RUN_ID"], + "github_run_attempt": os.environ["RUN_ATTEMPT"], + "commit_sha": os.environ["COMMIT_SHA"], + "ref_name": os.environ["REF_NAME"], + } + (root / "run-manifest.json").write_text( + json.dumps(metadata, indent=2) + "\n", + encoding="utf-8", + ) + + lines = [ + "# System One Code Locator — offline fixture", + "", + f"- Query: `{metadata['query']}`", + f"- Directories: {len(result.get('directories', []))}", + f"- Files: {len(result.get('files', []))}", + f"- Snippets: {len(result.get('snippets', []))}", + f"- Metrics: `{result.get('metrics', {})}`", + ] + (root / "summary.md").write_text("\n".join(lines) + "\n") + PY + + - name: Publish summary + if: always() + run: | + if test -f artifacts/system-one-code-locator/offline/summary.md; then + cat artifacts/system-one-code-locator/offline/summary.md >> "$GITHUB_STEP_SUMMARY" + fi + + - name: Upload offline experiment artifacts + if: always() + uses: actions/upload-artifact@v4 + with: + name: system-one-code-locator-offline-${{ github.run_id }}-${{ github.run_attempt }} + path: artifacts/system-one-code-locator/offline/ + if-no-files-found: error + retention-days: 90 + + typesafe: + name: Real TypeSafe code-locator experiment + if: ${{ github.event_name == 'workflow_dispatch' && inputs.run_typesafe }} + runs-on: ubuntu-24.04 + environment: typesafe + timeout-minutes: 120 + + steps: + - name: Checkout Narness harness + uses: actions/checkout@v4 + with: + path: narness + fetch-depth: 1 + persist-credentials: false + + - name: Checkout subject repository + uses: actions/checkout@v4 + with: + repository: ${{ inputs.subject_repository }} + ref: ${{ inputs.subject_ref }} + path: subject + fetch-depth: 1 + persist-credentials: false + + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Require TypeSafe credentials + env: + TYPESAFE_API_KEY: ${{ secrets.TYPESAFE_API_KEY || vars.TYPESAFE_API_KEY }} + run: | + test -n "$TYPESAFE_API_KEY" || { + echo "TYPESAFE_API_KEY is required in the typesafe environment." >&2 + exit 2 + } + + - name: Prepare artifact directory + run: mkdir -p artifacts/system-one-code-locator/typesafe + + - name: Save run manifest + env: + SUBJECT_REPOSITORY: ${{ inputs.subject_repository }} + SUBJECT_REF: ${{ inputs.subject_ref }} + RUN_ID: ${{ github.run_id }} + RUN_ATTEMPT: ${{ github.run_attempt }} + HARNESS_SHA: ${{ github.sha }} + run: | + python3 - <<'PY' + import json + import os + import subprocess + from pathlib import Path + + def revision(path): + return subprocess.check_output( + ["git", "-C", path, "rev-parse", "HEAD"], + text=True, + ).strip() + + payload = { + "schema_version": 1, + "mode": "typesafe", + "query": os.environ["QUERY"], + "model": os.environ["TYPESAFE_MODEL"], + "thresholds": { + "directory": float(os.environ["DIRECTORY_THRESHOLD"]), + "file": float(os.environ["FILE_THRESHOLD"]), + "line": float(os.environ["LINE_THRESHOLD"]), + }, + "batch_size": int(os.environ["BATCH_SIZE"]), + "harness": { + "repository": os.environ.get("GITHUB_REPOSITORY"), + "revision": revision("narness"), + }, + "subject": { + "repository": os.environ["SUBJECT_REPOSITORY"], + "requested_ref": os.environ["SUBJECT_REF"], + "revision": revision("subject"), + }, + "github_run_id": os.environ["RUN_ID"], + "github_run_attempt": os.environ["RUN_ATTEMPT"], + } + root = Path("artifacts/system-one-code-locator/typesafe") + (root / "run-manifest.json").write_text( + json.dumps(payload, indent=2) + "\n", + encoding="utf-8", + ) + PY + + - name: Run real System One code locator + env: + TYPESAFE_API_KEY: ${{ secrets.TYPESAFE_API_KEY || vars.TYPESAFE_API_KEY }} + TYPESAFE_API_URL: ${{ vars.TYPESAFE_API_URL || 'https://api.typesafe.ai/v1/systemone' }} + shell: bash + run: | + set -o pipefail + python3 narness/research/code-locator/src/system_one_code_locator.py subject "$QUERY" --directory-threshold "$DIRECTORY_THRESHOLD" --file-threshold "$FILE_THRESHOLD" --line-threshold "$LINE_THRESHOLD" --batch-size "$BATCH_SIZE" --model "$TYPESAFE_MODEL" --trace-file artifacts/system-one-code-locator/typesafe/trace.jsonl --output-json artifacts/system-one-code-locator/typesafe/result.json 2>&1 | tee artifacts/system-one-code-locator/typesafe/run.log + + - name: Build research summary + if: always() + run: | + python3 - <<'PY' + import json + import os + from pathlib import Path + + root = Path("artifacts/system-one-code-locator/typesafe") + manifest_path = root / "run-manifest.json" + result_path = root / "result.json" + + lines = ["# System One Code Locator — TypeSafe run", ""] + + if manifest_path.exists(): + manifest = json.loads(manifest_path.read_text()) + lines += [ + f"- Subject: `{manifest['subject']['repository']}@{manifest['subject']['revision']}`", + f"- Query: `{manifest['query']}`", + f"- Model: `{manifest['model']}`", + f"- Thresholds: `{manifest['thresholds']}`", + "", + ] + + if result_path.exists(): + result = json.loads(result_path.read_text()) + lines += [ + "## Result", + "", + f"- Directories: {len(result.get('directories', []))}", + f"- Files: {len(result.get('files', []))}", + f"- Snippets: {len(result.get('snippets', []))}", + f"- Metrics: `{result.get('metrics', {})}`", + "", + "### Selected files", + "", + ] + lines += [ + f"- `{item['id']}` ({item['score']:.3f})" + for item in result.get("files", []) + ] + else: + lines += [ + "## Result", + "", + "The experiment did not produce result.json. Inspect run.log and trace.jsonl.", + ] + + summary = "\n".join(lines) + "\n" + (root / "summary.md").write_text(summary) + with open(Path(os.environ["GITHUB_STEP_SUMMARY"]), "a") as handle: + handle.write(summary) + PY + env: + PYTHONUNBUFFERED: "1" + + - name: Upload TypeSafe experiment artifacts + if: always() + uses: actions/upload-artifact@v4 + with: + name: system-one-code-locator-typesafe-${{ github.run_id }}-${{ github.run_attempt }} + path: artifacts/system-one-code-locator/typesafe/ + if-no-files-found: warn + retention-days: 90 + + persist-record: + name: Persist experiment record + needs: + - offline + - typesafe + if: ${{ always() && github.event_name != 'pull_request' }} + runs-on: ubuntu-24.04 + timeout-minutes: 10 + permissions: + contents: write + + steps: + - name: Checkout experiment branch + uses: actions/checkout@v4 + with: + ref: ${{ github.ref_name }} + fetch-depth: 1 + persist-credentials: true + + - name: Download offline evidence + uses: actions/download-artifact@v4 + continue-on-error: true + with: + name: system-one-code-locator-offline-${{ github.run_id }}-${{ github.run_attempt }} + path: .experiment-record/offline + + - name: Download TypeSafe evidence + if: ${{ github.event_name == 'workflow_dispatch' && inputs.run_typesafe }} + uses: actions/download-artifact@v4 + continue-on-error: true + with: + name: system-one-code-locator-typesafe-${{ github.run_id }}-${{ github.run_attempt }} + path: .experiment-record/typesafe + + - name: Materialize repository record + env: + OFFLINE_RESULT: ${{ needs.offline.result }} + TYPESAFE_RESULT: ${{ needs.typesafe.result }} + RUN_URL: https://github.com/${{ github.repository }}/actions/runs/${{ github.run_id }} + shell: bash + run: | + set -euo pipefail + + stamp="$(TZ=Asia/Shanghai date +'%Y%m%dT%H%M%S%z')" + record="research/code-locator/runs/${stamp}-run-${GITHUB_RUN_ID}-attempt-${GITHUB_RUN_ATTEMPT}" + mkdir -p "$record" + + if test -d .experiment-record/offline; then + mkdir -p "$record/offline" + cp -a .experiment-record/offline/. "$record/offline/" + fi + + if test -d .experiment-record/typesafe; then + mkdir -p "$record/typesafe" + cp -a .experiment-record/typesafe/. "$record/typesafe/" + fi + + { + echo "# System One Code Locator experiment" + echo + printf -- '- Recorded at: `%s` (UTC+8)\n' "$stamp" + printf -- '- GitHub run: [%s](%s)\n' "$GITHUB_RUN_ID" "$RUN_URL" + printf -- '- Run attempt: `%s`\n' "$GITHUB_RUN_ATTEMPT" + printf -- '- Event: `%s`\n' "$GITHUB_EVENT_NAME" + printf -- '- Repository: `%s`\n' "$GITHUB_REPOSITORY" + printf -- '- Ref: `%s`\n' "$GITHUB_REF_NAME" + printf -- '- Trigger SHA: `%s`\n' "$GITHUB_SHA" + printf -- '- Offline job: `%s`\n' "$OFFLINE_RESULT" + printf -- '- TypeSafe job: `%s`\n' "$TYPESAFE_RESULT" + echo + echo "## Evidence" + echo + echo "- [Offline evidence](offline/) when the offline validation produced an artifact." + echo "- [TypeSafe evidence](typesafe/) when a real TypeSafe experiment was requested." + echo + echo "The files in this directory are an immutable snapshot of the evidence produced by this workflow run." + } > "$record/README.md" + + echo "RECORD_DIR=$record" >> "$GITHUB_ENV" + + - name: Commit experiment record + shell: bash + run: | + set -euo pipefail + git config user.name "github-actions[bot]" + git config user.email "41898282+github-actions[bot]@users.noreply.github.com" + + git add "$RECORD_DIR" + + if git diff --cached --quiet; then + echo "No experiment evidence to commit." + exit 0 + fi + + git commit -m "research: record code locator run ${GITHUB_RUN_ID}" + git pull --rebase origin "${GITHUB_REF_NAME}" + git push origin "HEAD:${GITHUB_REF_NAME}" diff --git a/README.md b/README.md index 06ec4a77..d80e5dc7 100644 --- a/README.md +++ b/README.md @@ -224,7 +224,7 @@ bash plugins/narness/scripts/narness-rust-test-integration.sh examples/rust-work The example contains a root workspace contract, a task Skill with progressive disclosure, unit and integration evidence, an environment declaration, and a stable CI required gate. -The repository also contains research prototypes that are intentionally outside the canonical Narness runtime scope, including the [System One Kubernetes command generator](examples/system-one-k8s/README.md) used by the progressive-action-space topic. +The repository also contains research prototypes that are intentionally outside the canonical Narness runtime scope, including the [System One Kubernetes command generator](examples/system-one-k8s/README.md) and [System One code locator](research/code-locator/README.md) used by the progressive-action-space topic. ## Adopt Narness @@ -259,6 +259,7 @@ Start with [docs/adoption.md](docs/adoption.md): - [Change-to-Evidence Planning topic](docs/topics/change-to-evidence-planning/README.md) - [System One Progressive Action Spaces topic](docs/topics/system-one-progressive-action-space/README.md) - [System One Kubernetes command generator](examples/system-one-k8s/README.md) +- [System One code locator](research/code-locator/README.md) - [AI Workspace architecture](docs/architecture.md) - [Adoption guide](docs/adoption.md) - [Runnable Rust example](examples/rust-workspace/README.md) diff --git a/docs/topics/system-one-progressive-action-space/README.md b/docs/topics/system-one-progressive-action-space/README.md index f13ae080..e5e5efd4 100644 --- a/docs/topics/system-one-progressive-action-space/README.md +++ b/docs/topics/system-one-progressive-action-space/README.md @@ -587,7 +587,57 @@ The result should distinguish between tasks that are naturally frontier-driven a - Can the same runtime interface serve Kubernetes, coding, and browser environments? - How much state should be sent to System One at each branch? -## 18. Relationship to Narness +## 18. Code localization experiment + +The second reference experiment applies the same progressive-disclosure idea to source-code localization. + +```text +repository + -> directory candidates + -> relevant directories + -> file candidates + -> relevant files + -> source-line candidates + -> grounded code snippets +``` + +Unlike Kubernetes action selection, this stage is not a single-winner decision. Multiple directories, files, and source ranges can all be relevant, so the prototype uses independent Noul judgments rather than Choice. + +The harness owns traversal, filesystem IO, batching, thresholds, provenance, and range merging. System One only estimates candidate relevance. + +This broadens the working runtime abstraction: + +```text +StateSpaceGenerator + -> CandidateSet + -> DecisionPrimitive + -> TransitionPolicy + -> EvidenceRecorder +``` + +The decision primitive can therefore vary by state: + +- `Choice` for selecting one grounded action from a local frontier; +- `Noul` for independently retaining multiple relevant candidates; +- deterministic short-circuiting where model judgment is unnecessary. + +See [Hierarchical Code Localization with System One](../../../research/code-locator/docs/design.md) for the hypotheses, threshold/recall analysis, evidence contract, and experiment plan. + +## 19. Research evidence workflows + +The experiments use separate workflows because they exercise different decision primitives, inputs, costs, and failure modes. + +- [System One Kubernetes experiment](../../../.github/workflows/system-one-k8s-experiment.yml) owns only the Kubernetes state-machine demo. +- [System One Code Locator experiment](../../../.github/workflows/system-one-code-locator.yml) owns only hierarchical code localization. + +The Code Locator workflow has two layers: + +1. deterministic fixture validation on Code Locator changes, requiring no external model service; +2. manually dispatched real-System-One localization against a configurable subject repository. + +Its artifact contains only Code Locator evidence: run metadata, execution log, `trace.jsonl`, `result.json`, and `summary.md`. This keeps repository-localization research independent from Kubernetes results and makes failures, costs, and parameter sweeps attributable to one experiment. + +## 20. Relationship to Narness Narness currently defines itself as AI Workspace Engineering and explicitly does not aim to become a general Agent Runtime. @@ -599,9 +649,13 @@ It is still relevant because it explores a neighboring harness-engineering quest If the conclusions stabilize, some of them may graduate into Narness concepts around progressive disclosure, capability representation, deterministic execution, policy, evidence, and agent observability without requiring Narness itself to own a runtime. -## Reference implementation +## Reference implementations -See [System One Kubernetes Command Generator](../../../examples/system-one-k8s/README.md). +- [System One Kubernetes Command Generator](../../../examples/system-one-k8s/README.md) +- [System One Code Locator](../../../research/code-locator/README.md) +- [Code localization research note](../../../research/code-locator/docs/design.md) +- [System One Kubernetes experiment workflow](../../../.github/workflows/system-one-k8s-experiment.yml) +- [System One Code Locator experiment workflow](../../../.github/workflows/system-one-code-locator.yml) ## Related Narness topics diff --git a/research/code-locator/README.md b/research/code-locator/README.md new file mode 100644 index 00000000..9c534649 --- /dev/null +++ b/research/code-locator/README.md @@ -0,0 +1,219 @@ +# System One Code Locator + +> Experimental research project. This is a localization harness, not a code-changing agent and not part of the canonical Narness runtime. + +This directory is the complete Code Locator research unit: implementation, fixtures, tests, design notes, pilot analyses, and immutable run records live together here. + +## Project layout + +```text +research/code-locator/ +├── README.md +├── src/ +│ └── system_one_code_locator.py +├── tests/ +│ └── test_system_one_code_locator.py +├── fixtures/ +│ └── repository/ +├── docs/ +│ ├── design.md +│ └── pilots/ +│ └── nession-websocket-2026-09-23.md +└── runs/ + └──