diff --git a/.github/workflows/file-discovery-root-instruction-context.yml b/.github/workflows/file-discovery-root-instruction-context.yml new file mode 100644 index 0000000..760fa2a --- /dev/null +++ b/.github/workflows/file-discovery-root-instruction-context.yml @@ -0,0 +1,119 @@ +name: File Discovery — Root Instruction Context (Issue 15) + +on: + push: + branches: [main] + paths: + - fixtures/file-discovery/global-directory-root-instructions.json + workflow_dispatch: + +permissions: + contents: read + +concurrency: + group: file-discovery-root-instruction-context + cancel-in-progress: false + +jobs: + validate: + runs-on: ubuntu-24.04 + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: '3.13' + - name: Run tests before experiment + run: python3 -m unittest discover -s tests -v + + experiment: + needs: validate + runs-on: ubuntu-24.04 + environment: typesafe + timeout-minutes: 120 + strategy: + fail-fast: false + max-parallel: 3 + matrix: + include: + - slug: nession + repository: BestNathan/nession + revision: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + - slug: codex + repository: openai/codex + revision: d7b07d45517a793acfba4cbf8de697d723cceb46 + - slug: openclaw + repository: openclaw/openclaw + revision: 932abb0a841b522ebaa5b81921119a61b6a80b21 + steps: + - uses: actions/checkout@v4 + - uses: actions/checkout@v4 + with: + repository: ${{ matrix.repository }} + ref: ${{ matrix.revision }} + path: subject + persist-credentials: false + - uses: actions/setup-python@v5 + with: + python-version: '3.13' + - name: Verify frozen revision and run both arms + env: + TYPESAFE_MODEL: ${{ vars.TYPESAFE_MODEL || 'jev-latest' }} + TYPESAFE_API_URL: ${{ vars.TYPESAFE_API_URL || 'https://api.typesafe.ai/v1/systemone' }} + TYPESAFE_API_KEY: ${{ secrets.TYPESAFE_API_KEY || secrets.TYPESAFE_KEY || secrets.SYSTEM_ONE_API_KEY || secrets.TYPE_SAFE_API_KEY || vars.TYPESAFE_API_KEY || vars.TYPESAFE_KEY || vars.SYSTEM_ONE_API_KEY || vars.TYPE_SAFE_API_KEY }} + run: | + test "$GITHUB_ACTOR" = BestNathan + test -n "$TYPESAFE_API_KEY" + test "$(git -C subject rev-parse HEAD)" = "${{ matrix.revision }}" + python3 src/file_discovery_global_directory.py \ + --root subject --repository '${{ matrix.repository }}' \ + --output-dir artifacts/${{ matrix.slug }} \ + --config fixtures/file-discovery/global-directory-root-instructions.json \ + --run-url '${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}' + - name: Add repository results to run summary + if: always() + run: | + if [ -f artifacts/${{ matrix.slug }}/report.md ]; then + cat artifacts/${{ matrix.slug }}/report.md >> "$GITHUB_STEP_SUMMARY" + fi + - uses: actions/upload-artifact@v4 + if: always() + with: + name: issue15-root-instructions-${{ matrix.slug }} + path: artifacts/${{ matrix.slug }}/ + retention-days: 90 + if-no-files-found: warn + + report: + if: always() + needs: [validate, experiment] + runs-on: ubuntu-24.04 + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: '3.13' + - uses: actions/download-artifact@v4 + continue-on-error: true + with: + pattern: issue15-root-instructions-* + path: artifacts/raw + - name: Aggregate metrics and create Workflow metadata + env: + VALIDATE_RESULT: ${{ needs.validate.result }} + EXPERIMENT_RESULT: ${{ needs.experiment.result }} + run: | + python3 src/file_discovery_global_directory.py \ + --aggregate artifacts/raw --output-dir artifacts/report \ + --config fixtures/file-discovery/global-directory-root-instructions.json \ + --run-url '${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}' \ + --run-id '${{ github.run_id }}' --commit '${{ github.sha }}' \ + --validate-status "$VALIDATE_RESULT" --experiment-status "$EXPERIMENT_RESULT" \ + --report-run-url '${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}' \ + --report-run-id '${{ github.run_id }}' + cat artifacts/report/report.md >> "$GITHUB_STEP_SUMMARY" + - uses: actions/upload-artifact@v4 + if: always() + with: + name: issue15-aggregate-${{ github.run_id }} + path: artifacts/report/ + retention-days: 90 diff --git a/fixtures/file-discovery/global-directory-root-instructions.json b/fixtures/file-discovery/global-directory-root-instructions.json new file mode 100644 index 0000000..d6d77bc --- /dev/null +++ b/fixtures/file-discovery/global-directory-root-instructions.json @@ -0,0 +1,24 @@ +{ + "experiment": "issue-15-root-instruction-context", + "model": "jev-latest", + "profile": "compact_state_v4", + "repeats": 1, + "directory_threshold": 0.65, + "file_threshold": 0.65, + "workers": 4, + "batch_size": 64, + "max_candidate_bytes": 48000, + "load_root_instruction_context": true, + "repository_revisions": { + "BestNathan/nession": "7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df", + "openai/codex": "d7b07d45517a793acfba4cbf8de697d723cceb46", + "openclaw/openclaw": "932abb0a841b522ebaa5b81921119a61b6a80b21" + }, + "notes": [ + "Issue 14 architecture and thresholds are frozen unchanged.", + "Phase 1 loads root CLAUDE.md verbatim when present, otherwise root AGENTS.md verbatim.", + "No summary, generated repository prior, nested instruction lookup, or filename-derived context is permitted.", + "Repository instruction context is injected into Phase 1 directory state only; Phase 2 file scoring remains unchanged.", + "The complete eligible directory population is fixed before model decisions; no ancestor gating or recursive expansion is permitted." + ] +} diff --git a/src/file_discovery_global_directory.py b/src/file_discovery_global_directory.py index 3fc474c..c385727 100644 --- a/src/file_discovery_global_directory.py +++ b/src/file_discovery_global_directory.py @@ -20,6 +20,31 @@ MAX_CANDIDATE_BYTES = 48000 DIRECTORY_THRESHOLD = 0.65 FILE_THRESHOLD = 0.65 +ROOT_INSTRUCTION_FILES = ("CLAUDE.md", "AGENTS.md") + + +def load_root_instruction_context(root): + """Load one repository-authored root instruction file verbatim. + + CLAUDE.md takes precedence because some repositories expose AGENTS.md as an + alias to the same content. No summarization, nested lookup, or generated + repository knowledge is introduced here. + """ + root = Path(root).resolve() + for filename in ROOT_INSTRUCTION_FILES: + path = root / filename + try: + if not path.is_file(): + continue + content = path.read_text(encoding="utf-8") + except (OSError, UnicodeError): + continue + return { + "source": filename, + "content": content, + "bytes": len(content.encode("utf-8")), + } + return None def eligible_file(entry): @@ -130,14 +155,21 @@ def score_candidates(self, query, stage, candidates): "false": "The direct files in this exact directory are unlikely to contain material task evidence.", }, } - response, usage = self.send("global_directory_direct_file_relevance", { + state = { "goal": query, "phase": "global_directory_classification", "population_size": getattr(self, "logical_directory_population", len(candidates)), "source_visible": False, "contract": ("Independent multi-label decisions over one fixed global directory population. " "Directory scores do not expose, hide, or gate any other directory."), - }, questions) + } + instruction_context = getattr(self, "root_instruction_context", None) + if instruction_context: + state["repository_instruction_context"] = { + "source": instruction_context["source"], + "content": instruction_context["content"], + } + response, usage = self.send("global_directory_direct_file_relevance", state, questions) answers = response.get("answers", {}) result = [] for index, candidate in enumerate(candidates): @@ -264,6 +296,8 @@ def flat_baseline(root, query, scorer): def run_repository(args): config = json.loads(Path(args.config).read_text(encoding="utf-8")) + instruction_context = (load_root_instruction_context(args.root) + if config.get("load_root_instruction_context") else None) cases = json.loads(Path(args.cases).read_text(encoding="utf-8"))["cases"] cases = [case for case in cases if case["repository"] == args.repository] output = Path(args.output_dir) @@ -278,6 +312,8 @@ def run_repository(args): profile=config["profile"], repository_context={}, endpoint=os.getenv("TYPESAFE_API_URL", API_URL), model=os.getenv("TYPESAFE_MODEL", config["model"] or MODEL)) + decider.root_instruction_context = (instruction_context + if arm == "global_directory" else None) scorer = BatchScorer(case["goal"], decider, workers=int(config["workers"]), batch_size=int(config["batch_size"]), max_candidate_bytes=int(config["max_candidate_bytes"])) @@ -290,6 +326,10 @@ def run_repository(args): workers=int(config["workers"]), batch_size=int(config["batch_size"]), max_candidate_bytes=int(config["max_candidate_bytes"]))) + active_context = decider.root_instruction_context + result["root_instruction_context_present"] = bool(active_context) + result["root_instruction_context_source"] = (active_context or {}).get("source") + result["root_instruction_context_bytes"] = int((active_context or {}).get("bytes", 0)) selected_files = {item["path"] for item in result["promoted_files"]} scored_files = set(result["file_scores"]) primary = set(case["primary_files"]) @@ -464,7 +504,9 @@ def aggregate(args): "candidate_file_count", "file_decisions", "directory_decisions", "model_visible_nodes", "candidate_recall", "primary_recall", "missing_primary", "missing_primary_causes", "target_details", "phase_usage", "physical_usage", "pricing", "phase1_wall_time_ms", - "phase2_enumeration_wall_time_ms", "phase2_scoring_wall_time_ms", "wall_time_ms", "error") + "phase2_enumeration_wall_time_ms", "phase2_scoring_wall_time_ms", "wall_time_ms", + "root_instruction_context_present", "root_instruction_context_source", + "root_instruction_context_bytes", "error") compact_rows = [] for row in rows: compact = {key: row[key] for key in summary_fields if key in row} diff --git a/tests/test_file_discovery_global_directory.py b/tests/test_file_discovery_global_directory.py index fcbcb38..053ffe5 100644 --- a/tests/test_file_discovery_global_directory.py +++ b/tests/test_file_discovery_global_directory.py @@ -5,7 +5,12 @@ sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src")) -from file_discovery_global_directory import enumerate_directories, enumerate_direct_files, two_stage_discovery +from file_discovery_global_directory import ( + enumerate_directories, + enumerate_direct_files, + load_root_instruction_context, + two_stage_discovery, +) class FakeScorer: @@ -42,6 +47,24 @@ def test_nested_directory_is_visible_even_when_parent_scores_low(self): self.assertEqual(["pkg/sub/target.py"], list(result["file_scores"])) self.assertNotIn("pkg/sub/child/hidden.py", result["file_scores"]) + def test_root_instruction_context_prefers_claude_and_does_not_merge_agents(self): + with tempfile.TemporaryDirectory() as temp: + root = Path(temp) + (root / "CLAUDE.md").write_text("claude context", encoding="utf-8") + (root / "AGENTS.md").write_text("agents context", encoding="utf-8") + context = load_root_instruction_context(root) + self.assertEqual("CLAUDE.md", context["source"]) + self.assertEqual("claude context", context["content"]) + self.assertEqual(len(b"claude context"), context["bytes"]) + + def test_root_instruction_context_falls_back_to_agents(self): + with tempfile.TemporaryDirectory() as temp: + root = Path(temp) + (root / "AGENTS.md").write_text("agents context", encoding="utf-8") + context = load_root_instruction_context(root) + self.assertEqual("AGENTS.md", context["source"]) + self.assertEqual("agents context", context["content"]) + def test_direct_file_enumeration_deduplicates_selected_directories(self): with tempfile.TemporaryDirectory() as temp: root = Path(temp)