Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
119 changes: 119 additions & 0 deletions .github/workflows/file-discovery-root-instruction-context.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,119 @@
name: File Discovery — Root Instruction Context (Issue 15)

on:
push:
branches: [main]
paths:
- fixtures/file-discovery/global-directory-root-instructions.json
workflow_dispatch:

permissions:
contents: read

concurrency:
group: file-discovery-root-instruction-context
cancel-in-progress: false

jobs:
validate:
runs-on: ubuntu-24.04
steps:
- uses: actions/checkout@v4
- uses: actions/setup-python@v5
with:
python-version: '3.13'
- name: Run tests before experiment
run: python3 -m unittest discover -s tests -v

experiment:
needs: validate
runs-on: ubuntu-24.04
environment: typesafe
timeout-minutes: 120
strategy:
fail-fast: false
max-parallel: 3
matrix:
include:
- slug: nession
repository: BestNathan/nession
revision: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df
- slug: codex
repository: openai/codex
revision: d7b07d45517a793acfba4cbf8de697d723cceb46
- slug: openclaw
repository: openclaw/openclaw
revision: 932abb0a841b522ebaa5b81921119a61b6a80b21
steps:
- uses: actions/checkout@v4
- uses: actions/checkout@v4
with:
repository: ${{ matrix.repository }}
ref: ${{ matrix.revision }}
path: subject
persist-credentials: false
- uses: actions/setup-python@v5
with:
python-version: '3.13'
- name: Verify frozen revision and run both arms
env:
TYPESAFE_MODEL: ${{ vars.TYPESAFE_MODEL || 'jev-latest' }}
TYPESAFE_API_URL: ${{ vars.TYPESAFE_API_URL || 'https://api.typesafe.ai/v1/systemone' }}
TYPESAFE_API_KEY: ${{ secrets.TYPESAFE_API_KEY || secrets.TYPESAFE_KEY || secrets.SYSTEM_ONE_API_KEY || secrets.TYPE_SAFE_API_KEY || vars.TYPESAFE_API_KEY || vars.TYPESAFE_KEY || vars.SYSTEM_ONE_API_KEY || vars.TYPE_SAFE_API_KEY }}
run: |
test "$GITHUB_ACTOR" = BestNathan
test -n "$TYPESAFE_API_KEY"
test "$(git -C subject rev-parse HEAD)" = "${{ matrix.revision }}"
python3 src/file_discovery_global_directory.py \
--root subject --repository '${{ matrix.repository }}' \
--output-dir artifacts/${{ matrix.slug }} \
--config fixtures/file-discovery/global-directory-root-instructions.json \
--run-url '${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}'
- name: Add repository results to run summary
if: always()
run: |
if [ -f artifacts/${{ matrix.slug }}/report.md ]; then
cat artifacts/${{ matrix.slug }}/report.md >> "$GITHUB_STEP_SUMMARY"
fi
- uses: actions/upload-artifact@v4
if: always()
with:
name: issue15-root-instructions-${{ matrix.slug }}
path: artifacts/${{ matrix.slug }}/
retention-days: 90
if-no-files-found: warn

report:
if: always()
needs: [validate, experiment]
runs-on: ubuntu-24.04
steps:
- uses: actions/checkout@v4
- uses: actions/setup-python@v5
with:
python-version: '3.13'
- uses: actions/download-artifact@v4
continue-on-error: true
with:
pattern: issue15-root-instructions-*
path: artifacts/raw
- name: Aggregate metrics and create Workflow metadata
env:
VALIDATE_RESULT: ${{ needs.validate.result }}
EXPERIMENT_RESULT: ${{ needs.experiment.result }}
run: |
python3 src/file_discovery_global_directory.py \
--aggregate artifacts/raw --output-dir artifacts/report \
--config fixtures/file-discovery/global-directory-root-instructions.json \
--run-url '${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}' \
--run-id '${{ github.run_id }}' --commit '${{ github.sha }}' \
--validate-status "$VALIDATE_RESULT" --experiment-status "$EXPERIMENT_RESULT" \
--report-run-url '${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}' \
--report-run-id '${{ github.run_id }}'
cat artifacts/report/report.md >> "$GITHUB_STEP_SUMMARY"
- uses: actions/upload-artifact@v4
if: always()
with:
name: issue15-aggregate-${{ github.run_id }}
path: artifacts/report/
retention-days: 90
24 changes: 24 additions & 0 deletions fixtures/file-discovery/global-directory-root-instructions.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,24 @@
{
"experiment": "issue-15-root-instruction-context",
"model": "jev-latest",
"profile": "compact_state_v4",
"repeats": 1,
"directory_threshold": 0.65,
"file_threshold": 0.65,
"workers": 4,
"batch_size": 64,
"max_candidate_bytes": 48000,
"load_root_instruction_context": true,
"repository_revisions": {
"BestNathan/nession": "7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df",
"openai/codex": "d7b07d45517a793acfba4cbf8de697d723cceb46",
"openclaw/openclaw": "932abb0a841b522ebaa5b81921119a61b6a80b21"
},
"notes": [
"Issue 14 architecture and thresholds are frozen unchanged.",
"Phase 1 loads root CLAUDE.md verbatim when present, otherwise root AGENTS.md verbatim.",
"No summary, generated repository prior, nested instruction lookup, or filename-derived context is permitted.",
"Repository instruction context is injected into Phase 1 directory state only; Phase 2 file scoring remains unchanged.",
"The complete eligible directory population is fixed before model decisions; no ancestor gating or recursive expansion is permitted."
]
}
48 changes: 45 additions & 3 deletions src/file_discovery_global_directory.py
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,31 @@
MAX_CANDIDATE_BYTES = 48000
DIRECTORY_THRESHOLD = 0.65
FILE_THRESHOLD = 0.65
ROOT_INSTRUCTION_FILES = ("CLAUDE.md", "AGENTS.md")


def load_root_instruction_context(root):
"""Load one repository-authored root instruction file verbatim.

CLAUDE.md takes precedence because some repositories expose AGENTS.md as an
alias to the same content. No summarization, nested lookup, or generated
repository knowledge is introduced here.
"""
root = Path(root).resolve()
for filename in ROOT_INSTRUCTION_FILES:
path = root / filename
try:
if not path.is_file():
continue
content = path.read_text(encoding="utf-8")
except (OSError, UnicodeError):
continue
return {
"source": filename,
"content": content,
"bytes": len(content.encode("utf-8")),
}
return None


def eligible_file(entry):
Expand Down Expand Up @@ -130,14 +155,21 @@ def score_candidates(self, query, stage, candidates):
"false": "The direct files in this exact directory are unlikely to contain material task evidence.",
},
}
response, usage = self.send("global_directory_direct_file_relevance", {
state = {
"goal": query,
"phase": "global_directory_classification",
"population_size": getattr(self, "logical_directory_population", len(candidates)),
"source_visible": False,
"contract": ("Independent multi-label decisions over one fixed global directory population. "
"Directory scores do not expose, hide, or gate any other directory."),
}, questions)
}
instruction_context = getattr(self, "root_instruction_context", None)
if instruction_context:
state["repository_instruction_context"] = {
"source": instruction_context["source"],
"content": instruction_context["content"],
}
response, usage = self.send("global_directory_direct_file_relevance", state, questions)
answers = response.get("answers", {})
result = []
for index, candidate in enumerate(candidates):
Expand Down Expand Up @@ -264,6 +296,8 @@ def flat_baseline(root, query, scorer):

def run_repository(args):
config = json.loads(Path(args.config).read_text(encoding="utf-8"))
instruction_context = (load_root_instruction_context(args.root)
if config.get("load_root_instruction_context") else None)
cases = json.loads(Path(args.cases).read_text(encoding="utf-8"))["cases"]
cases = [case for case in cases if case["repository"] == args.repository]
output = Path(args.output_dir)
Expand All @@ -278,6 +312,8 @@ def run_repository(args):
profile=config["profile"], repository_context={},
endpoint=os.getenv("TYPESAFE_API_URL", API_URL),
model=os.getenv("TYPESAFE_MODEL", config["model"] or MODEL))
decider.root_instruction_context = (instruction_context
if arm == "global_directory" else None)
scorer = BatchScorer(case["goal"], decider, workers=int(config["workers"]),
batch_size=int(config["batch_size"]),
max_candidate_bytes=int(config["max_candidate_bytes"]))
Expand All @@ -290,6 +326,10 @@ def run_repository(args):
workers=int(config["workers"]),
batch_size=int(config["batch_size"]),
max_candidate_bytes=int(config["max_candidate_bytes"])))
active_context = decider.root_instruction_context
result["root_instruction_context_present"] = bool(active_context)
result["root_instruction_context_source"] = (active_context or {}).get("source")
result["root_instruction_context_bytes"] = int((active_context or {}).get("bytes", 0))
selected_files = {item["path"] for item in result["promoted_files"]}
scored_files = set(result["file_scores"])
primary = set(case["primary_files"])
Expand Down Expand Up @@ -464,7 +504,9 @@ def aggregate(args):
"candidate_file_count", "file_decisions", "directory_decisions", "model_visible_nodes",
"candidate_recall", "primary_recall", "missing_primary", "missing_primary_causes",
"target_details", "phase_usage", "physical_usage", "pricing", "phase1_wall_time_ms",
"phase2_enumeration_wall_time_ms", "phase2_scoring_wall_time_ms", "wall_time_ms", "error")
"phase2_enumeration_wall_time_ms", "phase2_scoring_wall_time_ms", "wall_time_ms",
"root_instruction_context_present", "root_instruction_context_source",
"root_instruction_context_bytes", "error")
compact_rows = []
for row in rows:
compact = {key: row[key] for key in summary_fields if key in row}
Expand Down
25 changes: 24 additions & 1 deletion tests/test_file_discovery_global_directory.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,12 @@

sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))

from file_discovery_global_directory import enumerate_directories, enumerate_direct_files, two_stage_discovery
from file_discovery_global_directory import (
enumerate_directories,
enumerate_direct_files,
load_root_instruction_context,
two_stage_discovery,
)


class FakeScorer:
Expand Down Expand Up @@ -42,6 +47,24 @@ def test_nested_directory_is_visible_even_when_parent_scores_low(self):
self.assertEqual(["pkg/sub/target.py"], list(result["file_scores"]))
self.assertNotIn("pkg/sub/child/hidden.py", result["file_scores"])

def test_root_instruction_context_prefers_claude_and_does_not_merge_agents(self):
with tempfile.TemporaryDirectory() as temp:
root = Path(temp)
(root / "CLAUDE.md").write_text("claude context", encoding="utf-8")
(root / "AGENTS.md").write_text("agents context", encoding="utf-8")
context = load_root_instruction_context(root)
self.assertEqual("CLAUDE.md", context["source"])
self.assertEqual("claude context", context["content"])
self.assertEqual(len(b"claude context"), context["bytes"])

def test_root_instruction_context_falls_back_to_agents(self):
with tempfile.TemporaryDirectory() as temp:
root = Path(temp)
(root / "AGENTS.md").write_text("agents context", encoding="utf-8")
context = load_root_instruction_context(root)
self.assertEqual("AGENTS.md", context["source"])
self.assertEqual("agents context", context["content"])

def test_direct_file_enumeration_deduplicates_selected_directories(self):
with tempfile.TemporaryDirectory() as temp:
root = Path(temp)
Expand Down
Loading