-
-
Notifications
You must be signed in to change notification settings - Fork 150
Expand file tree
/
Copy pathconfig.example.yaml
More file actions
374 lines (345 loc) · 16.4 KB
/
Copy pathconfig.example.yaml
File metadata and controls
374 lines (345 loc) · 16.4 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
# OpenSwarm Configuration
# Environment variables can be used in ${VAR_NAME} or ${VAR_NAME:-default} format
# Copy this file to config.yaml to use
# Default CLI adapter for worker/reviewer stages
# Options: codex, codex-responses, cc-router, cursor, gpt, local, lmstudio, ollama-cloud, openrouter, atlascloud, claude
# For ChatGPT OAuth: run `openswarm auth login --provider gpt`
# `codex-responses` runs OpenSwarm's native tool loop with the same login.
# For OpenRouter: run `openswarm auth login --provider openrouter`
# (PKCE browser flow → stores a sk-or-* API key; falls back to manual paste)
# For Atlas Cloud: set ATLASCLOUD_API_KEY
# For local: start Ollama, LMStudio, or llama.cpp server
# For lmstudio: start LM Studio Local Server (default http://localhost:1234)
# Optional env: LMSTUDIO_BASE_URL, LMSTUDIO_MODEL, LMSTUDIO_API_KEY
# If LMSTUDIO_MODEL is unset, the adapter auto-selects the first loaded model.
# web_search: set OPENSWARM_SEARXNG_URL (vega-search, e.g. http://searxng:8080)
# or VEGA_SEARXNG_URL. Optional OPENSWARM_SEARXNG_KEY / VEGA_SEARXNG_KEY for
# the public X-VEGA-Key header. TAVILY_KEY / BRAVE_SEARCH_KEY still work.
adapter: codex-responses
# reviewAdapter: openrouter # reviewer for `openswarm review` AND `review --max`; omit to follow `adapter` (AGT-4292/4299)
discord:
token: ${DISCORD_TOKEN}
channelId: ${DISCORD_CHANNEL_ID}
webhookUrl: ${DISCORD_WEBHOOK_URL:-} # Optional
# Optional fail-closed boundary for unattended agents. When enabled, all
# external human collaboration surfaces are read-only. Delegated CLIs and
# diagnostics stay disabled. Native-loop bash is exposed only after the
# separate network-none companion attests its Unix socket, bwrap mount/PID
# namespaces, exact /work roots and limits. Missing/mismatched proof leaves it
# disabled; typed DevOps/data MCP writes keep their existing grants.
humanSurfaceReadOnly:
enabled: false
sandboxExecutor:
enabled: false
socketPath: /run/openswarm-sandbox/executor.sock
allowedRoots: [/work]
connectTimeoutMs: 1000
maxRequestBytes: 65536
maxOutputBytes: 524288
maxTimeoutMs: 900000
maxConcurrent: 8
linear:
apiKey: ${LINEAR_API_KEY}
teamId: ${LINEAR_TEAM_ID}
github:
repos:
- owner/repo # owner/repo format
checkInterval: 300000 # 5 minutes (ms)
# Agent list
agents:
- name: main
projectPath: ~/dev/your-project
heartbeatInterval: 1800000 # 30 minutes (ms)
linearLabel: main # Label for Linear issue filtering
enabled: true
paused: false
# Default heartbeat interval (ms)
defaultHeartbeatInterval: 1800000 # 30 minutes
# Autonomous execution mode settings
autonomous:
enabled: true
pairMode: true
schedule: "*/15 * * * *" # Every 15 minutes
maxAttempts: 3
maxConcurrentTasks: 64 # Operator-controlled global capacity (maximum 256)
# In Progress is a live ownership signal, not a parking lot. If neither the
# scheduler nor a durable lease owns an issue for this long, return it to Backlog.
stalledInProgressHours: 6
worktreeMode: true # Required for same-project parallel tasks
allowSameProjectConcurrent: true
# What durable admission does with a task whose write scope could not be
# resolved while another run in the same repository is live. admit (default)
# relies on isolated worktrees; serialize is the Codex-era fail-closed hold.
# unknownScopeAdmission: admit
# An infrastructure failure is not the task's fault and is never counted
# toward STUCK — but the same infrastructure failure on this many consecutive
# attempts is not a blip. Park the run for the operator with the cause named
# instead of backing off forever (`openswarm work` redispatches it). 0 disables.
# infraFailureCircuit: 6
# maxConcurrentPerProject: 64 # Optional hard cap; omit for weighted, work-conserving project fairness
automationLedgerMode: primary # off | shadow | primary (durable fail-closed execution truth)
automationLeaseMs: 600000 # Renewed every ~3m; stale callbacks are fenced
shutdownGraceMs: 30000 # Wait for real executor exit before service teardown
# Durable agent board + human/operator coordination. Use one project-scoped
# issue whose comments form the append-only Linear board.
coordinationBoardIssueId: AGT-3993
# `primary` must name the same adapter this file selects above (adapter:), or
# routing silently disables — the worker only consults a policy whose primary
# is the adapter it is actually running.
# Only typed failures may route to CC-Router or an authenticated Cursor CLI;
# a failing test or a reviewer verdict never does.
# quota — provider usage limit / 429 on the primary
# infra — spawn, auth, or timeout failure before the agent ran
# capability — the primary is not installed or not serving on this machine
adapterRouting:
primary: codex-responses
fallbacks: [cc-router, cursor]
allowReasons: [quota, infra, capability]
# Exact role-scoped MCP grants. Reads are allowed from listed servers; writes
# and destructive operations require exact tool names.
mcpPolicies:
orchestrator:
servers: [github, linear, cloudflare]
writeTools: [linear__save_comment]
worker:
servers: [github, linear]
writeTools: [linear__save_comment]
reviewer:
servers: [github, linear]
# Periodic review agents are read-only and leased per repository/profile.
periodicReviews:
- { profile: permissions, schedule: "17 */6 * * *" }
- { profile: hygiene, schedule: "43 */6 * * *" }
- { profile: security, schedule: "11 2 * * *", adapter: codex }
# High-capability project supervisor. Actionable board requests trigger a
# debounced sweep immediately; the cron is a reconciliation fallback. It runs
# in a scratch directory with shell access withheld, so use a native-loop
# adapter (delegated codex/claude/cursor CLIs are rejected).
orchestrator:
enabled: true
schedule: "17 */2 * * *"
eventDriven: true
eventDebounceMs: 1000
adapter: codex-responses
model: gpt-5.6-sol
reasoningEffort: high
timeoutMs: 600000
maxTurns: 12
allowedProjects:
- ~/dev/your-project
# Task decomposition settings (Planner Agent)
decomposition:
enabled: true # Enable decomposition
thresholdMinutes: 30 # Decompose if estimated time exceeds this
# maxChildrenPerTask: structural cap on any one parent — shared by the
# autonomous runner and human `/plan` dispatch (AGT-4123 Option 2).
maxChildrenPerTask: 5
# dailyLimit: paces unsupervised automation only. `/plan` enforces the
# child cap above but does not reserve against this budget (AGT-4123).
dailyLimit: 20
# decomposeAfterFailures: a task that failed this many times whole is split
# on its next attempt — even when it resumes a preserved worktree, where the
# size check used to be skipped — and the planner's "fits the budget" answer
# is not honoured (the failures already disproved it). 3 = the attempt
# before STUCK at 4 retries. 0 never forces (AGT-4287).
decomposeAfterFailures: 3
plannerModel: gpt-5.6-sol # High-leverage decomposition stays on Sol
# Backlog grooming (Planner compares fetched open queue states against the repo).
# Safe default: comment recommendations only. Set mode: apply to update
# descriptions and move strongly stale issues to Done.
backlogGrooming:
enabled: false
cadenceHours: 24
mode: comment
plannerModel: gpt-5.6-terra # Bulk queue analysis favors the balanced tier
maxIssues: 80
# Per-role settings
# GPT-5.6: Luna=high-volume/light, Terra=balanced default, Sol=quality gate.
defaultRoles:
worker:
enabled: true
model: gpt-5.6-terra # Default implementation tier
escalateModel: gpt-5.6-sol # Repeated failure earns the frontier tier
escalateAfterIteration: 2
timeoutMs: 1800000 # 30 minutes
# Turn ceiling for the agentic loop. Omitted or 0 = none (AGT-4388): a coding
# run ends on completion, the repeated-tool-call guard, or timeoutMs — a turn
# count is not a property of the task. Set a number only to cap cost hard.
# maxTurns: 0
#
# Declarative tool scoping per role. `allow` can only NARROW the role's
# default tools — it can never grant one the role would not otherwise have,
# so an allow-list naming `bash` on a read-only role stays withheld. `deny`
# is applied after `allow` (deny wins), and supports a trailing `*`
# (`scratch_*`). Use this to keep a stage's blast radius explicit instead
# of relying on the stage's defaults.
# tools:
# allow: [read_file, write_file, edit_file, bash]
# deny: [web_fetch, web_search]
# reasoning effort for this role's native-loop adapter (low|medium|high)
# effort: medium
reviewer:
enabled: true
model: gpt-5.6-sol # Correctness gate; light profile lowers this to Terra
# model / timeoutMs / maxTurns also budget the PR-time fresh review that
# runs against every publication (even with `enabled: false`, when that
# review is the only semantic review). timeoutMs 0 there keeps the CLI's
# diff-scaled default (300s base); a slow model needs more — measured
# 2026-09-17: half the reviews died at 300s. maxTurns 0 = no ceiling.
timeoutMs: 600000
# Second, independently-prompted review of the SAME diff. Its only permitted
# effect is to add findings the reviewer missed and to raise severity — it
# can never soften the reviewer's verdict, and any raised severity without a
# concrete finding is discarded. DISABLED by default: it is a second paid
# call on every review.
#
# Its model must come from a DIFFERENT family than the reviewer's. On the
# reviewer's own model the advisor is a second identical opinion — the same
# weights re-deriving the same blind spots (the trap modelCompat.ts
# documents for `escalate`). The shipped reviewer default is
# deepseek/deepseek-v4-flash; z-ai/glm-5.2 is the family-independent
# alternative measured at 100% detect / 6% false-reject (6s avg) on the
# planted-defect fixtures where the reviewer scored 0% false-reject at 36s.
advisor:
enabled: false
model: z-ai/glm-5.2
timeoutMs: 45000 # 45s single-turn ceiling, as the guard arbiter uses
tester:
enabled: false
model: gpt-5.6-terra # Used only if deterministic verify cannot run
documenter:
enabled: false
model: gpt-5.6-luna
timeoutMs: 120000 # 2 minutes
auditor:
enabled: false
model: gpt-5.6-sol
timeoutMs: 300000 # 5 minutes
skill-documenter:
enabled: false
model: gpt-5.6-luna
timeoutMs: 120000 # 2 minutes
# Model for the draft (task-brief) stage that runs before every worker.
# Unset → the adapter's built-in drafter default (openrouter: qwen3-235b).
# Set it to run the whole fleet on one model.
# draftModel: deepseek/deepseek-v4-flash
# OS fence for the worker's bash tool (AGT-4387). on (default): every bash
# call runs under sandbox-exec (macOS) / bwrap (Linux) with writes limited to
# the worktree, temp dir and package caches; network stays open. off: bash
# runs as the daemon user with only the regex denylist (pre-AGT-4387).
# workerSandbox: on
# Deterministic baseline-diff verification (enabled by default).
# Define repo-specific commands in .openswarm/verify.yaml; see templates/verify.example.yaml.
verify:
enabled: true
blockOnNewFailures: true
maxCommands: 4
# CodeQL security baseline-diff gate. Off by default: a full audit is minutes
# to tens of minutes per language and parked autonomous PRs (AGT-4160).
# Set enabled: true only when CodeQL is installed and latency is acceptable.
# Tool absence still blocks an edit when this is on.
securityAudit:
enabled: false
maxThreads: 2
# Memory target passed to each CodeQL run, in MB. Best-effort by CodeQL's
# own account — a large database can still exceed it.
maxRamMb: 4096
# How many CodeQL runs may execute at once is a PROCESS-wide bound, not a
# per-audit setting, so it lives in the environment rather than here:
# OPENSWARM_CODEQL_MAX_CONCURRENT (default 1, max 8). Raise it only on a
# host with cores to spare — the audit runs twice per task, and letting it
# scale with maxConcurrentTasks is what starved the daemon (AGT-4062).
# Pipeline guards
guards:
qualityGate: false # deprecated whole-tree gate; prefer verify above
fakeDataGuard: true
conventionalCommits: true
branchValidation: true
uncertaintyDetection: true
registryCheck: true
bsDetector: true # blocks on critical code-smell patterns
dependencyAntiPatternCheck: true # block package recreation/version spoofing after import failures (INT-2388 #1)
contractEvidenceCheck: true # block self-referential contract tests without producer/consumer evidence (INT-2388 #2)
verifiedMetricEvidenceCheck: true # block verified evidence deletion / metric changes without distributions (INT-2388 #4)
deadModuleCheck: true # warn on new modules nothing imports/calls (INT-2388 #5)
reformatCheck: true # warn on reformat-only files / oversized diffs (INT-2388 #6)
rewriteCheck: true # block an unacknowledged whole-file rewrite (>30% of a 20+ line file deleted;
# the worker may declare `REWRITE: <path>` in its summary); restores stripped
# trailing newlines in place (AGT-4406)
claimEvidenceCheck: true # block a docs-only change that replaces figures the run never printed;
# flag approval claims without a link/date (AGT-4408)
# Self-repair reflection budget: max objective (lint/bs/test) failures tolerated
# before the loop gives up on bad edits. Lower it to cap token burn when
# reflection stops making progress; the loop also bails early on stagnation
# (identical errors twice in a row). Default: 3.
maxReflections: 3
# Optional repository-owned generator declarations. An output enters a
# binding worker write scope only if the task explicitly names the command.
# projectAgents:
# - projectPath: ~/dev/example
# generatedOutputRules:
# - command: npm run generate:catalog
# outputs: [docs/DATA-CATALOG.md]
# Job profiles override the default Worker/Reviewer pair by task estimate.
jobProfiles:
- name: light
minMinutes: 1
maxMinutes: 29
effort: low
roles:
worker: gpt-5.6-luna
reviewer: gpt-5.6-terra
- name: heavy
minMinutes: 30
effort: high
roles:
worker: gpt-5.6-terra
reviewer: gpt-5.6-sol
# Long-running task monitoring (RunPod training, batch processing, etc.)
#
# checkCommand is a literal argv array — no shell is invoked, so pipes,
# redirects, and substitutions do not work. Put shell logic in a script
# you control and call that script instead (e.g. ["/opt/probe.sh", "arg"]).
#
# monitors:
# - id: runpod-training
# name: "LSTM model training"
# issueId: INT-456
# checkCommand: ["ssh", "runpod", "tail -1 /workspace/train.log"]
# completionCheck:
# type: output-regex
# successPattern: "Training finished"
# failurePattern: "Error|FAILED"
# checkInterval: 2 # Every 2 heartbeats
# maxDurationHours: 24
# notify: true
# PR Auto-Improvement
prProcessor:
enabled: true
schedule: "*/15 * * * *" # Check every 15 minutes
maxIterations: 3 # Max 3 Worker-Reviewer iterations
# Anonymous usage telemetry (opt-out). Sends command name, version, and OS only —
# never code, prompts, paths, or personal data. Disable here, or via
# OPENSWARM_TELEMETRY=0 / DO_NOT_TRACK=1. CI environments are excluded automatically.
telemetry:
enabled: true
# MCP servers available to the daemon. Role policies below decide which tools
# each autonomous role actually sees; no policy means no MCP access.
mcp:
servers:
github:
surface: devops
command: npx
args: ["-y", "@modelcontextprotocol/server-github"]
linear:
surface: devops
preset: linear
cloudflare:
surface: devops
command: npx
args: ["-y", "@cloudflare/mcp-server-cloudflare"]
# Label opaque aliases explicitly. Human-surface tools remain globally
# read-only even when a role lists them in writeTools/destructiveTools.
# company-comms:
# surface: human
# url: https://mcp.example.com/comms