-
Notifications
You must be signed in to change notification settings - Fork 13
Expand file tree
/
Copy pathdocker-compose.yml.example
More file actions
237 lines (230 loc) · 9.01 KB
/
Copy pathdocker-compose.yml.example
File metadata and controls
237 lines (230 loc) · 9.01 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
services:
mt5:
image: dockurr/windows:5.14
environment:
RAM_SIZE: "4G"
RAM_CHECK: "N"
CPU_CORES: "2"
DISK_SIZE: "32G"
devices:
- /dev/kvm
cap_add:
- NET_ADMIN
ports:
- "${NOVNC_PORT:-8006}:8006"
volumes:
- ./data/storage:/storage
- ./data/oem:/oem
- ./data/shared:/shared
- ./assets:/shared/assets:ro
- ./data/win.iso:/boot.iso
deploy:
resources:
limits:
memory: 512M
memswap_limit: 5G
healthcheck:
test: ["CMD", "sh", "/shared/scripts/healthcheck.sh"]
interval: 30s
timeout: 30s
retries: 10
start_period: 120s
restart: unless-stopped
stop_grace_period: 2m
# Wickworks TA sidecar — shares mt5's net namespace so it's reachable
# ONLY from the mt5 container (and from the Windows VM via the dockurr
# gateway 20.20.20.1:8000). No ports published; nothing else on the
# docker network can talk to it.
#
# netns-shared sidecars are NOT re-joined when the owning VM container is
# recreated: Docker leaves wickworks running in the old, now-orphaned
# network namespace, where its own loopback healthcheck still passes while
# the VM can no longer reach it (every TA call then 502s). This healthcheck
# DETECTS the orphan and reports unhealthy. It deliberately does NOT kill
# the process: a healthcheck cannot repair its own immutable
# NetworkMode=container:<old-owner-id> binding, and killing the sidecar
# makes compose restart it into a netns that no longer exists.
# Recovery is a Compose-level operation: recreate the VM together with its
# sidecar so it gets a fresh NetworkMode binding:
# docker compose up -d --force-recreate mt5 wickworks
# (or ./scripts/recreate-vm.sh mt5)
# Restarting the VM alone is NOT a reliable recovery: the owner ID survives
# a restart, but the sidecar's netns does not (verified in the lifecycle
# integration test). See scripts/wickworks-healthcheck.py.
wickworks:
image: psyb0t/wickworks:v0.7.0@sha256:2973055356e8879a4a9a4025422d5835e18235eece898c780fa61ed563c43590
restart: unless-stopped
network_mode: "service:mt5"
environment:
LOG_LEVEL: INFO
MAX_BARS: "5000"
MIN_BARS: "50"
volumes:
- ./scripts/wickworks-healthcheck.py:/wickworks-healthcheck.py:ro
healthcheck:
test: ["CMD", "python", "/wickworks-healthcheck.py"]
interval: 15s
# Must exceed the script's worst case (self-health probe + concurrent
# gateway sweep, each up to 2s), or Docker aborts the orphan-detection
# check before it can report.
timeout: 12s
start_period: 30s
retries: 3
depends_on:
- mt5
# Daily log rotator. Rotates data/shared/logs/*.log to *.log.YYYYMMDD
# at the day boundary and prunes those archives plus dated MT5 terminal,
# Tester, and Tester Agent journals older than RETAIN_DAYS.
# Truncate-in-place so the Python API's open log handles keep working
# without reopening. Hourly check, idempotent (keyed on yesterday's
# archive existing).
log-rotator:
image: alpine:3.20
restart: unless-stopped
environment:
LOG_DIR: /logs
TERMINALS_DIR: /terminals
RETAIN_DAYS: "7"
MAX_LOG_BYTES: "2147483648"
IDLE_MINUTES: "30"
INTERVAL: "3600"
volumes:
- ./data/shared/logs:/logs
- ./data/shared/terminals:/terminals
- ./scripts/rotate-logs.sh:/rotate.sh:ro
command: ["sh", "/rotate.sh"]
# VM crash watchdog. dockurr/windows keeps the container up while the
# Windows guest may have crashed internally, so restart: unless-stopped
# never fires and every terminal API in that VM stays dead. This sidecar
# polls Docker health through the socket and restarts a VM only after its
# health has stayed unhealthy for a sustained FailingStreak, with
# exponential backoff and bounded retries; state lives on a named volume.
# Runs inside the compose project (docker compose up -d), no host cron.
# See scripts/vm-watchdog.py.
vm-watchdog:
# Needs the docker CLI + compose plugin to run scripts/recreate-vm.sh, so
# it is built rather than pulled. The base image is digest-pinned inside
# the Dockerfile (this container mounts the Docker socket).
build:
context: .
dockerfile: Dockerfile.watchdog
restart: unless-stopped
command: ["python", "-u", "/vm-watchdog.py"]
volumes:
- /var/run/docker.sock:/var/run/docker.sock
- ./scripts/vm-watchdog.py:/vm-watchdog.py:ro
- vm-watchdog-state:/state
# The project itself, at the SAME absolute path the host uses. Compose
# resolves the relative bind mounts in this file client-side, so a
# different path in here would rewrite every mount to somewhere that
# does not exist on the host. run.sh exports MT5_PROJECT_DIR; the
# watchdog refuses to act (and says so at startup) if it is unset.
- ${MT5_PROJECT_DIR:?MT5_PROJECT_DIR must be the absolute host path of this project (run.sh exports it; otherwise set it in .env)}:${MT5_PROJECT_DIR}:ro
environment:
WATCHDOG_STATE_DIR: /state
WATCHDOG_PROJECT_DIR: ${MT5_PROJECT_DIR}
WATCHDOG_RECREATE_SCRIPT: ${MT5_PROJECT_DIR}/scripts/recreate-vm.sh
# The helper the watchdog runs calls `docker compose`, which interpolates
# ${MT5_PROJECT_DIR:?} in THIS file again. The watchdog sets it in the
# helper's environment itself; it is handed through here as well so a
# human running `docker compose exec vm-watchdog …/recreate-vm.sh` gets
# the same environment the watchdog uses.
MT5_PROJECT_DIR: ${MT5_PROJECT_DIR}
logging:
driver: json-file
options:
max-size: "10m"
max-file: "3"
mcpunifier:
build:
context: .
dockerfile: Dockerfile.mcpunifier
restart: unless-stopped
environment:
MT5_HOST: mt5
LOG_LEVEL: ${MCP_LOG_LEVEL:-info}
volumes:
# Read-only: this service only ever reads the terminal table.
- ./config/config.yaml:/app/config/config.yaml:ro
security_opt:
- "no-new-privileges:true"
cap_drop:
- ALL
read_only: true
tmpfs:
- /tmp:rw,noexec,nosuid,size=16m
- /var/log/mcpunifier:rw,noexec,nosuid,size=64m
init: true
deploy:
resources:
limits:
memory: 256M
cpus: "0.5"
pids: 128
logging:
driver: json-file
options:
max-size: "10m"
max-file: "5"
healthcheck:
test: ["CMD", "python", "-c", "import urllib.request,sys; sys.exit(0 if urllib.request.urlopen('http://127.0.0.1:6600/health', timeout=3).status == 200 else 1)"]
interval: 30s
timeout: 5s
retries: 3
start_period: 15s
depends_on:
- mt5
nginx:
image: nginx:1.30.0-alpine3.23
restart: unless-stopped
ports:
- "127.0.0.1:${API_HOST_PORT:-8888}:80"
volumes:
- ./.data/nginx/nginx.conf:/etc/nginx/nginx.conf:ro
depends_on:
- mt5
- mcpunifier
# ── Cloudflare Tunnel (uncomment to expose API publicly) ──────────
# Point your cloudflared ingress to http://nginx:80 (single backend,
# nginx routes per-terminal paths). Drop creds + config in
# ./.data/cloudflared/.
# cloudflared:
# image: cloudflare/cloudflared:2026.3.0
# restart: unless-stopped
# command: tunnel --config /etc/cloudflared/config.yml run
# volumes:
# - ./.data/cloudflared/config.yml:/etc/cloudflared/config.yml:ro
# - ./.data/cloudflared/creds.json:/etc/cloudflared/creds.json:ro
# depends_on:
# - nginx
# ── Tailscale (uncomment to expose API over tailnet HTTP) ─────────
# Set tailscale.auth_key in config/config.yaml; for Headscale, also set
# tailscale.login_server. run.sh reads both and writes .env + wires
# tailscale serve via the CLI inside the sidecar. URL scheme:
# http://mt5-httpapi/<broker>/<account>/...
# Plain HTTP — bare MagicDNS hostnames don't have matching certs, and
# the wireguard layer already encrypts everything inside the tailnet.
# The sidecar runs in its OWN netns (bridge mode, not host) so it gets
# its own tailnet identity — host's tailscale (if any) stays clean and
# ACLs scope to the container's node only.
# tailscale:
# image: tailscale/tailscale:v1.96.5
# restart: unless-stopped
# environment:
# - TS_AUTHKEY=${TS_AUTHKEY:-}
# - TS_HOSTNAME=${TS_HOSTNAME:-mt5-httpapi}
# - TS_STATE_DIR=/var/lib/tailscale
# - TS_USERSPACE=false # real tailscale0 in this netns — outbound to 100.64.x goes via SIDECAR's identity, not the host's. Requires /dev/net/tun + NET_ADMIN + NET_RAW (all set below).
# - TS_EXTRA_ARGS=${TS_EXTRA_ARGS:---accept-dns=false}
# volumes:
# - ./.data/tailscale/state:/var/lib/tailscale
# - /dev/net/tun:/dev/net/tun
# cap_add:
# - NET_ADMIN
# - NET_RAW
# depends_on:
# - nginx
volumes:
# Persistent per-container watchdog state (last restart, attempts,
# healthy-since) so backoff survives the watchdog's own restarts.
vm-watchdog-state: