Repository navigation
Expand file tree
/
Copy pathbuild.sh
More file actions
executable file
·387 lines (343 loc) · 16.2 KB
/
Copy pathbuild.sh
File metadata and controls
executable file
·387 lines (343 loc) · 16.2 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
#!/usr/bin/env bash
set -e
set -o pipefail
root="$PWD"
# The CDN serves the S3 bucket; narinfo/NAR paths have no cache-name prefix.
ICEDOS_CACHE_URL="${ICEDOS_SUBSTITUTER:-https://icedos.fyi}"
# secret-key-files is restricted, so the untrusted CI user's copies land unsigned
# and clients reject them. The store's secret-key param is not, and signs on write.
# Single-threaded xz (the file:// default) pushed a cuda closure past the 30m staging
# timeout on the 4-vCPU runner. zstd is several times faster; narinfos record it per path.
ICEDOS_STORE_PARAMS="?compression=zstd¶llel-compression=true"
if [ -n "${ICEDOS_SIGNING_KEY:-}" ]; then
printf '%s\n' "$ICEDOS_SIGNING_KEY" > "$root/nix-secret.pem"
chmod 600 "$root/nix-secret.pem"
export NIX_CONFIG="secret-key-files = $root/nix-secret.pem"
ICEDOS_STORE_PARAMS="$ICEDOS_STORE_PARAMS&secret-key=$root/nix-secret.pem"
fi
ICEDOS_KEY_NAME="${ICEDOS_KEY_NAME:-$(cut -d: -f1 <"$root/nix-public.pem" 2>/dev/null || true)}"
# Returns 0 if the path is cached AND signed by us: an unsigned narinfo is
# unusable to clients, and counting it as a hit would skip the config forever.
is_path_cached() {
local store_path="$1"
local hash="${store_path#/nix/store/}"
hash="${hash:0:32}"
local body
body=$(curl -sf --max-time 30 "$ICEDOS_CACHE_URL/$hash.narinfo") || return 1
[ -n "$body" ] || return 1
[ -n "${ICEDOS_KEY_NAME:-}" ] || return 0
printf '%s\n' "$body" | grep -q "^Sig:.*[ :]$ICEDOS_KEY_NAME:"
}
[ -d build ] && rm -rf build
mkdir -p build/status build/seedpool
# Health gate: a broken origin makes nix treat CI-built paths as unavailable and
# recompile the world. Abort; the next cycle retries. 403 = missing via CloudFront+OAC.
probe_code=$(curl -s -o /dev/null -w '%{http_code}' --max-time 30 \
"$ICEDOS_CACHE_URL/00000000000000000000000000000000.narinfo") || probe_code=000
case "$probe_code" in
200|403|404) ;;
*) echo "::error::cache server unhealthy (HTTP $probe_code) — aborting build so the next cycle retries against a healthy server" >&2; exit 1 ;;
esac
# The work dir is baked into the closure via `icedos.configurationLocation`, so a mktemp base
# gives every run fresh hashes for ~38 paths per config. CI pins ICEDOS_WORKBASE instead.
workbase="${ICEDOS_WORKBASE:-}"
if [ -n "$workbase" ]; then
# Clear the CONTENTS, never the directory: CI creates it under root-owned /mnt, so the
# runner can write inside it but `rm -rf "$workbase"` fails with EACCES.
clean_workbase() { find "$workbase" -mindepth 1 -maxdepth 1 -exec rm -rf {} + 2>/dev/null || true; }
mkdir -p "$workbase"
clean_workbase
else
workbase="$(mktemp -d -t icedos-cache-XXXXXX)"
clean_workbase() { rm -rf "$workbase"; }
fi
trap clean_workbase EXIT
max_parallel="${ICEDOS_MAX_PARALLEL:-6}"
# ICEDOS_BUILD_CONFIGS (external mode): space-separated basenames of the configs
# affected by the source repo, derived by nix-build.yml. The base warm-up always
# runs: it seeds BASE_LOCK with the pinned rev and realizes the shared closure.
# force overrides the subset; without it everything outside the subset is
# untouched and stays covered by the cache.
selected=()
if [ -n "${ICEDOS_BUILD_CONFIGS:-}" ] && [ -z "${ICEDOS_FORCE_BUILD:-}" ]; then
for name in $ICEDOS_BUILD_CONFIGS; do
[ -f "config/$name" ] || { echo "ICEDOS_BUILD_CONFIGS: no such config: $name" >&2; exit 1; }
selected+=("$name")
done
else
for cfg in config/*.toml; do
selected+=("$(basename "$cfg")")
done
fi
in_selected() {
local name="$1" s
for s in "${selected[@]}"; do [ "$s" = "$name" ] && return 0; done
return 1
}
# Apply the unpin list to the seed: nodes this run is meant to advance are
# removed so they resolve fresh (external: the pinned source repo; internal:
# nixpkgs/home-manager on the nixpkgs PR, the PR's leaf otherwise). Everything
# else keeps the rev the cache was last built with.
apply_unpin() {
local seed="$1"
[ -f "$seed" ] && [ -n "${ICEDOS_UNPIN:-}" ] || return 0
python3 - "$seed" $ICEDOS_UNPIN <<'PYEOF'
import json, sys
path, patterns = sys.argv[1], sys.argv[2:]
lock = json.load(open(path))
nodes = lock.get("nodes", {})
# Same key lookup as cache-server's tracked-revs.py: exact match, else a
# "-<name>" suffixed node key. A root input also resolves through its reference:
# nix keys a clashing node `nixpkgs_2`, which neither match would catch.
root_inputs = nodes.get("root", {}).get("inputs", {})
targets = set()
for name in patterns:
if name in nodes:
targets.add(name)
targets.update(k for k in nodes if k.endswith("-" + name))
ref = root_inputs.get(name)
if isinstance(ref, str):
targets.add(ref)
for k in targets:
nodes.pop(k, None)
def strip_refs(inputs):
if not isinstance(inputs, dict):
return
for name, ref in list(inputs.items()):
if isinstance(ref, str) and ref in targets:
del inputs[name]
for node in nodes.values():
strip_refs(node.get("inputs"))
strip_refs(nodes.get("root", {}).get("inputs"))
json.dump(lock, open(path, "w"), indent=2)
print(f"unpinned {len(targets)} node(s) from the seed: {sorted(targets)}")
PYEOF
}
# Build one config in an isolated, git-less work dir so the flake eval sees the untracked
# config.toml. Pushes go behind a flock: the 1-core server only chunks one closure at a time.
build_and_push() {
local cfg="$1"
local name work out result top_path stage pushed attempt push_rc push_out eval_err
name="$(basename "$cfg" .toml)"
work="$workbase/$name"
out="$work/out"
(
set -e
# Isolated copy so parallel builds never race on config.toml or the flake state.
rsync -a --exclude=build --exclude=.git "$root/" "$work/"
cp "$cfg" "$work/config.toml"
# Seed per config: the exact lock this config was last built with (the root
# state.lock only covers whichever config published last). Placed before
# the base-lock seed below: the warm-up's evolved lock wins for the rest.
seed="$ICEDOS_SEED_DIR/$name.lock"
[ -f "$seed" ] || seed="${ICEDOS_SEED_LOCK:-}"
if [ -n "$seed" ] && [ -f "$seed" ]; then
mkdir -p "$work/build/.state"
cp "$seed" "$work/build/.state/flake.lock"
apply_unpin "$work/build/.state/flake.lock"
fi
# Seed the base build's resolved lock so this build only resolves its OWN repos; nix
# adds the missing ones in-memory (--no-update-lock-file doesn't block additions).
if [ -n "${BASE_LOCK:-}" ] && [ -f "${BASE_LOCK:-}" ] && [ "$cfg" != "$base" ]; then
mkdir -p "$work/build/.state"
cp "$BASE_LOCK" "$work/build/.state/flake.lock"
# Re-apply the unpin AFTER the base-lock copy: the base lock carries every
# node already resolved (the source repo at main, not the PR head), and
# --no-update-lock-file keeps existing nodes as-is. Popping the source repo
# here forces a re-resolve from the config URL, which staging pinned to the
# PR head. Without this, external gates build main and merge unvalidated PRs.
apply_unpin "$work/build/.state/flake.lock"
fi
# Heal mode: rebuild from each config's own last-resolved lock so the copied
# closure matches what that config actually serves its users.
if [ -n "${ICEDOS_HEAL:-}" ] && [ -f "$root/build/locks/$name.lock" ]; then
mkdir -p "$work/build/.state"
cp "$root/build/locks/$name.lock" "$work/build/.state/flake.lock"
fi
mkdir -p "$out"
cd "$work"
# Skip when the top-level closure is already cached; genflake runs first for the pure outPath eval.
# pipe-operators is REQUIRED — core/lib/icedos.nix uses `|>`; stderr kept so failed evals aren't silent.
top_path=""
if [ -z "${ICEDOS_HEAL:-}" ] && [ -n "${ICEDOS_CACHE_URL:-}" ] && [ -z "${ICEDOS_FORCE_BUILD:-}" ]; then
# Evals are silent — a stalled fetch here hangs the whole fan-out, so bound them.
if timeout 15m env TMPDIR="$out" nix run path:.#icedos -- --genflake-only; then
# Write the resolved lock: the skip check eval reads it, and CI pins the built
# input hashes from build/locks (gitignored).
timeout 10m nix --extra-experimental-features "nix-command flakes" flake lock "$work/build/.state"
mkdir -p "$root/build/locks"
cp "$work/build/.state/flake.lock" "$root/build/locks/$name.lock"
eval_err="$root/build/$name.eval.err"
top_path=$(timeout 10m nix eval --raw --no-write-lock-file \
--extra-experimental-features "nix-command flakes pipe-operators" \
"path:$work/build/.state#nixosConfigurations.icedos.config.system.build.toplevel.outPath" \
2>"$eval_err") || {
top_path=""
echo "$cfg: could not evaluate the top-level closure, building unconditionally:" >&2
cat "$eval_err" >&2
}
fi
if [ -n "$top_path" ] && is_path_cached "$top_path"; then
echo "$cfg: top-level closure already in cache ($top_path), skipping build"
echo ok >"$root/build/status/$name"
return 0
fi
fi
echo "building $cfg..."
TMPDIR="$out" nix run path:.#icedos -- --build \
--nh-args --no-nom \
--build-args \
-L \
--extra-substituters "$ICEDOS_CACHE_URL?priority=100" \
--extra-trusted-public-keys "$(cat nix-public.pem)" \
--extra-substituters "https://attic.xuyh0120.win/lantian?priority=90" \
--extra-trusted-public-keys "lantian:EeAUQ+W+6r7EtwnmYjeVwx5kOGEBpjlBfPlzGlTNvHc="
shopt -s nullglob
local results=("$out"/*/result)
shopt -u nullglob
[ "${#results[@]}" -eq 1 ] || {
echo "expected 1 result under $out, found ${#results[@]}" >&2
exit 1
}
result="$(readlink "${results[0]}")"
# Refresh the persisted lock: the build may have resolved newer revs.
timeout 10m nix --extra-experimental-features "nix-command flakes" flake lock "$work/build/.state"
mkdir -p "$root/build/locks"
cp "$work/build/.state/flake.lock" "$root/build/locks/$name.lock"
echo "pushing $cfg..."
# `nix copy` always writes whole closures, so filtering its path list changes
# nothing. Seed a staging cache with narinfos for every path already served
# (ours, signed; or upstream, which users prefer at priority 40 < 100) and nix
# treats them as present, writing only the genuinely new paths.
stage="$out/push-cache"
mkdir -p "$stage"
mapfile -t closure_paths < <(nix path-info -r "$result")
printf '%s\n' "${closure_paths[@]}" | \
ICEDOS_CACHE_URL="$ICEDOS_CACHE_URL" ICEDOS_KEY_NAME="${ICEDOS_KEY_NAME:-}" stage="$stage" pool="$root/build/seedpool" \
xargs -P 8 -I{} bash -c '
h=$(basename "{}" | cut -c1-32)
# The pool holds what earlier configs in this run pushed; the CDN can still
# be serving a stale miss for those.
cp "$pool/$h.narinfo" "$stage/$h.narinfo" 2>/dev/null && exit 0
body=$(curl -sf --max-time 30 "$ICEDOS_CACHE_URL/$h.narinfo" || true)
if [ -n "$body" ] && printf "%s\n" "$body" | grep -q "^Sig:.*[ :]$ICEDOS_KEY_NAME:"; then
printf "%s\n" "$body" >"$stage/$h.narinfo"
exit 0
fi
code=000
for i in 1 2 3; do
code=$(curl -s -o "$stage/$h.tmp" -w "%{http_code}" --max-time 20 "https://cache.nixos.org/$h.narinfo")
[ "$code" = 000 ] || break
sleep $((i * 2))
done
# a failed probe is not proof of absence; pushing beats permanent skip
if [ "$code" = 200 ]; then
mv "$stage/$h.tmp" "$stage/$h.narinfo"
else
rm -f "$stage/$h.tmp"
printf "%s\n" "{}"
fi
' >"$out/missing-paths" || true
mapfile -t missing_paths < "$out/missing-paths"
find "$stage" -maxdepth 1 -name '*.narinfo' -printf '%f\n' >"$out/seeded"
echo "$cfg: $(wc -l <"$out/seeded") paths already served, pushing ${#missing_paths[@]}"
# nix copy fails the whole invocation on one bad path, which would leave the
# closure half-uploaded. Retry before giving up.
pushed=0
for attempt in 1 2 3; do
push_rc=0
push_out="$(timeout 30m nix copy --to "file://$stage$ICEDOS_STORE_PARAMS" "$result" 2>&1)" || push_rc=$?
printf '%s\n' "$push_out"
[ "$push_rc" -eq 0 ] && { pushed=1; break; }
echo "$cfg: staging attempt $attempt failed (rc=$push_rc), retrying" >&2
sleep $((attempt * 15))
done
[ "$pushed" -eq 1 ] || {
echo "$cfg: staging failed after 3 attempts; the cached closure for $result is incomplete" >&2
exit 1
}
# Seeds describe objects the cache already holds; only what nix wrote is new.
while read -r f; do rm -f "$stage/$f"; done <"$out/seeded"
rm -f "$stage/nix-cache-info"
# An unsigned narinfo is unusable to clients and would be skipped forever
# after. Never upload one; a dropped key param is a build failure, not a warning.
if unsigned=$(grep -L "^Sig:.*[ :]${ICEDOS_KEY_NAME}:" "$stage"/*.narinfo 2>/dev/null) \
&& [ -n "$unsigned" ]; then
echo "$cfg: $(printf "%s\\n" "$unsigned" | wc -l) narinfos would upload unsigned; refusing" >&2
exit 1
fi
pushed=0
for attempt in 1 2 3; do
# NARs before narinfos: an interrupted upload then leaves an orphan NAR,
# not a narinfo clients resolve to a 404.
(
flock 9
{ [ ! -d "$stage/nar" ] || timeout 15m aws s3 sync "$stage/nar" "s3://$ICEDOS_S3_BUCKET/nar/" \
--region "$AWS_REGION" --only-show-errors; } &&
timeout 15m aws s3 sync "$stage" "s3://$ICEDOS_S3_BUCKET/" --region "$AWS_REGION" \
--only-show-errors --exclude '*' --include '*.narinfo'
) 9>"$root/build/push.lock" && {
pushed=1
break
}
echo "$cfg: upload attempt $attempt failed, retrying" >&2
sleep $((attempt * 15))
done
[ "$pushed" -eq 1 ] || {
echo "$cfg: upload failed after 3 attempts; the cached closure for $result is incomplete" >&2
exit 1
}
# Feed the pool so later configs seed these without waiting on the CDN.
cp "$stage"/*.narinfo "$root/build/seedpool/" 2>/dev/null || true
# nix never writes nix-cache-info to S3; nix drops the substituter without it
printf 'StoreDir: /nix/store\nWantMassQuery: 1\nPriority: 100\n' \
| aws s3 cp - "s3://$ICEDOS_S3_BUCKET/nix-cache-info" --region "$AWS_REGION" >/dev/null 2>&1 || \
echo "warning: failed to upload nix-cache-info" >&2
# Persist the config's resolved lock so the weekly heal job can restore
# exactly this closure if the lifecycle rule expires any of its paths.
aws s3 cp "$root/build/locks/$name.lock" "s3://$ICEDOS_S3_BUCKET/locks/$name.lock" --region "$AWS_REGION" >/dev/null 2>&1 || \
echo "$cfg: warning — could not upload the config lock" >&2
echo "$cfg successfully built and uploaded to the cache server!"
) && echo ok >"$root/build/status/$name" || echo fail >"$root/build/status/$name"
}
# Build the bare base alone first so its shared closure is realized and pushed once; the
# parallel builds then only handle their own deltas instead of racing on a cold store.
base="config/00-base.toml"
BASE_LOCK=""
if [ -f "$base" ]; then
echo "=== warming shared base: $base ==="
build_and_push "$base"
if [ "$(cat "$root/build/status/00-base" 2>/dev/null)" = "ok" ]; then
BASE_LOCK="$workbase/00-base/build/.state/flake.lock"
else
echo "WARNING: base warm-up failed; parallel builds will each resolve their own inputs" >&2
fi
fi
# Fan out the remaining configs in parallel against the now-warm store, throttled.
# Subset mode skips configs outside the selection (including the base itself,
# which the warm-up above already handled).
for cfg in config/*.toml; do
[ "$cfg" = "$base" ] && continue
in_selected "$(basename "$cfg")" || continue
while [ "$(jobs -r | wc -l)" -ge "$max_parallel" ]; do wait -n || true; done
build_and_push "$cfg" &
done
wait
failed=()
for cfg in config/*.toml; do
[ "$cfg" = "$base" ] && continue
in_selected "$(basename "$cfg")" || continue
name="$(basename "$cfg" .toml)"
[ "$(cat "$root/build/status/$name" 2>/dev/null)" = "ok" ] || failed+=("$cfg")
done
if [ "${#failed[@]}" -gt 0 ]; then
echo "Build/upload failed for ${#failed[@]} config(s):" >/dev/stderr
printf ' %s\n' "${failed[@]}" >/dev/stderr
exit 1
fi
# Expose the built state lock for the publish step: it becomes the next run's
# seed, so the cache branch always describes what the cache was built with.
if [ -n "${BASE_LOCK:-}" ] && [ -f "${BASE_LOCK:-}" ]; then
cp "$BASE_LOCK" "$root/state.lock"
fi
echo "All configs built successfully!"