Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
202 changes: 202 additions & 0 deletions .github/scripts/render_schema.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,202 @@
#!/usr/bin/env python3
"""Render the definitions that differ between schemas into their published trees.

Every binary in the field computes its own store URL and cannot be taught a new
one, so the unprefixed path has to keep carrying what those binaries expect.
That is the oldest schema still supported, and it is where a definition lives by
default: services/<name>.yaml is authored, published as it stands, and copied
nowhere.

A definition is authored in the highest schema tree it needs, and every tree
below renders down from it. One needing nothing newer is authored in services/
and published as it stands; one carrying a key that would be wrong to publish to
an older binary is authored in schema/N/services/ instead, and services/ gets the
downgrade rendered from it. A schema tree is sparse, holding only the definitions
authored there plus its own index, since the client tries its bases in order and
a 404 falls straight through to the tree below.

Usage: render_schema.py
"""

import copy
import json
import pathlib
import sys
import yaml

ROOT = pathlib.Path(__file__).resolve().parents[2]
LEGACY = ROOT / "services"
SCHEMA_DIR = ROOT / "schema"

INDEX_FIELDS = ["name", "description", "family", "dashboard", "image", "category",
"icon", "color", "admin_for", "admin_rank", "depends_on",
"versions", "default_version", "env_role"]


def load_yaml(path):
with open(path, "r", encoding="utf-8") as handle:
return yaml.safe_load(handle) or {}


def schemas():
out = []
for path in sorted(SCHEMA_DIR.glob("*.yaml"), key=lambda p: int(p.stem)):
spec = load_yaml(path)
out.append((int(spec["schema"]), spec))
return out


def resolve(doc, path):
node = doc
for part in path.split("."):
if not isinstance(node, dict) or part not in node:
return None
node = node[part]
return node


def drop(doc, path):
parts = path.split(".")
node = doc
for part in parts[:-1]:
if not isinstance(node, dict) or part not in node:
return
node = node[part]
if isinstance(node, dict):
node.pop(parts[-1], None)


def apply_change(doc, change, guard):
"""Render one change backwards, newer schema to older.

A `when` guard reads the document as it entered this schema's step, not the
half-rendered one: a rule commonly depends on a key an earlier rule in the
same step has already dropped.
"""
if "when" in change and not resolve(guard, change["when"]):
return
path, how = change["path"], change["downgrade"]
if how == "drop":
drop(doc, path)
elif isinstance(how, dict) and "join" in how:
value = resolve(doc, path)
if isinstance(value, list):
parts = path.split(".")
node = doc
for part in parts[:-1]:
node = node[part]
node[parts[-1]] = how["join"].join(str(v) for v in value)
elif isinstance(how, dict) and "rename" in how:
value = resolve(doc, path)
if value is not None:
drop(doc, path)
doc[how["rename"]] = value
else:
raise SystemExit(f"unknown downgrade {how!r} for {path}")


def render_to(doc, target, specs):
out = copy.deepcopy(doc)
for version, spec in sorted(specs, reverse=True):
if version <= target:
continue
guard = copy.deepcopy(out)
for change in spec.get("changes", []):
apply_change(out, change, guard)
return out


def dump(doc, path):
with open(path, "w", encoding="utf-8") as handle:
yaml.safe_dump(doc, handle, sort_keys=False, default_flow_style=False, allow_unicode=True)


def index_for(docs):
entries = []
for doc in sorted(docs, key=lambda d: d.get("name", "")):
entries.append({f: doc[f] for f in INDEX_FIELDS
if f in doc and doc[f] not in (None, "", [], {})})
return {"services": entries}


def write_index(path, published, docs, owned):
"""Rewrite only the entries this render owns.

The published index is hand-maintained and does not always match what a
projection of the YAML would produce. Regenerating it wholesale would push
those differences to every install as a change nobody asked for, so entries
for definitions this render does not touch are carried through exactly as
they stand.
"""
entries = index_for(docs)["services"]
fresh = {e["name"]: e for e in entries if e.get("name") in owned}
out, seen = [], set()
if published.exists():
with open(published, "r", encoding="utf-8") as handle:
for entry in json.load(handle).get("services", []):
name = entry.get("name")
out.append(fresh.get(name, entry))
seen.add(name)
for entry in entries:
if entry.get("name") not in seen:
out.append(entry)
with open(path, "w", encoding="utf-8") as handle:
json.dump({"services": out}, handle, indent=2)
handle.write("\n")


def main():
specs = schemas()
if not specs:
print("no schema deltas; nothing to render")
return 0
oldest = 1
newest = max(v for v, _ in specs)

# A definition is authored in the highest schema tree it needs, and every
# tree below renders down from it. One that needs nothing newer is authored
# in services/ and published as it stands.
top = SCHEMA_DIR / str(newest) / "services"
sources = {p.stem: load_yaml(p) for p in sorted(top.glob("*.yaml"))} if top.exists() else {}
plain = {p.stem: load_yaml(p) for p in sorted(LEGACY.glob("*.yaml")) if p.stem not in sources}

inert = []
for name, doc in sorted(sources.items()):
low = render_to(doc, oldest, specs)
dump(low, LEGACY / f"{name}.yaml")
if low == doc:
inert.append(name)

write_index(LEGACY / "index.json", LEGACY / "index.json",
list(plain.values()) + [render_to(d, oldest, specs) for d in sources.values()],
set(sources))
print(f"schema {oldest}: {len(plain)} authored in place, {len(sources)} rendered -> services")

for version, _ in sorted(specs):
if version <= oldest:
continue
out_dir = SCHEMA_DIR / str(version) / "services"
out_dir.mkdir(parents=True, exist_ok=True)
carried = 0
for name, doc in sorted(sources.items()):
# The newest tree is authored, not rendered: leave its bytes alone.
if version < newest:
high = render_to(doc, version, specs)
if high != render_to(doc, oldest, specs):
dump(high, out_dir / f"{name}.yaml")
carried += 1
else:
carried += 1
write_index(out_dir / "index.json", LEGACY / "index.json",
list(plain.values()) + [render_to(d, version, specs) for d in sources.values()],
set(sources))
print(f"schema {version}: {carried} definition(s) -> {out_dir.relative_to(ROOT)}")

if inert:
print(f"note: {', '.join(inert)} render the same in every schema and need no source")
print(f"authored schema is {newest}")
return 0


if __name__ == "__main__":
sys.exit(main())
23 changes: 23 additions & 0 deletions .github/scripts/schema_guard.py
Original file line number Diff line number Diff line change
Expand Up @@ -75,6 +75,26 @@ def profile(root, pattern):
return out


def schema_dropped_paths(root):
"""Paths a schema delta renders away, which are removals by design.

The published legacy tree is the oldest schema, so every key a later schema
added is absent from it on purpose. Without this the guard would read each
of those as a key that disappeared.
"""
dropped = set()
for path in sorted(pathlib.Path(root, "schema").glob("*.yaml")):
try:
with open(path, "r", encoding="utf-8") as handle:
spec = yaml.safe_load(handle) or {}
except (OSError, yaml.YAMLError):
continue
for change in spec.get("changes", []):
if change.get("downgrade") == "drop" and "path" in change:
dropped.add(change["path"])
return dropped


def closed_sets(profiles):
"""Union every file's observed values per path, keeping the small ones."""
merged = {}
Expand All @@ -96,6 +116,7 @@ def main():
published = profile(published_dir, pattern)
candidate = profile(candidate_dir, pattern)
enums = closed_sets(published)
by_design = schema_dropped_paths(candidate_dir)

failures, warnings = [], []

Expand All @@ -110,6 +131,8 @@ def main():

for path, kinds in sorted(old_types.items()):
if path not in new_types:
if path.split("[]")[0] in by_design:
continue
failures.append(f"{rel}: key `{path}` was removed, an older lerd still reads it")
continue
if kinds != new_types[path] and not kinds & new_types[path]:
Expand Down
30 changes: 24 additions & 6 deletions .github/workflows/schema-guard.yml
Original file line number Diff line number Diff line change
Expand Up @@ -3,22 +3,40 @@ name: Schema guard
on:
pull_request:

# A store definition reaches every install within a day, whatever version of
# lerd it runs, so a pull request may only grow the schema. This refuses a key
# that disappeared or changed type, and reports a value no published definition
# has used before.
# A definition is authored in the highest schema tree it needs and rendered down
# into every tree below. Two things have to hold on every pull request: the lower
# trees are what the authored ones render to, and the lowest stays readable by
# the binaries that fetch it.
jobs:
render:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: actions/setup-python@v5
with:
python-version: "3.12"
- run: pip install pyyaml

- name: Render every schema
run: python3 .github/scripts/render_schema.py

- name: The published trees must match the sources
run: |
if ! git diff --quiet; then
echo "The rendered trees are stale. Run .github/scripts/render_schema.py and commit the result."
git diff --stat
exit 1
fi

schema-guard:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
with:
fetch-depth: 0

- uses: actions/setup-python@v5
with:
python-version: "3.12"

- run: pip install pyyaml

- name: Extract the published tree
Expand Down
41 changes: 41 additions & 0 deletions schema/2.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,41 @@
# Schema 2, as understood by lerd 1.36.0 and later.
#
# A schema states only what changed since the one before it, and how to render a
# document back down to that one. A definition is authored in the highest schema
# tree it needs and rendered down into every tree below, the lowest being the
# unprefixed path: that is what every binary up to 1.35.0 computes for itself and
# cannot be taught otherwise.
schema: 2
since_lerd: "1.36.0"

changes:
# Keys schema 1 has no field for. An old binary ignores them, so dropping is
# about publishing a tree that means what it says rather than about safety.
- path: admin_rank
downgrade: drop
- path: dashboard_follows_color_scheme
downgrade: drop
- path: dashboard_login
downgrade: drop
- path: dashboard_proxy_at_path
downgrade: drop
- path: dashboard_proxy_keep_host
downgrade: drop
- path: dashboard_proxy_rebase
downgrade: drop
- path: dashboard_proxy_reroute
downgrade: drop
- path: dashboard_proxy_strip
downgrade: drop
- path: dashboard_scheme_key
downgrade: drop

# This one is not cosmetic. Schema 2 proxies a dashboard at /_svc/<name>/ and
# strips the upstream's X-Frame-Options and frame-ancestors on the way through.
# Schema 1 has no such proxy and opens the dashboard URL directly in an iframe,
# so a UI that refuses framing, which is every upstream needing the strip flag,
# renders a blocked panel. Better to publish no dashboard to schema 1 than a
# card that cannot work: the service still installs, runs and serves its port.
- path: dashboard
downgrade: drop
when: dashboard_proxy_strip
Loading
Loading