Repository navigation
Expand file tree
/
Copy pathdialect_package_codegen.py
More file actions
2211 lines (1995 loc) · 121 KB
/
Copy pathdialect_package_codegen.py
File metadata and controls
2211 lines (1995 loc) · 121 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
#!/usr/bin/env python3
"""dialect_package_codegen.py — build-time codegen for an insight-canon dialect package.
Projects ONE `<dialect>.dialect.yaml` declaration into the complete C++ content of an
`insight-canon` semantic package: the manifest, every row kind, the channel and revision
vocabularies, the derived writer projection, the compile-time fences, and the code-tier
hooks REFERENCED BY NAME. The output is an `.inc` textually included by a thin
hand-written module interface unit, which carries the module name, the imports and the
code-tier declarations and no ruleset content at all.
WHY A SEPARATE ENTRY POINT, and not a mode of `intent_library_codegen.py` (DN-17.D19):
* the grammars are disjoint and must stay CLOSED — one root is `intent:`, the other
`dialect:`, and a union key set would let an intent key land in a dialect file, which
dissolves the closed-grammar guarantee that is the whole safety story;
* the moat rules are opposite — the Intent library's declarations stay PRIVATE and its
fixtures are synthetic by construction, while dialect declarations are PUBLIC and live
in insight-canon;
* THE SORT RULES ARE OPPOSITE, and this is the one that would silently corrupt a
ruleset. The library codegen iterates `sorted(entries)` — correct there, because its
entries are a set discovered from the filesystem and sorting is what makes discovery
order-independent. Here, DECLARED ORDER IS CONTENT: level-lift matching is
first-match-in-declared-order, and `serialize_manifest` walks every row span in
declared order, so a reordering silently changes both what the dialect recognizes and
the composed `semantic_identity`. This tool never sorts a row.
WHAT THIS TOOL CANNOT DO, BY CONSTRUCTION (DN-17.D20). Rows are hashable DATA, never
callables: a declaration only COMBINES the closed core enums, so there is no regex, no
computed prefix, no conditional row, no predicate and no arithmetic here. A dialect
needing a new parse or emit SHAPE is a core grammar-version bump — a new enumerator in
`canon.spi.cppm` — never an algorithm tier in this file. The same rule governs the code
tier: hooks are NAMED, never generated, and a third hook kind is a core SPI change.
THE WRITER PROJECTION IS DERIVED, NEVER DECLARED (DN-17.D15). There is no `emits:` key,
and adding one is a schema error rather than an option: two authorable projections is
exactly the silent divergence the concept exists to forbid. Each declared marker yields
exactly one emit row, in declared order, with `emit = dual(extract)` — and the generator
emits the CALL to core's `dual()`, never a `PayloadEmit` enumerator, so a fifth extractor
added to core cannot be silently mis-derived by a stale table in this file.
A MARKER ROW MAY DECLARE ITS PAYLOAD'S VERSION COORDINATE. The optional mapping
`version_coordinate: {introducer, shape}` on a marker row states where that unit's name
holds its version: the introducer byte sequence, and the payload shape it applies to, a
member of a closed core enum. It is data the core applies; no version syntax is parsed
here. A row that declares none omits the key, keeps its content hash and emits the bytes
it emitted before. The coordinate is a reader-side derivation, so it appears on the
recognition row alone and the derived emit row carries no copy of it.
A DIALECT MAY DECLARE A KEY A STREAM'S ACQUIRER SUPPLIES A VALUE UNDER. A row of the
optional `declared_values:` section names the key and the markers behind which the core
masks a digit run equal to the declared value (DN-133.D1): the dialect owns the vocabulary
(`PR-`), the acquirer owns the value (a run's own pull-request number), and the core owns the
mechanism. No value appears here, ever. A declaration with no such section keeps its content
hash and emits the bytes it emitted before.
A MARKER ROW MAY OPEN A UNIT WITHOUT NAMING IT. The optional `role: Opens` states that the
row's line opens a unit of its kind and carries no identity; the unit is named by a
naming row of the same kind that follows. A row that declares no role NAMES its unit, as
every row did before, so "names" has exactly one spelling: the key's absence, which keeps
the row's content hash and its emitted bytes. An opening row declares no `child_order:`,
no `extract:` and no `version_coordinate:` — it carries no payload, so it has no class,
no instance and no version — and its kind must have a naming row in the same section, or
it could never take effect. When an opener takes effect is the core's algorithm, applied
by the consumer that segments; nothing here decides it. The derived emit row carries the
role, so a writer selecting a unit's banner selects the naming row and never the opener.
A ROLE ROW MAY MATCH THE LINE'S WHOLE SHAPE INSTEAD OF ITS PREFIX (ADR-17.D14). A row declares
exactly one of `prefix:` and `shape:`. A shape is literal bytes with `{n}` holes, each a decimal
number, and the core matches it against the whole content after the transport peel and a
trailing-whitespace trim. It is one closed core kind, not a pattern language: no other hole, no
escape, no literal brace, and no shape whose match would depend on how much a hole takes. The
`Progress` role is declared by a shape alone, because the lines it samples share their prefix
with lines that are content. The emitted row carries the shape in the bytes member and adds
`.match = RoleMatchKind::Shape`; a prefix row emits no match kind, so a declaration of prefix
rows alone keeps its content hash and its emitted bytes.
A MARKER ROW MAY CLOSE THE STEPS OF ITS JOB. The optional `role: Closes` states that the
row's line enters the current job's epilogue, a step-level unit the core supplies and names
(DN-89.D40): the runner's own teardown, which no line names. A closing row is legal only on
`kind: Step`, and like an opening row it declares no `child_order:`, no `extract:` and no
`version_coordinate:`, since the unit it enters carries no class, instance or version of the
dialect's. What the row does to the open unit is the core's algorithm; the derived emit row
carries the role, so a writer never selects it as a banner.
DETERMINISM (DN-17.D19). Strict YAML subset -> canonical content hash -> byte-stable
emission, strings end-to-end with no typed conversion, LF-only output, write-if-changed.
The emitted text is Python output and cannot vary with the C++ compiler; the axes that
CAN move it are the interpreter version, the hash seed, the locale, filesystem
enumeration order and line endings, and `--selftest` asserts on those rather than on the
compiler.
ENCODING (DN-17.D24). Everything that enters the content hash is ASCII and is rejected
otherwise, naming the offending byte. `why:` prose does not enter the hash and is UTF-8,
byte-preserved with no normalisation — the warning glyphs an argument uses to make a
reader stop are load-bearing typography. Four sequences are refused inside a `why:`
scalar, each for a named hazard: a trailing backslash (C++ splices lines in phase 2,
BEFORE comment removal, so a `//` line ending in `\\` would swallow the row beneath it),
U+2028/U+2029, U+FEFF, and the bidirectional overrides (a `why:` exists to be read, the
format is public, and a bidi override makes text render in an order different from its
bytes).
"""
from __future__ import annotations
import argparse
import hashlib
import os
import subprocess
import sys
import tempfile
from pathlib import Path
from codegen_common import (
CODEGEN_COMMON_VERSION,
DeclarationError,
canonical_hash,
cpp_escape,
expect_keys,
fail,
parse_subset_yaml,
read_text,
reject_non_ascii_field,
required_scalar,
)
TOOL_VERSION = "7" # 7: a marker row may close the steps of its job
SCHEMA_VERSION = "1"
DIALECT_FILE_SUFFIX = ".dialect.yaml"
# ── the closed core vocabularies, transcribed as TEXT ────────────────────────
#
# Transcription, not typing: a declaration value is carried to the emitted enumerator as a
# string, so no locale- or stdlib-dependent conversion exists anywhere in the path
# (determinism MUST 3). Growing one of these sets is a core grammar-version bump — a new
# enumerator in `canon.spi.cppm` / `canon.api.cppm` — and this file follows it, never
# leads it.
_STRUCTURAL_ROLES = ("None", "GroupBegin", "GroupEnd", "Terminator", "Progress")
# The role a row may announce ONLY through a shape: a sample line shares its prefix with lines
# that are content (ADR-17.D14), so a prefix row announcing it would take them too.
_SHAPE_ONLY_ROLES = ("Progress",)
# The one hole a shape admits: a decimal number, `\d+(\.\d+)?`, matched by the core.
_SHAPE_HOLE = "{n}"
_MARKER_KINDS = ("None", "Job", "Step")
_CHILD_ORDERS = ("Ordered", "Unordered")
_PAYLOAD_EXTRACTS = ("None", "RemainderAfterPrefix", "RemainderToClosingParen",
"NumericFieldThenRemainder")
# The payload shapes a marker row's version coordinate may name. A row that declares no
# coordinate omits the key: the core's `None` shape is an ABSENCE here, never an authorable
# value, so "declared none" has exactly one spelling.
_VERSION_PAYLOAD_SHAPES = ("OneToken",)
# The roles a marker row may declare. A row that declares none NAMES its unit: the core's
# `Names` role is an ABSENCE here, never an authorable value, so "names" has exactly one
# spelling.
_MARKER_ROLES = ("Opens", "Closes")
_ROLE_ROW_KEYS = ("prefix", "kind", "role", "dialect_gate", "channel_gate", "why")
_LOG_LEVELS = ("Trace", "Debug", "Info", "Warn", "Error", "Fatal")
_RUN_OUTCOMES = ("Unknown", "Success", "Failure", "Unstable", "Aborted")
# The gate spelling is `self` / `any`, NEVER a literal dialect name. `all_dialect_gates_owned`
# admits exactly two values per gated row — `kAnyDialect` or the manifest's own name — so a
# free-string gate would re-admit the third, illegal case and lean on a consteval fence to
# catch it. This spelling makes that case UNREPRESENTABLE.
_DIALECT_GATE_SELF = "self"
_DIALECT_GATE_ANY = "any"
_DIALECT_GATES = (_DIALECT_GATE_SELF, _DIALECT_GATE_ANY)
_CHANNEL_GATE_ANY = "any"
# ── the section roster (DN-17.D26) ───────────────────────────────────────────
#
# A section name has THREE independent statuses and they are not one axis: whether it is
# LEGAL AT ALL (a fact about `SemanticPackageManifest`'s row surfaces, stable as dialects
# come and go), whether it is authorable NON-EMPTY in schema v1 (does the schema specify a
# row shape this tool can project?), and whether it is authorable EMPTY — the
# declared-absence seat, which lives at the intersection.
_ROW_SECTIONS = ("roles", "markers", "level_lifts", "outcome_tokens",
"outcome_markers", "locations", "value_classes", "declared_values")
_VOCABULARY_SECTIONS = ("channels", "revisions")
_SECTION_NAMES = _ROW_SECTIONS + _VOCABULARY_SECTIONS
# Legal, but EMPTY-only in schema v1: the declared-absence seat and nothing more. The row
# shape arrives with the dialect that needs it, which keeps "a generator capability no
# declaration exercises is dormant plumbing" intact.
_EMPTY_ONLY_SECTIONS = {
"outcome_markers": "its row shape arrives with the Jenkins dialect",
"locations": "its row shape arrives with the test_frameworks package",
"value_classes": ("its row shape is gated on a determinism decision — `scale` is an "
"int64 and would be this schema's first numeric field (DN-17.D19)"),
}
# Three keys earn their OWN refusal rather than the generic unknown-key one. `emits:` above
# all: the manifest HAS that member, so it is the name an implementer will type, and
# "unknown key" would be a far worse answer than the reason. A closed grammar's value is
# the quality of its refusals.
_DIALECT_REJECTIONS = {
"emits": ("derived from `markers:`, never declared; there is no authorable writer "
"projection. Each marker yields exactly one emit row with "
"`emit = dual(extract)` (DN-17.D15) — two authorable projections is the "
"silent divergence the concept exists to forbid."),
"recoverability": ("RESERVED in schema v1, and rejected rather than ignored "
"(DN-17.D18). The block is specified and its fence is expressible, "
"but a grammar that ACCEPTS a claim it cannot project lets someone "
"author a claim that silently does nothing. It arrives with the "
"first real claim, as one additive grammar bump."),
}
_ROOT_KEYS = ("name", "version", "why", "code_tier") + _SECTION_NAMES
# The code tier is a mapping of CLOSED hook kinds — not a section: it carries no list and
# has no empty-section seat. One key per nullable manifest member; an unknown key is a
# generation error, because a third hook kind is a core SPI change (a new nullable member
# with its own signature type), never a tool feature.
_HOOK_KINDS = {
"echoed_source": "insight::semantic::ProvenanceHook",
"strategy": "insight::semantic::StrategyFactory",
}
_HOOK_MEMBER = {"echoed_source": "echoed_source", "strategy": "strategy"}
# ── the `why:` fences (DN-17.D24 / DN-17.D27) ────────────────────────────────
_WHY_FORBIDDEN = {
"
": ("U+2028 LINE SEPARATOR — some tooling treats it as a line terminator, "
"which would end the generated comment early and expose the row beneath"),
"
": ("U+2029 PARAGRAPH SEPARATOR — some tooling treats it as a line terminator, "
"which would end the generated comment early and expose the row beneath"),
"": "U+FEFF — a byte-order mark has no place inside a scalar",
"": "U+202A — a bidirectional override makes text render in an order different "
"from its bytes (the Trojan-Source class); a `why:` exists to be read",
"": "U+202B — a bidirectional override makes text render in an order different "
"from its bytes (the Trojan-Source class); a `why:` exists to be read",
"": "U+202C — a bidirectional override makes text render in an order different "
"from its bytes (the Trojan-Source class); a `why:` exists to be read",
"": "U+202D — a bidirectional override makes text render in an order different "
"from its bytes (the Trojan-Source class); a `why:` exists to be read",
"": "U+202E — a bidirectional override makes text render in an order different "
"from its bytes (the Trojan-Source class); a `why:` exists to be read",
"": "U+2066 — a bidirectional isolate makes text render in an order different "
"from its bytes (the Trojan-Source class); a `why:` exists to be read",
"": "U+2067 — a bidirectional isolate makes text render in an order different "
"from its bytes (the Trojan-Source class); a `why:` exists to be read",
"": "U+2068 — a bidirectional isolate makes text render in an order different "
"from its bytes (the Trojan-Source class); a `why:` exists to be read",
"": "U+2069 — a bidirectional isolate makes text render in an order different "
"from its bytes (the Trojan-Source class); a `why:` exists to be read",
}
# The C0 block and DEL, refused by name. A `why:` is copied VERBATIM into a `//` comment,
# so a control code is not an escaping question here — it is a byte in a source file: NUL
# is a diagnostic on every compile leg, FF/VT/BS make the emitted line render as something
# other than its bytes, and BEL/ESC turn a code review into terminal output. The table
# names the ones an author might plausibly paste; anything else in the range is refused by
# codepoint, because the fence is the RANGE and the names are courtesy.
_WHY_CONTROL_NAMES = {0x00: "NUL", 0x07: "BEL", 0x08: "BS", 0x09: "TAB", 0x0B: "VT",
0x0C: "FF", 0x1B: "ESC", 0x7F: "DEL"}
def _validate_why(node: dict, context: str, source: str) -> list[str]:
"""Validate and return the `why:` block of any mapping the schema defines.
`why:` is an optional key on EVERY mapping — rows, sections, vocabulary entries, the
code tier, each hook, and the document root. That is one rule rather than four seats,
and it is what makes an argument attachable at its point of use.
"""
raw = node.get("why")
if raw is None:
return []
if not isinstance(raw, list):
fail(source, None,
f"{context}: `why:` is a SEQUENCE of single-line scalars, one per line of "
"argument — a bare scalar is not the shape (it is emitted verbatim as "
"comment lines above what it annotates)")
if not any(isinstance(line, str) and line.strip() for line in raw):
fail(source, None,
f"{context}: `why:` carries no argument — an empty one is noise, and the "
"key is optional precisely so that nothing has to be written where there is "
"nothing to say")
lines: list[str] = []
for position, line in enumerate(raw):
where = f"{context}: why[{position}]"
if not isinstance(line, str):
fail(source, None, f"{where}: must be a scalar")
for char in line:
code = ord(char)
if code < 0x20 or code == 0x7F:
named = _WHY_CONTROL_NAMES.get(code)
fail(source, None,
f"{where}: U+{code:04X}"
+ (f" {named}" if named else "")
+ " — a control character. The argument is emitted verbatim as a "
"`//` comment line, so this is not an escaping question: it is a "
"byte in a C++ source file, where a NUL is a diagnostic on every "
"compile leg and a form feed or a backspace makes the line render "
"as something other than what it says. A `why:` carries prose.")
if not line.strip():
# An interior empty scalar is a PARAGRAPH BREAK — it emits a bare `//`, which
# is what makes a multi-paragraph argument readable at the row it annotates.
# It is refused at either end, and doubled, because those emit a stray comment
# line and nothing else: the emitted shape has to be a function of the
# argument, not of the author's spacing habits.
if position in (0, len(raw) - 1):
fail(source, None,
f"{where}: a `why:` may not open or close on an empty line — an "
"empty scalar is a paragraph break, and a break at the edge breaks "
"nothing")
if not raw[position - 1].strip():
fail(source, None,
f"{where}: two consecutive empty lines — one paragraph break is a "
"break; two is whitespace with an opinion")
lines.append("")
continue
for char, reason in _WHY_FORBIDDEN.items():
if char in line:
fail(source, None, f"{where}: {reason}")
if line.endswith("\\"):
fail(source, None,
f"{where}: a trailing backslash. C++ splices lines in translation phase 2, "
"BEFORE comments are removed in phase 3, so a `//` comment ending in a "
"backslash swallows the line beneath it — and in a generated file the line "
"beneath it is a ROW.")
if '\\"' in line:
fail(source, None,
f'{where}: a double quote needs no escape here — `\\"` is a leftover from '
"when this text lived in a C++ string literal; write the quote. A `why:` "
"carries the ARGUMENT as it should READ, never its escaping accidents; "
"the quoting layer owns whatever escaping the scalar itself needs.")
lines.append(line)
return lines
# ── validation ───────────────────────────────────────────────────────────────
def _ascii_field(value: str, context: str, source: str) -> str:
reject_non_ascii_field(value, source, context)
return value
def _enum_value(node: dict, key: str, allowed: tuple[str, ...], context: str,
source: str) -> str:
value = required_scalar(node, key, context, source)
_ascii_field(value, f"{context}: `{key}:`", source)
if value not in allowed:
fail(source, None,
f"{context}: `{key}: {value}` is outside the closed core vocabulary "
f"{{{', '.join(allowed)}}} — growing it is a canon grammar-version bump, "
"never a declaration or a tool feature")
return value
def _dialect_gate(node: dict, context: str, source: str) -> str:
value = required_scalar(node, "dialect_gate", context, source)
if value not in _DIALECT_GATES:
fail(source, None,
f"{context}: `dialect_gate: {value}` — the gate is `self` (this package's own "
f"name) or `any` (fires whatever the caller declared), never a literal dialect "
"name. A row gating to another package's name would reach across a boundary "
"this package does not own; spelling the gate this way makes that case "
"unrepresentable rather than merely asserted (DN-17.D13).")
return value
def _channel_gate(node: dict, channels: list[str], context: str, source: str) -> str:
value = node.get("channel_gate")
if value is None:
fail(source, None, f"{context}: missing required key `channel_gate:`")
if not isinstance(value, str):
fail(source, None, f"{context}: `channel_gate:` must be a scalar")
if value == _CHANNEL_GATE_ANY:
return value
if value not in channels:
declared = ", ".join(channels) if channels else "(this dialect declares none)"
fail(source, None,
f"{context}: `channel_gate: {value}` names no channel this document declares "
f"— the declared vocabulary is: {declared}. This is the generation-time dual "
"of core's `all_channel_gates_declared`, and an unknown channel is a MISTAKE "
"where an absent one is a choice: they must not share a code path.")
return value
def _prefix(node: dict, key: str, context: str, source: str) -> str:
value = required_scalar(node, key, context, source)
_ascii_field(value, f"{context}: `{key}:`", source)
if not value:
fail(source, None, f"{context}: `{key}:` must be non-empty")
return value
def _shape(node: dict, context: str, source: str) -> str:
"""A whole-line SHAPE: literal bytes with `{n}` holes, each a decimal number.
The core matches it against the line's whole content after the transport peel and a
trailing-whitespace trim, greedily and without backtracking, so every refusal below
names a shape whose match would depend on how much a hole takes: a hole is exactly the
number the line prints, never a part of one.
"""
value = _prefix(node, "shape", context, source)
if _SHAPE_HOLE not in value:
fail(source, None,
f"{context}: `shape: {value}` has no hole — a shape carries at least one hole "
f"`{_SHAPE_HOLE}`, the sampled quantity; a line with none is no sample")
if value != value.rstrip():
fail(source, None,
f"{context}: `shape:` ends in trailing whitespace — the core trims a line's "
"trailing whitespace before matching, so this shape could never match")
literals = value.split(_SHAPE_HOLE)
if any("{" in literal or "}" in literal for literal in literals):
fail(source, None,
f"{context}: `shape: {value}` — a hole is spelled `{_SHAPE_HOLE}` and nothing "
"else, and a brace is never literal: the hole is the one closed core kind")
if any(not between for between in literals[1:-1]):
fail(source, None,
f"{context}: `shape: {value}` puts a hole against a digit or another hole — "
"the line's number would then be split between them")
for before, after in zip(literals, literals[1:]):
if (before and before[-1].isdigit()) or (after and after[0].isdigit()):
fail(source, None,
f"{context}: `shape: {value}` puts a hole against a digit or another hole — "
"the line's number would then be split between them")
if (after.startswith(".") and (len(after) == 1 or after[1].isdigit())) \
or (before.endswith(".") and len(before) > 1 and before[-2].isdigit()):
fail(source, None,
f"{context}: `shape: {value}` puts a decimal point between a hole and a "
"digit or a hole — a hole takes a decimal fraction, so the line's number "
"would then be split")
return value
def _validate_role_row(row: dict, context: str, source: str) -> dict:
"""A role row announces its role by a line PREFIX, or by the line's whole SHAPE.
The validated row carries `prefix` or `shape` and never both, so a declaration of prefix
rows alone keeps its content hash and its emitted bytes.
"""
expect_keys(row, ("prefix", "shape", "role", "dialect_gate", "why"), context, source,
_DIALECT_REJECTIONS)
if ("prefix" in row) == ("shape" in row):
fail(source, None,
f"{context}: a role row declares exactly one of `prefix:` and `shape:` — the "
"line's opening bytes, or its whole content with `{n}` holes")
role = _enum_value(row, "role", _STRUCTURAL_ROLES, context, source)
if "prefix" in row:
if role in _SHAPE_ONLY_ROLES:
fail(source, None,
f"{context}: `role: {role}` on a prefix row — the role is declared by its "
"exact shape, because the lines it samples share their prefix with lines "
"that are content (ADR-17.D14)")
matched = {"prefix": _prefix(row, "prefix", context, source)}
else:
matched = {"shape": _shape(row, context, source)}
return {
**matched,
"role": role,
"dialect_gate": _dialect_gate(row, context, source),
"why": _validate_why(row, context, source),
}
def _validate_version_coordinate(node, context: str, source: str) -> dict:
"""Where the row's payload holds its unit's VERSION: an introducer and a payload shape.
The dialect states its own version syntax as data; the core applies it and knows no
platform's reference grammar. Both members are required: a coordinate is declared whole
or not at all, which is the rule composition enforces on a hand-written package.
"""
if not isinstance(node, dict):
fail(source, None,
f"{context}: `version_coordinate:` must be a mapping of `introducer:` and "
"`shape:` — a row that declares no version coordinate omits the key")
expect_keys(node, ("introducer", "shape", "why"), context, source, _DIALECT_REJECTIONS)
return {
"introducer": _prefix(node, "introducer", f"{context} version_coordinate", source),
"shape": _enum_value(node, "shape", _VERSION_PAYLOAD_SHAPES,
f"{context} version_coordinate", source),
"why": _validate_why(node, f"{context} version_coordinate", source),
}
def _validate_unit_role_row(row: dict, channels: list[str], context: str, source: str) -> dict:
"""A row that OPENS a unit of its kind, or CLOSES the steps of its job: a prefix, never a payload.
An opener's unit is named by the naming row that follows, and a closer's unit is the
core's supplied epilogue, so everything a payload derives — the class, the instance, the
version — is never this row's, and a row declaring any of it would be a second, silent
source of the same fact.
"""
role = _enum_value(row, "role", _MARKER_ROLES, context, source)
for key in ("child_order", "extract", "version_coordinate"):
if key in row:
fail(source, None,
f"{context}: `{key}:` on a row declaring `role: {role}` — a row that opens "
"or closes a unit carries no identity: an opened unit is named by the naming "
"row of its kind that follows, a closed job's epilogue is the core's own unit, "
"and neither takes its class, instance or version from this row")
expect_keys(row, _ROLE_ROW_KEYS, context, source, _DIALECT_REJECTIONS)
kind = _enum_value(row, "kind", _MARKER_KINDS, context, source)
if role == "Opens" and kind == "None":
fail(source, None,
f"{context}: `kind: None` on a row declaring `role: Opens` — an opening row "
"opens a unit of a kind, and `None` is no unit")
if role == "Closes" and kind != "Step":
fail(source, None,
f"{context}: `kind: {kind}` on a row declaring `role: Closes` — a closing row "
"enters its job's epilogue, a step-level unit, so it is legal only on `kind: Step`")
return {
"prefix": _prefix(row, "prefix", context, source),
"kind": kind,
"role": role,
"dialect_gate": _dialect_gate(row, context, source),
"channel_gate": _channel_gate(row, channels, context, source),
"why": _validate_why(row, context, source),
}
def _validate_marker_row(row: dict, channels: list[str], context: str, source: str) -> dict:
# The key enters the validated row ONLY when declared, so a declaration whose rows all
# name their units keeps its content hash and its emitted bytes.
if "role" in row:
return _validate_unit_role_row(row, channels, context, source)
expect_keys(row, ("prefix", "kind", "child_order", "dialect_gate", "extract",
"channel_gate", "version_coordinate", "why"), context, source,
_DIALECT_REJECTIONS)
validated = {
"prefix": _prefix(row, "prefix", context, source),
"kind": _enum_value(row, "kind", _MARKER_KINDS, context, source),
"child_order": _enum_value(row, "child_order", _CHILD_ORDERS, context, source),
"dialect_gate": _dialect_gate(row, context, source),
"extract": _enum_value(row, "extract", _PAYLOAD_EXTRACTS, context, source),
"channel_gate": _channel_gate(row, channels, context, source),
"why": _validate_why(row, context, source),
}
# The key enters the validated row ONLY when declared, so a declaration that names no
# version coordinate keeps its content hash and its emitted bytes.
if "version_coordinate" in row:
validated["version_coordinate"] = _validate_version_coordinate(
row["version_coordinate"], context, source)
return validated
def _validate_level_lift_row(row: dict, context: str, source: str) -> dict:
expect_keys(row, ("prefix", "level", "dialect_gate", "why"), context, source,
_DIALECT_REJECTIONS)
return {
"prefix": _prefix(row, "prefix", context, source),
"level": _enum_value(row, "level", _LOG_LEVELS, context, source),
"dialect_gate": _dialect_gate(row, context, source),
"why": _validate_why(row, context, source),
}
def _validate_outcome_token_row(row: dict, context: str, source: str) -> dict:
expect_keys(row, ("token", "outcome", "dialect_gate", "why"), context, source,
_DIALECT_REJECTIONS)
return {
"token": _prefix(row, "token", context, source),
"outcome": _enum_value(row, "outcome", _RUN_OUTCOMES, context, source),
"dialect_gate": _dialect_gate(row, context, source),
"why": _validate_why(row, context, source),
}
def _validate_declared_value_row(row: dict, context: str, source: str) -> dict:
"""A key a stream's acquirer supplies a value under, and the markers the core masks behind.
The key is what a caller types (`--changed-context pull_request=<n>`), so it is spelled like
a dialect name. A marker ending in a digit is refused here as composition refuses it: the
digit run after it would not be the run the marker introduces.
"""
expect_keys(row, ("key", "markers", "dialect_gate", "why"), context, source,
_DIALECT_REJECTIONS)
key = required_scalar(row, "key", context, source)
_ascii_field(key, f"{context}: `key:`", source)
if not key or not key.replace("_", "a").isalnum() or not key[0].isalpha() \
or key != key.lower():
fail(source, None,
f"{context}: `key: {key}` must match [a-z][a-z0-9_]* — it is what a caller "
"declares a value under")
markers = row.get("markers")
if not isinstance(markers, list) or not markers:
fail(source, None,
f"{context}: `markers:` must be a non-empty sequence — a key with no marker "
"could never mask anything")
seen: set[str] = set()
for position, marker in enumerate(markers):
marker_context = f"{context} markers[{position}]"
if not isinstance(marker, str) or not marker:
fail(source, None, f"{marker_context}: a marker is a non-empty scalar")
_ascii_field(marker, marker_context, source)
if marker[-1].isdigit():
fail(source, None,
f"{marker_context}: {marker!r} ends in a digit — the digit run after a "
"marker would then not be the run the marker introduces")
if marker in seen:
fail(source, None, f"{marker_context}: duplicate marker {marker!r}")
seen.add(marker)
return {
"key": key,
"markers": list(markers),
"dialect_gate": _dialect_gate(row, context, source),
"why": _validate_why(row, context, source),
}
_ROW_VALIDATORS = {
"declared_values": _validate_declared_value_row,
"roles": _validate_role_row,
"level_lifts": _validate_level_lift_row,
"outcome_tokens": _validate_outcome_token_row,
}
def _validate_section(node, name: str, source: str, channels: list[str] | None = None) -> dict:
context = f"section `{name}:`"
if not isinstance(node, dict):
fail(source, None,
f"{context}: a section is a MAPPING that carries its list — "
f"`{name}: {{why: [...], rows: [...]}}` — so that an argument has a seat on "
"the section itself, not only on its rows")
payload_key = "names" if name in _VOCABULARY_SECTIONS else "rows"
expect_keys(node, ("why", payload_key), context, source, _DIALECT_REJECTIONS)
why = _validate_why(node, context, source)
items = node.get(payload_key)
if items is None:
fail(source, None, f"{context}: missing required key `{payload_key}:`")
if not isinstance(items, list):
fail(source, None, f"{context}: `{payload_key}:` must be a sequence")
if not items:
# A section declared with an EMPTY list and a mandatory `why:` IS the declared
# absence — "we looked, here is the measurement, there is nothing to declare".
# Without the argument it is noise, and the seat would become a place to enumerate
# nothing; writing the argument is the work, which is what makes it self-limiting.
if not why:
fail(source, None,
f"{context}: an empty section carrying no `why:` is noise — omit the "
"section. An empty section WITH an argument is a declared ABSENCE, which "
"is a claim: it is how an omission and an exclusion stop looking alike.")
return {"why": why, payload_key: []}
if name in _EMPTY_ONLY_SECTIONS:
fail(source, None,
f"{context}: `{name}` may be declared EMPTY with an argument in schema v1 — "
f"{_EMPTY_ONLY_SECTIONS[name]}. Schema v1 specifies no row shape for it, so "
"there is nothing this tool could project from the rows above.")
if name in _VOCABULARY_SECTIONS:
return {"why": why, "names": _validate_vocabulary(items, name, source)}
if name == "markers":
rows = [_validate_marker_row(_as_mapping(item, name, position, source),
channels or [],
f"section `{name}:` rows[{position}]", source)
for position, item in enumerate(items)]
named = {row["kind"] for row in rows if "role" not in row}
for position, row in enumerate(rows):
if row.get("role") == "Opens" and row["kind"] not in named:
fail(source, None,
f"section `{name}:` rows[{position}]: an opening row of kind "
f"`{row['kind']}`, and no row of this section names a unit of that "
"kind — an opener takes effect only when a naming row of its kind "
"follows it, so this row could never take effect")
else:
validator = _ROW_VALIDATORS[name]
rows = [validator(_as_mapping(item, name, position, source),
f"section `{name}:` rows[{position}]", source)
for position, item in enumerate(items)]
return {"why": why, "rows": rows}
def _as_mapping(item, name: str, position: int, source: str) -> dict:
if not isinstance(item, dict):
fail(source, None,
f"section `{name}:` rows[{position}]: a row is a mapping of its declared "
"fields — every field is required, because a row with a defaulted gate is a "
"row nobody decided")
return item
def _validate_vocabulary(items: list, name: str, source: str) -> list[dict]:
"""A vocabulary entry is a bare scalar, or `{name:, why?:}` when it carries an argument.
The one polymorphism the schema admits, and its bound is stated rather than discovered:
an entry carrying NO `why:` may be written as a bare scalar. The two forms are
discriminable by type with no ambiguity, and the bare form is the dominant case —
forcing `- name: v1` on every dialect forever would be pure tax. Row entries stay
mappings always; sections stay mappings always.
"""
entries: list[dict] = []
seen: set[str] = set()
for position, item in enumerate(items):
context = f"section `{name}:` names[{position}]"
if isinstance(item, str):
entry_name, why = item, []
elif isinstance(item, dict):
expect_keys(item, ("name", "why"), context, source, _DIALECT_REJECTIONS)
entry_name = required_scalar(item, "name", context, source)
why = _validate_why(item, context, source)
if not why:
fail(source, None,
f"{context}: the mapping form exists to carry a `why:` — an entry "
"with no argument is written as a bare scalar")
else:
fail(source, None, f"{context}: must be a scalar or a `{{name:, why:}}` mapping")
_ascii_field(entry_name, context, source)
if not entry_name:
fail(source, None,
f"{context}: an empty name IS the any-sentinel on this axis, so it may "
"not also name a concrete member")
if entry_name in seen:
fail(source, None, f"{context}: duplicate name {entry_name!r}")
seen.add(entry_name)
entries.append({"name": entry_name, "why": why})
return entries
def _validate_code_tier(node, package_dir: Path, source: str) -> dict:
context = "`code_tier:`"
if not isinstance(node, dict):
fail(source, None, f"{context}: must be a mapping of hook kinds")
expect_keys(node, ("why",) + tuple(_HOOK_KINDS), context, source, _DIALECT_REJECTIONS)
tier: dict[str, object] = {"why": _validate_why(node, context, source)}
for kind in _HOOK_KINDS:
entry = node.get(kind)
if entry is None:
continue
entry_context = f"{context} `{kind}:`"
if not isinstance(entry, dict):
fail(source, None, f"{entry_context}: must be `{{symbol:, unit:, why?:}}`")
expect_keys(entry, ("symbol", "unit", "why"), entry_context, source,
_DIALECT_REJECTIONS)
symbol = required_scalar(entry, "symbol", entry_context, source)
_ascii_field(symbol, entry_context, source)
if not symbol.replace("_", "a").isalnum() or not symbol[0].isalpha() \
or symbol != symbol.lower():
fail(source, None,
f"{entry_context}: symbol {symbol!r} outside [a-z_][a-z0-9_]* — the hook "
"is NAMED here and DEFINED in C++; this tool does not parse C++ and must "
"not, because a C++ front end inside a generator is an algorithm tier")
unit = required_scalar(entry, "unit", entry_context, source)
_ascii_field(unit, entry_context, source)
_check_unit_path(unit, package_dir, entry_context, source)
tier[kind] = {"symbol": symbol, "unit": unit,
"why": _validate_why(entry, entry_context, source)}
return tier
def _check_unit_path(unit: str, package_dir: Path, context: str, source: str) -> None:
"""The earliest of the three declared failure sites for a hook that does not exist.
Earliest first: an absent `unit:` file is a GENERATION error here; a symbol the wrapper
does not declare is a COMPILE error at the generated `&<symbol>`; a symbol declared but
never defined, or defined at a different signature, is a LINK error naming the symbol.
The third is the declared residual, not a hole: this tool does not parse C++.
The path is normalised relative to the package directory and may be neither absolute
nor contain `..`, so generation stays host-independent — and its RESULT decides success,
never output bytes.
"""
candidate = Path(unit)
if candidate.is_absolute() or ".." in candidate.parts:
fail(source, None,
f"{context}: `unit: {unit}` — the path is relative to the package directory, "
"with no `..` and no absolute form: generation must stay host-independent")
if not (package_dir / candidate).is_file():
fail(source, None,
f"{context}: `unit: {unit}` names no file under {package_dir} — the code tier "
"is REFERENCED by name, so the reference must resolve at generation time")
def validate_declaration(document: dict, stem: str, package_dir: Path, source: str) -> dict:
"""Validate one dialect declaration; return the annotated declaration.
The returned structure carries `why:` for emission; `hashed_content()` strips it, and
that stripped view is exactly what the content hash covers.
"""
expect_keys(document, ("dialect",), "document root", source, _DIALECT_REJECTIONS)
if "dialect" not in document:
fail(source, None, "document root: missing required key `dialect:`")
dialect = document["dialect"]
if not isinstance(dialect, dict):
fail(source, None, "`dialect:` must be a mapping")
expect_keys(dialect, _ROOT_KEYS, "dialect", source, _DIALECT_REJECTIONS)
name = required_scalar(dialect, "name", "dialect", source)
_ascii_field(name, "dialect: `name:`", source)
if not name or not name.replace("_", "a").isalnum() or not name[0].isalpha() \
or name != name.lower():
fail(source, None,
f"dialect name {name!r}: must match [a-z][a-z0-9_]* — it is the manifest "
"name, the value every `self`-gated row carries, and what a caller declares")
if name != stem:
fail(source, None,
f"dialect name {name!r} does not match its file name (expected "
f"`{stem}{DIALECT_FILE_SUFFIX}` to declare `name: {stem}`) — one dialect, "
"one file")
version = required_scalar(dialect, "version", "dialect", source)
_ascii_field(version, "dialect: `version:`", source)
if not version:
fail(source, None, "dialect: `version:` must be non-empty")
declaration: dict[str, object] = {
"name": name,
"version": version,
"why": _validate_why(dialect, "dialect", source),
}
# Channels first: a marker's `channel_gate` is validated against this document's own
# declared vocabulary, so the vocabulary has to exist before any row is read.
for section in _VOCABULARY_SECTIONS:
if section in dialect:
declaration[section] = _validate_section(dialect[section], section, source)
channels = [entry["name"] for entry in
declaration.get("channels", {}).get("names", [])]
if "revisions" not in declaration:
fail(source, None,
"dialect: missing required section `revisions:` — the VENDOR generation these "
"rows recognize. Core requires a non-empty vocabulary with unique, non-empty "
"names (`all_revisions_named`): a package that recognizes nothing in "
"particular is not a state the grammar admits.")
revisions = declaration["revisions"]["names"]
if len(revisions) != 1:
fail(source, None,
f"section `revisions:`: schema v1 admits exactly ONE revision, got "
f"{len(revisions)}. The bound is a SCHEMA bound, not a core one (ADR-17.D9): "
"core carries the general span so that the day a vendor ships a second syntax "
"generation is a data change rather than a redesign, and the v2 subject "
"re-opens this by bumping the declaration schema version.")
for section in _ROW_SECTIONS:
if section in dialect:
declaration[section] = _validate_section(dialect[section], section, source,
channels)
if "code_tier" in dialect:
declaration["code_tier"] = _validate_code_tier(dialect["code_tier"], package_dir,
source)
_check_row_uniqueness(declaration, source)
return declaration
def _check_row_uniqueness(declaration: dict, source: str) -> None:
"""Two rows with the same key inside ONE package are a typo, never a decision.
Composition already refuses a duplicate key across the composed SET; catching it here
names the declaration and the row position instead of the composed table.
"""
keys = {
"roles": lambda row: ("prefix" in row, row.get("prefix", row.get("shape")),
row["dialect_gate"]),
"markers": lambda row: (row["prefix"], row["dialect_gate"], row["channel_gate"]),
"level_lifts": lambda row: (row["prefix"], row["dialect_gate"]),
"outcome_tokens": lambda row: (row["token"], row["dialect_gate"]),
"declared_values": lambda row: (row["key"],),
}
for section, key_of in keys.items():
seen: dict[tuple, int] = {}
for position, row in enumerate(declaration.get(section, {}).get("rows", [])):
key = key_of(row)
if key in seen:
fail(source, None,
f"section `{section}:` rows[{position}]: duplicates rows[{seen[key]}] "
f"on {key} — within one package a duplicate row is a typo, and the "
"second one could never fire")
seen[key] = position
# ── the content hash (DN-17.D19 MUST 4 / DN-17.D20) ──────────────────────────
def hashed_content(declaration: dict) -> dict:
"""The declaration's SEMANTIC content: every `why:` stripped, the NAME stripped.
`why:` is prose, not ruleset content — editing an argument must not revoke a
ratification, which is the same carve-out the hash already applies to formatting. The
name stays out because resolution keys on it before any hash is compared, so a rename
surfaces as a loud stale-record fault rather than as a quiet mismatch.
"""
def strip(node):
if isinstance(node, dict):
return {key: strip(value) for key, value in node.items() if key != "why"}
if isinstance(node, list):
return [strip(item) for item in node]
return node
return {key: strip(value) for key, value in declaration.items()
if key not in ("why", "name")}
def declaration_hash(declaration: dict) -> str:
return canonical_hash(hashed_content(declaration))
# ── C++ emission ─────────────────────────────────────────────────────────────
def _channel_symbol(name: str) -> str:
return "kChannel" + "".join(part.capitalize() for part in name.split("_"))
def _gate_expression(gate: str) -> str:
return "kDialect" if gate == _DIALECT_GATE_SELF else "kAnyDialect"
def _channel_expression(gate: str) -> str:
return "kAnyChannel" if gate == _CHANNEL_GATE_ANY else _channel_symbol(gate)
def _quoted(value: str) -> str:
return f'"{cpp_escape(value)}"'
# The AUTHOR-VOICE marker (DN-17.D35). Every line of `why:` prose reaches the emitted file
# behind `// > `; every other comment line in the emitted file is this tool's own words. The
# rule it enforces: a TOOL's explanation and an AUTHOR's argument answer different questions
# and must be distinguishable in the emitted file.
#
# It is not a style preference, and the near-miss that earned it is worth stating. The tool's
# reason for an empty `locations:` is "its row shape arrives with the test_frameworks package"
# — a claim about what the SCHEMA can project. The declaration's reason is "that is the
# test_frameworks package" — a claim about what THIS DIALECT has to say and who owns the
# vocabulary instead. Same package name, nearly the same words, opposite altitudes. A reader
# scanning for "is this absence explained?" finds the noun, finds a sentence, and stops — so
# sharing the identifying token makes adjacency read as coverage. The two land at ONE SITE
# once a section is declared empty, which is exactly where the confusion would be built in.
#
# Mechanical by construction, which is the point: `^\s*// > ` selects the declaration's voice
# and NOTHING else, so a reviewer scoring whether an argument was carried can do it by
# selection rather than by judgement, without knowing which line this tool wrote. `--selftest`
# holds both halves of that equivalence.
_WHY_MARKER = "// > "
def _every_why_line(declaration: dict) -> list[str]:
"""Every `why:` line in a declaration, at every depth — the author's whole voice.
Walks the parsed declaration rather than a hand-kept list of seats, so a seat added to the
schema is covered the day it exists: the selftest cases below assert a SET EQUIVALENCE, and
an enumeration that could go short would silently weaken one half of it.
"""
lines: list[str] = []
def walk(node: object) -> None:
if isinstance(node, dict):
for key, value in node.items():
if key == "why" and isinstance(value, list):
lines.extend(value)
else:
walk(value)
elif isinstance(node, list):
for item in node:
walk(item)
walk(declaration)
return lines
def _emit_why(out: list[str], why: list[str], indent: str = "") -> None:
for line in why:
out.append(f"{indent}{_WHY_MARKER}{line}" if line
else f"{indent}{_WHY_MARKER.rstrip()}")
def _emit_row(out: list[str], fields: list[tuple[str, str]], why: list[str]) -> None:
_emit_why(out, why, " ")
opening = f" {{.{fields[0][0]} = {fields[0][1]},"
out.append(opening)
pad = " "
for name, value in fields[1:-1]:
out.append(f"{pad}.{name} = {value},")
last_name, last_value = fields[-1]
out.append(f"{pad}.{last_name} = {last_value}}},")
def _emit_array(out: list[str], cpp_type: str, symbol: str, count: int) -> None:
out.append(f"inline constexpr std::array<{cpp_type}, {count}> {symbol}{{{{")
def emit_inc(declaration: dict, *, module_suffix: str, source_name: str) -> str:
name = declaration["name"]
namespace = f"insight::semantic::{name}{module_suffix}"
digest = declaration_hash(declaration)
out: list[str] = []
out.append("// GENERATED by dialect_package_codegen.py -- DO NOT EDIT.")
out.append(f"// tool version: {TOOL_VERSION}, codegen_common: {CODEGEN_COMMON_VERSION}, "
f"declaration schema: {SCHEMA_VERSION}")
out.append("// (all three enter this header and none of them enters the declaration "
"hash -- a tool")
out.append("// upgrade must not mass-revoke, so a stale version string here is "
"cosmetic.)")
out.append(f"// declaration: {source_name}")
out.append(f"// declaration hash: {digest}")