From 3b3f11fe14a874c6b13c89056590bd4feb371725 Mon Sep 17 00:00:00 2001 From: Esteban Zimanyi Date: Thu, 1 Oct 2026 14:03:52 +0200 Subject: [PATCH] Restate each function's wire from the codecs its classes state Each function's network and wire are stated again once the codec pass settles every class (parser/enrich.py restate_wire, run right after state_type_encodings), through the same assess that first states them, and the exposable count with them. A serialized value reads and writes through its class's in and out: a Temporal parameter reads through temporal_from_hexwkb, whose WKB carries the subtype, and a Temporal result writes through temporal_as_mfjson with its trailing inputs. Why. The wire was stated before the codec pass and kept the per-class pick that pass replaces: every Temporal parameter read through tbigint_in, every Set through bigintset_in. A tfloat read so fails: tbigint_in('[1.5@2026-01-01 00:00:00+00, 2.5@2026-01-02 00:00:00+00]') raises 'invalid input syntax for type bigint: "1.5"'. Measured. Over MobilityDB 72abdf86c2, 4194 wire values named a codec their class does not state and none does now: 1821 Temporal parameters from tbigint_in and 689 results from temporal_out, 461 Set, 435 Span and 296 SpanSet parameters from the bigint readers and 492 of their results from *_out; each parameter now reads through its class's hex-WKB reader, each Temporal result writes through temporal_as_mfjson and each Set, Span and SpanSet result through its hex-WKB writer. The six Raster functions writing a raster become exposable through raster_as_hexwkb, 3228 to 3234, and none stops being exposable. The same tfloat, written as hex WKB, reads through temporal_from_hexwkb and writes back through temporal_as_mfjson as {"type":"MovingFloat","values":[1.5,2.5],...}. Witness. tests/test_enrich.py restates a parameter and a result through a settled codec and the exposable count with them, and over the catalog finds every wire value naming its class's codec; both fail on the code without this change. The suite floor goes from 395 to 398. --- .github/workflows/pytest.yml | 2 +- docs/enrichment.md | 15 ++++++++--- parser/enrich.py | 17 ++++++++++++ run.py | 4 ++- tests/test_enrich.py | 51 +++++++++++++++++++++++++++++++++++- 5 files changed, 82 insertions(+), 7 deletions(-) diff --git a/.github/workflows/pytest.yml b/.github/workflows/pytest.yml index e06b987..025c04b 100644 --- a/.github/workflows/pytest.yml +++ b/.github/workflows/pytest.yml @@ -96,7 +96,7 @@ jobs: # carries, or a change to them is not exercised until after it merges. # Consumers use the action; this repository owns the rules. - name: Refuse a skip, and a suite that shrank - run: tools/check-test-outcome.py "$RUNNER_TEMP/pytest.log" --min-tests 395 + run: tools/check-test-outcome.py "$RUNNER_TEMP/pytest.log" --min-tests 398 # The rules earn their place by refusing a log that carries what they # name. Both fixtures are written here rather than tracked, and the diff --git a/docs/enrichment.md b/docs/enrichment.md index efab2ea..fd59650 100644 --- a/docs/enrichment.md +++ b/docs/enrichment.md @@ -109,15 +109,22 @@ For every function: "network": { "exposable": true, "method": "POST", "reason": null }, "wire": { "params": [ - { "name": "temp1", "kind": "serialized", "cType": "Temporal *", - "decode": "temporal_in", "encodings": ["mfjson","text","wkb"] }, - { "name": "temp2", "kind": "serialized", "cType": "Temporal *", - "decode": "temporal_in", "encodings": ["mfjson","text","wkb"] } + { "name": "temp1", "kind": "serialized", "cType": "const Temporal *", + "decode": "temporal_from_hexwkb", "decode_aux": [], + "encodings": ["mfjson","text","wkb"] }, + { "name": "temp2", "kind": "serialized", "cType": "const Temporal *", + "decode": "temporal_from_hexwkb", "decode_aux": [], + "encodings": ["mfjson","text","wkb"] } ], "result": { "kind": "json", "json": "boolean" } } ``` +A `serialized` value reads and writes through its class's `in` and `out` with +their `in_aux` and `out_aux`, as section 2 states them once the SQL signatures +are in (`parser/enrich.py` `restate_wire`): a `Temporal` reads through +`temporal_from_hexwkb`, whose WKB carries the subtype. + `wire` element `kind`: | kind | meaning | JSON Schema hint | diff --git a/parser/enrich.py b/parser/enrich.py index 47b5c71..d716ffb 100644 --- a/parser/enrich.py +++ b/parser/enrich.py @@ -626,3 +626,20 @@ def enrich_idl(idl: dict) -> dict: ), } return idl + + +def restate_wire(idl: dict) -> dict: + """Restate each function's ``network`` and ``wire``, and the exposable count, from + ``idl["typeEncodings"]`` as it stands, through #assess as #enrich_idl first states + them. #state_type_encodings of parser/codecs.py settles each class's codec after + enrich (``Temporal`` reads through ``temporal_from_hexwkb``, not enrich's + ``tbigint_in``), so a wire read before it names codecs the catalog no longer + states.""" + enum_names = {e["name"] for e in idl.get("enums", [])} + type_encodings = idl.get("typeEncodings", {}) + functions = idl.get("functions", []) + for fn in functions: + fn["network"], fn["wire"] = assess(fn, type_encodings, enum_names) + idl["enrichment"]["exposableFunctions"] = sum( + 1 for fn in functions if fn["network"]["exposable"]) + return idl diff --git a/run.py b/run.py index d5e544a..5f8c4aa 100644 --- a/run.py +++ b/run.py @@ -18,7 +18,7 @@ from parser.boundargs import (attach_call_literals, merge_boundargs, resolve_bound_names, strip_call_literals) from parser.compositions import attach_compositions -from parser.enrich import enrich_idl +from parser.enrich import enrich_idl, restate_wire from parser.codecs import state_type_encodings from parser.sqlfn import (attach_sqlfn_map, attach_aggfn_map, attach_row_sources, attach_sqlaggfn_map, lint_container_family_csqlfn, @@ -333,6 +333,8 @@ def main(): if codec_errors: raise ValueError("type encodings that contradict themselves:\n " + "\n ".join(codec_errors)) + # Each function's wire names the codec its classes now state. + idl = restate_wire(idl) # 6. Attach the temporal-covering descriptor (Parquet/Iceberg projection) print(f" Attaching temporal covering from {COVERING_PATH}...", diff --git a/tests/test_enrich.py b/tests/test_enrich.py index 9ac41d4..9a72e0a 100644 --- a/tests/test_enrich.py +++ b/tests/test_enrich.py @@ -15,7 +15,7 @@ sys.path.insert(0, str(Path(__file__).resolve().parents[1])) from parser.enrich import (_aux_specs, build_type_encodings, classify_category, - enrich_idl) + enrich_idl, restate_wire) def fn(name, ret, *params): @@ -285,6 +285,55 @@ def test_a_value_type_without_a_codec_registers_nothing(self): self.assertEqual(self.fns["datum_hash"]["network"]["reason"], "no-decoder:Datum") +class RestatedWireTests(unittest.TestCase): + """A wire read after the classes settle their codec, as #state_type_encodings of + parser/codecs.py settles them, names the codec they state, as #ExposabilityTests + reads the wire enrich first states.""" + + def setUp(self): + self.idl = make_idl() + te = self.idl["typeEncodings"]["Temporal"] + te["in"], te["in_aux"] = "temporal_from_hexwkb", [] + te["out"] = "temporal_as_mfjson" + restate_wire(self.idl) + self.fns = by_name(self.idl) + + def test_a_parameter_and_a_result_read_the_settled_codec(self): + w = self.fns["tpoint_speed"]["wire"] + self.assertEqual(w["params"][0]["decode"], "temporal_from_hexwkb") + self.assertEqual(w["result"]["encode"], "temporal_as_mfjson") + + def test_the_exposable_count_follows_the_wire(self): + self.assertEqual(self.idl["enrichment"]["exposableFunctions"], + sum(f["network"]["exposable"] for f in self.idl["functions"])) + + +class WireCatalogTests(unittest.TestCase): + """Over the generated catalog, every value on the wire reads and writes through the + codec its class states, as #RestatedWireTests states.""" + + def test_every_wire_value_names_its_class_codec(self): + idl_path = Path(__file__).resolve().parents[1] / "output" / "meos-idl.json" + if not idl_path.exists(): + self.skipTest(f"{idl_path} not generated; run `python run.py` first") + idl = json.loads(idl_path.read_text()) + te = idl["typeEncodings"] + stale = [] + for f in idl["functions"]: + w = f["wire"] + vals = [(p, "decode", "in") for p in w["params"] if p["kind"] == "serialized"] + vals += [(p["element"], "decode", "in") for p in w["params"] + if p["kind"] == "array"] + if w["result"].get("kind") == "serialized": + vals.append((w["result"], "encode", "out")) + for v, key, side in vals: + cls = " ".join(v["cType"].replace("const", " ").replace("struct", " ") + .replace("*", " ").split()) + if v[key] != te[cls][side]: + stale.append((f["name"], v[key], te[cls][side])) + self.assertEqual(stale, []) + + class StandardIntegerTests(unittest.TestCase): """An integer the catalog states by its C standard name (``int64_t``, ``uint8_t``), as #normalize_canonical of parser/typerecover.py states every integer typedef,