From dd693c3c87dfece385318a039f15fe9275e93b93 Mon Sep 17 00:00:00 2001 From: Esteban Zimanyi Date: Mon, 28 Sep 2026 14:26:45 +0200 Subject: [PATCH] State the byte codec of each type beside its wire encodings typeEncodings gives each type, under bytes, the reader and the writer of its WKB bytes: a reader T *f(const uint8_t *wkb, size_t size) and a writer uint8_t *f(const T *, uint8_t variant, size_t *size_out), recognised by shape, the generic _from_wkb and _as_wkb preferred as the wire codecs prefer theirs. The bytes are no wire string, so they add no entry to encodings, which keeps the string forms a network generator serves. A binding that holds values in process, as the JVM engines do, reads the codec there instead of naming it itself. Witness. test_byte_codec_beside_the_wire_encodings states the Temporal pair, leaves the encodings at mfjson and text, and states no codec for a type with a writer and no reader. Measured. Over the headers of MobilityDB 3f9ee86639, twelve types carry the entry: Temporal, Set, Span, SpanSet, STBox, TBox, Cbuffer, Npoint, Pose, PoseChain, Raquet and Raster. The catalog is otherwise identical to the one derived without the change. The suite runs its 37 test files, all passing. Why. The Spark generator names the WKB functions in a table of its own and the Flink generator finds only the hex codec here, so the two engines carry values in two forms; stated once in the catalog, both take the same codec from it. --- parser/enrich.py | 30 +++++++++++++++++++++++++++++- tests/test_enrich.py | 20 ++++++++++++++++++++ 2 files changed, 49 insertions(+), 1 deletion(-) diff --git a/parser/enrich.py b/parser/enrich.py index 30aa980..af5b4a5 100644 --- a/parser/enrich.py +++ b/parser/enrich.py @@ -6,7 +6,8 @@ - ``category`` — a coarse semantic class (constructor, predicate, io, ...). - ``typeEncodings``— for each opaque C type, how it round-trips to the wire - (text / MF-JSON / WKB) and the function names that do it. + (text / MF-JSON / WKB) and the function names that do it, + and under ``bytes`` the reader and writer of its WKB bytes. - ``network`` — whether the function can be projected onto a *stateless* endpoint, and if not, why. - ``wire`` — per-parameter and return value, the concrete request / @@ -238,6 +239,28 @@ def choose(cands: dict, base: str, suffix: str) -> str: generic = base.lower() + suffix return generic if generic in cands else sorted(cands)[0] + # The byte codec of a type, beside its wire encodings: a reader + # ``T *f(const uint8_t *wkb, size_t size)`` and a writer + # ``uint8_t *f(const T *, uint8_t variant, size_t *size_out)``, recognised by + # shape. The bytes are no wire string, so they add no ``encodings`` entry; an + # in-process binding carries a value as them (the JVM engines). + bread: dict[str, list] = {} + bwrite: dict[str, list] = {} + for fn in functions: + params = fn.get("params", []) + ret = fn["returnType"]["canonical"] + cs = [p["canonical"] for p in params] + if (len(cs) == 2 and _base(cs[0]) == "uint8_t" and _ptr_depth(cs[0]) == 1 + and _base(cs[1]) == "size_t" and _ptr_depth(cs[1]) == 0 + and _ptr_depth(ret) == 1 and _base(ret) in structs): + bread.setdefault(_base(ret), []).append(fn["name"]) + if (len(cs) == 3 and _base(ret) == "uint8_t" and _ptr_depth(ret) == 1 + and _ptr_depth(cs[0]) == 1 and _base(cs[0]) in structs + and _base(cs[1]) in ("uint8_t", "unsigned char") + and _ptr_depth(cs[1]) == 0 + and _base(cs[2]) == "size_t" and _ptr_depth(cs[2]) == 1): + bwrite.setdefault(_base(cs[0]), []).append(fn["name"]) + out: dict[str, dict] = {} for base, s in enc.items(): dec = {e: choose(c, base, dec_suffix[e]) @@ -255,6 +278,11 @@ def choose(cands: dict, base: str, suffix: str) -> str: "in_aux": s["decoders"][in_e][dec[in_e]] if in_e else [], "out_aux": s["encoders"][out_e][encd[out_e]] if out_e else [], } + if base in bread and base in bwrite: + out[base]["bytes"] = { + "decoder": choose(dict.fromkeys(bread[base]), base, "_from_wkb"), + "encoder": choose(dict.fromkeys(bwrite[base]), base, "_as_wkb"), + } return out diff --git a/tests/test_enrich.py b/tests/test_enrich.py index 6723b87..213982d 100644 --- a/tests/test_enrich.py +++ b/tests/test_enrich.py @@ -62,6 +62,15 @@ def fn(name, ret, *params): ("const struct Box *", "box"), ("int", "maxdd")), fn("weird_in", "struct Weird *", ("const char *", "str"), ("int", "basetype")), + # The byte codec: a reader of (bytes, length) and a writer of (value, + # variant, *size_out). Box has a writer and no reader, so no byte codec. + fn("temporal_from_wkb", "struct Temporal *", + ("const uint8_t *", "wkb"), ("size_t", "size")), + fn("temporal_as_wkb", "uint8_t *", + (T, "temp"), ("uint8_t", "variant"), ("size_t *", "size_out")), + fn("box_as_wkb", "uint8_t *", + ("const struct Box *", "box"), ("uint8_t", "variant"), + ("size_t *", "size_out")), # An otherwise-exposable function carrying an internal doxygen group: it # must be policy-excluded (api=internal), like the programmer Datum API. dict(fn("internal_op", "struct Temporal *", (T, "temp")), @@ -149,6 +158,17 @@ def test_defaultable_aux_accepted_type_tag_rejected(self): # Weird gets no decoder at all. self.assertNotIn("Weird", self.te) + def test_byte_codec_beside_the_wire_encodings(self): + # the byte-codec twin of #test_struct_prefix_stripped_and_round_trip + self.assertEqual(self.te["Temporal"]["bytes"], + {"decoder": "temporal_from_wkb", + "encoder": "temporal_as_wkb"}) + # the bytes are no wire string: the encodings stay the string forms + self.assertEqual(self.te["Temporal"]["encodings"], ["mfjson", "text"]) + # a writer without a reader states no codec + self.assertNotIn("bytes", self.te["Box"]) + self.assertNotIn("bytes", self.te["Set"]) + def test_no_primitive_or_intermediate_false_positives(self): self.assertNotIn("int", self.te) # was a real false positive self.assertNotIn("char", self.te)