From c7b9f45cad8533027bf5c66d596d18b5693c2401 Mon Sep 17 00:00:00 2001 From: Esteban Zimanyi Date: Thu, 1 Oct 2026 09:48:43 +0200 Subject: [PATCH] State every scalar type by its definition, never by the platform's A slot declared by a scalar typedef states what MEOS defines it as, read from the typedef chain of the unit the catalog parses: the parser records every typedef one step down, the C library's and libh3's included. A type PostgreSQL declares in MobilityDB's vendored pgtypes keeps its name: TimestampTz, DateADT, TimeADT, Oid, Datum. Any other integer typedef is stated by the C standard type its chain reaches: int32 is int32_t, int16 is int16_t, H3Index, Quadbin and S2CellId are uint64_t. A chain meeting neither ends at the C scalar it names, float8 at double. libclang's resolution on the host is never the answer: it states TimeADT as long, 32 bits on Windows, and int32 as int. A name defined as its own name, as PostgreSQL 18's c.h defines the historical names for types in (typedef int32_t int32), and a name for a C floating type (typedef double float8) name a width and not a type of their own, so the chain goes on through them, whichever header defines them. The other typedefs keep the spelling _TYPE_MAP gives them (Jsonb, GSERIALIZED), and _TYPE_MAP still recovers a name the preprocessor erased to int. The service projection reads the C standard's integer types as integers, as it reads int and long: a parameter or result spelled int64_t, uint64_t, uint32_t, uint8_t or size_t is a JSON integer. Why. Bindings key on canonical. A binding that read TimeADT as long read 32 bits on Windows, one that read uint64_t never met the int32 of the same header spelled the same way, and every typedef missing from a hand list reached the bindings as whatever libclang made of it on the host. The chain leaves no list to complete: a new typedef is stated by its definition. Measured. Over MobilityDB 72abdf86c2, whose pg_basetypes.h defines int8 to uint64 as PostgreSQL 18's c.h does, 557 slots change: 258 int32, 111 int16 and 77 int32_t slots read int32_t and int16_t for int and short, 35 uint8_t read uint8_t for unsigned char, 73 TimeADT read TimeADT for long, and 3 Oid read Oid for unsigned int. 368 functions taking or returning a C standard integer enter the service projection, 2516 exposable becoming 2860; 23 time functions and text_cmp leave it, TimeADT and Oid joining the TimestampTz and DateADT the projection maps to no wire type. Witness. tests/test_typerecover.py follows the chains of a width name, a cell, a PostgreSQL type, Datum over uintptr_t, a width name spelled as no type, a float and a name no scalar; reads PostgreSQL's names from every scalar typedef of a pgtypes tree, whichever header declares it, without its structs and pointers; and states a slot by its chain. Over the catalog, no slot declared by a typedef reads as a C integer, and TimeADT, DateADT, int32 and uint8_t slots read their definition. tests/test_enrich.py reads int64_t, int32_t and uint32_t as JSON integers. The suite floor goes from 343 to 354. --- .github/workflows/pytest.yml | 2 +- parser/enrich.py | 7 +- parser/parser.py | 12 +++- parser/typerecover.py | 99 +++++++++++++++++++++++---- run.py | 6 +- tests/test_enrich.py | 31 +++++++++ tests/test_typerecover.py | 125 +++++++++++++++++++++++++++++++++++ 7 files changed, 264 insertions(+), 18 deletions(-) diff --git a/.github/workflows/pytest.yml b/.github/workflows/pytest.yml index 5aed1c7..d1024c1 100644 --- a/.github/workflows/pytest.yml +++ b/.github/workflows/pytest.yml @@ -96,7 +96,7 @@ jobs: # carries, or a change to them is not exercised until after it merges. # Consumers use the action; this repository owns the rules. - name: Refuse a skip, and a suite that shrank - run: tools/check-test-outcome.py "$RUNNER_TEMP/pytest.log" --min-tests 343 + run: tools/check-test-outcome.py "$RUNNER_TEMP/pytest.log" --min-tests 354 # The rules earn their place by refusing a log that carries what they # name. Both fixtures are written here rather than tracked, and the diff --git a/parser/enrich.py b/parser/enrich.py index af5b4a5..98a9b38 100644 --- a/parser/enrich.py +++ b/parser/enrich.py @@ -42,11 +42,16 @@ "other", # anything not classified above ) -# Canonical scalar spellings as emitted by libclang. +# Canonical integer spellings: C's own names, and the fixed-width and size types the C +# standard names, which #normalize_canonical of parser/typerecover.py states a MEOS or +# PostgreSQL integer typedef by (`int32` is `int32_t`, `H3Index` is `uint64_t`). _INT_BASES = { "char", "signed char", "unsigned char", "short", "unsigned short", "int", "unsigned int", "long", "unsigned long", "long long", "unsigned long long", + "int8_t", "int16_t", "int32_t", "int64_t", + "uint8_t", "uint16_t", "uint32_t", "uint64_t", + "intptr_t", "uintptr_t", "size_t", "ptrdiff_t", } _FLOAT_BASES = {"float", "double", "long double"} _BOOL_BASES = {"bool", "_Bool"} diff --git a/parser/parser.py b/parser/parser.py index 04c3924..a41b790 100644 --- a/parser/parser.py +++ b/parser/parser.py @@ -132,9 +132,16 @@ def parse_meos(entry: Path, include_dir: Path, own_files = {str(p.resolve()) for p in include_dir.glob("**/*.h")} own_files.update(str(h.resolve()) for h in extra_headers) - # First pass: build a mapping "anonymous struct location -> typedef name" + # First pass: build a mapping "anonymous struct location -> typedef name", and + # record every typedef of the unit, the C library's and an external family's + # (`h3api.h`) included, as the type it names one step down: the chain + # #normalize_canonical of parser/typerecover.py follows to state a scalar. typedef_map: dict[str, str] = {} + typedefs: dict[str, str] = {} for node in tu.cursor.walk_preorder(): + if node.kind == clang.cindex.CursorKind.TYPEDEF_DECL: + typedefs.setdefault(node.spelling, + node.underlying_typedef_type.spelling) loc = node.location.file if not loc or str(Path(loc.name).resolve()) not in own_files: continue @@ -207,8 +214,9 @@ def _dedup_structs(items: list) -> list: enums = _dedup(enums) macros = _dedup(macros) + # `_typedefs` serves the type passes of run.py and leaves the catalog with them. idl = {"functions": functions, "structs": structs, "enums": enums, - "macros": macros} + "macros": macros, "_typedefs": typedefs} # Resolve types if the mappings file exists mappings_path = Path("./meta/type-mappings.json") diff --git a/parser/typerecover.py b/parser/typerecover.py index b374693..4a9c05c 100644 --- a/parser/typerecover.py +++ b/parser/typerecover.py @@ -202,24 +202,99 @@ def _base_name(t): return _BASE_RE.sub(" ", t or "").strip() -def normalize_canonical(idl): +# The fixed-width and size types the C standard names (, ), the +# spellings #_c_base of parser/sqlfn.py reads as MEOS's own: the same width on every +# platform, so a chain of typedefs reaching one is stated by it. +_C_STANDARD = re.compile(r"^(?:u?int(?:8|16|32|64)_t|u?intptr_t|size_t|ptrdiff_t)$") +# C's own scalar type names, which a typedef chain ends at when it meets no name above. +_C_SCALARS = { + "char", "signed char", "unsigned char", "short", "signed short", "unsigned short", + "short int", "signed short int", "unsigned short int", "int", "signed", "signed int", + "unsigned", "unsigned int", "long", "signed long", "unsigned long", "long int", + "signed long int", "unsigned long int", "long long", "signed long long", + "unsigned long long", "long long int", "signed long long int", + "unsigned long long int", "float", "double", "long double", "_Bool", "bool", +} +# A typedef of one scalar or of another name: no struct, union, enum, pointer or array. +_TYPEDEF_DECL = re.compile( + r"^\s*typedef\s+(?!(?:struct|union|enum)\b)([A-Za-z_][\w ]*?)\s+([A-Za-z_]\w*)\s*;", + re.M) + + +def postgres_scalar_names(pgtypes_root): + """The scalar typedefs of MobilityDB's vendored PostgreSQL, ``pgtypes/``, the + directory #_public_pgtypes_headers of run.py reads (``TimestampTz``, ``DateADT``, + ``TimeADT``, ``Oid``, ``Datum``, ``int32``, ``float8``, ...): the names a ``typedef`` + of one scalar or of another name declares there, whichever header declares it.""" + root = Path(pgtypes_root) + if not root.is_dir(): + return frozenset() + names = set() + for path in root.glob("**/*.h"): + text = re.sub(r"/\*.*?\*/|//[^\n]*", " ", path.read_text(errors="ignore"), + flags=re.S) + names |= {name for _, name in _TYPEDEF_DECL.findall(text)} + return frozenset(names) + + +# The floating types of C, which have no fixed-width name of their own. +_C_FLOATS = {"float", "double", "long double"} + + +def scalar_spelling(name, typedefs, pg_names): + """How the catalog states the scalar type ``name``, read from its typedef chain, as + #_recovery reads a collapsed name from its declaration. + + A C standard type (``int64_t``, ``size_t``) is stated by itself. A name defined as + its own ```` name, as PostgreSQL 18's ``c.h`` defines the "historical + names for types in " (``typedef int32_t int32``), and a name for a C + floating type (``typedef double float8``) are stated by the type they name. Any + other typedef PostgreSQL declares (#postgres_scalar_names) is a type of its own + (``TimestampTz``, ``DateADT``, ``Oid``, ``Datum``) and keeps its name. A chain + meeting none of these ends at the C scalar it names. None for a name that is no + typedef of a scalar.""" + seen, cur = set(), name + while cur not in seen: + seen.add(cur) + if _C_STANDARD.match(cur): + return cur + nxt = typedefs.get(cur) + if nxt is None: + return cur if cur in _C_SCALARS and cur != name else None + nxt = " ".join(re.sub(r"\b(?:const|volatile)\b", " ", nxt).split()) + if nxt == cur + "_t" or nxt in _C_FLOATS: + cur = nxt + continue + if cur in pg_names: + return cur + cur = nxt + return None + + +def normalize_canonical(idl, pg_names=frozenset()): """Re-derive each type slot's ``canonical`` from its ``cType`` typedef. - ``canonical`` is the MEOS/PG typedef the public API exposes (``_TYPE_MAP``), - not libclang's fully-resolved platform type. The self-contained (installed) - header parse resolves ``TimestampTz`` -> ``long`` and ``Jsonb *`` -> ``struct - varlena *`` while ``cType`` keeps the faithful typedef, so re-derive - ``canonical`` from ``cType`` -- a binding generator keys on ``canonical`` and - must see the semantic type (a timestamp, a jsonb), never its platform width. - Idempotent; a no-op on the source parse (``canonical`` already equals the - typedef) and on non-typedef slots (``Temporal *``, ``int *``). Complements - ``recover_collapsed_types``: that recovers a ``cType`` the preprocessor erased - to ``int``; this trusts a faithful ``cType`` and only re-spells ``canonical``. + A scalar typedef is stated by its own typedef chain (#scalar_spelling), as the + unit the catalog parsed declares it (``_typedefs``, recorded by the parser): a type + PostgreSQL declares keeps its name (``TimestampTz``, ``TimeADT``), and any other + reaches the C standard type of its width (``int32`` -> ``int32_t``, ``H3Index`` -> + ``uint64_t``), never the platform spelling libclang resolves it to (``TimeADT`` -> + ``long``, 32 bits on Windows). Any other typedef keeps the spelling ``_TYPE_MAP`` + gives it (``Jsonb``, ``GSERIALIZED``), not libclang's (``struct varlena *``). + + A binding generator keys on ``canonical`` and must see the type MEOS declares, + never a platform width. Idempotent; a no-op on non-typedef slots (``Temporal *``, + ``int *``). Complements ``recover_collapsed_types``: that recovers a ``cType`` the + preprocessor erased to ``int``; this trusts a faithful ``cType`` and only + re-spells ``canonical``. """ fixed = 0 + typedefs = idl.pop("_typedefs", None) or {} def want(ctype): - mapped = _TYPE_MAP.get(_base_name(ctype)) + base = _base_name(ctype) + mapped = (scalar_spelling(base, typedefs, pg_names) if base in typedefs + else None) or _TYPE_MAP.get(base) if not mapped: return None const = "const " if re.search(r"\bconst\b", ctype) else "" diff --git a/run.py b/run.py index 443a7b9..2e186b6 100644 --- a/run.py +++ b/run.py @@ -8,7 +8,8 @@ from parser.portable import (attach_portable_aliases, attach_position_names, classify_backing_sqlfn) from parser.covering import attach_temporal_covering -from parser.typerecover import recover_collapsed_types, normalize_canonical +from parser.typerecover import (recover_collapsed_types, normalize_canonical, + postgres_scalar_names) from parser.header_types import reconcile from parser.shapeinfer import infer_shapes from parser.nullable import merge_nullable @@ -125,7 +126,8 @@ def main(): # source parse leaves them as the typedef. Deriving canonical from the # faithful cType makes both parses agree, so a binding generator (which # keys on canonical) marshals timestamps/jsonb rather than dropping them. - idl, ncanon = normalize_canonical(idl) + idl, ncanon = normalize_canonical(idl, postgres_scalar_names( + Path(os.environ.get("MDB_SRC_ROOT", "./_mobilitydb")) / "pgtypes")) if ncanon: print(f" normalized {ncanon} canonical spellings to the cType typedef", file=sys.stderr) diff --git a/tests/test_enrich.py b/tests/test_enrich.py index 213982d..3d76bb6 100644 --- a/tests/test_enrich.py +++ b/tests/test_enrich.py @@ -236,6 +236,37 @@ def test_lifecycle_and_index_not_exposable(self): self.assertIn("index", self.n("rtree_insert")["reason"]) +class StandardIntegerTests(unittest.TestCase): + """An integer the catalog states by its C standard name (``int64_t``, ``uint8_t``), + as #normalize_canonical of parser/typerecover.py states every integer typedef, + reads as an integer, as the builtin spellings in #ExposabilityTests do.""" + + def setUp(self): + self.fns = by_name(enrich_idl({ + "functions": [ + fn("bigint_to_set", "struct Set *", ("int64_t", "i")), + fn("set_round", "struct Set *", + ("const struct Set *", "s"), ("int32_t", "maxdd")), + fn("set_hash", "uint32_t", ("const struct Set *", "s")), + fn("bigintset_in", "struct Set *", ("const char *", "str")), + fn("bigintset_out", "char *", ("const struct Set *", "set")), + ], + "structs": [{"name": "Set", "fields": []}], + "enums": [], + })) + + def test_a_standard_integer_parameter_is_a_json_integer(self): + f = self.fns["bigint_to_set"] + self.assertEqual(f["wire"]["params"][0], + {"name": "i", "kind": "json", "json": "integer"}) + self.assertTrue(f["network"]["exposable"]) + self.assertEqual(self.fns["set_round"]["wire"]["params"][1]["json"], "integer") + + def test_a_standard_integer_result_is_a_json_integer(self): + self.assertEqual(self.fns["set_hash"]["wire"]["result"], + {"kind": "json", "json": "integer"}) + + class ApiClassificationTests(unittest.TestCase): def setUp(self): self.fns = by_name(make_idl()) diff --git a/tests/test_typerecover.py b/tests/test_typerecover.py index 77c75c9..cbce880 100644 --- a/tests/test_typerecover.py +++ b/tests/test_typerecover.py @@ -213,6 +213,131 @@ def test_genuine_int_left_untouched(self): self.assertEqual(self._ret("intspan_width"), "int") # genuine scalar int self.assertEqual(self._ret("tint_values"), "int *") # genuine int array + # ---- the class: no typedef reads as a platform integer ------------------ + + def test_no_typedef_reads_as_a_platform_integer(self): + # A slot declared by a name (`int32`, `TimeADT`, `H3Index`) states that name's + # definition, never the C integer libclang resolves it to on the host: `long` + # is 64 bits on Linux and 32 on Windows, so a binding keying on it reads the + # wrong width. The assertion runs over every slot, so a new typedef is held + # to it without being named here. + idl = json.loads(IDL.read_text()) + slots = [s for f in idl["functions"] + for s in [f["returnType"]] + f.get("params", [])] + slots += [fl for st in idl.get("structs", []) for fl in st.get("fields", [])] + bad = sorted({(_base(s.get("c") or s.get("cType")), _base(s.get("canonical"))) + for s in slots + if _base(s.get("c") or s.get("cType")) not in _C_INTEGERS + and _base(s.get("canonical")) in _C_INTEGERS}) + self.assertEqual(bad, [], f"typedefs stated as a platform integer: {bad}") + + def test_postgres_and_standard_types_keep_their_definition(self): + def canon(name, pname): + p = next(p for p in self.by_name[name]["params"] if p["name"] == pname) + return p["canonical"] + # TimeADT is PostgreSQL's `time`, as DateADT is its `date`. + self.assertEqual(canon("pg_time_out", "time"), "TimeADT") + self.assertEqual(canon("date_to_timestamp", "date"), "DateADT") + self.assertEqual(self.by_name["pg_time_in"]["returnType"]["canonical"], "TimeADT") + # a width name reaches the C standard type PostgreSQL 18 defines it as + self.assertEqual(canon("pg_time_in", "typmod"), "int32_t") + self.assertEqual(canon("set_as_wkb", "variant"), "uint8_t") + + +# C's own integer names: what a typedef must never be stated as. +_C_INTEGERS = { + "char", "signed char", "unsigned char", "short", "signed short", "unsigned short", + "short int", "int", "signed", "signed int", "unsigned", "unsigned int", "long", + "signed long", "unsigned long", "long int", "unsigned long int", "long long", + "unsigned long long", "long long int", "unsigned long long int", +} + + +def _base(t): + return " ".join(t.replace("const", " ").replace("struct", " ").replace("*", " ") + .split()) if t else "" + + +class ScalarSpellingTests(unittest.TestCase): + """#scalar_spelling and #postgres_scalar_names of parser/typerecover.py, over + typedef chains as the parser records them, one step each.""" + + TYPEDEFS = { + "int32": "int32_t", "int32_t": "__int32_t", "__int32_t": "int", + "int64": "int64_t", "int64_t": "__int64_t", "__int64_t": "long", + "uint64": "uint64_t", "uint64_t": "__uint64_t", "__uint64_t": "unsigned long", + "Quadbin": "uint64", "H3Index": "uint64_t", + "TimeADT": "int64", "DateADT": "int32", "Oid": "unsigned int", + "Datum": "uintptr_t", "uintptr_t": "unsigned long", + "float8": "double", "raw16": "signed short", "int16": "signed short", + "MeosType": "enum MeosType", "Loop": "Loop", + } + # every scalar typedef pgtypes declares, its base types included + PG = frozenset({"TimeADT", "DateADT", "Oid", "Datum", "int32", "int64", "uint64", + "float8", "int16"}) + + def spell(self, name): + from parser.typerecover import scalar_spelling + return scalar_spelling(name, self.TYPEDEFS, self.PG) + + def test_a_width_name_reaches_its_standard_type(self): + self.assertEqual(self.spell("int32"), "int32_t") + self.assertEqual(self.spell("int64"), "int64_t") + + def test_a_cell_reaches_uint64_t_through_uint64(self): + self.assertEqual(self.spell("Quadbin"), "uint64_t") + self.assertEqual(self.spell("H3Index"), "uint64_t") + + def test_a_postgres_type_keeps_its_name(self): + # as #test_typedef_canonical_not_platform_resolved holds TimestampTz + self.assertEqual(self.spell("TimeADT"), "TimeADT") + self.assertEqual(self.spell("DateADT"), "DateADT") + self.assertEqual(self.spell("Oid"), "Oid") + # a C standard type below it does not make Datum a width name + self.assertEqual(self.spell("Datum"), "Datum") + # `typedef signed short int16` names no type, so the chain stops at + # PostgreSQL's name rather than at the platform's short + self.assertEqual(self.spell("int16"), "int16") + + def test_a_chain_ending_at_a_builtin_states_the_builtin(self): + self.assertEqual(self.spell("float8"), "double") + self.assertEqual(self.spell("raw16"), "signed short") + + def test_no_scalar_no_spelling(self): + self.assertIsNone(self.spell("MeosType")) + self.assertIsNone(self.spell("Loop")) + self.assertIsNone(self.spell("int")) + + def test_the_postgres_names_are_every_scalar_typedef_wherever_it_sits(self): + import tempfile + from parser.typerecover import postgres_scalar_names + with tempfile.TemporaryDirectory() as d: + root = Path(d) + (root / "pg_basetypes.h").write_text( + "typedef int32_t int32;\ntypedef double float8;\n" + "#ifndef DATE_H\ntypedef int32 DateADT;\n#endif\n") + (root / "datatype").mkdir() + (root / "datatype" / "timestamp.h").write_text( + "typedef int64 TimestampTz;\n/* typedef int64 Commented; */\n" + "typedef struct varlena bytea;\ntypedef char *Pointer;\n") + self.assertEqual(postgres_scalar_names(root), + {"int32", "float8", "DateADT", "TimestampTz"}) + + def test_normalize_states_a_slot_by_its_chain(self): + from parser.typerecover import normalize_canonical + idl = {"_typedefs": dict(self.TYPEDEFS), "functions": [ + {"name": "pg_time_in", "returnType": {"c": "TimeADT", "canonical": "long"}, + "params": [{"name": "typmod", "cType": "int32", "canonical": "int"}, + {"name": "cells", "cType": "const Quadbin *", + "canonical": "const unsigned long *"}]}]} + idl, fixed = normalize_canonical(idl, self.PG) + f = idl["functions"][0] + self.assertEqual(f["returnType"]["canonical"], "TimeADT") + self.assertEqual([p["canonical"] for p in f["params"]], + ["int32_t", "const uint64_t *"]) + self.assertEqual(fixed, 3) + self.assertNotIn("_typedefs", idl) + if __name__ == "__main__": unittest.main()