Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/workflows/pytest.yml
Original file line number Diff line number Diff line change
Expand Up @@ -96,7 +96,7 @@ jobs:
# carries, or a change to them is not exercised until after it merges.
# Consumers use the action; this repository owns the rules.
- name: Refuse a skip, and a suite that shrank
run: tools/check-test-outcome.py "$RUNNER_TEMP/pytest.log" --min-tests 343
run: tools/check-test-outcome.py "$RUNNER_TEMP/pytest.log" --min-tests 354

# The rules earn their place by refusing a log that carries what they
# name. Both fixtures are written here rather than tracked, and the
Expand Down
7 changes: 6 additions & 1 deletion parser/enrich.py
Original file line number Diff line number Diff line change
Expand Up @@ -42,11 +42,16 @@
"other", # anything not classified above
)

# Canonical scalar spellings as emitted by libclang.
# Canonical integer spellings: C's own names, and the fixed-width and size types the C
# standard names, which #normalize_canonical of parser/typerecover.py states a MEOS or
# PostgreSQL integer typedef by (`int32` is `int32_t`, `H3Index` is `uint64_t`).
_INT_BASES = {
"char", "signed char", "unsigned char",
"short", "unsigned short", "int", "unsigned int",
"long", "unsigned long", "long long", "unsigned long long",
"int8_t", "int16_t", "int32_t", "int64_t",
"uint8_t", "uint16_t", "uint32_t", "uint64_t",
"intptr_t", "uintptr_t", "size_t", "ptrdiff_t",
}
_FLOAT_BASES = {"float", "double", "long double"}
_BOOL_BASES = {"bool", "_Bool"}
Expand Down
12 changes: 10 additions & 2 deletions parser/parser.py
Original file line number Diff line number Diff line change
Expand Up @@ -132,9 +132,16 @@ def parse_meos(entry: Path, include_dir: Path,
own_files = {str(p.resolve()) for p in include_dir.glob("**/*.h")}
own_files.update(str(h.resolve()) for h in extra_headers)

# First pass: build a mapping "anonymous struct location -> typedef name"
# First pass: build a mapping "anonymous struct location -> typedef name", and
# record every typedef of the unit, the C library's and an external family's
# (`h3api.h`) included, as the type it names one step down: the chain
# #normalize_canonical of parser/typerecover.py follows to state a scalar.
typedef_map: dict[str, str] = {}
typedefs: dict[str, str] = {}
for node in tu.cursor.walk_preorder():
if node.kind == clang.cindex.CursorKind.TYPEDEF_DECL:
typedefs.setdefault(node.spelling,
node.underlying_typedef_type.spelling)
loc = node.location.file
if not loc or str(Path(loc.name).resolve()) not in own_files:
continue
Expand Down Expand Up @@ -207,8 +214,9 @@ def _dedup_structs(items: list) -> list:
enums = _dedup(enums)
macros = _dedup(macros)

# `_typedefs` serves the type passes of run.py and leaves the catalog with them.
idl = {"functions": functions, "structs": structs, "enums": enums,
"macros": macros}
"macros": macros, "_typedefs": typedefs}

# Resolve types if the mappings file exists
mappings_path = Path("./meta/type-mappings.json")
Expand Down
99 changes: 87 additions & 12 deletions parser/typerecover.py
Original file line number Diff line number Diff line change
Expand Up @@ -202,24 +202,99 @@ def _base_name(t):
return _BASE_RE.sub(" ", t or "").strip()


def normalize_canonical(idl):
# The fixed-width and size types the C standard names (<stdint.h>, <stddef.h>), the
# spellings #_c_base of parser/sqlfn.py reads as MEOS's own: the same width on every
# platform, so a chain of typedefs reaching one is stated by it.
_C_STANDARD = re.compile(r"^(?:u?int(?:8|16|32|64)_t|u?intptr_t|size_t|ptrdiff_t)$")
# C's own scalar type names, which a typedef chain ends at when it meets no name above.
_C_SCALARS = {
"char", "signed char", "unsigned char", "short", "signed short", "unsigned short",
"short int", "signed short int", "unsigned short int", "int", "signed", "signed int",
"unsigned", "unsigned int", "long", "signed long", "unsigned long", "long int",
"signed long int", "unsigned long int", "long long", "signed long long",
"unsigned long long", "long long int", "signed long long int",
"unsigned long long int", "float", "double", "long double", "_Bool", "bool",
}
# A typedef of one scalar or of another name: no struct, union, enum, pointer or array.
_TYPEDEF_DECL = re.compile(
r"^\s*typedef\s+(?!(?:struct|union|enum)\b)([A-Za-z_][\w ]*?)\s+([A-Za-z_]\w*)\s*;",
re.M)


def postgres_scalar_names(pgtypes_root):
"""The scalar typedefs of MobilityDB's vendored PostgreSQL, ``pgtypes/``, the
directory #_public_pgtypes_headers of run.py reads (``TimestampTz``, ``DateADT``,
``TimeADT``, ``Oid``, ``Datum``, ``int32``, ``float8``, ...): the names a ``typedef``
of one scalar or of another name declares there, whichever header declares it."""
root = Path(pgtypes_root)
if not root.is_dir():
return frozenset()
names = set()
for path in root.glob("**/*.h"):
text = re.sub(r"/\*.*?\*/|//[^\n]*", " ", path.read_text(errors="ignore"),
flags=re.S)
names |= {name for _, name in _TYPEDEF_DECL.findall(text)}
return frozenset(names)


# The floating types of C, which have no fixed-width name of their own.
_C_FLOATS = {"float", "double", "long double"}


def scalar_spelling(name, typedefs, pg_names):
"""How the catalog states the scalar type ``name``, read from its typedef chain, as
#_recovery reads a collapsed name from its declaration.

A C standard type (``int64_t``, ``size_t``) is stated by itself. A name defined as
its own ``<stdint.h>`` name, as PostgreSQL 18's ``c.h`` defines the "historical
names for types in <stdint.h>" (``typedef int32_t int32``), and a name for a C
floating type (``typedef double float8``) are stated by the type they name. Any
other typedef PostgreSQL declares (#postgres_scalar_names) is a type of its own
(``TimestampTz``, ``DateADT``, ``Oid``, ``Datum``) and keeps its name. A chain
meeting none of these ends at the C scalar it names. None for a name that is no
typedef of a scalar."""
seen, cur = set(), name
while cur not in seen:
seen.add(cur)
if _C_STANDARD.match(cur):
return cur
nxt = typedefs.get(cur)
if nxt is None:
return cur if cur in _C_SCALARS and cur != name else None
nxt = " ".join(re.sub(r"\b(?:const|volatile)\b", " ", nxt).split())
if nxt == cur + "_t" or nxt in _C_FLOATS:
cur = nxt
continue
if cur in pg_names:
return cur
cur = nxt
return None


def normalize_canonical(idl, pg_names=frozenset()):
"""Re-derive each type slot's ``canonical`` from its ``cType`` typedef.

``canonical`` is the MEOS/PG typedef the public API exposes (``_TYPE_MAP``),
not libclang's fully-resolved platform type. The self-contained (installed)
header parse resolves ``TimestampTz`` -> ``long`` and ``Jsonb *`` -> ``struct
varlena *`` while ``cType`` keeps the faithful typedef, so re-derive
``canonical`` from ``cType`` -- a binding generator keys on ``canonical`` and
must see the semantic type (a timestamp, a jsonb), never its platform width.
Idempotent; a no-op on the source parse (``canonical`` already equals the
typedef) and on non-typedef slots (``Temporal *``, ``int *``). Complements
``recover_collapsed_types``: that recovers a ``cType`` the preprocessor erased
to ``int``; this trusts a faithful ``cType`` and only re-spells ``canonical``.
A scalar typedef is stated by its own typedef chain (#scalar_spelling), as the
unit the catalog parsed declares it (``_typedefs``, recorded by the parser): a type
PostgreSQL declares keeps its name (``TimestampTz``, ``TimeADT``), and any other
reaches the C standard type of its width (``int32`` -> ``int32_t``, ``H3Index`` ->
``uint64_t``), never the platform spelling libclang resolves it to (``TimeADT`` ->
``long``, 32 bits on Windows). Any other typedef keeps the spelling ``_TYPE_MAP``
gives it (``Jsonb``, ``GSERIALIZED``), not libclang's (``struct varlena *``).

A binding generator keys on ``canonical`` and must see the type MEOS declares,
never a platform width. Idempotent; a no-op on non-typedef slots (``Temporal *``,
``int *``). Complements ``recover_collapsed_types``: that recovers a ``cType`` the
preprocessor erased to ``int``; this trusts a faithful ``cType`` and only
re-spells ``canonical``.
"""
fixed = 0
typedefs = idl.pop("_typedefs", None) or {}

def want(ctype):
mapped = _TYPE_MAP.get(_base_name(ctype))
base = _base_name(ctype)
mapped = (scalar_spelling(base, typedefs, pg_names) if base in typedefs
else None) or _TYPE_MAP.get(base)
if not mapped:
return None
const = "const " if re.search(r"\bconst\b", ctype) else ""
Expand Down
6 changes: 4 additions & 2 deletions run.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,7 +8,8 @@
from parser.portable import (attach_portable_aliases, attach_position_names,
classify_backing_sqlfn)
from parser.covering import attach_temporal_covering
from parser.typerecover import recover_collapsed_types, normalize_canonical
from parser.typerecover import (recover_collapsed_types, normalize_canonical,
postgres_scalar_names)
from parser.header_types import reconcile
from parser.shapeinfer import infer_shapes
from parser.nullable import merge_nullable
Expand Down Expand Up @@ -125,7 +126,8 @@ def main():
# source parse leaves them as the typedef. Deriving canonical from the
# faithful cType makes both parses agree, so a binding generator (which
# keys on canonical) marshals timestamps/jsonb rather than dropping them.
idl, ncanon = normalize_canonical(idl)
idl, ncanon = normalize_canonical(idl, postgres_scalar_names(
Path(os.environ.get("MDB_SRC_ROOT", "./_mobilitydb")) / "pgtypes"))
if ncanon:
print(f" normalized {ncanon} canonical spellings to the cType typedef",
file=sys.stderr)
Expand Down
31 changes: 31 additions & 0 deletions tests/test_enrich.py
Original file line number Diff line number Diff line change
Expand Up @@ -236,6 +236,37 @@ def test_lifecycle_and_index_not_exposable(self):
self.assertIn("index", self.n("rtree_insert")["reason"])


class StandardIntegerTests(unittest.TestCase):
"""An integer the catalog states by its C standard name (``int64_t``, ``uint8_t``),
as #normalize_canonical of parser/typerecover.py states every integer typedef,
reads as an integer, as the builtin spellings in #ExposabilityTests do."""

def setUp(self):
self.fns = by_name(enrich_idl({
"functions": [
fn("bigint_to_set", "struct Set *", ("int64_t", "i")),
fn("set_round", "struct Set *",
("const struct Set *", "s"), ("int32_t", "maxdd")),
fn("set_hash", "uint32_t", ("const struct Set *", "s")),
fn("bigintset_in", "struct Set *", ("const char *", "str")),
fn("bigintset_out", "char *", ("const struct Set *", "set")),
],
"structs": [{"name": "Set", "fields": []}],
"enums": [],
}))

def test_a_standard_integer_parameter_is_a_json_integer(self):
f = self.fns["bigint_to_set"]
self.assertEqual(f["wire"]["params"][0],
{"name": "i", "kind": "json", "json": "integer"})
self.assertTrue(f["network"]["exposable"])
self.assertEqual(self.fns["set_round"]["wire"]["params"][1]["json"], "integer")

def test_a_standard_integer_result_is_a_json_integer(self):
self.assertEqual(self.fns["set_hash"]["wire"]["result"],
{"kind": "json", "json": "integer"})


class ApiClassificationTests(unittest.TestCase):
def setUp(self):
self.fns = by_name(make_idl())
Expand Down
125 changes: 125 additions & 0 deletions tests/test_typerecover.py
Original file line number Diff line number Diff line change
Expand Up @@ -213,6 +213,131 @@ def test_genuine_int_left_untouched(self):
self.assertEqual(self._ret("intspan_width"), "int") # genuine scalar int
self.assertEqual(self._ret("tint_values"), "int *") # genuine int array

# ---- the class: no typedef reads as a platform integer ------------------

def test_no_typedef_reads_as_a_platform_integer(self):
# A slot declared by a name (`int32`, `TimeADT`, `H3Index`) states that name's
# definition, never the C integer libclang resolves it to on the host: `long`
# is 64 bits on Linux and 32 on Windows, so a binding keying on it reads the
# wrong width. The assertion runs over every slot, so a new typedef is held
# to it without being named here.
idl = json.loads(IDL.read_text())
slots = [s for f in idl["functions"]
for s in [f["returnType"]] + f.get("params", [])]
slots += [fl for st in idl.get("structs", []) for fl in st.get("fields", [])]
bad = sorted({(_base(s.get("c") or s.get("cType")), _base(s.get("canonical")))
for s in slots
if _base(s.get("c") or s.get("cType")) not in _C_INTEGERS
and _base(s.get("canonical")) in _C_INTEGERS})
self.assertEqual(bad, [], f"typedefs stated as a platform integer: {bad}")

def test_postgres_and_standard_types_keep_their_definition(self):
def canon(name, pname):
p = next(p for p in self.by_name[name]["params"] if p["name"] == pname)
return p["canonical"]
# TimeADT is PostgreSQL's `time`, as DateADT is its `date`.
self.assertEqual(canon("pg_time_out", "time"), "TimeADT")
self.assertEqual(canon("date_to_timestamp", "date"), "DateADT")
self.assertEqual(self.by_name["pg_time_in"]["returnType"]["canonical"], "TimeADT")
# a width name reaches the C standard type PostgreSQL 18 defines it as
self.assertEqual(canon("pg_time_in", "typmod"), "int32_t")
self.assertEqual(canon("set_as_wkb", "variant"), "uint8_t")


# C's own integer names: what a typedef must never be stated as.
_C_INTEGERS = {
"char", "signed char", "unsigned char", "short", "signed short", "unsigned short",
"short int", "int", "signed", "signed int", "unsigned", "unsigned int", "long",
"signed long", "unsigned long", "long int", "unsigned long int", "long long",
"unsigned long long", "long long int", "unsigned long long int",
}


def _base(t):
return " ".join(t.replace("const", " ").replace("struct", " ").replace("*", " ")
.split()) if t else ""


class ScalarSpellingTests(unittest.TestCase):
"""#scalar_spelling and #postgres_scalar_names of parser/typerecover.py, over
typedef chains as the parser records them, one step each."""

TYPEDEFS = {
"int32": "int32_t", "int32_t": "__int32_t", "__int32_t": "int",
"int64": "int64_t", "int64_t": "__int64_t", "__int64_t": "long",
"uint64": "uint64_t", "uint64_t": "__uint64_t", "__uint64_t": "unsigned long",
"Quadbin": "uint64", "H3Index": "uint64_t",
"TimeADT": "int64", "DateADT": "int32", "Oid": "unsigned int",
"Datum": "uintptr_t", "uintptr_t": "unsigned long",
"float8": "double", "raw16": "signed short", "int16": "signed short",
"MeosType": "enum MeosType", "Loop": "Loop",
}
# every scalar typedef pgtypes declares, its base types included
PG = frozenset({"TimeADT", "DateADT", "Oid", "Datum", "int32", "int64", "uint64",
"float8", "int16"})

def spell(self, name):
from parser.typerecover import scalar_spelling
return scalar_spelling(name, self.TYPEDEFS, self.PG)

def test_a_width_name_reaches_its_standard_type(self):
self.assertEqual(self.spell("int32"), "int32_t")
self.assertEqual(self.spell("int64"), "int64_t")

def test_a_cell_reaches_uint64_t_through_uint64(self):
self.assertEqual(self.spell("Quadbin"), "uint64_t")
self.assertEqual(self.spell("H3Index"), "uint64_t")

def test_a_postgres_type_keeps_its_name(self):
# as #test_typedef_canonical_not_platform_resolved holds TimestampTz
self.assertEqual(self.spell("TimeADT"), "TimeADT")
self.assertEqual(self.spell("DateADT"), "DateADT")
self.assertEqual(self.spell("Oid"), "Oid")
# a C standard type below it does not make Datum a width name
self.assertEqual(self.spell("Datum"), "Datum")
# `typedef signed short int16` names no <stdint.h> type, so the chain stops at
# PostgreSQL's name rather than at the platform's short
self.assertEqual(self.spell("int16"), "int16")

def test_a_chain_ending_at_a_builtin_states_the_builtin(self):
self.assertEqual(self.spell("float8"), "double")
self.assertEqual(self.spell("raw16"), "signed short")

def test_no_scalar_no_spelling(self):
self.assertIsNone(self.spell("MeosType"))
self.assertIsNone(self.spell("Loop"))
self.assertIsNone(self.spell("int"))

def test_the_postgres_names_are_every_scalar_typedef_wherever_it_sits(self):
import tempfile
from parser.typerecover import postgres_scalar_names
with tempfile.TemporaryDirectory() as d:
root = Path(d)
(root / "pg_basetypes.h").write_text(
"typedef int32_t int32;\ntypedef double float8;\n"
"#ifndef DATE_H\ntypedef int32 DateADT;\n#endif\n")
(root / "datatype").mkdir()
(root / "datatype" / "timestamp.h").write_text(
"typedef int64 TimestampTz;\n/* typedef int64 Commented; */\n"
"typedef struct varlena bytea;\ntypedef char *Pointer;\n")
self.assertEqual(postgres_scalar_names(root),
{"int32", "float8", "DateADT", "TimestampTz"})

def test_normalize_states_a_slot_by_its_chain(self):
from parser.typerecover import normalize_canonical
idl = {"_typedefs": dict(self.TYPEDEFS), "functions": [
{"name": "pg_time_in", "returnType": {"c": "TimeADT", "canonical": "long"},
"params": [{"name": "typmod", "cType": "int32", "canonical": "int"},
{"name": "cells", "cType": "const Quadbin *",
"canonical": "const unsigned long *"}]}]}
idl, fixed = normalize_canonical(idl, self.PG)
f = idl["functions"][0]
self.assertEqual(f["returnType"]["canonical"], "TimeADT")
self.assertEqual([p["canonical"] for p in f["params"]],
["int32_t", "const uint64_t *"])
self.assertEqual(fixed, 3)
self.assertNotIn("_typedefs", idl)


if __name__ == "__main__":
unittest.main()
Loading