From c4515c0fe1d455db94b281156976aa8e6658244b Mon Sep 17 00:00:00 2001 From: Esteban Zimanyi Date: Wed, 30 Sep 2026 16:45:00 +0200 Subject: [PATCH] Name the C value feeding each column of a returned SQL row MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A SQL signature returning a row lists its columns, each naming the C value that feeds it. PostgreSQL declares the columns of a function returning record as its OUT arguments, and those of a function returning a composite type as the members of that type. An OUT argument is no input, so it leaves args, argDefaults and required: eDisjointPairs(tgeometry[], tgeometry[]) is {args: [tgeometry[], tgeometry[]], ret: record, retSet: true, columns: [i, j]}: the two arrays are its only arguments. attach_row_sources matches each column to its C source by type, so the SQL column order and the C parameter order need not agree. from is return, the value or array the function returns, or an out-parameter, with element or field inside it: valueTimeSplit rows (number, time, tnumber) read the out-parameters value_bins and time_bins and the returned fragments, dynTimeWarpPath rows (i, j) the fields of each returned Match, quaternion the four elements of the returned array. A bool returned beside out-parameters says whether there is a row and feeds no column, and a struct a class stands for, a TBox tile, is one value and not its fields. meta/sql-columns.json states what only the wrapper computes: the index of a tile numbers the rows from 1 (from: ordinal), and the …Pairs wrappers add 1 to the C indices i and j to make them SQL array positions (offset: 1). A row with a column no C value feeds, fed alike by two, or with a C value feeding no column stops the catalog. Why. A binding that returns the rows of a SQL function reads them from the C call: it has to know which C value is which column, and which argument of a record-returning signature is passed rather than returned. Measured. Over MobilityDB 8a23781e4c, 147 SQL signatures carried by 64 functions state their columns: 37 returning record (the …Pairs kernels) and 110 returning one of 61 composite types. Of their columns, 200 read the C return (95 the value or array whole, 74 an element plus the declared offset, 24 a struct field, 7 an element of a fixed array), 96 an out-parameter, and 18 the ordinal. The 37 record signatures drop their OUT arguments from args, and the 12 …Pairs functions their sqlArity from 4, 5 or 6 to 2 or 3. The catalog is otherwise identical to the one derived without the change. Witness. tests/test_sqlfn_rows.py takes the OUT arguments out of args and the members of a composite return type as columns, matches columns by type whatever their order, reads struct fields, fixed-array elements and a pair's elements beside an out-parameter array, applies ordinal and offset only where declared, stops on an unfed, doubly fed or surplus source, and validates meta/sql-columns.json against its schema. The suite floor goes from 317 to 337. --- .github/workflows/pytest.yml | 2 +- meta/sql-columns.json | 62 +++++++ meta/sql-columns.schema.json | 49 +++++ parser/sqlfn.py | 242 ++++++++++++++++++++++++- run.py | 8 +- tests/test_sqlfn_rows.py | 340 +++++++++++++++++++++++++++++++++++ 6 files changed, 692 insertions(+), 11 deletions(-) create mode 100644 meta/sql-columns.json create mode 100644 meta/sql-columns.schema.json create mode 100644 tests/test_sqlfn_rows.py diff --git a/.github/workflows/pytest.yml b/.github/workflows/pytest.yml index d06c02d..df8e0b3 100644 --- a/.github/workflows/pytest.yml +++ b/.github/workflows/pytest.yml @@ -96,7 +96,7 @@ jobs: # carries, or a change to them is not exercised until after it merges. # Consumers use the action; this repository owns the rules. - name: Refuse a skip, and a suite that shrank - run: tools/check-test-outcome.py "$RUNNER_TEMP/pytest.log" --min-tests 317 + run: tools/check-test-outcome.py "$RUNNER_TEMP/pytest.log" --min-tests 337 # The rules earn their place by refusing a log that carries what they # name. Both fixtures are written here rather than tracked, and the diff --git a/meta/sql-columns.json b/meta/sql-columns.json new file mode 100644 index 0000000..cb2decb --- /dev/null +++ b/meta/sql-columns.json @@ -0,0 +1,62 @@ +{ + "$schema": "./sql-columns.schema.json", + "description": "The columns of rows a SQL function returns whose value MEOS does not state. parser/sqlfn.py feeds every column of a returned row from the C value of its type: the value or array the function returns, or an out-parameter. A column its wrapper computes rather than reads is declared here, keyed by the composite type the function returns, or by its SQL name when it returns record; a row with a column neither fed nor declared stops the catalog.", + "rows": { + "index_tbox": { + "note": "The tiles of a temporal box: index numbers them from 1 in the order the MEOS array returns them.", + "columns": {"index": {"from": "ordinal"}} + }, + "index_stbox": { + "note": "The tiles of a spatiotemporal box: index numbers them from 1 in the order the MEOS array returns them.", + "columns": {"index": {"from": "ordinal"}} + }, + "aDisjointPairs": { + "note": "i and j are positions in the SQL arrays, from 1; MEOS returns the C indices, from 0.", + "columns": {"i": {"offset": 1}, "j": {"offset": 1}} + }, + "aDwithinPairs": { + "note": "i and j are positions in the SQL arrays, from 1; MEOS returns the C indices, from 0.", + "columns": {"i": {"offset": 1}, "j": {"offset": 1}} + }, + "aIntersectsPairs": { + "note": "i and j are positions in the SQL arrays, from 1; MEOS returns the C indices, from 0.", + "columns": {"i": {"offset": 1}, "j": {"offset": 1}} + }, + "aTouchesPairs": { + "note": "i and j are positions in the SQL arrays, from 1; MEOS returns the C indices, from 0.", + "columns": {"i": {"offset": 1}, "j": {"offset": 1}} + }, + "eDisjointPairs": { + "note": "i and j are positions in the SQL arrays, from 1; MEOS returns the C indices, from 0.", + "columns": {"i": {"offset": 1}, "j": {"offset": 1}} + }, + "eDwithinPairs": { + "note": "i and j are positions in the SQL arrays, from 1; MEOS returns the C indices, from 0.", + "columns": {"i": {"offset": 1}, "j": {"offset": 1}} + }, + "eIntersectsPairs": { + "note": "i and j are positions in the SQL arrays, from 1; MEOS returns the C indices, from 0.", + "columns": {"i": {"offset": 1}, "j": {"offset": 1}} + }, + "eTouchesPairs": { + "note": "i and j are positions in the SQL arrays, from 1; MEOS returns the C indices, from 0.", + "columns": {"i": {"offset": 1}, "j": {"offset": 1}} + }, + "tDisjointPairs": { + "note": "i and j are positions in the SQL arrays, from 1; MEOS returns the C indices, from 0.", + "columns": {"i": {"offset": 1}, "j": {"offset": 1}} + }, + "tDwithinPairs": { + "note": "i and j are positions in the SQL arrays, from 1; MEOS returns the C indices, from 0.", + "columns": {"i": {"offset": 1}, "j": {"offset": 1}} + }, + "tIntersectsPairs": { + "note": "i and j are positions in the SQL arrays, from 1; MEOS returns the C indices, from 0.", + "columns": {"i": {"offset": 1}, "j": {"offset": 1}} + }, + "tTouchesPairs": { + "note": "i and j are positions in the SQL arrays, from 1; MEOS returns the C indices, from 0.", + "columns": {"i": {"offset": 1}, "j": {"offset": 1}} + } + } +} diff --git a/meta/sql-columns.schema.json b/meta/sql-columns.schema.json new file mode 100644 index 0000000..cef752f --- /dev/null +++ b/meta/sql-columns.schema.json @@ -0,0 +1,49 @@ +{ + "$schema": "http://json-schema.org/draft-07/schema#", + "$id": "sql-columns.schema.json", + "title": "Columns of returned SQL rows that MEOS does not state", + "description": "Declared values of the columns of rows a SQL function returns that its wrapper computes rather than reads from MEOS. Consumed by parser/sqlfn.py, which fails generation on any row with a column neither fed by a C value nor declared here.", + "type": "object", + "required": ["rows"], + "additionalProperties": false, + "properties": { + "$schema": { "type": "string" }, + "description": { "type": "string" }, + "rows": { + "type": "object", + "description": "Keyed by the composite type a function returns, or by its SQL name when it returns record.", + "additionalProperties": { + "type": "object", + "required": ["note", "columns"], + "additionalProperties": false, + "properties": { + "note": { + "type": "string", + "description": "What the wrapper computes, and why MEOS does not state it.", + "minLength": 1 + }, + "columns": { + "type": "object", + "description": "Keyed by column name.", + "minProperties": 1, + "additionalProperties": { + "type": "object", + "minProperties": 1, + "additionalProperties": false, + "properties": { + "from": { + "description": "ordinal: the column numbers the rows from 1, as PostgreSQL's WITH ORDINALITY.", + "const": "ordinal" + }, + "offset": { + "description": "A constant the wrapper adds to the C value feeding the column.", + "type": "integer" + } + } + } + } + } + } + } + } +} diff --git a/parser/sqlfn.py b/parser/sqlfn.py index 84b75f6..d2fc21f 100644 --- a/parser/sqlfn.py +++ b/parser/sqlfn.py @@ -14,10 +14,13 @@ Adds per function (when the chain resolves): `sqlfn`, `sqlop`, `mdbC`. """ +import json import re from pathlib import Path -from parser.typescope import (TypeFacts, declared_scopes, read_bodies, +from parser.shapeinfer import _out_count_param +from parser.typescope import (C_BASE_TYPES, TypeFacts, declared_scopes, read_bodies, + sql_spellings, require_scopes, resolve_scope, signatures_for) # A @csqlfn tag carries one OR MORE #Wrapper() references — comma- or @@ -89,6 +92,12 @@ def _split_top_commas(s): _ARGMODE = re.compile(r"^(?:IN|OUT|INOUT|VARIADIC)\s+", re.I) +# The argument modes naming a column of the row a function returns: PostgreSQL +# declares a record-returning function's columns as OUT (or INOUT) arguments. +_OUTMODE = re.compile(r"^(?:OUT|INOUT)\s+", re.I) +# `CREATE TYPE name AS (` — a composite type, whose members are the columns of the +# row a function returning it gives. +_CREATE_COMPOSITE = re.compile(r"CREATE\s+TYPE\s+(\w+)\s+AS\s*\(", re.I) def _arg_default(decl): @@ -108,6 +117,29 @@ def _bare_type(decl): return re.split(r"\bDEFAULT\b|=", a, maxsplit=1, flags=re.I)[0].strip() +def _is_in_arg(decl): + """Whether a CREATE FUNCTION argument is passed by the caller: every mode but OUT.""" + return not re.match(r"^OUT\s+", decl.strip(), re.I) + + +def _column(decl): + """`(name, type)` of a column declaration: an OUT argument (`OUT i integer`) or a + composite member (`value integer`, `times bigint[]`).""" + name, _, typ = _bare_type(decl).partition(" ") + return name, " ".join(typ.split()) + + +def _composite_types(text): + """Yield (typeName, [(column, type), ...]) for every composite type in `text`.""" + for m in _CREATE_COMPOSITE.finditer(text): + i, depth = m.end(), 1 + while i < len(text) and depth: + depth += (text[i] == "(") - (text[i] == ")") + i += 1 + yield m.group(1), [_column(c) for c in _split_top_commas(text[m.end():i - 1]) + if c.strip()] + + def _arg_type(decl, vocab): """The concrete SQL type of one argument, resolved MECHANICALLY (no hardcoded type list). `vocab` is the .in.sql's own type surface, gathered from the unambiguous @@ -206,21 +238,26 @@ def _create_fn_stmts(text): def _wrapper_sql_sigs(sql_src): """MobilityDB-C wrapper name -> list of per-overload SQL signatures - {sqlName, args:[type,...], required, ret, retSet}, straight from the CREATE FUNCTION + {sqlName, args:[type,...], required, ret, retSet, columns}, straight from the CREATE FUNCTION statements. The .in.sql CREATE FUNCTION set IS the exact SQL registration surface, so a binding emits ONE registration per signature over the concrete arg types with NO type-scope heuristic — e.g. `minInstant` lands on exactly its four overloads {tint,tbigint,tfloat,ttext}, never over tbool or the geo types. `required` counts the non-DEFAULT args (args beyond it are SQL-optional); `ret` is the concrete SQL subtype - the polymorphic `Temporal *` C return loses. Two passes: gather the type vocabulary - from the unambiguous positions, then resolve every arg's type against it.""" + the polymorphic `Temporal *` C return loses. `args` are the arguments a caller + passes; the OUT arguments are no input but the columns of the row returned, and + `columns` lists them, or the members of the composite type returned, as + (name, type) pairs, None for a function returning one value. Two passes: gather + the type vocabulary and the composite types, then resolve every arg's type + against the vocabulary.""" out = {} sql_src = Path(sql_src) if not sql_src.exists(): return out - stmts, vocab = [], set() + stmts, vocab, composites = [], set(), {} for sf in sorted(sql_src.rglob("*.sql")): text = _strip_sql_comments(sf.read_text(errors="ignore")) + composites.update(_composite_types(text)) for sqlname, argdecls, ret, wrapper, retset in _create_fn_stmts(text): stmts.append((sqlname, argdecls, ret, wrapper, retset)) if ret: @@ -232,12 +269,17 @@ def _wrapper_sql_sigs(sql_src): for sqlname, argdecls, ret, wrapper, retset in stmts: if wrapper is None: continue # LANGUAGE SQL / $$ body — no C symbol - args = [_arg_type(a, vocab) for a in argdecls] - arg_defaults = [_arg_default(a) for a in argdecls] - required = sum(1 for a in argdecls if not re.search(r"\bDEFAULT\b", a, re.I)) + indecls = [a for a in argdecls if _is_in_arg(a)] + args = [_arg_type(a, vocab) for a in indecls] + arg_defaults = [_arg_default(a) for a in indecls] + required = sum(1 for a in indecls if not re.search(r"\bDEFAULT\b", a, re.I)) + outcols = [_column(_OUTMODE.sub("", a.strip())) for a in argdecls + if _OUTMODE.match(a.strip())] + columns = outcols or composites.get(ret) out.setdefault(wrapper, []).append( {"sqlName": sqlname, "args": args, "required": required, - "argDefaults": arg_defaults, "ret": ret, "retSet": retset}) + "argDefaults": arg_defaults, "ret": ret, "retSet": retset, + "columns": columns if columns and len(columns) > 1 else None}) return out @@ -374,6 +416,186 @@ def _meos_direct_sql(meos_src): return out +_COLUMNS_META = Path(__file__).resolve().parent.parent / "meta" / "sql-columns.json" + + +def declared_columns(path=_COLUMNS_META): + """The column facts stated in `meta/sql-columns.json`, keyed by the composite + type a function returns, or by its SQL name when it returns `record`.""" + doc = json.loads(Path(path).read_text()) + return {key: entry["columns"] for key, entry in doc["rows"].items()} + + +def _c_base(ctype): + """A C type without `const`, `struct` and its pointer levels, with the + `` spellings read as MEOS's own (`int64_t` is `int64`), and the number of + pointer levels: `const Temporal **` is (`Temporal`, 2).""" + t = re.sub(r"\b(?:const|struct)\b", "", ctype or "") + stars = t.count("*") + t = " ".join(t.replace("*", " ").split()) + return re.sub(r"^(u?int(?:8|16|32|64))_t$", r"\1", t), stars + + +# The cell ids MEOS declares by their own name, `typedef uint64 H3Index` and alike, +# which #_TYPE_MAP of parser/typerecover.py spells `uint64_t` catalog-wide, beside the +# SQL type each one is. +_CELL_IDS = {"H3Index": "h3index", "Quadbin": "quadbin", "S2CellId": "s2cell"} + + +def _sql_ctypes(idl): + """SQL type name -> the C types a value of it can arrive as, read from the + catalog: a class is its `cType` (the object model names `TInt`, `TsTzSpanSet`, + `Geometry` after the SQL types, lower-cased), every temporal type a `Temporal`, + a base type its C spelling (#C_BASE_TYPES of parser/typescope.py, in + PostgreSQL's spelling through #sql_spellings), a cell id `uint64`, and every + base type of the type relations also a `Datum`.""" + out = {} + for cls, rec in ((idl.get("objectModel") or {}).get("classes") or {}).items(): + if rec.get("cType"): + out.setdefault(cls.lower(), set()).add(_c_base(rec["cType"])[0]) + for temptype in idl.get("temporalTypes") or {}: + out.setdefault(temptype, set()).add("Temporal") + for sqltype in _CELL_IDS.values(): + out.setdefault(sqltype, set()).add("uint64") + for c, meos in C_BASE_TYPES.items(): + for m in (meos if isinstance(meos, tuple) else (meos,)): + for name in sql_spellings({m}): + out.setdefault(name, set()).add(c) + for base in ((idl.get("typeRelations") or {}).get("byBase") or {}): + for name in sql_spellings({base}): + out.setdefault(name, set()).add("Datum") + return out + + +def _row_slots(func, retset, width, struct, classes): + """The C values that can feed the columns of one row of `func`, in the order a + row lists them: each is (source, C type, pointer levels) where `source` is the + column's `from` and, inside one C value, its `element` or `field`. + + A set of rows reads one element of each array per row: the returned array + (split into `groupSize` elements per row when it is flattened, into the fields + of its struct when it is an array of structs) and each out-parameter array. + One row reads the returned value, each element of a returned fixed array (a + quaternion), and each out-parameter, an array out-parameter whole; a `bool` + returned beside out-parameters says whether there is a row, and feeds none. + The count of an array feeds no column, and a struct a class stands for (a + `TBox` tile) is one value, not its fields.""" + shape = func.get("shape") or {} + ar = shape.get("arrayReturn") + params = [(p["name"], p.get("cType")) for p in func.get("params") or ()] + length = _out_count_param(func) + outs = set(shape.get("outParams") or ()) + outarrays = {a["param"] for a in shape.get("outputArrays") or ()} + slots = [] + if ar: + elem, stars = _c_base(ar["element"]["c"]) + if ar.get("groupSize"): + slots += [({"from": "return", "element": k}, elem, stars) + for k in range(ar["groupSize"])] + elif struct and not stars and elem not in classes: + slots += [({"from": "return", "field": f["name"]}, *_c_base(f["cType"])) + for f in struct["fields"]] + elif not retset and not outarrays: + slots += [({"from": "return", "element": k}, elem, stars) + for k in range(width)] + else: + slots.append(({"from": "return"}, elem, stars)) + else: + ret, stars = _c_base((func.get("returnType") or {}).get("c")) + if ret not in ("void", "bool") or not outs: + slots.append(({"from": "return"}, ret, stars)) + for name, ctype in params: + if name in outs and name != length: + base, stars = _c_base(ctype) + # An out-parameter points at what it returns: one pointer level less. A + # set of rows reads one element of an array out-parameter per row. + stars -= 1 + (retset and name in outarrays) + slots.append(({"from": name}, base, stars)) + return slots + + +def _fits(sqltype, cbase, stars, sqlc): + """Whether a C value of base type `cbase` behind `stars` pointer levels can be a + value of SQL type `sqltype`: a `Datum` or a by-value base type bare (`int`, + `TimestampTz`), a struct bare (a `TBox` array element) or behind one pointer + (`Temporal *`, `SpanSet *`), an SQL array a C array of its elements.""" + if sqltype.endswith("[]"): + return stars >= 1 and _fits(sqltype[:-2], cbase, stars - 1, sqlc) + if cbase not in sqlc.get(sqltype, ()): + return False + return stars == 0 if cbase == "Datum" else stars <= 1 + + +def _column_sources(func, sig, sqlc, declared, struct): + """The columns of the row `sig` returns, each naming the C value feeding it by + `from`: `return`, the value or array the function returns, or an out-parameter, + and `element` or `field` inside it. Each column is fed by the first C value of + its type not already feeding one, so the SQL column order and the C parameter + order need not agree: `valueTimeSplit` rows `(number, time, tnumber)` read the + out-parameters `value_bins`, `time_bins` and the returned fragments, + `tDisjointPairs` rows `(i, j, periods)` the two elements of the returned index + pair and the out-parameter `periods`. None when a column fits no C value, when + it fits values of two C sources, or when a C value feeds no column. + + `meta/sql-columns.json` states what only the wrapper does: a column numbering + the rows from 1, as PostgreSQL's WITH ORDINALITY (`"from": "ordinal"`, the index + of a tile), and a constant the wrapper adds (`"offset": 1`, turning a C array + index into a SQL array position).""" + stated = declared.get(sig["ret"]) or declared.get(sig.get("sqlName") or func.get("sqlfn")) or {} + cols = sig["columns"] + fed = [c for c in cols if "from" not in stated.get(c["name"], {})] + classes = {c for cs in sqlc.values() for c in cs} + slots = _row_slots(func, sig.get("retSet", False), len(fed), struct, classes) + used, out = set(), {} + for c in fed: + fits = [k for k, (_, base, stars) in enumerate(slots) + if k not in used and _fits(c["type"], base, stars, sqlc)] + if not fits or len({slots[k][0]["from"] for k in fits}) > 1: + return None + used.add(fits[0]) + out[c["name"]] = dict(slots[fits[0]][0]) + if len(used) != len(slots): + return None + result = [] + for c in cols: + entry = {"name": c["name"], "type": c["type"]} + entry.update(out.get(c["name"], {})) + entry.update(stated.get(c["name"], {})) + result.append(entry) + return result + + +def attach_row_sources(idl, declared=None): + """Name the C value feeding each column of every row a SQL signature returns. + + Runs once the object model and the type relations are attached, since a column + is matched to a C value by type. A row whose columns no C value and no + declaration feeds stops the catalog: guessing a column's source would hand a + binding a row PostgreSQL does not return.""" + declared = declared_columns() if declared is None else declared + sqlc = _sql_ctypes(idl) + structs = {s["name"]: s for s in idl.get("structs") or ()} + unfed, n = [], 0 + for f in idl["functions"]: + ar = (f.get("shape") or {}).get("arrayReturn") + struct = structs.get(_c_base(ar["element"]["c"])[0]) if ar else None + for s in f.get("sqlSignatures") or (): + if not s.get("columns"): + continue + columns = _column_sources(f, s, sqlc, declared, struct) + if columns is None: + unfed.append(f"{f['name']}: {s.get('sqlName', f.get('sqlfn'))}" + f"({', '.join(s['args'])}) RETURNS {s['ret']}") + else: + s["columns"] = columns + n += 1 + if unfed: + raise ValueError( + "SQL rows with a column MEOS states no source for; state it in " + "meta/sql-columns.json:\n " + "\n ".join(sorted(set(unfed)))) + return idl, n + + def attach_sqlfn_map(idl, meos_src, mdb_src, sql_src=None): m2d = _meos_to_mdb(meos_src) d2s = _mdb_to_sql(mdb_src) @@ -536,6 +758,8 @@ def attach_sqlfn_map(idl, meos_src, mdb_src, sql_src=None): entry = {"args": s["args"], "ret": s["ret"]} if s["retSet"]: entry["retSet"] = True + if s["columns"]: + entry["columns"] = [{"name": c, "type": t} for c, t in s["columns"]] if any(d is not None for d in s["argDefaults"]): entry["argDefaults"] = s["argDefaults"] if multiname or s["sqlName"] != f["sqlfn"]: diff --git a/run.py b/run.py index 1b7a84e..443a7b9 100644 --- a/run.py +++ b/run.py @@ -16,7 +16,7 @@ from parser.outparam import extract_param_names, merge_outparams from parser.boundargs import merge_boundargs, resolve_bound_names from parser.enrich import enrich_idl -from parser.sqlfn import (attach_sqlfn_map, attach_aggfn_map, +from parser.sqlfn import (attach_sqlfn_map, attach_aggfn_map, attach_row_sources, attach_sqlaggfn_map, lint_container_family_csqlfn, lint_ea_sqlfn, lint_positional_sqlfn, lint_sqlfn_case_collisions) @@ -334,6 +334,12 @@ def main(): # moment a family is added. idl = attach_temporal_types(idl, MOBILITYDB_SRC) + # Name the C value feeding each column of every row a SQL signature returns, matched + # by type once the object model and the type relations state the C type of each SQL + # type; a row with a column nothing feeds stops the catalog. + idl, nrows = attach_row_sources(idl) + print(f" SQL rows with every column's C source: {nrows}", file=sys.stderr) + # Stamp the MobilityDB source commit so the catalog is SELF-DESCRIBING about its freshness: # a consumer proves it is current by comparing sourceCommit to live upstream master, never by # inspecting whatever directory a vendored copy sits in. None when the source is not a git diff --git a/tests/test_sqlfn_rows.py b/tests/test_sqlfn_rows.py new file mode 100644 index 0000000..042341d --- /dev/null +++ b/tests/test_sqlfn_rows.py @@ -0,0 +1,340 @@ +"""The columns of the row a SQL function returns, each fed by a named C value. + +PostgreSQL declares the columns of a record-returning function as its OUT +arguments, and those of a function returning a composite type as the members +of that type. An OUT argument is no input: it leaves `args`, `argDefaults` and +`required` and becomes a column. Each column then names the C value feeding it, +matched by type, so the SQL column order and the C parameter order need not +agree: `from` is `return` (the value or array the function returns) or an +out-parameter, with `element` or `field` inside it. A column the wrapper +computes rather than reads is stated in `meta/sql-columns.json`: `ordinal` +numbers the rows from 1, `offset` is a constant the wrapper adds. A row with a +column nothing feeds, or fed from two C values alike, stops the catalog. + +The catalog signature is built as #_attach of tests/test_sqlfn_setof.py builds +it, and the declarations are validated as #SchemaTests of tests/test_covering.py +validates its descriptor. Plain unittest, no pytest dependency; synthetic +sources via a temp dir. +""" +import copy +import json +import tempfile +import unittest +from pathlib import Path + +from parser.sqlfn import (_wrapper_sql_sigs, attach_row_sources, attach_sqlfn_map, + declared_columns) + +ROOT = Path(__file__).resolve().parents[1] +COLUMNS = ROOT / "meta" / "sql-columns.json" +SCHEMA = ROOT / "meta" / "sql-columns.schema.json" + +MEOS_C = """ +/** + * @ingroup meos_geo_rel_ever + * @brief Return the pairs of indices of temporal geometries that are ever disjoint + * @csqlfn #Edisjoint_tgeoarr_tgeoarr() + */ +int * +edisjoint_tgeoarr_tgeoarr(const Temporal **arr1, int count1, + const Temporal **arr2, int count2, int *count) +{ +} +""" + +MDB_C = """ +/** + * @brief Return the pairs of indices of temporal geometries that are ever disjoint + * @sqlfn eDisjointPairs() + */ +Datum +Edisjoint_tgeoarr_tgeoarr(PG_FUNCTION_ARGS) +{ +} +""" + +MDB_SQL = """ +CREATE TYPE index_tbox AS ( + index integer, + tile tbox +); +CREATE FUNCTION eDisjointPairs(tgeometry[], tgeometry[], OUT i integer, OUT j integer) + RETURNS SETOF record + AS 'MODULE_PATHNAME', 'Edisjoint_tgeoarr_tgeoarr' + LANGUAGE C IMMUTABLE STRICT PARALLEL SAFE; +CREATE FUNCTION tDisjointPairs(tgeometry[], tgeometry[], OUT i integer, OUT j integer, + OUT periods tstzspanset) + RETURNS SETOF record + AS 'MODULE_PATHNAME', 'Tdisjoint_tgeoarr_tgeoarr' + LANGUAGE C IMMUTABLE STRICT PARALLEL SAFE; +CREATE FUNCTION timeTiles(tbox, interval, timestamptz DEFAULT '2000-01-03') + RETURNS SETOF index_tbox + AS 'MODULE_PATHNAME', 'Tbox_time_tiles' + LANGUAGE C IMMUTABLE STRICT PARALLEL SAFE; +CREATE FUNCTION unnest(intset) + RETURNS SETOF integer + AS 'MODULE_PATHNAME', 'Set_unnest' + LANGUAGE C IMMUTABLE STRICT PARALLEL SAFE; +""" + + +def _sigs(): + with tempfile.TemporaryDirectory() as d: + (Path(d) / "x.sql").write_text(MDB_SQL) + return _wrapper_sql_sigs(d) + + +class ColumnDeclarationTests(unittest.TestCase): + + def test_out_arguments_are_columns_not_args(self): + s = _sigs()["Edisjoint_tgeoarr_tgeoarr"][0] + self.assertEqual(s["args"], ["tgeometry[]", "tgeometry[]"]) + self.assertEqual(s["required"], 2) + self.assertEqual(s["argDefaults"], [None, None]) + self.assertEqual(s["columns"], [("i", "integer"), ("j", "integer")]) + + def test_every_out_argument_is_a_column(self): + s = _sigs()["Tdisjoint_tgeoarr_tgeoarr"][0] + self.assertEqual(s["args"], ["tgeometry[]", "tgeometry[]"]) + self.assertEqual(s["columns"], [("i", "integer"), ("j", "integer"), + ("periods", "tstzspanset")]) + + def test_a_composite_return_type_gives_its_members(self): + s = _sigs()["Tbox_time_tiles"][0] + self.assertEqual(s["ret"], "index_tbox") + self.assertEqual(s["columns"], [("index", "integer"), ("tile", "tbox")]) + self.assertEqual(s["argDefaults"], [None, None, "'2000-01-03'"]) + + def test_one_value_has_no_columns(self): + self.assertIsNone(_sigs()["Set_unnest"][0]["columns"]) + + def test_the_catalog_signature_lists_its_columns(self): + idl = {"functions": [{"name": "edisjoint_tgeoarr_tgeoarr", "api": "public"}]} + with tempfile.TemporaryDirectory() as d: + meos, mdb, sql = (Path(d) / "meos" / "src", Path(d) / "mdb", Path(d) / "sql") + for p in (meos / "temporal", mdb, sql): + p.mkdir(parents=True) + (meos / "temporal" / "meos_catalog.c").write_text("") + (meos / "x.c").write_text(MEOS_C) + (mdb / "y.c").write_text(MDB_C) + (sql / "z.sql").write_text(MDB_SQL) + idl, _, _ = attach_sqlfn_map(idl, str(meos), str(mdb), str(sql)) + sig = idl["functions"][0]["sqlSignatures"][0] + self.assertEqual(sig["args"], ["tgeometry[]", "tgeometry[]"]) + self.assertEqual(sig["columns"], [{"name": "i", "type": "integer"}, + {"name": "j", "type": "integer"}]) + self.assertNotIn("argDefaults", sig) + + +# The catalog facts the matching reads: the C type of each class, the temporal +# types, and the structs a returned array can hold. +CATALOG = { + "objectModel": {"classes": { + "TBox": {"cType": "TBox"}, + "TsTzSpanSet": {"cType": "SpanSet"}, + }}, + "temporalTypes": {"tbigint": {}, "tgeometry": {}}, + "structs": [ + {"name": "Match", "fields": [{"name": "i", "cType": "int"}, + {"name": "j", "cType": "int"}]}, + {"name": "TBox", "fields": [{"name": "period", "cType": "Span"}, + {"name": "span", "cType": "Span"}]}, + ], +} + + +def _cols(*pairs): + return [{"name": n, "type": t} for n, t in pairs] + + +def _count(): + return {"name": "count", "cType": "int *"} + + +def _attach(func, declared=None): + idl = copy.deepcopy(CATALOG) + idl["functions"] = [func] + idl, n = attach_row_sources(idl, declared or {}) + return idl["functions"][0]["sqlSignatures"][0]["columns"], n + + +VALUE_TIME_SPLIT = { + "name": "tbigint_value_time_split", + "sqlfn": "valueTimeSplit", + "params": [{"name": "temp", "cType": "const Temporal *"}, + {"name": "vsize", "cType": "int64_t"}, + {"name": "value_bins", "cType": "int64_t **"}, + {"name": "time_bins", "cType": "TimestampTz **"}, + _count()], + "returnType": {"c": "Temporal **"}, + "shape": {"arrayReturn": {"element": {"c": "Temporal *"}}, + "outputArrays": [{"param": "value_bins"}, {"param": "time_bins"}], + "outParams": ["value_bins", "time_bins", "count"]}, + "sqlSignatures": [{"args": ["tbigint", "bigint"], "ret": "number_time_tbigint", + "retSet": True, + "columns": _cols(("number", "bigint"), ("time", "timestamptz"), + ("tnumber", "tbigint"))}], +} + + +def _pairs(with_periods): + params = [{"name": "arr1", "cType": "const Temporal **"}, + {"name": "count1", "cType": "int"}, + {"name": "arr2", "cType": "const Temporal **"}, + {"name": "count2", "cType": "int"}, + _count()] + shape = {"arrayReturn": {"element": {"c": "int"}, "groupSize": 2}, + "outParams": ["count"]} + cols = [("i", "integer"), ("j", "integer")] + if with_periods: + params.append({"name": "periods", "cType": "SpanSet ***"}) + shape["outParams"].append("periods") + shape["outputArrays"] = [{"param": "periods"}] + cols.append(("periods", "tstzspanset")) + return {"name": "tdisjoint_tgeoarr_tgeoarr" if with_periods + else "edisjoint_tgeoarr_tgeoarr", + "sqlfn": "tDisjointPairs" if with_periods else "eDisjointPairs", + "params": params, "returnType": {"c": "int *"}, "shape": shape, + "sqlSignatures": [{"args": ["tgeometry[]", "tgeometry[]"], "ret": "record", + "retSet": True, "columns": _cols(*cols)}]} + + +PAIRS_OFFSET = {"i": {"offset": 1}, "j": {"offset": 1}} + + +class RowSourceTests(unittest.TestCase): + + def test_columns_match_their_source_by_type_not_by_position(self): + """The SQL row is (number, time, tnumber); C returns the fragments and + writes the bins to its out-parameters.""" + cols, n = _attach(copy.deepcopy(VALUE_TIME_SPLIT)) + self.assertEqual(n, 1) + self.assertEqual([(c["name"], c["from"]) for c in cols], + [("number", "value_bins"), ("time", "time_bins"), + ("tnumber", "return")]) + + def test_a_flattened_pair_feeds_one_column_per_element(self): + cols, _ = _attach(_pairs(False), {"eDisjointPairs": PAIRS_OFFSET}) + self.assertEqual(cols, [ + {"name": "i", "type": "integer", "from": "return", "element": 0, "offset": 1}, + {"name": "j", "type": "integer", "from": "return", "element": 1, "offset": 1}]) + + def test_an_out_parameter_array_feeds_the_column_after_the_pair(self): + cols, _ = _attach(_pairs(True), {"tDisjointPairs": PAIRS_OFFSET}) + self.assertEqual(cols[2], {"name": "periods", "type": "tstzspanset", + "from": "periods"}) + self.assertEqual([c.get("element") for c in cols[:2]], [0, 1]) + + def test_a_struct_element_feeds_one_column_per_field(self): + func = {"name": "temporal_dyntimewarp_path", "sqlfn": "dynTimeWarpPath", + "params": [{"name": "temp1", "cType": "const Temporal *"}, _count()], + "returnType": {"c": "Match *"}, + "shape": {"arrayReturn": {"element": {"c": "Match"}}, + "outParams": ["count"]}, + "sqlSignatures": [{"args": ["tgeometry"], "ret": "warp", "retSet": True, + "columns": _cols(("i", "integer"), ("j", "integer"))}]} + cols, _ = _attach(func) + self.assertEqual([(c["from"], c["field"]) for c in cols], + [("return", "i"), ("return", "j")]) + + def test_a_fixed_array_feeds_one_column_per_element(self): + func = {"name": "pose_quaternion", "sqlfn": "quaternion", + "params": [{"name": "pose", "cType": "const Pose *"}, _count()], + "returnType": {"c": "double *"}, + "shape": {"arrayReturn": {"element": {"c": "double"}}, + "outParams": ["count"]}, + "sqlSignatures": [{"args": ["pose"], "ret": "quaternion", + "columns": _cols(("W", "float"), ("X", "float"), + ("Y", "float"), ("Z", "float"))}]} + cols, _ = _attach(func) + self.assertEqual([(c["from"], c["element"]) for c in cols], + [("return", k) for k in range(4)]) + + def test_an_ordinal_column_is_declared_and_a_class_is_one_value(self): + """A `TBox` tile is one value of the class, not its fields.""" + func = {"name": "tintbox_time_tiles", "sqlfn": "timeTiles", + "params": [{"name": "box", "cType": "const TBox *"}, _count()], + "returnType": {"c": "TBox *"}, + "shape": {"arrayReturn": {"element": {"c": "TBox"}}, + "outParams": ["count"]}, + "sqlSignatures": [{"args": ["tbox"], "ret": "index_tbox", "retSet": True, + "columns": _cols(("index", "integer"), ("tile", "tbox"))}]} + cols, _ = _attach(func, {"index_tbox": {"index": {"from": "ordinal"}}}) + self.assertEqual(cols, [ + {"name": "index", "type": "integer", "from": "ordinal"}, + {"name": "tile", "type": "tbox", "from": "return"}]) + + def test_a_found_flag_feeds_no_column(self): + """A `bool` returned beside out-parameters says whether there is a row.""" + func = {"name": "tpoint_as_mvtgeom", "sqlfn": "asMVTGeom", + "params": [{"name": "temp", "cType": "const Temporal *"}, + {"name": "gsarr", "cType": "GSERIALIZED **"}, + {"name": "timesarr", "cType": "int64_t **"}, _count()], + "returnType": {"c": "bool"}, + "shape": {"outputArrays": [{"param": "timesarr"}], + "outParams": ["gsarr", "timesarr", "count"]}, + "sqlSignatures": [{"args": ["tgeometry"], "ret": "geom_times", + "columns": _cols(("geom", "geometry"), + ("times", "bigint[]"))}]} + cols, _ = _attach(func) + self.assertEqual([(c["name"], c["from"]) for c in cols], + [("geom", "gsarr"), ("times", "timesarr")]) + + def test_a_row_without_columns_is_left_alone(self): + func = copy.deepcopy(VALUE_TIME_SPLIT) + del func["sqlSignatures"][0]["columns"] + idl = copy.deepcopy(CATALOG) + idl["functions"] = [func] + _, n = attach_row_sources(idl, {}) + self.assertEqual(n, 0) + + +class UnfedRowTests(unittest.TestCase): + + def test_a_column_no_c_value_fits_stops_the_catalog(self): + func = copy.deepcopy(VALUE_TIME_SPLIT) + func["sqlSignatures"][0]["columns"][0]["type"] = "text" + with self.assertRaisesRegex(ValueError, "valueTimeSplit.*number_time_tbigint"): + _attach(func) + + def test_a_column_two_c_values_fit_stops_the_catalog(self): + func = copy.deepcopy(VALUE_TIME_SPLIT) + func["params"][3] = {"name": "time_bins", "cType": "int64_t **"} + func["sqlSignatures"][0]["columns"][1]["type"] = "bigint" + with self.assertRaises(ValueError): + _attach(func) + + def test_a_c_value_feeding_no_column_stops_the_catalog(self): + func = copy.deepcopy(VALUE_TIME_SPLIT) + del func["sqlSignatures"][0]["columns"][1] + with self.assertRaises(ValueError): + _attach(func) + + def test_an_undeclared_offset_is_not_invented(self): + """Without its declaration the pair reads the C indices as they are.""" + cols, _ = _attach(_pairs(False)) + self.assertNotIn("offset", cols[0]) + + +class DeclaredColumnsTests(unittest.TestCase): + + def test_the_declarations_validate(self): + import jsonschema + jsonschema.validate(json.loads(COLUMNS.read_text()), + json.loads(SCHEMA.read_text())) + + def test_an_unknown_source_is_refused(self): + import jsonschema + doc = json.loads(COLUMNS.read_text()) + doc["rows"]["index_tbox"]["columns"]["index"] = {"from": "position"} + with self.assertRaises(jsonschema.ValidationError): + jsonschema.validate(doc, json.loads(SCHEMA.read_text())) + + def test_the_declarations_are_read_by_row(self): + declared = declared_columns(COLUMNS) + self.assertEqual(declared["index_tbox"], {"index": {"from": "ordinal"}}) + self.assertEqual(declared["tDisjointPairs"], PAIRS_OFFSET) + + +if __name__ == "__main__": + unittest.main()