From 0d6fd7ba0947e7915d7c079fac19a555b83aeca8 Mon Sep 17 00:00:00 2001 From: Esteban Zimanyi Date: Sun, 4 Oct 2026 18:54:53 +0200 Subject: [PATCH] Read a geometry and a geography through the HexEWKB reader of each The codec patterns of parser/enrich.py read a hex reader as _from_hex_?e?wkb and a hex writer as _as_hex_?e?wkb, so they take the spelling MEOS gives the functions of the geometries, geom_from_hexewkb, geog_from_hexewkb and geo_as_hexewkb, beside the _from_hexwkb and _as_hexwkb of every other type. MobilityDB backs the SQL functions geometryFromHexEWKB(text), geographyFromHexEWKB(text) and asHexEWKB(geometry|geography, text) with them, so the readers carry SQL signatures returning geometry and geography, and the class they share, GSERIALIZED, states its wkb encoding as Set or Temporal states theirs: readers keyed by SQL type, geom_from_hexewkb for geometry and geog_from_hexewkb for geography, and the one writer geo_as_hexewkb as its wkb encoder, its endian left NULL, which wkb_variant_from_endian reads as the machine's byte order. Its text and MF-JSON forms carry no SQL signature, PostGIS owning geometry and geography, and keep the functions the C shapes give them. Witness: over MobilityDB master 2d13c5c42d, typeEncodings.GSERIALIZED states readers.wkb = {geography: geog_from_hexewkb, geometry: geom_from_hexewkb} and encoders.wkb = geo_as_hexewkb, where it stated no wkb encoding, so a binding reading a codec per SQL type found none for geometry or geography. Measured over MobilityDB master 2d13c5c42d, the catalog derived from the installed headers of a libmeos built by tools/provision-meos.sh, against the catalog the same headers give on master: GSERIALIZED is the one class whose codec changes; its 473 parameters gain the wkb encoding in their wire; the three functions are filed under io rather than conversion; nothing else moves. The suite passes its 497 tests, 25 skipped as on master, tests/test_codecs.py stating the codec on a synthetic catalog and on the derived one. Why: one GSERIALIZED stands for two SQL types, so a single decoder cannot tell a binding which of them a string names; the reader each SQL type's own function provides can, as intset_in and floatset_in do for Set. --- docs/enrichment.md | 12 +++++++----- parser/codecs.py | 6 +++--- parser/enrich.py | 4 ++-- tests/test_codecs.py | 21 ++++++++++++++++++++- 4 files changed, 32 insertions(+), 11 deletions(-) diff --git a/docs/enrichment.md b/docs/enrichment.md index fd59650..d60db36 100644 --- a/docs/enrichment.md +++ b/docs/enrichment.md @@ -59,8 +59,8 @@ literals are in the catalog. ``` - **decoder** — a public `const char * (+ aux) → T` (`*_in`, `*_from_mfjson`, - `*_from_hexwkb`, …); **encoder** — a public `T (+ aux) → char *` (`*_out`, - `*_as_mfjson`, `*_as_hexwkb`, …). A size the function writes back is an + `*_from_hexwkb`, `*_from_hexewkb`, …); **encoder** — a public `T (+ aux) → char *` + (`*_out`, `*_as_mfjson`, `*_as_hexwkb`, `*_as_hexewkb`, …). A size the function writes back is an out-parameter, stated in its `shape.outParams` as for any other function. - **A PostgreSQL type passed by value** (`TimestampTz`, `Timestamp`, `TimeADT`, `DateADT`) is a class like a pointer type, its decoder returning it and its @@ -68,14 +68,16 @@ literals are in the catalog. PostgreSQL's spelling `pg_X` yields to the name MEOS states it under, `X`, when both serve a class (`timetz_in`, never `pg_timetz_in`). - **`readers` / `writers`** — a class several SQL types share (`Set`, `Span`, - `SpanSet`, `Temporal`) reads and writes each type through that type's own public + `SpanSet`, `Temporal`, `GSERIALIZED`) reads and writes each type through that type's own public function, keyed by the SQL type its signature returns (a reader) or takes (a writer). An encoding with a `readers` or `writers` entry has no single `decoders` or `encoders` entry. Within one SQL type, the function the encoding table names first wins (`cbuffer_out` before `cbuffer_as_ewkt`), then the narrower one (`cbufferset_out` before `spatialset_out`); two that tie stop the catalog. An - encoding whose functions carry no SQL signature (`GSERIALIZED`, whose geometry and - geography are PostGIS's types) keeps the function the C shapes give it. + encoding whose functions carry no SQL signature keeps the function the C shapes give + it: the text and MF-JSON forms of `GSERIALIZED`, whose geometry and geography are + PostGIS's types, while its HexEWKB readers `geom_from_hexewkb` and + `geog_from_hexewkb` are keyed by `geometry` and `geography`. - **`in` / `out`** — the class's single decoder and encoder in the order `text` > `mfjson` > `wkb`, absent when no single function serves the whole class. - **Trailing inputs** — every decoder, encoder, reader and writer states the inputs diff --git a/parser/codecs.py b/parser/codecs.py index ab8eb7a..ef9cc4d 100644 --- a/parser/codecs.py +++ b/parser/codecs.py @@ -168,9 +168,9 @@ def resolve(cls, enc, cands, prior, sql_type, side, table): """(function and aux, or None; {SQL type: (name, aux)}) of one encoding: what the SQL signatures key (#keyed), else the function #build_type_encodings of parser/enrich.py states for the encoding, else the encoding's one public - candidate. A class whose readers and writers carry no SQL signature - (``GSERIALIZED``, whose geometry and geography are PostGIS's types) keeps what - enrich states.""" + candidate. An encoding whose readers and writers carry no SQL signature + (the text and MF-JSON forms of ``GSERIALIZED``, whose geometry and geography are + PostGIS's types) keeps what enrich states.""" pick, by_type = keyed(cls, cands, sql_type, side, table) if cands else (None, {}) if pick or by_type: return pick, by_type diff --git a/parser/enrich.py b/parser/enrich.py index d716ffb..a518e1c 100644 --- a/parser/enrich.py +++ b/parser/enrich.py @@ -65,7 +65,7 @@ (re.compile(r"_as_e?wkt$"), "text"), (re.compile(r"_as_mfjson$"), "mfjson"), (re.compile(r"_as_geojson$"), "mfjson"), - (re.compile(r"_as_hex_?wkb$"), "wkb"), + (re.compile(r"_as_hex_?e?wkb$"), "wkb"), (re.compile(r"_as_e?wkb$"), "wkb"), ] _DECODERS = [ @@ -74,7 +74,7 @@ (re.compile(r"_from_text$"), "text"), (re.compile(r"_from_mfjson$"), "mfjson"), (re.compile(r"_from_geojson$"), "mfjson"), - (re.compile(r"_from_hex_?wkb$"), "wkb"), + (re.compile(r"_from_hex_?e?wkb$"), "wkb"), (re.compile(r"_from_e?wkb$"), "wkb"), ] _IO_RE = [rx for rx, _ in _DECODERS + _ENCODERS] diff --git a/tests/test_codecs.py b/tests/test_codecs.py index 46e28cb..6344877 100644 --- a/tests/test_codecs.py +++ b/tests/test_codecs.py @@ -74,6 +74,13 @@ def sig(name, args, ret, **kw): # a cell: a value of its own over uint64_t, read and written by its own functions fn("h3index_in", "uint64_t", [("const char *", "str")], typedef="H3Index"), fn("h3index_out", "char *", [("uint64_t", "cell")], typedef="H3Index"), + # geometry and geography share one class: a HexEWKB reader per type, one writer for both + fn("geom_from_hexewkb", "GSERIALIZED *", [("const char *", "hexwkb")], + [sig(None, ["text"], "geometry")]), + fn("geog_from_hexewkb", "GSERIALIZED *", [("const char *", "hexwkb")], + [sig(None, ["text"], "geography")]), + fn("geo_as_hexewkb", "char *", [("const GSERIALIZED *", "gs"), ("const char *", "endian")], + [sig(None, ["geometry", "text"], "text"), sig(None, ["geography", "text"], "text")]), ] @@ -81,7 +88,7 @@ def _idl(functions=FUNCTIONS): return {"functions": json.loads(json.dumps(functions)), "macros": [{"name": "WKB_EXTENDED", "value": 4}], "structs": [{"name": "Set", "fields": []}, {"name": "Raster", "fields": []}], - "typeEncodings": {"Set": {}, "Raster": {}}} + "typeEncodings": {"Set": {}, "Raster": {}, "GSERIALIZED": {}}} class CodecTests(unittest.TestCase): @@ -123,6 +130,15 @@ def test_without_a_send_the_variant_is_the_sql_writers_default(self): self.assertEqual(r["out"], "raster_as_hexwkb") self.assertEqual(r["encoderAux"]["wkb"][0]["default"], 0) + def test_a_hexewkb_reader_per_type_and_one_writer_for_both(self): + g = self.te["GSERIALIZED"] + self.assertEqual(g["readers"]["wkb"], {"geometry": "geom_from_hexewkb", + "geography": "geog_from_hexewkb"}) + self.assertNotIn("wkb", g["decoders"]) + self.assertEqual(g["encoders"]["wkb"], "geo_as_hexewkb") + self.assertEqual(g["encoderAux"]["wkb"], + [{"name": "endian", "kind": "string", "default": None}]) + def test_a_cell_is_a_class_of_its_own(self): h = self.te["H3Index"] self.assertEqual((h["in"], h["out"]), ("h3index_in", "h3index_out")) @@ -174,6 +190,9 @@ def test_the_classes_a_binding_reads(self): self.assertEqual(te["Set"]["readers"]["text"]["intset"], "intset_in") self.assertEqual(te["Temporal"]["writers"]["text"]["tfloat"], "tfloat_out") self.assertEqual(te["Raster"]["out"], "raster_as_hexwkb") + self.assertEqual(te["GSERIALIZED"]["readers"]["wkb"], + {"geometry": "geom_from_hexewkb", "geography": "geog_from_hexewkb"}) + self.assertEqual(te["GSERIALIZED"]["encoders"]["wkb"], "geo_as_hexewkb") for cell in ("H3Index", "Quadbin", "S2CellId"): self.assertIn(cell, te)