* Add handling for pgvector and PostGIS extensions in migration scripts * Implement operator class and storage parameters for vector indexes * Update tests to validate new index behaviors and extension creation
460 lines
15 KiB
Go
460 lines
15 KiB
Go
package pgsql
|
|
|
|
import (
|
|
"sort"
|
|
"strings"
|
|
)
|
|
|
|
// Extension describes a PostgreSQL extension RelSpec recognizes, along with the schema
|
|
// artefacts that imply it: the types it provides (declared on TypeSpec.Extension), the
|
|
// index access methods and operator classes it installs, and the functions whose use in a
|
|
// default, check constraint, index predicate, or view body requires it.
|
|
type Extension struct {
|
|
Name string
|
|
Category string
|
|
Description string
|
|
|
|
// Requires lists extensions that must be created before this one.
|
|
Requires []string
|
|
|
|
// IndexMethods are access methods usable as Index.Type.
|
|
IndexMethods []string
|
|
|
|
// OperatorClasses are operator classes the extension installs.
|
|
OperatorClasses []string
|
|
|
|
// Functions are function names whose use implies the extension.
|
|
Functions []string
|
|
|
|
// FunctionPrefixes match whole families of functions (e.g. "st_" for PostGIS).
|
|
FunctionPrefixes []string
|
|
}
|
|
|
|
// postgresExtensions is the set of extensions RelSpec knows how to detect and emit.
|
|
var postgresExtensions = map[string]Extension{
|
|
"amcheck": {
|
|
Name: "amcheck", Category: "integrity",
|
|
Description: "Verifies B-tree and related structure consistency to help detect corruption.",
|
|
Functions: []string{"bt_index_check", "bt_index_parent_check", "verify_heapam"},
|
|
},
|
|
"btree_gin": {
|
|
Name: "btree_gin", Category: "indexing",
|
|
Description: "Adds GIN operator classes for common scalar data types.",
|
|
},
|
|
"btree_gist": {
|
|
Name: "btree_gist", Category: "indexing",
|
|
Description: "Adds GiST operator classes for common scalar data types and exclusion constraints.",
|
|
},
|
|
"citext": {
|
|
Name: "citext", Category: "text",
|
|
Description: "Provides case-insensitive text columns and operators.",
|
|
Functions: []string{"citext"},
|
|
},
|
|
"fuzzystrmatch": {
|
|
Name: "fuzzystrmatch", Category: "text",
|
|
Description: "Adds phonetic and fuzzy matching helpers like Soundex and Levenshtein.",
|
|
Functions: []string{
|
|
"soundex", "difference", "levenshtein", "levenshtein_less_equal",
|
|
"metaphone", "dmetaphone", "dmetaphone_alt",
|
|
},
|
|
},
|
|
"hstore": {
|
|
Name: "hstore", Category: "document",
|
|
Description: "Adds a lightweight key/value data type for semi-structured attributes.",
|
|
OperatorClasses: []string{"gin_hstore_ops", "gist_hstore_ops", "hash_hstore_ops", "btree_hstore_ops"},
|
|
Functions: []string{
|
|
"hstore", "akeys", "avals", "skeys", "svals",
|
|
"hstore_to_json", "hstore_to_jsonb", "hstore_to_array", "hstore_to_matrix",
|
|
},
|
|
},
|
|
"http": {
|
|
Name: "http", Category: "integration",
|
|
Description: "Lets SQL functions make outbound HTTP requests.",
|
|
Functions: []string{
|
|
"http", "http_get", "http_post", "http_put", "http_patch", "http_delete",
|
|
"http_head", "urlencode",
|
|
},
|
|
},
|
|
"pg_background": {
|
|
Name: "pg_background", Category: "jobs",
|
|
Description: "Runs SQL asynchronously in PostgreSQL background workers.",
|
|
Functions: []string{"pg_background_launch", "pg_background_result", "pg_background_detach"},
|
|
},
|
|
"pg_cron": {
|
|
Name: "pg_cron", Category: "scheduling",
|
|
Description: "Schedules recurring SQL jobs inside PostgreSQL.",
|
|
FunctionPrefixes: []string{"cron."},
|
|
},
|
|
"pg_jsonschema": {
|
|
Name: "pg_jsonschema", Category: "validation",
|
|
Description: "Validates json and jsonb values against JSON Schema.",
|
|
Functions: []string{"json_matches_schema", "jsonb_matches_schema", "jsonschema_is_valid"},
|
|
},
|
|
"pg_partman": {
|
|
Name: "pg_partman", Category: "partitioning",
|
|
Description: "Automates time-based and serial-based partition management.",
|
|
FunctionPrefixes: []string{"partman."},
|
|
},
|
|
"pg_qualstats": {
|
|
Name: "pg_qualstats", Category: "observability",
|
|
Description: "Tracks predicate usage in WHERE and JOIN clauses for tuning and index advice.",
|
|
},
|
|
"pg_repack": {
|
|
Name: "pg_repack", Category: "maintenance",
|
|
Description: "Rebuilds bloated tables and indexes online with minimal locking.",
|
|
},
|
|
"pg_search": {
|
|
Name: "pg_search", Category: "search",
|
|
Description: "Provides ParadeDB full-text and relevance search features.",
|
|
// bm25 is also the access method name used by pg_textsearch; pg_search is the
|
|
// canonical provider, so a bm25 index resolves to it.
|
|
IndexMethods: []string{"bm25"},
|
|
FunctionPrefixes: []string{"paradedb."},
|
|
},
|
|
"pg_stat_statements": {
|
|
Name: "pg_stat_statements", Category: "observability",
|
|
Description: "Tracks normalized query execution statistics.",
|
|
},
|
|
"pg_textsearch": {
|
|
Name: "pg_textsearch", Category: "search",
|
|
Description: "Adds BM25-style text search support.",
|
|
},
|
|
"pg_trgm": {
|
|
Name: "pg_trgm", Category: "text",
|
|
Description: "Adds trigram similarity search and fast fuzzy matching indexes.",
|
|
OperatorClasses: []string{"gin_trgm_ops", "gist_trgm_ops"},
|
|
Functions: []string{
|
|
"similarity", "word_similarity", "strict_word_similarity",
|
|
"show_trgm", "show_limit", "set_limit",
|
|
},
|
|
},
|
|
"pgcrypto": {
|
|
Name: "pgcrypto", Category: "security",
|
|
Description: "Adds hashing, encryption, random bytes, and UUID helpers.",
|
|
// gen_random_uuid is deliberately absent: it is built in since PostgreSQL 13.
|
|
Functions: []string{
|
|
"crypt", "gen_salt", "gen_random_bytes", "digest", "hmac",
|
|
"pgp_sym_encrypt", "pgp_sym_decrypt", "pgp_pub_encrypt", "pgp_pub_decrypt",
|
|
"armor", "dearmor",
|
|
},
|
|
},
|
|
"pgrouting": {
|
|
Name: "pgrouting", Category: "geospatial",
|
|
Description: "Adds routing and graph algorithms on top of PostGIS data.",
|
|
Requires: []string{"postgis"},
|
|
FunctionPrefixes: []string{"pgr_"},
|
|
},
|
|
"pgstattuple": {
|
|
Name: "pgstattuple", Category: "maintenance",
|
|
Description: "Reports table and index tuple density and bloat information.",
|
|
Functions: []string{"pgstattuple", "pgstatindex", "pgstatginindex", "pg_relpages"},
|
|
},
|
|
"plpython3u": {
|
|
Name: "plpython3u", Category: "procedural",
|
|
Description: "Lets you write PostgreSQL functions in Python 3.",
|
|
},
|
|
"postgis": {
|
|
Name: "postgis", Category: "geospatial",
|
|
Description: "Adds spatial data types, functions, and indexes.",
|
|
IndexMethods: nil, // uses the built-in gist/spgist/brin access methods
|
|
OperatorClasses: []string{
|
|
"gist_geometry_ops_2d", "gist_geometry_ops_nd", "gist_geography_ops",
|
|
"spgist_geometry_ops_2d", "spgist_geometry_ops_3d", "spgist_geometry_ops_nd",
|
|
"brin_geometry_inclusion_ops_2d", "brin_geometry_inclusion_ops_3d",
|
|
"brin_geometry_inclusion_ops_4d", "brin_geography_inclusion_ops_2d",
|
|
"btree_geometry_ops", "btree_geography_ops",
|
|
},
|
|
FunctionPrefixes: []string{"st_"},
|
|
Functions: []string{
|
|
"geometrytype", "addgeometrycolumn", "dropgeometrycolumn", "updategeometrysrid",
|
|
"find_srid", "postgis_version", "postgis_full_version",
|
|
},
|
|
},
|
|
"postgis_raster": {
|
|
Name: "postgis_raster", Category: "geospatial",
|
|
Description: "Adds the raster type and raster analysis functions.",
|
|
Requires: []string{"postgis"},
|
|
},
|
|
"postgis_topology": {
|
|
Name: "postgis_topology", Category: "geospatial",
|
|
Description: "Adds topology-aware spatial models and validation tools.",
|
|
Requires: []string{"postgis"},
|
|
FunctionPrefixes: []string{"topology."},
|
|
},
|
|
"postgres_fdw": {
|
|
Name: "postgres_fdw", Category: "federation",
|
|
Description: "Connects PostgreSQL tables to other PostgreSQL servers.",
|
|
},
|
|
"timescaledb": {
|
|
Name: "timescaledb", Category: "time-series",
|
|
Description: "Adds hypertables, compression, retention, and time-series optimizations.",
|
|
Functions: []string{
|
|
"create_hypertable", "add_dimension", "time_bucket", "time_bucket_gapfill",
|
|
"add_retention_policy", "add_compression_policy", "locf", "interpolate",
|
|
},
|
|
},
|
|
"unaccent": {
|
|
Name: "unaccent", Category: "text",
|
|
Description: "Removes accents and diacritics for normalized text search.",
|
|
Functions: []string{"unaccent"},
|
|
},
|
|
"uuid-ossp": {
|
|
Name: "uuid-ossp", Category: "utility",
|
|
Description: "Generates UUIDs using several algorithms.",
|
|
Functions: []string{
|
|
"uuid_generate_v1", "uuid_generate_v1mc", "uuid_generate_v3",
|
|
"uuid_generate_v4", "uuid_generate_v5",
|
|
"uuid_nil", "uuid_ns_dns", "uuid_ns_url", "uuid_ns_oid", "uuid_ns_x500",
|
|
},
|
|
},
|
|
"vector": {
|
|
Name: "vector", Category: "ai/search",
|
|
Description: "Adds vector data types and similarity search for embeddings.",
|
|
IndexMethods: []string{"hnsw", "ivfflat"},
|
|
OperatorClasses: []string{
|
|
"vector_l2_ops", "vector_ip_ops", "vector_cosine_ops", "vector_l1_ops",
|
|
"halfvec_l2_ops", "halfvec_ip_ops", "halfvec_cosine_ops", "halfvec_l1_ops",
|
|
"sparsevec_l2_ops", "sparsevec_ip_ops", "sparsevec_cosine_ops", "sparsevec_l1_ops",
|
|
"bit_hamming_ops", "bit_jaccard_ops",
|
|
},
|
|
Functions: []string{"l2_distance", "inner_product", "cosine_distance", "l1_distance", "vector_dims", "vector_norm"},
|
|
},
|
|
"vchord": {
|
|
Name: "vchord", Category: "ai/search",
|
|
Description: "Adds VectorChord scalable disk-friendly vector indexes compatible with pgvector data types.",
|
|
Requires: []string{"vector"},
|
|
IndexMethods: []string{"vchordrq", "vchordg"},
|
|
},
|
|
"ltree": {
|
|
Name: "ltree", Category: "document",
|
|
Description: "Adds a hierarchical label tree type.",
|
|
OperatorClasses: []string{"gist_ltree_ops", "gin_ltree_ops", "gist__ltree_ops"},
|
|
Functions: []string{"subltree", "subpath", "nlevel", "lca", "ltree2text", "text2ltree"},
|
|
},
|
|
}
|
|
|
|
// extensionIndexMethods maps an index access method to the extension providing it.
|
|
var extensionIndexMethods = buildExtensionIndex(func(ext Extension) []string { return ext.IndexMethods })
|
|
|
|
// extensionOperatorClasses maps an operator class to the extension providing it.
|
|
var extensionOperatorClasses = buildExtensionIndex(func(ext Extension) []string { return ext.OperatorClasses })
|
|
|
|
// extensionFunctions maps a function name to the extension providing it.
|
|
var extensionFunctions = buildExtensionIndex(func(ext Extension) []string { return ext.Functions })
|
|
|
|
// extensionFunctionPrefixes maps a function name prefix to the extension providing it.
|
|
var extensionFunctionPrefixes = buildExtensionIndex(func(ext Extension) []string { return ext.FunctionPrefixes })
|
|
|
|
func buildExtensionIndex(keys func(Extension) []string) map[string]string {
|
|
index := make(map[string]string)
|
|
for _, ext := range postgresExtensions {
|
|
for _, key := range keys(ext) {
|
|
// Deterministic on collision: the alphabetically first extension wins.
|
|
if existing, ok := index[key]; ok && existing < ext.Name {
|
|
continue
|
|
}
|
|
index[key] = ext.Name
|
|
}
|
|
}
|
|
return index
|
|
}
|
|
|
|
// LookupExtension returns the registered extension by name.
|
|
func LookupExtension(name string) (Extension, bool) {
|
|
ext, ok := postgresExtensions[strings.ToLower(strings.TrimSpace(name))]
|
|
return ext, ok
|
|
}
|
|
|
|
// IsKnownExtension reports whether the named extension is registered.
|
|
func IsKnownExtension(name string) bool {
|
|
_, ok := LookupExtension(name)
|
|
return ok
|
|
}
|
|
|
|
// GetExtensions returns every registered extension name, sorted.
|
|
func GetExtensions() []string {
|
|
names := make([]string, 0, len(postgresExtensions))
|
|
for name := range postgresExtensions {
|
|
names = append(names, name)
|
|
}
|
|
sort.Strings(names)
|
|
return names
|
|
}
|
|
|
|
// IndexMethodExtension returns the extension providing an index access method
|
|
// ("hnsw" -> "vector", "vchordrq" -> "vchord"). Built-in methods return "".
|
|
func IndexMethodExtension(method string) string {
|
|
return extensionIndexMethods[strings.ToLower(strings.TrimSpace(method))]
|
|
}
|
|
|
|
// OperatorClassExtension returns the extension providing an operator class
|
|
// ("gin_trgm_ops" -> "pg_trgm"). Built-in operator classes return "".
|
|
func OperatorClassExtension(opClass string) string {
|
|
return extensionOperatorClasses[strings.ToLower(strings.TrimSpace(opClass))]
|
|
}
|
|
|
|
// ExtensionsForExpression returns the extensions whose functions appear in a SQL
|
|
// expression such as a column default, check constraint, index predicate, or view body.
|
|
// The result is sorted and deduplicated.
|
|
func ExtensionsForExpression(expression string) []string {
|
|
if strings.TrimSpace(expression) == "" {
|
|
return nil
|
|
}
|
|
|
|
lower := strings.ToLower(expression)
|
|
found := make(map[string]bool)
|
|
|
|
for _, call := range sqlFunctionCalls(lower) {
|
|
if ext, ok := extensionFunctions[call]; ok {
|
|
found[ext] = true
|
|
continue
|
|
}
|
|
for prefix, ext := range extensionFunctionPrefixes {
|
|
if strings.HasPrefix(call, prefix) {
|
|
found[ext] = true
|
|
break
|
|
}
|
|
}
|
|
}
|
|
|
|
if len(found) == 0 {
|
|
return nil
|
|
}
|
|
|
|
names := make([]string, 0, len(found))
|
|
for name := range found {
|
|
names = append(names, name)
|
|
}
|
|
sort.Strings(names)
|
|
return names
|
|
}
|
|
|
|
// sqlFunctionCalls returns the lowercase names of every function call in an expression.
|
|
// A call is an identifier (optionally schema-qualified) immediately followed by "(".
|
|
func sqlFunctionCalls(lowerExpression string) []string {
|
|
calls := make([]string, 0, 4)
|
|
end := 0
|
|
|
|
for i := 0; i < len(lowerExpression); i++ {
|
|
if lowerExpression[i] != '(' {
|
|
continue
|
|
}
|
|
|
|
end = i
|
|
// Allow whitespace between the identifier and its opening parenthesis.
|
|
for end > 0 && isSQLSpace(lowerExpression[end-1]) {
|
|
end--
|
|
}
|
|
|
|
start := end
|
|
for start > 0 && isSQLIdentifierByte(lowerExpression[start-1]) {
|
|
start--
|
|
}
|
|
if start == end {
|
|
continue
|
|
}
|
|
// A leading digit means this is not an identifier (e.g. "2(").
|
|
if lowerExpression[start] >= '0' && lowerExpression[start] <= '9' {
|
|
continue
|
|
}
|
|
calls = append(calls, lowerExpression[start:end])
|
|
}
|
|
|
|
return calls
|
|
}
|
|
|
|
func isSQLIdentifierByte(b byte) bool {
|
|
switch {
|
|
case b >= 'a' && b <= 'z', b >= 'A' && b <= 'Z', b >= '0' && b <= '9':
|
|
return true
|
|
case b == '_', b == '.':
|
|
return true
|
|
default:
|
|
return false
|
|
}
|
|
}
|
|
|
|
func isSQLSpace(b byte) bool {
|
|
return b == ' ' || b == '\t' || b == '\n' || b == '\r'
|
|
}
|
|
|
|
// SortExtensions orders extension names so that dependencies come first (postgis before
|
|
// postgis_topology, vector before vchord), with alphabetical order breaking ties.
|
|
// Duplicates are removed; unknown names are kept and sorted alphabetically.
|
|
func SortExtensions(names []string) []string {
|
|
unique := make(map[string]bool, len(names))
|
|
for _, name := range names {
|
|
name = strings.ToLower(strings.TrimSpace(name))
|
|
if name != "" {
|
|
unique[name] = true
|
|
}
|
|
}
|
|
if len(unique) == 0 {
|
|
return nil
|
|
}
|
|
|
|
pending := make([]string, 0, len(unique))
|
|
for name := range unique {
|
|
pending = append(pending, name)
|
|
}
|
|
sort.Strings(pending)
|
|
|
|
sorted := make([]string, 0, len(pending))
|
|
emitted := make(map[string]bool, len(pending))
|
|
|
|
var emit func(name string, seen map[string]bool)
|
|
emit = func(name string, seen map[string]bool) {
|
|
if emitted[name] || seen[name] {
|
|
return
|
|
}
|
|
seen[name] = true
|
|
|
|
if ext, ok := LookupExtension(name); ok {
|
|
for _, dependency := range ext.Requires {
|
|
// Only order dependencies that are actually being created.
|
|
if unique[dependency] {
|
|
emit(dependency, seen)
|
|
}
|
|
}
|
|
}
|
|
|
|
emitted[name] = true
|
|
sorted = append(sorted, name)
|
|
}
|
|
|
|
for _, name := range pending {
|
|
emit(name, make(map[string]bool))
|
|
}
|
|
return sorted
|
|
}
|
|
|
|
// ExtensionDependencies returns the extensions a given extension requires, sorted.
|
|
func ExtensionDependencies(name string) []string {
|
|
ext, ok := LookupExtension(name)
|
|
if !ok || len(ext.Requires) == 0 {
|
|
return nil
|
|
}
|
|
requires := append([]string(nil), ext.Requires...)
|
|
sort.Strings(requires)
|
|
return requires
|
|
}
|
|
|
|
// QuoteExtensionName quotes an extension name when it is not a bare SQL identifier,
|
|
// e.g. uuid-ossp -> "uuid-ossp".
|
|
func QuoteExtensionName(name string) string {
|
|
name = strings.TrimSpace(name)
|
|
if name == "" {
|
|
return ""
|
|
}
|
|
for i := 0; i < len(name); i++ {
|
|
b := name[i]
|
|
switch {
|
|
case b >= 'a' && b <= 'z', b == '_':
|
|
case b >= '0' && b <= '9' && i > 0:
|
|
default:
|
|
return `"` + strings.ReplaceAll(name, `"`, `""`) + `"`
|
|
}
|
|
}
|
|
return name
|
|
}
|