Files
PgTidy/pkg/format/dml.go
T
Hein 04711cf7b2
CI / Test (push) Successful in 28s
CI / Build (push) Successful in 25s
fix(cmd): handle errors and improve output formatting
* update error handling in various commands to use blank identifier
* enhance output formatting for better readability
* add golangci-lint to Makefile for linting checks
2026-07-01 12:53:25 +02:00

699 lines
16 KiB
Go

package format
import (
"strings"
"git.warky.dev/wdevs/pgtidy/pkg/config"
"git.warky.dev/wdevs/pgtidy/pkg/cst"
"git.warky.dev/wdevs/pgtidy/pkg/lexer"
)
// isDMLStart reports whether toks begins with a DML statement keyword.
func isDMLStart(toks []cst.Tok) bool {
if len(toks) == 0 || toks[0].Tok.Kind != lexer.Ident {
return false
}
switch lowerASCII(toks[0].Tok.Text) {
case "select", "insert", "update", "delete", "with":
return true
}
return false
}
// dmlSeg is one major clause of a DML statement.
type dmlSeg struct {
kw []cst.Tok // clause keyword tokens (possibly multi-word)
body []cst.Tok // remaining tokens up to the next clause boundary
}
// formatDML formats a top-level DML statement from its token slice, applying
// keyword casing, clause-per-line layout, and leading-comma column lists for
// SELECT and UPDATE SET clauses. Falls back to verbatim on comment-heavy input.
func formatDML(toks []cst.Tok, st config.Style) string {
// Strip the trailing semicolon so segments don't see it.
semi := ""
if n := len(toks); n > 0 && toks[n-1].Tok.Kind == lexer.Semicolon {
semi = ";"
toks = toks[:n-1]
}
segs := segmentDML(toks)
if len(segs) == 0 {
return verbatimSpan(toks) + semi
}
nl := st.Newline
var b strings.Builder
for i, seg := range segs {
if i > 0 {
b.WriteString(nl)
}
b.WriteString(dmlSegText(seg, st))
}
b.WriteString(semi)
return b.String()
}
// segmentDML splits toks into clause segments at depth-0 clause boundaries.
// Tokens inside parentheses (depth > 0) are never treated as clause starters,
// so subqueries and function calls are kept intact.
func segmentDML(toks []cst.Tok) []dmlSeg {
var segs []dmlSeg
depth := 0
segStart := 0
kwEnd := 0
started := false
flush := func(end int) {
if !started || end <= segStart {
return
}
segs = append(segs, dmlSeg{
kw: toks[segStart:kwEnd],
body: toks[kwEnd:end],
})
}
i := 0
for i < len(toks) {
t := toks[i]
switch t.Tok.Kind {
case lexer.LParen, lexer.LBracket:
depth++
i++
continue
case lexer.RParen, lexer.RBracket:
if depth > 0 {
depth--
}
i++
continue
}
if depth == 0 && t.Tok.Kind == lexer.Ident {
kw := lowerASCII(t.Tok.Text)
if dmlIsClauseKw(kw, toks, i) {
flush(i)
started = true
segStart = i
i = dmlConsumeKw(toks, i)
kwEnd = i
continue
}
}
i++
}
flush(len(toks))
return segs
}
// dmlIsClauseKw reports whether the keyword at toks[i] starts a new DML clause.
func dmlIsClauseKw(kw string, toks []cst.Tok, i int) bool {
switch kw {
case "select", "from", "where", "having", "limit", "offset",
"returning", "with", "into", "values", "set",
"union", "intersect", "except",
"insert", "update", "delete",
"join", "left", "right", "inner", "full", "cross", "natural":
return true
case "group", "order":
return i+1 < len(toks) && toks[i+1].Is("by")
case "on":
return i+1 < len(toks) && toks[i+1].Is("conflict")
}
return false
}
// dmlConsumeKw advances past multi-word clause keywords (e.g. GROUP BY,
// LEFT OUTER JOIN, INSERT INTO, DELETE FROM) and returns the new index.
func dmlConsumeKw(toks []cst.Tok, i int) int {
if i >= len(toks) {
return i
}
kw := lowerASCII(toks[i].Tok.Text)
i++
switch kw {
case "group", "order":
if i < len(toks) && toks[i].Is("by") {
i++
}
case "left", "right", "full":
if i < len(toks) && toks[i].Is("outer") {
i++
}
if i < len(toks) && toks[i].Is("join") {
i++
}
case "inner", "cross", "natural":
if i < len(toks) && toks[i].Is("join") {
i++
}
case "on":
if i < len(toks) && toks[i].Is("conflict") {
i++
}
case "delete":
// DELETE FROM — consume the FROM so it isn't treated as a separate clause.
if i < len(toks) && toks[i].Is("from") {
i++
}
case "insert":
// INSERT INTO — consume INTO.
if i < len(toks) && toks[i].Is("into") {
i++
}
}
return i
}
// dmlSegText formats one DML clause segment into a text line (or lines for
// column-list clauses and WITH bodies).
func dmlSegText(seg dmlSeg, st config.Style) string {
kwText := dmlInline(seg.kw, st)
kw := ""
if len(seg.kw) > 0 {
kw = lowerASCII(seg.kw[0].Tok.Text)
}
switch kw {
case "select", "returning":
items := dmlSplitCommas(seg.body)
return dmlColListSelect(kwText, items, st)
case "set":
items := dmlSplitCommas(seg.body)
return dmlColListSet(kwText, items, st)
case "where":
return dmlWhereClause(kwText, seg.body, st)
case "join", "left", "right", "inner", "full", "cross", "natural":
return dmlJoinClause(kwText, seg.body, st)
case "with":
return formatWithBody(kwText, seg.body, st)
default:
body := dmlInline(seg.body, st)
if body == "" {
return kwText
}
return kwText + " " + body
}
}
// dmlJoinClause formats a JOIN clause, applying indent_join when configured.
func dmlJoinClause(kwText string, body []cst.Tok, st config.Style) string {
text := dmlInline(body, st)
line := kwText
if text != "" {
line += " " + text
}
if !st.IndentJoin {
return line
}
indent := strings.Repeat(st.Indent, st.JoinIndentSize)
nl := st.Newline
var b strings.Builder
for i, part := range strings.Split(line, nl) {
if i > 0 {
b.WriteString(nl)
}
b.WriteString(indent)
b.WriteString(part)
}
return b.String()
}
// dmlWhereClause formats a WHERE clause, splitting AND/OR conditions per
// the where_wrap and where_and_or_indent settings.
func dmlWhereClause(kwText string, body []cst.Tok, st config.Style) string {
if st.WhereWrap == config.WrapNever {
text := dmlInline(body, st)
if text == "" {
return kwText
}
return kwText + " " + text
}
// Split at depth-0 AND/OR.
conditions := dmlSplitAndOr(body)
if len(conditions) <= 1 {
text := dmlInline(body, st)
if text == "" {
return kwText
}
return kwText + " " + text
}
nl := st.Newline
var b strings.Builder
b.WriteString(kwText)
for i, cond := range conditions {
b.WriteString(nl)
text := dmlInline(cond, st)
if st.WhereAndOrIndent {
b.WriteString(st.Indent)
}
if i == 0 {
// First condition: no leading AND/OR
b.WriteString(" ") // align with AND/OR token width
b.WriteString(text)
} else {
b.WriteString(text)
}
}
return b.String()
}
// dmlSplitAndOr splits toks at depth-0 AND/OR tokens, keeping the AND/OR with
// the following condition.
func dmlSplitAndOr(toks []cst.Tok) [][]cst.Tok {
var result [][]cst.Tok
depth := 0
start := 0
for i, t := range toks {
switch t.Tok.Kind {
case lexer.LParen, lexer.LBracket:
depth++
case lexer.RParen, lexer.RBracket:
if depth > 0 {
depth--
}
}
if depth == 0 && t.Tok.Kind == lexer.Ident {
low := lowerASCII(t.Tok.Text)
if (low == "and" || low == "or") && i > start {
result = append(result, toks[start:i])
start = i
}
}
}
result = append(result, toks[start:])
return result
}
// formatWithBody formats the body of a WITH clause by splitting CTE definitions
// at depth-0 commas and formatting the subquery inside each AS (...) block.
func formatWithBody(kwText string, body []cst.Tok, st config.Style) string {
nl := st.Newline
cteDefs := dmlSplitCommas(body)
// Filter spurious empty items.
var kept [][]cst.Tok
for _, d := range cteDefs {
if len(d) > 0 {
kept = append(kept, d)
}
}
cteDefs = kept
switch len(cteDefs) {
case 0:
return kwText
case 1:
return kwText + " " + formatCTEDef(cteDefs[0], st)
default:
// Multiple CTEs: one per line with the configured comma style.
first := st.Indent + " "
cont := st.Indent + ","
contPad := strings.Repeat(" ", len(cont)) // same width as cont, no comma
var b strings.Builder
b.WriteString(kwText)
for i, cteDef := range cteDefs {
b.WriteString(nl)
var headPfx, tailPfx string
if i == 0 || st.Commas != config.CommaLeading {
headPfx = first
tailPfx = first
} else {
headPfx = cont
tailPfx = contPad
}
cteText := formatCTEDef(cteDef, st)
cteLines := strings.Split(cteText, nl)
for j, line := range cteLines {
if j > 0 {
b.WriteString(nl)
b.WriteString(tailPfx)
} else {
b.WriteString(headPfx)
}
b.WriteString(line)
}
}
return b.String()
}
}
// formatCTEDef formats one CTE definition of the form:
//
// name [column_list] AS [NOT] [MATERIALIZED] (subquery)
//
// The subquery is formatted as DML, indented by st.Indent inside the parentheses.
// Falls back to dmlInline if the expected structure is not found.
func formatCTEDef(toks []cst.Tok, st config.Style) string {
nl := st.Newline
// Find the AS keyword at depth 0.
asIdx := dmlKeywordIdx(toks, 0, "as")
if asIdx < 0 {
return dmlInline(toks, st)
}
// Find the opening '(' after AS (may be preceded by NOT / MATERIALIZED).
parenOpen := -1
for i := asIdx + 1; i < len(toks); i++ {
if toks[i].Tok.Kind == lexer.LParen {
parenOpen = i
break
}
if toks[i].Tok.Kind != lexer.Ident {
// Unexpected token before '(' — fall back.
break
}
}
if parenOpen < 0 {
return dmlInline(toks, st)
}
// Find the matching ')'.
parenClose := dmlMatchParen(toks, parenOpen)
if parenClose < 0 {
return dmlInline(toks, st)
}
// Format the header (name, optional column list, AS, optional MATERIALIZED).
header := dmlInline(toks[:parenOpen], st)
// Format the subquery as DML.
subToks := toks[parenOpen+1 : parenClose]
subFormatted := strings.TrimRight(formatDML(subToks, st), nl)
if subFormatted == "" {
return header + " ()"
}
// Indent every non-empty line of the subquery by st.Indent.
indent := st.Indent
var indented strings.Builder
for i, line := range strings.Split(subFormatted, nl) {
if i > 0 {
indented.WriteString(nl)
}
if line != "" {
indented.WriteString(indent)
}
indented.WriteString(line)
}
return header + " (" + nl + indented.String() + nl + ")"
}
// dmlKeywordIdx returns the index of the first token equal to kw at paren depth 0,
// starting from `from`. Returns -1 if not found.
func dmlKeywordIdx(toks []cst.Tok, from int, kw string) int {
depth := 0
for i := from; i < len(toks); i++ {
switch toks[i].Tok.Kind {
case lexer.LParen, lexer.LBracket:
depth++
case lexer.RParen, lexer.RBracket:
if depth > 0 {
depth--
}
}
if depth == 0 && toks[i].Is(kw) {
return i
}
}
return -1
}
// dmlMatchParen returns the index of the ')' matching the '(' at toks[open].
// Returns -1 if no matching paren is found.
func dmlMatchParen(toks []cst.Tok, open int) int {
depth := 1
for i := open + 1; i < len(toks); i++ {
switch toks[i].Tok.Kind {
case lexer.LParen, lexer.LBracket:
depth++
case lexer.RParen:
depth--
if depth == 0 {
return i
}
case lexer.RBracket:
if depth > 0 {
depth--
}
}
}
return -1
}
// dmlInline renders toks on one line with keyword casing and proper spacing.
// If toks[1:] contains comment trivia the function falls back to verbatimSpan
// so no comment is lost.
func dmlInline(toks []cst.Tok, st config.Style) string {
if len(toks) == 0 {
return ""
}
if anyComment(toks[1:]) {
return verbatimSpan(toks)
}
var b strings.Builder
for i, t := range toks {
if i > 0 && needSpace(toks[i-1].Tok, t.Tok) {
b.WriteByte(' ')
}
// Space after comma in calls: func(a, b) vs func(a,b).
if st.SpaceAfterCommaInCalls && i > 0 && toks[i-1].Tok.Kind == lexer.Comma {
// Only inside parens (caller manages this at depth > 0, but we add space
// when the comma is not a clause-level comma — heuristic: always add).
b.WriteByte(' ')
}
var prev lexer.Token
if i > 0 {
prev = toks[i-1].Tok
}
nextIsLParen := i+1 < len(toks) && toks[i+1].Tok.Kind == lexer.LParen
b.WriteString(caseTextCtx(t.Tok, prev, nextIsLParen, st))
}
return b.String()
}
// dmlSplitCommas splits toks at depth-0 commas and returns the items between
// them (the comma tokens themselves are discarded).
func dmlSplitCommas(toks []cst.Tok) [][]cst.Tok {
var items [][]cst.Tok
depth := 0
start := 0
for i, t := range toks {
switch t.Tok.Kind {
case lexer.LParen, lexer.LBracket:
depth++
case lexer.RParen, lexer.RBracket:
if depth > 0 {
depth--
}
case lexer.Comma:
if depth == 0 {
items = append(items, toks[start:i])
start = i + 1
}
}
}
// Remaining tokens after the last comma (or all tokens if no comma found).
items = append(items, toks[start:])
return items
}
// dmlColListSelect formats a SELECT / RETURNING column list with optional
// align_columns and select_align_as settings.
func dmlColListSelect(kwText string, items [][]cst.Tok, st config.Style) string {
var kept [][]cst.Tok
for _, item := range items {
if len(item) > 0 {
kept = append(kept, item)
}
}
items = kept
nl := st.Newline
switch len(items) {
case 0:
return kwText
case 1:
body := dmlInline(items[0], st)
if body == "" {
return kwText
}
return kwText + " " + body
}
// Render each item text.
texts := make([]string, len(items))
for i, item := range items {
texts[i] = dmlInline(item, st)
}
// align_columns / select_align_as: pad expressions so AS and aliases align.
if (st.AlignColumns || st.SelectAlignAs) && len(texts) > 1 {
texts = alignSelectItems(texts, st)
}
first := st.Indent + " "
cont := st.Indent + ","
var b strings.Builder
b.WriteString(kwText)
for i, text := range texts {
b.WriteString(nl)
if i == 0 || st.Commas != config.CommaLeading {
b.WriteString(first)
b.WriteString(text)
if st.Commas == config.CommaTrailing && i < len(items)-1 {
b.WriteString(",")
}
} else {
b.WriteString(cont)
b.WriteString(text)
}
}
return b.String()
}
// dmlColListSet formats an UPDATE SET column list with optional set_align_equal.
func dmlColListSet(kwText string, items [][]cst.Tok, st config.Style) string {
var kept [][]cst.Tok
for _, item := range items {
if len(item) > 0 {
kept = append(kept, item)
}
}
items = kept
nl := st.Newline
switch len(items) {
case 0:
return kwText
case 1:
body := dmlInline(items[0], st)
if body == "" {
return kwText
}
return kwText + " " + body
}
texts := make([]string, len(items))
for i, item := range items {
texts[i] = dmlInline(item, st)
}
// set_align_equal: pad lhs so = signs align.
if st.SetAlignEqual && len(texts) > 1 {
texts = alignSetItems(texts)
}
first := st.Indent + " "
cont := st.Indent + ","
var b strings.Builder
b.WriteString(kwText)
for i, text := range texts {
b.WriteString(nl)
if i == 0 || st.Commas != config.CommaLeading {
b.WriteString(first)
b.WriteString(text)
if st.Commas == config.CommaTrailing && i < len(items)-1 {
b.WriteString(",")
}
} else {
b.WriteString(cont)
b.WriteString(text)
}
}
return b.String()
}
// alignSelectItems pads SELECT list item expressions so that AS keywords and
// alias names align vertically.
func alignSelectItems(texts []string, st config.Style) []string {
// Split each text into (expr, " AS ", alias) or keep as-is.
type part struct {
expr, alias string
hasAs bool
}
parts := make([]part, len(texts))
maxExpr := 0
for i, t := range texts {
// Find " AS " or " as " (case-insensitive).
if idx := findAsIndex(t); idx >= 0 {
parts[i] = part{expr: t[:idx], alias: t[idx:], hasAs: true}
if l := len(t[:idx]); l > maxExpr {
maxExpr = l
}
} else {
parts[i] = part{expr: t}
if st.AlignColumns {
if l := len(t); l > maxExpr {
maxExpr = l
}
}
}
}
out := make([]string, len(texts))
for i, p := range parts {
if !p.hasAs || maxExpr == 0 {
out[i] = texts[i]
continue
}
pad := strings.Repeat(" ", maxExpr-len(p.expr))
out[i] = p.expr + pad + p.alias
}
return out
}
// findAsIndex returns the byte index of " AS " (case-insensitive) in s,
// or -1 if not present at depth 0.
func findAsIndex(s string) int {
low := lowerASCII(s)
// Look for " as " boundary.
for i := 0; i < len(low)-3; i++ {
if low[i] == ' ' && low[i+1] == 'a' && low[i+2] == 's' && low[i+3] == ' ' {
return i + 1 // index of 'a'
}
}
return -1
}
// alignSetItems pads SET assignment lhs values so that = signs align.
func alignSetItems(texts []string) []string {
maxLhs := 0
lhsWidths := make([]int, len(texts))
for i, t := range texts {
idx := strings.Index(t, " = ")
if idx < 0 {
idx = strings.Index(t, "=")
}
if idx >= 0 {
lhsWidths[i] = idx
if idx > maxLhs {
maxLhs = idx
}
}
}
if maxLhs == 0 {
return texts
}
out := make([]string, len(texts))
for i, t := range texts {
if lhsWidths[i] == 0 || lhsWidths[i] == maxLhs {
out[i] = t
continue
}
idx := lhsWidths[i]
pad := strings.Repeat(" ", maxLhs-idx)
out[i] = t[:idx] + pad + t[idx:]
}
return out
}