mirror of
https://github.com/bitechdev/ResolveSpec.git
synced 2026-10-05 04:51:58 +00:00
fix(dbmanager): keep the pool alive across errors and restarts
Implements the fixes from audit/pkg/dbmanager.audit.md. - Stop closing the shared *sql.DB to recover from errors. Adapter factories and the health checker no longer call Reconnect; Reconnect is atomic and operator-only. - Postgres uses a custom driver.Connector: Reconnect retires pooled connections by generation without closing the pool, so held Bun/GORM handles keep working. Verified against a live server restart. - Add TCP keepalive, TCP_USER_TIMEOUT, a bounded reuse ping and statement_timeout as a runtime parameter; drop the 2 min timeout floor. - Health check pings without holding the connection lock. - Listener: single goroutine pair, bounded Close without UNLISTEN, and serialised use of the pgx connection (fixes conn busy and a close race). - Fix Connect/Close/Connect/Close panic, idempotent Connect, dial outside the manager lock, clean up on partial failure. - SQLite: pin :memory: to one connection, pragmas via DSN. - Escape credentials in Postgres/MSSQL/Mongo DSNs; sslmode defaults to prefer. Wire retry settings, publish metrics, fix logger calls. - NewConnectionFromDB: Close is a no-op with a warning (caller owns the pool); Reconnect only pings. - Document correct usage in the README; mark the audit with what was done. Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 5.5
parent
bc8bff7955
commit
da1af1487e
@@ -7,6 +7,8 @@ import (
|
||||
"sync"
|
||||
|
||||
"go.mongodb.org/mongo-driver/mongo"
|
||||
|
||||
"github.com/bitechdev/ResolveSpec/pkg/logger"
|
||||
)
|
||||
|
||||
// ExistingDBProvider wraps an existing *sql.DB connection
|
||||
@@ -44,16 +46,27 @@ func (p *ExistingDBProvider) Connect(ctx context.Context, cfg ConnectionConfig)
|
||||
return nil
|
||||
}
|
||||
|
||||
// Close closes the underlying database connection
|
||||
func (p *ExistingDBProvider) Close() error {
|
||||
p.mu.Lock()
|
||||
defer p.mu.Unlock()
|
||||
// Refresh verifies the wrapped database is still reachable. The pool belongs to
|
||||
// the caller and cannot be re-dialed here, so it is never closed to "reconnect".
|
||||
func (p *ExistingDBProvider) Refresh(ctx context.Context) error {
|
||||
p.mu.RLock()
|
||||
defer p.mu.RUnlock()
|
||||
|
||||
if p.db == nil {
|
||||
return nil
|
||||
return fmt.Errorf("database connection is nil")
|
||||
}
|
||||
return p.db.PingContext(ctx)
|
||||
}
|
||||
|
||||
return p.db.Close()
|
||||
// OwnsDB reports whether Close releases the wrapped database. It never does:
|
||||
// the *sql.DB was opened by the caller, who is responsible for closing it.
|
||||
func (p *ExistingDBProvider) OwnsDB() bool { return false }
|
||||
|
||||
// Close is a no-op for the wrapped database. The pool belongs to the caller, so
|
||||
// closing it here would break the caller's other users of it.
|
||||
func (p *ExistingDBProvider) Close() error {
|
||||
logger.Warn("Not closing externally provided database: name=%s; the caller owns this *sql.DB and must close it", p.name)
|
||||
return nil
|
||||
}
|
||||
|
||||
// HealthCheck verifies the connection is alive
|
||||
|
||||
@@ -164,7 +164,7 @@ func TestExistingDBProvider_Stats(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestExistingDBProvider_Close(t *testing.T) {
|
||||
func TestExistingDBProvider_Close_LeavesDBOpen(t *testing.T) {
|
||||
db, err := sql.Open("sqlite3", ":memory:")
|
||||
if err != nil {
|
||||
t.Fatalf("Failed to open database: %v", err)
|
||||
@@ -177,10 +177,10 @@ func TestExistingDBProvider_Close(t *testing.T) {
|
||||
t.Errorf("Expected Close to succeed, got error: %v", err)
|
||||
}
|
||||
|
||||
// Verify the database is closed
|
||||
err = db.Ping()
|
||||
if err == nil {
|
||||
t.Error("Expected database to be closed")
|
||||
// The caller owns the database, so Close must leave it open
|
||||
defer db.Close()
|
||||
if err := db.Ping(); err != nil {
|
||||
t.Errorf("Expected caller's database to stay open, got: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -41,9 +41,10 @@ func (p *MongoProvider) Connect(ctx context.Context, cfg ConnectionConfig) error
|
||||
clientOpts.SetMaxPoolSize(maxPoolSize)
|
||||
}
|
||||
|
||||
if cfg.GetMaxIdleConns() != nil {
|
||||
minPoolSize := uint64(*cfg.GetMaxIdleConns())
|
||||
clientOpts.SetMinPoolSize(minPoolSize)
|
||||
// MaxIdleConns is a ceiling on idle connections, not a pre-warmed minimum
|
||||
// (MinPoolSize), so only the idle-time limit maps onto the Mongo pool.
|
||||
if cfg.GetConnMaxIdleTime() != nil {
|
||||
clientOpts.SetMaxConnIdleTime(*cfg.GetConnMaxIdleTime())
|
||||
}
|
||||
|
||||
// Set timeouts
|
||||
@@ -65,12 +66,11 @@ func (p *MongoProvider) Connect(ctx context.Context, cfg ConnectionConfig) error
|
||||
var client *mongo.Client
|
||||
var lastErr error
|
||||
|
||||
retryAttempts := 3
|
||||
retryDelay := 1 * time.Second
|
||||
retryAttempts, retryDelay, retryMaxDelay := retryPolicy(cfg)
|
||||
|
||||
for attempt := 0; attempt < retryAttempts; attempt++ {
|
||||
if attempt > 0 {
|
||||
delay := calculateBackoff(attempt, retryDelay, 10*time.Second)
|
||||
delay := calculateBackoff(attempt, retryDelay, retryMaxDelay)
|
||||
if cfg.GetEnableLogging() {
|
||||
logger.Info("Retrying MongoDB connection: attempt=%d/%d, delay=%v", attempt+1, retryAttempts, delay)
|
||||
}
|
||||
@@ -87,7 +87,7 @@ func (p *MongoProvider) Connect(ctx context.Context, cfg ConnectionConfig) error
|
||||
if err != nil {
|
||||
lastErr = err
|
||||
if cfg.GetEnableLogging() {
|
||||
logger.Warn("Failed to connect to MongoDB", "error", err)
|
||||
logger.Warn("Failed to connect to MongoDB: %v", err)
|
||||
}
|
||||
continue
|
||||
}
|
||||
@@ -101,7 +101,7 @@ func (p *MongoProvider) Connect(ctx context.Context, cfg ConnectionConfig) error
|
||||
lastErr = err
|
||||
_ = client.Disconnect(ctx)
|
||||
if cfg.GetEnableLogging() {
|
||||
logger.Warn("Failed to ping MongoDB", "error", err)
|
||||
logger.Warn("Failed to ping MongoDB: %v", err)
|
||||
}
|
||||
continue
|
||||
}
|
||||
|
||||
@@ -35,12 +35,11 @@ func (p *MSSQLProvider) Connect(ctx context.Context, cfg ConnectionConfig) error
|
||||
var db *sql.DB
|
||||
var lastErr error
|
||||
|
||||
retryAttempts := 3 // Default retry attempts
|
||||
retryDelay := 1 * time.Second
|
||||
retryAttempts, retryDelay, retryMaxDelay := retryPolicy(cfg)
|
||||
|
||||
for attempt := 0; attempt < retryAttempts; attempt++ {
|
||||
if attempt > 0 {
|
||||
delay := calculateBackoff(attempt, retryDelay, 10*time.Second)
|
||||
delay := calculateBackoff(attempt, retryDelay, retryMaxDelay)
|
||||
if cfg.GetEnableLogging() {
|
||||
logger.Info("Retrying MSSQL connection: attempt=%d/%d, delay=%v", attempt+1, retryAttempts, delay)
|
||||
}
|
||||
@@ -57,7 +56,7 @@ func (p *MSSQLProvider) Connect(ctx context.Context, cfg ConnectionConfig) error
|
||||
if err != nil {
|
||||
lastErr = err
|
||||
if cfg.GetEnableLogging() {
|
||||
logger.Warn("Failed to open MSSQL connection", "error", err)
|
||||
logger.Warn("Failed to open MSSQL connection: %v", err)
|
||||
}
|
||||
continue
|
||||
}
|
||||
@@ -71,7 +70,7 @@ func (p *MSSQLProvider) Connect(ctx context.Context, cfg ConnectionConfig) error
|
||||
lastErr = err
|
||||
db.Close()
|
||||
if cfg.GetEnableLogging() {
|
||||
logger.Warn("Failed to ping MSSQL database", "error", err)
|
||||
logger.Warn("Failed to ping MSSQL database: %v", err)
|
||||
}
|
||||
continue
|
||||
}
|
||||
|
||||
@@ -0,0 +1,132 @@
|
||||
package providers
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql/driver"
|
||||
"fmt"
|
||||
"net"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
"github.com/jackc/pgx/v5"
|
||||
"github.com/jackc/pgx/v5/stdlib"
|
||||
)
|
||||
|
||||
const (
|
||||
// tcpKeepAlive is how often keepalive probes are sent on idle connections.
|
||||
tcpKeepAlive = 30 * time.Second
|
||||
// tcpUserTimeout bounds how long written data may stay unacknowledged before
|
||||
// the kernel drops the socket. Without it a query on a silently dead peer
|
||||
// waits for tcp_retries2 (about 15 minutes).
|
||||
tcpUserTimeout = 30 * time.Second
|
||||
// resetSessionTimeout bounds the liveness ping database/sql triggers when a
|
||||
// pooled connection is reused, which otherwise runs on the request context.
|
||||
resetSessionTimeout = 5 * time.Second
|
||||
)
|
||||
|
||||
// pgConnector is a driver.Connector whose connections can be retired without
|
||||
// closing the *sql.DB. Reconnecting bumps a generation; connections created
|
||||
// under an older generation report themselves invalid and database/sql
|
||||
// discards them and dials new ones. Every handle wrapping the *sql.DB keeps
|
||||
// working across a reconnect.
|
||||
type pgConnector struct {
|
||||
inner atomic.Pointer[connectorState]
|
||||
generation atomic.Uint64
|
||||
}
|
||||
|
||||
type connectorState struct {
|
||||
connector driver.Connector
|
||||
gen uint64
|
||||
}
|
||||
|
||||
func newPGConnector(cfg *pgx.ConnConfig) *pgConnector {
|
||||
c := &pgConnector{}
|
||||
c.swap(cfg)
|
||||
return c
|
||||
}
|
||||
|
||||
// swap installs a new connection config under a fresh generation.
|
||||
func (c *pgConnector) swap(cfg *pgx.ConnConfig) {
|
||||
gen := c.generation.Add(1)
|
||||
c.inner.Store(&connectorState{connector: stdlib.GetConnector(*cfg), gen: gen})
|
||||
}
|
||||
|
||||
func (c *pgConnector) Connect(ctx context.Context) (driver.Conn, error) {
|
||||
st := c.inner.Load()
|
||||
conn, err := st.connector.Connect(ctx)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
sc, ok := conn.(*stdlib.Conn)
|
||||
if !ok {
|
||||
conn.Close()
|
||||
return nil, fmt.Errorf("unexpected pgx driver connection type %T", conn)
|
||||
}
|
||||
return &pgConn{Conn: sc, owner: c, gen: st.gen}, nil
|
||||
}
|
||||
|
||||
func (c *pgConnector) Driver() driver.Driver {
|
||||
return stdlib.GetDefaultDriver()
|
||||
}
|
||||
|
||||
// pgConn embeds *stdlib.Conn, so every optional driver interface (context
|
||||
// queries, Pinger, NamedValueChecker, ...) is promoted unchanged.
|
||||
type pgConn struct {
|
||||
*stdlib.Conn
|
||||
owner *pgConnector
|
||||
gen uint64
|
||||
}
|
||||
|
||||
func (c *pgConn) stale() bool { return c.gen != c.owner.generation.Load() }
|
||||
|
||||
// IsValid implements driver.Validator: stale or closed connections are dropped
|
||||
// when returned to the pool.
|
||||
func (c *pgConn) IsValid() bool {
|
||||
return !c.stale() && !c.Conn.Conn().IsClosed()
|
||||
}
|
||||
|
||||
// ResetSession runs when a pooled connection is reused. It discards stale
|
||||
// connections and bounds pgx's liveness ping so a dead socket fails in seconds
|
||||
// rather than blocking on the caller's context.
|
||||
func (c *pgConn) ResetSession(ctx context.Context) error {
|
||||
if c.stale() {
|
||||
return driver.ErrBadConn
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(ctx, resetSessionTimeout)
|
||||
defer cancel()
|
||||
return c.Conn.ResetSession(ctx)
|
||||
}
|
||||
|
||||
// newDialFunc returns a pgconn dial function with TCP keepalive and, where the
|
||||
// platform supports it, TCP_USER_TIMEOUT.
|
||||
func newDialFunc(connectTimeout time.Duration) func(ctx context.Context, network, addr string) (net.Conn, error) {
|
||||
d := &net.Dialer{
|
||||
Timeout: connectTimeout,
|
||||
KeepAlive: tcpKeepAlive,
|
||||
Control: setTCPUserTimeout(tcpUserTimeout),
|
||||
}
|
||||
return d.DialContext
|
||||
}
|
||||
|
||||
// buildPGXConfig parses the DSN and applies client-side hardening: bounded
|
||||
// dialing, TCP timeouts, and statement_timeout, which is set as a runtime
|
||||
// parameter so it also applies to caller-supplied DSNs.
|
||||
func buildPGXConfig(cfg ConnectionConfig) (*pgx.ConnConfig, error) {
|
||||
dsn, err := cfg.BuildDSN()
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to build DSN: %w", err)
|
||||
}
|
||||
|
||||
cc, err := pgx.ParseConfig(dsn)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to parse connection config: %w", err)
|
||||
}
|
||||
|
||||
cc.DialFunc = newDialFunc(cfg.GetConnectTimeout())
|
||||
if cfg.GetQueryTimeout() > 0 {
|
||||
if _, set := cc.RuntimeParams["statement_timeout"]; !set {
|
||||
cc.RuntimeParams["statement_timeout"] = fmt.Sprintf("%d", cfg.GetQueryTimeout().Milliseconds())
|
||||
}
|
||||
}
|
||||
return cc, nil
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
package providers
|
||||
|
||||
import (
|
||||
"database/sql/driver"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/jackc/pgx/v5"
|
||||
)
|
||||
|
||||
func TestConnectorGenerationInvalidatesConns(t *testing.T) {
|
||||
cfg, err := pgx.ParseConfig("postgres://u:p@127.0.0.1:1/db")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
c := newPGConnector(cfg)
|
||||
conn := &pgConn{owner: c, gen: c.generation.Load()}
|
||||
if conn.stale() {
|
||||
t.Fatal("fresh connection reported stale")
|
||||
}
|
||||
c.swap(cfg)
|
||||
if !conn.stale() {
|
||||
t.Fatal("connection from an older generation must be stale")
|
||||
}
|
||||
if err := conn.ResetSession(t.Context()); err != driver.ErrBadConn {
|
||||
t.Fatalf("ResetSession on stale conn = %v, want ErrBadConn", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDialFuncHasTimeout(t *testing.T) {
|
||||
if newDialFunc(2*time.Second) == nil {
|
||||
t.Fatal("nil dial func")
|
||||
}
|
||||
}
|
||||
@@ -3,12 +3,12 @@ package providers
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"errors"
|
||||
"fmt"
|
||||
"math"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
_ "github.com/jackc/pgx/v5/stdlib" // PostgreSQL driver
|
||||
"go.mongodb.org/mongo-driver/mongo"
|
||||
|
||||
"github.com/bitechdev/ResolveSpec/pkg/logger"
|
||||
@@ -16,10 +16,11 @@ import (
|
||||
|
||||
// PostgresProvider implements Provider for PostgreSQL databases
|
||||
type PostgresProvider struct {
|
||||
db *sql.DB
|
||||
config ConnectionConfig
|
||||
listener *PostgresListener
|
||||
mu sync.Mutex
|
||||
db *sql.DB
|
||||
connector *pgConnector
|
||||
config ConnectionConfig
|
||||
listener *PostgresListener
|
||||
mu sync.Mutex
|
||||
}
|
||||
|
||||
// NewPostgresProvider creates a new PostgreSQL provider
|
||||
@@ -29,22 +30,24 @@ func NewPostgresProvider() *PostgresProvider {
|
||||
|
||||
// Connect establishes a PostgreSQL connection
|
||||
func (p *PostgresProvider) Connect(ctx context.Context, cfg ConnectionConfig) error {
|
||||
// Build DSN
|
||||
dsn, err := cfg.BuildDSN()
|
||||
connCfg, err := buildPGXConfig(cfg)
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to build DSN: %w", err)
|
||||
return err
|
||||
}
|
||||
|
||||
// The connector and *sql.DB are created once; the pool is never closed to
|
||||
// recover from errors (see Refresh).
|
||||
connector := newPGConnector(connCfg)
|
||||
db := sql.OpenDB(connector)
|
||||
|
||||
// Connect with retry logic
|
||||
var db *sql.DB
|
||||
var lastErr error
|
||||
retryAttempts, retryDelay, retryMaxDelay := retryPolicy(cfg)
|
||||
|
||||
retryAttempts := 3 // Default retry attempts
|
||||
retryDelay := 1 * time.Second
|
||||
|
||||
connected := false
|
||||
for attempt := 0; attempt < retryAttempts; attempt++ {
|
||||
if attempt > 0 {
|
||||
delay := calculateBackoff(attempt, retryDelay, 10*time.Second)
|
||||
delay := calculateBackoff(attempt, retryDelay, retryMaxDelay)
|
||||
if cfg.GetEnableLogging() {
|
||||
logger.Info("Retrying PostgreSQL connection: attempt=%d/%d, delay=%v", attempt+1, retryAttempts, delay)
|
||||
}
|
||||
@@ -52,20 +55,11 @@ func (p *PostgresProvider) Connect(ctx context.Context, cfg ConnectionConfig) er
|
||||
select {
|
||||
case <-time.After(delay):
|
||||
case <-ctx.Done():
|
||||
db.Close()
|
||||
return ctx.Err()
|
||||
}
|
||||
}
|
||||
|
||||
// Open database connection
|
||||
db, err = sql.Open("pgx", dsn)
|
||||
if err != nil {
|
||||
lastErr = err
|
||||
if cfg.GetEnableLogging() {
|
||||
logger.Warn("Failed to open PostgreSQL connection", "error", err)
|
||||
}
|
||||
continue
|
||||
}
|
||||
|
||||
// Test the connection with context timeout
|
||||
connectCtx, cancel := context.WithTimeout(ctx, cfg.GetConnectTimeout())
|
||||
err = db.PingContext(connectCtx)
|
||||
@@ -73,18 +67,18 @@ func (p *PostgresProvider) Connect(ctx context.Context, cfg ConnectionConfig) er
|
||||
|
||||
if err != nil {
|
||||
lastErr = err
|
||||
db.Close()
|
||||
if cfg.GetEnableLogging() {
|
||||
logger.Warn("Failed to ping PostgreSQL database", "error", err)
|
||||
logger.Warn("Failed to ping PostgreSQL database: %v", err)
|
||||
}
|
||||
continue
|
||||
}
|
||||
|
||||
// Connection successful
|
||||
connected = true
|
||||
break
|
||||
}
|
||||
|
||||
if err != nil {
|
||||
if !connected {
|
||||
db.Close()
|
||||
return fmt.Errorf("failed to connect after %d attempts: %w", retryAttempts, lastErr)
|
||||
}
|
||||
|
||||
@@ -103,6 +97,7 @@ func (p *PostgresProvider) Connect(ctx context.Context, cfg ConnectionConfig) er
|
||||
}
|
||||
|
||||
p.db = db
|
||||
p.connector = connector
|
||||
p.config = cfg
|
||||
|
||||
if cfg.GetEnableLogging() {
|
||||
@@ -112,34 +107,55 @@ func (p *PostgresProvider) Connect(ctx context.Context, cfg ConnectionConfig) er
|
||||
return nil
|
||||
}
|
||||
|
||||
// Close closes the PostgreSQL connection
|
||||
func (p *PostgresProvider) Close() error {
|
||||
// Close listener if it exists
|
||||
p.mu.Lock()
|
||||
if p.listener != nil {
|
||||
if err := p.listener.Close(); err != nil {
|
||||
p.mu.Unlock()
|
||||
return fmt.Errorf("failed to close listener: %w", err)
|
||||
}
|
||||
p.listener = nil
|
||||
// Refresh retires every pooled connection and dials fresh ones on demand,
|
||||
// without closing the *sql.DB. Handles already handed out keep working:
|
||||
// connections in use finish their current query and are then discarded.
|
||||
func (p *PostgresProvider) Refresh(ctx context.Context) error {
|
||||
if p.db == nil || p.connector == nil {
|
||||
return fmt.Errorf("database connection is not initialized")
|
||||
}
|
||||
|
||||
connCfg, err := buildPGXConfig(p.config)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
p.connector.swap(connCfg)
|
||||
|
||||
pingCtx, cancel := context.WithTimeout(ctx, p.config.GetConnectTimeout())
|
||||
defer cancel()
|
||||
if err := p.db.PingContext(pingCtx); err != nil {
|
||||
return fmt.Errorf("failed to ping after refresh: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Close closes the PostgreSQL connection. A listener failure does not stop the
|
||||
// pool from being closed.
|
||||
func (p *PostgresProvider) Close() error {
|
||||
var errs []error
|
||||
|
||||
p.mu.Lock()
|
||||
listener := p.listener
|
||||
p.listener = nil
|
||||
p.mu.Unlock()
|
||||
|
||||
if p.db == nil {
|
||||
return nil
|
||||
if listener != nil {
|
||||
if err := listener.Close(); err != nil {
|
||||
errs = append(errs, fmt.Errorf("failed to close listener: %w", err))
|
||||
}
|
||||
}
|
||||
|
||||
err := p.db.Close()
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to close PostgreSQL connection: %w", err)
|
||||
if p.db != nil {
|
||||
if err := p.db.Close(); err != nil {
|
||||
errs = append(errs, fmt.Errorf("failed to close PostgreSQL connection: %w", err))
|
||||
} else if p.config.GetEnableLogging() {
|
||||
logger.Info("PostgreSQL connection closed: name=%s", p.config.GetName())
|
||||
}
|
||||
p.db = nil
|
||||
p.connector = nil
|
||||
}
|
||||
|
||||
if p.config.GetEnableLogging() {
|
||||
logger.Info("PostgreSQL connection closed: name=%s", p.config.GetName())
|
||||
}
|
||||
|
||||
p.db = nil
|
||||
return nil
|
||||
return errors.Join(errs...)
|
||||
}
|
||||
|
||||
// HealthCheck verifies the PostgreSQL connection is alive
|
||||
|
||||
@@ -23,6 +23,10 @@ type PostgresListener struct {
|
||||
// Channel subscriptions
|
||||
channels map[string]NotificationHandler
|
||||
mu sync.RWMutex
|
||||
// connMu serialises use of the single pgx.Conn: it is not safe for
|
||||
// concurrent use, so the notification wait and LISTEN/UNLISTEN/NOTIFY take
|
||||
// turns. Lock order: connMu before mu.
|
||||
connMu sync.Mutex
|
||||
|
||||
// Lifecycle management
|
||||
ctx context.Context
|
||||
@@ -30,6 +34,7 @@ type PostgresListener struct {
|
||||
closed bool
|
||||
closeMu sync.Mutex
|
||||
reconnectC chan struct{}
|
||||
startOnce sync.Once // background goroutines start exactly once
|
||||
}
|
||||
|
||||
// NewPostgresListener creates a new PostgreSQL listener
|
||||
@@ -44,76 +49,20 @@ func NewPostgresListener(cfg ConnectionConfig) *PostgresListener {
|
||||
}
|
||||
}
|
||||
|
||||
// Connect establishes a dedicated connection for listening
|
||||
// Connect establishes a dedicated connection for listening and starts the
|
||||
// background loops (once per listener).
|
||||
func (l *PostgresListener) Connect(ctx context.Context) error {
|
||||
dsn, err := l.config.BuildDSN()
|
||||
conn, err := l.dial(ctx)
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to build DSN: %w", err)
|
||||
return err
|
||||
}
|
||||
|
||||
// Parse connection config
|
||||
connConfig, err := pgx.ParseConfig(dsn)
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to parse connection config: %w", err)
|
||||
}
|
||||
l.swapConn(conn)
|
||||
|
||||
// Connect with retry logic
|
||||
var conn *pgx.Conn
|
||||
var lastErr error
|
||||
|
||||
retryAttempts := 3
|
||||
retryDelay := 1 * time.Second
|
||||
|
||||
for attempt := 0; attempt < retryAttempts; attempt++ {
|
||||
if attempt > 0 {
|
||||
delay := calculateBackoff(attempt, retryDelay, 10*time.Second)
|
||||
if l.config.GetEnableLogging() {
|
||||
logger.Info("Retrying PostgreSQL listener connection: attempt=%d/%d, delay=%v", attempt+1, retryAttempts, delay)
|
||||
}
|
||||
|
||||
select {
|
||||
case <-time.After(delay):
|
||||
case <-ctx.Done():
|
||||
return ctx.Err()
|
||||
}
|
||||
}
|
||||
|
||||
conn, err = pgx.ConnectConfig(ctx, connConfig)
|
||||
if err != nil {
|
||||
lastErr = err
|
||||
if l.config.GetEnableLogging() {
|
||||
logger.Warn("Failed to connect PostgreSQL listener", "error", err)
|
||||
}
|
||||
continue
|
||||
}
|
||||
|
||||
// Test the connection
|
||||
if err = conn.Ping(ctx); err != nil {
|
||||
lastErr = err
|
||||
conn.Close(ctx)
|
||||
if l.config.GetEnableLogging() {
|
||||
logger.Warn("Failed to ping PostgreSQL listener", "error", err)
|
||||
}
|
||||
continue
|
||||
}
|
||||
|
||||
// Connection successful
|
||||
break
|
||||
}
|
||||
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to connect listener after %d attempts: %w", retryAttempts, lastErr)
|
||||
}
|
||||
|
||||
l.mu.Lock()
|
||||
l.conn = conn
|
||||
l.mu.Unlock()
|
||||
|
||||
// Start notification handler
|
||||
go l.handleNotifications()
|
||||
|
||||
// Start reconnection handler
|
||||
go l.handleReconnection()
|
||||
l.startOnce.Do(func() {
|
||||
go l.handleNotifications()
|
||||
go l.handleReconnection()
|
||||
})
|
||||
|
||||
if l.config.GetEnableLogging() {
|
||||
logger.Info("PostgreSQL listener connected: name=%s", l.config.GetName())
|
||||
@@ -122,30 +71,105 @@ func (l *PostgresListener) Connect(ctx context.Context) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// dial opens and verifies a new dedicated connection, with retries.
|
||||
func (l *PostgresListener) dial(ctx context.Context) (*pgx.Conn, error) {
|
||||
connConfig, err := buildPGXConfig(l.config)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
var lastErr error
|
||||
|
||||
retryAttempts, retryDelay, retryMaxDelay := retryPolicy(l.config)
|
||||
|
||||
for attempt := 0; attempt < retryAttempts; attempt++ {
|
||||
if attempt > 0 {
|
||||
delay := calculateBackoff(attempt, retryDelay, retryMaxDelay)
|
||||
if l.config.GetEnableLogging() {
|
||||
logger.Info("Retrying PostgreSQL listener connection: attempt=%d/%d, delay=%v", attempt+1, retryAttempts, delay)
|
||||
}
|
||||
|
||||
select {
|
||||
case <-time.After(delay):
|
||||
case <-ctx.Done():
|
||||
return nil, ctx.Err()
|
||||
}
|
||||
}
|
||||
|
||||
conn, err := pgx.ConnectConfig(ctx, connConfig)
|
||||
if err != nil {
|
||||
lastErr = err
|
||||
if l.config.GetEnableLogging() {
|
||||
logger.Warn("Failed to connect PostgreSQL listener: %v", err)
|
||||
}
|
||||
continue
|
||||
}
|
||||
|
||||
// Test the connection
|
||||
if err = conn.Ping(ctx); err != nil {
|
||||
lastErr = err
|
||||
closeConnBounded(conn)
|
||||
if l.config.GetEnableLogging() {
|
||||
logger.Warn("Failed to ping PostgreSQL listener: %v", err)
|
||||
}
|
||||
continue
|
||||
}
|
||||
|
||||
return conn, nil
|
||||
}
|
||||
|
||||
return nil, fmt.Errorf("failed to connect listener after %d attempts: %w", retryAttempts, lastErr)
|
||||
}
|
||||
|
||||
// closeConnBounded closes a pgx connection without ever waiting on a dead socket.
|
||||
func closeConnBounded(conn *pgx.Conn) error {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), listenerCloseTimeout)
|
||||
defer cancel()
|
||||
return conn.Close(ctx)
|
||||
}
|
||||
|
||||
const (
|
||||
listenerCloseTimeout = 2 * time.Second
|
||||
notificationPollInterval = 500 * time.Millisecond
|
||||
)
|
||||
|
||||
// currentConn returns the live connection, or an error if the listener is
|
||||
// closed or not yet connected.
|
||||
func (l *PostgresListener) currentConn() (*pgx.Conn, error) {
|
||||
l.closeMu.Lock()
|
||||
closed := l.closed
|
||||
l.closeMu.Unlock()
|
||||
if closed {
|
||||
return nil, fmt.Errorf("listener is closed")
|
||||
}
|
||||
|
||||
l.mu.RLock()
|
||||
conn := l.conn
|
||||
l.mu.RUnlock()
|
||||
if conn == nil {
|
||||
return nil, fmt.Errorf("listener connection is not initialized")
|
||||
}
|
||||
return conn, nil
|
||||
}
|
||||
|
||||
// Listen subscribes to a PostgreSQL notification channel
|
||||
func (l *PostgresListener) Listen(channel string, handler NotificationHandler) error {
|
||||
l.closeMu.Lock()
|
||||
if l.closed {
|
||||
l.closeMu.Unlock()
|
||||
return fmt.Errorf("listener is closed")
|
||||
}
|
||||
l.closeMu.Unlock()
|
||||
// Take the connection between notification waits (each wait is short).
|
||||
l.connMu.Lock()
|
||||
defer l.connMu.Unlock()
|
||||
|
||||
l.mu.Lock()
|
||||
defer l.mu.Unlock()
|
||||
|
||||
if l.conn == nil {
|
||||
return fmt.Errorf("listener connection is not initialized")
|
||||
}
|
||||
|
||||
// Execute LISTEN command
|
||||
_, err := l.conn.Exec(l.ctx, fmt.Sprintf("LISTEN %s", pgx.Identifier{channel}.Sanitize()))
|
||||
conn, err := l.currentConn()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
if _, err := conn.Exec(l.ctx, fmt.Sprintf("LISTEN %s", pgx.Identifier{channel}.Sanitize())); err != nil {
|
||||
return fmt.Errorf("failed to listen on channel %s: %w", channel, err)
|
||||
}
|
||||
|
||||
// Store the handler
|
||||
l.mu.Lock()
|
||||
l.channels[channel] = handler
|
||||
l.mu.Unlock()
|
||||
|
||||
if l.config.GetEnableLogging() {
|
||||
logger.Info("Listening on channel: name=%s, channel=%s", l.config.GetName(), channel)
|
||||
@@ -156,28 +180,21 @@ func (l *PostgresListener) Listen(channel string, handler NotificationHandler) e
|
||||
|
||||
// Unlisten unsubscribes from a PostgreSQL notification channel
|
||||
func (l *PostgresListener) Unlisten(channel string) error {
|
||||
l.closeMu.Lock()
|
||||
if l.closed {
|
||||
l.closeMu.Unlock()
|
||||
return fmt.Errorf("listener is closed")
|
||||
}
|
||||
l.closeMu.Unlock()
|
||||
l.connMu.Lock()
|
||||
defer l.connMu.Unlock()
|
||||
|
||||
l.mu.Lock()
|
||||
defer l.mu.Unlock()
|
||||
|
||||
if l.conn == nil {
|
||||
return fmt.Errorf("listener connection is not initialized")
|
||||
}
|
||||
|
||||
// Execute UNLISTEN command
|
||||
_, err := l.conn.Exec(l.ctx, fmt.Sprintf("UNLISTEN %s", pgx.Identifier{channel}.Sanitize()))
|
||||
conn, err := l.currentConn()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
if _, err := conn.Exec(l.ctx, fmt.Sprintf("UNLISTEN %s", pgx.Identifier{channel}.Sanitize())); err != nil {
|
||||
return fmt.Errorf("failed to unlisten from channel %s: %w", channel, err)
|
||||
}
|
||||
|
||||
// Remove the handler
|
||||
l.mu.Lock()
|
||||
delete(l.channels, channel)
|
||||
l.mu.Unlock()
|
||||
|
||||
if l.config.GetEnableLogging() {
|
||||
logger.Info("Unlistened from channel: name=%s, channel=%s", l.config.GetName(), channel)
|
||||
@@ -188,31 +205,24 @@ func (l *PostgresListener) Unlisten(channel string) error {
|
||||
|
||||
// Notify sends a notification to a PostgreSQL channel
|
||||
func (l *PostgresListener) Notify(ctx context.Context, channel string, payload string) error {
|
||||
l.closeMu.Lock()
|
||||
if l.closed {
|
||||
l.closeMu.Unlock()
|
||||
return fmt.Errorf("listener is closed")
|
||||
}
|
||||
l.closeMu.Unlock()
|
||||
l.connMu.Lock()
|
||||
defer l.connMu.Unlock()
|
||||
|
||||
l.mu.RLock()
|
||||
conn := l.conn
|
||||
l.mu.RUnlock()
|
||||
|
||||
if conn == nil {
|
||||
return fmt.Errorf("listener connection is not initialized")
|
||||
}
|
||||
|
||||
// Execute NOTIFY command
|
||||
_, err := conn.Exec(ctx, "SELECT pg_notify($1, $2)", channel, payload)
|
||||
conn, err := l.currentConn()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
if _, err := conn.Exec(ctx, "SELECT pg_notify($1, $2)", channel, payload); err != nil {
|
||||
return fmt.Errorf("failed to notify channel %s: %w", channel, err)
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// Close closes the listener and all subscriptions
|
||||
// Close closes the listener and all subscriptions. Closing the connection drops
|
||||
// every subscription server-side, so no UNLISTEN round trips are needed, and
|
||||
// the close itself is bounded so a dead socket cannot hang the caller.
|
||||
func (l *PostgresListener) Close() error {
|
||||
l.closeMu.Lock()
|
||||
if l.closed {
|
||||
@@ -225,27 +235,26 @@ func (l *PostgresListener) Close() error {
|
||||
// Cancel context to stop background goroutines
|
||||
l.cancel()
|
||||
|
||||
// The cancelled ctx makes the notification wait return promptly, releasing
|
||||
// connMu; closing the conn while it is being read would race inside pgx.
|
||||
l.connMu.Lock()
|
||||
l.mu.Lock()
|
||||
defer l.mu.Unlock()
|
||||
conn := l.conn
|
||||
l.conn = nil
|
||||
l.channels = make(map[string]NotificationHandler)
|
||||
l.mu.Unlock()
|
||||
|
||||
if l.conn == nil {
|
||||
if conn == nil {
|
||||
l.connMu.Unlock()
|
||||
return nil
|
||||
}
|
||||
|
||||
// Unlisten from all channels
|
||||
for channel := range l.channels {
|
||||
_, _ = l.conn.Exec(context.Background(), fmt.Sprintf("UNLISTEN %s", pgx.Identifier{channel}.Sanitize()))
|
||||
}
|
||||
|
||||
// Close connection
|
||||
err := l.conn.Close(context.Background())
|
||||
err := closeConnBounded(conn)
|
||||
l.connMu.Unlock()
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to close listener connection: %w", err)
|
||||
}
|
||||
|
||||
l.conn = nil
|
||||
l.channels = make(map[string]NotificationHandler)
|
||||
|
||||
if l.config.GetEnableLogging() {
|
||||
logger.Info("PostgreSQL listener closed: name=%s", l.config.GetName())
|
||||
}
|
||||
@@ -262,20 +271,26 @@ func (l *PostgresListener) handleNotifications() {
|
||||
default:
|
||||
}
|
||||
|
||||
l.connMu.Lock()
|
||||
l.mu.RLock()
|
||||
conn := l.conn
|
||||
l.mu.RUnlock()
|
||||
|
||||
if conn == nil {
|
||||
l.connMu.Unlock()
|
||||
// Connection not available, wait for reconnection
|
||||
time.Sleep(100 * time.Millisecond)
|
||||
if !l.sleep(100 * time.Millisecond) {
|
||||
return
|
||||
}
|
||||
continue
|
||||
}
|
||||
|
||||
// Wait for notification with timeout
|
||||
ctx, cancel := context.WithTimeout(l.ctx, 5*time.Second)
|
||||
// Wait for a notification with a short timeout, so Listen/Unlisten/Notify
|
||||
// waiting on connMu are served promptly.
|
||||
ctx, cancel := context.WithTimeout(l.ctx, notificationPollInterval)
|
||||
notification, err := conn.WaitForNotification(ctx)
|
||||
cancel()
|
||||
l.connMu.Unlock()
|
||||
|
||||
if err != nil {
|
||||
// Check if context was cancelled
|
||||
@@ -291,13 +306,15 @@ func (l *PostgresListener) handleNotifications() {
|
||||
|
||||
// Connection error, trigger reconnection
|
||||
if l.config.GetEnableLogging() {
|
||||
logger.Warn("Notification error, triggering reconnection", "error", err)
|
||||
logger.Warn("Notification error, triggering reconnection: %v", err)
|
||||
}
|
||||
select {
|
||||
case l.reconnectC <- struct{}{}:
|
||||
default:
|
||||
}
|
||||
time.Sleep(1 * time.Second)
|
||||
if !l.sleep(1 * time.Second) {
|
||||
return
|
||||
}
|
||||
continue
|
||||
}
|
||||
|
||||
@@ -322,7 +339,22 @@ func (l *PostgresListener) handleNotifications() {
|
||||
}
|
||||
}
|
||||
|
||||
// handleReconnection manages automatic reconnection
|
||||
// sleep waits for d or until the listener is closed; it reports whether the
|
||||
// listener is still running.
|
||||
func (l *PostgresListener) sleep(d time.Duration) bool {
|
||||
t := time.NewTimer(d)
|
||||
defer t.Stop()
|
||||
select {
|
||||
case <-t.C:
|
||||
return true
|
||||
case <-l.ctx.Done():
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
// handleReconnection manages automatic reconnection. It runs as a single
|
||||
// goroutine and dials replacement connections directly rather than through the
|
||||
// public Connect, so no extra loops are started.
|
||||
func (l *PostgresListener) handleReconnection() {
|
||||
for {
|
||||
select {
|
||||
@@ -333,31 +365,21 @@ func (l *PostgresListener) handleReconnection() {
|
||||
logger.Info("Attempting to reconnect listener: name=%s", l.config.GetName())
|
||||
}
|
||||
|
||||
// Close existing connection
|
||||
l.mu.Lock()
|
||||
if l.conn != nil {
|
||||
l.conn.Close(context.Background())
|
||||
l.conn = nil
|
||||
}
|
||||
|
||||
// Save current subscriptions
|
||||
channels := make(map[string]NotificationHandler)
|
||||
for ch, handler := range l.channels {
|
||||
channels[ch] = handler
|
||||
}
|
||||
l.mu.Unlock()
|
||||
|
||||
// Attempt reconnection
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
err := l.Connect(ctx)
|
||||
ctx, cancel := context.WithTimeout(l.ctx, 30*time.Second)
|
||||
err := l.reconnect(ctx)
|
||||
cancel()
|
||||
|
||||
if err != nil {
|
||||
if l.ctx.Err() != nil {
|
||||
return
|
||||
}
|
||||
if l.config.GetEnableLogging() {
|
||||
logger.Error("Failed to reconnect listener: name=%s, error=%v", l.config.GetName(), err)
|
||||
}
|
||||
// Retry after delay
|
||||
time.Sleep(5 * time.Second)
|
||||
if !l.sleep(5 * time.Second) {
|
||||
return
|
||||
}
|
||||
select {
|
||||
case l.reconnectC <- struct{}{}:
|
||||
default:
|
||||
@@ -365,15 +387,6 @@ func (l *PostgresListener) handleReconnection() {
|
||||
continue
|
||||
}
|
||||
|
||||
// Resubscribe to all channels
|
||||
for channel, handler := range channels {
|
||||
if err := l.Listen(channel, handler); err != nil {
|
||||
if l.config.GetEnableLogging() {
|
||||
logger.Error("Failed to resubscribe to channel: name=%s, channel=%s, error=%v", l.config.GetName(), channel, err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if l.config.GetEnableLogging() {
|
||||
logger.Info("Listener reconnected successfully: name=%s", l.config.GetName())
|
||||
}
|
||||
@@ -381,6 +394,51 @@ func (l *PostgresListener) handleReconnection() {
|
||||
}
|
||||
}
|
||||
|
||||
// reconnect replaces the connection and resubscribes every channel on the new
|
||||
// connection before publishing it, so the notification loop never touches a
|
||||
// half-initialised conn.
|
||||
func (l *PostgresListener) reconnect(ctx context.Context) error {
|
||||
conn, err := l.dial(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
l.mu.RLock()
|
||||
channels := make([]string, 0, len(l.channels))
|
||||
for ch := range l.channels {
|
||||
channels = append(channels, ch)
|
||||
}
|
||||
l.mu.RUnlock()
|
||||
|
||||
for _, ch := range channels {
|
||||
if _, err := conn.Exec(ctx, fmt.Sprintf("LISTEN %s", pgx.Identifier{ch}.Sanitize())); err != nil {
|
||||
closeConnBounded(conn)
|
||||
return fmt.Errorf("failed to resubscribe to channel %s: %w", ch, err)
|
||||
}
|
||||
}
|
||||
|
||||
if l.ctx.Err() != nil {
|
||||
closeConnBounded(conn)
|
||||
return l.ctx.Err()
|
||||
}
|
||||
l.swapConn(conn)
|
||||
return nil
|
||||
}
|
||||
|
||||
// swapConn installs conn and closes the previous one. The old connection is
|
||||
// closed under connMu so it is never closed while another goroutine is using it.
|
||||
func (l *PostgresListener) swapConn(conn *pgx.Conn) {
|
||||
l.connMu.Lock()
|
||||
l.mu.Lock()
|
||||
old := l.conn
|
||||
l.conn = conn
|
||||
l.mu.Unlock()
|
||||
if old != nil {
|
||||
closeConnBounded(old)
|
||||
}
|
||||
l.connMu.Unlock()
|
||||
}
|
||||
|
||||
// IsConnected returns true if the listener is connected
|
||||
func (l *PostgresListener) IsConnected() bool {
|
||||
l.mu.RLock()
|
||||
|
||||
@@ -4,17 +4,11 @@ import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"errors"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"go.mongodb.org/mongo-driver/mongo"
|
||||
)
|
||||
|
||||
// isDBClosed reports whether err indicates the *sql.DB has been closed.
|
||||
func isDBClosed(err error) bool {
|
||||
return err != nil && strings.Contains(err.Error(), "sql: database is closed")
|
||||
}
|
||||
|
||||
// Common errors
|
||||
var (
|
||||
// ErrNotSQLDatabase is returned when attempting SQL operations on a non-SQL database
|
||||
@@ -63,6 +57,31 @@ type ConnectionConfig interface {
|
||||
GetConnMaxLifetime() *time.Duration
|
||||
GetConnMaxIdleTime() *time.Duration
|
||||
GetReadPreference() string
|
||||
GetRetryAttempts() int
|
||||
GetRetryDelay() time.Duration
|
||||
GetRetryMaxDelay() time.Duration
|
||||
}
|
||||
|
||||
// retryPolicy returns the configured retry settings, falling back to defaults.
|
||||
func retryPolicy(cfg ConnectionConfig) (attempts int, delay, maxDelay time.Duration) {
|
||||
attempts, delay, maxDelay = cfg.GetRetryAttempts(), cfg.GetRetryDelay(), cfg.GetRetryMaxDelay()
|
||||
if attempts <= 0 {
|
||||
attempts = 3
|
||||
}
|
||||
if delay <= 0 {
|
||||
delay = time.Second
|
||||
}
|
||||
if maxDelay <= 0 {
|
||||
maxDelay = 10 * time.Second
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
// Refresher is implemented by providers that can retire their pooled
|
||||
// connections and dial fresh ones without closing the shared *sql.DB, so
|
||||
// handles already handed out keep working.
|
||||
type Refresher interface {
|
||||
Refresh(ctx context.Context) error
|
||||
}
|
||||
|
||||
// Provider creates and manages the underlying database connection
|
||||
|
||||
@@ -4,6 +4,7 @@ import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"fmt"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
@@ -15,10 +16,9 @@ import (
|
||||
|
||||
// SQLiteProvider implements Provider for SQLite databases
|
||||
type SQLiteProvider struct {
|
||||
db *sql.DB
|
||||
dbMu sync.RWMutex
|
||||
dbFactory func() (*sql.DB, error)
|
||||
config ConnectionConfig
|
||||
db *sql.DB
|
||||
dbMu sync.RWMutex
|
||||
config ConnectionConfig
|
||||
}
|
||||
|
||||
// NewSQLiteProvider creates a new SQLite provider
|
||||
@@ -26,6 +26,22 @@ func NewSQLiteProvider() *SQLiteProvider {
|
||||
return &SQLiteProvider{}
|
||||
}
|
||||
|
||||
// isMemoryDSN reports whether the SQLite DSN refers to a private in-memory
|
||||
// database (each pooled connection would get its own empty database).
|
||||
func isMemoryDSN(dsn string) bool {
|
||||
path := dsn
|
||||
if i := strings.IndexByte(path, '?'); i >= 0 {
|
||||
path = path[:i]
|
||||
}
|
||||
if path == ":memory:" || path == "" {
|
||||
return true
|
||||
}
|
||||
if strings.Contains(dsn, "mode=memory") && !strings.Contains(dsn, "cache=shared") {
|
||||
return true
|
||||
}
|
||||
return path == "file::memory:" && !strings.Contains(dsn, "cache=shared")
|
||||
}
|
||||
|
||||
// Connect establishes a SQLite connection
|
||||
func (p *SQLiteProvider) Connect(ctx context.Context, cfg ConnectionConfig) error {
|
||||
// Build DSN
|
||||
@@ -50,48 +66,35 @@ func (p *SQLiteProvider) Connect(ctx context.Context, cfg ConnectionConfig) erro
|
||||
return fmt.Errorf("failed to ping SQLite database: %w", err)
|
||||
}
|
||||
|
||||
// Configure connection pool
|
||||
// Note: SQLite works best with MaxOpenConns=1 for write operations
|
||||
// but can handle multiple readers
|
||||
if cfg.GetMaxOpenConns() != nil {
|
||||
db.SetMaxOpenConns(*cfg.GetMaxOpenConns())
|
||||
} else {
|
||||
// Default to 1 for SQLite to avoid "database is locked" errors
|
||||
if isMemoryDSN(dsn) {
|
||||
// A private in-memory database exists per connection and disappears when
|
||||
// that connection closes, so pin the pool to one connection that is
|
||||
// never recycled.
|
||||
db.SetMaxOpenConns(1)
|
||||
}
|
||||
|
||||
if cfg.GetMaxIdleConns() != nil {
|
||||
db.SetMaxIdleConns(*cfg.GetMaxIdleConns())
|
||||
}
|
||||
if cfg.GetConnMaxLifetime() != nil {
|
||||
db.SetConnMaxLifetime(*cfg.GetConnMaxLifetime())
|
||||
}
|
||||
if cfg.GetConnMaxIdleTime() != nil {
|
||||
db.SetConnMaxIdleTime(*cfg.GetConnMaxIdleTime())
|
||||
}
|
||||
|
||||
// Enable WAL mode for better concurrent access
|
||||
_, err = db.ExecContext(ctx, "PRAGMA journal_mode=WAL")
|
||||
if err != nil {
|
||||
if cfg.GetEnableLogging() {
|
||||
logger.Warn("Failed to enable WAL mode for SQLite", "error", err)
|
||||
db.SetMaxIdleConns(1)
|
||||
db.SetConnMaxLifetime(0)
|
||||
db.SetConnMaxIdleTime(0)
|
||||
} else {
|
||||
// SQLite works best with few writers; default to 1 unless configured.
|
||||
if cfg.GetMaxOpenConns() != nil {
|
||||
db.SetMaxOpenConns(*cfg.GetMaxOpenConns())
|
||||
} else {
|
||||
db.SetMaxOpenConns(1)
|
||||
}
|
||||
// Don't fail connection if WAL mode cannot be enabled
|
||||
}
|
||||
|
||||
// Set busy timeout to handle locked database (minimum 2 minutes = 120000ms)
|
||||
busyTimeout := cfg.GetQueryTimeout().Milliseconds()
|
||||
if busyTimeout < 120000 {
|
||||
busyTimeout = 120000 // Enforce minimum of 2 minutes
|
||||
}
|
||||
_, err = db.ExecContext(ctx, fmt.Sprintf("PRAGMA busy_timeout=%d", busyTimeout))
|
||||
if err != nil {
|
||||
if cfg.GetEnableLogging() {
|
||||
logger.Warn("Failed to set busy timeout for SQLite", "error", err)
|
||||
if cfg.GetMaxIdleConns() != nil {
|
||||
db.SetMaxIdleConns(*cfg.GetMaxIdleConns())
|
||||
}
|
||||
if cfg.GetConnMaxLifetime() != nil {
|
||||
db.SetConnMaxLifetime(*cfg.GetConnMaxLifetime())
|
||||
}
|
||||
if cfg.GetConnMaxIdleTime() != nil {
|
||||
db.SetConnMaxIdleTime(*cfg.GetConnMaxIdleTime())
|
||||
}
|
||||
}
|
||||
|
||||
p.dbMu.Lock()
|
||||
p.db = db
|
||||
p.dbMu.Unlock()
|
||||
p.config = cfg
|
||||
|
||||
if cfg.GetEnableLogging() {
|
||||
@@ -132,14 +135,7 @@ func (p *SQLiteProvider) HealthCheck(ctx context.Context) error {
|
||||
|
||||
// Execute a simple query to verify the database is accessible
|
||||
var result int
|
||||
run := func() error { return p.getDB().QueryRowContext(healthCtx, "SELECT 1").Scan(&result) }
|
||||
err := run()
|
||||
if isDBClosed(err) {
|
||||
if reconnErr := p.reconnectDB(); reconnErr == nil {
|
||||
err = run()
|
||||
}
|
||||
}
|
||||
if err != nil {
|
||||
if err := p.getDB().QueryRowContext(healthCtx, "SELECT 1").Scan(&result); err != nil {
|
||||
return fmt.Errorf("health check failed: %w", err)
|
||||
}
|
||||
|
||||
@@ -150,32 +146,12 @@ func (p *SQLiteProvider) HealthCheck(ctx context.Context) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// WithDBFactory configures a factory used to reopen the database connection if it is closed.
|
||||
func (p *SQLiteProvider) WithDBFactory(factory func() (*sql.DB, error)) *SQLiteProvider {
|
||||
p.dbFactory = factory
|
||||
return p
|
||||
}
|
||||
|
||||
func (p *SQLiteProvider) getDB() *sql.DB {
|
||||
p.dbMu.RLock()
|
||||
defer p.dbMu.RUnlock()
|
||||
return p.db
|
||||
}
|
||||
|
||||
func (p *SQLiteProvider) reconnectDB() error {
|
||||
if p.dbFactory == nil {
|
||||
return fmt.Errorf("no db factory configured for reconnect")
|
||||
}
|
||||
newDB, err := p.dbFactory()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
p.dbMu.Lock()
|
||||
p.db = newDB
|
||||
p.dbMu.Unlock()
|
||||
return nil
|
||||
}
|
||||
|
||||
// GetNative returns the native *sql.DB connection
|
||||
func (p *SQLiteProvider) GetNative() (*sql.DB, error) {
|
||||
if p.db == nil {
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
//go:build linux
|
||||
|
||||
package providers
|
||||
|
||||
import (
|
||||
"syscall"
|
||||
"time"
|
||||
|
||||
"golang.org/x/sys/unix"
|
||||
)
|
||||
|
||||
// setTCPUserTimeout returns a net.Dialer Control func setting TCP_USER_TIMEOUT.
|
||||
func setTCPUserTimeout(d time.Duration) func(network, address string, c syscall.RawConn) error {
|
||||
return func(network, address string, c syscall.RawConn) error {
|
||||
var sockErr error
|
||||
err := c.Control(func(fd uintptr) {
|
||||
sockErr = unix.SetsockoptInt(int(fd), unix.IPPROTO_TCP, unix.TCP_USER_TIMEOUT, int(d.Milliseconds()))
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return sockErr
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
//go:build !linux
|
||||
|
||||
package providers
|
||||
|
||||
import (
|
||||
"syscall"
|
||||
"time"
|
||||
)
|
||||
|
||||
// setTCPUserTimeout is a no-op where TCP_USER_TIMEOUT is unavailable; TCP
|
||||
// keepalive still applies.
|
||||
func setTCPUserTimeout(time.Duration) func(network, address string, c syscall.RawConn) error {
|
||||
return nil
|
||||
}
|
||||
Reference in New Issue
Block a user