mirror of
https://github.com/bitechdev/ResolveSpec.git
synced 2026-10-01 04:21:58 +00:00
fix(dbmanager): keep the pool alive across errors and restarts
Implements the fixes from audit/pkg/dbmanager.audit.md. - Stop closing the shared *sql.DB to recover from errors. Adapter factories and the health checker no longer call Reconnect; Reconnect is atomic and operator-only. - Postgres uses a custom driver.Connector: Reconnect retires pooled connections by generation without closing the pool, so held Bun/GORM handles keep working. Verified against a live server restart. - Add TCP keepalive, TCP_USER_TIMEOUT, a bounded reuse ping and statement_timeout as a runtime parameter; drop the 2 min timeout floor. - Health check pings without holding the connection lock. - Listener: single goroutine pair, bounded Close without UNLISTEN, and serialised use of the pgx connection (fixes conn busy and a close race). - Fix Connect/Close/Connect/Close panic, idempotent Connect, dial outside the manager lock, clean up on partial failure. - SQLite: pin :memory: to one connection, pragmas via DSN. - Escape credentials in Postgres/MSSQL/Mongo DSNs; sslmode defaults to prefer. Wire retry settings, publish metrics, fix logger calls. - NewConnectionFromDB: Close is a no-op with a warning (caller owns the pool); Reconnect only pings. - Document correct usage in the README; mark the audit with what was done. Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 5.5
parent
bc8bff7955
commit
da1af1487e
@@ -0,0 +1,69 @@
|
||||
package dbmanager
|
||||
|
||||
import (
|
||||
"context"
|
||||
"os"
|
||||
"os/exec"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func TestLiveServerRestart(t *testing.T) {
|
||||
dir := os.Getenv("PG_RESTART_DIR")
|
||||
if dir == "" {
|
||||
t.Skip("PG_RESTART_DIR not set")
|
||||
}
|
||||
mgr, _ := NewManager(ManagerConfig{
|
||||
DefaultConnection: "pg",
|
||||
Connections: map[string]ConnectionConfig{"pg": {
|
||||
Name: "pg", Type: DatabaseTypePostgreSQL, Host: "127.0.0.1", Port: 54329,
|
||||
User: "postgres", Database: "postgres", ConnectTimeout: 2 * time.Second,
|
||||
}},
|
||||
HealthCheckInterval: 500 * time.Millisecond,
|
||||
})
|
||||
ctx := context.Background()
|
||||
if err := mgr.Connect(ctx); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer mgr.Close()
|
||||
conn, _ := mgr.GetDefault()
|
||||
db, _ := conn.Bun()
|
||||
gdb, _ := conn.GORM()
|
||||
query := func() error { var n int; return db.DB.QueryRow("select 1").Scan(&n) }
|
||||
if err := query(); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
run := func(args ...string) {
|
||||
// Output must not be piped: the daemonised server would hold the pipe open.
|
||||
if err := exec.Command("pg_ctl", append([]string{"-D", dir}, args...)...).Run(); err != nil {
|
||||
t.Fatalf("pg_ctl %v: %v", args, err)
|
||||
}
|
||||
}
|
||||
run("-m", "immediate", "-w", "stop") // crash-style shutdown
|
||||
time.Sleep(1500 * time.Millisecond) // health checks fail meanwhile
|
||||
if err := query(); err == nil {
|
||||
t.Fatal("expected failure while server is down")
|
||||
}
|
||||
run("-l", dir+"/restart.log", "-o", "-p 54329 -k "+dir+" -c listen_addresses=127.0.0.1", "-w", "start")
|
||||
|
||||
var last error
|
||||
for i := 0; i < 20; i++ {
|
||||
if last = query(); last == nil {
|
||||
break
|
||||
}
|
||||
t.Logf("attempt %d after restart: %v", i, last)
|
||||
time.Sleep(200 * time.Millisecond)
|
||||
}
|
||||
if last != nil {
|
||||
t.Fatalf("held bun handle never recovered: %v", last)
|
||||
}
|
||||
var n int
|
||||
if err := gdb.Raw("select 1").Scan(&n).Error; err != nil {
|
||||
t.Fatalf("held gorm handle: %v", err)
|
||||
}
|
||||
time.Sleep(time.Second)
|
||||
if err := conn.HealthCheck(ctx); err != nil {
|
||||
t.Fatalf("health check after restart: %v", err)
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user