fix: reduce reconnect redeploy and metrics load (#476)

* fix: reduce reconnect redeploy and metrics load

Throttle node-online redeploy retries and lower the agent metric cadence so brief reconnect churn no longer fans out into repeated runtime syncs and backend connection pressure.

* docs: add follow-up implementation design notes

Document the planned flow upload batching work and the local remote-address toggle so the next changesets can implement them against an agreed design.
This commit is contained in:
sagit
2026-04-26 23:35:35 +08:00
committed by GitHub
parent 87a1a34ad5
commit 58d2e89147
9 changed files with 1324 additions and 66 deletions
+15 -9
View File
@@ -44,8 +44,11 @@ type Handler struct {
jobsStarted bool
jobsWG sync.WaitGroup
upgradeMu sync.Mutex
pendingUpgradeRedeploy map[int64]struct{}
upgradeMu sync.Mutex
pendingUpgradeRedeploy map[int64]struct{}
nodeOnlineRedeployAt map[int64]time.Time
nodeOnlineRedeployQueued map[int64]struct{}
nodeOnlineRedeploying map[int64]struct{}
qualityProber *tunnelQualityProber
}
@@ -97,13 +100,16 @@ const (
func New(repo *repo.Repository, jwtSecret string) *Handler {
h := &Handler{
repo: repo,
jwtSecret: jwtSecret,
wsServer: ws.NewServer(repo, jwtSecret),
metrics: metrics.NewIngestionService(repo),
healthCheck: nil,
captchaTokens: make(map[string]int64),
pendingUpgradeRedeploy: make(map[int64]struct{}),
repo: repo,
jwtSecret: jwtSecret,
wsServer: ws.NewServer(repo, jwtSecret),
metrics: metrics.NewIngestionService(repo),
healthCheck: nil,
captchaTokens: make(map[string]int64),
pendingUpgradeRedeploy: make(map[int64]struct{}),
nodeOnlineRedeployAt: make(map[int64]time.Time),
nodeOnlineRedeployQueued: make(map[int64]struct{}),
nodeOnlineRedeploying: make(map[int64]struct{}),
}
h.healthCheck = health.NewChecker(repo, h.wsServer)
h.qualityProber = newTunnelQualityProber(h)
+112 -12
View File
@@ -39,6 +39,8 @@ var (
testKeywordPattern = regexp.MustCompile(`(?i)(alpha|beta|rc)`)
)
const nodeOnlineRedeployCooldown = 30 * time.Second
type githubRelease struct {
TagName string `json:"tag_name"`
Name string `json:"name"`
@@ -396,23 +398,120 @@ func (h *Handler) consumeNodePendingUpgradeRedeploy(nodeID int64) bool {
}
func (h *Handler) onNodeOnline(nodeID int64) {
h.consumeNodePendingUpgradeRedeploy(nodeID)
// Always redeploy rules on reconnection, not just for pending upgrade nodes.
// This handles cases where the node restarted and lost its in-memory config
// before persistence had time to flush, or if the panel also restarted.
h.redeployNodeRuntimeAfterUpgrade(nodeID)
if !h.startNodeOnlineRedeploy(nodeID, time.Now()) {
return
}
defer h.finishNodeOnlineRedeploy(nodeID)
// Reconcile node runtime on the first reconnect, but suppress rapid flapping
// so websocket churn does not trigger repeated full redeploy storms.
if !h.redeployNodeRuntimeAfterUpgrade(nodeID) {
h.markNodePendingUpgradeRedeploy(nodeID)
}
}
func (h *Handler) redeployNodeRuntimeAfterUpgrade(nodeID int64) {
func (h *Handler) startNodeOnlineRedeploy(nodeID int64, now time.Time) bool {
if h == nil || nodeID <= 0 {
return false
}
if now.IsZero() {
now = time.Now()
}
h.upgradeMu.Lock()
defer h.upgradeMu.Unlock()
if h.pendingUpgradeRedeploy == nil {
h.pendingUpgradeRedeploy = make(map[int64]struct{})
}
if h.nodeOnlineRedeployAt == nil {
h.nodeOnlineRedeployAt = make(map[int64]time.Time)
}
if h.nodeOnlineRedeployQueued == nil {
h.nodeOnlineRedeployQueued = make(map[int64]struct{})
}
if h.nodeOnlineRedeploying == nil {
h.nodeOnlineRedeploying = make(map[int64]struct{})
}
_, pendingUpgrade := h.pendingUpgradeRedeploy[nodeID]
lastRedeployAt := h.nodeOnlineRedeployAt[nodeID]
_, inFlight := h.nodeOnlineRedeploying[nodeID]
if fireAt, start := nextNodeOnlineRedeployFireAt(lastRedeployAt, now, pendingUpgrade, inFlight); !start {
h.queueNodeOnlineRedeployLocked(nodeID, fireAt)
return false
}
delete(h.pendingUpgradeRedeploy, nodeID)
h.nodeOnlineRedeployAt[nodeID] = now
h.nodeOnlineRedeploying[nodeID] = struct{}{}
return true
}
func nextNodeOnlineRedeployFireAt(lastRedeployAt, now time.Time, pendingUpgrade bool, inFlight bool) (time.Time, bool) {
if now.IsZero() {
now = time.Now()
}
if inFlight {
fireAt := now.Add(nodeOnlineRedeployCooldown)
if !lastRedeployAt.IsZero() {
cooldownAt := lastRedeployAt.Add(nodeOnlineRedeployCooldown)
if cooldownAt.After(now) {
fireAt = cooldownAt
}
}
return fireAt, false
}
if !pendingUpgrade && !lastRedeployAt.IsZero() && now.Sub(lastRedeployAt) < nodeOnlineRedeployCooldown {
return lastRedeployAt.Add(nodeOnlineRedeployCooldown), false
}
return time.Time{}, true
}
func (h *Handler) queueNodeOnlineRedeployLocked(nodeID int64, fireAt time.Time) {
if h == nil || nodeID <= 0 {
return
}
if h.nodeOnlineRedeployQueued == nil {
h.nodeOnlineRedeployQueued = make(map[int64]struct{})
}
if _, queued := h.nodeOnlineRedeployQueued[nodeID]; queued {
return
}
if fireAt.IsZero() {
fireAt = time.Now().Add(nodeOnlineRedeployCooldown)
}
delay := time.Until(fireAt)
if delay < 0 {
delay = 0
}
h.nodeOnlineRedeployQueued[nodeID] = struct{}{}
time.AfterFunc(delay, func() {
h.upgradeMu.Lock()
delete(h.nodeOnlineRedeployQueued, nodeID)
h.upgradeMu.Unlock()
h.onNodeOnline(nodeID)
})
}
func (h *Handler) finishNodeOnlineRedeploy(nodeID int64) {
if h == nil || nodeID <= 0 {
return
}
h.upgradeMu.Lock()
delete(h.nodeOnlineRedeploying, nodeID)
h.upgradeMu.Unlock()
}
func (h *Handler) redeployNodeRuntimeAfterUpgrade(nodeID int64) bool {
tunnelIDs, err := h.repo.ListActiveTunnelIDsByNode(nodeID)
if err != nil {
fmt.Printf("post-upgrade redeploy: list tunnels for node %d failed: %v\n", nodeID, err)
return
return false
}
forwardIDs, err := h.repo.ListForwardIDsByNode(nodeID)
if err != nil {
fmt.Printf("post-upgrade redeploy: list forwards for node %d failed: %v\n", nodeID, err)
return
return false
}
// First pass: deploy everything
@@ -442,7 +541,7 @@ func (h *Handler) redeployNodeRuntimeAfterUpgrade(nodeID int64) {
}
// Retry failed items with exponential backoff (max 3 attempts)
h.retryFailedRedeploys(nodeID, tunnelFailed, failedForwards)
return h.retryFailedRedeploys(nodeID, tunnelFailed, failedForwards)
}
// isRetryableError returns true if the error looks transient and worth retrying.
@@ -463,9 +562,9 @@ func isRetryableError(err error) bool {
}
// retryFailedRedeploys retries failed tunnels and forwards with exponential backoff.
func (h *Handler) retryFailedRedeploys(nodeID int64, tunnelFailed map[int64]struct{}, failedForwards []failedForward) {
func (h *Handler) retryFailedRedeploys(nodeID int64, tunnelFailed map[int64]struct{}, failedForwards []failedForward) bool {
if len(tunnelFailed) == 0 && len(failedForwards) == 0 {
return
return true
}
const maxRetries = 3
@@ -507,7 +606,7 @@ func (h *Handler) retryFailedRedeploys(nodeID int64, tunnelFailed map[int64]stru
if len(tunnelFailed) == 0 && len(failedForwards) == 0 {
fmt.Printf("post-upgrade redeploy retry: all items recovered on node %d\n", nodeID)
return
return true
}
}
@@ -518,4 +617,5 @@ func (h *Handler) retryFailedRedeploys(nodeID int64, tunnelFailed map[int64]stru
for _, ff := range failedForwards {
fmt.Printf("post-upgrade redeploy: forward %d permanently failed on node %d after retries\n", ff.id, nodeID)
}
return false
}
@@ -0,0 +1,111 @@
package handler
import (
"testing"
"time"
)
func TestStartNodeOnlineRedeploySkipsRecentReconnects(t *testing.T) {
h := &Handler{
pendingUpgradeRedeploy: map[int64]struct{}{},
nodeOnlineRedeployAt: map[int64]time.Time{},
nodeOnlineRedeployQueued: map[int64]struct{}{},
nodeOnlineRedeploying: map[int64]struct{}{},
}
now := time.Unix(1_777_176_720, 0)
if !h.startNodeOnlineRedeploy(54, now) {
t.Fatalf("expected first reconnect to redeploy")
}
h.finishNodeOnlineRedeploy(54)
if h.startNodeOnlineRedeploy(54, now.Add(5*time.Second)) {
t.Fatalf("expected recent reconnect to skip redeploy")
}
if h.consumeNodePendingUpgradeRedeploy(54) {
t.Fatalf("did not expect pending upgrade marker to be consumed")
}
}
func TestStartNodeOnlineRedeployAllowsPendingUpgradeDuringCooldown(t *testing.T) {
h := &Handler{
pendingUpgradeRedeploy: map[int64]struct{}{},
nodeOnlineRedeployAt: map[int64]time.Time{},
nodeOnlineRedeployQueued: map[int64]struct{}{},
nodeOnlineRedeploying: map[int64]struct{}{},
}
now := time.Unix(1_777_176_720, 0)
if !h.startNodeOnlineRedeploy(54, now) {
t.Fatalf("expected first reconnect to redeploy")
}
h.finishNodeOnlineRedeploy(54)
h.markNodePendingUpgradeRedeploy(54)
if !h.startNodeOnlineRedeploy(54, now.Add(5*time.Second)) {
t.Fatalf("expected pending upgrade reconnect to bypass cooldown")
}
if h.consumeNodePendingUpgradeRedeploy(54) {
t.Fatalf("expected pending upgrade marker to be consumed during redeploy")
}
}
func TestStartNodeOnlineRedeployQueuesCooldownReconnect(t *testing.T) {
h := &Handler{
pendingUpgradeRedeploy: map[int64]struct{}{},
nodeOnlineRedeployAt: map[int64]time.Time{},
nodeOnlineRedeployQueued: map[int64]struct{}{},
nodeOnlineRedeploying: map[int64]struct{}{},
}
now := time.Unix(1_777_176_720, 0)
if !h.startNodeOnlineRedeploy(54, now) {
t.Fatalf("expected first reconnect to redeploy")
}
h.finishNodeOnlineRedeploy(54)
if h.startNodeOnlineRedeploy(54, now.Add(5*time.Second)) {
t.Fatalf("expected cooldown reconnect to skip immediate redeploy")
}
if _, queued := h.nodeOnlineRedeployQueued[54]; !queued {
t.Fatalf("expected cooldown reconnect to queue a follow-up redeploy")
}
}
func TestStartNodeOnlineRedeployKeepsPendingUpgradeWhileInFlight(t *testing.T) {
h := &Handler{
pendingUpgradeRedeploy: map[int64]struct{}{},
nodeOnlineRedeployAt: map[int64]time.Time{},
nodeOnlineRedeployQueued: map[int64]struct{}{},
nodeOnlineRedeploying: map[int64]struct{}{},
}
now := time.Unix(1_777_176_720, 0)
if !h.startNodeOnlineRedeploy(54, now) {
t.Fatalf("expected first reconnect to redeploy")
}
h.markNodePendingUpgradeRedeploy(54)
if h.startNodeOnlineRedeploy(54, now.Add(time.Second)) {
t.Fatalf("expected in-flight redeploy to suppress parallel restart")
}
if !h.consumeNodePendingUpgradeRedeploy(54) {
t.Fatalf("expected pending upgrade marker to remain for the next retry")
}
h.finishNodeOnlineRedeploy(54)
}
func TestNextNodeOnlineRedeployFireAtDefersExpiredInFlightReconnect(t *testing.T) {
now := time.Unix(1_777_176_720, 0)
last := now.Add(-nodeOnlineRedeployCooldown - 5*time.Second)
fireAt, start := nextNodeOnlineRedeployFireAt(last, now, false, true)
if start {
t.Fatalf("expected in-flight reconnect to queue instead of starting immediately")
}
want := now.Add(nodeOnlineRedeployCooldown)
if !fireAt.Equal(want) {
t.Fatalf("expected queued reconnect at %s, got %s", want, fireAt)
}
}
+55 -37
View File
@@ -22,6 +22,13 @@ import (
"go-backend/internal/store/model"
)
const (
defaultPostgresMaxOpenConns = 32
defaultPostgresMaxIdleConns = 8
defaultPostgresConnMaxIdle = 5 * time.Minute
defaultPostgresConnMaxLife = 30 * time.Minute
)
// ─── Type aliases for backward compatibility ─────────────────────────
// Handlers still reference repo.User, repo.BackupData, etc.
@@ -210,6 +217,7 @@ func OpenPostgres(dsn string) (*Repository, error) {
if err != nil {
return nil, err
}
configurePostgresPool(sqlDB)
if err := sqlDB.Ping(); err != nil {
_ = sqlDB.Close()
return nil, err
@@ -235,6 +243,16 @@ func OpenPostgres(dsn string) (*Repository, error) {
return &Repository{db: db}, nil
}
func configurePostgresPool(sqlDB *sql.DB) {
if sqlDB == nil {
return
}
sqlDB.SetMaxOpenConns(defaultPostgresMaxOpenConns)
sqlDB.SetMaxIdleConns(defaultPostgresMaxIdleConns)
sqlDB.SetConnMaxIdleTime(defaultPostgresConnMaxIdle)
sqlDB.SetConnMaxLifetime(defaultPostgresConnMaxLife)
}
func (r *Repository) Close() error {
if r == nil || r.db == nil {
return nil
@@ -771,11 +789,11 @@ func (r *Repository) ListNodes() ([]map[string]interface{}, error) {
"version": nullableString(n.Version),
"http": n.HTTP, "tls": n.TLS, "socks": n.Socks,
"status": n.Status, "isRemote": n.IsRemote,
"remoteUrl": nullableString(n.RemoteURL),
"remoteToken": nullableString(n.RemoteToken),
"remoteConfig": nullableString(n.RemoteConfig),
"expiryReminderDismissed": n.ExpiryReminderDismissed,
"interfaceName": nullableString(n.InterfaceName),
"remoteUrl": nullableString(n.RemoteURL),
"remoteToken": nullableString(n.RemoteToken),
"remoteConfig": nullableString(n.RemoteConfig),
"expiryReminderDismissed": n.ExpiryReminderDismissed,
"interfaceName": nullableString(n.InterfaceName),
})
}
return items, nil
@@ -806,7 +824,7 @@ func (r *Repository) ListUsers() ([]map[string]interface{}, error) {
"flowResetTime": u.FlowResetTime, "createdTime": u.CreatedTime,
"updatedTime": nullableInt64(u.UpdatedTime),
"inFlow": u.InFlow, "outFlow": u.OutFlow,
"maxConn": u.MaxConn,
"maxConn": u.MaxConn,
}
if quota := quotaMap[u.ID]; quota != nil {
item["dailyQuotaGB"] = quota.DailyLimitGB
@@ -847,22 +865,22 @@ func (r *Repository) ListForwards() ([]map[string]interface{}, error) {
}
type fwdRow struct {
ID int64
UserID int64
UserName string
Name string
TunnelID int64
TunnelName string
TrafficRatio float64
RemoteAddr string
Strategy string
InFlow int64
OutFlow int64
CreatedTime int64
Status int
Inx int
SpeedID sql.NullInt64
MaxConn int
ID int64
UserID int64
UserName string
Name string
TunnelID int64
TunnelName string
TrafficRatio float64
RemoteAddr string
Strategy string
InFlow int64
OutFlow int64
CreatedTime int64
Status int
Inx int
SpeedID sql.NullInt64
MaxConn int
ProxyProtocol int
}
@@ -890,7 +908,7 @@ func (r *Repository) ListForwards() ([]map[string]interface{}, error) {
"remoteAddr": row.RemoteAddr, "strategy": row.Strategy,
"inFlow": row.InFlow, "outFlow": row.OutFlow,
"createdTime": row.CreatedTime, "status": row.Status, "inx": int64(row.Inx),
"maxConn": row.MaxConn,
"maxConn": row.MaxConn,
"proxyProtocol": row.ProxyProtocol,
}
if row.SpeedID.Valid {
@@ -2469,19 +2487,19 @@ func importForwards(tx *gorm.DB, forwards []model.ForwardBackup, now int64) (int
count := 0
for _, f := range forwards {
item := model.Forward{
ID: f.ID,
UserID: f.UserID,
UserName: f.UserName,
Name: f.Name,
TunnelID: f.TunnelID,
RemoteAddr: f.RemoteAddr,
Strategy: f.Strategy,
InFlow: f.InFlow,
OutFlow: f.OutFlow,
CreatedTime: f.CreatedTime,
UpdatedTime: now,
Status: f.Status,
Inx: f.Inx,
ID: f.ID,
UserID: f.UserID,
UserName: f.UserName,
Name: f.Name,
TunnelID: f.TunnelID,
RemoteAddr: f.RemoteAddr,
Strategy: f.Strategy,
InFlow: f.InFlow,
OutFlow: f.OutFlow,
CreatedTime: f.CreatedTime,
UpdatedTime: now,
Status: f.Status,
Inx: f.Inx,
ProxyProtocol: f.ProxyProtocol,
}
err := tx.Clauses(clause.OnConflict{
@@ -3436,7 +3454,7 @@ func (r *Repository) GetNodeMetrics(nodeID int64, startMs, endMs int64) ([]model
rangeMs := endMs - startMs
const maxRawRangeMs = int64(60 * 60 * 1000) // 1 hour — return raw data for short ranges
const targetPoints = 500 // target number of chart points for downsampled data
const targetPoints = 500 // target number of chart points for downsampled data
// For short ranges, return raw data (full resolution).
if rangeMs <= maxRawRangeMs {
@@ -0,0 +1,32 @@
package repo
import (
"testing"
gsqlite "github.com/glebarez/sqlite"
"gorm.io/gorm"
"gorm.io/gorm/logger"
)
func TestConfigurePostgresPoolSetsMaxOpenConnections(t *testing.T) {
db, err := gorm.Open(gsqlite.Open(":memory:"), &gorm.Config{
Logger: logger.Default.LogMode(logger.Silent),
})
if err != nil {
t.Fatalf("open sqlite: %v", err)
}
sqlDB, err := db.DB()
if err != nil {
t.Fatalf("db handle: %v", err)
}
t.Cleanup(func() {
_ = sqlDB.Close()
})
configurePostgresPool(sqlDB)
if got := sqlDB.Stats().MaxOpenConnections; got != defaultPostgresMaxOpenConns {
t.Fatalf("expected max open conns %d, got %d", defaultPostgresMaxOpenConns, got)
}
}