fix: reduce reconnect redeploy and metrics load (#476)

* fix: reduce reconnect redeploy and metrics load

Throttle node-online redeploy retries and lower the agent metric cadence so brief reconnect churn no longer fans out into repeated runtime syncs and backend connection pressure.

* docs: add follow-up implementation design notes

Document the planned flow upload batching work and the local remote-address toggle so the next changesets can implement them against an agreed design.
This commit is contained in:
sagit
2026-04-26 23:35:35 +08:00
committed by GitHub
parent 87a1a34ad5
commit 58d2e89147
9 changed files with 1324 additions and 66 deletions
+15 -9
View File
@@ -44,8 +44,11 @@ type Handler struct {
jobsStarted bool
jobsWG sync.WaitGroup
upgradeMu sync.Mutex
pendingUpgradeRedeploy map[int64]struct{}
upgradeMu sync.Mutex
pendingUpgradeRedeploy map[int64]struct{}
nodeOnlineRedeployAt map[int64]time.Time
nodeOnlineRedeployQueued map[int64]struct{}
nodeOnlineRedeploying map[int64]struct{}
qualityProber *tunnelQualityProber
}
@@ -97,13 +100,16 @@ const (
func New(repo *repo.Repository, jwtSecret string) *Handler {
h := &Handler{
repo: repo,
jwtSecret: jwtSecret,
wsServer: ws.NewServer(repo, jwtSecret),
metrics: metrics.NewIngestionService(repo),
healthCheck: nil,
captchaTokens: make(map[string]int64),
pendingUpgradeRedeploy: make(map[int64]struct{}),
repo: repo,
jwtSecret: jwtSecret,
wsServer: ws.NewServer(repo, jwtSecret),
metrics: metrics.NewIngestionService(repo),
healthCheck: nil,
captchaTokens: make(map[string]int64),
pendingUpgradeRedeploy: make(map[int64]struct{}),
nodeOnlineRedeployAt: make(map[int64]time.Time),
nodeOnlineRedeployQueued: make(map[int64]struct{}),
nodeOnlineRedeploying: make(map[int64]struct{}),
}
h.healthCheck = health.NewChecker(repo, h.wsServer)
h.qualityProber = newTunnelQualityProber(h)
+112 -12
View File
@@ -39,6 +39,8 @@ var (
testKeywordPattern = regexp.MustCompile(`(?i)(alpha|beta|rc)`)
)
const nodeOnlineRedeployCooldown = 30 * time.Second
type githubRelease struct {
TagName string `json:"tag_name"`
Name string `json:"name"`
@@ -396,23 +398,120 @@ func (h *Handler) consumeNodePendingUpgradeRedeploy(nodeID int64) bool {
}
func (h *Handler) onNodeOnline(nodeID int64) {
h.consumeNodePendingUpgradeRedeploy(nodeID)
// Always redeploy rules on reconnection, not just for pending upgrade nodes.
// This handles cases where the node restarted and lost its in-memory config
// before persistence had time to flush, or if the panel also restarted.
h.redeployNodeRuntimeAfterUpgrade(nodeID)
if !h.startNodeOnlineRedeploy(nodeID, time.Now()) {
return
}
defer h.finishNodeOnlineRedeploy(nodeID)
// Reconcile node runtime on the first reconnect, but suppress rapid flapping
// so websocket churn does not trigger repeated full redeploy storms.
if !h.redeployNodeRuntimeAfterUpgrade(nodeID) {
h.markNodePendingUpgradeRedeploy(nodeID)
}
}
func (h *Handler) redeployNodeRuntimeAfterUpgrade(nodeID int64) {
func (h *Handler) startNodeOnlineRedeploy(nodeID int64, now time.Time) bool {
if h == nil || nodeID <= 0 {
return false
}
if now.IsZero() {
now = time.Now()
}
h.upgradeMu.Lock()
defer h.upgradeMu.Unlock()
if h.pendingUpgradeRedeploy == nil {
h.pendingUpgradeRedeploy = make(map[int64]struct{})
}
if h.nodeOnlineRedeployAt == nil {
h.nodeOnlineRedeployAt = make(map[int64]time.Time)
}
if h.nodeOnlineRedeployQueued == nil {
h.nodeOnlineRedeployQueued = make(map[int64]struct{})
}
if h.nodeOnlineRedeploying == nil {
h.nodeOnlineRedeploying = make(map[int64]struct{})
}
_, pendingUpgrade := h.pendingUpgradeRedeploy[nodeID]
lastRedeployAt := h.nodeOnlineRedeployAt[nodeID]
_, inFlight := h.nodeOnlineRedeploying[nodeID]
if fireAt, start := nextNodeOnlineRedeployFireAt(lastRedeployAt, now, pendingUpgrade, inFlight); !start {
h.queueNodeOnlineRedeployLocked(nodeID, fireAt)
return false
}
delete(h.pendingUpgradeRedeploy, nodeID)
h.nodeOnlineRedeployAt[nodeID] = now
h.nodeOnlineRedeploying[nodeID] = struct{}{}
return true
}
func nextNodeOnlineRedeployFireAt(lastRedeployAt, now time.Time, pendingUpgrade bool, inFlight bool) (time.Time, bool) {
if now.IsZero() {
now = time.Now()
}
if inFlight {
fireAt := now.Add(nodeOnlineRedeployCooldown)
if !lastRedeployAt.IsZero() {
cooldownAt := lastRedeployAt.Add(nodeOnlineRedeployCooldown)
if cooldownAt.After(now) {
fireAt = cooldownAt
}
}
return fireAt, false
}
if !pendingUpgrade && !lastRedeployAt.IsZero() && now.Sub(lastRedeployAt) < nodeOnlineRedeployCooldown {
return lastRedeployAt.Add(nodeOnlineRedeployCooldown), false
}
return time.Time{}, true
}
func (h *Handler) queueNodeOnlineRedeployLocked(nodeID int64, fireAt time.Time) {
if h == nil || nodeID <= 0 {
return
}
if h.nodeOnlineRedeployQueued == nil {
h.nodeOnlineRedeployQueued = make(map[int64]struct{})
}
if _, queued := h.nodeOnlineRedeployQueued[nodeID]; queued {
return
}
if fireAt.IsZero() {
fireAt = time.Now().Add(nodeOnlineRedeployCooldown)
}
delay := time.Until(fireAt)
if delay < 0 {
delay = 0
}
h.nodeOnlineRedeployQueued[nodeID] = struct{}{}
time.AfterFunc(delay, func() {
h.upgradeMu.Lock()
delete(h.nodeOnlineRedeployQueued, nodeID)
h.upgradeMu.Unlock()
h.onNodeOnline(nodeID)
})
}
func (h *Handler) finishNodeOnlineRedeploy(nodeID int64) {
if h == nil || nodeID <= 0 {
return
}
h.upgradeMu.Lock()
delete(h.nodeOnlineRedeploying, nodeID)
h.upgradeMu.Unlock()
}
func (h *Handler) redeployNodeRuntimeAfterUpgrade(nodeID int64) bool {
tunnelIDs, err := h.repo.ListActiveTunnelIDsByNode(nodeID)
if err != nil {
fmt.Printf("post-upgrade redeploy: list tunnels for node %d failed: %v\n", nodeID, err)
return
return false
}
forwardIDs, err := h.repo.ListForwardIDsByNode(nodeID)
if err != nil {
fmt.Printf("post-upgrade redeploy: list forwards for node %d failed: %v\n", nodeID, err)
return
return false
}
// First pass: deploy everything
@@ -442,7 +541,7 @@ func (h *Handler) redeployNodeRuntimeAfterUpgrade(nodeID int64) {
}
// Retry failed items with exponential backoff (max 3 attempts)
h.retryFailedRedeploys(nodeID, tunnelFailed, failedForwards)
return h.retryFailedRedeploys(nodeID, tunnelFailed, failedForwards)
}
// isRetryableError returns true if the error looks transient and worth retrying.
@@ -463,9 +562,9 @@ func isRetryableError(err error) bool {
}
// retryFailedRedeploys retries failed tunnels and forwards with exponential backoff.
func (h *Handler) retryFailedRedeploys(nodeID int64, tunnelFailed map[int64]struct{}, failedForwards []failedForward) {
func (h *Handler) retryFailedRedeploys(nodeID int64, tunnelFailed map[int64]struct{}, failedForwards []failedForward) bool {
if len(tunnelFailed) == 0 && len(failedForwards) == 0 {
return
return true
}
const maxRetries = 3
@@ -507,7 +606,7 @@ func (h *Handler) retryFailedRedeploys(nodeID int64, tunnelFailed map[int64]stru
if len(tunnelFailed) == 0 && len(failedForwards) == 0 {
fmt.Printf("post-upgrade redeploy retry: all items recovered on node %d\n", nodeID)
return
return true
}
}
@@ -518,4 +617,5 @@ func (h *Handler) retryFailedRedeploys(nodeID int64, tunnelFailed map[int64]stru
for _, ff := range failedForwards {
fmt.Printf("post-upgrade redeploy: forward %d permanently failed on node %d after retries\n", ff.id, nodeID)
}
return false
}
@@ -0,0 +1,111 @@
package handler
import (
"testing"
"time"
)
func TestStartNodeOnlineRedeploySkipsRecentReconnects(t *testing.T) {
h := &Handler{
pendingUpgradeRedeploy: map[int64]struct{}{},
nodeOnlineRedeployAt: map[int64]time.Time{},
nodeOnlineRedeployQueued: map[int64]struct{}{},
nodeOnlineRedeploying: map[int64]struct{}{},
}
now := time.Unix(1_777_176_720, 0)
if !h.startNodeOnlineRedeploy(54, now) {
t.Fatalf("expected first reconnect to redeploy")
}
h.finishNodeOnlineRedeploy(54)
if h.startNodeOnlineRedeploy(54, now.Add(5*time.Second)) {
t.Fatalf("expected recent reconnect to skip redeploy")
}
if h.consumeNodePendingUpgradeRedeploy(54) {
t.Fatalf("did not expect pending upgrade marker to be consumed")
}
}
func TestStartNodeOnlineRedeployAllowsPendingUpgradeDuringCooldown(t *testing.T) {
h := &Handler{
pendingUpgradeRedeploy: map[int64]struct{}{},
nodeOnlineRedeployAt: map[int64]time.Time{},
nodeOnlineRedeployQueued: map[int64]struct{}{},
nodeOnlineRedeploying: map[int64]struct{}{},
}
now := time.Unix(1_777_176_720, 0)
if !h.startNodeOnlineRedeploy(54, now) {
t.Fatalf("expected first reconnect to redeploy")
}
h.finishNodeOnlineRedeploy(54)
h.markNodePendingUpgradeRedeploy(54)
if !h.startNodeOnlineRedeploy(54, now.Add(5*time.Second)) {
t.Fatalf("expected pending upgrade reconnect to bypass cooldown")
}
if h.consumeNodePendingUpgradeRedeploy(54) {
t.Fatalf("expected pending upgrade marker to be consumed during redeploy")
}
}
func TestStartNodeOnlineRedeployQueuesCooldownReconnect(t *testing.T) {
h := &Handler{
pendingUpgradeRedeploy: map[int64]struct{}{},
nodeOnlineRedeployAt: map[int64]time.Time{},
nodeOnlineRedeployQueued: map[int64]struct{}{},
nodeOnlineRedeploying: map[int64]struct{}{},
}
now := time.Unix(1_777_176_720, 0)
if !h.startNodeOnlineRedeploy(54, now) {
t.Fatalf("expected first reconnect to redeploy")
}
h.finishNodeOnlineRedeploy(54)
if h.startNodeOnlineRedeploy(54, now.Add(5*time.Second)) {
t.Fatalf("expected cooldown reconnect to skip immediate redeploy")
}
if _, queued := h.nodeOnlineRedeployQueued[54]; !queued {
t.Fatalf("expected cooldown reconnect to queue a follow-up redeploy")
}
}
func TestStartNodeOnlineRedeployKeepsPendingUpgradeWhileInFlight(t *testing.T) {
h := &Handler{
pendingUpgradeRedeploy: map[int64]struct{}{},
nodeOnlineRedeployAt: map[int64]time.Time{},
nodeOnlineRedeployQueued: map[int64]struct{}{},
nodeOnlineRedeploying: map[int64]struct{}{},
}
now := time.Unix(1_777_176_720, 0)
if !h.startNodeOnlineRedeploy(54, now) {
t.Fatalf("expected first reconnect to redeploy")
}
h.markNodePendingUpgradeRedeploy(54)
if h.startNodeOnlineRedeploy(54, now.Add(time.Second)) {
t.Fatalf("expected in-flight redeploy to suppress parallel restart")
}
if !h.consumeNodePendingUpgradeRedeploy(54) {
t.Fatalf("expected pending upgrade marker to remain for the next retry")
}
h.finishNodeOnlineRedeploy(54)
}
func TestNextNodeOnlineRedeployFireAtDefersExpiredInFlightReconnect(t *testing.T) {
now := time.Unix(1_777_176_720, 0)
last := now.Add(-nodeOnlineRedeployCooldown - 5*time.Second)
fireAt, start := nextNodeOnlineRedeployFireAt(last, now, false, true)
if start {
t.Fatalf("expected in-flight reconnect to queue instead of starting immediately")
}
want := now.Add(nodeOnlineRedeployCooldown)
if !fireAt.Equal(want) {
t.Fatalf("expected queued reconnect at %s, got %s", want, fireAt)
}
}