fix(clickhouse): harden R/W path P0–P3 (cleanup, durability, rollups)

Honest TTL cleanup semantics; enqueue-safe dedup with flush retry and writer
metrics; model insert hooks; latest-per-node and hourly metric/openresty
rollups; small-host pool/async defaults, traffic hourly TTL, and UV labeling.
This commit is contained in:
ryan
2026-07-10 10:34:04 +08:00
parent 9b3555c569
commit 160e63558f
37 changed files with 1779 additions and 269 deletions
+25 -11
View File
@@ -25,7 +25,7 @@ func newDedupSet() *dedupSet {
// markIfNew records key when it has not been seen within dedupTTL.
func (s *dedupSet) markIfNew(key string) bool {
if key == "" {
if s == nil || key == "" {
return false
}
@@ -33,19 +33,33 @@ func (s *dedupSet) markIfNew(key string) bool {
s.mu.Lock()
defer s.mu.Unlock()
// Periodically clean up all expired keys (e.g., every 30 seconds)
if now.Sub(s.lastCleanup) >= 30*time.Second {
for existing, expiresAt := range s.keys {
if now.After(expiresAt) {
delete(s.keys, existing)
}
}
s.lastCleanup = now
}
s.cleanupExpiredLocked(now)
if expiresAt, exists := s.keys[key]; exists && now.Before(expiresAt) {
return false
}
s.keys[key] = now.Add(dedupTTL)
return true
}
}
// unmark removes a key so a later enqueue or flush retry may accept it again.
func (s *dedupSet) unmark(key string) {
if s == nil || key == "" {
return
}
s.mu.Lock()
defer s.mu.Unlock()
delete(s.keys, key)
}
func (s *dedupSet) cleanupExpiredLocked(now time.Time) {
if now.Sub(s.lastCleanup) < 30*time.Second {
return
}
for existing, expiresAt := range s.keys {
if now.After(expiresAt) {
delete(s.keys, existing)
}
}
s.lastCleanup = now
}
+164 -2
View File
@@ -3,7 +3,16 @@
package chwriter
import "testing"
import (
"context"
"errors"
"sync"
"testing"
"time"
"github.com/Rain-kl/Wavelet/internal/db/batchwriter"
analyticsmodel "github.com/Rain-kl/Wavelet/internal/model/analytics"
)
func TestDedupSetMarkIfNew(t *testing.T) {
t.Parallel()
@@ -21,4 +30,157 @@ func TestDedupSetMarkIfNew(t *testing.T) {
if set.markIfNew("") {
t.Fatal("markIfNew() = true, want false on empty key")
}
}
}
func TestDedupSetUnmarkAllowsRetry(t *testing.T) {
t.Parallel()
set := newDedupSet()
if !set.markIfNew("k") {
t.Fatal("markIfNew() = false, want true")
}
set.unmark("k")
if !set.markIfNew("k") {
t.Fatal("markIfNew() after unmark = false, want true")
}
}
func TestQueueWithDedupDoesNotMarkWhenEnqueueFails(t *testing.T) {
t.Parallel()
cfg := batchwriter.DefaultConfig()
cfg.QueueSize = 1
cfg.MaxBatchSize = 10
cfg.FlushInterval = time.Hour
// Block the worker so the queue stays full after one enqueue.
block := make(chan struct{})
writer, err := batchwriter.New[int](cfg, func(context.Context, []int) error {
<-block
return nil
})
if err != nil {
t.Fatalf("New() error = %v", err)
}
writer.Start(context.Background())
t.Cleanup(func() {
close(block)
stopCtx, cancel := context.WithTimeout(context.Background(), time.Second)
defer cancel()
_ = writer.Stop(stopCtx)
})
// Fill the channel buffer (and the worker's current receive slot may empty one).
// Keep enqueueing until full so subsequent queueWithDedup fails.
for i := 0; i < cfg.QueueSize+2; i++ {
_ = writer.TryEnqueue(i)
if writer.IsFull() {
break
}
}
if !writer.IsFull() {
t.Fatal("writer not full after filling; cannot test enqueue failure path")
}
dedup := newDedupSet()
queueWithDedup(writer, dedup, "dedup-key", 99)
// Key must not remain marked after failed enqueue.
if !dedup.markIfNew("dedup-key") {
t.Fatal("dedup key still marked after failed enqueue; want unmark")
}
}
func TestQueueWithDedupMarksOnlyOnSuccess(t *testing.T) {
t.Parallel()
cfg := batchwriter.DefaultConfig()
cfg.MaxBatchSize = 100
cfg.FlushInterval = time.Hour
writer, err := batchwriter.New[int](cfg, func(context.Context, []int) error { return nil })
if err != nil {
t.Fatalf("New() error = %v", err)
}
writer.Start(context.Background())
t.Cleanup(func() {
stopCtx, cancel := context.WithTimeout(context.Background(), time.Second)
defer cancel()
_ = writer.Stop(stopCtx)
})
dedup := newDedupSet()
queueWithDedup(writer, dedup, "ok-key", 1)
if dedup.markIfNew("ok-key") {
t.Fatal("markIfNew() = true after successful enqueue, want false (key marked)")
}
}
func TestFlushErrorHandlerUnmarksKeys(t *testing.T) {
t.Parallel()
dedup := newDedupSet()
flushErr := errors.New("ch down")
var (
mu sync.Mutex
errCount int
)
cfg := batchwriter.Config{
Name: "test_obs",
QueueSize: 10,
MaxBatchSize: 1,
FlushInterval: time.Hour,
}
keyFn := func(s analyticsmodel.NodeMetricSnapshot) string {
return metricSnapshotKey(s)
}
writer, err := batchwriter.New(
cfg,
func(context.Context, []analyticsmodel.NodeMetricSnapshot) error { return flushErr },
batchwriter.WithFlushErrorHandler[analyticsmodel.NodeMetricSnapshot](func(_ context.Context, items []analyticsmodel.NodeMetricSnapshot, err error) {
mu.Lock()
errCount++
mu.Unlock()
for _, item := range items {
dedup.unmark(keyFn(item))
}
}),
)
if err != nil {
t.Fatalf("New() error = %v", err)
}
writer.Start(context.Background())
t.Cleanup(func() {
stopCtx, cancel := context.WithTimeout(context.Background(), time.Second)
defer cancel()
_ = writer.Stop(stopCtx)
})
item := analyticsmodel.NodeMetricSnapshot{
NodeID: "n1",
CapturedAt: time.Unix(1, 0).UTC(),
}
key := keyFn(item)
if !dedup.markIfNew(key) {
t.Fatal("markIfNew failed")
}
if !writer.TryEnqueue(item) {
t.Fatal("TryEnqueue failed")
}
deadline := time.Now().Add(time.Second)
for {
mu.Lock()
ready := errCount >= 1
mu.Unlock()
if ready || time.Now().After(deadline) {
break
}
time.Sleep(5 * time.Millisecond)
}
if !dedup.markIfNew(key) {
t.Fatal("key still marked after flush error unmark; want available for retry")
}
}
+159 -57
View File
@@ -14,6 +14,7 @@ import (
"github.com/Rain-kl/Wavelet/internal/config"
"github.com/Rain-kl/Wavelet/internal/db/batchwriter"
"github.com/Rain-kl/Wavelet/internal/lifecycle"
"github.com/Rain-kl/Wavelet/internal/model"
analyticsmodel "github.com/Rain-kl/Wavelet/internal/model/analytics"
analyticsrepo "github.com/Rain-kl/Wavelet/internal/repository/analytics"
"github.com/Rain-kl/Wavelet/pkg/logger"
@@ -33,6 +34,10 @@ const (
nodeAccessLogMinBatchSize = 50
nodeAccessLogFlushEvery = 2 * time.Second
nodeAccessLogMaxFlushWait = 5 * time.Second
// flushAttempts is total tries (1 initial + short retries) before giving up a batch.
flushAttempts = 2
flushRetryBackoff = 50 * time.Millisecond
)
var (
@@ -65,11 +70,36 @@ func Init(ctx context.Context) {
frpsDedup = newDedupSet()
frpcDedup = newDedupSet()
metricSnapshotWriter = mustNewObservabilityWriter("metric_snapshots", analyticsrepo.BatchInsertNodeMetricSnapshots)
requestReportWriter = mustNewObservabilityWriter("request_reports", analyticsrepo.BatchInsertNodeRequestReports)
openrestyWriter = mustNewObservabilityWriter("openresty_obs", analyticsrepo.BatchInsertNodeObsOpenresty)
frpsWriter = mustNewObservabilityWriter("frps_obs", analyticsrepo.BatchInsertNodeObsFrps)
frpcWriter = mustNewObservabilityWriter("frpc_obs", analyticsrepo.BatchInsertNodeObsFrpc)
metricSnapshotWriter = mustNewObservabilityWriter(
"metric_snapshots",
withFlushRetries(analyticsrepo.BatchInsertNodeMetricSnapshots),
metricSnapshotDedup,
metricSnapshotKey,
)
requestReportWriter = mustNewObservabilityWriter(
"request_reports",
withFlushRetries(analyticsrepo.BatchInsertNodeRequestReports),
requestReportDedup,
requestReportKey,
)
openrestyWriter = mustNewObservabilityWriter(
"openresty_obs",
withFlushRetries(analyticsrepo.BatchInsertNodeObsOpenresty),
openrestyDedup,
openrestyKey,
)
frpsWriter = mustNewObservabilityWriter(
"frps_obs",
withFlushRetries(analyticsrepo.BatchInsertNodeObsFrps),
frpsDedup,
frpsKey,
)
frpcWriter = mustNewObservabilityWriter(
"frpc_obs",
withFlushRetries(analyticsrepo.BatchInsertNodeObsFrpc),
frpcDedup,
frpcKey,
)
nodeAccessLogWriter = mustNewNodeAccessLogWriter()
metricSnapshotWriter.Start(ctx)
@@ -79,6 +109,7 @@ func Init(ctx context.Context) {
frpcWriter.Start(ctx)
nodeAccessLogWriter.Start(ctx)
wireModelInsertHooks()
lifecycle.OnShutdown("openflare_chwriter", Stop)
})
}
@@ -108,69 +139,49 @@ func Stop(ctx context.Context) error {
return firstErr
}
// WriterStats returns queue depth and failure counters for all OpenFlare writers.
func WriterStats() []batchwriter.Stats {
writers := []statsProvider{
metricSnapshotWriter,
requestReportWriter,
openrestyWriter,
frpsWriter,
frpcWriter,
nodeAccessLogWriter,
}
out := make([]batchwriter.Stats, 0, len(writers))
for _, w := range writers {
if w == nil {
continue
}
out = append(out, w.Stats())
}
return out
}
// QueueMetricSnapshot enqueues a metric snapshot for asynchronous flush.
func QueueMetricSnapshot(snapshot analyticsmodel.NodeMetricSnapshot) {
if metricSnapshotWriter == nil {
return
}
key := fmt.Sprintf("%s|%d", snapshot.NodeID, snapshot.CapturedAt.UTC().UnixNano())
if !metricSnapshotDedup.markIfNew(key) {
return
}
metricSnapshotWriter.TryEnqueue(snapshot)
queueWithDedup(metricSnapshotWriter, metricSnapshotDedup, metricSnapshotKey(snapshot), snapshot)
}
// QueueRequestReport enqueues a request report for asynchronous flush.
func QueueRequestReport(report analyticsmodel.NodeRequestReport) {
if requestReportWriter == nil {
return
}
key := fmt.Sprintf(
"%s|%d|%d",
report.NodeID,
report.WindowStartedAt.UTC().UnixNano(),
report.WindowEndedAt.UTC().UnixNano(),
)
if !requestReportDedup.markIfNew(key) {
return
}
requestReportWriter.TryEnqueue(report)
queueWithDedup(requestReportWriter, requestReportDedup, requestReportKey(report), report)
}
// QueueOpenrestyObservation enqueues an OpenResty observation for asynchronous flush.
func QueueOpenrestyObservation(observation analyticsmodel.NodeObsOpenresty) {
if openrestyWriter == nil {
return
}
key := fmt.Sprintf("%s|%d", observation.NodeID, observation.CapturedAt.UTC().UnixNano())
if !openrestyDedup.markIfNew(key) {
return
}
openrestyWriter.TryEnqueue(observation)
queueWithDedup(openrestyWriter, openrestyDedup, openrestyKey(observation), observation)
}
// QueueFrpsObservation enqueues an FRPS observation for asynchronous flush.
func QueueFrpsObservation(observation analyticsmodel.NodeObsFrps) {
if frpsWriter == nil {
return
}
key := fmt.Sprintf("%s|%d", observation.NodeID, observation.CapturedAt.UTC().UnixNano())
if !frpsDedup.markIfNew(key) {
return
}
frpsWriter.TryEnqueue(observation)
queueWithDedup(frpsWriter, frpsDedup, frpsKey(observation), observation)
}
// QueueFrpcObservation enqueues an FRPC observation for asynchronous flush.
func QueueFrpcObservation(observation analyticsmodel.NodeObsFrpc) {
if frpcWriter == nil {
return
}
key := fmt.Sprintf("%s|%d", observation.NodeID, observation.CapturedAt.UTC().UnixNano())
if !frpcDedup.markIfNew(key) {
return
}
frpcWriter.TryEnqueue(observation)
queueWithDedup(frpcWriter, frpcDedup, frpcKey(observation), observation)
}
// QueueNodeAccessLogs enqueues node access logs for asynchronous flush.
@@ -183,7 +194,26 @@ func QueueNodeAccessLogs(logs []analyticsmodel.NodeAccessLog) {
}
}
func mustNewObservabilityWriter[T any](name string, flush batchwriter.FlushFunc[T]) *batchwriter.Writer[T] {
func queueWithDedup[T any](writer *batchwriter.Writer[T], dedup *dedupSet, key string, item T) {
if writer == nil {
return
}
// Mark first so concurrent duplicates still collapse; release on enqueue failure
// so a full queue does not permanently suppress the item.
if !dedup.markIfNew(key) {
return
}
if !writer.TryEnqueue(item) {
dedup.unmark(key)
}
}
func mustNewObservabilityWriter[T any](
name string,
flush batchwriter.FlushFunc[T],
dedup *dedupSet,
keyFn func(T) string,
) *batchwriter.Writer[T] {
cfg := batchwriter.Config{
Name: name,
QueueSize: observabilityQueueSize,
@@ -196,8 +226,14 @@ func mustNewObservabilityWriter[T any](name string, flush batchwriter.FlushFunc[
cfg,
flush,
withObservabilityDropHandler[T](name),
batchwriter.WithFlushErrorHandler[T](func(ctx context.Context, batchSize int, err error) {
logger.ErrorF(ctx, "[OpenFlare] flush %s failed (batch=%d): %v", name, batchSize, err)
batchwriter.WithFlushErrorHandler[T](func(ctx context.Context, items []T, err error) {
logger.ErrorF(ctx, "[OpenFlare] flush %s failed (batch=%d): %v", name, len(items), err)
if dedup == nil || keyFn == nil {
return
}
for _, item := range items {
dedup.unmark(keyFn(item))
}
}),
)
if err != nil {
@@ -215,12 +251,14 @@ func mustNewNodeAccessLogWriter() *batchwriter.Writer[analyticsmodel.NodeAccessL
FlushInterval: nodeAccessLogFlushEvery,
MaxFlushWait: nodeAccessLogMaxFlushWait,
}
writer, err := batchwriter.New[analyticsmodel.NodeAccessLog](cfg, analyticsrepo.BatchInsertNodeAccessLogs,
writer, err := batchwriter.New[analyticsmodel.NodeAccessLog](
cfg,
withFlushRetries(analyticsrepo.BatchInsertNodeAccessLogs),
batchwriter.WithDropHandler[analyticsmodel.NodeAccessLog](func(item analyticsmodel.NodeAccessLog) {
logger.WarnF(context.Background(), "[OpenFlare] node access log queue full, dropping log for node %s path %s", item.NodeID, item.Path)
}),
batchwriter.WithFlushErrorHandler[analyticsmodel.NodeAccessLog](func(ctx context.Context, batchSize int, err error) {
logger.ErrorF(ctx, "[OpenFlare] flush node access logs failed (batch=%d): %v", batchSize, err)
batchwriter.WithFlushErrorHandler[analyticsmodel.NodeAccessLog](func(ctx context.Context, items []analyticsmodel.NodeAccessLog, err error) {
logger.ErrorF(ctx, "[OpenFlare] flush node access logs failed (batch=%d): %v", len(items), err)
}),
)
if err != nil {
@@ -235,10 +273,74 @@ func withObservabilityDropHandler[T any](name string) batchwriter.Option[T] {
})
}
// withFlushRetries wraps a flush function with a short retry to ride out brief CH blips.
func withFlushRetries[T any](flush batchwriter.FlushFunc[T]) batchwriter.FlushFunc[T] {
return func(ctx context.Context, items []T) error {
var err error
for attempt := 1; attempt <= flushAttempts; attempt++ {
err = flush(ctx, items)
if err == nil {
return nil
}
if attempt == flushAttempts {
break
}
select {
case <-ctx.Done():
return ctx.Err()
case <-time.After(flushRetryBackoff * time.Duration(attempt)):
}
}
return err
}
}
func wireModelInsertHooks() {
model.SetObservabilityInsertHooks(model.ObservabilityInsertHooks{
QueueMetricSnapshot: QueueMetricSnapshot,
QueueRequestReport: QueueRequestReport,
QueueOpenrestyObservation: QueueOpenrestyObservation,
QueueFrpsObservation: QueueFrpsObservation,
QueueFrpcObservation: QueueFrpcObservation,
})
model.SetAccessLogInsertHooks(model.AccessLogInsertHooks{
QueueNodeAccessLogs: QueueNodeAccessLogs,
})
}
func metricSnapshotKey(snapshot analyticsmodel.NodeMetricSnapshot) string {
return fmt.Sprintf("%s|%d", snapshot.NodeID, snapshot.CapturedAt.UTC().UnixNano())
}
func requestReportKey(report analyticsmodel.NodeRequestReport) string {
return fmt.Sprintf(
"%s|%d|%d",
report.NodeID,
report.WindowStartedAt.UTC().UnixNano(),
report.WindowEndedAt.UTC().UnixNano(),
)
}
func openrestyKey(observation analyticsmodel.NodeObsOpenresty) string {
return fmt.Sprintf("%s|%d", observation.NodeID, observation.CapturedAt.UTC().UnixNano())
}
func frpsKey(observation analyticsmodel.NodeObsFrps) string {
return fmt.Sprintf("%s|%d", observation.NodeID, observation.CapturedAt.UTC().UnixNano())
}
func frpcKey(observation analyticsmodel.NodeObsFrpc) string {
return fmt.Sprintf("%s|%d", observation.NodeID, observation.CapturedAt.UTC().UnixNano())
}
type batchStopper interface {
Stop(ctx context.Context) error
}
type statsProvider interface {
Stats() batchwriter.Stats
}
func running() bool {
return metricSnapshotWriter != nil && metricSnapshotWriter.Running()
}
}