mirror of
https://github.com/Rain-kl/OpenFlare.git
synced 2026-10-01 22:46:38 +08:00
fix(clickhouse): harden R/W path P0–P3 (cleanup, durability, rollups)
Honest TTL cleanup semantics; enqueue-safe dedup with flush retry and writer metrics; model insert hooks; latest-per-node and hourly metric/openresty rollups; small-host pool/async defaults, traffic hourly TTL, and UV labeling.
This commit is contained in:
@@ -6,16 +6,19 @@ package status
|
||||
import (
|
||||
"net/http"
|
||||
|
||||
"github.com/Rain-kl/Wavelet/internal/apps/openflare/chwriter"
|
||||
"github.com/Rain-kl/Wavelet/internal/apps/risk_control"
|
||||
"github.com/Rain-kl/Wavelet/internal/common/response"
|
||||
"github.com/Rain-kl/Wavelet/internal/config"
|
||||
"github.com/Rain-kl/Wavelet/internal/db"
|
||||
"github.com/Rain-kl/Wavelet/internal/db/batchwriter"
|
||||
analyticsrepo "github.com/Rain-kl/Wavelet/internal/repository/analytics"
|
||||
"github.com/gin-gonic/gin"
|
||||
)
|
||||
|
||||
// GetClickHouseStatus returns ClickHouse operational metrics for administrators.
|
||||
// @Summary 获取 ClickHouse 运行指标
|
||||
// @Description 返回 ClickHouse parts、mutation、async_insert 队列等运维指标,需要管理员权限
|
||||
// @Description 返回 ClickHouse parts、mutation、async_insert 队列及进程内 batch writer 指标,需要管理员权限
|
||||
// @Tags admin
|
||||
// @Produce json
|
||||
// @Security SessionCookie
|
||||
@@ -36,5 +39,15 @@ func GetClickHouseStatus(c *gin.Context) {
|
||||
response.AbortInternal(c, "获取 ClickHouse 运行指标失败")
|
||||
return
|
||||
}
|
||||
stats.BatchWriters = collectBatchWriterStats()
|
||||
c.JSON(http.StatusOK, response.OK(stats))
|
||||
}
|
||||
}
|
||||
|
||||
func collectBatchWriterStats() []batchwriter.Stats {
|
||||
out := chwriter.WriterStats()
|
||||
if out == nil {
|
||||
out = make([]batchwriter.Stats, 0, 1)
|
||||
}
|
||||
out = append(out, risk_control.LogWriterStats())
|
||||
return out
|
||||
}
|
||||
|
||||
@@ -25,7 +25,7 @@ func newDedupSet() *dedupSet {
|
||||
|
||||
// markIfNew records key when it has not been seen within dedupTTL.
|
||||
func (s *dedupSet) markIfNew(key string) bool {
|
||||
if key == "" {
|
||||
if s == nil || key == "" {
|
||||
return false
|
||||
}
|
||||
|
||||
@@ -33,19 +33,33 @@ func (s *dedupSet) markIfNew(key string) bool {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
|
||||
// Periodically clean up all expired keys (e.g., every 30 seconds)
|
||||
if now.Sub(s.lastCleanup) >= 30*time.Second {
|
||||
for existing, expiresAt := range s.keys {
|
||||
if now.After(expiresAt) {
|
||||
delete(s.keys, existing)
|
||||
}
|
||||
}
|
||||
s.lastCleanup = now
|
||||
}
|
||||
s.cleanupExpiredLocked(now)
|
||||
|
||||
if expiresAt, exists := s.keys[key]; exists && now.Before(expiresAt) {
|
||||
return false
|
||||
}
|
||||
s.keys[key] = now.Add(dedupTTL)
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
// unmark removes a key so a later enqueue or flush retry may accept it again.
|
||||
func (s *dedupSet) unmark(key string) {
|
||||
if s == nil || key == "" {
|
||||
return
|
||||
}
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
delete(s.keys, key)
|
||||
}
|
||||
|
||||
func (s *dedupSet) cleanupExpiredLocked(now time.Time) {
|
||||
if now.Sub(s.lastCleanup) < 30*time.Second {
|
||||
return
|
||||
}
|
||||
for existing, expiresAt := range s.keys {
|
||||
if now.After(expiresAt) {
|
||||
delete(s.keys, existing)
|
||||
}
|
||||
}
|
||||
s.lastCleanup = now
|
||||
}
|
||||
|
||||
@@ -3,7 +3,16 @@
|
||||
|
||||
package chwriter
|
||||
|
||||
import "testing"
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/Rain-kl/Wavelet/internal/db/batchwriter"
|
||||
analyticsmodel "github.com/Rain-kl/Wavelet/internal/model/analytics"
|
||||
)
|
||||
|
||||
func TestDedupSetMarkIfNew(t *testing.T) {
|
||||
t.Parallel()
|
||||
@@ -21,4 +30,157 @@ func TestDedupSetMarkIfNew(t *testing.T) {
|
||||
if set.markIfNew("") {
|
||||
t.Fatal("markIfNew() = true, want false on empty key")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestDedupSetUnmarkAllowsRetry(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
set := newDedupSet()
|
||||
if !set.markIfNew("k") {
|
||||
t.Fatal("markIfNew() = false, want true")
|
||||
}
|
||||
set.unmark("k")
|
||||
if !set.markIfNew("k") {
|
||||
t.Fatal("markIfNew() after unmark = false, want true")
|
||||
}
|
||||
}
|
||||
|
||||
func TestQueueWithDedupDoesNotMarkWhenEnqueueFails(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
cfg := batchwriter.DefaultConfig()
|
||||
cfg.QueueSize = 1
|
||||
cfg.MaxBatchSize = 10
|
||||
cfg.FlushInterval = time.Hour
|
||||
|
||||
// Block the worker so the queue stays full after one enqueue.
|
||||
block := make(chan struct{})
|
||||
writer, err := batchwriter.New[int](cfg, func(context.Context, []int) error {
|
||||
<-block
|
||||
return nil
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("New() error = %v", err)
|
||||
}
|
||||
writer.Start(context.Background())
|
||||
t.Cleanup(func() {
|
||||
close(block)
|
||||
stopCtx, cancel := context.WithTimeout(context.Background(), time.Second)
|
||||
defer cancel()
|
||||
_ = writer.Stop(stopCtx)
|
||||
})
|
||||
|
||||
// Fill the channel buffer (and the worker's current receive slot may empty one).
|
||||
// Keep enqueueing until full so subsequent queueWithDedup fails.
|
||||
for i := 0; i < cfg.QueueSize+2; i++ {
|
||||
_ = writer.TryEnqueue(i)
|
||||
if writer.IsFull() {
|
||||
break
|
||||
}
|
||||
}
|
||||
if !writer.IsFull() {
|
||||
t.Fatal("writer not full after filling; cannot test enqueue failure path")
|
||||
}
|
||||
|
||||
dedup := newDedupSet()
|
||||
queueWithDedup(writer, dedup, "dedup-key", 99)
|
||||
// Key must not remain marked after failed enqueue.
|
||||
if !dedup.markIfNew("dedup-key") {
|
||||
t.Fatal("dedup key still marked after failed enqueue; want unmark")
|
||||
}
|
||||
}
|
||||
|
||||
func TestQueueWithDedupMarksOnlyOnSuccess(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
cfg := batchwriter.DefaultConfig()
|
||||
cfg.MaxBatchSize = 100
|
||||
cfg.FlushInterval = time.Hour
|
||||
|
||||
writer, err := batchwriter.New[int](cfg, func(context.Context, []int) error { return nil })
|
||||
if err != nil {
|
||||
t.Fatalf("New() error = %v", err)
|
||||
}
|
||||
writer.Start(context.Background())
|
||||
t.Cleanup(func() {
|
||||
stopCtx, cancel := context.WithTimeout(context.Background(), time.Second)
|
||||
defer cancel()
|
||||
_ = writer.Stop(stopCtx)
|
||||
})
|
||||
|
||||
dedup := newDedupSet()
|
||||
queueWithDedup(writer, dedup, "ok-key", 1)
|
||||
if dedup.markIfNew("ok-key") {
|
||||
t.Fatal("markIfNew() = true after successful enqueue, want false (key marked)")
|
||||
}
|
||||
}
|
||||
|
||||
func TestFlushErrorHandlerUnmarksKeys(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dedup := newDedupSet()
|
||||
flushErr := errors.New("ch down")
|
||||
|
||||
var (
|
||||
mu sync.Mutex
|
||||
errCount int
|
||||
)
|
||||
|
||||
cfg := batchwriter.Config{
|
||||
Name: "test_obs",
|
||||
QueueSize: 10,
|
||||
MaxBatchSize: 1,
|
||||
FlushInterval: time.Hour,
|
||||
}
|
||||
keyFn := func(s analyticsmodel.NodeMetricSnapshot) string {
|
||||
return metricSnapshotKey(s)
|
||||
}
|
||||
writer, err := batchwriter.New(
|
||||
cfg,
|
||||
func(context.Context, []analyticsmodel.NodeMetricSnapshot) error { return flushErr },
|
||||
batchwriter.WithFlushErrorHandler[analyticsmodel.NodeMetricSnapshot](func(_ context.Context, items []analyticsmodel.NodeMetricSnapshot, err error) {
|
||||
mu.Lock()
|
||||
errCount++
|
||||
mu.Unlock()
|
||||
for _, item := range items {
|
||||
dedup.unmark(keyFn(item))
|
||||
}
|
||||
}),
|
||||
)
|
||||
if err != nil {
|
||||
t.Fatalf("New() error = %v", err)
|
||||
}
|
||||
writer.Start(context.Background())
|
||||
t.Cleanup(func() {
|
||||
stopCtx, cancel := context.WithTimeout(context.Background(), time.Second)
|
||||
defer cancel()
|
||||
_ = writer.Stop(stopCtx)
|
||||
})
|
||||
|
||||
item := analyticsmodel.NodeMetricSnapshot{
|
||||
NodeID: "n1",
|
||||
CapturedAt: time.Unix(1, 0).UTC(),
|
||||
}
|
||||
key := keyFn(item)
|
||||
if !dedup.markIfNew(key) {
|
||||
t.Fatal("markIfNew failed")
|
||||
}
|
||||
if !writer.TryEnqueue(item) {
|
||||
t.Fatal("TryEnqueue failed")
|
||||
}
|
||||
|
||||
deadline := time.Now().Add(time.Second)
|
||||
for {
|
||||
mu.Lock()
|
||||
ready := errCount >= 1
|
||||
mu.Unlock()
|
||||
if ready || time.Now().After(deadline) {
|
||||
break
|
||||
}
|
||||
time.Sleep(5 * time.Millisecond)
|
||||
}
|
||||
|
||||
if !dedup.markIfNew(key) {
|
||||
t.Fatal("key still marked after flush error unmark; want available for retry")
|
||||
}
|
||||
}
|
||||
|
||||
@@ -14,6 +14,7 @@ import (
|
||||
"github.com/Rain-kl/Wavelet/internal/config"
|
||||
"github.com/Rain-kl/Wavelet/internal/db/batchwriter"
|
||||
"github.com/Rain-kl/Wavelet/internal/lifecycle"
|
||||
"github.com/Rain-kl/Wavelet/internal/model"
|
||||
analyticsmodel "github.com/Rain-kl/Wavelet/internal/model/analytics"
|
||||
analyticsrepo "github.com/Rain-kl/Wavelet/internal/repository/analytics"
|
||||
"github.com/Rain-kl/Wavelet/pkg/logger"
|
||||
@@ -33,6 +34,10 @@ const (
|
||||
nodeAccessLogMinBatchSize = 50
|
||||
nodeAccessLogFlushEvery = 2 * time.Second
|
||||
nodeAccessLogMaxFlushWait = 5 * time.Second
|
||||
|
||||
// flushAttempts is total tries (1 initial + short retries) before giving up a batch.
|
||||
flushAttempts = 2
|
||||
flushRetryBackoff = 50 * time.Millisecond
|
||||
)
|
||||
|
||||
var (
|
||||
@@ -65,11 +70,36 @@ func Init(ctx context.Context) {
|
||||
frpsDedup = newDedupSet()
|
||||
frpcDedup = newDedupSet()
|
||||
|
||||
metricSnapshotWriter = mustNewObservabilityWriter("metric_snapshots", analyticsrepo.BatchInsertNodeMetricSnapshots)
|
||||
requestReportWriter = mustNewObservabilityWriter("request_reports", analyticsrepo.BatchInsertNodeRequestReports)
|
||||
openrestyWriter = mustNewObservabilityWriter("openresty_obs", analyticsrepo.BatchInsertNodeObsOpenresty)
|
||||
frpsWriter = mustNewObservabilityWriter("frps_obs", analyticsrepo.BatchInsertNodeObsFrps)
|
||||
frpcWriter = mustNewObservabilityWriter("frpc_obs", analyticsrepo.BatchInsertNodeObsFrpc)
|
||||
metricSnapshotWriter = mustNewObservabilityWriter(
|
||||
"metric_snapshots",
|
||||
withFlushRetries(analyticsrepo.BatchInsertNodeMetricSnapshots),
|
||||
metricSnapshotDedup,
|
||||
metricSnapshotKey,
|
||||
)
|
||||
requestReportWriter = mustNewObservabilityWriter(
|
||||
"request_reports",
|
||||
withFlushRetries(analyticsrepo.BatchInsertNodeRequestReports),
|
||||
requestReportDedup,
|
||||
requestReportKey,
|
||||
)
|
||||
openrestyWriter = mustNewObservabilityWriter(
|
||||
"openresty_obs",
|
||||
withFlushRetries(analyticsrepo.BatchInsertNodeObsOpenresty),
|
||||
openrestyDedup,
|
||||
openrestyKey,
|
||||
)
|
||||
frpsWriter = mustNewObservabilityWriter(
|
||||
"frps_obs",
|
||||
withFlushRetries(analyticsrepo.BatchInsertNodeObsFrps),
|
||||
frpsDedup,
|
||||
frpsKey,
|
||||
)
|
||||
frpcWriter = mustNewObservabilityWriter(
|
||||
"frpc_obs",
|
||||
withFlushRetries(analyticsrepo.BatchInsertNodeObsFrpc),
|
||||
frpcDedup,
|
||||
frpcKey,
|
||||
)
|
||||
nodeAccessLogWriter = mustNewNodeAccessLogWriter()
|
||||
|
||||
metricSnapshotWriter.Start(ctx)
|
||||
@@ -79,6 +109,7 @@ func Init(ctx context.Context) {
|
||||
frpcWriter.Start(ctx)
|
||||
nodeAccessLogWriter.Start(ctx)
|
||||
|
||||
wireModelInsertHooks()
|
||||
lifecycle.OnShutdown("openflare_chwriter", Stop)
|
||||
})
|
||||
}
|
||||
@@ -108,69 +139,49 @@ func Stop(ctx context.Context) error {
|
||||
return firstErr
|
||||
}
|
||||
|
||||
// WriterStats returns queue depth and failure counters for all OpenFlare writers.
|
||||
func WriterStats() []batchwriter.Stats {
|
||||
writers := []statsProvider{
|
||||
metricSnapshotWriter,
|
||||
requestReportWriter,
|
||||
openrestyWriter,
|
||||
frpsWriter,
|
||||
frpcWriter,
|
||||
nodeAccessLogWriter,
|
||||
}
|
||||
out := make([]batchwriter.Stats, 0, len(writers))
|
||||
for _, w := range writers {
|
||||
if w == nil {
|
||||
continue
|
||||
}
|
||||
out = append(out, w.Stats())
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// QueueMetricSnapshot enqueues a metric snapshot for asynchronous flush.
|
||||
func QueueMetricSnapshot(snapshot analyticsmodel.NodeMetricSnapshot) {
|
||||
if metricSnapshotWriter == nil {
|
||||
return
|
||||
}
|
||||
key := fmt.Sprintf("%s|%d", snapshot.NodeID, snapshot.CapturedAt.UTC().UnixNano())
|
||||
if !metricSnapshotDedup.markIfNew(key) {
|
||||
return
|
||||
}
|
||||
metricSnapshotWriter.TryEnqueue(snapshot)
|
||||
queueWithDedup(metricSnapshotWriter, metricSnapshotDedup, metricSnapshotKey(snapshot), snapshot)
|
||||
}
|
||||
|
||||
// QueueRequestReport enqueues a request report for asynchronous flush.
|
||||
func QueueRequestReport(report analyticsmodel.NodeRequestReport) {
|
||||
if requestReportWriter == nil {
|
||||
return
|
||||
}
|
||||
key := fmt.Sprintf(
|
||||
"%s|%d|%d",
|
||||
report.NodeID,
|
||||
report.WindowStartedAt.UTC().UnixNano(),
|
||||
report.WindowEndedAt.UTC().UnixNano(),
|
||||
)
|
||||
if !requestReportDedup.markIfNew(key) {
|
||||
return
|
||||
}
|
||||
requestReportWriter.TryEnqueue(report)
|
||||
queueWithDedup(requestReportWriter, requestReportDedup, requestReportKey(report), report)
|
||||
}
|
||||
|
||||
// QueueOpenrestyObservation enqueues an OpenResty observation for asynchronous flush.
|
||||
func QueueOpenrestyObservation(observation analyticsmodel.NodeObsOpenresty) {
|
||||
if openrestyWriter == nil {
|
||||
return
|
||||
}
|
||||
key := fmt.Sprintf("%s|%d", observation.NodeID, observation.CapturedAt.UTC().UnixNano())
|
||||
if !openrestyDedup.markIfNew(key) {
|
||||
return
|
||||
}
|
||||
openrestyWriter.TryEnqueue(observation)
|
||||
queueWithDedup(openrestyWriter, openrestyDedup, openrestyKey(observation), observation)
|
||||
}
|
||||
|
||||
// QueueFrpsObservation enqueues an FRPS observation for asynchronous flush.
|
||||
func QueueFrpsObservation(observation analyticsmodel.NodeObsFrps) {
|
||||
if frpsWriter == nil {
|
||||
return
|
||||
}
|
||||
key := fmt.Sprintf("%s|%d", observation.NodeID, observation.CapturedAt.UTC().UnixNano())
|
||||
if !frpsDedup.markIfNew(key) {
|
||||
return
|
||||
}
|
||||
frpsWriter.TryEnqueue(observation)
|
||||
queueWithDedup(frpsWriter, frpsDedup, frpsKey(observation), observation)
|
||||
}
|
||||
|
||||
// QueueFrpcObservation enqueues an FRPC observation for asynchronous flush.
|
||||
func QueueFrpcObservation(observation analyticsmodel.NodeObsFrpc) {
|
||||
if frpcWriter == nil {
|
||||
return
|
||||
}
|
||||
key := fmt.Sprintf("%s|%d", observation.NodeID, observation.CapturedAt.UTC().UnixNano())
|
||||
if !frpcDedup.markIfNew(key) {
|
||||
return
|
||||
}
|
||||
frpcWriter.TryEnqueue(observation)
|
||||
queueWithDedup(frpcWriter, frpcDedup, frpcKey(observation), observation)
|
||||
}
|
||||
|
||||
// QueueNodeAccessLogs enqueues node access logs for asynchronous flush.
|
||||
@@ -183,7 +194,26 @@ func QueueNodeAccessLogs(logs []analyticsmodel.NodeAccessLog) {
|
||||
}
|
||||
}
|
||||
|
||||
func mustNewObservabilityWriter[T any](name string, flush batchwriter.FlushFunc[T]) *batchwriter.Writer[T] {
|
||||
func queueWithDedup[T any](writer *batchwriter.Writer[T], dedup *dedupSet, key string, item T) {
|
||||
if writer == nil {
|
||||
return
|
||||
}
|
||||
// Mark first so concurrent duplicates still collapse; release on enqueue failure
|
||||
// so a full queue does not permanently suppress the item.
|
||||
if !dedup.markIfNew(key) {
|
||||
return
|
||||
}
|
||||
if !writer.TryEnqueue(item) {
|
||||
dedup.unmark(key)
|
||||
}
|
||||
}
|
||||
|
||||
func mustNewObservabilityWriter[T any](
|
||||
name string,
|
||||
flush batchwriter.FlushFunc[T],
|
||||
dedup *dedupSet,
|
||||
keyFn func(T) string,
|
||||
) *batchwriter.Writer[T] {
|
||||
cfg := batchwriter.Config{
|
||||
Name: name,
|
||||
QueueSize: observabilityQueueSize,
|
||||
@@ -196,8 +226,14 @@ func mustNewObservabilityWriter[T any](name string, flush batchwriter.FlushFunc[
|
||||
cfg,
|
||||
flush,
|
||||
withObservabilityDropHandler[T](name),
|
||||
batchwriter.WithFlushErrorHandler[T](func(ctx context.Context, batchSize int, err error) {
|
||||
logger.ErrorF(ctx, "[OpenFlare] flush %s failed (batch=%d): %v", name, batchSize, err)
|
||||
batchwriter.WithFlushErrorHandler[T](func(ctx context.Context, items []T, err error) {
|
||||
logger.ErrorF(ctx, "[OpenFlare] flush %s failed (batch=%d): %v", name, len(items), err)
|
||||
if dedup == nil || keyFn == nil {
|
||||
return
|
||||
}
|
||||
for _, item := range items {
|
||||
dedup.unmark(keyFn(item))
|
||||
}
|
||||
}),
|
||||
)
|
||||
if err != nil {
|
||||
@@ -215,12 +251,14 @@ func mustNewNodeAccessLogWriter() *batchwriter.Writer[analyticsmodel.NodeAccessL
|
||||
FlushInterval: nodeAccessLogFlushEvery,
|
||||
MaxFlushWait: nodeAccessLogMaxFlushWait,
|
||||
}
|
||||
writer, err := batchwriter.New[analyticsmodel.NodeAccessLog](cfg, analyticsrepo.BatchInsertNodeAccessLogs,
|
||||
writer, err := batchwriter.New[analyticsmodel.NodeAccessLog](
|
||||
cfg,
|
||||
withFlushRetries(analyticsrepo.BatchInsertNodeAccessLogs),
|
||||
batchwriter.WithDropHandler[analyticsmodel.NodeAccessLog](func(item analyticsmodel.NodeAccessLog) {
|
||||
logger.WarnF(context.Background(), "[OpenFlare] node access log queue full, dropping log for node %s path %s", item.NodeID, item.Path)
|
||||
}),
|
||||
batchwriter.WithFlushErrorHandler[analyticsmodel.NodeAccessLog](func(ctx context.Context, batchSize int, err error) {
|
||||
logger.ErrorF(ctx, "[OpenFlare] flush node access logs failed (batch=%d): %v", batchSize, err)
|
||||
batchwriter.WithFlushErrorHandler[analyticsmodel.NodeAccessLog](func(ctx context.Context, items []analyticsmodel.NodeAccessLog, err error) {
|
||||
logger.ErrorF(ctx, "[OpenFlare] flush node access logs failed (batch=%d): %v", len(items), err)
|
||||
}),
|
||||
)
|
||||
if err != nil {
|
||||
@@ -235,10 +273,74 @@ func withObservabilityDropHandler[T any](name string) batchwriter.Option[T] {
|
||||
})
|
||||
}
|
||||
|
||||
// withFlushRetries wraps a flush function with a short retry to ride out brief CH blips.
|
||||
func withFlushRetries[T any](flush batchwriter.FlushFunc[T]) batchwriter.FlushFunc[T] {
|
||||
return func(ctx context.Context, items []T) error {
|
||||
var err error
|
||||
for attempt := 1; attempt <= flushAttempts; attempt++ {
|
||||
err = flush(ctx, items)
|
||||
if err == nil {
|
||||
return nil
|
||||
}
|
||||
if attempt == flushAttempts {
|
||||
break
|
||||
}
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return ctx.Err()
|
||||
case <-time.After(flushRetryBackoff * time.Duration(attempt)):
|
||||
}
|
||||
}
|
||||
return err
|
||||
}
|
||||
}
|
||||
|
||||
func wireModelInsertHooks() {
|
||||
model.SetObservabilityInsertHooks(model.ObservabilityInsertHooks{
|
||||
QueueMetricSnapshot: QueueMetricSnapshot,
|
||||
QueueRequestReport: QueueRequestReport,
|
||||
QueueOpenrestyObservation: QueueOpenrestyObservation,
|
||||
QueueFrpsObservation: QueueFrpsObservation,
|
||||
QueueFrpcObservation: QueueFrpcObservation,
|
||||
})
|
||||
model.SetAccessLogInsertHooks(model.AccessLogInsertHooks{
|
||||
QueueNodeAccessLogs: QueueNodeAccessLogs,
|
||||
})
|
||||
}
|
||||
|
||||
func metricSnapshotKey(snapshot analyticsmodel.NodeMetricSnapshot) string {
|
||||
return fmt.Sprintf("%s|%d", snapshot.NodeID, snapshot.CapturedAt.UTC().UnixNano())
|
||||
}
|
||||
|
||||
func requestReportKey(report analyticsmodel.NodeRequestReport) string {
|
||||
return fmt.Sprintf(
|
||||
"%s|%d|%d",
|
||||
report.NodeID,
|
||||
report.WindowStartedAt.UTC().UnixNano(),
|
||||
report.WindowEndedAt.UTC().UnixNano(),
|
||||
)
|
||||
}
|
||||
|
||||
func openrestyKey(observation analyticsmodel.NodeObsOpenresty) string {
|
||||
return fmt.Sprintf("%s|%d", observation.NodeID, observation.CapturedAt.UTC().UnixNano())
|
||||
}
|
||||
|
||||
func frpsKey(observation analyticsmodel.NodeObsFrps) string {
|
||||
return fmt.Sprintf("%s|%d", observation.NodeID, observation.CapturedAt.UTC().UnixNano())
|
||||
}
|
||||
|
||||
func frpcKey(observation analyticsmodel.NodeObsFrpc) string {
|
||||
return fmt.Sprintf("%s|%d", observation.NodeID, observation.CapturedAt.UTC().UnixNano())
|
||||
}
|
||||
|
||||
type batchStopper interface {
|
||||
Stop(ctx context.Context) error
|
||||
}
|
||||
|
||||
type statsProvider interface {
|
||||
Stats() batchwriter.Stats
|
||||
}
|
||||
|
||||
func running() bool {
|
||||
return metricSnapshotWriter != nil && metricSnapshotWriter.Running()
|
||||
}
|
||||
}
|
||||
|
||||
@@ -117,6 +117,16 @@ func buildOverviewView(ctx context.Context) (*OverviewView, error) {
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
// Latest-per-node health: dedicated LIMIT 1 BY queries (not a global raw LIMIT).
|
||||
latestSnapshotRows, err := model.ListOpenFlareLatestMetricSnapshotsSince(ctx, "", since)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
latestTrafficRows, err := model.ListOpenFlareLatestRequestReportsSince(ctx, "", since)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
// Bounded raw windows remain for distributions and trend fallbacks; trends prefer hourly rollups.
|
||||
snapshots, err := model.ListOpenFlareMetricSnapshotsSince(ctx, "", since, dashboardOverviewSnapshotLimit)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
@@ -146,8 +156,8 @@ func buildOverviewView(ctx context.Context) (*OverviewView, error) {
|
||||
|
||||
var cpuNodeCount int
|
||||
var memoryNodeCount int
|
||||
latestSnapshots := observability.LatestMetricSnapshotsByNode(snapshots)
|
||||
latestTrafficReports := observability.LatestTrafficReportsByNode(reports)
|
||||
latestSnapshots := observability.LatestMetricSnapshotsByNode(latestSnapshotRows)
|
||||
latestTrafficReports := observability.LatestTrafficReportsByNode(latestTrafficRows)
|
||||
activeEventsByNode := observability.ActiveHealthEventsByNode(activeEvents)
|
||||
|
||||
for _, node := range nodes {
|
||||
|
||||
@@ -58,6 +58,32 @@ func TestGetOverviewStructure(t *testing.T) {
|
||||
OpenrestyStatus: "unknown",
|
||||
}).Error)
|
||||
|
||||
// Seed older + newer snapshots per node; health must use latest-per-node, not a global raw limit.
|
||||
require.NoError(t, model.InsertOpenFlareMetricSnapshot(ctx, &model.OpenFlareMetricSnapshot{
|
||||
NodeID: "node-dashboard-1",
|
||||
CapturedAt: now.Add(-2 * time.Hour),
|
||||
CPUUsagePercent: 10,
|
||||
MemoryUsedBytes: 1,
|
||||
MemoryTotalBytes: 10,
|
||||
}))
|
||||
require.NoError(t, model.InsertOpenFlareMetricSnapshot(ctx, &model.OpenFlareMetricSnapshot{
|
||||
NodeID: "node-dashboard-1",
|
||||
CapturedAt: now.Add(-time.Minute),
|
||||
CPUUsagePercent: 55,
|
||||
MemoryUsedBytes: 5,
|
||||
MemoryTotalBytes: 10,
|
||||
StorageUsedBytes: 2,
|
||||
StorageTotalBytes: 10,
|
||||
}))
|
||||
require.NoError(t, model.InsertOpenFlareRequestReport(ctx, &model.OpenFlareRequestReport{
|
||||
NodeID: "node-dashboard-1",
|
||||
WindowStartedAt: now.Add(-2 * time.Minute),
|
||||
WindowEndedAt: now.Add(-time.Minute),
|
||||
RequestCount: 12,
|
||||
ErrorCount: 1,
|
||||
UniqueVisitorCount: 4,
|
||||
}))
|
||||
|
||||
overview, err := GetOverview(ctx)
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, overview)
|
||||
@@ -69,14 +95,14 @@ func TestGetOverviewStructure(t *testing.T) {
|
||||
assert.Equal(t, 0, overview.Summary.OfflineNodes)
|
||||
assert.Equal(t, 0, overview.Summary.UnhealthyNodes)
|
||||
|
||||
assert.Equal(t, int64(0), overview.Traffic.RequestCount)
|
||||
assert.Equal(t, int64(0), overview.Traffic.UniqueVisitors)
|
||||
assert.Equal(t, int64(0), overview.Traffic.ErrorCount)
|
||||
assert.Equal(t, float64(0), overview.Traffic.EstimatedQPS)
|
||||
assert.Equal(t, 0, overview.Traffic.ReportedNodes)
|
||||
assert.Equal(t, int64(12), overview.Traffic.RequestCount)
|
||||
assert.Equal(t, int64(4), overview.Traffic.UniqueVisitors)
|
||||
assert.Equal(t, int64(1), overview.Traffic.ErrorCount)
|
||||
assert.InDelta(t, 0.2, overview.Traffic.EstimatedQPS, 0.0001)
|
||||
assert.Equal(t, 1, overview.Traffic.ReportedNodes)
|
||||
|
||||
assert.Equal(t, float64(0), overview.Capacity.AverageCPUUsagePercent)
|
||||
assert.Equal(t, float64(0), overview.Capacity.AverageMemoryUsagePercent)
|
||||
assert.Equal(t, 55.0, overview.Capacity.AverageCPUUsagePercent)
|
||||
assert.Equal(t, 50.0, overview.Capacity.AverageMemoryUsagePercent)
|
||||
assert.Equal(t, 0, overview.Capacity.HighCPUNodes)
|
||||
assert.Equal(t, 0, overview.Capacity.HighMemoryNodes)
|
||||
assert.Equal(t, 0, overview.Capacity.HighStorageNodes)
|
||||
@@ -120,10 +146,20 @@ func TestGetOverviewStructure(t *testing.T) {
|
||||
assert.Equal(t, "Edge 1", onlineNode[2])
|
||||
assert.Equal(t, "online", onlineNode[6])
|
||||
assert.Equal(t, "healthy", onlineNode[7])
|
||||
// Latest-per-node health fields (indexes match compressDashboardNodes).
|
||||
assert.Equal(t, 55.0, onlineNode[11]) // cpu_usage_percent from latest snapshot
|
||||
assert.Equal(t, 50.0, onlineNode[12]) // memory_usage_percent
|
||||
assert.Equal(t, int64(12), onlineNode[14])
|
||||
assert.Equal(t, int64(1), onlineNode[15])
|
||||
assert.Equal(t, int64(4), onlineNode[16])
|
||||
|
||||
pendingNode := nodeByID["node-dashboard-2"]
|
||||
require.NotNil(t, pendingNode)
|
||||
assert.Equal(t, "Edge 2", pendingNode[2])
|
||||
assert.Equal(t, "pending", pendingNode[6])
|
||||
assert.Equal(t, "unknown", pendingNode[7])
|
||||
|
||||
assert.Equal(t, 55.0, overview.Capacity.AverageCPUUsagePercent)
|
||||
assert.Equal(t, 1, overview.Traffic.ReportedNodes)
|
||||
assert.Equal(t, int64(4), overview.Traffic.UniqueVisitors)
|
||||
}
|
||||
|
||||
@@ -59,6 +59,9 @@ type databaseCleanupResult struct {
|
||||
Target string `json:"target"`
|
||||
TargetLabel string `json:"target_label"`
|
||||
DeletedCount int64 `json:"deleted_count"`
|
||||
EligibleCount int64 `json:"eligible_count,omitempty"`
|
||||
CleanupMode string `json:"cleanup_mode,omitempty"`
|
||||
TableTTLDays int `json:"table_ttl_days,omitempty"`
|
||||
DeleteAll bool `json:"delete_all"`
|
||||
RetentionDays *int `json:"retention_days,omitempty"`
|
||||
}
|
||||
@@ -198,6 +201,9 @@ func cleanupDatabaseObservability(ctx context.Context, input databaseCleanupInpu
|
||||
Target: result.Target,
|
||||
TargetLabel: result.TargetLabel,
|
||||
DeletedCount: result.DeletedCount,
|
||||
EligibleCount: result.EligibleCount,
|
||||
CleanupMode: result.CleanupMode,
|
||||
TableTTLDays: result.TableTTLDays,
|
||||
DeleteAll: result.DeleteAll,
|
||||
RetentionDays: result.RetentionDays,
|
||||
}, nil
|
||||
|
||||
@@ -146,21 +146,26 @@ func TestCleanupDatabaseObservabilityDeletesRows(t *testing.T) {
|
||||
},
|
||||
}))
|
||||
|
||||
retention := 7
|
||||
result, err := cleanupDatabaseObservability(ctx, databaseCleanupInput{
|
||||
// Retention shorter than table TTL (90d for access logs) must be rejected.
|
||||
shortRetention := 7
|
||||
_, err := cleanupDatabaseObservability(ctx, databaseCleanupInput{
|
||||
Target: "node_access_logs",
|
||||
RetentionDays: &retention,
|
||||
RetentionDays: &shortRetention,
|
||||
})
|
||||
require.Error(t, err)
|
||||
|
||||
// Full truncate still hard-deletes all rows.
|
||||
result, err := cleanupDatabaseObservability(ctx, databaseCleanupInput{
|
||||
Target: "node_access_logs",
|
||||
})
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, "node_access_logs", result.Target)
|
||||
assert.Equal(t, "访问日志", result.TargetLabel)
|
||||
assert.Equal(t, int64(1), result.DeletedCount)
|
||||
assert.False(t, result.DeleteAll)
|
||||
require.NotNil(t, result.RetentionDays)
|
||||
assert.Equal(t, 7, *result.RetentionDays)
|
||||
assert.Equal(t, int64(2), result.DeletedCount)
|
||||
assert.True(t, result.DeleteAll)
|
||||
assert.Equal(t, "truncate", result.CleanupMode)
|
||||
|
||||
rows, err := model.ListOpenFlareAccessLogs(ctx, model.OpenFlareAccessLogQuery{Page: 0, PageSize: 10})
|
||||
require.NoError(t, err)
|
||||
require.Len(t, rows, 1)
|
||||
assert.Equal(t, "/recent", rows[0].Path)
|
||||
assert.Empty(t, rows)
|
||||
}
|
||||
|
||||
@@ -34,9 +34,19 @@ var databaseCleanupTargets = map[string]string{
|
||||
DatabaseCleanupTargetAccessLogs: "访问日志",
|
||||
DatabaseCleanupTargetMetricSnapshots: "性能快照",
|
||||
DatabaseCleanupTargetRequestReports: "请求聚合",
|
||||
DatabaseCleanupTargetObsOpenresty: "OpenResty 观测",
|
||||
DatabaseCleanupTargetObsFrps: "FRPS 观测",
|
||||
DatabaseCleanupTargetObsFrpc: "FRPC 观测",
|
||||
DatabaseCleanupTargetObsOpenresty: "OpenResty 观测",
|
||||
DatabaseCleanupTargetObsFrps: "FRPS 观测",
|
||||
DatabaseCleanupTargetObsFrpc: "FRPC 观测",
|
||||
}
|
||||
|
||||
// databaseCleanupTableTTLDays maps API targets to ClickHouse DDL TTL days.
|
||||
var databaseCleanupTableTTLDays = map[string]int{
|
||||
DatabaseCleanupTargetAccessLogs: analyticsrepo.TableTTLDaysNodeAccessLogs,
|
||||
DatabaseCleanupTargetMetricSnapshots: analyticsrepo.TableTTLDaysNodeMetricSnapshots,
|
||||
DatabaseCleanupTargetRequestReports: analyticsrepo.TableTTLDaysNodeRequestReports,
|
||||
DatabaseCleanupTargetObsOpenresty: analyticsrepo.TableTTLDaysNodeObs,
|
||||
DatabaseCleanupTargetObsFrps: analyticsrepo.TableTTLDaysNodeObs,
|
||||
DatabaseCleanupTargetObsFrpc: analyticsrepo.TableTTLDaysNodeObs,
|
||||
}
|
||||
|
||||
// DatabaseCleanupInput describes a manual observability cleanup request.
|
||||
@@ -46,14 +56,21 @@ type DatabaseCleanupInput struct {
|
||||
}
|
||||
|
||||
// DatabaseCleanupResult summarizes a manual observability cleanup run.
|
||||
//
|
||||
// Semantics:
|
||||
// - delete_all / cleanup_mode=truncate: DeletedCount is hard-deleted rows (TRUNCATE).
|
||||
// - retention path / cleanup_mode=ttl_materialize: DeletedCount is always 0;
|
||||
// EligibleCount estimates rows past the table DDL TTL (not an arbitrary younger cutoff).
|
||||
type DatabaseCleanupResult struct {
|
||||
Target string `json:"target"`
|
||||
TargetLabel string `json:"target_label"`
|
||||
DeletedCount int64 `json:"deleted_count"`
|
||||
CleanupMode string `json:"cleanup_mode,omitempty"`
|
||||
DeleteAll bool `json:"delete_all"`
|
||||
RetentionDays *int `json:"retention_days,omitempty"`
|
||||
Cutoff *time.Time `json:"cutoff,omitempty"`
|
||||
Target string `json:"target"`
|
||||
TargetLabel string `json:"target_label"`
|
||||
DeletedCount int64 `json:"deleted_count"`
|
||||
EligibleCount int64 `json:"eligible_count,omitempty"`
|
||||
CleanupMode string `json:"cleanup_mode,omitempty"`
|
||||
TableTTLDays int `json:"table_ttl_days,omitempty"`
|
||||
DeleteAll bool `json:"delete_all"`
|
||||
RetentionDays *int `json:"retention_days,omitempty"`
|
||||
Cutoff *time.Time `json:"cutoff,omitempty"`
|
||||
}
|
||||
|
||||
// DatabaseAutoCleanupSummary summarizes a scheduled auto-cleanup run.
|
||||
@@ -63,7 +80,17 @@ type DatabaseAutoCleanupSummary struct {
|
||||
Results []DatabaseCleanupResult `json:"results"`
|
||||
}
|
||||
|
||||
// TableTTLDaysForCleanupTarget returns the DDL TTL days for a cleanup target.
|
||||
func TableTTLDaysForCleanupTarget(target string) (int, bool) {
|
||||
days, ok := databaseCleanupTableTTLDays[strings.TrimSpace(target)]
|
||||
return days, ok
|
||||
}
|
||||
|
||||
// CleanupDatabaseObservability deletes observability rows for the given target.
|
||||
//
|
||||
// When RetentionDays is nil, rows are hard-deleted via TRUNCATE.
|
||||
// When RetentionDays is set, ClickHouse only force-materializes the table TTL policy:
|
||||
// retention_days shorter than the table TTL is rejected (do not fake success).
|
||||
func CleanupDatabaseObservability(ctx context.Context, input DatabaseCleanupInput) (*DatabaseCleanupResult, error) {
|
||||
target := strings.TrimSpace(input.Target)
|
||||
targetLabel, ok := databaseCleanupTargets[target]
|
||||
@@ -74,10 +101,12 @@ func CleanupDatabaseObservability(ctx context.Context, input DatabaseCleanupInpu
|
||||
return nil, errors.New("retention_days 必须为大于 0 的整数")
|
||||
}
|
||||
|
||||
tableTTLDays := databaseCleanupTableTTLDays[target]
|
||||
result := &DatabaseCleanupResult{
|
||||
Target: target,
|
||||
TargetLabel: targetLabel,
|
||||
DeleteAll: input.RetentionDays == nil,
|
||||
Target: target,
|
||||
TargetLabel: targetLabel,
|
||||
DeleteAll: input.RetentionDays == nil,
|
||||
TableTTLDays: tableTTLDays,
|
||||
}
|
||||
|
||||
if input.RetentionDays == nil {
|
||||
@@ -86,24 +115,37 @@ func CleanupDatabaseObservability(ctx context.Context, input DatabaseCleanupInpu
|
||||
return nil, err
|
||||
}
|
||||
result.DeletedCount = deleted
|
||||
result.EligibleCount = deleted
|
||||
result.CleanupMode = mode
|
||||
return result, nil
|
||||
}
|
||||
|
||||
retentionDays := *input.RetentionDays
|
||||
cutoff := time.Now().UTC().Add(-time.Duration(retentionDays) * 24 * time.Hour)
|
||||
deleted, mode, err := deleteObservabilityRowsBefore(ctx, target, cutoff)
|
||||
if retentionDays < tableTTLDays {
|
||||
return nil, fmt.Errorf(
|
||||
"retention_days 不能小于表 TTL(%d 天);ClickHouse 仅支持按表 TTL 物化过期,更短保留请使用清空全部或调整 DDL",
|
||||
tableTTLDays,
|
||||
)
|
||||
}
|
||||
|
||||
// MATERIALIZE TTL only enforces DDL policy; cutoff reported is the table TTL boundary.
|
||||
tableCutoff := time.Now().UTC().Add(-time.Duration(tableTTLDays) * 24 * time.Hour)
|
||||
eligible, mode, err := materializeObservabilityTableTTL(ctx, target)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
result.DeletedCount = deleted
|
||||
result.DeletedCount = 0
|
||||
result.EligibleCount = eligible
|
||||
result.CleanupMode = mode
|
||||
result.RetentionDays = &retentionDays
|
||||
result.Cutoff = &cutoff
|
||||
result.Cutoff = &tableCutoff
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// RunDatabaseAutoCleanupOnce runs retention-based cleanup for all observability targets.
|
||||
//
|
||||
// Configured retention shorter than a target's table TTL is clamped up to the table TTL
|
||||
// so the scheduled job can force-materialize each table policy without failing.
|
||||
func RunDatabaseAutoCleanupOnce(ctx context.Context, now time.Time) (*DatabaseAutoCleanupSummary, error) {
|
||||
enabled, err := repository.GetBoolByKey(ctx, model.ConfigKeyDatabaseAutoCleanupEnabled)
|
||||
if err != nil {
|
||||
@@ -128,9 +170,13 @@ func RunDatabaseAutoCleanupOnce(ctx context.Context, now time.Time) (*DatabaseAu
|
||||
DatabaseCleanupTargetObsFrps,
|
||||
DatabaseCleanupTargetObsFrpc,
|
||||
} {
|
||||
effectiveDays := retentionDays
|
||||
if ttl, ok := databaseCleanupTableTTLDays[target]; ok && effectiveDays < ttl {
|
||||
effectiveDays = ttl
|
||||
}
|
||||
result, err := CleanupDatabaseObservability(ctx, DatabaseCleanupInput{
|
||||
Target: target,
|
||||
RetentionDays: &retentionDays,
|
||||
RetentionDays: &effectiveDays,
|
||||
})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
@@ -172,29 +218,37 @@ func deleteAllObservabilityRows(ctx context.Context, target string) (int64, stri
|
||||
return deleted, analyticsrepo.CleanupModeTruncate, nil
|
||||
}
|
||||
|
||||
func deleteObservabilityRowsBefore(ctx context.Context, target string, cutoff time.Time) (int64, string, error) {
|
||||
// materializeObservabilityTableTTL triggers table-TTL materialize (or memory-store delete-before
|
||||
// with the table TTL cutoff for tests) and returns the eligible/estimate row count.
|
||||
func materializeObservabilityTableTTL(ctx context.Context, target string) (int64, string, error) {
|
||||
ttlDays, ok := databaseCleanupTableTTLDays[target]
|
||||
if !ok {
|
||||
return 0, "", errors.New("unsupported cleanup target")
|
||||
}
|
||||
cutoff := time.Now().UTC().Add(-time.Duration(ttlDays) * 24 * time.Hour)
|
||||
|
||||
var (
|
||||
deleted int64
|
||||
err error
|
||||
eligible int64
|
||||
err error
|
||||
)
|
||||
switch target {
|
||||
case DatabaseCleanupTargetAccessLogs:
|
||||
deleted, err = model.DeleteOpenFlareAccessLogsBefore(ctx, cutoff)
|
||||
eligible, err = model.DeleteOpenFlareAccessLogsBefore(ctx, cutoff)
|
||||
case DatabaseCleanupTargetMetricSnapshots:
|
||||
deleted, err = model.DeleteOpenFlareMetricSnapshotsBefore(ctx, cutoff)
|
||||
eligible, err = model.DeleteOpenFlareMetricSnapshotsBefore(ctx, cutoff)
|
||||
case DatabaseCleanupTargetRequestReports:
|
||||
deleted, err = model.DeleteOpenFlareRequestReportsBefore(ctx, cutoff)
|
||||
eligible, err = model.DeleteOpenFlareRequestReportsBefore(ctx, cutoff)
|
||||
case DatabaseCleanupTargetObsOpenresty:
|
||||
deleted, err = model.DeleteOpenFlareNodeObservationOpenrestyBefore(ctx, cutoff)
|
||||
eligible, err = model.DeleteOpenFlareNodeObservationOpenrestyBefore(ctx, cutoff)
|
||||
case DatabaseCleanupTargetObsFrps:
|
||||
deleted, err = model.DeleteOpenFlareNodeObservationFrpsBefore(ctx, cutoff)
|
||||
eligible, err = model.DeleteOpenFlareNodeObservationFrpsBefore(ctx, cutoff)
|
||||
case DatabaseCleanupTargetObsFrpc:
|
||||
deleted, err = model.DeleteOpenFlareNodeObservationFrpcBefore(ctx, cutoff)
|
||||
eligible, err = model.DeleteOpenFlareNodeObservationFrpcBefore(ctx, cutoff)
|
||||
default:
|
||||
return 0, "", errors.New("unsupported cleanup target")
|
||||
}
|
||||
if err != nil {
|
||||
return 0, "", err
|
||||
}
|
||||
return deleted, analyticsrepo.CleanupModeTTLMaterialize, nil
|
||||
return eligible, analyticsrepo.CleanupModeTTLMaterialize, nil
|
||||
}
|
||||
|
||||
@@ -11,6 +11,7 @@ import (
|
||||
"github.com/Rain-kl/Wavelet/internal/db"
|
||||
"github.com/Rain-kl/Wavelet/internal/model"
|
||||
"github.com/Rain-kl/Wavelet/internal/repository"
|
||||
analyticsrepo "github.com/Rain-kl/Wavelet/internal/repository/analytics"
|
||||
"github.com/glebarez/sqlite"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
@@ -36,13 +37,41 @@ func setupDatabaseCleanupTestDB(t *testing.T) context.Context {
|
||||
return context.Background()
|
||||
}
|
||||
|
||||
func TestCleanupDatabaseObservabilityDeletesTargetedRows(t *testing.T) {
|
||||
func TestCleanupDatabaseObservabilityRejectsRetentionShorterThanTableTTL(t *testing.T) {
|
||||
ctx := setupDatabaseCleanupTestDB(t)
|
||||
|
||||
retentionDays := 7 // metric snapshots DDL TTL is 30 days
|
||||
result, err := CleanupDatabaseObservability(ctx, DatabaseCleanupInput{
|
||||
Target: DatabaseCleanupTargetMetricSnapshots,
|
||||
RetentionDays: &retentionDays,
|
||||
})
|
||||
require.Error(t, err)
|
||||
assert.Nil(t, result)
|
||||
assert.Contains(t, err.Error(), "不能小于表 TTL")
|
||||
assert.Contains(t, err.Error(), "30")
|
||||
}
|
||||
|
||||
func TestCleanupDatabaseObservabilityRejectsAccessLogRetentionShorterThanTableTTL(t *testing.T) {
|
||||
ctx := setupDatabaseCleanupTestDB(t)
|
||||
|
||||
retentionDays := 30 // access logs DDL TTL is 90 days
|
||||
result, err := CleanupDatabaseObservability(ctx, DatabaseCleanupInput{
|
||||
Target: DatabaseCleanupTargetAccessLogs,
|
||||
RetentionDays: &retentionDays,
|
||||
})
|
||||
require.Error(t, err)
|
||||
assert.Nil(t, result)
|
||||
assert.Contains(t, err.Error(), "90")
|
||||
}
|
||||
|
||||
func TestCleanupDatabaseObservabilityMaterializeDoesNotClaimHardDelete(t *testing.T) {
|
||||
ctx := setupDatabaseCleanupTestDB(t)
|
||||
now := time.Now().UTC()
|
||||
|
||||
// One row past metric table TTL (30d), one still inside the window.
|
||||
require.NoError(t, model.InsertOpenFlareMetricSnapshot(ctx, &model.OpenFlareMetricSnapshot{
|
||||
NodeID: "node-a",
|
||||
CapturedAt: now.Add(-10 * 24 * time.Hour),
|
||||
CapturedAt: now.Add(-40 * 24 * time.Hour),
|
||||
CPUUsagePercent: 10,
|
||||
}))
|
||||
require.NoError(t, model.InsertOpenFlareMetricSnapshot(ctx, &model.OpenFlareMetricSnapshot{
|
||||
@@ -51,15 +80,22 @@ func TestCleanupDatabaseObservabilityDeletesTargetedRows(t *testing.T) {
|
||||
CPUUsagePercent: 20,
|
||||
}))
|
||||
|
||||
retentionDays := 7
|
||||
retentionDays := analyticsrepo.TableTTLDaysNodeMetricSnapshots
|
||||
result, err := CleanupDatabaseObservability(ctx, DatabaseCleanupInput{
|
||||
Target: DatabaseCleanupTargetMetricSnapshots,
|
||||
RetentionDays: &retentionDays,
|
||||
})
|
||||
require.NoError(t, err)
|
||||
assert.False(t, result.DeleteAll)
|
||||
assert.Equal(t, int64(1), result.DeletedCount)
|
||||
assert.Equal(t, analyticsrepo.CleanupModeTTLMaterialize, result.CleanupMode)
|
||||
assert.Equal(t, analyticsrepo.TableTTLDaysNodeMetricSnapshots, result.TableTTLDays)
|
||||
// MATERIALIZE is not a counted hard delete.
|
||||
assert.Equal(t, int64(0), result.DeletedCount)
|
||||
assert.Equal(t, int64(1), result.EligibleCount)
|
||||
require.NotNil(t, result.Cutoff)
|
||||
assert.True(t, result.Cutoff.Before(now.Add(-29*24*time.Hour)))
|
||||
|
||||
// Memory store applies the table-TTL cutoff for tests; only the recent row remains.
|
||||
rows, err := model.ListOpenFlareMetricSnapshotsSince(ctx, "", time.Time{}, 0)
|
||||
require.NoError(t, err)
|
||||
require.Len(t, rows, 1)
|
||||
@@ -94,20 +130,23 @@ func TestCleanupDatabaseObservabilityDeletesAllRowsWhenRetentionMissing(t *testi
|
||||
})
|
||||
require.NoError(t, err)
|
||||
assert.True(t, result.DeleteAll)
|
||||
assert.Equal(t, analyticsrepo.CleanupModeTruncate, result.CleanupMode)
|
||||
assert.Equal(t, int64(2), result.DeletedCount)
|
||||
assert.Equal(t, int64(2), result.EligibleCount)
|
||||
|
||||
rows, err := model.ListOpenFlareAccessLogs(ctx, model.OpenFlareAccessLogQuery{Page: 0, PageSize: 10})
|
||||
require.NoError(t, err)
|
||||
assert.Empty(t, rows)
|
||||
}
|
||||
|
||||
func TestRunDatabaseAutoCleanupOnceDeletesAllObservabilityTargets(t *testing.T) {
|
||||
func TestRunDatabaseAutoCleanupOnceClampsRetentionToTableTTL(t *testing.T) {
|
||||
ctx := setupDatabaseCleanupTestDB(t)
|
||||
now := time.Now().UTC()
|
||||
|
||||
// Access logs TTL=90d, metrics TTL=30d. Config retention=1 must clamp, not reject.
|
||||
require.NoError(t, model.InsertOpenFlareAccessLogsBatch(ctx, []*model.OpenFlareAccessLog{{
|
||||
NodeID: "node-a",
|
||||
LoggedAt: now.Add(-48 * time.Hour),
|
||||
LoggedAt: now.Add(-100 * 24 * time.Hour),
|
||||
RemoteAddr: "203.0.113.10",
|
||||
Host: "example.com",
|
||||
Path: "/access",
|
||||
@@ -115,13 +154,13 @@ func TestRunDatabaseAutoCleanupOnceDeletesAllObservabilityTargets(t *testing.T)
|
||||
}}))
|
||||
require.NoError(t, model.InsertOpenFlareMetricSnapshot(ctx, &model.OpenFlareMetricSnapshot{
|
||||
NodeID: "node-a",
|
||||
CapturedAt: now.Add(-48 * time.Hour),
|
||||
CapturedAt: now.Add(-40 * 24 * time.Hour),
|
||||
CPUUsagePercent: 10,
|
||||
}))
|
||||
require.NoError(t, model.InsertOpenFlareRequestReport(ctx, &model.OpenFlareRequestReport{
|
||||
NodeID: "node-a",
|
||||
WindowStartedAt: now.Add(-49 * time.Hour),
|
||||
WindowEndedAt: now.Add(-48 * time.Hour),
|
||||
WindowStartedAt: now.Add(-41 * 24 * time.Hour),
|
||||
WindowEndedAt: now.Add(-40 * 24 * time.Hour),
|
||||
RequestCount: 15,
|
||||
}))
|
||||
|
||||
@@ -132,6 +171,15 @@ func TestRunDatabaseAutoCleanupOnceDeletesAllObservabilityTargets(t *testing.T)
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, summary)
|
||||
require.Len(t, summary.Results, 6)
|
||||
assert.Equal(t, 1, summary.RetentionDays)
|
||||
|
||||
for _, result := range summary.Results {
|
||||
assert.Equal(t, analyticsrepo.CleanupModeTTLMaterialize, result.CleanupMode)
|
||||
assert.Equal(t, int64(0), result.DeletedCount, "target %s must not claim hard delete", result.Target)
|
||||
assert.GreaterOrEqual(t, result.TableTTLDays, 30)
|
||||
require.NotNil(t, result.RetentionDays)
|
||||
assert.GreaterOrEqual(t, *result.RetentionDays, result.TableTTLDays)
|
||||
}
|
||||
|
||||
accessLogs, err := model.ListOpenFlareAccessLogs(ctx, model.OpenFlareAccessLogQuery{Page: 0, PageSize: 10})
|
||||
require.NoError(t, err)
|
||||
@@ -145,3 +193,16 @@ func TestRunDatabaseAutoCleanupOnceDeletesAllObservabilityTargets(t *testing.T)
|
||||
require.NoError(t, err)
|
||||
assert.Empty(t, requestReports)
|
||||
}
|
||||
|
||||
func TestTableTTLDaysForCleanupTarget(t *testing.T) {
|
||||
days, ok := TableTTLDaysForCleanupTarget(DatabaseCleanupTargetAccessLogs)
|
||||
require.True(t, ok)
|
||||
assert.Equal(t, 90, days)
|
||||
|
||||
days, ok = TableTTLDaysForCleanupTarget(DatabaseCleanupTargetMetricSnapshots)
|
||||
require.True(t, ok)
|
||||
assert.Equal(t, 30, days)
|
||||
|
||||
_, ok = TableTTLDaysForCleanupTarget("unknown")
|
||||
assert.False(t, ok)
|
||||
}
|
||||
|
||||
@@ -6,6 +6,7 @@ package risk_control
|
||||
import (
|
||||
"context"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/Rain-kl/Wavelet/internal/config"
|
||||
"github.com/Rain-kl/Wavelet/internal/db/batchwriter"
|
||||
@@ -15,6 +16,11 @@ import (
|
||||
"github.com/Rain-kl/Wavelet/pkg/logger"
|
||||
)
|
||||
|
||||
const (
|
||||
// Bound visibility lag for sparse access-log traffic when MinBatchSize is not met.
|
||||
accessLogMaxFlushWait = 3 * time.Second
|
||||
)
|
||||
|
||||
var (
|
||||
logWriterMu sync.RWMutex
|
||||
logWriter *batchwriter.Writer[*analytics.UserAccessLog]
|
||||
@@ -33,6 +39,8 @@ func InitLogWriter(ctx context.Context) {
|
||||
}
|
||||
|
||||
cfg := batchwriter.DefaultConfig()
|
||||
cfg.Name = "user_access_logs"
|
||||
cfg.MaxFlushWait = accessLogMaxFlushWait
|
||||
writer, err := batchwriter.New[*analytics.UserAccessLog](cfg, func(ctx context.Context, items []*analytics.UserAccessLog) error {
|
||||
rows := make([]analytics.UserAccessLog, 0, len(items))
|
||||
for _, item := range items {
|
||||
@@ -50,8 +58,8 @@ func InitLogWriter(ctx context.Context) {
|
||||
}
|
||||
logger.WarnF(context.Background(), "[RiskControl] Log queue full, dropping log item for path: %s", path)
|
||||
}),
|
||||
batchwriter.WithFlushErrorHandler[*analytics.UserAccessLog](func(ctx context.Context, batchSize int, err error) {
|
||||
logger.ErrorF(ctx, "[RiskControl] Send ClickHouse batch failed (batch=%d): %v", batchSize, err)
|
||||
batchwriter.WithFlushErrorHandler[*analytics.UserAccessLog](func(ctx context.Context, items []*analytics.UserAccessLog, err error) {
|
||||
logger.ErrorF(ctx, "[RiskControl] Send ClickHouse batch failed (batch=%d): %v", len(items), err)
|
||||
}),
|
||||
)
|
||||
if err != nil {
|
||||
@@ -82,6 +90,16 @@ func IsBufferFull() bool {
|
||||
return writer.IsFull()
|
||||
}
|
||||
|
||||
// LogWriterStats returns queue depth and failure counters for the access-log writer.
|
||||
// When the writer is not initialized, it returns a zero-value Stats with the expected name.
|
||||
func LogWriterStats() batchwriter.Stats {
|
||||
writer := currentLogWriter()
|
||||
if writer == nil {
|
||||
return batchwriter.Stats{Name: "user_access_logs"}
|
||||
}
|
||||
return writer.Stats()
|
||||
}
|
||||
|
||||
// QueueAccessLog enqueues an access log without blocking.
|
||||
func QueueAccessLog(logItem *analytics.UserAccessLog) {
|
||||
writer := currentLogWriter()
|
||||
@@ -108,4 +126,4 @@ func currentLogWriter() *batchwriter.Writer[*analytics.UserAccessLog] {
|
||||
logWriterMu.RLock()
|
||||
defer logWriterMu.RUnlock()
|
||||
return logWriter
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,30 @@
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
package risk_control
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func TestAccessLogMaxFlushWaitInRange(t *testing.T) {
|
||||
t.Parallel()
|
||||
if accessLogMaxFlushWait < 2*time.Second || accessLogMaxFlushWait > 5*time.Second {
|
||||
t.Fatalf("accessLogMaxFlushWait = %v, want in [2s, 5s]", accessLogMaxFlushWait)
|
||||
}
|
||||
}
|
||||
|
||||
func TestLogWriterStatsWhenNil(t *testing.T) {
|
||||
t.Parallel()
|
||||
reset := SetLogWriterForTest(nil)
|
||||
t.Cleanup(reset)
|
||||
|
||||
stats := LogWriterStats()
|
||||
if stats.Name != "user_access_logs" {
|
||||
t.Fatalf("LogWriterStats().Name = %q, want user_access_logs", stats.Name)
|
||||
}
|
||||
if stats.Running {
|
||||
t.Fatal("LogWriterStats().Running = true for nil writer, want false")
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user