mirror of
https://github.com/Rain-kl/OpenFlare.git
synced 2026-10-08 00:26:37 +08:00
refactor(backend): rename OpenFlare directory to lowercase openflare
This commit is contained in:
@@ -0,0 +1,85 @@
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
package observability
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func TestResolveAccessLogIPSummaryWindowHours(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
since, until, hours, err := resolveAccessLogIPSummaryWindow("", "", 0)
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, defaultAccessLogQueryDays*24, hours)
|
||||
assert.WithinDuration(t, time.Now().UTC(), until, 2*time.Second)
|
||||
assert.WithinDuration(t, until.Add(-time.Duration(hours)*time.Hour), since, time.Second)
|
||||
|
||||
_, _, hours, err = resolveAccessLogIPSummaryWindow("", "", 24)
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, 24, hours)
|
||||
|
||||
_, _, hours, err = resolveAccessLogIPSummaryWindow("", "", 9999)
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, maxAccessLogOverviewHours, hours)
|
||||
}
|
||||
|
||||
func TestResolveAccessLogIPSummaryWindowCustomRange(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
start := time.Date(2026, 7, 1, 0, 0, 0, 0, time.UTC)
|
||||
end := start.Add(72 * time.Hour)
|
||||
since, until, hours, err := resolveAccessLogIPSummaryWindow(
|
||||
start.Format(time.RFC3339),
|
||||
end.Format(time.RFC3339),
|
||||
24,
|
||||
)
|
||||
require.NoError(t, err)
|
||||
assert.True(t, since.Equal(start))
|
||||
assert.True(t, until.Equal(end))
|
||||
assert.Equal(t, 72, hours)
|
||||
}
|
||||
|
||||
func TestResolveAccessLogIPSummaryWindowErrors(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
_, _, _, err := resolveAccessLogIPSummaryWindow("2026-07-01T00:00:00Z", "", 24)
|
||||
require.Error(t, err)
|
||||
|
||||
_, _, _, err = resolveAccessLogIPSummaryWindow("bad", "2026-07-02T00:00:00Z", 24)
|
||||
require.Error(t, err)
|
||||
|
||||
start := time.Date(2026, 7, 2, 0, 0, 0, 0, time.UTC)
|
||||
end := start.Add(-time.Hour)
|
||||
_, _, _, err = resolveAccessLogIPSummaryWindow(
|
||||
start.Format(time.RFC3339),
|
||||
end.Format(time.RFC3339),
|
||||
24,
|
||||
)
|
||||
require.Error(t, err)
|
||||
|
||||
start = time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC)
|
||||
end = start.Add(40 * 24 * time.Hour)
|
||||
_, _, _, err = resolveAccessLogIPSummaryWindow(
|
||||
start.Format(time.RFC3339),
|
||||
end.Format(time.RFC3339),
|
||||
24,
|
||||
)
|
||||
require.Error(t, err)
|
||||
}
|
||||
|
||||
func TestNormalizeIPSummarySortBy(t *testing.T) {
|
||||
t.Parallel()
|
||||
assert.Equal(t, "total_requests", normalizeIPSummarySortBy(""))
|
||||
assert.Equal(t, "request_length", normalizeIPSummarySortBy("bytes_received"))
|
||||
assert.Equal(t, "request_length", normalizeIPSummarySortBy("request_length"))
|
||||
assert.Equal(t, "bytes_sent", normalizeIPSummarySortBy("bytes_sent"))
|
||||
assert.Equal(t, "success_ratio", normalizeIPSummarySortBy("success_ratio"))
|
||||
assert.Equal(t, "last_seen_at", normalizeIPSummarySortBy("last_seen_at"))
|
||||
assert.Equal(t, "remote_addr", normalizeIPSummarySortBy("remote_addr"))
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,697 @@
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
package observability
|
||||
|
||||
import (
|
||||
"context"
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"Wavelet/openflare/plugins/server/kernel/repository"
|
||||
|
||||
"Wavelet/openflare/plugins/server/kernel/model"
|
||||
)
|
||||
|
||||
const observabilityTrendBuckets = 24
|
||||
const unknownTrendNodeKey = "__unknown__"
|
||||
|
||||
const (
|
||||
healthEventStatusActive = "active"
|
||||
healthEventStatusResolved = "resolved"
|
||||
healthSeverityCritical = "critical"
|
||||
healthSeverityWarning = "warning"
|
||||
percentageMultiplier = 100
|
||||
sortOrderAsc = "asc"
|
||||
)
|
||||
|
||||
// DistributionItem is a key/value distribution entry.
|
||||
type DistributionItem struct {
|
||||
Key string `json:"key"`
|
||||
Value int64 `json:"value"`
|
||||
}
|
||||
|
||||
// TrafficDistributions groups traffic distribution charts.
|
||||
type TrafficDistributions struct {
|
||||
StatusCodes []DistributionItem `json:"status_codes"`
|
||||
TopDomains []DistributionItem `json:"top_domains"`
|
||||
SourceCountries []DistributionItem `json:"source_countries"`
|
||||
}
|
||||
|
||||
const metricSnapshotEdgeHealthMatchWindow = 2 * time.Minute
|
||||
|
||||
// NodeMetricSnapshotView is a metric snapshot enriched with edge health connections.
|
||||
type NodeMetricSnapshotView struct {
|
||||
ID uint `json:"id,omitempty"`
|
||||
NodeID string `json:"node_id,omitempty"`
|
||||
CapturedAt time.Time `json:"captured_at"`
|
||||
CPUUsagePercent float64 `json:"cpu_usage_percent"`
|
||||
MemoryUsedBytes int64 `json:"memory_used_bytes"`
|
||||
MemoryTotalBytes int64 `json:"memory_total_bytes"`
|
||||
StorageUsedBytes int64 `json:"storage_used_bytes"`
|
||||
StorageTotalBytes int64 `json:"storage_total_bytes"`
|
||||
DiskReadBytes int64 `json:"disk_read_bytes"`
|
||||
DiskWriteBytes int64 `json:"disk_write_bytes"`
|
||||
OpenrestyConnections int64 `json:"openresty_connections"`
|
||||
}
|
||||
|
||||
// TrafficWindowSummary summarizes a traffic reporting window.
|
||||
type TrafficWindowSummary struct {
|
||||
WindowStartedAt time.Time `json:"window_started_at"`
|
||||
WindowEndedAt time.Time `json:"window_ended_at"`
|
||||
RequestCount int64 `json:"request_count"`
|
||||
UniqueVisitorCount int64 `json:"unique_visitor_count"`
|
||||
ErrorCount int64 `json:"error_count"`
|
||||
EstimatedQPS float64 `json:"estimated_qps"`
|
||||
ErrorRatePercent float64 `json:"error_rate_percent"`
|
||||
}
|
||||
|
||||
// HealthSummary summarizes node health alerts and risks.
|
||||
type HealthSummary struct {
|
||||
ActiveAlerts int `json:"active_alerts"`
|
||||
CriticalAlerts int `json:"critical_alerts"`
|
||||
WarningAlerts int `json:"warning_alerts"`
|
||||
InfoAlerts int `json:"info_alerts"`
|
||||
ResolvedAlerts int `json:"resolved_alerts"`
|
||||
HasCapacityRisk bool `json:"has_capacity_risk"`
|
||||
HasTrafficRisk bool `json:"has_traffic_risk"`
|
||||
HasRuntimeRisk bool `json:"has_runtime_risk"`
|
||||
}
|
||||
|
||||
// TrafficTrendPoint is a traffic trend bucket.
|
||||
type TrafficTrendPoint struct {
|
||||
BucketStartedAt time.Time `json:"bucket_started_at"`
|
||||
RequestCount int64 `json:"request_count"`
|
||||
ErrorCount int64 `json:"error_count"`
|
||||
UniqueVisitorCount int64 `json:"unique_visitor_count"`
|
||||
Status2xxCount int64 `json:"status_2xx_count"`
|
||||
Status4xxCount int64 `json:"status_4xx_count"`
|
||||
Status5xxCount int64 `json:"status_5xx_count"`
|
||||
}
|
||||
|
||||
// CapacityTrendPoint is a capacity trend bucket.
|
||||
type CapacityTrendPoint struct {
|
||||
BucketStartedAt time.Time `json:"bucket_started_at"`
|
||||
AverageCPUUsagePercent float64 `json:"average_cpu_usage_percent"`
|
||||
AverageMemoryUsagePercent float64 `json:"average_memory_usage_percent"`
|
||||
ReportedNodes int `json:"reported_nodes"`
|
||||
}
|
||||
|
||||
// NetworkTrendPoint is a business-byte trend bucket from access logs (L1).
|
||||
// Host NIC trends are intentionally not exposed.
|
||||
type NetworkTrendPoint struct {
|
||||
BucketStartedAt time.Time `json:"bucket_started_at"`
|
||||
BytesReceived int64 `json:"bytes_received"` // sum(request_length)
|
||||
BytesProvided int64 `json:"bytes_provided"` // sum(bytes_sent)
|
||||
ReportedNodes int `json:"reported_nodes"`
|
||||
}
|
||||
|
||||
// DiskIOTrendPoint is a disk IO trend bucket.
|
||||
type DiskIOTrendPoint struct {
|
||||
BucketStartedAt time.Time `json:"bucket_started_at"`
|
||||
DiskReadBytes int64 `json:"disk_read_bytes"`
|
||||
DiskWriteBytes int64 `json:"disk_write_bytes"`
|
||||
ReportedNodes int `json:"reported_nodes"`
|
||||
}
|
||||
|
||||
type distributionAccumulator map[string]int64
|
||||
|
||||
type capacityTrendAccumulator struct {
|
||||
cpuSum float64
|
||||
cpuCount int
|
||||
memSum float64
|
||||
memCount int
|
||||
nodes map[string]struct{}
|
||||
}
|
||||
|
||||
type snapshotTrendAccumulator struct {
|
||||
nodes map[string]struct{}
|
||||
}
|
||||
|
||||
type diskCounterState struct {
|
||||
read int64
|
||||
write int64
|
||||
seen bool
|
||||
}
|
||||
|
||||
func buildTrafficWindowSummaryFromAccessLogs(
|
||||
ctx context.Context,
|
||||
nodeID string,
|
||||
since, until time.Time,
|
||||
) *TrafficWindowSummary {
|
||||
row, err := repository.TrafficSummaryOpenFlareAccessLogs(ctx, model.OpenFlareAccessLogQuery{
|
||||
NodeID: nodeID,
|
||||
Since: since,
|
||||
Until: until,
|
||||
})
|
||||
if err != nil || row.RequestCount <= 0 {
|
||||
return nil
|
||||
}
|
||||
summary := &TrafficWindowSummary{
|
||||
WindowStartedAt: since.UTC(),
|
||||
WindowEndedAt: until.UTC(),
|
||||
RequestCount: row.RequestCount,
|
||||
UniqueVisitorCount: row.UniqueIPCount,
|
||||
ErrorCount: row.ErrorCount,
|
||||
}
|
||||
if duration := until.Sub(since).Seconds(); duration > 0 {
|
||||
summary.EstimatedQPS = float64(row.RequestCount) / duration
|
||||
}
|
||||
if row.RequestCount > 0 {
|
||||
summary.ErrorRatePercent = (float64(row.ErrorCount) / float64(row.RequestCount)) * 100
|
||||
}
|
||||
return summary
|
||||
}
|
||||
|
||||
// BuildMetricSnapshotViews merges metric snapshots with edge health connections for API responses.
|
||||
func BuildMetricSnapshotViews(
|
||||
snapshots []*model.OpenFlareMetricSnapshot,
|
||||
edgeHealth []*model.OpenFlareEdgeHealth,
|
||||
) []*NodeMetricSnapshotView {
|
||||
if len(snapshots) == 0 {
|
||||
return []*NodeMetricSnapshotView{}
|
||||
}
|
||||
views := make([]*NodeMetricSnapshotView, 0, len(snapshots))
|
||||
for _, snapshot := range snapshots {
|
||||
if snapshot == nil {
|
||||
continue
|
||||
}
|
||||
view := &NodeMetricSnapshotView{
|
||||
ID: snapshot.ID,
|
||||
NodeID: snapshot.NodeID,
|
||||
CapturedAt: snapshot.CapturedAt,
|
||||
CPUUsagePercent: snapshot.CPUUsagePercent,
|
||||
MemoryUsedBytes: snapshot.MemoryUsedBytes,
|
||||
MemoryTotalBytes: snapshot.MemoryTotalBytes,
|
||||
StorageUsedBytes: snapshot.StorageUsedBytes,
|
||||
StorageTotalBytes: snapshot.StorageTotalBytes,
|
||||
DiskReadBytes: snapshot.DiskReadBytes,
|
||||
DiskWriteBytes: snapshot.DiskWriteBytes,
|
||||
}
|
||||
if matched := matchEdgeHealth(snapshot.CapturedAt, edgeHealth); matched != nil {
|
||||
view.OpenrestyConnections = matched.Connections
|
||||
}
|
||||
views = append(views, view)
|
||||
}
|
||||
return views
|
||||
}
|
||||
|
||||
// BuildTrafficDistributionsFromAccessLogs builds distributions from access logs (L1).
|
||||
func BuildTrafficDistributionsFromAccessLogs(
|
||||
ctx context.Context,
|
||||
since, until time.Time,
|
||||
limit int,
|
||||
accessLogRegions []*model.OpenFlareAccessLogRegionCount,
|
||||
) TrafficDistributions {
|
||||
statusCodes := make(distributionAccumulator)
|
||||
topDomains := make(distributionAccumulator)
|
||||
|
||||
query := model.OpenFlareAccessLogQuery{Since: since, Until: until}
|
||||
if statusRows, err := repository.ValueCountsOpenFlareAccessLogs(ctx, query, "status_code", limit); err == nil {
|
||||
for _, row := range statusRows {
|
||||
if strings.TrimSpace(row.Value) == "" || row.Count <= 0 {
|
||||
continue
|
||||
}
|
||||
statusCodes[row.Value] = row.Count
|
||||
}
|
||||
}
|
||||
if hostRows, err := repository.ValueCountsOpenFlareAccessLogs(ctx, query, "host", limit); err == nil {
|
||||
for _, row := range hostRows {
|
||||
if strings.TrimSpace(row.Value) == "" || row.Count <= 0 {
|
||||
continue
|
||||
}
|
||||
topDomains[row.Value] = row.Count
|
||||
}
|
||||
}
|
||||
|
||||
sourceCountries := make(distributionAccumulator)
|
||||
for _, item := range accessLogRegions {
|
||||
if item == nil || strings.TrimSpace(item.Region) == "" || item.Count <= 0 {
|
||||
continue
|
||||
}
|
||||
sourceCountries[item.Region] = item.Count
|
||||
}
|
||||
return TrafficDistributions{
|
||||
StatusCodes: toDistributionItems(statusCodes, limit),
|
||||
TopDomains: toDistributionItems(topDomains, limit),
|
||||
SourceCountries: toDistributionItems(sourceCountries, limit),
|
||||
}
|
||||
}
|
||||
|
||||
func buildHealthSummary(
|
||||
snapshot *model.OpenFlareMetricSnapshot,
|
||||
traffic *TrafficWindowSummary,
|
||||
events []*model.OpenFlareHealthEvent,
|
||||
) HealthSummary {
|
||||
summary := HealthSummary{}
|
||||
for _, event := range events {
|
||||
if event == nil {
|
||||
continue
|
||||
}
|
||||
if event.Status == healthEventStatusResolved {
|
||||
summary.ResolvedAlerts++
|
||||
continue
|
||||
}
|
||||
summary.ActiveAlerts++
|
||||
switch event.Severity {
|
||||
case healthSeverityCritical:
|
||||
summary.CriticalAlerts++
|
||||
case healthSeverityWarning:
|
||||
summary.WarningAlerts++
|
||||
default:
|
||||
summary.InfoAlerts++
|
||||
}
|
||||
}
|
||||
if snapshot != nil {
|
||||
memoryUsage := Percentage(snapshot.MemoryUsedBytes, snapshot.MemoryTotalBytes)
|
||||
storageUsage := Percentage(snapshot.StorageUsedBytes, snapshot.StorageTotalBytes)
|
||||
summary.HasCapacityRisk = snapshot.CPUUsagePercent >= 80 || memoryUsage >= 85 || storageUsage >= 85
|
||||
}
|
||||
if traffic != nil && traffic.RequestCount >= 100 {
|
||||
summary.HasTrafficRisk = (float64(traffic.ErrorCount) / float64(traffic.RequestCount)) >= 0.05
|
||||
}
|
||||
summary.HasRuntimeRisk = summary.ActiveAlerts > 0 || summary.HasCapacityRisk || summary.HasTrafficRisk
|
||||
return summary
|
||||
}
|
||||
|
||||
// BuildNodeTrends builds 24h trend series.
|
||||
// Business traffic (requests/errors and provided/received bytes) comes from access logs.
|
||||
// Host capacity/disk come from metric snapshots (hourly when available). Host NIC is not tracked.
|
||||
func BuildNodeTrends(
|
||||
ctx context.Context,
|
||||
now time.Time,
|
||||
nodeID string,
|
||||
snapshots []*model.OpenFlareMetricSnapshot,
|
||||
) NodeTrends {
|
||||
trendSince := now.Add(-24 * time.Hour)
|
||||
|
||||
trafficTrend := BuildTrafficTrendPointsFromAccessLogs(ctx, now, nodeID, trendSince)
|
||||
capacityTrend := BuildCapacityTrendPoints(now, snapshots)
|
||||
networkTrend := emptyNetworkTrendPoints(now)
|
||||
applyAccessLogBytesToNetworkTrend(ctx, now, nodeID, trendSince, networkTrend)
|
||||
diskIOTrend := BuildDiskIOTrendPoints(now, snapshots)
|
||||
|
||||
metricHourly, metricErr := repository.ListOpenFlareMetricHourlySince(ctx, nodeID, trendSince)
|
||||
if metricErr == nil && len(metricHourly) > 0 {
|
||||
capacityTrend = BuildCapacityTrendPointsFromHourly(now, metricHourly)
|
||||
diskIOTrend = BuildDiskIOTrendPointsFromHourly(now, metricHourly)
|
||||
}
|
||||
|
||||
return NodeTrends{
|
||||
Traffic24h: trafficTrend,
|
||||
Capacity24h: capacityTrend,
|
||||
Network24h: networkTrend,
|
||||
DiskIO24h: diskIOTrend,
|
||||
}
|
||||
}
|
||||
|
||||
// BuildTrafficTrendPointsFromAccessLogs builds 24h request/error/status buckets from access logs.
|
||||
// Uses raw bucket aggregates: the hourly rollup (of_access_log_hourly) has no per-status counts,
|
||||
// and the 24h window on the dashboard is cached, so the raw scan is acceptable.
|
||||
// UniqueVisitorCount from buckets is exact (uniqExact on raw); TrafficSummary is used elsewhere for UV.
|
||||
func BuildTrafficTrendPointsFromAccessLogs(ctx context.Context, now time.Time, nodeID string, since time.Time) []TrafficTrendPoint {
|
||||
start := trendWindowStart(now)
|
||||
points := make([]TrafficTrendPoint, observabilityTrendBuckets)
|
||||
for index := range points {
|
||||
points[index].BucketStartedAt = start.Add(time.Duration(index) * time.Hour)
|
||||
}
|
||||
|
||||
buckets, err := repository.ListOpenFlareAccessLogBuckets(ctx, model.OpenFlareAccessLogBucketQuery{
|
||||
NodeID: nodeID,
|
||||
Since: since,
|
||||
Until: now,
|
||||
FoldMinutes: 60,
|
||||
SortBy: defaultAccessLogSortBy,
|
||||
SortOrder: sortOrderAsc,
|
||||
})
|
||||
if err != nil || len(buckets) == 0 {
|
||||
return points
|
||||
}
|
||||
byEpoch := make(map[int64]*model.OpenFlareAccessLogBucketRow, len(buckets))
|
||||
for _, row := range buckets {
|
||||
if row == nil {
|
||||
continue
|
||||
}
|
||||
byEpoch[row.BucketEpoch] = row
|
||||
}
|
||||
for index := range points {
|
||||
epoch := points[index].BucketStartedAt.Unix()
|
||||
if row, ok := byEpoch[epoch]; ok {
|
||||
points[index].RequestCount = row.RequestCount
|
||||
points[index].ErrorCount = row.ServerErrorCount
|
||||
points[index].UniqueVisitorCount = row.UniqueIPCount
|
||||
points[index].Status2xxCount = row.Status2xxCount
|
||||
points[index].Status4xxCount = row.Status4xxCount
|
||||
points[index].Status5xxCount = row.Status5xxCount
|
||||
}
|
||||
}
|
||||
return points
|
||||
}
|
||||
|
||||
func applyAccessLogBytesToNetworkTrend(ctx context.Context, now time.Time, nodeID string, since time.Time, points []NetworkTrendPoint) {
|
||||
if len(points) == 0 {
|
||||
return
|
||||
}
|
||||
// Prefer of_access_log_hourly (summed across hosts).
|
||||
if hourly, err := analyticsListAccessLogHourlyBytes(ctx, nodeID, since); err == nil && len(hourly) > 0 {
|
||||
for hourUnix, totals := range hourly {
|
||||
for index := range points {
|
||||
if points[index].BucketStartedAt.Unix() == hourUnix {
|
||||
points[index].BytesProvided = totals.provided
|
||||
points[index].BytesReceived = totals.received
|
||||
}
|
||||
}
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
buckets, err := repository.ListOpenFlareAccessLogBuckets(ctx, model.OpenFlareAccessLogBucketQuery{
|
||||
NodeID: nodeID,
|
||||
Since: since,
|
||||
Until: now,
|
||||
FoldMinutes: 60,
|
||||
SortBy: defaultAccessLogSortBy,
|
||||
SortOrder: sortOrderAsc,
|
||||
})
|
||||
if err != nil || len(buckets) == 0 {
|
||||
return
|
||||
}
|
||||
byEpoch := make(map[int64]*model.OpenFlareAccessLogBucketRow, len(buckets))
|
||||
for _, row := range buckets {
|
||||
if row == nil {
|
||||
continue
|
||||
}
|
||||
byEpoch[row.BucketEpoch] = row
|
||||
}
|
||||
for index := range points {
|
||||
epoch := points[index].BucketStartedAt.Unix()
|
||||
if row, ok := byEpoch[epoch]; ok {
|
||||
points[index].BytesProvided = row.BytesSent
|
||||
points[index].BytesReceived = row.RequestLength
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
type accessLogHourBytes struct {
|
||||
provided int64
|
||||
received int64
|
||||
}
|
||||
|
||||
func analyticsListAccessLogHourlyBytes(ctx context.Context, nodeID string, since time.Time) (map[int64]accessLogHourBytes, error) {
|
||||
rows, err := repository.ListOpenFlareAccessLogHourlySince(ctx, nodeID, since)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out := make(map[int64]accessLogHourBytes)
|
||||
for _, row := range rows {
|
||||
if row == nil {
|
||||
continue
|
||||
}
|
||||
key := row.Hour.UTC().Truncate(time.Hour).Unix()
|
||||
cur := out[key]
|
||||
cur.provided += row.BytesSent
|
||||
cur.received += row.RequestLength
|
||||
out[key] = cur
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// BuildTrafficTrendPointsFromHourly builds 24h traffic trend buckets from hourly rollups.
|
||||
// UniqueVisitorCount is left at 0: hourly UV is not summed (use TrafficSummary for exact UV).
|
||||
func BuildTrafficTrendPointsFromHourly(now time.Time, hourly []*model.OpenFlareTrafficHourly) []TrafficTrendPoint {
|
||||
start := trendWindowStart(now)
|
||||
points := make([]TrafficTrendPoint, observabilityTrendBuckets)
|
||||
for index := range points {
|
||||
points[index].BucketStartedAt = start.Add(time.Duration(index) * time.Hour)
|
||||
}
|
||||
for _, row := range hourly {
|
||||
if row == nil {
|
||||
continue
|
||||
}
|
||||
index, ok := trendBucketIndex(row.Hour, start)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
points[index].RequestCount += row.RequestCount
|
||||
points[index].ErrorCount += row.ErrorCount
|
||||
}
|
||||
return points
|
||||
}
|
||||
|
||||
// BuildCapacityTrendPoints builds 24h capacity trend buckets.
|
||||
func BuildCapacityTrendPoints(now time.Time, snapshots []*model.OpenFlareMetricSnapshot) []CapacityTrendPoint {
|
||||
start := trendWindowStart(now)
|
||||
points := make([]CapacityTrendPoint, observabilityTrendBuckets)
|
||||
accumulators := make([]capacityTrendAccumulator, observabilityTrendBuckets)
|
||||
for index := range points {
|
||||
points[index].BucketStartedAt = start.Add(time.Duration(index) * time.Hour)
|
||||
accumulators[index].nodes = make(map[string]struct{})
|
||||
}
|
||||
for _, snapshot := range snapshots {
|
||||
index, ok := trendBucketIndex(snapshot.CapturedAt, start)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
if snapshot.CPUUsagePercent > 0 {
|
||||
accumulators[index].cpuSum += snapshot.CPUUsagePercent
|
||||
accumulators[index].cpuCount++
|
||||
}
|
||||
if memoryUsage := Percentage(snapshot.MemoryUsedBytes, snapshot.MemoryTotalBytes); memoryUsage > 0 {
|
||||
accumulators[index].memSum += memoryUsage
|
||||
accumulators[index].memCount++
|
||||
}
|
||||
if snapshot.NodeID != "" {
|
||||
accumulators[index].nodes[snapshot.NodeID] = struct{}{}
|
||||
}
|
||||
}
|
||||
for index := range points {
|
||||
if accumulators[index].cpuCount > 0 {
|
||||
points[index].AverageCPUUsagePercent = accumulators[index].cpuSum / float64(accumulators[index].cpuCount)
|
||||
}
|
||||
if accumulators[index].memCount > 0 {
|
||||
points[index].AverageMemoryUsagePercent = accumulators[index].memSum / float64(accumulators[index].memCount)
|
||||
}
|
||||
points[index].ReportedNodes = len(accumulators[index].nodes)
|
||||
}
|
||||
return points
|
||||
}
|
||||
|
||||
// BuildCapacityTrendPointsFromHourly builds 24h capacity trend buckets from hourly aggregates.
|
||||
func BuildCapacityTrendPointsFromHourly(now time.Time, hourly []*model.OpenFlareMetricHourly) []CapacityTrendPoint {
|
||||
start := trendWindowStart(now)
|
||||
points := make([]CapacityTrendPoint, observabilityTrendBuckets)
|
||||
for index := range points {
|
||||
points[index].BucketStartedAt = start.Add(time.Duration(index) * time.Hour)
|
||||
}
|
||||
for _, row := range hourly {
|
||||
if row == nil {
|
||||
continue
|
||||
}
|
||||
index, ok := trendBucketIndex(row.Hour, start)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
points[index].AverageCPUUsagePercent = row.AverageCPUUsagePercent
|
||||
points[index].AverageMemoryUsagePercent = row.AverageMemoryUsagePercent
|
||||
points[index].ReportedNodes = row.ReportedNodes
|
||||
}
|
||||
return points
|
||||
}
|
||||
|
||||
func emptyNetworkTrendPoints(now time.Time) []NetworkTrendPoint {
|
||||
start := trendWindowStart(now)
|
||||
points := make([]NetworkTrendPoint, observabilityTrendBuckets)
|
||||
for index := range points {
|
||||
points[index].BucketStartedAt = start.Add(time.Duration(index) * time.Hour)
|
||||
}
|
||||
return points
|
||||
}
|
||||
|
||||
// BuildDiskIOTrendPoints builds 24h disk IO trend buckets.
|
||||
func BuildDiskIOTrendPoints(now time.Time, snapshots []*model.OpenFlareMetricSnapshot) []DiskIOTrendPoint {
|
||||
start := trendWindowStart(now)
|
||||
points := make([]DiskIOTrendPoint, observabilityTrendBuckets)
|
||||
accumulators := make([]snapshotTrendAccumulator, observabilityTrendBuckets)
|
||||
for index := range points {
|
||||
points[index].BucketStartedAt = start.Add(time.Duration(index) * time.Hour)
|
||||
accumulators[index].nodes = make(map[string]struct{})
|
||||
}
|
||||
sort.Slice(snapshots, func(i int, j int) bool {
|
||||
if snapshots[i].CapturedAt.Equal(snapshots[j].CapturedAt) {
|
||||
return snapshots[i].NodeID < snapshots[j].NodeID
|
||||
}
|
||||
return snapshots[i].CapturedAt.Before(snapshots[j].CapturedAt)
|
||||
})
|
||||
previousByNode := make(map[string]diskCounterState, len(snapshots))
|
||||
for _, snapshot := range snapshots {
|
||||
nodeKey := snapshot.NodeID
|
||||
if nodeKey == "" {
|
||||
nodeKey = unknownTrendNodeKey
|
||||
}
|
||||
previous := previousByNode[nodeKey]
|
||||
previousByNode[nodeKey] = diskCounterState{
|
||||
read: snapshot.DiskReadBytes,
|
||||
write: snapshot.DiskWriteBytes,
|
||||
seen: true,
|
||||
}
|
||||
if !previous.seen {
|
||||
continue
|
||||
}
|
||||
index, ok := trendBucketIndex(snapshot.CapturedAt, start)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
points[index].DiskReadBytes += nonNegativeDelta(snapshot.DiskReadBytes, previous.read)
|
||||
points[index].DiskWriteBytes += nonNegativeDelta(snapshot.DiskWriteBytes, previous.write)
|
||||
if snapshot.NodeID != "" {
|
||||
accumulators[index].nodes[snapshot.NodeID] = struct{}{}
|
||||
}
|
||||
}
|
||||
for index := range points {
|
||||
points[index].ReportedNodes = len(accumulators[index].nodes)
|
||||
}
|
||||
return points
|
||||
}
|
||||
|
||||
// BuildDiskIOTrendPointsFromHourly builds 24h disk IO trend buckets from hourly aggregates.
|
||||
func BuildDiskIOTrendPointsFromHourly(now time.Time, hourly []*model.OpenFlareMetricHourly) []DiskIOTrendPoint {
|
||||
start := trendWindowStart(now)
|
||||
points := make([]DiskIOTrendPoint, observabilityTrendBuckets)
|
||||
for index := range points {
|
||||
points[index].BucketStartedAt = start.Add(time.Duration(index) * time.Hour)
|
||||
}
|
||||
for _, row := range hourly {
|
||||
if row == nil {
|
||||
continue
|
||||
}
|
||||
index, ok := trendBucketIndex(row.Hour, start)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
points[index].DiskReadBytes += row.DiskReadBytes
|
||||
points[index].DiskWriteBytes += row.DiskWriteBytes
|
||||
points[index].ReportedNodes = row.ReportedNodes
|
||||
}
|
||||
return points
|
||||
}
|
||||
|
||||
func nonNegativeDelta(current int64, previous int64) int64 {
|
||||
delta := current - previous
|
||||
if delta < 0 {
|
||||
return 0
|
||||
}
|
||||
return delta
|
||||
}
|
||||
|
||||
func latestMetricSnapshot(snapshots []*model.OpenFlareMetricSnapshot) *model.OpenFlareMetricSnapshot {
|
||||
var latest *model.OpenFlareMetricSnapshot
|
||||
for _, snapshot := range snapshots {
|
||||
if snapshot == nil {
|
||||
continue
|
||||
}
|
||||
if latest == nil || snapshot.CapturedAt.After(latest.CapturedAt) {
|
||||
latest = snapshot
|
||||
}
|
||||
}
|
||||
return latest
|
||||
}
|
||||
|
||||
func matchEdgeHealth(
|
||||
capturedAt time.Time,
|
||||
health []*model.OpenFlareEdgeHealth,
|
||||
) *model.OpenFlareEdgeHealth {
|
||||
var matched *model.OpenFlareEdgeHealth
|
||||
bestDelta := metricSnapshotEdgeHealthMatchWindow + time.Second
|
||||
for _, row := range health {
|
||||
if row == nil {
|
||||
continue
|
||||
}
|
||||
delta := capturedAt.Sub(row.CapturedAt)
|
||||
if delta < 0 {
|
||||
delta = -delta
|
||||
}
|
||||
if delta > metricSnapshotEdgeHealthMatchWindow {
|
||||
continue
|
||||
}
|
||||
if matched == nil || delta < bestDelta {
|
||||
matched = row
|
||||
bestDelta = delta
|
||||
}
|
||||
}
|
||||
return matched
|
||||
}
|
||||
|
||||
// LatestMetricSnapshotsByNode returns the latest snapshot per node.
|
||||
func LatestMetricSnapshotsByNode(snapshots []*model.OpenFlareMetricSnapshot) map[string]*model.OpenFlareMetricSnapshot {
|
||||
result := make(map[string]*model.OpenFlareMetricSnapshot, len(snapshots))
|
||||
for _, snapshot := range snapshots {
|
||||
if snapshot == nil || snapshot.NodeID == "" {
|
||||
continue
|
||||
}
|
||||
if existing, ok := result[snapshot.NodeID]; ok && !snapshot.CapturedAt.After(existing.CapturedAt) {
|
||||
continue
|
||||
}
|
||||
result[snapshot.NodeID] = snapshot
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
// ActiveHealthEventsByNode groups active health events by node id.
|
||||
func ActiveHealthEventsByNode(events []*model.OpenFlareHealthEvent) map[string][]*model.OpenFlareHealthEvent {
|
||||
result := make(map[string][]*model.OpenFlareHealthEvent)
|
||||
for _, event := range events {
|
||||
if event == nil || event.NodeID == "" {
|
||||
continue
|
||||
}
|
||||
result[event.NodeID] = append(result[event.NodeID], event)
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
// Percentage returns used/total as a percentage.
|
||||
func Percentage(used int64, total int64) float64 {
|
||||
if used <= 0 || total <= 0 {
|
||||
return 0
|
||||
}
|
||||
return (float64(used) / float64(total)) * percentageMultiplier
|
||||
}
|
||||
|
||||
func toDistributionItems(values distributionAccumulator, limit int) []DistributionItem {
|
||||
if len(values) == 0 {
|
||||
return []DistributionItem{}
|
||||
}
|
||||
items := make([]DistributionItem, 0, len(values))
|
||||
for key, value := range values {
|
||||
if strings.TrimSpace(key) == "" || value <= 0 {
|
||||
continue
|
||||
}
|
||||
items = append(items, DistributionItem{Key: key, Value: value})
|
||||
}
|
||||
sort.Slice(items, func(i int, j int) bool {
|
||||
if items[i].Value == items[j].Value {
|
||||
return items[i].Key < items[j].Key
|
||||
}
|
||||
return items[i].Value > items[j].Value
|
||||
})
|
||||
if limit > 0 && len(items) > limit {
|
||||
items = items[:limit]
|
||||
}
|
||||
return items
|
||||
}
|
||||
|
||||
func trendWindowStart(now time.Time) time.Time {
|
||||
return now.Truncate(time.Hour).Add(-(observabilityTrendBuckets - 1) * time.Hour)
|
||||
}
|
||||
|
||||
func trendBucketIndex(timestamp time.Time, start time.Time) (int, bool) {
|
||||
if timestamp.Before(start) {
|
||||
return 0, false
|
||||
}
|
||||
delta := timestamp.Sub(start)
|
||||
index := int(delta / time.Hour)
|
||||
if index < 0 || index >= observabilityTrendBuckets {
|
||||
return 0, false
|
||||
}
|
||||
return index, true
|
||||
}
|
||||
@@ -0,0 +1,168 @@
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
package observability
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"Wavelet/openflare/plugins/server/kernel/model"
|
||||
)
|
||||
|
||||
func TestBuildTrafficTrendPointsFromHourlyBucketsByHour(t *testing.T) {
|
||||
now := time.Date(2026, 7, 2, 15, 30, 0, 0, time.UTC)
|
||||
hourly := []*model.OpenFlareTrafficHourly{
|
||||
{
|
||||
NodeID: "node-a",
|
||||
Hour: now.Add(-2 * time.Hour).Truncate(time.Hour),
|
||||
RequestCount: 12,
|
||||
ErrorCount: 1,
|
||||
UniqueVisitorCount: 4,
|
||||
},
|
||||
}
|
||||
points := BuildTrafficTrendPointsFromHourly(now, hourly)
|
||||
if len(points) != observabilityTrendBuckets {
|
||||
t.Fatalf("BuildTrafficTrendPointsFromHourly() len = %d, want %d", len(points), observabilityTrendBuckets)
|
||||
}
|
||||
// Hourly UV must not be summed into trend points.
|
||||
for _, point := range points {
|
||||
if point.UniqueVisitorCount != 0 {
|
||||
t.Fatalf("UniqueVisitorCount = %d, want 0 on hourly path", point.UniqueVisitorCount)
|
||||
}
|
||||
}
|
||||
index, ok := trendBucketIndex(now.Add(-2*time.Hour).Truncate(time.Hour), trendWindowStart(now))
|
||||
if !ok {
|
||||
t.Fatal("expected valid bucket index")
|
||||
}
|
||||
if points[index].RequestCount != 12 {
|
||||
t.Fatalf("request_count = %d, want 12", points[index].RequestCount)
|
||||
}
|
||||
if points[index].ErrorCount != 1 {
|
||||
t.Fatalf("error_count = %d, want 1", points[index].ErrorCount)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildMetricSnapshotViewsMergesEdgeHealthConnections(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
capturedAt := time.Date(2026, 6, 19, 12, 0, 0, 0, time.UTC)
|
||||
snapshots := []*model.OpenFlareMetricSnapshot{
|
||||
{
|
||||
ID: 1,
|
||||
NodeID: "node-a",
|
||||
CapturedAt: capturedAt,
|
||||
CPUUsagePercent: 12.5,
|
||||
},
|
||||
}
|
||||
edgeHealth := []*model.OpenFlareEdgeHealth{
|
||||
{
|
||||
NodeID: "node-a",
|
||||
CapturedAt: capturedAt.Add(5 * time.Second),
|
||||
Status: "healthy",
|
||||
Connections: 7,
|
||||
},
|
||||
}
|
||||
|
||||
views := BuildMetricSnapshotViews(snapshots, edgeHealth)
|
||||
if len(views) != 1 {
|
||||
t.Fatalf("BuildMetricSnapshotViews() len = %d, want 1", len(views))
|
||||
}
|
||||
if views[0].OpenrestyConnections != 7 {
|
||||
t.Fatalf("OpenrestyConnections = %d, want 7", views[0].OpenrestyConnections)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildTrafficWindowSummaryFromAccessLogsNilWithoutData(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// Without an access-log store / data, summary is nil.
|
||||
if summary := buildTrafficWindowSummaryFromAccessLogs(t.Context(), "missing", time.Now().Add(-time.Hour), time.Now()); summary != nil {
|
||||
t.Fatalf("buildTrafficWindowSummaryFromAccessLogs() = %#v, want nil", summary)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildCapacityTrendPointsFromHourlyFillsBuckets(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
now := time.Date(2026, 7, 10, 9, 30, 0, 0, time.UTC)
|
||||
hourly := []*model.OpenFlareMetricHourly{
|
||||
{
|
||||
Hour: now.Add(-3 * time.Hour).Truncate(time.Hour),
|
||||
AverageCPUUsagePercent: 42.5,
|
||||
AverageMemoryUsagePercent: 61.2,
|
||||
ReportedNodes: 1,
|
||||
},
|
||||
{
|
||||
Hour: now.Truncate(time.Hour),
|
||||
AverageCPUUsagePercent: 12.0,
|
||||
AverageMemoryUsagePercent: 50.0,
|
||||
ReportedNodes: 2,
|
||||
},
|
||||
}
|
||||
|
||||
points := BuildCapacityTrendPointsFromHourly(now, hourly)
|
||||
if len(points) != observabilityTrendBuckets {
|
||||
t.Fatalf("len = %d, want %d", len(points), observabilityTrendBuckets)
|
||||
}
|
||||
if points[len(points)-4].AverageCPUUsagePercent != 42.5 {
|
||||
t.Fatalf("hour-3 cpu = %v, want 42.5", points[len(points)-4].AverageCPUUsagePercent)
|
||||
}
|
||||
if points[len(points)-1].ReportedNodes != 2 {
|
||||
t.Fatalf("current hour reported_nodes = %d, want 2", points[len(points)-1].ReportedNodes)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEmptyNetworkTrendPointsHas24Buckets(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
now := time.Date(2026, 7, 10, 9, 30, 0, 0, time.UTC)
|
||||
points := emptyNetworkTrendPoints(now)
|
||||
if len(points) != observabilityTrendBuckets {
|
||||
t.Fatalf("len(points) = %d, want %d", len(points), observabilityTrendBuckets)
|
||||
}
|
||||
if !points[0].BucketStartedAt.Before(points[len(points)-1].BucketStartedAt) {
|
||||
t.Fatalf("bucket order invalid: first=%v last=%v", points[0].BucketStartedAt, points[len(points)-1].BucketStartedAt)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildHealthSummaryUsesTrafficSummary(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
snapshot := &model.OpenFlareMetricSnapshot{
|
||||
CPUUsagePercent: 10,
|
||||
MemoryUsedBytes: 1,
|
||||
MemoryTotalBytes: 10,
|
||||
}
|
||||
traffic := &TrafficWindowSummary{
|
||||
RequestCount: 200,
|
||||
ErrorCount: 20, // 10% error rate
|
||||
}
|
||||
summary := buildHealthSummary(snapshot, traffic, nil)
|
||||
if !summary.HasTrafficRisk {
|
||||
t.Fatal("HasTrafficRisk = false, want true for 10% error rate with >=100 requests")
|
||||
}
|
||||
if summary.HasCapacityRisk {
|
||||
t.Fatal("HasCapacityRisk = true, want false")
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildDiskIOTrendPointsFromHourlyFillsBuckets(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
now := time.Date(2026, 7, 10, 9, 30, 0, 0, time.UTC)
|
||||
hourly := []*model.OpenFlareMetricHourly{
|
||||
{
|
||||
Hour: now.Add(-1 * time.Hour).Truncate(time.Hour),
|
||||
DiskReadBytes: 1024,
|
||||
DiskWriteBytes: 2048,
|
||||
ReportedNodes: 1,
|
||||
},
|
||||
}
|
||||
|
||||
points := BuildDiskIOTrendPointsFromHourly(now, hourly)
|
||||
prev := points[len(points)-2]
|
||||
if prev.DiskReadBytes != 1024 || prev.DiskWriteBytes != 2048 {
|
||||
t.Fatalf("previous hour disk io = %#v, want read=1024 write=2048", prev)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,65 @@
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
package chwriter
|
||||
|
||||
import (
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
const dedupTTL = 2 * time.Minute
|
||||
|
||||
type dedupSet struct {
|
||||
mu sync.Mutex
|
||||
keys map[string]time.Time
|
||||
lastCleanup time.Time
|
||||
}
|
||||
|
||||
func newDedupSet() *dedupSet {
|
||||
return &dedupSet{
|
||||
keys: make(map[string]time.Time),
|
||||
lastCleanup: time.Now(),
|
||||
}
|
||||
}
|
||||
|
||||
// markIfNew records key when it has not been seen within dedupTTL.
|
||||
func (s *dedupSet) markIfNew(key string) bool {
|
||||
if s == nil || key == "" {
|
||||
return false
|
||||
}
|
||||
|
||||
now := time.Now()
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
|
||||
s.cleanupExpiredLocked(now)
|
||||
|
||||
if expiresAt, exists := s.keys[key]; exists && now.Before(expiresAt) {
|
||||
return false
|
||||
}
|
||||
s.keys[key] = now.Add(dedupTTL)
|
||||
return true
|
||||
}
|
||||
|
||||
// unmark removes a key so a later enqueue or flush retry may accept it again.
|
||||
func (s *dedupSet) unmark(key string) {
|
||||
if s == nil || key == "" {
|
||||
return
|
||||
}
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
delete(s.keys, key)
|
||||
}
|
||||
|
||||
func (s *dedupSet) cleanupExpiredLocked(now time.Time) {
|
||||
if now.Sub(s.lastCleanup) < 30*time.Second {
|
||||
return
|
||||
}
|
||||
for existing, expiresAt := range s.keys {
|
||||
if now.After(expiresAt) {
|
||||
delete(s.keys, existing)
|
||||
}
|
||||
}
|
||||
s.lastCleanup = now
|
||||
}
|
||||
@@ -0,0 +1,186 @@
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
package chwriter
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
analyticsmodel "Wavelet/openflare/plugins/server/kernel/model/analytics"
|
||||
"Wavelet/pkg/batchwriter"
|
||||
)
|
||||
|
||||
func TestDedupSetMarkIfNew(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
set := newDedupSet()
|
||||
if !set.markIfNew("node-a|1") {
|
||||
t.Fatal("markIfNew() = false, want true on first key")
|
||||
}
|
||||
if set.markIfNew("node-a|1") {
|
||||
t.Fatal("markIfNew() = true, want false on duplicate key")
|
||||
}
|
||||
if !set.markIfNew("node-b|1") {
|
||||
t.Fatal("markIfNew() = false, want true on different key")
|
||||
}
|
||||
if set.markIfNew("") {
|
||||
t.Fatal("markIfNew() = true, want false on empty key")
|
||||
}
|
||||
}
|
||||
|
||||
func TestDedupSetUnmarkAllowsRetry(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
set := newDedupSet()
|
||||
if !set.markIfNew("k") {
|
||||
t.Fatal("markIfNew() = false, want true")
|
||||
}
|
||||
set.unmark("k")
|
||||
if !set.markIfNew("k") {
|
||||
t.Fatal("markIfNew() after unmark = false, want true")
|
||||
}
|
||||
}
|
||||
|
||||
func TestQueueWithDedupDoesNotMarkWhenEnqueueFails(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
cfg := batchwriter.DefaultConfig()
|
||||
cfg.QueueSize = 1
|
||||
cfg.MaxBatchSize = 10
|
||||
cfg.FlushInterval = time.Hour
|
||||
|
||||
// Block the worker so the queue stays full after one enqueue.
|
||||
block := make(chan struct{})
|
||||
writer, err := batchwriter.New[int](cfg, func(context.Context, []int) error {
|
||||
<-block
|
||||
return nil
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("New() error = %v", err)
|
||||
}
|
||||
writer.Start(context.Background())
|
||||
t.Cleanup(func() {
|
||||
close(block)
|
||||
stopCtx, cancel := context.WithTimeout(context.Background(), time.Second)
|
||||
defer cancel()
|
||||
_ = writer.Stop(stopCtx)
|
||||
})
|
||||
|
||||
// Fill the channel buffer (and the worker's current receive slot may empty one).
|
||||
// Keep enqueueing until full so subsequent queueWithDedup fails.
|
||||
for i := 0; i < cfg.QueueSize+2; i++ {
|
||||
_ = writer.TryEnqueue(i)
|
||||
if writer.IsFull() {
|
||||
break
|
||||
}
|
||||
}
|
||||
if !writer.IsFull() {
|
||||
t.Fatal("writer not full after filling; cannot test enqueue failure path")
|
||||
}
|
||||
|
||||
dedup := newDedupSet()
|
||||
queueWithDedup(writer, dedup, "dedup-key", 99)
|
||||
// Key must not remain marked after failed enqueue.
|
||||
if !dedup.markIfNew("dedup-key") {
|
||||
t.Fatal("dedup key still marked after failed enqueue; want unmark")
|
||||
}
|
||||
}
|
||||
|
||||
func TestQueueWithDedupMarksOnlyOnSuccess(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
cfg := batchwriter.DefaultConfig()
|
||||
cfg.MaxBatchSize = 100
|
||||
cfg.FlushInterval = time.Hour
|
||||
|
||||
writer, err := batchwriter.New[int](cfg, func(context.Context, []int) error { return nil })
|
||||
if err != nil {
|
||||
t.Fatalf("New() error = %v", err)
|
||||
}
|
||||
writer.Start(context.Background())
|
||||
t.Cleanup(func() {
|
||||
stopCtx, cancel := context.WithTimeout(context.Background(), time.Second)
|
||||
defer cancel()
|
||||
_ = writer.Stop(stopCtx)
|
||||
})
|
||||
|
||||
dedup := newDedupSet()
|
||||
queueWithDedup(writer, dedup, "ok-key", 1)
|
||||
if dedup.markIfNew("ok-key") {
|
||||
t.Fatal("markIfNew() = true after successful enqueue, want false (key marked)")
|
||||
}
|
||||
}
|
||||
|
||||
func TestFlushErrorHandlerUnmarksKeys(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dedup := newDedupSet()
|
||||
flushErr := errors.New("ch down")
|
||||
|
||||
var (
|
||||
mu sync.Mutex
|
||||
errCount int
|
||||
)
|
||||
|
||||
cfg := batchwriter.Config{
|
||||
Name: "test_obs",
|
||||
QueueSize: 10,
|
||||
MaxBatchSize: 1,
|
||||
FlushInterval: time.Hour,
|
||||
}
|
||||
keyFn := func(s analyticsmodel.NodeMetricSnapshot) string {
|
||||
return metricSnapshotKey(s)
|
||||
}
|
||||
writer, err := batchwriter.New(
|
||||
cfg,
|
||||
func(context.Context, []analyticsmodel.NodeMetricSnapshot) error { return flushErr },
|
||||
batchwriter.WithFlushErrorHandler[analyticsmodel.NodeMetricSnapshot](func(_ context.Context, items []analyticsmodel.NodeMetricSnapshot, err error) {
|
||||
mu.Lock()
|
||||
errCount++
|
||||
mu.Unlock()
|
||||
for _, item := range items {
|
||||
dedup.unmark(keyFn(item))
|
||||
}
|
||||
}),
|
||||
)
|
||||
if err != nil {
|
||||
t.Fatalf("New() error = %v", err)
|
||||
}
|
||||
writer.Start(context.Background())
|
||||
t.Cleanup(func() {
|
||||
stopCtx, cancel := context.WithTimeout(context.Background(), time.Second)
|
||||
defer cancel()
|
||||
_ = writer.Stop(stopCtx)
|
||||
})
|
||||
|
||||
item := analyticsmodel.NodeMetricSnapshot{
|
||||
NodeID: "n1",
|
||||
CapturedAt: time.Unix(1, 0).UTC(),
|
||||
}
|
||||
key := keyFn(item)
|
||||
if !dedup.markIfNew(key) {
|
||||
t.Fatal("markIfNew failed")
|
||||
}
|
||||
if !writer.TryEnqueue(item) {
|
||||
t.Fatal("TryEnqueue failed")
|
||||
}
|
||||
|
||||
deadline := time.Now().Add(time.Second)
|
||||
for {
|
||||
mu.Lock()
|
||||
ready := errCount >= 1
|
||||
mu.Unlock()
|
||||
if ready || time.Now().After(deadline) {
|
||||
break
|
||||
}
|
||||
time.Sleep(5 * time.Millisecond)
|
||||
}
|
||||
|
||||
if !dedup.markIfNew(key) {
|
||||
t.Fatal("key still marked after flush error unmark; want available for retry")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,88 @@
|
||||
//go:build live_ch
|
||||
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
package chwriter_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"Wavelet/openflare/plugins/server/kernel/repository"
|
||||
|
||||
"Wavelet/openflare/plugins/server/domain/observability/chwriter"
|
||||
"Wavelet/openflare/plugins/server/kernel/model"
|
||||
db "Wavelet/plugins/infra/database"
|
||||
)
|
||||
|
||||
// Run with Docker ClickHouse + config.yaml:
|
||||
//
|
||||
// go test -tags live_ch ./internal/apps/openflare/chwriter -run TestLiveAppWritePath -count=1 -timeout 2m
|
||||
func TestLiveAppWritePath(t *testing.T) {
|
||||
if db.ChConn == nil {
|
||||
t.Skip("ClickHouse connection not ready")
|
||||
}
|
||||
ctx := context.Background()
|
||||
chwriter.Init(ctx)
|
||||
|
||||
now := time.Now().UTC()
|
||||
nodeID := "e2e-app-write-" + now.Format("150405")
|
||||
if err := repository.InsertOpenFlareMetricSnapshot(ctx, &model.OpenFlareMetricSnapshot{
|
||||
NodeID: nodeID,
|
||||
CapturedAt: now,
|
||||
CPUUsagePercent: 33.3,
|
||||
MemoryUsedBytes: 111,
|
||||
MemoryTotalBytes: 1000,
|
||||
StorageUsedBytes: 222,
|
||||
StorageTotalBytes: 2000,
|
||||
DiskReadBytes: 10,
|
||||
DiskWriteBytes: 20,
|
||||
NetworkRxBytes: 30,
|
||||
NetworkTxBytes: 40,
|
||||
}); err != nil {
|
||||
t.Fatalf("InsertOpenFlareMetricSnapshot: %v", err)
|
||||
}
|
||||
|
||||
deadline := time.Now().Add(45 * time.Second)
|
||||
var found bool
|
||||
for time.Now().Before(deadline) {
|
||||
rows, err := repository.ListOpenFlareMetricSnapshotsSince(ctx, nodeID, now.Add(-time.Minute), 10)
|
||||
if err != nil {
|
||||
t.Fatalf("ListOpenFlareMetricSnapshotsSince: %v", err)
|
||||
}
|
||||
if len(rows) > 0 {
|
||||
found = true
|
||||
t.Logf("found snapshot id=%d cpu=%.1f after flush", rows[0].ID, rows[0].CPUUsagePercent)
|
||||
break
|
||||
}
|
||||
time.Sleep(2 * time.Second)
|
||||
}
|
||||
if !found {
|
||||
t.Fatal("metric snapshot not visible in ClickHouse after flush wait")
|
||||
}
|
||||
|
||||
latest, err := repository.ListOpenFlareLatestMetricSnapshotsSince(ctx, "", now.Add(-time.Hour))
|
||||
if err != nil {
|
||||
t.Fatalf("ListOpenFlareLatestMetricSnapshotsSince: %v", err)
|
||||
}
|
||||
var latestOK bool
|
||||
for _, row := range latest {
|
||||
if row != nil && row.NodeID == nodeID {
|
||||
latestOK = true
|
||||
break
|
||||
}
|
||||
}
|
||||
if !latestOK {
|
||||
t.Fatalf("latest-per-node query missing node %s (rows=%d)", nodeID, len(latest))
|
||||
}
|
||||
|
||||
stats := chwriter.WriterStats()
|
||||
if len(stats) == 0 {
|
||||
t.Fatal("WriterStats empty after Init")
|
||||
}
|
||||
for _, s := range stats {
|
||||
t.Logf("writer %s running=%v depth=%d drops=%d flush_err=%d", s.Name, s.Running, s.Depth, s.Drops, s.FlushErrors)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,400 @@
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
// Package chwriter queues OpenFlare ClickHouse writes and flushes them through
|
||||
// internal/infra/persistence/batchwriter with per-table writer instances.
|
||||
package chwriter
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
analyticsmodel "Wavelet/openflare/plugins/server/kernel/model/analytics"
|
||||
"Wavelet/openflare/plugins/server/kernel/repository/logstore"
|
||||
"Wavelet/pkg/batchwriter"
|
||||
"Wavelet/pkg/logger"
|
||||
)
|
||||
|
||||
const (
|
||||
// Observability traffic is sparse (heartbeat ~10s/node). Prefer larger batches to
|
||||
// cut ClickHouse parts/merges; MaxFlushWait bounds visibility lag for single-node labs.
|
||||
observabilityQueueSize = 5_000
|
||||
observabilityMaxBatchSize = 500
|
||||
observabilityMinBatchSize = 20
|
||||
observabilityFlushEvery = 10 * time.Second
|
||||
observabilityMaxFlushWait = 30 * time.Second
|
||||
|
||||
nodeAccessLogQueueSize = 10_000
|
||||
nodeAccessLogMaxBatchSize = 1_000
|
||||
nodeAccessLogMinBatchSize = 50
|
||||
nodeAccessLogFlushEvery = 2 * time.Second
|
||||
nodeAccessLogMaxFlushWait = 5 * time.Second
|
||||
|
||||
// flushAttempts is total tries (1 initial + short retries) before giving up a batch.
|
||||
flushAttempts = 2
|
||||
flushRetryBackoff = 50 * time.Millisecond
|
||||
)
|
||||
|
||||
var (
|
||||
initOnce sync.Once
|
||||
|
||||
metricSnapshotWriter *batchwriter.Writer[analyticsmodel.NodeMetricSnapshot]
|
||||
edgeHealthWriter *batchwriter.Writer[analyticsmodel.NodeEdgeHealth]
|
||||
frpsWriter *batchwriter.Writer[analyticsmodel.NodeObsFrps]
|
||||
frpcWriter *batchwriter.Writer[analyticsmodel.NodeObsFrpc]
|
||||
nodeAccessLogWriter *batchwriter.Writer[analyticsmodel.NodeAccessLog]
|
||||
|
||||
metricSnapshotDedup *dedupSet
|
||||
edgeHealthDedup *dedupSet
|
||||
frpsDedup *dedupSet
|
||||
frpcDedup *dedupSet
|
||||
)
|
||||
|
||||
// Init starts OpenFlare log batch writers. Safe to call multiple times.
|
||||
// Writers always initialize regardless of ClickHouse.enabled; the active log
|
||||
// store is resolved via logstore at flush time (PG/SQLite when CH is not active).
|
||||
func Init(ctx context.Context) {
|
||||
initOnce.Do(func() {
|
||||
metricSnapshotDedup = newDedupSet()
|
||||
edgeHealthDedup = newDedupSet()
|
||||
frpsDedup = newDedupSet()
|
||||
frpcDedup = newDedupSet()
|
||||
|
||||
metricSnapshotWriter = mustNewObservabilityWriter(
|
||||
"metric_snapshots",
|
||||
withFlushRetries(flushNodeMetricSnapshots),
|
||||
metricSnapshotDedup,
|
||||
metricSnapshotKey,
|
||||
)
|
||||
edgeHealthWriter = mustNewObservabilityWriter(
|
||||
"edge_health",
|
||||
withFlushRetries(flushNodeEdgeHealth),
|
||||
edgeHealthDedup,
|
||||
edgeHealthKey,
|
||||
)
|
||||
frpsWriter = mustNewObservabilityWriter(
|
||||
"frps_obs",
|
||||
withFlushRetries(flushNodeObsFrps),
|
||||
frpsDedup,
|
||||
frpsKey,
|
||||
)
|
||||
frpcWriter = mustNewObservabilityWriter(
|
||||
"frpc_obs",
|
||||
withFlushRetries(flushNodeObsFrpc),
|
||||
frpcDedup,
|
||||
frpcKey,
|
||||
)
|
||||
nodeAccessLogWriter = mustNewNodeAccessLogWriter()
|
||||
|
||||
metricSnapshotWriter.Start(ctx)
|
||||
edgeHealthWriter.Start(ctx)
|
||||
frpsWriter.Start(ctx)
|
||||
frpcWriter.Start(ctx)
|
||||
nodeAccessLogWriter.Start(ctx)
|
||||
|
||||
wireModelInsertHooks()
|
||||
})
|
||||
}
|
||||
|
||||
// Stop drains all OpenFlare ClickHouse writers.
|
||||
func Stop(ctx context.Context) error {
|
||||
if !running() {
|
||||
return nil
|
||||
}
|
||||
|
||||
var firstErr error
|
||||
for _, writer := range []batchStopper{
|
||||
metricSnapshotWriter,
|
||||
edgeHealthWriter,
|
||||
frpsWriter,
|
||||
frpcWriter,
|
||||
nodeAccessLogWriter,
|
||||
} {
|
||||
if writer == nil {
|
||||
continue
|
||||
}
|
||||
if err := writer.Stop(ctx); err != nil && firstErr == nil {
|
||||
firstErr = err
|
||||
}
|
||||
}
|
||||
return firstErr
|
||||
}
|
||||
|
||||
// Drain 等待所有 OpenFlare 日志 writer 的在途批次落库:轮询队列 Depth 归零后
|
||||
// 再保持一个最大 flush 周期(observabilityFlushEvery)持续为空才返回;
|
||||
// 不停止 writer(迁移冻结后由 ensureWritable 拒绝新写入)。未初始化时直接返回 nil。
|
||||
func Drain(ctx context.Context) error {
|
||||
return drainWriters(ctx, WriterStats, observabilityFlushEvery)
|
||||
}
|
||||
|
||||
// drainWriters 轮询 stats 直至所有队列 Depth=0 并持续 quietPeriod 无新积压。
|
||||
func drainWriters(ctx context.Context, stats func() []batchwriter.Stats, quietPeriod time.Duration) error {
|
||||
if !running() {
|
||||
return nil
|
||||
}
|
||||
ticker := time.NewTicker(drainPollInterval)
|
||||
defer ticker.Stop()
|
||||
var quietSince time.Time
|
||||
for {
|
||||
if allDepthZero(stats()) {
|
||||
if quietSince.IsZero() {
|
||||
quietSince = time.Now()
|
||||
} else if time.Since(quietSince) >= quietPeriod {
|
||||
return nil
|
||||
}
|
||||
} else {
|
||||
quietSince = time.Time{}
|
||||
}
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return ctx.Err()
|
||||
case <-ticker.C:
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// drainPollInterval 队列轮询间隔。
|
||||
const drainPollInterval = 50 * time.Millisecond
|
||||
|
||||
func allDepthZero(stats []batchwriter.Stats) bool {
|
||||
for _, s := range stats {
|
||||
if s.Depth > 0 {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// WriterStats returns queue depth and failure counters for all OpenFlare writers.
|
||||
func WriterStats() []batchwriter.Stats {
|
||||
writers := []statsProvider{
|
||||
metricSnapshotWriter,
|
||||
edgeHealthWriter,
|
||||
frpsWriter,
|
||||
frpcWriter,
|
||||
nodeAccessLogWriter,
|
||||
}
|
||||
out := make([]batchwriter.Stats, 0, len(writers))
|
||||
for _, w := range writers {
|
||||
if w == nil {
|
||||
continue
|
||||
}
|
||||
out = append(out, w.Stats())
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// QueueMetricSnapshot enqueues a metric snapshot for asynchronous flush.
|
||||
func QueueMetricSnapshot(snapshot analyticsmodel.NodeMetricSnapshot) {
|
||||
queueWithDedup(metricSnapshotWriter, metricSnapshotDedup, metricSnapshotKey(snapshot), snapshot)
|
||||
}
|
||||
|
||||
// QueueEdgeHealth enqueues an L2 edge health snapshot for asynchronous flush.
|
||||
func QueueEdgeHealth(row analyticsmodel.NodeEdgeHealth) {
|
||||
queueWithDedup(edgeHealthWriter, edgeHealthDedup, edgeHealthKey(row), row)
|
||||
}
|
||||
|
||||
// QueueFrpsObservation enqueues an FRPS observation for asynchronous flush.
|
||||
func QueueFrpsObservation(observation analyticsmodel.NodeObsFrps) {
|
||||
queueWithDedup(frpsWriter, frpsDedup, frpsKey(observation), observation)
|
||||
}
|
||||
|
||||
// QueueFrpcObservation enqueues an FRPC observation for asynchronous flush.
|
||||
func QueueFrpcObservation(observation analyticsmodel.NodeObsFrpc) {
|
||||
queueWithDedup(frpcWriter, frpcDedup, frpcKey(observation), observation)
|
||||
}
|
||||
|
||||
// QueueNodeAccessLogs enqueues node access logs for asynchronous flush.
|
||||
func QueueNodeAccessLogs(logs []analyticsmodel.NodeAccessLog) {
|
||||
if nodeAccessLogWriter == nil || len(logs) == 0 {
|
||||
return
|
||||
}
|
||||
for _, logItem := range logs {
|
||||
nodeAccessLogWriter.TryEnqueue(logItem)
|
||||
}
|
||||
}
|
||||
|
||||
func queueWithDedup[T any](writer *batchwriter.Writer[T], dedup *dedupSet, key string, item T) {
|
||||
if writer == nil {
|
||||
return
|
||||
}
|
||||
// Mark first so concurrent duplicates still collapse; release on enqueue failure
|
||||
// so a full queue does not permanently suppress the item.
|
||||
if !dedup.markIfNew(key) {
|
||||
return
|
||||
}
|
||||
if !writer.TryEnqueue(item) {
|
||||
dedup.unmark(key)
|
||||
}
|
||||
}
|
||||
|
||||
func mustNewObservabilityWriter[T any](
|
||||
name string,
|
||||
flush batchwriter.FlushFunc[T],
|
||||
dedup *dedupSet,
|
||||
keyFn func(T) string,
|
||||
) *batchwriter.Writer[T] {
|
||||
cfg := batchwriter.Config{
|
||||
Name: name,
|
||||
QueueSize: observabilityQueueSize,
|
||||
MaxBatchSize: observabilityMaxBatchSize,
|
||||
MinBatchSize: observabilityMinBatchSize,
|
||||
FlushInterval: observabilityFlushEvery,
|
||||
MaxFlushWait: observabilityMaxFlushWait,
|
||||
}
|
||||
writer, err := batchwriter.New(
|
||||
cfg,
|
||||
flush,
|
||||
withObservabilityDropHandler[T](name),
|
||||
batchwriter.WithFlushErrorHandler[T](func(ctx context.Context, items []T, err error) {
|
||||
logger.ErrorF(ctx, "[OpenFlare] flush %s failed (batch=%d): %v", name, len(items), err)
|
||||
if dedup == nil || keyFn == nil {
|
||||
return
|
||||
}
|
||||
for _, item := range items {
|
||||
dedup.unmark(keyFn(item))
|
||||
}
|
||||
}),
|
||||
)
|
||||
if err != nil {
|
||||
panic(fmt.Sprintf("openflare chwriter %s: %v", name, err))
|
||||
}
|
||||
return writer
|
||||
}
|
||||
|
||||
func mustNewNodeAccessLogWriter() *batchwriter.Writer[analyticsmodel.NodeAccessLog] {
|
||||
cfg := batchwriter.Config{
|
||||
Name: "node_access_logs",
|
||||
QueueSize: nodeAccessLogQueueSize,
|
||||
MaxBatchSize: nodeAccessLogMaxBatchSize,
|
||||
MinBatchSize: nodeAccessLogMinBatchSize,
|
||||
FlushInterval: nodeAccessLogFlushEvery,
|
||||
MaxFlushWait: nodeAccessLogMaxFlushWait,
|
||||
}
|
||||
writer, err := batchwriter.New[analyticsmodel.NodeAccessLog](
|
||||
cfg,
|
||||
withFlushRetries(flushNodeAccessLogs),
|
||||
batchwriter.WithDropHandler[analyticsmodel.NodeAccessLog](func(item analyticsmodel.NodeAccessLog) {
|
||||
logger.WarnF(context.Background(), "[OpenFlare] node access log queue full, dropping log for node %s path %s", item.NodeID, item.Path)
|
||||
}),
|
||||
batchwriter.WithFlushErrorHandler[analyticsmodel.NodeAccessLog](func(ctx context.Context, items []analyticsmodel.NodeAccessLog, err error) {
|
||||
logger.ErrorF(ctx, "[OpenFlare] flush node access logs failed (batch=%d): %v", len(items), err)
|
||||
}),
|
||||
)
|
||||
if err != nil {
|
||||
panic(fmt.Sprintf("openflare chwriter node_access_logs: %v", err))
|
||||
}
|
||||
return writer
|
||||
}
|
||||
|
||||
func withObservabilityDropHandler[T any](name string) batchwriter.Option[T] {
|
||||
return batchwriter.WithDropHandler(func(_ T) {
|
||||
logger.WarnF(context.Background(), "[OpenFlare] %s queue full, dropping observability item", name)
|
||||
})
|
||||
}
|
||||
|
||||
// withFlushRetries wraps a flush function with a short retry to ride out brief CH blips.
|
||||
func withFlushRetries[T any](flush batchwriter.FlushFunc[T]) batchwriter.FlushFunc[T] {
|
||||
return func(ctx context.Context, items []T) error {
|
||||
var err error
|
||||
for attempt := 1; attempt <= flushAttempts; attempt++ {
|
||||
err = flush(ctx, items)
|
||||
if err == nil {
|
||||
return nil
|
||||
}
|
||||
if attempt == flushAttempts {
|
||||
break
|
||||
}
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return ctx.Err()
|
||||
case <-time.After(flushRetryBackoff * time.Duration(attempt)):
|
||||
}
|
||||
}
|
||||
return err
|
||||
}
|
||||
}
|
||||
|
||||
func wireModelInsertHooks() {
|
||||
logstore.SetObservabilityHooks(logstore.ObservabilityHooks{
|
||||
QueueMetricSnapshot: QueueMetricSnapshot,
|
||||
QueueEdgeHealth: QueueEdgeHealth,
|
||||
QueueNodeObsFrps: QueueFrpsObservation,
|
||||
QueueNodeObsFrpc: QueueFrpcObservation,
|
||||
})
|
||||
logstore.SetAccessLogHooks(logstore.AccessLogHooks{
|
||||
QueueNodeAccessLogs: QueueNodeAccessLogs,
|
||||
})
|
||||
}
|
||||
|
||||
// 以下 flush 函数作为 batchwriter 的落库目标:激活库由 logstore 在 flush 时决定。
|
||||
|
||||
func flushNodeMetricSnapshots(ctx context.Context, rows []analyticsmodel.NodeMetricSnapshot) error {
|
||||
s, err := logstore.Active(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return s.Observability.BatchInsertNodeMetricSnapshots(ctx, rows)
|
||||
}
|
||||
|
||||
func flushNodeEdgeHealth(ctx context.Context, rows []analyticsmodel.NodeEdgeHealth) error {
|
||||
s, err := logstore.Active(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return s.Observability.BatchInsertNodeEdgeHealth(ctx, rows)
|
||||
}
|
||||
|
||||
func flushNodeObsFrps(ctx context.Context, rows []analyticsmodel.NodeObsFrps) error {
|
||||
s, err := logstore.Active(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return s.Observability.BatchInsertNodeObsFrps(ctx, rows)
|
||||
}
|
||||
|
||||
func flushNodeObsFrpc(ctx context.Context, rows []analyticsmodel.NodeObsFrpc) error {
|
||||
s, err := logstore.Active(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return s.Observability.BatchInsertNodeObsFrpc(ctx, rows)
|
||||
}
|
||||
|
||||
func flushNodeAccessLogs(ctx context.Context, rows []analyticsmodel.NodeAccessLog) error {
|
||||
s, err := logstore.Active(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return s.AccessLogs.BatchInsertNodeAccessLogs(ctx, rows)
|
||||
}
|
||||
|
||||
func metricSnapshotKey(snapshot analyticsmodel.NodeMetricSnapshot) string {
|
||||
return fmt.Sprintf("%s|%d", snapshot.NodeID, snapshot.CapturedAt.UTC().UnixNano())
|
||||
}
|
||||
|
||||
func edgeHealthKey(row analyticsmodel.NodeEdgeHealth) string {
|
||||
return fmt.Sprintf("%s|%d", row.NodeID, row.CapturedAt.UTC().UnixNano())
|
||||
}
|
||||
|
||||
func frpsKey(observation analyticsmodel.NodeObsFrps) string {
|
||||
return fmt.Sprintf("%s|%d", observation.NodeID, observation.CapturedAt.UTC().UnixNano())
|
||||
}
|
||||
|
||||
func frpcKey(observation analyticsmodel.NodeObsFrpc) string {
|
||||
return fmt.Sprintf("%s|%d", observation.NodeID, observation.CapturedAt.UTC().UnixNano())
|
||||
}
|
||||
|
||||
type batchStopper interface {
|
||||
Stop(ctx context.Context) error
|
||||
}
|
||||
|
||||
type statsProvider interface {
|
||||
Stats() batchwriter.Stats
|
||||
}
|
||||
|
||||
func running() bool {
|
||||
return metricSnapshotWriter != nil && metricSnapshotWriter.Running()
|
||||
}
|
||||
@@ -0,0 +1,9 @@
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
// Package observability defines shared error messages for observability operations.
|
||||
package observability
|
||||
|
||||
const (
|
||||
errInvalidStatusCode = "status_code 必须为 100-599 之间的整数"
|
||||
)
|
||||
@@ -0,0 +1,398 @@
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
package observability
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"time"
|
||||
|
||||
"Wavelet/openflare/plugins/server/domain/observability/chwriter"
|
||||
"Wavelet/openflare/plugins/server/kernel/model"
|
||||
analyticsmodel "Wavelet/openflare/plugins/server/kernel/model/analytics"
|
||||
"Wavelet/openflare/plugins/server/kernel/repository"
|
||||
"Wavelet/openflare/plugins/server/kernel/repository/logstore"
|
||||
"Wavelet/openflare/plugins/server/kernel/runtimeconfig"
|
||||
"Wavelet/openflare/plugins/server/kernel/task"
|
||||
"Wavelet/pkg/logger"
|
||||
)
|
||||
|
||||
const copyBatchSize = 1000
|
||||
|
||||
// 迁移目标库名常量(normalizeTarget 归一化后的取值)。
|
||||
const (
|
||||
targetPostgres = "postgres"
|
||||
targetSQLite = "sqlite"
|
||||
targetClickHouse = "clickhouse"
|
||||
)
|
||||
|
||||
type logDBSwitchPayload struct {
|
||||
Target string `json:"target"`
|
||||
}
|
||||
|
||||
// LogDBSwitchHandler 切换日志数据库任务处理器。
|
||||
type LogDBSwitchHandler struct{}
|
||||
|
||||
// ValidatePayload 校验并规范化参数。
|
||||
func (h *LogDBSwitchHandler) ValidatePayload(payload []byte) ([]byte, error) {
|
||||
var p logDBSwitchPayload
|
||||
if err := json.Unmarshal(payload, &p); err != nil {
|
||||
return nil, fmt.Errorf("参数解析失败: %w", err)
|
||||
}
|
||||
p.Target = normalizeTarget(p.Target)
|
||||
if !validTarget(p.Target) {
|
||||
return nil, fmt.Errorf("目标日志库不合法: %s", p.Target)
|
||||
}
|
||||
out, err := json.Marshal(p)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func normalizeTarget(v string) string {
|
||||
switch v {
|
||||
case targetPostgres, "postgresql":
|
||||
return targetPostgres
|
||||
case targetSQLite, "sqlite3":
|
||||
return targetSQLite
|
||||
case targetClickHouse, "ch":
|
||||
return targetClickHouse
|
||||
}
|
||||
return v
|
||||
}
|
||||
|
||||
func validTarget(v string) bool {
|
||||
return v == targetPostgres || v == targetSQLite || v == targetClickHouse
|
||||
}
|
||||
|
||||
// Execute 执行迁移。
|
||||
func (h *LogDBSwitchHandler) Execute(ctx context.Context, payload []byte) (*task.TaskResult, error) {
|
||||
var p logDBSwitchPayload
|
||||
if err := json.Unmarshal(payload, &p); err != nil {
|
||||
return nil, fmt.Errorf("参数解析失败: %w", err)
|
||||
}
|
||||
p.Target = normalizeTarget(p.Target)
|
||||
if err := validateSwitch(ctx, p.Target); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
source, err := currentLogDatabase(ctx)
|
||||
if err != nil {
|
||||
task.AppendLog(ctx, "读取日志主库失败: %v", err)
|
||||
return nil, err
|
||||
}
|
||||
task.AppendLog(ctx, "开始切换日志数据库:%s -> %s", source, p.Target)
|
||||
|
||||
// 设置迁移冻结标记(置位后由 ensureWritable 拒绝新写入)。
|
||||
if err := setMigrationFlag(ctx, "migrating"); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
// 失败也清除(SaveOrUpdateSystemConfig 会失效 RAM 缓存并广播),保持源库可写。
|
||||
defer func() {
|
||||
if err := setMigrationFlag(ctx, ""); err != nil {
|
||||
logger.ErrorF(ctx, "清除日志迁移冻结标记失败: %v", err)
|
||||
}
|
||||
}()
|
||||
|
||||
// 冻结标记置位后再排空在途批次(chwriter + 用户访问日志 writer),
|
||||
// 保证排空完成后不再有新批次进入源库。
|
||||
if err := drainLogWriters(ctx); err != nil {
|
||||
return nil, fmt.Errorf("排空日志写入队列失败: %w", err)
|
||||
}
|
||||
|
||||
src, err := logstore.Active(ctx)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
dst, err := buildTargetStore(ctx, p.Target)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
// 清空目标库日志表(幂等重试前提)。
|
||||
if err := clearTargetLogTables(ctx, dst); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
// PG 目标:按源库时间范围预建分区,避免历史数据复制报 "no partition of relation found"。
|
||||
if err := ensureTargetPartitions(ctx, src, dst, p.Target); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
// 逐表复制(6 张日志表)。
|
||||
if err := copyAccessLogs(ctx, src, dst); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := copyUserAccessLogs(ctx, src, dst); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := copyObservability(ctx, src, dst); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
// 翻转主库标记。
|
||||
if err := flipLogDatabase(ctx, p.Target); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
task.AppendLog(ctx, "日志数据库已切换为 %s,写入恢复", p.Target)
|
||||
return &task.TaskResult{Message: fmt.Sprintf("日志数据库已从 %s 切换为 %s", source, p.Target)}, nil
|
||||
}
|
||||
|
||||
func validateSwitch(ctx context.Context, target string) error {
|
||||
source, err := currentLogDatabase(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if source == target {
|
||||
return errors.New("目标日志库与当前日志库相同,无需迁移")
|
||||
}
|
||||
switch target {
|
||||
case "clickhouse":
|
||||
if !runtimeconfig.ClickHouseEnabled() {
|
||||
return errors.New("ClickHouse 未启用,无法迁移到 ClickHouse")
|
||||
}
|
||||
case "postgres":
|
||||
if !runtimeconfig.DatabaseEnabled() {
|
||||
return errors.New("PostgreSQL 未启用(当前主库为 SQLite),无法迁移到 PostgreSQL")
|
||||
}
|
||||
case "sqlite":
|
||||
if runtimeconfig.DatabaseEnabled() {
|
||||
return errors.New("当前主库为 PostgreSQL,日志库不能设置为 SQLite")
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func currentLogDatabase(ctx context.Context) (string, error) {
|
||||
cfg, err := repository.GetSystemConfigByKey(ctx, model.ConfigKeyLogDatabase)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("读取日志主库失败: %w", err)
|
||||
}
|
||||
if cfg.Value == "" {
|
||||
return "", errors.New("日志主库配置为空")
|
||||
}
|
||||
return cfg.Value, nil
|
||||
}
|
||||
|
||||
// drainLogWriters 等待 chwriter(节点访问日志 + 可观测 4 表)的在途批次全部落库。
|
||||
// 见设计 §7.2:先排空再冻结。用户访问日志(w_user_access_logs)记录已禁用,无在途批次。
|
||||
func drainLogWriters(ctx context.Context) error {
|
||||
return chwriter.Drain(ctx)
|
||||
}
|
||||
|
||||
// setMigrationFlag 写入迁移冻结标记。用 SaveOrUpdateSystemConfig:行缺失时 upsert,
|
||||
// 并失效 RAM 缓存 + 广播其他节点,保证 logstore.Migrating/resolveDatabase 立即生效。
|
||||
func setMigrationFlag(ctx context.Context, v string) error {
|
||||
return repository.SaveOrUpdateSystemConfig(ctx, model.ConfigKeyLogDBMigration, v)
|
||||
}
|
||||
|
||||
// flipLogDatabase 翻转日志主库。同上用 SaveOrUpdateSystemConfig,确保各进程缓存失效后指向新库。
|
||||
func flipLogDatabase(ctx context.Context, target string) error {
|
||||
return repository.SaveOrUpdateSystemConfig(ctx, model.ConfigKeyLogDatabase, target)
|
||||
}
|
||||
|
||||
// buildTargetStore 构造目标库 Store(不经过 Active 缓存,直接 Build)。
|
||||
// 迁移期间冻结标记已置位,目标库的清空/复制写入必须放行,故使用 BuildForMigration。
|
||||
func buildTargetStore(ctx context.Context, database string) (*logstore.Store, error) {
|
||||
return logstore.BuildForMigration(ctx, database)
|
||||
}
|
||||
|
||||
func clearTargetLogTables(ctx context.Context, dst *logstore.Store) error {
|
||||
// 依次清空 6 张表:AccessLogs.DeleteAll、UserAccessLogs.DeleteAll、Observability.DeleteAll*
|
||||
// (SQLite/PG 用 DeleteAll;CH 用 TRUNCATE 语义)。
|
||||
if _, err := dst.AccessLogs.DeleteAll(ctx); err != nil {
|
||||
return fmt.Errorf("清空目标访问日志失败: %w", err)
|
||||
}
|
||||
if _, err := dst.UserAccessLogs.DeleteAll(ctx); err != nil {
|
||||
return fmt.Errorf("清空目标用户访问日志失败: %w", err)
|
||||
}
|
||||
for _, fn := range []func(context.Context) (int64, error){
|
||||
dst.Observability.DeleteAllMetricSnapshots,
|
||||
dst.Observability.DeleteAllEdgeHealth,
|
||||
dst.Observability.DeleteAllNodeObservationFrps,
|
||||
dst.Observability.DeleteAllNodeObservationFrpc,
|
||||
} {
|
||||
if _, err := fn(ctx); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// ensureTargetPartitions 目标为 PG 时,按源库时间范围(两表合并)预建分区,
|
||||
// 否则复制历史数据会报 "no partition of relation found";目标非 PG 为 no-op。
|
||||
func ensureTargetPartitions(ctx context.Context, src, dst *logstore.Store, target string) error {
|
||||
if target != targetPostgres {
|
||||
return nil
|
||||
}
|
||||
from, to, err := migrationRange(ctx, src)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if from.IsZero() || to.IsZero() {
|
||||
task.AppendLog(ctx, "源库无日志数据,跳过分区预建")
|
||||
return nil
|
||||
}
|
||||
if err := dst.AccessLogs.EnsurePartitions(ctx, from, to.AddDate(0, 1, 0)); err != nil {
|
||||
return fmt.Errorf("预建目标 PG 分区失败: %w", err)
|
||||
}
|
||||
task.AppendLog(ctx, "已为目标 PG 预建分区 %s ~ %s", from.Format("2006-01"), to.Format("2006-01"))
|
||||
return nil
|
||||
}
|
||||
|
||||
// migrationRange 合并源库节点访问日志(logged_at)与用户访问日志(created_at)
|
||||
// 的最小/最大时间;任一表为空时忽略该表。
|
||||
func migrationRange(ctx context.Context, src *logstore.Store) (time.Time, time.Time, error) {
|
||||
fromAccess, toAccess, err := src.AccessLogs.MigrationRange(ctx)
|
||||
if err != nil {
|
||||
return time.Time{}, time.Time{}, fmt.Errorf("读取源访问日志时间范围失败: %w", err)
|
||||
}
|
||||
fromUser, toUser, err := src.UserAccessLogs.MigrationRange(ctx)
|
||||
if err != nil {
|
||||
return time.Time{}, time.Time{}, fmt.Errorf("读取源用户访问日志时间范围失败: %w", err)
|
||||
}
|
||||
return minTime(fromAccess, fromUser), maxTime(toAccess, toUser), nil
|
||||
}
|
||||
|
||||
func minTime(a, b time.Time) time.Time {
|
||||
switch {
|
||||
case a.IsZero():
|
||||
return b
|
||||
case b.IsZero():
|
||||
return a
|
||||
case a.Before(b):
|
||||
return a
|
||||
default:
|
||||
return b
|
||||
}
|
||||
}
|
||||
|
||||
func maxTime(a, b time.Time) time.Time {
|
||||
switch {
|
||||
case a.IsZero():
|
||||
return b
|
||||
case b.IsZero():
|
||||
return a
|
||||
case a.After(b):
|
||||
return a
|
||||
default:
|
||||
return b
|
||||
}
|
||||
}
|
||||
|
||||
// copyAccessLogs 从 src 复制节点访问日志到 dst。
|
||||
func copyAccessLogs(ctx context.Context, src, dst *logstore.Store) error {
|
||||
// 注意:迁移期间 src 已冻结,但复制读取不受冻结影响;每批按 id 升序扫描。
|
||||
var lastID uint64
|
||||
for {
|
||||
rows, err := listNodeAccessLogsByID(ctx, src, lastID, copyBatchSize)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if len(rows) == 0 {
|
||||
break
|
||||
}
|
||||
if err := dst.AccessLogs.BatchInsertNodeAccessLogs(ctx, rows); err != nil {
|
||||
return fmt.Errorf("写入目标访问日志失败(批 %d): %w", lastID, err)
|
||||
}
|
||||
task.AppendLog(ctx, "已复制访问日志 %d 条(截至 id=%d)", len(rows), rows[len(rows)-1].ID)
|
||||
lastID = rows[len(rows)-1].ID
|
||||
if len(rows) < copyBatchSize {
|
||||
break
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func listNodeAccessLogsByID(ctx context.Context, src *logstore.Store, afterID uint64, limit int) ([]analyticsmodel.NodeAccessLog, error) {
|
||||
return src.AccessLogs.ListForMigration(ctx, afterID, limit)
|
||||
}
|
||||
|
||||
// copyUserAccessLogs 从 src 复制用户访问日志到 dst(按 id 升序分批)。
|
||||
func copyUserAccessLogs(ctx context.Context, src, dst *logstore.Store) error {
|
||||
var lastID uint64
|
||||
for {
|
||||
rows, err := src.UserAccessLogs.ListForMigration(ctx, lastID, copyBatchSize)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if len(rows) == 0 {
|
||||
return nil
|
||||
}
|
||||
if err := dst.UserAccessLogs.BatchInsert(ctx, rows); err != nil {
|
||||
return fmt.Errorf("写入目标用户访问日志失败(批 %d): %w", lastID, err)
|
||||
}
|
||||
lastID = rows[len(rows)-1].ID
|
||||
task.AppendLog(ctx, "已复制用户访问日志 %d 条(截至 id=%d)", len(rows), lastID)
|
||||
if len(rows) < copyBatchSize {
|
||||
return nil
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// copyObservability 复制 4 张可观测表,每张表按 id 升序分批复制,
|
||||
// 以每批最后一条 id 作为下一批游标(不使用 len 近似)。
|
||||
func copyObservability(ctx context.Context, src, dst *logstore.Store) error {
|
||||
if err := copyObsTable(ctx, "metric_snapshots",
|
||||
src.Observability.ListMetricSnapshotsForMigration,
|
||||
dst.Observability.BatchInsertNodeMetricSnapshots,
|
||||
lastMetricSnapshotID); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := copyObsTable(ctx, "edge_health",
|
||||
src.Observability.ListEdgeHealthForMigration,
|
||||
dst.Observability.BatchInsertNodeEdgeHealth,
|
||||
lastEdgeHealthID); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := copyObsTable(ctx, "obs_frps",
|
||||
src.Observability.ListNodeObsFrpsForMigration,
|
||||
dst.Observability.BatchInsertNodeObsFrps,
|
||||
lastObsFrpsID); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := copyObsTable(ctx, "obs_frpc",
|
||||
src.Observability.ListNodeObsFrpcForMigration,
|
||||
dst.Observability.BatchInsertNodeObsFrpc,
|
||||
lastObsFrpcID); err != nil {
|
||||
return err
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// copyObsTable 按 id 升序分批复制单张可观测表;idOf 返回批内最后一条 id。
|
||||
func copyObsTable[T any](ctx context.Context, name string,
|
||||
list func(context.Context, uint64, int) ([]T, error),
|
||||
insert func(context.Context, []T) error,
|
||||
idOf func([]T) uint64,
|
||||
) error {
|
||||
var lastID uint64
|
||||
for {
|
||||
rows, err := list(ctx, lastID, copyBatchSize)
|
||||
if err != nil {
|
||||
return fmt.Errorf("复制 %s 失败: %w", name, err)
|
||||
}
|
||||
if len(rows) == 0 {
|
||||
return nil
|
||||
}
|
||||
if err := insert(ctx, rows); err != nil {
|
||||
return fmt.Errorf("复制 %s 失败: %w", name, err)
|
||||
}
|
||||
lastID = idOf(rows)
|
||||
task.AppendLog(ctx, "已复制 %s %d 条(截至 id=%d)", name, len(rows), lastID)
|
||||
if len(rows) < copyBatchSize {
|
||||
return nil
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func lastMetricSnapshotID(rows []analyticsmodel.NodeMetricSnapshot) uint64 {
|
||||
return rows[len(rows)-1].ID
|
||||
}
|
||||
func lastEdgeHealthID(rows []analyticsmodel.NodeEdgeHealth) uint64 { return rows[len(rows)-1].ID }
|
||||
func lastObsFrpsID(rows []analyticsmodel.NodeObsFrps) uint64 { return rows[len(rows)-1].ID }
|
||||
func lastObsFrpcID(rows []analyticsmodel.NodeObsFrpc) uint64 { return rows[len(rows)-1].ID }
|
||||
@@ -0,0 +1,381 @@
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
package observability
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/glebarez/sqlite"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
"gorm.io/gorm"
|
||||
"gorm.io/gorm/logger"
|
||||
|
||||
"Wavelet/openflare/plugins/server/kernel/model"
|
||||
analyticsmodel "Wavelet/openflare/plugins/server/kernel/model/analytics"
|
||||
"Wavelet/openflare/plugins/server/kernel/repository"
|
||||
"Wavelet/openflare/plugins/server/kernel/repository/logstore"
|
||||
"Wavelet/openflare/plugins/server/kernel/runtimeconfig"
|
||||
db "Wavelet/plugins/infra/database"
|
||||
)
|
||||
|
||||
var logDBSwitchDBSeq int64
|
||||
|
||||
// newLogDBSwitchDB 构造内存 sqlite 库(含日志 5 表 + 系统配置表)。
|
||||
func newLogDBSwitchDB(t *testing.T) *gorm.DB {
|
||||
t.Helper()
|
||||
dsn := fmt.Sprintf("file:log-db-switch-%d?mode=memory&cache=shared", atomic.AddInt64(&logDBSwitchDBSeq, 1))
|
||||
gdb, err := gorm.Open(sqlite.Open(dsn), &gorm.Config{Logger: logger.Default.LogMode(logger.Silent)})
|
||||
require.NoError(t, err)
|
||||
require.NoError(t, gdb.AutoMigrate(
|
||||
&model.SystemConfig{},
|
||||
&analyticsmodel.NodeAccessLog{},
|
||||
&analyticsmodel.NodeMetricSnapshot{},
|
||||
&analyticsmodel.NodeEdgeHealth{},
|
||||
&analyticsmodel.NodeObsFrps{},
|
||||
&analyticsmodel.NodeObsFrpc{},
|
||||
&analyticsmodel.UserAccessLog{},
|
||||
))
|
||||
return gdb
|
||||
}
|
||||
|
||||
// TestCopyAccessLogsPreservesIDs sqlite→sqlite 模拟:源 store 3 条,目标空库,
|
||||
// copyAccessLogs 后 ID 保留、数量一致。
|
||||
func TestCopyAccessLogsPreservesIDs(t *testing.T) {
|
||||
t.Cleanup(runtimeconfig.Override(false, false))
|
||||
logstore.ResetForTest()
|
||||
defer logstore.ResetForTest()
|
||||
|
||||
ctx := context.Background()
|
||||
srcDB := newLogDBSwitchDB(t)
|
||||
dstDB := newLogDBSwitchDB(t)
|
||||
|
||||
db.SetDB(srcDB)
|
||||
src, err := logstore.Active(ctx) // 无 reader 时按 seed 规则解析为 sqlite
|
||||
require.NoError(t, err)
|
||||
db.SetDB(dstDB)
|
||||
dst, err := logstore.BuildForMigration(ctx, "sqlite")
|
||||
require.NoError(t, err)
|
||||
t.Cleanup(func() { db.SetDB(nil) })
|
||||
|
||||
now := time.Now().UTC()
|
||||
rows := []analyticsmodel.NodeAccessLog{
|
||||
{ID: 101, NodeID: "n1", LoggedAt: now, RemoteAddr: "1.1.1.1", Host: "a.example.com", Path: "/"},
|
||||
{ID: 202, NodeID: "n2", LoggedAt: now, RemoteAddr: "2.2.2.2", Host: "b.example.com", Path: "/x"},
|
||||
{ID: 303, NodeID: "n1", LoggedAt: now, RemoteAddr: "3.3.3.3", Host: "c.example.com", Path: "/y"},
|
||||
}
|
||||
require.NoError(t, src.AccessLogs.BatchInsertNodeAccessLogs(ctx, rows))
|
||||
|
||||
require.NoError(t, copyAccessLogs(ctx, src, dst))
|
||||
|
||||
var got []analyticsmodel.NodeAccessLog
|
||||
require.NoError(t, dstDB.Order("id ASC").Find(&got).Error)
|
||||
require.Len(t, got, 3)
|
||||
for i, wantID := range []uint64{101, 202, 303} {
|
||||
assert.Equal(t, wantID, got[i].ID, "row %d id preserved", i)
|
||||
}
|
||||
assert.Equal(t, "n1", got[0].NodeID)
|
||||
assert.Equal(t, "n2", got[1].NodeID)
|
||||
assert.Equal(t, "n1", got[2].NodeID)
|
||||
assert.Equal(t, "1.1.1.1", got[0].RemoteAddr)
|
||||
|
||||
// 源库保持不变。
|
||||
var srcCount int64
|
||||
require.NoError(t, srcDB.Model(&analyticsmodel.NodeAccessLog{}).Count(&srcCount).Error)
|
||||
assert.Equal(t, int64(3), srcCount)
|
||||
}
|
||||
|
||||
// TestCopyUserAccessLogsPreservesIDs sqlite→sqlite 模拟:源库用户访问日志按 id 升序
|
||||
// 复制到目标库,ID 保留、数量一致,且源库保持不变。
|
||||
func TestCopyUserAccessLogsPreservesIDs(t *testing.T) {
|
||||
t.Cleanup(runtimeconfig.Override(false, false))
|
||||
logstore.ResetForTest()
|
||||
defer logstore.ResetForTest()
|
||||
|
||||
ctx := context.Background()
|
||||
srcDB := newLogDBSwitchDB(t)
|
||||
dstDB := newLogDBSwitchDB(t)
|
||||
|
||||
db.SetDB(srcDB)
|
||||
src, err := logstore.Active(ctx)
|
||||
require.NoError(t, err)
|
||||
db.SetDB(dstDB)
|
||||
dst, err := logstore.BuildForMigration(ctx, "sqlite")
|
||||
require.NoError(t, err)
|
||||
t.Cleanup(func() { db.SetDB(nil) })
|
||||
|
||||
now := time.Now().UTC()
|
||||
rows := []analyticsmodel.UserAccessLog{
|
||||
{ID: 11, UserID: 1, Path: "/a", CreatedAt: now},
|
||||
{ID: 22, UserID: 2, Path: "/b", CreatedAt: now.Add(time.Second)},
|
||||
{ID: 33, UserID: 1, Path: "/c", CreatedAt: now.Add(2 * time.Second)},
|
||||
}
|
||||
require.NoError(t, src.UserAccessLogs.BatchInsert(ctx, rows))
|
||||
|
||||
require.NoError(t, copyUserAccessLogs(ctx, src, dst))
|
||||
|
||||
var got []analyticsmodel.UserAccessLog
|
||||
require.NoError(t, dstDB.Order("id ASC").Find(&got).Error)
|
||||
require.Len(t, got, 3)
|
||||
for i, wantID := range []uint64{11, 22, 33} {
|
||||
assert.Equal(t, wantID, got[i].ID, "row %d id preserved", i)
|
||||
}
|
||||
|
||||
var srcCount int64
|
||||
require.NoError(t, srcDB.Model(&analyticsmodel.UserAccessLog{}).Count(&srcCount).Error)
|
||||
assert.Equal(t, int64(3), srcCount)
|
||||
}
|
||||
|
||||
// TestClearTargetLogTablesClearsUserAccessLogs 验证清空目标包含用户访问日志表
|
||||
// (6 张日志表之一),迁移「覆盖目标库已有日志」幂等前提成立。
|
||||
func TestClearTargetLogTablesClearsUserAccessLogs(t *testing.T) {
|
||||
t.Cleanup(runtimeconfig.Override(false, false))
|
||||
logstore.ResetForTest()
|
||||
defer logstore.ResetForTest()
|
||||
|
||||
ctx := context.Background()
|
||||
dstDB := newLogDBSwitchDB(t)
|
||||
db.SetDB(dstDB)
|
||||
t.Cleanup(func() { db.SetDB(nil) })
|
||||
dst, err := logstore.BuildForMigration(ctx, "sqlite")
|
||||
require.NoError(t, err)
|
||||
|
||||
now := time.Now().UTC()
|
||||
require.NoError(t, dst.UserAccessLogs.BatchInsert(ctx, []analyticsmodel.UserAccessLog{
|
||||
{ID: 1, UserID: 1, Path: "/a", CreatedAt: now},
|
||||
{ID: 2, UserID: 2, Path: "/b", CreatedAt: now},
|
||||
}))
|
||||
|
||||
require.NoError(t, clearTargetLogTables(ctx, dst))
|
||||
|
||||
var count int64
|
||||
require.NoError(t, dstDB.Model(&analyticsmodel.UserAccessLog{}).Count(&count).Error)
|
||||
assert.Zero(t, count, "用户访问日志应被清空")
|
||||
}
|
||||
|
||||
// TestClearTargetLogTablesDuringMigration 回归:冻结标记置位后,BuildForMigration 构造的
|
||||
// 目标 store 必须放行用户访问日志清空/写入。skipFreeze 未传播到 UserAccessLogs store 时
|
||||
// DeleteAll 会误报 ErrMigrating,导致真实切换任务在清空目标库阶段失败。
|
||||
func TestClearTargetLogTablesDuringMigration(t *testing.T) {
|
||||
logstore.ResetForTest()
|
||||
defer logstore.ResetForTest()
|
||||
|
||||
gdb := newLogDBSwitchDB(t)
|
||||
db.SetDB(gdb)
|
||||
t.Cleanup(func() { db.SetDB(nil) })
|
||||
ctx := context.Background()
|
||||
|
||||
logstore.SetConfigReader(func(ctx context.Context, key string) (string, error) {
|
||||
cfg, err := repository.GetSystemConfigByKey(ctx, key)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return cfg.Value, nil
|
||||
})
|
||||
|
||||
// 预置目标库已有日志(迁移「覆盖目标库已有日志」幂等前提)。
|
||||
now := time.Now().UTC()
|
||||
require.NoError(t, gdb.Create(&analyticsmodel.UserAccessLog{ID: 1, UserID: 1, Path: "/a", CreatedAt: now}).Error)
|
||||
|
||||
// 冻结标记置位(与真实任务 Execute 流程一致)。
|
||||
require.NoError(t, setMigrationFlag(ctx, "migrating"))
|
||||
t.Cleanup(func() { _ = setMigrationFlag(ctx, "") })
|
||||
require.True(t, logstore.Migrating(ctx))
|
||||
|
||||
dst, err := logstore.BuildForMigration(ctx, "sqlite")
|
||||
require.NoError(t, err)
|
||||
|
||||
require.NoError(t, clearTargetLogTables(ctx, dst), "迁移冻结期间目标库清空必须放行")
|
||||
|
||||
var count int64
|
||||
require.NoError(t, gdb.Model(&analyticsmodel.UserAccessLog{}).Count(&count).Error)
|
||||
assert.Zero(t, count, "用户访问日志应被清空")
|
||||
}
|
||||
|
||||
// TestValidateSwitch 各非法组合报错。
|
||||
func TestValidateSwitch(t *testing.T) {
|
||||
t.Cleanup(func() {
|
||||
})
|
||||
|
||||
gdb := newLogDBSwitchDB(t)
|
||||
db.SetDB(gdb)
|
||||
t.Cleanup(func() { db.SetDB(nil) })
|
||||
ctx := context.Background()
|
||||
setLogDB := func(v string) {
|
||||
require.NoError(t, repository.SaveOrUpdateSystemConfig(ctx, model.ConfigKeyLogDatabase, v))
|
||||
}
|
||||
|
||||
t.Run("same target rejected", func(t *testing.T) {
|
||||
setLogDB("sqlite")
|
||||
t.Cleanup(runtimeconfig.Override(false, false))
|
||||
err := validateSwitch(ctx, "sqlite")
|
||||
require.Error(t, err)
|
||||
assert.Contains(t, err.Error(), "相同")
|
||||
})
|
||||
t.Run("clickhouse disabled rejected", func(t *testing.T) {
|
||||
setLogDB("sqlite")
|
||||
t.Cleanup(runtimeconfig.Override(false, false))
|
||||
err := validateSwitch(ctx, "clickhouse")
|
||||
require.Error(t, err)
|
||||
assert.Contains(t, err.Error(), "ClickHouse 未启用")
|
||||
})
|
||||
t.Run("postgres requires main db enabled", func(t *testing.T) {
|
||||
setLogDB("sqlite")
|
||||
t.Cleanup(runtimeconfig.Override(false, false))
|
||||
err := validateSwitch(ctx, "postgres")
|
||||
require.Error(t, err)
|
||||
assert.Contains(t, err.Error(), "PostgreSQL 未启用")
|
||||
})
|
||||
t.Run("sqlite rejected when main db is postgres", func(t *testing.T) {
|
||||
setLogDB("postgres")
|
||||
t.Cleanup(runtimeconfig.Override(true, false))
|
||||
err := validateSwitch(ctx, "sqlite")
|
||||
require.Error(t, err)
|
||||
assert.Contains(t, err.Error(), "SQLite")
|
||||
})
|
||||
t.Run("valid postgres migration", func(t *testing.T) {
|
||||
setLogDB("sqlite")
|
||||
t.Cleanup(runtimeconfig.Override(true, false))
|
||||
require.NoError(t, validateSwitch(ctx, "postgres"))
|
||||
})
|
||||
}
|
||||
|
||||
// TestLogDBSwitchValidatePayload 参数归一化与非法值拒绝。
|
||||
func TestLogDBSwitchValidatePayload(t *testing.T) {
|
||||
h := &LogDBSwitchHandler{}
|
||||
cases := []struct {
|
||||
name string
|
||||
in string
|
||||
want string
|
||||
ok bool
|
||||
}{
|
||||
{name: "postgresql normalized", in: `{"target":"postgresql"}`, want: "postgres", ok: true},
|
||||
{name: "sqlite3 normalized", in: `{"target":"sqlite3"}`, want: "sqlite", ok: true},
|
||||
{name: "ch normalized", in: `{"target":"ch"}`, want: "clickhouse", ok: true},
|
||||
{name: "postgres passthrough", in: `{"target":"postgres"}`, want: "postgres", ok: true},
|
||||
{name: "invalid target", in: `{"target":"mysql"}`, ok: false},
|
||||
{name: "malformed json", in: `not-json`, ok: false},
|
||||
}
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
out, err := h.ValidatePayload([]byte(c.in))
|
||||
if !c.ok {
|
||||
require.Error(t, err)
|
||||
return
|
||||
}
|
||||
require.NoError(t, err)
|
||||
var p logDBSwitchPayload
|
||||
require.NoError(t, json.Unmarshal(out, &p))
|
||||
assert.Equal(t, c.want, p.Target)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestExecuteFailureClearsMigrationFlag 迁移失败后 log_db_migration 冻结标记被清除。
|
||||
// 在 FRESH DB(不预置 log_db_migration 行)上验证:setMigrationFlag 必须 upsert 建行,
|
||||
// 且失败后经缓存路径(GetSystemConfigByKey)可观察为空。
|
||||
func TestExecuteFailureClearsMigrationFlag(t *testing.T) {
|
||||
t.Cleanup(runtimeconfig.Override(true, runtimeconfig.ClickHouseEnabled()))
|
||||
|
||||
logstore.ResetForTest()
|
||||
defer logstore.ResetForTest()
|
||||
|
||||
gdb := newLogDBSwitchDB(t)
|
||||
db.SetDB(gdb)
|
||||
t.Cleanup(func() { db.SetDB(nil) })
|
||||
ctx := context.Background()
|
||||
|
||||
// FRESH DB:log_db_migration 行不存在(不预置),log_database 预置为 sqlite。
|
||||
require.NoError(t, repository.SaveOrUpdateSystemConfig(ctx, model.ConfigKeyLogDatabase, "sqlite"))
|
||||
_, err := repository.GetSystemConfigByKey(ctx, model.ConfigKeyLogDBMigration)
|
||||
require.ErrorIs(t, err, gorm.ErrRecordNotFound)
|
||||
|
||||
// configReader 对 log_database 报错,使 logstore.Active 在冻结标记置位后失败。
|
||||
logstore.SetConfigReader(func(_ context.Context, key string) (string, error) {
|
||||
if key == model.ConfigKeyLogDatabase {
|
||||
return "", errors.New("reader error")
|
||||
}
|
||||
return "", nil
|
||||
})
|
||||
|
||||
_, err = (&LogDBSwitchHandler{}).Execute(ctx, []byte(`{"target":"postgres"}`))
|
||||
require.Error(t, err)
|
||||
assert.Contains(t, err.Error(), "reader error")
|
||||
|
||||
// 冻结标记必须被 upsert 持久化(行存在)并经缓存路径可观察为空,源库恢复可写。
|
||||
cfg, err := repository.GetSystemConfigByKey(ctx, model.ConfigKeyLogDBMigration)
|
||||
require.NoError(t, err, "setMigrationFlag 应 upsert 创建 log_db_migration 行")
|
||||
assert.Empty(t, cfg.Value, "失败后冻结标记必须清除,源库保持可写")
|
||||
assert.False(t, logstore.Migrating(ctx))
|
||||
}
|
||||
|
||||
// TestSetMigrationFlagObservableThroughCache 在 FRESH DB 上验证 setMigrationFlag 写入
|
||||
// 经缓存路径(logstore.Migrating → repository 读取)实时反映:置位 true、清除 false。
|
||||
func TestSetMigrationFlagObservableThroughCache(t *testing.T) {
|
||||
logstore.ResetForTest()
|
||||
defer logstore.ResetForTest()
|
||||
|
||||
gdb := newLogDBSwitchDB(t)
|
||||
db.SetDB(gdb)
|
||||
t.Cleanup(func() { db.SetDB(nil) })
|
||||
ctx := context.Background()
|
||||
|
||||
// 按 bootstrap 同款注入 repository 读取,走 RAM 缓存路径。
|
||||
logstore.SetConfigReader(func(ctx context.Context, key string) (string, error) {
|
||||
cfg, err := repository.GetSystemConfigByKey(ctx, key)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return cfg.Value, nil
|
||||
})
|
||||
|
||||
// FRESH DB:行缺失 → fail-open false。
|
||||
assert.False(t, logstore.Migrating(ctx))
|
||||
|
||||
require.NoError(t, setMigrationFlag(ctx, "migrating"))
|
||||
assert.True(t, logstore.Migrating(ctx), "置位后缓存路径必须立即观察到 migrating")
|
||||
|
||||
require.NoError(t, setMigrationFlag(ctx, ""))
|
||||
assert.False(t, logstore.Migrating(ctx), "清除后缓存路径必须立即观察到非 migrating")
|
||||
}
|
||||
|
||||
// TestFlipLogDatabaseRefreshesCachedConfig 验证翻转日志主库后缓存路径立即反映新库
|
||||
// (logstore.ActiveDatabase / GetSystemConfigByKey),防止各进程继续写旧库(split-brain)。
|
||||
func TestFlipLogDatabaseRefreshesCachedConfig(t *testing.T) {
|
||||
logstore.ResetForTest()
|
||||
defer logstore.ResetForTest()
|
||||
|
||||
gdb := newLogDBSwitchDB(t)
|
||||
db.SetDB(gdb)
|
||||
t.Cleanup(func() { db.SetDB(nil) })
|
||||
ctx := context.Background()
|
||||
|
||||
logstore.SetConfigReader(func(ctx context.Context, key string) (string, error) {
|
||||
cfg, err := repository.GetSystemConfigByKey(ctx, key)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return cfg.Value, nil
|
||||
})
|
||||
|
||||
require.NoError(t, repository.SaveOrUpdateSystemConfig(ctx, model.ConfigKeyLogDatabase, "sqlite"))
|
||||
active, err := logstore.ActiveDatabase(ctx)
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, "sqlite", active) // 预热缓存
|
||||
|
||||
require.NoError(t, flipLogDatabase(ctx, "postgres"))
|
||||
|
||||
active, err = logstore.ActiveDatabase(ctx)
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, "postgres", active, "翻转后缓存路径必须立即反映新库")
|
||||
cfg, err := repository.GetSystemConfigByKey(ctx, model.ConfigKeyLogDatabase)
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, "postgres", cfg.Value)
|
||||
}
|
||||
@@ -0,0 +1,275 @@
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
package observability
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"Wavelet/openflare/plugins/server/kernel/repository"
|
||||
|
||||
"Wavelet/openflare/plugins/server/kernel/model"
|
||||
|
||||
"gorm.io/gorm"
|
||||
)
|
||||
|
||||
const (
|
||||
defaultObservabilityWindow = 24 * time.Hour
|
||||
defaultObservabilityLimit = 120
|
||||
maxObservabilityLimit = 500
|
||||
defaultTrafficDistributionLimit = 8
|
||||
nodeObservabilityCacheTTL = 15 * time.Second
|
||||
)
|
||||
|
||||
var nodeObservabilityCache struct {
|
||||
mu sync.Mutex
|
||||
views map[string]cachedNodeObservability
|
||||
}
|
||||
|
||||
type cachedNodeObservability struct {
|
||||
view *NodeView
|
||||
expiresAt time.Time
|
||||
}
|
||||
|
||||
// NodeQuery filters node observability data.
|
||||
type NodeQuery struct {
|
||||
Hours int `json:"hours"`
|
||||
Limit int `json:"limit"`
|
||||
}
|
||||
|
||||
// NodeAnalytics groups node observability analytics.
|
||||
type NodeAnalytics struct {
|
||||
Traffic *TrafficWindowSummary `json:"traffic"`
|
||||
Distributions TrafficDistributions `json:"distributions"`
|
||||
Health HealthSummary `json:"health"`
|
||||
}
|
||||
|
||||
// NodeTrends groups node observability trend series.
|
||||
type NodeTrends struct {
|
||||
Traffic24h []TrafficTrendPoint `json:"traffic_24h"`
|
||||
Capacity24h []CapacityTrendPoint `json:"capacity_24h"`
|
||||
Network24h []NetworkTrendPoint `json:"network_24h"`
|
||||
DiskIO24h []DiskIOTrendPoint `json:"disk_io_24h"`
|
||||
}
|
||||
|
||||
// RelayDashboardSnapshot summarizes tunnel relay status.
|
||||
type RelayDashboardSnapshot struct {
|
||||
TotalProxies int `json:"total_proxies"`
|
||||
OnlineProxies int `json:"online_proxies"`
|
||||
OfflineProxies int `json:"offline_proxies"`
|
||||
Proxies []RelayProxyStat `json:"proxies"`
|
||||
TotalConnections int `json:"total_connections"`
|
||||
ClientCounts int `json:"client_counts"`
|
||||
}
|
||||
|
||||
// RelayProxyStat is a single relay proxy entry.
|
||||
type RelayProxyStat struct {
|
||||
Name string `json:"name"`
|
||||
Type string `json:"type"`
|
||||
Status string `json:"status"`
|
||||
ClientVersion string `json:"client_version"`
|
||||
LastStartTime string `json:"last_start_time"`
|
||||
LastCloseTime string `json:"last_close_time"`
|
||||
ClientAddr string `json:"client_addr"`
|
||||
}
|
||||
|
||||
// NodeView is the node observability API response.
|
||||
type NodeView struct {
|
||||
NodeID string `json:"node_id"`
|
||||
Profile *model.OpenFlareNodeSystemProfile `json:"profile"`
|
||||
MetricSnapshots []*NodeMetricSnapshotView `json:"metric_snapshots"`
|
||||
HealthEvents []*model.OpenFlareHealthEvent `json:"health_events"`
|
||||
Analytics NodeAnalytics `json:"analytics"`
|
||||
Trends NodeTrends `json:"trends"`
|
||||
RelayDashboard *RelayDashboardSnapshot `json:"relay_dashboard,omitempty"`
|
||||
}
|
||||
|
||||
// HealthEventCleanupResult reports health event cleanup outcome.
|
||||
type HealthEventCleanupResult struct {
|
||||
NodeID string `json:"node_id"`
|
||||
DeletedCount int64 `json:"deleted_count"`
|
||||
}
|
||||
|
||||
// GetNodeObservability returns observability details for a node.
|
||||
func GetNodeObservability(ctx context.Context, id uint, query NodeQuery) (*NodeView, error) {
|
||||
now := time.Now()
|
||||
node, err := repository.GetOpenFlareNodeByID(ctx, id)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if view, ok := getCachedNodeObservability(node.NodeID); ok {
|
||||
return view, nil
|
||||
}
|
||||
|
||||
limit := normalizeObservabilityLimit(query.Limit)
|
||||
since := now.Add(-normalizeObservabilityWindow(query.Hours))
|
||||
|
||||
profile, err := repository.GetOpenFlareNodeSystemProfile(ctx, node.NodeID)
|
||||
if err != nil && !errors.Is(err, gorm.ErrRecordNotFound) {
|
||||
return nil, err
|
||||
}
|
||||
if errors.Is(err, gorm.ErrRecordNotFound) {
|
||||
profile = nil
|
||||
}
|
||||
|
||||
snapshots, err := repository.ListOpenFlareMetricSnapshotsSince(ctx, node.NodeID, since, limit)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
edgeHealth, err := repository.ListOpenFlareEdgeHealth(ctx, node.NodeID, since, limit)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
accessLogRegions, err := repository.ListOpenFlareAccessLogRegionCounts(ctx, node.NodeID, since, defaultTrafficDistributionLimit)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
events, err := repository.ListOpenFlareHealthEvents(ctx, node.NodeID, false, limit)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
distributions := BuildTrafficDistributionsFromAccessLogs(
|
||||
ctx, since, now, defaultTrafficDistributionLimit, accessLogRegions,
|
||||
)
|
||||
trafficSummary := buildTrafficWindowSummaryFromAccessLogs(ctx, node.NodeID, since, now)
|
||||
view := &NodeView{
|
||||
NodeID: node.NodeID,
|
||||
Profile: profile,
|
||||
MetricSnapshots: BuildMetricSnapshotViews(snapshots, edgeHealth),
|
||||
HealthEvents: events,
|
||||
Analytics: NodeAnalytics{
|
||||
Traffic: trafficSummary,
|
||||
Distributions: distributions,
|
||||
Health: buildHealthSummary(latestMetricSnapshot(snapshots), trafficSummary, events),
|
||||
},
|
||||
Trends: BuildNodeTrends(ctx, now, node.NodeID, snapshots),
|
||||
}
|
||||
if node.NodeType == "tunnel_relay" {
|
||||
frpsObs, frpsErr := repository.ListOpenFlareNodeObservationFrps(ctx, node.NodeID, time.Time{}, 1)
|
||||
if frpsErr != nil {
|
||||
return nil, frpsErr
|
||||
}
|
||||
var latestFrps *model.OpenFlareNodeObservationFrps
|
||||
if len(frpsObs) > 0 {
|
||||
latestFrps = frpsObs[0]
|
||||
}
|
||||
view.RelayDashboard = buildRelayDashboardSnapshot(node, latestFrps)
|
||||
}
|
||||
setCachedNodeObservability(node.NodeID, view)
|
||||
return view, nil
|
||||
}
|
||||
|
||||
func getCachedNodeObservability(nodeID string) (*NodeView, bool) {
|
||||
nodeObservabilityCache.mu.Lock()
|
||||
defer nodeObservabilityCache.mu.Unlock()
|
||||
if nodeObservabilityCache.views == nil {
|
||||
return nil, false
|
||||
}
|
||||
entry, ok := nodeObservabilityCache.views[nodeID]
|
||||
if !ok || time.Now().After(entry.expiresAt) {
|
||||
return nil, false
|
||||
}
|
||||
return entry.view, true
|
||||
}
|
||||
|
||||
func setCachedNodeObservability(nodeID string, view *NodeView) {
|
||||
nodeObservabilityCache.mu.Lock()
|
||||
defer nodeObservabilityCache.mu.Unlock()
|
||||
if nodeObservabilityCache.views == nil {
|
||||
nodeObservabilityCache.views = make(map[string]cachedNodeObservability)
|
||||
}
|
||||
nodeObservabilityCache.views[nodeID] = cachedNodeObservability{
|
||||
view: view,
|
||||
expiresAt: time.Now().Add(nodeObservabilityCacheTTL),
|
||||
}
|
||||
}
|
||||
|
||||
// CleanupHealthEvents removes all health events for a node.
|
||||
func CleanupHealthEvents(ctx context.Context, id uint) (*HealthEventCleanupResult, error) {
|
||||
node, err := repository.GetOpenFlareNodeByID(ctx, id)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
deletedCount, err := repository.DeleteOpenFlareHealthEventsByNodeID(ctx, node.NodeID)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return &HealthEventCleanupResult{
|
||||
NodeID: node.NodeID,
|
||||
DeletedCount: deletedCount,
|
||||
}, nil
|
||||
}
|
||||
|
||||
func buildRelayDashboardSnapshot(node *model.OpenFlareNode, obs *model.OpenFlareNodeObservationFrps) *RelayDashboardSnapshot {
|
||||
if node == nil {
|
||||
return nil
|
||||
}
|
||||
totalProxies := 0
|
||||
totalConnections := 0
|
||||
clientCounts := 0
|
||||
proxies := []RelayProxyStat{}
|
||||
|
||||
if obs != nil {
|
||||
totalProxies = obs.FrpsProxyCount
|
||||
totalConnections = obs.FrpsConnections
|
||||
clientCounts = obs.FrpsClientCount
|
||||
if obs.FrpsProxies != "" {
|
||||
var decoded []RelayProxyStat
|
||||
if err := json.Unmarshal([]byte(obs.FrpsProxies), &decoded); err == nil {
|
||||
proxies = decoded
|
||||
}
|
||||
}
|
||||
}
|
||||
if totalProxies < 0 {
|
||||
totalProxies = 0
|
||||
}
|
||||
onlineProxies := 0
|
||||
for _, proxy := range proxies {
|
||||
if proxy.Status == "online" {
|
||||
onlineProxies++
|
||||
}
|
||||
}
|
||||
if len(proxies) == 0 {
|
||||
onlineProxies = totalProxies
|
||||
if node.RelayStatus != "healthy" {
|
||||
onlineProxies = 0
|
||||
}
|
||||
}
|
||||
|
||||
return &RelayDashboardSnapshot{
|
||||
TotalProxies: totalProxies,
|
||||
OnlineProxies: onlineProxies,
|
||||
OfflineProxies: totalProxies - onlineProxies,
|
||||
Proxies: proxies,
|
||||
TotalConnections: maxInt(totalConnections, 0),
|
||||
ClientCounts: maxInt(clientCounts, 0),
|
||||
}
|
||||
}
|
||||
|
||||
func maxInt(a int, b int) int {
|
||||
if a > b {
|
||||
return a
|
||||
}
|
||||
return b
|
||||
}
|
||||
|
||||
func normalizeObservabilityLimit(limit int) int {
|
||||
if limit <= 0 {
|
||||
return defaultObservabilityLimit
|
||||
}
|
||||
if limit > maxObservabilityLimit {
|
||||
return maxObservabilityLimit
|
||||
}
|
||||
return limit
|
||||
}
|
||||
|
||||
func normalizeObservabilityWindow(hours int) time.Duration {
|
||||
if hours <= 0 {
|
||||
return defaultObservabilityWindow
|
||||
}
|
||||
return time.Duration(hours) * time.Hour
|
||||
}
|
||||
@@ -0,0 +1,130 @@
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
package observability
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"testing"
|
||||
|
||||
"github.com/gin-gonic/gin"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func TestReadQueryStringArrayAcceptsHostsBracketForm(t *testing.T) {
|
||||
gin.SetMode(gin.TestMode)
|
||||
|
||||
cases := []struct {
|
||||
name string
|
||||
url string
|
||||
want []string
|
||||
}{
|
||||
{
|
||||
name: "axios brackets form",
|
||||
url: "/overview?hours=168&hosts%5B%5D=gist.arctel.de",
|
||||
want: []string{"gist.arctel.de"},
|
||||
},
|
||||
{
|
||||
name: "repeated hosts keys",
|
||||
url: "/overview?hosts=a.example&hosts=b.example",
|
||||
want: []string{"a.example", "b.example"},
|
||||
},
|
||||
{
|
||||
name: "single hosts key",
|
||||
url: "/overview?hosts=gist.arctel.de",
|
||||
want: []string{"gist.arctel.de"},
|
||||
},
|
||||
{
|
||||
name: "empty",
|
||||
url: "/overview?hours=24",
|
||||
want: nil,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
w := httptest.NewRecorder()
|
||||
c, _ := gin.CreateTestContext(w)
|
||||
req, err := http.NewRequest(http.MethodGet, tc.url, nil)
|
||||
require.NoError(t, err)
|
||||
c.Request = req
|
||||
|
||||
got := readQueryStringArray(c, "hosts")
|
||||
require.Equal(t, tc.want, got)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestReadAccessLogQueryIncludesStatusCode(t *testing.T) {
|
||||
gin.SetMode(gin.TestMode)
|
||||
w := httptest.NewRecorder()
|
||||
c, _ := gin.CreateTestContext(w)
|
||||
req, err := http.NewRequest(
|
||||
http.MethodGet,
|
||||
"/?node_id=n1&remote_addr=1.2.3.4&host=a.example&path=/api&status_code=404&p=2&page_size=50",
|
||||
nil,
|
||||
)
|
||||
require.NoError(t, err)
|
||||
c.Request = req
|
||||
|
||||
got, err := readAccessLogQuery(c)
|
||||
require.NoError(t, err)
|
||||
require.Equal(t, "n1", got.NodeID)
|
||||
require.Equal(t, "1.2.3.4", got.RemoteAddr)
|
||||
require.Equal(t, "a.example", got.Host)
|
||||
require.Equal(t, "/api", got.Path)
|
||||
require.Equal(t, 404, got.StatusCode)
|
||||
require.Equal(t, 2, got.Page)
|
||||
require.Equal(t, 50, got.PageSize)
|
||||
}
|
||||
|
||||
func TestReadAccessLogQueryRejectsInvalidStatusCode(t *testing.T) {
|
||||
gin.SetMode(gin.TestMode)
|
||||
for _, raw := range []string{"abc", "99", "600", "-1"} {
|
||||
w := httptest.NewRecorder()
|
||||
c, _ := gin.CreateTestContext(w)
|
||||
req, err := http.NewRequest(http.MethodGet, "/?status_code="+raw, nil)
|
||||
require.NoError(t, err)
|
||||
c.Request = req
|
||||
|
||||
_, err = readAccessLogQuery(c)
|
||||
require.Error(t, err, "status_code=%s should be rejected", raw)
|
||||
}
|
||||
|
||||
w := httptest.NewRecorder()
|
||||
c, _ := gin.CreateTestContext(w)
|
||||
req, err := http.NewRequest(http.MethodGet, "/", nil)
|
||||
require.NoError(t, err)
|
||||
c.Request = req
|
||||
got, err := readAccessLogQuery(c)
|
||||
require.NoError(t, err)
|
||||
require.Equal(t, 0, got.StatusCode)
|
||||
}
|
||||
|
||||
func TestResolveAccessLogWindow(t *testing.T) {
|
||||
since, until, err := resolveAccessLogWindow(
|
||||
"2026-08-01T00:00:00Z",
|
||||
"2026-08-02T00:00:00Z",
|
||||
)
|
||||
require.NoError(t, err)
|
||||
require.True(t, until.After(since))
|
||||
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
since string
|
||||
until string
|
||||
}{
|
||||
{name: "missing both", since: "", until: ""},
|
||||
{name: "only since", since: "2026-08-01T00:00:00Z", until: ""},
|
||||
{name: "only until", since: "", until: "2026-08-02T00:00:00Z"},
|
||||
{name: "bad since", since: "not-a-time", until: "2026-08-02T00:00:00Z"},
|
||||
{name: "bad until", since: "2026-08-01T00:00:00Z", until: "not-a-time"},
|
||||
{name: "reversed", since: "2026-08-02T00:00:00Z", until: "2026-08-01T00:00:00Z"},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
_, _, err := resolveAccessLogWindow(tc.since, tc.until)
|
||||
require.Error(t, err)
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,323 @@
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
package observability
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"net/http"
|
||||
"strconv"
|
||||
|
||||
"Wavelet/openflare/plugins/server/kernel/apiutil"
|
||||
"Wavelet/pkg/response"
|
||||
|
||||
"github.com/gin-gonic/gin"
|
||||
)
|
||||
|
||||
// GetAccessLogOverviewHandler 获取访问日志概览。
|
||||
// @Summary 获取访问日志概览
|
||||
// @Description 返回访问日志汇总指标、趋势与 Top 排行,需要管理员权限
|
||||
// @Tags openflare-observability
|
||||
// @Produce json
|
||||
// @Security SessionCookie
|
||||
// @Param node_id query string false "节点 ID"
|
||||
// @Param host query string false "请求 Host(单域名)"
|
||||
// @Param hosts query []string false "请求 Host 列表(多域名精确匹配)"
|
||||
// @Param hours query int false "统计时间范围(小时)"
|
||||
// @Param bucket_minutes query int false "趋势桶分钟数(1、3、5 或 60,默认 60)"
|
||||
// @Success 200 {object} response.Any{data=observability.AccessLogOverview} "访问日志概览"
|
||||
// @Failure 400 {object} response.Any "参数错误"
|
||||
// @Failure 401 {object} response.Any "未登录"
|
||||
// @Failure 404 {object} response.Any "无权限或不存在"
|
||||
// @Failure 500 {object} response.Any "内部错误"
|
||||
// @Router /api/v1/d/access-logs/overview [get]
|
||||
func GetAccessLogOverviewHandler(c *gin.Context) {
|
||||
result, err := GetAccessLogOverview(c.Request.Context(), AccessLogOverviewQuery{
|
||||
NodeID: c.Query("node_id"),
|
||||
Host: c.Query("host"),
|
||||
Hosts: readQueryStringArray(c, "hosts"),
|
||||
Hours: readQueryInt(c, "hours"),
|
||||
BucketMinutes: readQueryInt(c, "bucket_minutes"),
|
||||
})
|
||||
if apiutil.AbortBadRequestOnError(c, err) {
|
||||
return
|
||||
}
|
||||
c.JSON(http.StatusOK, response.OK(result))
|
||||
}
|
||||
|
||||
// GetAccessLogsHandler 分页列出访问日志。
|
||||
// @Summary 列出访问日志
|
||||
// @Description 分页返回 OpenFlare 访问日志,支持按节点、IP、主机、路径与状态码筛选,需要管理员权限
|
||||
// @Tags openflare-observability
|
||||
// @Produce json
|
||||
// @Security SessionCookie
|
||||
// @Param node_id query string false "节点 ID"
|
||||
// @Param remote_addr query string false "客户端 IP"
|
||||
// @Param host query string false "请求 Host"
|
||||
// @Param path query string false "请求路径"
|
||||
// @Param status_code query int false "HTTP 状态码(100-599)"
|
||||
// @Param since query string false "起始时间(RFC3339,需与 until 成对提供)"
|
||||
// @Param until query string false "结束时间(RFC3339,需与 since 成对提供)"
|
||||
// @Param p query int false "页码"
|
||||
// @Param page_size query int false "每页条数"
|
||||
// @Param sort_by query string false "排序字段"
|
||||
// @Param sort_order query string false "排序方向"
|
||||
// @Success 200 {object} response.Any{data=observability.AccessLogList} "访问日志列表"
|
||||
// @Failure 400 {object} response.Any "参数错误"
|
||||
// @Failure 401 {object} response.Any "未登录"
|
||||
// @Failure 404 {object} response.Any "无权限或不存在"
|
||||
// @Failure 500 {object} response.Any "内部错误"
|
||||
// @Router /api/v1/d/access-logs [get]
|
||||
func GetAccessLogsHandler(c *gin.Context) {
|
||||
query, err := readAccessLogQuery(c)
|
||||
if apiutil.AbortBadRequestOnError(c, err) {
|
||||
return
|
||||
}
|
||||
logs, err := ListAccessLogs(c.Request.Context(), query)
|
||||
if apiutil.AbortBadRequestOnError(c, err) {
|
||||
return
|
||||
}
|
||||
c.JSON(http.StatusOK, response.OK(logs))
|
||||
}
|
||||
|
||||
// GetFoldedAccessLogsHandler 分页列出折叠访问日志。
|
||||
// @Summary 列出折叠访问日志
|
||||
// @Description 按时间桶聚合访问日志并分页返回,需要管理员权限
|
||||
// @Tags openflare-observability
|
||||
// @Produce json
|
||||
// @Security SessionCookie
|
||||
// @Param node_id query string false "节点 ID"
|
||||
// @Param remote_addr query string false "客户端 IP"
|
||||
// @Param host query string false "请求 Host"
|
||||
// @Param path query string false "请求路径"
|
||||
// @Param fold_minutes query int false "折叠时间窗口(分钟)"
|
||||
// @Param p query int false "页码"
|
||||
// @Param page_size query int false "每页条数"
|
||||
// @Param sort_by query string false "排序字段"
|
||||
// @Param sort_order query string false "排序方向"
|
||||
// @Success 200 {object} response.Any{data=observability.FoldedAccessLogList} "折叠访问日志列表"
|
||||
// @Failure 400 {object} response.Any "参数错误"
|
||||
// @Failure 401 {object} response.Any "未登录"
|
||||
// @Failure 404 {object} response.Any "无权限或不存在"
|
||||
// @Failure 500 {object} response.Any "内部错误"
|
||||
// @Router /api/v1/d/access-logs/folds [get]
|
||||
func GetFoldedAccessLogsHandler(c *gin.Context) {
|
||||
query, err := readAccessLogQuery(c)
|
||||
if apiutil.AbortBadRequestOnError(c, err) {
|
||||
return
|
||||
}
|
||||
query.FoldMinutes = readQueryInt(c, "fold_minutes")
|
||||
logs, err := ListFoldedAccessLogs(c.Request.Context(), query)
|
||||
if apiutil.AbortBadRequestOnError(c, err) {
|
||||
return
|
||||
}
|
||||
c.JSON(http.StatusOK, response.OK(logs))
|
||||
}
|
||||
|
||||
// GetFoldedAccessLogIPsHandler 列出折叠桶内的 IP 汇总。
|
||||
// @Summary 列出折叠访问日志 IP 汇总
|
||||
// @Description 在指定时间桶内按 IP 聚合访问统计,需要管理员权限
|
||||
// @Tags openflare-observability
|
||||
// @Produce json
|
||||
// @Security SessionCookie
|
||||
// @Param node_id query string false "节点 ID"
|
||||
// @Param remote_addr query string false "客户端 IP"
|
||||
// @Param host query string false "请求 Host"
|
||||
// @Param path query string false "请求路径"
|
||||
// @Param bucket_started_at query string false "时间桶起始时间"
|
||||
// @Param fold_minutes query int false "折叠时间窗口(分钟)"
|
||||
// @Param p query int false "页码"
|
||||
// @Param page_size query int false "每页条数"
|
||||
// @Param sort_by query string false "排序字段"
|
||||
// @Param sort_order query string false "排序方向"
|
||||
// @Success 200 {object} response.Any{data=observability.FoldedAccessLogIPList} "折叠 IP 汇总列表"
|
||||
// @Failure 400 {object} response.Any "参数错误"
|
||||
// @Failure 401 {object} response.Any "未登录"
|
||||
// @Failure 404 {object} response.Any "无权限或不存在"
|
||||
// @Failure 500 {object} response.Any "内部错误"
|
||||
// @Router /api/v1/d/access-logs/folds/ip-summary [get]
|
||||
func GetFoldedAccessLogIPsHandler(c *gin.Context) {
|
||||
result, err := ListFoldedAccessLogIPs(c.Request.Context(), FoldedAccessLogIPQuery{
|
||||
NodeID: c.Query("node_id"),
|
||||
RemoteAddr: c.Query("remote_addr"),
|
||||
Host: c.Query("host"),
|
||||
Path: c.Query("path"),
|
||||
BucketStartedAt: c.Query("bucket_started_at"),
|
||||
FoldMinutes: readQueryInt(c, "fold_minutes"),
|
||||
Page: readQueryInt(c, "p"),
|
||||
PageSize: readQueryInt(c, "page_size"),
|
||||
SortBy: c.Query("sort_by"),
|
||||
SortOrder: c.Query("sort_order"),
|
||||
})
|
||||
if apiutil.AbortBadRequestOnError(c, err) {
|
||||
return
|
||||
}
|
||||
c.JSON(http.StatusOK, response.OK(result))
|
||||
}
|
||||
|
||||
// GetAccessLogIPSummariesHandler 列出访问日志 IP 汇总。
|
||||
// @Summary 列出访问日志 IP 汇总
|
||||
// @Description 按 IP 聚合访问日志统计并分页返回;支持 hours 或 since/until 时间窗,需要管理员权限
|
||||
// @Tags openflare-observability
|
||||
// @Produce json
|
||||
// @Security SessionCookie
|
||||
// @Param node_id query string false "节点 ID"
|
||||
// @Param remote_addr query string false "客户端 IP"
|
||||
// @Param host query string false "请求 Host"
|
||||
// @Param hours query int false "统计时间范围(小时,1-720,默认 168)"
|
||||
// @Param since query string false "开始时间 RFC3339(与 until 同时提供时优先于 hours)"
|
||||
// @Param until query string false "结束时间 RFC3339"
|
||||
// @Param p query int false "页码"
|
||||
// @Param page_size query int false "每页条数"
|
||||
// @Param sort_by query string false "排序字段 total_requests|request_length|bytes_sent|success_ratio|last_seen_at|remote_addr"
|
||||
// @Param sort_order query string false "排序方向"
|
||||
// @Success 200 {object} response.Any{data=observability.AccessLogIPSummaryList} "IP 汇总列表"
|
||||
// @Failure 400 {object} response.Any "参数错误"
|
||||
// @Failure 401 {object} response.Any "未登录"
|
||||
// @Failure 404 {object} response.Any "无权限或不存在"
|
||||
// @Failure 500 {object} response.Any "内部错误"
|
||||
// @Router /api/v1/d/access-logs/ip-summary [get]
|
||||
func GetAccessLogIPSummariesHandler(c *gin.Context) {
|
||||
result, err := ListAccessLogIPSummaries(c.Request.Context(), AccessLogIPSummaryQuery{
|
||||
NodeID: c.Query("node_id"),
|
||||
RemoteAddr: c.Query("remote_addr"),
|
||||
Host: c.Query("host"),
|
||||
Hours: readQueryInt(c, "hours"),
|
||||
Since: c.Query("since"),
|
||||
Until: c.Query("until"),
|
||||
Page: readQueryInt(c, "p"),
|
||||
PageSize: readQueryInt(c, "page_size"),
|
||||
SortBy: c.Query("sort_by"),
|
||||
SortOrder: c.Query("sort_order"),
|
||||
})
|
||||
if apiutil.AbortBadRequestOnError(c, err) {
|
||||
return
|
||||
}
|
||||
c.JSON(http.StatusOK, response.OK(result))
|
||||
}
|
||||
|
||||
// GetAccessLogIPTrendHandler 获取 IP 访问趋势。
|
||||
// @Summary 获取访问日志 IP 趋势
|
||||
// @Description 返回指定 IP 在时间范围内的访问趋势数据,需要管理员权限
|
||||
// @Tags openflare-observability
|
||||
// @Produce json
|
||||
// @Security SessionCookie
|
||||
// @Param node_id query string false "节点 ID"
|
||||
// @Param remote_addr query string false "客户端 IP"
|
||||
// @Param host query string false "请求 Host"
|
||||
// @Param hours query int false "统计时间范围(小时)"
|
||||
// @Param bucket_minutes query int false "时间桶粒度(分钟)"
|
||||
// @Success 200 {object} response.Any{data=observability.AccessLogIPTrendView} "IP 访问趋势"
|
||||
// @Failure 400 {object} response.Any "参数错误"
|
||||
// @Failure 401 {object} response.Any "未登录"
|
||||
// @Failure 404 {object} response.Any "无权限或不存在"
|
||||
// @Failure 500 {object} response.Any "内部错误"
|
||||
// @Router /api/v1/d/access-logs/ip-summary/trend [get]
|
||||
func GetAccessLogIPTrendHandler(c *gin.Context) {
|
||||
result, err := GetAccessLogIPTrend(c.Request.Context(), AccessLogIPTrendQuery{
|
||||
NodeID: c.Query("node_id"),
|
||||
RemoteAddr: c.Query("remote_addr"),
|
||||
Host: c.Query("host"),
|
||||
Hours: readQueryInt(c, "hours"),
|
||||
BucketMinutes: readQueryInt(c, "bucket_minutes"),
|
||||
})
|
||||
if apiutil.AbortBadRequestOnError(c, err) {
|
||||
return
|
||||
}
|
||||
c.JSON(http.StatusOK, response.OK(result))
|
||||
}
|
||||
|
||||
// GetAccessLogIPAnalysisHandler 获取单 IP 访问分析。
|
||||
// @Summary 获取访问日志 IP 分析
|
||||
// @Description 返回指定 IP 的汇总指标与 Top 分布,需要管理员权限
|
||||
// @Tags openflare-observability
|
||||
// @Produce json
|
||||
// @Security SessionCookie
|
||||
// @Param node_id query string false "节点 ID"
|
||||
// @Param remote_addr query string false "客户端 IP"
|
||||
// @Param host query string false "请求 Host"
|
||||
// @Param hours query int false "统计时间范围(小时)"
|
||||
// @Success 200 {object} response.Any{data=observability.AccessLogIPAnalysisView} "IP 访问分析"
|
||||
// @Failure 400 {object} response.Any "参数错误"
|
||||
// @Failure 401 {object} response.Any "未登录"
|
||||
// @Failure 404 {object} response.Any "无权限或不存在"
|
||||
// @Failure 500 {object} response.Any "内部错误"
|
||||
// @Router /api/v1/d/access-logs/ip-summary/analysis [get]
|
||||
func GetAccessLogIPAnalysisHandler(c *gin.Context) {
|
||||
result, err := GetAccessLogIPAnalysis(c.Request.Context(), AccessLogIPAnalysisQuery{
|
||||
NodeID: c.Query("node_id"),
|
||||
RemoteAddr: c.Query("remote_addr"),
|
||||
Host: c.Query("host"),
|
||||
Hours: readQueryInt(c, "hours"),
|
||||
})
|
||||
if apiutil.AbortBadRequestOnError(c, err) {
|
||||
return
|
||||
}
|
||||
c.JSON(http.StatusOK, response.OK(result))
|
||||
}
|
||||
|
||||
// CleanupAccessLogsHandler 清理过期访问日志。
|
||||
// @Summary 清理访问日志
|
||||
// @Description 按保留天数清理过期访问日志记录,需要管理员权限
|
||||
// @Tags openflare-observability
|
||||
// @Accept json
|
||||
// @Produce json
|
||||
// @Security SessionCookie
|
||||
// @Param request body observability.AccessLogCleanupInput true "清理参数"
|
||||
// @Success 200 {object} response.Any{data=observability.AccessLogCleanupResult} "清理结果"
|
||||
// @Failure 400 {object} response.Any "参数错误"
|
||||
// @Failure 401 {object} response.Any "未登录"
|
||||
// @Failure 404 {object} response.Any "无权限或不存在"
|
||||
// @Failure 500 {object} response.Any "内部错误"
|
||||
// @Router /api/v1/d/access-logs/cleanup [post]
|
||||
func CleanupAccessLogsHandler(c *gin.Context) {
|
||||
var input AccessLogCleanupInput
|
||||
if !apiutil.BindJSON(c, &input) {
|
||||
return
|
||||
}
|
||||
result, err := CleanupAccessLogs(c.Request.Context(), input)
|
||||
if apiutil.AbortBadRequestOnError(c, err) {
|
||||
return
|
||||
}
|
||||
c.JSON(http.StatusOK, response.OK(result))
|
||||
}
|
||||
|
||||
func readAccessLogQuery(c *gin.Context) (AccessLogQuery, error) {
|
||||
query := AccessLogQuery{
|
||||
NodeID: c.Query("node_id"),
|
||||
RemoteAddr: c.Query("remote_addr"),
|
||||
Host: c.Query("host"),
|
||||
Path: c.Query("path"),
|
||||
Since: c.Query("since"),
|
||||
Until: c.Query("until"),
|
||||
Page: readQueryInt(c, "p"),
|
||||
PageSize: readQueryInt(c, "page_size"),
|
||||
SortBy: c.Query("sort_by"),
|
||||
SortOrder: c.Query("sort_order"),
|
||||
}
|
||||
if raw := c.Query("status_code"); raw != "" {
|
||||
code, err := strconv.Atoi(raw)
|
||||
if err != nil || code < 100 || code > 599 {
|
||||
return AccessLogQuery{}, errors.New(errInvalidStatusCode)
|
||||
}
|
||||
query.StatusCode = code
|
||||
}
|
||||
return query, nil
|
||||
}
|
||||
|
||||
func readQueryInt(c *gin.Context, key string) int {
|
||||
value, _ := strconv.Atoi(c.DefaultQuery(key, "0"))
|
||||
return value
|
||||
}
|
||||
|
||||
// readQueryStringArray reads repeated query values for key, and also accepts
|
||||
// the Axios/jQuery bracket form key[] / key%5B%5D which Gin does not map to key.
|
||||
func readQueryStringArray(c *gin.Context, key string) []string {
|
||||
if values := c.QueryArray(key); len(values) > 0 {
|
||||
return values
|
||||
}
|
||||
if values := c.QueryArray(key + "[]"); len(values) > 0 {
|
||||
return values
|
||||
}
|
||||
return nil
|
||||
}
|
||||
Reference in New Issue
Block a user