feat(obs): 访问日志 SSOT 与 edge_health,去掉协议兼容层

Agent 仅上报 host_metrics/edge_health/access_logs;业务流量与 UV 由
Server 侧访问日志聚合。新增 of_node_edge_health 与 of_access_log_hourly,
删除 request_reports/openresty 吞吐路径;API 不再暴露 traffic_reports
与 openresty_rx|tx。心跳/离线默认阈值与回填迁移一并入库。
This commit is contained in:
ryan
2026-07-18 11:53:11 +08:00
parent 9a0974cce8
commit f0e234df1f
60 changed files with 1799 additions and 2164 deletions
+1 -1
View File
@@ -24,7 +24,7 @@ const (
randomTokenBytes = 16
maxDatabaseTextLength = 16000
defaultAgentHeartbeatInterval = 10000 // 默认心跳间隔 10 秒(毫秒)
defaultAgentHeartbeatInterval = 3000 // 默认心跳间隔 3 秒(毫秒)
defaultAgentUpdateRepo = "Rain-kl/OpenFlare"
)
+52 -61
View File
@@ -6,7 +6,6 @@ package agent
import (
"context"
"encoding/json"
"errors"
"log/slog"
"strings"
"time"
@@ -28,16 +27,16 @@ const (
healthEventMessageMaxLength = 4096
)
// PersistHeartbeatObservability stores profile, snapshots, traffic, access logs, and health events.
// PersistHeartbeatObservability stores profile, host metrics, edge health, and access logs.
func PersistHeartbeatObservability(ctx context.Context, nodeID string, payload NodePayload, reportedAt time.Time) {
if strings.TrimSpace(nodeID) == "" {
return
}
if payload.Profile == nil &&
payload.Snapshot == nil &&
payload.TrafficReport == nil &&
payload.HostMetrics == nil &&
payload.EdgeHealth == nil &&
len(payload.AccessLogs) == 0 &&
len(payload.BufferedObservability) == 0 &&
len(payload.Buffered) == 0 &&
payload.HealthEvents == nil {
return
}
@@ -47,7 +46,7 @@ func PersistHeartbeatObservability(ctx context.Context, nodeID string, payload N
return
}
accessLogRecords, err := buildNodeAccessLogRecords(nodeID, payload.AccessLogs, payload.BufferedObservability, reportedAt)
accessLogRecords, err := buildNodeAccessLogRecords(nodeID, payload.AccessLogs, payload.Buffered, reportedAt)
if err != nil {
zap.L().Error("build heartbeat access logs failed", zap.String("node_id", nodeID), zap.Error(err))
return
@@ -68,17 +67,14 @@ func PersistHeartbeatObservability(ctx context.Context, nodeID string, payload N
return
}
if err := persistBufferedObservability(ctx, nodeID, payload.BufferedObservability, reportedAt); err != nil {
if err := persistBufferedObservability(ctx, nodeID, payload.Buffered, reportedAt); err != nil {
zap.L().Error("persist buffered observability failed", zap.String("node_id", nodeID), zap.Error(err))
}
if err := persistNodeMetricSnapshot(ctx, nodeID, payload.Snapshot, reportedAt); err != nil {
if err := persistNodeMetricSnapshot(ctx, nodeID, payload.HostMetrics, reportedAt); err != nil {
zap.L().Error("persist metric snapshot failed", zap.String("node_id", nodeID), zap.Error(err))
}
if err := persistNodeOpenrestyObservation(ctx, nodeID, payload.OpenrestyObservation, reportedAt); err != nil {
zap.L().Error("persist openresty observation failed", zap.String("node_id", nodeID), zap.Error(err))
}
if err := persistNodeTrafficReport(ctx, nodeID, payload.TrafficReport, reportedAt); err != nil {
zap.L().Error("persist traffic report failed", zap.String("node_id", nodeID), zap.Error(err))
if err := persistNodeEdgeHealth(ctx, nodeID, payload.EdgeHealth, payload.OpenrestyStatus, reportedAt); err != nil {
zap.L().Error("persist edge health failed", zap.String("node_id", nodeID), zap.Error(err))
}
if err := persistNodeAccessLogs(ctx, nodeID, accessLogRecords, reportedAt); err != nil {
@@ -88,19 +84,35 @@ func PersistHeartbeatObservability(ctx context.Context, nodeID string, payload N
func persistBufferedObservability(ctx context.Context, nodeID string, records []BufferedObservabilityRecord, reportedAt time.Time) error {
for _, record := range records {
if err := persistNodeMetricSnapshot(ctx, nodeID, record.Snapshot, reportedAt); err != nil {
if err := persistNodeMetricSnapshot(ctx, nodeID, record.HostMetrics, reportedAt); err != nil {
return err
}
if err := persistNodeOpenrestyObservation(ctx, nodeID, record.OpenrestyObservation, reportedAt); err != nil {
return err
}
if err := persistNodeTrafficReport(ctx, nodeID, record.TrafficReport, reportedAt); err != nil {
if err := persistNodeEdgeHealth(ctx, nodeID, record.EdgeHealth, "", reportedAt); err != nil {
return err
}
}
return nil
}
func persistNodeEdgeHealth(ctx context.Context, nodeID string, health *NodeEdgeHealth, fallbackStatus string, reportedAt time.Time) error {
if health == nil {
return nil
}
status := strings.TrimSpace(health.Status)
if status == "" {
status = strings.TrimSpace(fallbackStatus)
}
if status == "" {
status = openrestyStatusUnknown
}
return model.InsertOpenFlareEdgeHealth(ctx, &model.OpenFlareEdgeHealth{
NodeID: nodeID,
CapturedAt: timeFromUnix(health.CapturedAtUnix, reportedAt),
Status: status,
Connections: health.Connections,
})
}
func persistNodeSystemProfile(tx *gorm.DB, nodeID string, profile *NodeSystemProfile, reportedAt time.Time) error {
if profile == nil {
return nil
@@ -158,41 +170,6 @@ func persistNodeMetricSnapshot(ctx context.Context, nodeID string, snapshot *Nod
return model.InsertOpenFlareMetricSnapshot(ctx, record)
}
func persistNodeOpenrestyObservation(ctx context.Context, nodeID string, obs *NodeOpenrestyObservation, reportedAt time.Time) error {
if obs == nil {
return nil
}
record := &model.OpenFlareNodeObservationOpenresty{
NodeID: nodeID,
CapturedAt: timeFromUnix(obs.CapturedAtUnix, reportedAt),
OpenrestyRxBytes: obs.OpenrestyRxBytes,
OpenrestyTxBytes: obs.OpenrestyTxBytes,
OpenrestyConnections: obs.OpenrestyConnections,
}
return model.InsertOpenFlareNodeObservationOpenresty(ctx, record)
}
func persistNodeTrafficReport(ctx context.Context, nodeID string, report *NodeTrafficReport, reportedAt time.Time) error {
if report == nil {
return nil
}
if report.WindowEndedAtUnix > 0 && report.WindowStartedAtUnix > report.WindowEndedAtUnix {
return errors.New("traffic report window_started_at_unix 不能大于 window_ended_at_unix")
}
record := &model.OpenFlareRequestReport{
NodeID: nodeID,
WindowStartedAt: timeFromUnix(report.WindowStartedAtUnix, reportedAt),
WindowEndedAt: timeFromUnix(report.WindowEndedAtUnix, reportedAt),
RequestCount: report.RequestCount,
ErrorCount: report.ErrorCount,
UniqueVisitorCount: report.UniqueVisitorCount,
StatusCodesJSON: marshalJSON(report.StatusCodes),
TopDomainsJSON: marshalJSON(report.TopDomains),
SourceCountriesJSON: marshalJSON(report.SourceCountries),
}
return model.InsertOpenFlareRequestReport(ctx, record)
}
func buildNodeAccessLogRecords(nodeID string, direct []NodeAccessLog, buffered []BufferedObservabilityRecord, reportedAt time.Time) ([]*model.OpenFlareAccessLog, error) {
total := len(direct)
for _, record := range buffered {
@@ -213,15 +190,29 @@ func buildNodeAccessLogRecords(nodeID string, direct []NodeAccessLog, buffered [
records := make([]*model.OpenFlareAccessLog, 0, total)
appendLogs := func(logs []NodeAccessLog) {
for _, item := range logs {
bytesSent := item.BytesSent
if bytesSent < 0 {
bytesSent = 0
}
requestLength := item.RequestLength
if requestLength < 0 {
requestLength = 0
}
requestTimeMs := item.RequestTimeMs
if requestTimeMs < 0 {
requestTimeMs = 0
}
record := &model.OpenFlareAccessLog{
NodeID: nodeID,
LoggedAt: timeFromUnix(item.LoggedAtUnix, reportedAt),
RemoteAddr: strings.TrimSpace(item.RemoteAddr),
Region: "",
Host: strings.TrimSpace(item.Host),
Path: truncateForDatabase(strings.TrimSpace(item.Path), accessLogPathMaxLength),
StatusCode: item.StatusCode,
BytesSent: item.BytesSent,
NodeID: nodeID,
LoggedAt: timeFromUnix(item.LoggedAtUnix, reportedAt),
RemoteAddr: strings.TrimSpace(item.RemoteAddr),
Region: "",
Host: strings.TrimSpace(item.Host),
Path: truncateForDatabase(strings.TrimSpace(item.Path), accessLogPathMaxLength),
StatusCode: item.StatusCode,
BytesSent: bytesSent,
RequestLength: requestLength,
RequestTimeMs: requestTimeMs,
}
if resolver != nil {
record.Region = resolver.Resolve(record.RemoteAddr)
@@ -14,11 +14,8 @@ type NodeSystemProfile = pkgprotocol.NodeSystemProfile
// NodeMetricSnapshot holds a point-in-time resource-usage sample from an agent.
type NodeMetricSnapshot = pkgprotocol.NodeMetricSnapshot
// NodeOpenrestyObservation reports the OpenResty process health observed by an agent.
type NodeOpenrestyObservation = pkgprotocol.NodeOpenrestyObservation
// NodeTrafficReport aggregates traffic counters collected by an agent.
type NodeTrafficReport = pkgprotocol.NodeTrafficReport
// NodeEdgeHealth is L2 OpenResty health + connections.
type NodeEdgeHealth = pkgprotocol.NodeEdgeHealth
// NodeAccessLog is a single access-log record forwarded by an agent.
type NodeAccessLog = pkgprotocol.NodeAccessLog
+20 -47
View File
@@ -44,15 +44,13 @@ var (
initOnce sync.Once
metricSnapshotWriter *batchwriter.Writer[analyticsmodel.NodeMetricSnapshot]
requestReportWriter *batchwriter.Writer[analyticsmodel.NodeRequestReport]
openrestyWriter *batchwriter.Writer[analyticsmodel.NodeObsOpenresty]
edgeHealthWriter *batchwriter.Writer[analyticsmodel.NodeEdgeHealth]
frpsWriter *batchwriter.Writer[analyticsmodel.NodeObsFrps]
frpcWriter *batchwriter.Writer[analyticsmodel.NodeObsFrpc]
nodeAccessLogWriter *batchwriter.Writer[analyticsmodel.NodeAccessLog]
metricSnapshotDedup *dedupSet
requestReportDedup *dedupSet
openrestyDedup *dedupSet
edgeHealthDedup *dedupSet
frpsDedup *dedupSet
frpcDedup *dedupSet
)
@@ -65,8 +63,7 @@ func Init(ctx context.Context) {
initOnce.Do(func() {
metricSnapshotDedup = newDedupSet()
requestReportDedup = newDedupSet()
openrestyDedup = newDedupSet()
edgeHealthDedup = newDedupSet()
frpsDedup = newDedupSet()
frpcDedup = newDedupSet()
@@ -76,17 +73,11 @@ func Init(ctx context.Context) {
metricSnapshotDedup,
metricSnapshotKey,
)
requestReportWriter = mustNewObservabilityWriter(
"request_reports",
withFlushRetries(analyticsrepo.BatchInsertNodeRequestReports),
requestReportDedup,
requestReportKey,
)
openrestyWriter = mustNewObservabilityWriter(
"openresty_obs",
withFlushRetries(analyticsrepo.BatchInsertNodeObsOpenresty),
openrestyDedup,
openrestyKey,
edgeHealthWriter = mustNewObservabilityWriter(
"edge_health",
withFlushRetries(analyticsrepo.BatchInsertNodeEdgeHealth),
edgeHealthDedup,
edgeHealthKey,
)
frpsWriter = mustNewObservabilityWriter(
"frps_obs",
@@ -103,8 +94,7 @@ func Init(ctx context.Context) {
nodeAccessLogWriter = mustNewNodeAccessLogWriter()
metricSnapshotWriter.Start(ctx)
requestReportWriter.Start(ctx)
openrestyWriter.Start(ctx)
edgeHealthWriter.Start(ctx)
frpsWriter.Start(ctx)
frpcWriter.Start(ctx)
nodeAccessLogWriter.Start(ctx)
@@ -123,8 +113,7 @@ func Stop(ctx context.Context) error {
var firstErr error
for _, writer := range []batchStopper{
metricSnapshotWriter,
requestReportWriter,
openrestyWriter,
edgeHealthWriter,
frpsWriter,
frpcWriter,
nodeAccessLogWriter,
@@ -143,8 +132,7 @@ func Stop(ctx context.Context) error {
func WriterStats() []batchwriter.Stats {
writers := []statsProvider{
metricSnapshotWriter,
requestReportWriter,
openrestyWriter,
edgeHealthWriter,
frpsWriter,
frpcWriter,
nodeAccessLogWriter,
@@ -164,14 +152,9 @@ func QueueMetricSnapshot(snapshot analyticsmodel.NodeMetricSnapshot) {
queueWithDedup(metricSnapshotWriter, metricSnapshotDedup, metricSnapshotKey(snapshot), snapshot)
}
// QueueRequestReport enqueues a request report for asynchronous flush.
func QueueRequestReport(report analyticsmodel.NodeRequestReport) {
queueWithDedup(requestReportWriter, requestReportDedup, requestReportKey(report), report)
}
// QueueOpenrestyObservation enqueues an OpenResty observation for asynchronous flush.
func QueueOpenrestyObservation(observation analyticsmodel.NodeObsOpenresty) {
queueWithDedup(openrestyWriter, openrestyDedup, openrestyKey(observation), observation)
// QueueEdgeHealth enqueues an L2 edge health snapshot for asynchronous flush.
func QueueEdgeHealth(row analyticsmodel.NodeEdgeHealth) {
queueWithDedup(edgeHealthWriter, edgeHealthDedup, edgeHealthKey(row), row)
}
// QueueFrpsObservation enqueues an FRPS observation for asynchronous flush.
@@ -297,11 +280,10 @@ func withFlushRetries[T any](flush batchwriter.FlushFunc[T]) batchwriter.FlushFu
func wireModelInsertHooks() {
model.SetObservabilityInsertHooks(model.ObservabilityInsertHooks{
QueueMetricSnapshot: QueueMetricSnapshot,
QueueRequestReport: QueueRequestReport,
QueueOpenrestyObservation: QueueOpenrestyObservation,
QueueFrpsObservation: QueueFrpsObservation,
QueueFrpcObservation: QueueFrpcObservation,
QueueMetricSnapshot: QueueMetricSnapshot,
QueueEdgeHealth: QueueEdgeHealth,
QueueFrpsObservation: QueueFrpsObservation,
QueueFrpcObservation: QueueFrpcObservation,
})
model.SetAccessLogInsertHooks(model.AccessLogInsertHooks{
QueueNodeAccessLogs: QueueNodeAccessLogs,
@@ -312,17 +294,8 @@ func metricSnapshotKey(snapshot analyticsmodel.NodeMetricSnapshot) string {
return fmt.Sprintf("%s|%d", snapshot.NodeID, snapshot.CapturedAt.UTC().UnixNano())
}
func requestReportKey(report analyticsmodel.NodeRequestReport) string {
return fmt.Sprintf(
"%s|%d|%d",
report.NodeID,
report.WindowStartedAt.UTC().UnixNano(),
report.WindowEndedAt.UTC().UnixNano(),
)
}
func openrestyKey(observation analyticsmodel.NodeObsOpenresty) string {
return fmt.Sprintf("%s|%d", observation.NodeID, observation.CapturedAt.UTC().UnixNano())
func edgeHealthKey(row analyticsmodel.NodeEdgeHealth) string {
return fmt.Sprintf("%s|%d", row.NodeID, row.CapturedAt.UTC().UnixNano())
}
func frpsKey(observation analyticsmodel.NodeObsFrps) string {
+2 -2
View File
@@ -31,8 +31,8 @@ func computeNodeStatus(node *model.OpenFlareNode) string {
if node.LastSeenAt == nil || node.LastSeenAt.IsZero() {
return nodeStatusPending
}
// 使用默认阈值 2 分钟
threshold := 2 * time.Minute
// 默认离线阈值 60 秒(与 node_offline_threshold 默认一致)
threshold := 60 * time.Second
if time.Since(*node.LastSeenAt) > threshold {
return nodeStatusOffline
}
+48 -52
View File
@@ -122,19 +122,10 @@ func buildOverviewView(ctx context.Context) (*OverviewView, error) {
if err != nil {
return nil, err
}
latestTrafficRows, err := model.ListOpenFlareLatestRequestReportsSince(ctx, "", since)
if err != nil {
return nil, err
}
// Bounded raw windows remain for distributions and trend fallbacks; trends prefer hourly rollups.
snapshots, err := model.ListOpenFlareMetricSnapshotsSince(ctx, "", since, dashboardOverviewSnapshotLimit)
if err != nil {
return nil, err
}
reports, err := model.ListOpenFlareRequestReportsSince(ctx, "", since, dashboardOverviewSnapshotLimit)
if err != nil {
return nil, err
}
accessLogRegions, err := model.ListOpenFlareAccessLogRegionCounts(ctx, "", since, dashboardDistributionLimit)
if err != nil {
return nil, err
@@ -143,21 +134,48 @@ func buildOverviewView(ctx context.Context) (*OverviewView, error) {
if err != nil {
return nil, err
}
openrestySnapshots, err := model.ListOpenFlareNodeObservationOpenresty(ctx, "", since, dashboardOverviewSnapshotLimit)
if err != nil {
return nil, err
}
// L1 business: trends + distributions + totals from access logs only.
view := &OverviewView{
GeneratedAt: now,
Nodes: make([]NodeHealth, 0, len(nodes)),
Distributions: observability.BuildTrafficDistributions(reports, accessLogRegions, dashboardDistributionLimit),
Trends: observability.BuildNodeTrends(ctx, now, "", snapshots, openrestySnapshots, reports),
GeneratedAt: now,
Nodes: make([]NodeHealth, 0, len(nodes)),
Distributions: observability.BuildTrafficDistributionsFromAccessLogs(
ctx, since, now, dashboardDistributionLimit, accessLogRegions,
),
Trends: observability.BuildNodeTrends(ctx, now, "", snapshots),
}
// Global traffic summary uses true window uniqExact for UV (not sum of hourly uniques).
if summary, sumErr := model.TrafficSummaryOpenFlareAccessLogs(ctx, model.OpenFlareAccessLogQuery{
Since: since,
Until: now,
}); sumErr == nil {
view.Traffic.RequestCount = summary.RequestCount
view.Traffic.ErrorCount = summary.ErrorCount
view.Traffic.UniqueVisitors = summary.UniqueIPCount
view.Traffic.ReportedNodes = int(summary.NodeCount)
if summary.RequestCount > 0 {
// Average QPS over the 24h window.
view.Traffic.EstimatedQPS = float64(summary.RequestCount) / (24 * 3600)
}
} else {
// Fallback: sum hourly request/error buckets only (UV left from summary path).
applyTrafficTotalsFromTrend(&view.Traffic, view.Trends.Traffic24h)
}
nodeTraffic := map[string]model.OpenFlareAccessLogNodeAggregate{}
if aggregates, aggErr := model.NodeAggregatesOpenFlareAccessLogs(ctx, model.OpenFlareAccessLogQuery{
Since: since,
Until: now,
}); aggErr == nil {
for _, row := range aggregates {
nodeTraffic[row.NodeID] = row
}
}
var cpuNodeCount int
var memoryNodeCount int
latestSnapshots := observability.LatestMetricSnapshotsByNode(latestSnapshotRows)
latestTrafficReports := observability.LatestTrafficReportsByNode(latestTrafficRows)
activeEventsByNode := observability.ActiveHealthEventsByNode(activeEvents)
for _, node := range nodes {
@@ -175,7 +193,6 @@ func buildOverviewView(ctx context.Context) (*OverviewView, error) {
}
latestSnapshot := latestSnapshots[node.NodeID]
latestTraffic := latestTrafficReports[node.NodeID]
nodeActiveEvents := activeEventsByNode[node.NodeID]
nodeHealth := NodeHealth{
@@ -193,14 +210,15 @@ func buildOverviewView(ctx context.Context) (*OverviewView, error) {
}
cpuNodeCount, memoryNodeCount = applyNodeSnapshotMetrics(&nodeHealth, latestSnapshot, view, cpuNodeCount, memoryNodeCount)
applyNodeTrafficMetrics(&nodeHealth, latestTraffic)
if agg, ok := nodeTraffic[node.NodeID]; ok {
nodeHealth.RequestCount = agg.RequestCount
nodeHealth.ErrorCount = agg.ErrorCount
nodeHealth.UniqueVisitorCount = agg.UniqueIPCount
}
view.Nodes = append(view.Nodes, nodeHealth)
}
applyTrafficTotalsFromTrend(&view.Traffic, view.Trends.Traffic24h)
applyTrafficRuntimeMetrics(&view.Traffic, latestTrafficReports)
view.Summary.TotalNodes = len(nodes)
if cpuNodeCount > 0 {
view.Capacity.AverageCPUUsagePercent /= float64(cpuNodeCount)
@@ -246,43 +264,19 @@ func applyNodeSnapshotMetrics(nodeHealth *NodeHealth, snapshot *model.OpenFlareM
return cpuNodeCount, memoryNodeCount
}
func applyNodeTrafficMetrics(nodeHealth *NodeHealth, traffic *model.OpenFlareRequestReport) {
if traffic == nil {
return
}
nodeHealth.RequestCount = traffic.RequestCount
nodeHealth.ErrorCount = traffic.ErrorCount
nodeHealth.UniqueVisitorCount = traffic.UniqueVisitorCount
}
func applyTrafficTotalsFromTrend(traffic *Traffic, points []observability.TrafficTrendPoint) {
if traffic == nil {
return
}
traffic.RequestCount = 0
traffic.ErrorCount = 0
// Do not sum hourly unique visitors — that overcounts. UV must come from TrafficSummary.
for _, point := range points {
traffic.RequestCount += point.RequestCount
traffic.ErrorCount += point.ErrorCount
}
}
func applyTrafficRuntimeMetrics(traffic *Traffic, latestReports map[string]*model.OpenFlareRequestReport) {
if traffic == nil {
return
}
traffic.UniqueVisitors = 0
traffic.EstimatedQPS = 0
traffic.ReportedNodes = 0
for _, report := range latestReports {
if report == nil {
continue
}
traffic.UniqueVisitors += report.UniqueVisitorCount
traffic.ReportedNodes++
if duration := report.WindowEndedAt.Sub(report.WindowStartedAt).Seconds(); duration > 0 {
traffic.EstimatedQPS += float64(report.RequestCount) / duration
}
if traffic.RequestCount > 0 && traffic.ReportedNodes == 0 {
traffic.ReportedNodes = 1
}
}
@@ -360,12 +354,14 @@ func compressCapacityTrendPoints(points []observability.CapacityTrendPoint) [][]
func compressNetworkTrendPoints(points []observability.NetworkTrendPoint) [][]any {
rows := make([][]any, 0, len(points))
for _, point := range points {
// Compact layout (stable positions):
// [0] bucket, [1] host_rx, [2] host_tx, [3] bytes_received, [4] bytes_provided, [5] reported_nodes
rows = append(rows, []any{
point.BucketStartedAt,
point.NetworkRxBytes,
point.NetworkTxBytes,
point.OpenrestyRxBytes,
point.OpenrestyTxBytes,
point.BytesReceived,
point.BytesProvided,
point.ReportedNodes,
})
}
@@ -39,7 +39,7 @@ func TestGetOverviewStructure(t *testing.T) {
ctx := context.Background()
now := time.Now().UTC()
lastSeen := now.Add(-time.Minute)
lastSeen := now.Add(-15 * time.Second) // within default 60s offline threshold
require.NoError(t, db.DB(ctx).Create(&model.OpenFlareNode{
NodeID: "node-dashboard-1",
@@ -75,14 +75,33 @@ func TestGetOverviewStructure(t *testing.T) {
StorageUsedBytes: 2,
StorageTotalBytes: 10,
}))
require.NoError(t, model.InsertOpenFlareRequestReport(ctx, &model.OpenFlareRequestReport{
NodeID: "node-dashboard-1",
WindowStartedAt: now.Add(-2 * time.Minute),
WindowEndedAt: now.Add(-time.Minute),
RequestCount: 12,
ErrorCount: 1,
UniqueVisitorCount: 4,
}))
// Business traffic from access logs (L1 authority): 12 requests, 1 server error, 4 unique IPs.
logs := make([]*model.OpenFlareAccessLog, 0, 12)
for i := 0; i < 11; i++ {
logs = append(logs, &model.OpenFlareAccessLog{
NodeID: "node-dashboard-1",
LoggedAt: now.Add(-time.Minute),
RemoteAddr: "10.0.0." + string(rune('1'+i%4)), // rough; fixed below
Host: "app.example.com",
Path: "/",
StatusCode: 200,
BytesSent: 100,
})
}
ips := []string{"10.0.0.10", "10.0.0.11", "10.0.0.12", "10.0.0.13"}
for i := 0; i < 11; i++ {
logs[i].RemoteAddr = ips[i%4]
}
logs = append(logs, &model.OpenFlareAccessLog{
NodeID: "node-dashboard-1",
LoggedAt: now.Add(-time.Minute),
RemoteAddr: ips[0],
Host: "app.example.com",
Path: "/err",
StatusCode: 502,
BytesSent: 10,
})
require.NoError(t, model.InsertOpenFlareAccessLogsBatch(ctx, logs))
overview, err := GetOverview(ctx)
require.NoError(t, err)
@@ -98,8 +117,12 @@ func TestGetOverviewStructure(t *testing.T) {
assert.Equal(t, int64(12), overview.Traffic.RequestCount)
assert.Equal(t, int64(4), overview.Traffic.UniqueVisitors)
assert.Equal(t, int64(1), overview.Traffic.ErrorCount)
assert.InDelta(t, 0.2, overview.Traffic.EstimatedQPS, 0.0001)
// QPS over 24h window
assert.InDelta(t, 12.0/(24*3600.0), overview.Traffic.EstimatedQPS, 0.0001)
assert.Equal(t, 1, overview.Traffic.ReportedNodes)
// Node-level traffic from access log aggregates
onlineNodeCheck := overview.Nodes
require.NotEmpty(t, onlineNodeCheck)
assert.Equal(t, 55.0, overview.Capacity.AverageCPUUsagePercent)
assert.Equal(t, 50.0, overview.Capacity.AverageMemoryUsagePercent)
@@ -110,8 +133,9 @@ func TestGetOverviewStructure(t *testing.T) {
require.NotNil(t, overview.Distributions.StatusCodes)
require.NotNil(t, overview.Distributions.TopDomains)
require.NotNil(t, overview.Distributions.SourceCountries)
assert.Empty(t, overview.Distributions.StatusCodes)
assert.Empty(t, overview.Distributions.TopDomains)
// Status/top domains come from access logs.
assert.NotEmpty(t, overview.Distributions.StatusCodes)
assert.NotEmpty(t, overview.Distributions.TopDomains)
assert.Empty(t, overview.Distributions.SourceCountries)
require.Len(t, overview.Trends.Traffic24h, 24)
@@ -147,11 +171,11 @@ func TestGetOverviewStructure(t *testing.T) {
assert.Equal(t, "online", onlineNode[6])
assert.Equal(t, "healthy", onlineNode[7])
// Latest-per-node health fields (indexes match compressDashboardNodes).
assert.Equal(t, 55.0, onlineNode[11]) // cpu_usage_percent from latest snapshot
assert.Equal(t, 50.0, onlineNode[12]) // memory_usage_percent
assert.Equal(t, int64(12), onlineNode[14])
assert.Equal(t, int64(1), onlineNode[15])
assert.Equal(t, int64(4), onlineNode[16])
assert.Equal(t, 55.0, onlineNode[11]) // cpu_usage_percent from latest snapshot
assert.Equal(t, 50.0, onlineNode[12]) // memory_usage_percent
assert.Equal(t, int64(12), onlineNode[14]) // request_count from access logs
assert.Equal(t, int64(1), onlineNode[15]) // error_count
assert.Equal(t, int64(4), onlineNode[16]) // unique visitors
pendingNode := nodeByID["node-dashboard-2"]
require.NotNil(t, pendingNode)
@@ -161,5 +185,4 @@ func TestGetOverviewStructure(t *testing.T) {
assert.Equal(t, 55.0, overview.Capacity.AverageCPUUsagePercent)
assert.Equal(t, 1, overview.Traffic.ReportedNodes)
assert.Equal(t, int64(4), overview.Traffic.UniqueVisitors)
}
+2 -2
View File
@@ -194,9 +194,9 @@ func computeNodeStatus(node *model.OpenFlareNode) string {
if node.LastSeenAt == nil || node.LastSeenAt.IsZero() {
return nodeStatusPending
}
// 使用默认阈值 2 分钟,避免在这里读取配置
// 默认离线阈值 60 秒(与 node_offline_threshold 默认一致),避免在这里读取配置
// 实际阈值会在需要精确判断的地方通过 getNodeOfflineThreshold 读取
threshold := 2 * time.Minute
threshold := 60 * time.Second
if time.Since(*node.LastSeenAt) > threshold {
return nodeStatusOffline
}
+2 -2
View File
@@ -322,8 +322,8 @@ func TestComputeNodeStatus(t *testing.T) {
online := &model.OpenFlareNode{LastSeenAt: &now}
assert.Equal(t, nodeStatusOnline, computeNodeStatus(online))
// computeNodeStatus 使用默认阈值 2 分钟
offlineAt := now.Add(-2*time.Minute - time.Minute)
// computeNodeStatus 使用默认阈值 60 秒
offlineAt := now.Add(-61 * time.Second)
offline := &model.OpenFlareNode{LastSeenAt: &offlineAt}
assert.Equal(t, nodeStatusOffline, computeNodeStatus(offline))
}
@@ -16,6 +16,7 @@ const (
maxAccessLogPageSize = 200
defaultAccessLogSortBy = "logged_at"
defaultAccessLogSortOrder = "desc"
accessLogSortOrderAsc = "asc"
defaultAccessLogFoldMinute = 3
defaultIPTrendHours = 24
defaultIPTrendBucketMinute = 30
@@ -605,8 +606,8 @@ func normalizeAccessLogSortBy(sortBy string) string {
}
func normalizeAccessLogSortOrder(sortOrder string) string {
if strings.EqualFold(strings.TrimSpace(sortOrder), "asc") {
return "asc"
if strings.EqualFold(strings.TrimSpace(sortOrder), accessLogSortOrderAsc) {
return accessLogSortOrderAsc
}
return defaultAccessLogSortOrder
}
+215 -197
View File
@@ -5,7 +5,6 @@ package observability
import (
"context"
"encoding/json"
"sort"
"strings"
"time"
@@ -22,6 +21,7 @@ const (
healthSeverityCritical = "critical"
healthSeverityWarning = "warning"
percentageMultiplier = 100
sortOrderAsc = "asc"
)
// DistributionItem is a key/value distribution entry.
@@ -37,9 +37,9 @@ type TrafficDistributions struct {
SourceCountries []DistributionItem `json:"source_countries"`
}
const metricSnapshotOpenrestyMatchWindow = 2 * time.Minute
const metricSnapshotEdgeHealthMatchWindow = 2 * time.Minute
// NodeMetricSnapshotView is a metric snapshot enriched with OpenResty observations.
// NodeMetricSnapshotView is a metric snapshot enriched with edge health connections.
type NodeMetricSnapshotView struct {
ID uint `json:"id,omitempty"`
NodeID string `json:"node_id,omitempty"`
@@ -53,8 +53,6 @@ type NodeMetricSnapshotView struct {
DiskWriteBytes int64 `json:"disk_write_bytes"`
NetworkRxBytes int64 `json:"network_rx_bytes"`
NetworkTxBytes int64 `json:"network_tx_bytes"`
OpenrestyRxBytes int64 `json:"openresty_rx_bytes"`
OpenrestyTxBytes int64 `json:"openresty_tx_bytes"`
OpenrestyConnections int64 `json:"openresty_connections"`
}
@@ -98,13 +96,15 @@ type CapacityTrendPoint struct {
}
// NetworkTrendPoint is a network trend bucket.
// Host network_* is L3 (宿主机网卡).
// bytes_received/provided are L1 business bytes from access logs.
type NetworkTrendPoint struct {
BucketStartedAt time.Time `json:"bucket_started_at"`
NetworkRxBytes int64 `json:"network_rx_bytes"`
NetworkTxBytes int64 `json:"network_tx_bytes"`
OpenrestyRxBytes int64 `json:"openresty_rx_bytes"`
OpenrestyTxBytes int64 `json:"openresty_tx_bytes"`
ReportedNodes int `json:"reported_nodes"`
BucketStartedAt time.Time `json:"bucket_started_at"`
NetworkRxBytes int64 `json:"network_rx_bytes"`
NetworkTxBytes int64 `json:"network_tx_bytes"`
BytesReceived int64 `json:"bytes_received"` // sum(request_length)
BytesProvided int64 `json:"bytes_provided"` // sum(bytes_sent)
ReportedNodes int `json:"reported_nodes"`
}
// DiskIOTrendPoint is a disk IO trend bucket.
@@ -141,30 +141,39 @@ type networkCounterState struct {
seen bool
}
func buildTrafficWindowSummary(report *model.OpenFlareRequestReport) *TrafficWindowSummary {
if report == nil {
func buildTrafficWindowSummaryFromAccessLogs(
ctx context.Context,
nodeID string,
since, until time.Time,
) *TrafficWindowSummary {
row, err := model.TrafficSummaryOpenFlareAccessLogs(ctx, model.OpenFlareAccessLogQuery{
NodeID: nodeID,
Since: since,
Until: until,
})
if err != nil || row.RequestCount <= 0 {
return nil
}
summary := TrafficWindowSummary{
WindowStartedAt: report.WindowStartedAt,
WindowEndedAt: report.WindowEndedAt,
RequestCount: report.RequestCount,
UniqueVisitorCount: report.UniqueVisitorCount,
ErrorCount: report.ErrorCount,
summary := &TrafficWindowSummary{
WindowStartedAt: since.UTC(),
WindowEndedAt: until.UTC(),
RequestCount: row.RequestCount,
UniqueVisitorCount: row.UniqueIPCount,
ErrorCount: row.ErrorCount,
}
if duration := report.WindowEndedAt.Sub(report.WindowStartedAt).Seconds(); duration > 0 {
summary.EstimatedQPS = float64(report.RequestCount) / duration
if duration := until.Sub(since).Seconds(); duration > 0 {
summary.EstimatedQPS = float64(row.RequestCount) / duration
}
if report.RequestCount > 0 {
summary.ErrorRatePercent = (float64(report.ErrorCount) / float64(report.RequestCount)) * 100
if row.RequestCount > 0 {
summary.ErrorRatePercent = (float64(row.ErrorCount) / float64(row.RequestCount)) * 100
}
return &summary
return summary
}
// BuildMetricSnapshotViews merges metric snapshots with OpenResty observations for API responses.
// BuildMetricSnapshotViews merges metric snapshots with edge health connections for API responses.
func BuildMetricSnapshotViews(
snapshots []*model.OpenFlareMetricSnapshot,
openrestyObs []*model.OpenFlareNodeObservationOpenresty,
edgeHealth []*model.OpenFlareEdgeHealth,
) []*NodeMetricSnapshotView {
if len(snapshots) == 0 {
return []*NodeMetricSnapshotView{}
@@ -188,40 +197,49 @@ func BuildMetricSnapshotViews(
NetworkRxBytes: snapshot.NetworkRxBytes,
NetworkTxBytes: snapshot.NetworkTxBytes,
}
if matched := matchOpenrestyObservation(snapshot.CapturedAt, openrestyObs); matched != nil {
view.OpenrestyRxBytes = matched.OpenrestyRxBytes
view.OpenrestyTxBytes = matched.OpenrestyTxBytes
view.OpenrestyConnections = matched.OpenrestyConnections
if matched := matchEdgeHealth(snapshot.CapturedAt, edgeHealth); matched != nil {
view.OpenrestyConnections = matched.Connections
}
views = append(views, view)
}
return views
}
// BuildTrafficDistributions aggregates traffic distribution charts.
func BuildTrafficDistributions(
reports []*model.OpenFlareRequestReport,
accessLogRegions []*model.OpenFlareAccessLogRegionCount,
// BuildTrafficDistributionsFromAccessLogs builds distributions from access logs (L1).
func BuildTrafficDistributionsFromAccessLogs(
ctx context.Context,
since, until time.Time,
limit int,
accessLogRegions []*model.OpenFlareAccessLogRegionCount,
) TrafficDistributions {
statusCodes := make(distributionAccumulator)
topDomains := make(distributionAccumulator)
reportSourceCountries := make(distributionAccumulator)
for _, report := range reports {
mergeJSONCounts(statusCodes, report.StatusCodesJSON)
mergeJSONCounts(topDomains, report.TopDomainsJSON)
mergeJSONCounts(reportSourceCountries, report.SourceCountriesJSON)
}
sourceCountries := reportSourceCountries
if len(accessLogRegions) > 0 {
sourceCountries = make(distributionAccumulator, len(accessLogRegions))
for _, item := range accessLogRegions {
if item == nil || strings.TrimSpace(item.Region) == "" || item.Count <= 0 {
query := model.OpenFlareAccessLogQuery{Since: since, Until: until}
if statusRows, err := model.ValueCountsOpenFlareAccessLogs(ctx, query, "status_code", limit); err == nil {
for _, row := range statusRows {
if strings.TrimSpace(row.Value) == "" || row.Count <= 0 {
continue
}
sourceCountries[item.Region] = item.Count
statusCodes[row.Value] = row.Count
}
}
if hostRows, err := model.ValueCountsOpenFlareAccessLogs(ctx, query, "host", limit); err == nil {
for _, row := range hostRows {
if strings.TrimSpace(row.Value) == "" || row.Count <= 0 {
continue
}
topDomains[row.Value] = row.Count
}
}
sourceCountries := make(distributionAccumulator)
for _, item := range accessLogRegions {
if item == nil || strings.TrimSpace(item.Region) == "" || item.Count <= 0 {
continue
}
sourceCountries[item.Region] = item.Count
}
return TrafficDistributions{
StatusCodes: toDistributionItems(statusCodes, limit),
TopDomains: toDistributionItems(topDomains, limit),
@@ -231,7 +249,7 @@ func BuildTrafficDistributions(
func buildHealthSummary(
snapshot *model.OpenFlareMetricSnapshot,
report *model.OpenFlareRequestReport,
traffic *TrafficWindowSummary,
events []*model.OpenFlareHealthEvent,
) HealthSummary {
summary := HealthSummary{}
@@ -258,41 +276,37 @@ func buildHealthSummary(
storageUsage := Percentage(snapshot.StorageUsedBytes, snapshot.StorageTotalBytes)
summary.HasCapacityRisk = snapshot.CPUUsagePercent >= 80 || memoryUsage >= 85 || storageUsage >= 85
}
if report != nil && report.RequestCount >= 100 {
summary.HasTrafficRisk = (float64(report.ErrorCount) / float64(report.RequestCount)) >= 0.05
if traffic != nil && traffic.RequestCount >= 100 {
summary.HasTrafficRisk = (float64(traffic.ErrorCount) / float64(traffic.RequestCount)) >= 0.05
}
summary.HasRuntimeRisk = summary.ActiveAlerts > 0 || summary.HasCapacityRisk || summary.HasTrafficRisk
return summary
}
// BuildNodeTrends builds 24h trend series, preferring ClickHouse hourly aggregates
// over limited raw snapshot windows so capacity/network/disk charts stay complete.
// BuildNodeTrends builds 24h trend series.
// Business traffic (requests/errors/UV and provided/received bytes) comes from access logs.
// Host capacity/disk/network come from metric snapshots (hourly when available).
func BuildNodeTrends(
ctx context.Context,
now time.Time,
nodeID string,
snapshots []*model.OpenFlareMetricSnapshot,
openrestyObs []*model.OpenFlareNodeObservationOpenresty,
reports []*model.OpenFlareRequestReport,
) NodeTrends {
trendSince := now.Add(-24 * time.Hour)
trafficTrend := BuildTrafficTrendPoints(now, reports)
if trafficHourly, err := model.ListOpenFlareTrafficHourlySince(ctx, nodeID, trendSince); err == nil && len(trafficHourly) > 0 {
trafficTrend = BuildTrafficTrendPointsFromHourly(now, trafficHourly)
}
trafficTrend := BuildTrafficTrendPointsFromAccessLogs(ctx, now, nodeID, trendSince)
capacityTrend := BuildCapacityTrendPoints(now, snapshots)
networkTrend := BuildNetworkTrendPoints(now, snapshots, openrestyObs)
networkTrend := BuildNetworkTrendPoints(now, snapshots)
// Overlay L1 business bytes onto network points.
applyAccessLogBytesToNetworkTrend(ctx, now, nodeID, trendSince, networkTrend)
diskIOTrend := BuildDiskIOTrendPoints(now, snapshots)
metricHourly, metricErr := model.ListOpenFlareMetricHourlySince(ctx, nodeID, trendSince)
if metricErr == nil && len(metricHourly) > 0 {
capacityTrend = BuildCapacityTrendPointsFromHourly(now, metricHourly)
diskIOTrend = BuildDiskIOTrendPointsFromHourly(now, metricHourly)
}
openrestyHourly, openrestyErr := model.ListOpenFlareOpenrestyHourlySince(ctx, nodeID, trendSince)
if metricErr == nil && openrestyErr == nil && (len(metricHourly) > 0 || len(openrestyHourly) > 0) {
networkTrend = BuildNetworkTrendPointsFromHourly(now, metricHourly, openrestyHourly)
networkTrend = BuildNetworkTrendPointsFromHourly(now, metricHourly)
applyAccessLogBytesToNetworkTrend(ctx, now, nodeID, trendSince, networkTrend)
}
return NodeTrends{
@@ -303,7 +317,131 @@ func BuildNodeTrends(
}
}
// BuildTrafficTrendPointsFromAccessLogs builds 24h request/error buckets from access logs.
// Prefers of_access_log_hourly when available; falls back to raw bucket aggregates.
// UniqueVisitorCount on hourly path is 0 (use TrafficSummary for exact UV).
func BuildTrafficTrendPointsFromAccessLogs(ctx context.Context, now time.Time, nodeID string, since time.Time) []TrafficTrendPoint {
start := trendWindowStart(now)
points := make([]TrafficTrendPoint, observabilityTrendBuckets)
for index := range points {
points[index].BucketStartedAt = start.Add(time.Duration(index) * time.Hour)
}
if hourly, err := model.ListOpenFlareTrafficHourlySince(ctx, nodeID, since); err == nil && len(hourly) > 0 {
for _, row := range hourly {
if row == nil {
continue
}
index, ok := trendBucketIndex(row.Hour, start)
if !ok {
continue
}
points[index].RequestCount += row.RequestCount
points[index].ErrorCount += row.ErrorCount
// UniqueVisitorCount intentionally not summed from hourly rollup (always 0 / overcounts).
}
return points
}
buckets, err := model.ListOpenFlareAccessLogBuckets(ctx, model.OpenFlareAccessLogBucketQuery{
NodeID: nodeID,
Since: since,
Until: now,
FoldMinutes: 60,
SortBy: "logged_at",
SortOrder: sortOrderAsc,
})
if err != nil || len(buckets) == 0 {
return points
}
byEpoch := make(map[int64]*model.OpenFlareAccessLogBucketRow, len(buckets))
for _, row := range buckets {
if row == nil {
continue
}
byEpoch[row.BucketEpoch] = row
}
for index := range points {
epoch := points[index].BucketStartedAt.Unix()
if row, ok := byEpoch[epoch]; ok {
points[index].RequestCount = row.RequestCount
points[index].ErrorCount = row.ServerErrorCount
points[index].UniqueVisitorCount = row.UniqueIPCount
}
}
return points
}
func applyAccessLogBytesToNetworkTrend(ctx context.Context, now time.Time, nodeID string, since time.Time, points []NetworkTrendPoint) {
if len(points) == 0 {
return
}
// Prefer of_access_log_hourly (summed across hosts).
if hourly, err := analyticsListAccessLogHourlyBytes(ctx, nodeID, since); err == nil && len(hourly) > 0 {
for hourUnix, totals := range hourly {
for index := range points {
if points[index].BucketStartedAt.Unix() == hourUnix {
points[index].BytesProvided = totals.provided
points[index].BytesReceived = totals.received
}
}
}
return
}
buckets, err := model.ListOpenFlareAccessLogBuckets(ctx, model.OpenFlareAccessLogBucketQuery{
NodeID: nodeID,
Since: since,
Until: now,
FoldMinutes: 60,
SortBy: "logged_at",
SortOrder: sortOrderAsc,
})
if err != nil || len(buckets) == 0 {
return
}
byEpoch := make(map[int64]*model.OpenFlareAccessLogBucketRow, len(buckets))
for _, row := range buckets {
if row == nil {
continue
}
byEpoch[row.BucketEpoch] = row
}
for index := range points {
epoch := points[index].BucketStartedAt.Unix()
if row, ok := byEpoch[epoch]; ok {
points[index].BytesProvided = row.BytesSent
points[index].BytesReceived = row.RequestLength
}
}
}
type accessLogHourBytes struct {
provided int64
received int64
}
func analyticsListAccessLogHourlyBytes(ctx context.Context, nodeID string, since time.Time) (map[int64]accessLogHourBytes, error) {
rows, err := model.ListOpenFlareAccessLogHourlySince(ctx, nodeID, since)
if err != nil {
return nil, err
}
out := make(map[int64]accessLogHourBytes)
for _, row := range rows {
if row == nil {
continue
}
key := row.Hour.UTC().Truncate(time.Hour).Unix()
cur := out[key]
cur.provided += row.BytesSent
cur.received += row.RequestLength
out[key] = cur
}
return out, nil
}
// BuildTrafficTrendPointsFromHourly builds 24h traffic trend buckets from hourly rollups.
// UniqueVisitorCount is left at 0: hourly UV is not summed (use TrafficSummary for exact UV).
func BuildTrafficTrendPointsFromHourly(now time.Time, hourly []*model.OpenFlareTrafficHourly) []TrafficTrendPoint {
start := trendWindowStart(now)
points := make([]TrafficTrendPoint, observabilityTrendBuckets)
@@ -320,26 +458,6 @@ func BuildTrafficTrendPointsFromHourly(now time.Time, hourly []*model.OpenFlareT
}
points[index].RequestCount += row.RequestCount
points[index].ErrorCount += row.ErrorCount
points[index].UniqueVisitorCount += row.UniqueVisitorCount
}
return points
}
// BuildTrafficTrendPoints builds 24h traffic trend buckets.
func BuildTrafficTrendPoints(now time.Time, reports []*model.OpenFlareRequestReport) []TrafficTrendPoint {
start := trendWindowStart(now)
points := make([]TrafficTrendPoint, observabilityTrendBuckets)
for index := range points {
points[index].BucketStartedAt = start.Add(time.Duration(index) * time.Hour)
}
for _, report := range reports {
index, ok := trendBucketIndex(report.WindowEndedAt, start)
if !ok {
continue
}
points[index].RequestCount += report.RequestCount
points[index].ErrorCount += report.ErrorCount
points[index].UniqueVisitorCount += report.UniqueVisitorCount
}
return points
}
@@ -404,12 +522,12 @@ func BuildCapacityTrendPointsFromHourly(now time.Time, hourly []*model.OpenFlare
return points
}
// BuildNetworkTrendPoints builds 24h network trend buckets.
// Host and OpenResty counters are cumulative; values are consecutive deltas.
// BuildNetworkTrendPoints builds 24h host-network trend buckets.
// Host network counters must be process-lifetime cumulative values; this function
// converts consecutive samples into deltas.
func BuildNetworkTrendPoints(
now time.Time,
snapshots []*model.OpenFlareMetricSnapshot,
openrestyObs []*model.OpenFlareNodeObservationOpenresty,
) []NetworkTrendPoint {
start := trendWindowStart(now)
points := make([]NetworkTrendPoint, observabilityTrendBuckets)
@@ -452,51 +570,17 @@ func BuildNetworkTrendPoints(
accumulators[index].nodes[snapshot.NodeID] = struct{}{}
}
}
sort.Slice(openrestyObs, func(i int, j int) bool {
if openrestyObs[i].CapturedAt.Equal(openrestyObs[j].CapturedAt) {
return openrestyObs[i].NodeID < openrestyObs[j].NodeID
}
return openrestyObs[i].CapturedAt.Before(openrestyObs[j].CapturedAt)
})
previousOpenrestyByNode := make(map[string]networkCounterState, len(openrestyObs))
for _, obs := range openrestyObs {
if obs == nil {
continue
}
nodeKey := obs.NodeID
if nodeKey == "" {
nodeKey = unknownTrendNodeKey
}
previous := previousOpenrestyByNode[nodeKey]
previousOpenrestyByNode[nodeKey] = networkCounterState{
rx: obs.OpenrestyRxBytes,
tx: obs.OpenrestyTxBytes,
seen: true,
}
if !previous.seen {
continue
}
index, ok := trendBucketIndex(obs.CapturedAt, start)
if !ok {
continue
}
points[index].OpenrestyRxBytes += nonNegativeDelta(obs.OpenrestyRxBytes, previous.rx)
points[index].OpenrestyTxBytes += nonNegativeDelta(obs.OpenrestyTxBytes, previous.tx)
if obs.NodeID != "" {
accumulators[index].nodes[obs.NodeID] = struct{}{}
}
}
for index := range points {
points[index].ReportedNodes = len(accumulators[index].nodes)
}
return points
}
// BuildNetworkTrendPointsFromHourly builds 24h network trend buckets from hourly aggregates.
// BuildNetworkTrendPointsFromHourly builds 24h host-network trend buckets from metric hourly aggregates.
// Business bytes (已提供/接收) are applied separately via applyAccessLogBytesToNetworkTrend.
func BuildNetworkTrendPointsFromHourly(
now time.Time,
metricHourly []*model.OpenFlareMetricHourly,
openrestyHourly []*model.OpenFlareOpenrestyHourly,
) []NetworkTrendPoint {
start := trendWindowStart(now)
points := make([]NetworkTrendPoint, observabilityTrendBuckets)
@@ -517,20 +601,6 @@ func BuildNetworkTrendPointsFromHourly(
points[index].ReportedNodes = row.ReportedNodes
}
}
for _, row := range openrestyHourly {
if row == nil {
continue
}
index, ok := trendBucketIndex(row.Hour, start)
if !ok {
continue
}
points[index].OpenrestyRxBytes += row.OpenrestyRxBytes
points[index].OpenrestyTxBytes += row.OpenrestyTxBytes
if row.ReportedNodes > points[index].ReportedNodes {
points[index].ReportedNodes = row.ReportedNodes
}
}
return points
}
@@ -623,38 +693,25 @@ func latestMetricSnapshot(snapshots []*model.OpenFlareMetricSnapshot) *model.Ope
return latest
}
func latestTrafficReport(reports []*model.OpenFlareRequestReport) *model.OpenFlareRequestReport {
var latest *model.OpenFlareRequestReport
for _, report := range reports {
if report == nil {
continue
}
if latest == nil || report.WindowEndedAt.After(latest.WindowEndedAt) {
latest = report
}
}
return latest
}
func matchOpenrestyObservation(
func matchEdgeHealth(
capturedAt time.Time,
observations []*model.OpenFlareNodeObservationOpenresty,
) *model.OpenFlareNodeObservationOpenresty {
var matched *model.OpenFlareNodeObservationOpenresty
bestDelta := metricSnapshotOpenrestyMatchWindow + time.Second
for _, observation := range observations {
if observation == nil {
health []*model.OpenFlareEdgeHealth,
) *model.OpenFlareEdgeHealth {
var matched *model.OpenFlareEdgeHealth
bestDelta := metricSnapshotEdgeHealthMatchWindow + time.Second
for _, row := range health {
if row == nil {
continue
}
delta := capturedAt.Sub(observation.CapturedAt)
delta := capturedAt.Sub(row.CapturedAt)
if delta < 0 {
delta = -delta
}
if delta > metricSnapshotOpenrestyMatchWindow {
if delta > metricSnapshotEdgeHealthMatchWindow {
continue
}
if matched == nil || delta < bestDelta {
matched = observation
matched = row
bestDelta = delta
}
}
@@ -676,21 +733,6 @@ func LatestMetricSnapshotsByNode(snapshots []*model.OpenFlareMetricSnapshot) map
return result
}
// LatestTrafficReportsByNode returns the latest traffic report per node.
func LatestTrafficReportsByNode(reports []*model.OpenFlareRequestReport) map[string]*model.OpenFlareRequestReport {
result := make(map[string]*model.OpenFlareRequestReport, len(reports))
for _, report := range reports {
if report == nil || report.NodeID == "" {
continue
}
if existing, ok := result[report.NodeID]; ok && !report.WindowEndedAt.After(existing.WindowEndedAt) {
continue
}
result[report.NodeID] = report
}
return result
}
// ActiveHealthEventsByNode groups active health events by node id.
func ActiveHealthEventsByNode(events []*model.OpenFlareHealthEvent) map[string][]*model.OpenFlareHealthEvent {
result := make(map[string][]*model.OpenFlareHealthEvent)
@@ -711,30 +753,6 @@ func Percentage(used int64, total int64) float64 {
return (float64(used) / float64(total)) * percentageMultiplier
}
func mergeJSONCounts(target distributionAccumulator, raw string) {
if len(target) == 0 && strings.TrimSpace(raw) == "" {
return
}
values := parseJSONCounts(raw)
for key, value := range values {
if strings.TrimSpace(key) == "" || value <= 0 {
continue
}
target[key] += value
}
}
func parseJSONCounts(raw string) map[string]int64 {
if strings.TrimSpace(raw) == "" {
return nil
}
values := make(map[string]int64)
if err := json.Unmarshal([]byte(raw), &values); err != nil {
return nil
}
return values
}
func toDistributionItems(values distributionAccumulator, limit int) []DistributionItem {
if len(values) == 0 {
return []DistributionItem{}
@@ -25,52 +25,25 @@ func TestBuildTrafficTrendPointsFromHourlyBucketsByHour(t *testing.T) {
if len(points) != observabilityTrendBuckets {
t.Fatalf("BuildTrafficTrendPointsFromHourly() len = %d, want %d", len(points), observabilityTrendBuckets)
}
}
func TestBuildTrafficTrendPointsBucketsByHour(t *testing.T) {
t.Parallel()
now := time.Date(2026, 6, 19, 17, 30, 0, 0, time.UTC)
reports := []*model.OpenFlareRequestReport{
{
NodeID: "node-a",
WindowStartedAt: now.Add(-3 * time.Hour),
WindowEndedAt: now.Add(-3*time.Hour + time.Minute),
RequestCount: 10,
ErrorCount: 1,
},
{
NodeID: "node-a",
WindowStartedAt: now.Add(-30 * time.Minute),
WindowEndedAt: now.Add(-29 * time.Minute),
RequestCount: 6,
ErrorCount: 0,
},
}
points := BuildTrafficTrendPoints(now, reports)
if len(points) != observabilityTrendBuckets {
t.Fatalf("BuildTrafficTrendPoints() len = %d, want %d", len(points), observabilityTrendBuckets)
}
var totalRequests int64
// Hourly UV must not be summed into trend points.
for _, point := range points {
totalRequests += point.RequestCount
if point.UniqueVisitorCount != 0 {
t.Fatalf("UniqueVisitorCount = %d, want 0 on hourly path", point.UniqueVisitorCount)
}
}
if totalRequests != 16 {
t.Fatalf("total request_count = %d, want 16", totalRequests)
index, ok := trendBucketIndex(now.Add(-2*time.Hour).Truncate(time.Hour), trendWindowStart(now))
if !ok {
t.Fatal("expected valid bucket index")
}
currentHour := points[len(points)-1]
if currentHour.RequestCount != 6 {
t.Fatalf("current hour request_count = %d, want 6", currentHour.RequestCount)
if points[index].RequestCount != 12 {
t.Fatalf("request_count = %d, want 12", points[index].RequestCount)
}
if currentHour.ErrorCount != 0 {
t.Fatalf("current hour error_count = %d, want 0", currentHour.ErrorCount)
if points[index].ErrorCount != 1 {
t.Fatalf("error_count = %d, want 1", points[index].ErrorCount)
}
}
func TestBuildMetricSnapshotViewsMergesOpenrestyObservation(t *testing.T) {
func TestBuildMetricSnapshotViewsMergesEdgeHealthConnections(t *testing.T) {
t.Parallel()
capturedAt := time.Date(2026, 6, 19, 12, 0, 0, 0, time.UTC)
@@ -82,54 +55,30 @@ func TestBuildMetricSnapshotViewsMergesOpenrestyObservation(t *testing.T) {
CPUUsagePercent: 12.5,
},
}
openrestyObs := []*model.OpenFlareNodeObservationOpenresty{
edgeHealth := []*model.OpenFlareEdgeHealth{
{
NodeID: "node-a",
CapturedAt: capturedAt.Add(5 * time.Second),
OpenrestyRxBytes: 4096,
OpenrestyTxBytes: 8192,
OpenrestyConnections: 7,
NodeID: "node-a",
CapturedAt: capturedAt.Add(5 * time.Second),
Status: "healthy",
Connections: 7,
},
}
views := BuildMetricSnapshotViews(snapshots, openrestyObs)
views := BuildMetricSnapshotViews(snapshots, edgeHealth)
if len(views) != 1 {
t.Fatalf("BuildMetricSnapshotViews() len = %d, want 1", len(views))
}
if views[0].OpenrestyRxBytes != 4096 {
t.Fatalf("OpenrestyRxBytes = %d, want 4096", views[0].OpenrestyRxBytes)
}
if views[0].OpenrestyTxBytes != 8192 {
t.Fatalf("OpenrestyTxBytes = %d, want 8192", views[0].OpenrestyTxBytes)
}
if views[0].OpenrestyConnections != 7 {
t.Fatalf("OpenrestyConnections = %d, want 7", views[0].OpenrestyConnections)
}
}
func TestLatestTrafficReportUsesLatestWindowEndedAt(t *testing.T) {
func TestBuildTrafficWindowSummaryFromAccessLogsNilWithoutData(t *testing.T) {
t.Parallel()
older := &model.OpenFlareRequestReport{
WindowEndedAt: time.Date(2026, 6, 19, 10, 0, 0, 0, time.UTC),
RequestCount: 3,
}
newer := &model.OpenFlareRequestReport{
WindowEndedAt: time.Date(2026, 6, 19, 11, 0, 0, 0, time.UTC),
RequestCount: 9,
}
latest := latestTrafficReport([]*model.OpenFlareRequestReport{older, newer})
if latest == nil || latest.RequestCount != 9 {
t.Fatalf("latestTrafficReport() = %#v, want newer report with request_count 9", latest)
}
}
func TestBuildTrafficWindowSummaryNilWithoutReport(t *testing.T) {
t.Parallel()
if summary := buildTrafficWindowSummary(nil); summary != nil {
t.Fatalf("buildTrafficWindowSummary(nil) = %#v, want nil", summary)
// Without an access-log store / data, summary is nil.
if summary := buildTrafficWindowSummaryFromAccessLogs(t.Context(), "missing", time.Now().Add(-time.Hour), time.Now()); summary != nil {
t.Fatalf("buildTrafficWindowSummaryFromAccessLogs() = %#v, want nil", summary)
}
}
@@ -173,12 +122,8 @@ func TestBuildNetworkTrendPointsUsesCounterDeltas(t *testing.T) {
{NodeID: "n1", CapturedAt: base.Add(10 * time.Minute), NetworkRxBytes: 1000, NetworkTxBytes: 2000},
{NodeID: "n1", CapturedAt: base.Add(20 * time.Minute), NetworkRxBytes: 1500, NetworkTxBytes: 2600},
}
openrestyObs := []*model.OpenFlareNodeObservationOpenresty{
{NodeID: "n1", CapturedAt: base.Add(10 * time.Minute), OpenrestyRxBytes: 100, OpenrestyTxBytes: 200},
{NodeID: "n1", CapturedAt: base.Add(20 * time.Minute), OpenrestyRxBytes: 180, OpenrestyTxBytes: 250},
}
points := BuildNetworkTrendPoints(now, snapshots, openrestyObs)
points := BuildNetworkTrendPoints(now, snapshots)
current := points[len(points)-1]
if current.NetworkRxBytes != 500 {
t.Fatalf("network_rx_bytes = %d, want 500", current.NetworkRxBytes)
@@ -186,11 +131,30 @@ func TestBuildNetworkTrendPointsUsesCounterDeltas(t *testing.T) {
if current.NetworkTxBytes != 600 {
t.Fatalf("network_tx_bytes = %d, want 600", current.NetworkTxBytes)
}
if current.OpenrestyRxBytes != 80 {
t.Fatalf("openresty_rx_bytes = %d, want 80", current.OpenrestyRxBytes)
if current.BytesReceived != 0 || current.BytesProvided != 0 {
t.Fatalf("business bytes should be 0 without access logs overlay, got received=%d provided=%d",
current.BytesReceived, current.BytesProvided)
}
if current.OpenrestyTxBytes != 50 {
t.Fatalf("openresty_tx_bytes = %d, want 50", current.OpenrestyTxBytes)
}
func TestBuildHealthSummaryUsesTrafficSummary(t *testing.T) {
t.Parallel()
snapshot := &model.OpenFlareMetricSnapshot{
CPUUsagePercent: 10,
MemoryUsedBytes: 1,
MemoryTotalBytes: 10,
}
traffic := &TrafficWindowSummary{
RequestCount: 200,
ErrorCount: 20, // 10% error rate
}
summary := buildHealthSummary(snapshot, traffic, nil)
if !summary.HasTrafficRisk {
t.Fatal("HasTrafficRisk = false, want true for 10% error rate with >=100 requests")
}
if summary.HasCapacityRisk {
t.Fatal("HasCapacityRisk = true, want false")
}
}
@@ -79,7 +79,6 @@ type NodeView struct {
NodeID string `json:"node_id"`
Profile *model.OpenFlareNodeSystemProfile `json:"profile"`
MetricSnapshots []*NodeMetricSnapshotView `json:"metric_snapshots"`
TrafficReports []*model.OpenFlareRequestReport `json:"traffic_reports"`
HealthEvents []*model.OpenFlareHealthEvent `json:"health_events"`
Analytics NodeAnalytics `json:"analytics"`
Trends NodeTrends `json:"trends"`
@@ -118,11 +117,7 @@ func GetNodeObservability(ctx context.Context, id uint, query NodeQuery) (*NodeV
if err != nil {
return nil, err
}
openrestyObs, err := model.ListOpenFlareNodeObservationOpenresty(ctx, node.NodeID, since, limit)
if err != nil {
return nil, err
}
reports, err := model.ListOpenFlareRequestReportsSince(ctx, node.NodeID, since, limit)
edgeHealth, err := model.ListOpenFlareEdgeHealth(ctx, node.NodeID, since, limit)
if err != nil {
return nil, err
}
@@ -134,18 +129,21 @@ func GetNodeObservability(ctx context.Context, id uint, query NodeQuery) (*NodeV
if err != nil {
return nil, err
}
distributions := BuildTrafficDistributionsFromAccessLogs(
ctx, since, now, defaultTrafficDistributionLimit, accessLogRegions,
)
trafficSummary := buildTrafficWindowSummaryFromAccessLogs(ctx, node.NodeID, since, now)
view := &NodeView{
NodeID: node.NodeID,
Profile: profile,
MetricSnapshots: BuildMetricSnapshotViews(snapshots, openrestyObs),
TrafficReports: reports,
MetricSnapshots: BuildMetricSnapshotViews(snapshots, edgeHealth),
HealthEvents: events,
Analytics: NodeAnalytics{
Traffic: buildTrafficWindowSummary(latestTrafficReport(reports)),
Distributions: BuildTrafficDistributions(reports, accessLogRegions, defaultTrafficDistributionLimit),
Health: buildHealthSummary(latestMetricSnapshot(snapshots), latestTrafficReport(reports), events),
Traffic: trafficSummary,
Distributions: distributions,
Health: buildHealthSummary(latestMetricSnapshot(snapshots), trafficSummary, events),
},
Trends: BuildNodeTrends(ctx, now, node.NodeID, snapshots, openrestyObs, reports),
Trends: BuildNodeTrends(ctx, now, node.NodeID, snapshots),
}
if node.NodeType == "tunnel_relay" {
frpsObs, frpsErr := model.ListOpenFlareNodeObservationFrps(ctx, node.NodeID, time.Time{}, 1)
+1 -1
View File
@@ -16,7 +16,7 @@ const (
relayStatusUnhealthy = "unhealthy"
releaseChannelStable = "stable"
defaultAgentHeartbeatInterval = 10000 // 默认心跳间隔 10 秒(毫秒)
defaultAgentHeartbeatInterval = 3000 // 默认心跳间隔 3 秒(毫秒)
defaultAgentUpdateRepo = "Rain-kl/OpenFlare"
)
@@ -47,7 +47,7 @@ func reconcileRelayHealthEvents(ctx context.Context, nodeID string, relayStatus
func persistRelayHeartbeatObservability(ctx context.Context, nodeID string, payload HeartbeatPayload, reportedAt time.Time) {
agent.PersistHeartbeatObservability(ctx, nodeID, agent.NodePayload{
Profile: payload.Profile,
Snapshot: payload.Snapshot,
HostMetrics: payload.Snapshot,
HealthEvents: payload.HealthEvents,
}, reportedAt)
@@ -20,10 +20,8 @@ const (
DatabaseCleanupTargetAccessLogs = "node_access_logs"
// DatabaseCleanupTargetMetricSnapshots is the API cleanup target for metric snapshots.
DatabaseCleanupTargetMetricSnapshots = "node_metric_snapshots"
// DatabaseCleanupTargetRequestReports is the API cleanup target for request reports.
DatabaseCleanupTargetRequestReports = "node_request_reports"
// DatabaseCleanupTargetObsOpenresty is the API cleanup target for OpenResty observations.
DatabaseCleanupTargetObsOpenresty = "node_obs_openresty"
// DatabaseCleanupTargetEdgeHealth is the API cleanup target for OpenResty edge health (connections).
DatabaseCleanupTargetEdgeHealth = "node_edge_health"
// DatabaseCleanupTargetObsFrps is the API cleanup target for FRPS observations.
DatabaseCleanupTargetObsFrps = "node_obs_frps"
// DatabaseCleanupTargetObsFrpc is the API cleanup target for FRPC observations.
@@ -33,8 +31,7 @@ const (
var databaseCleanupTargets = map[string]string{
DatabaseCleanupTargetAccessLogs: "访问日志",
DatabaseCleanupTargetMetricSnapshots: "性能快照",
DatabaseCleanupTargetRequestReports: "请求聚合",
DatabaseCleanupTargetObsOpenresty: "OpenResty 观测",
DatabaseCleanupTargetEdgeHealth: "OpenResty 健康(连接)",
DatabaseCleanupTargetObsFrps: "FRPS 观测",
DatabaseCleanupTargetObsFrpc: "FRPC 观测",
}
@@ -43,8 +40,7 @@ var databaseCleanupTargets = map[string]string{
var databaseCleanupTableTTLDays = map[string]int{
DatabaseCleanupTargetAccessLogs: analyticsrepo.TableTTLDaysNodeAccessLogs,
DatabaseCleanupTargetMetricSnapshots: analyticsrepo.TableTTLDaysNodeMetricSnapshots,
DatabaseCleanupTargetRequestReports: analyticsrepo.TableTTLDaysNodeRequestReports,
DatabaseCleanupTargetObsOpenresty: analyticsrepo.TableTTLDaysNodeObs,
DatabaseCleanupTargetEdgeHealth: analyticsrepo.TableTTLDaysNodeObs,
DatabaseCleanupTargetObsFrps: analyticsrepo.TableTTLDaysNodeObs,
DatabaseCleanupTargetObsFrpc: analyticsrepo.TableTTLDaysNodeObs,
}
@@ -165,8 +161,7 @@ func RunDatabaseAutoCleanupOnce(ctx context.Context, now time.Time) (*DatabaseAu
for _, target := range []string{
DatabaseCleanupTargetAccessLogs,
DatabaseCleanupTargetMetricSnapshots,
DatabaseCleanupTargetRequestReports,
DatabaseCleanupTargetObsOpenresty,
DatabaseCleanupTargetEdgeHealth,
DatabaseCleanupTargetObsFrps,
DatabaseCleanupTargetObsFrpc,
} {
@@ -201,10 +196,8 @@ func deleteAllObservabilityRows(ctx context.Context, target string) (int64, stri
deleted, err = model.DeleteAllOpenFlareAccessLogs(ctx)
case DatabaseCleanupTargetMetricSnapshots:
deleted, err = model.DeleteAllOpenFlareMetricSnapshots(ctx)
case DatabaseCleanupTargetRequestReports:
deleted, err = model.DeleteAllOpenFlareRequestReports(ctx)
case DatabaseCleanupTargetObsOpenresty:
deleted, err = model.DeleteAllOpenFlareNodeObservationOpenresty(ctx)
case DatabaseCleanupTargetEdgeHealth:
deleted, err = model.DeleteAllOpenFlareEdgeHealth(ctx)
case DatabaseCleanupTargetObsFrps:
deleted, err = model.DeleteAllOpenFlareNodeObservationFrps(ctx)
case DatabaseCleanupTargetObsFrpc:
@@ -236,10 +229,8 @@ func materializeObservabilityTableTTL(ctx context.Context, target string) (int64
eligible, err = model.DeleteOpenFlareAccessLogsBefore(ctx, cutoff)
case DatabaseCleanupTargetMetricSnapshots:
eligible, err = model.DeleteOpenFlareMetricSnapshotsBefore(ctx, cutoff)
case DatabaseCleanupTargetRequestReports:
eligible, err = model.DeleteOpenFlareRequestReportsBefore(ctx, cutoff)
case DatabaseCleanupTargetObsOpenresty:
eligible, err = model.DeleteOpenFlareNodeObservationOpenrestyBefore(ctx, cutoff)
case DatabaseCleanupTargetEdgeHealth:
eligible, err = model.DeleteOpenFlareEdgeHealthBefore(ctx, cutoff)
case DatabaseCleanupTargetObsFrps:
eligible, err = model.DeleteOpenFlareNodeObservationFrpsBefore(ctx, cutoff)
case DatabaseCleanupTargetObsFrpc:
@@ -157,11 +157,11 @@ func TestRunDatabaseAutoCleanupOnceClampsRetentionToTableTTL(t *testing.T) {
CapturedAt: now.Add(-40 * 24 * time.Hour),
CPUUsagePercent: 10,
}))
require.NoError(t, model.InsertOpenFlareRequestReport(ctx, &model.OpenFlareRequestReport{
NodeID: "node-a",
WindowStartedAt: now.Add(-41 * 24 * time.Hour),
WindowEndedAt: now.Add(-40 * 24 * time.Hour),
RequestCount: 15,
require.NoError(t, model.InsertOpenFlareEdgeHealth(ctx, &model.OpenFlareEdgeHealth{
NodeID: "node-a",
CapturedAt: now.Add(-40 * 24 * time.Hour),
Status: "healthy",
Connections: 2,
}))
require.NoError(t, repository.SaveOrUpdateSystemConfig(ctx, model.ConfigKeyDatabaseAutoCleanupEnabled, "true"))
@@ -170,7 +170,7 @@ func TestRunDatabaseAutoCleanupOnceClampsRetentionToTableTTL(t *testing.T) {
summary, err := RunDatabaseAutoCleanupOnce(ctx, now)
require.NoError(t, err)
require.NotNil(t, summary)
require.Len(t, summary.Results, 6)
require.Len(t, summary.Results, 5)
assert.Equal(t, 1, summary.RetentionDays)
for _, result := range summary.Results {
@@ -189,9 +189,9 @@ func TestRunDatabaseAutoCleanupOnceClampsRetentionToTableTTL(t *testing.T) {
require.NoError(t, err)
assert.Empty(t, metricSnapshots)
requestReports, err := model.ListOpenFlareRequestReportsSince(ctx, "", time.Time{}, 0)
edgeHealth, err := model.ListOpenFlareEdgeHealth(ctx, "", time.Time{}, 0)
require.NoError(t, err)
assert.Empty(t, requestReports)
assert.Empty(t, edgeHealth)
}
func TestTableTTLDaysForCleanupTarget(t *testing.T) {