feat(obs): 访问日志 SSOT 与 edge_health,去掉协议兼容层

Agent 仅上报 host_metrics/edge_health/access_logs;业务流量与 UV 由
Server 侧访问日志聚合。新增 of_node_edge_health 与 of_access_log_hourly,
删除 request_reports/openresty 吞吐路径;API 不再暴露 traffic_reports
与 openresty_rx|tx。心跳/离线默认阈值与回填迁移一并入库。
This commit is contained in:
ryan
2026-07-18 11:53:11 +08:00
parent 9a0974cce8
commit f0e234df1f
60 changed files with 1799 additions and 2164 deletions
+15 -15
View File
@@ -403,11 +403,11 @@ func TestRunnerHeartbeatPayloadIncludesObservabilityExtensions(t *testing.T) {
if firstPayload.Profile == nil { if firstPayload.Profile == nil {
t.Fatal("expected first heartbeat payload to include system profile") t.Fatal("expected first heartbeat payload to include system profile")
} }
if firstPayload.Snapshot == nil { if firstPayload.HostMetrics == nil {
t.Fatal("expected first heartbeat payload to include metric snapshot") t.Fatal("expected first heartbeat payload to include host metrics")
} }
if firstPayload.TrafficReport == nil || firstPayload.TrafficReport.RequestCount != 1 { if firstPayload.SchemaVersion != 2 {
t.Fatalf("expected first heartbeat payload to include traffic report, got %+v", firstPayload.TrafficReport) t.Fatalf("expected schema_version 2, got %d", firstPayload.SchemaVersion)
} }
if len(firstPayload.AccessLogs) != 1 || firstPayload.AccessLogs[0].Path != "/" { if len(firstPayload.AccessLogs) != 1 || firstPayload.AccessLogs[0].Path != "/" {
t.Fatalf("expected first heartbeat payload to include access logs, got %+v", firstPayload.AccessLogs) t.Fatalf("expected first heartbeat payload to include access logs, got %+v", firstPayload.AccessLogs)
@@ -420,11 +420,8 @@ func TestRunnerHeartbeatPayloadIncludesObservabilityExtensions(t *testing.T) {
if secondPayload.Profile != nil { if secondPayload.Profile != nil {
t.Fatal("expected unchanged profile to be omitted on subsequent heartbeat") t.Fatal("expected unchanged profile to be omitted on subsequent heartbeat")
} }
if secondPayload.Snapshot == nil { if secondPayload.HostMetrics == nil {
t.Fatal("expected metric snapshot to continue reporting on subsequent heartbeat") t.Fatal("expected host metrics to continue reporting on subsequent heartbeat")
}
if secondPayload.TrafficReport != nil {
t.Fatalf("expected unchanged traffic window to be omitted on subsequent heartbeat, got %+v", secondPayload.TrafficReport)
} }
if len(secondPayload.AccessLogs) != 0 { if len(secondPayload.AccessLogs) != 0 {
t.Fatalf("expected unchanged access log delta to be omitted on subsequent heartbeat, got %+v", secondPayload.AccessLogs) t.Fatalf("expected unchanged access log delta to be omitted on subsequent heartbeat, got %+v", secondPayload.AccessLogs)
@@ -442,8 +439,8 @@ func TestRunnerReplaysBufferedObservabilityAfterHeartbeatRecovery(t *testing.T)
bufferWindow := nowUnix - (nowUnix % 60) - 60 bufferWindow := nowUnix - (nowUnix % 60) - 60
if err := bufferStore.Upsert(state.ObservabilityBufferRecord{ if err := bufferStore.Upsert(state.ObservabilityBufferRecord{
WindowStartedAtUnix: bufferWindow, WindowStartedAtUnix: bufferWindow,
Snapshot: &protocol.NodeMetricSnapshot{CapturedAtUnix: bufferWindow + 5, CPUUsagePercent: 30}, HostMetrics: &protocol.NodeMetricSnapshot{CapturedAtUnix: bufferWindow + 5, CPUUsagePercent: 30},
TrafficReport: &protocol.NodeTrafficReport{WindowStartedAtUnix: bufferWindow, WindowEndedAtUnix: bufferWindow + 60, RequestCount: 8}, EdgeHealth: &protocol.NodeEdgeHealth{CapturedAtUnix: bufferWindow + 5, Connections: 3, Status: "healthy"},
QueuedAtUnix: bufferWindow + 60, QueuedAtUnix: bufferWindow + 60,
}, 0); err != nil { }, 0); err != nil {
t.Fatalf("failed to seed observability buffer: %v", err) t.Fatalf("failed to seed observability buffer: %v", err)
@@ -493,11 +490,14 @@ func TestRunnerReplaysBufferedObservabilityAfterHeartbeatRecovery(t *testing.T)
t.Fatalf("expected two heartbeat payloads, got %d", len(heartbeatService.heartbeatPayloads)) t.Fatalf("expected two heartbeat payloads, got %d", len(heartbeatService.heartbeatPayloads))
} }
secondPayload := heartbeatService.heartbeatPayloads[1] secondPayload := heartbeatService.heartbeatPayloads[1]
if len(secondPayload.BufferedObservability) != 1 { if len(secondPayload.Buffered) != 1 {
t.Fatalf("expected second heartbeat to replay one buffered observation, got %+v", secondPayload.BufferedObservability) t.Fatalf("expected second heartbeat to replay one buffered observation, got %+v", secondPayload.Buffered)
} }
if len(secondPayload.BufferedObservability[0].AccessLogs) != 0 { if len(secondPayload.Buffered[0].AccessLogs) != 0 {
t.Fatalf("expected seeded buffered observation to keep empty access logs, got %+v", secondPayload.BufferedObservability[0].AccessLogs) t.Fatalf("expected seeded buffered observation to keep empty access logs, got %+v", secondPayload.Buffered[0].AccessLogs)
}
if secondPayload.Buffered[0].EdgeHealth == nil || secondPayload.Buffered[0].EdgeHealth.Connections != 3 {
t.Fatalf("expected buffered edge health, got %+v", secondPayload.Buffered[0].EdgeHealth)
} }
replayable, err := bufferStore.Replayable(0, 0) replayable, err := bufferStore.Replayable(0, 0)
+2 -2
View File
@@ -29,11 +29,11 @@ const (
defaultStateRelativePath = "var/lib/openflare/agent-state.json" defaultStateRelativePath = "var/lib/openflare/agent-state.json"
defaultObservabilityBufferRelativePath = "var/lib/openflare/observability-buffer.json" defaultObservabilityBufferRelativePath = "var/lib/openflare/observability-buffer.json"
defaultOpenRestyObservabilityPort = 18081 defaultOpenRestyObservabilityPort = 18081
defaultObservabilityReplayMinutes = 15 defaultObservabilityReplayMinutes = 60
defaultMMDBUpdateInterval = 24 * time.Hour defaultMMDBUpdateInterval = 24 * time.Hour
defaultMMDBDownloadURL = "https://github.com/FyraLabs/geolite2/releases/latest/download/GeoLite2-Country.mmdb" defaultMMDBDownloadURL = "https://github.com/FyraLabs/geolite2/releases/latest/download/GeoLite2-Country.mmdb"
defaultCityMMDBDownloadURL = "https://github.com/FyraLabs/geolite2/releases/latest/download/GeoLite2-City.mmdb" defaultCityMMDBDownloadURL = "https://github.com/FyraLabs/geolite2/releases/latest/download/GeoLite2-City.mmdb"
defaultHeartbeatInterval = 10 * time.Second defaultHeartbeatInterval = 3 * time.Second
defaultRequestTimeout = 10 * time.Second defaultRequestTimeout = 10 * time.Second
configFilePerm = 0o600 configFilePerm = 0o600
) )
+30 -35
View File
@@ -90,13 +90,10 @@ func (c *Cycle) NodePayload(ctx context.Context, nodeID string) protocol.NodePay
openrestyStatus = protocol.OpenrestyStatusUnknown openrestyStatus = protocol.OpenrestyStatusUnknown
} }
profile := observability.BuildProfile(c.Config, c.StateStore) profile := observability.BuildProfile(c.Config, c.StateStore)
managedOpenRestyMetrics := observability.CollectManagedOpenRestyMetrics(ctx, c.Config) edgeSnapshot := observability.CollectEdgeHealth(ctx, c.Config)
trafficReport, accessLogs, fallbackMetrics := observability.BuildTrafficObservability(c.Config, c.StateStore, managedOpenRestyMetrics) accessLogs := observability.CollectAccessLogs(c.Config, c.StateStore)
if managedOpenRestyMetrics == nil {
managedOpenRestyMetrics = fallbackMetrics
}
metricSnapshot := observability.BuildSnapshot(c.Config, c.StateStore) metricSnapshot := observability.BuildSnapshot(c.Config, c.StateStore)
openrestyObservation := observability.BuildOpenrestyObservation(managedOpenRestyMetrics) edgeHealth := observability.BuildEdgeHealth(edgeSnapshot, openrestyStatus, snapshot.OpenrestyMessage)
healthEvents := observability.BuildHealthEvents(snapshot) healthEvents := observability.BuildHealthEvents(snapshot)
ip := c.Config.NodeIP ip := c.Config.NodeIP
@@ -105,21 +102,21 @@ func (c *Cycle) NodePayload(ctx context.Context, nodeID string) protocol.NodePay
} }
payload := protocol.NodePayload{ payload := protocol.NodePayload{
NodeID: nodeID, SchemaVersion: 2,
Name: c.Config.NodeName, NodeID: nodeID,
IP: ip, Name: c.Config.NodeName,
Version: c.Config.Version, IP: ip,
ExtVersion: c.Config.ExtVersion, Version: c.Config.Version,
CurrentVersion: snapshot.CurrentVersion, ExtVersion: c.Config.ExtVersion,
LastError: snapshot.LastError, CurrentVersion: snapshot.CurrentVersion,
OpenrestyStatus: openrestyStatus, LastError: snapshot.LastError,
OpenrestyMessage: snapshot.OpenrestyMessage, OpenrestyStatus: openrestyStatus,
Profile: profile, OpenrestyMessage: snapshot.OpenrestyMessage,
Snapshot: metricSnapshot, Profile: profile,
OpenrestyObservation: openrestyObservation, HostMetrics: metricSnapshot,
TrafficReport: trafficReport, EdgeHealth: edgeHealth,
AccessLogs: accessLogs, AccessLogs: accessLogs,
HealthEvents: healthEvents, HealthEvents: healthEvents,
} }
if c.Sync != nil { if c.Sync != nil {
checksums, err := c.Sync.WAFIPGroupChecksums() checksums, err := c.Sync.WAFIPGroupChecksums()
@@ -135,23 +132,22 @@ func (c *Cycle) NodePayload(ctx context.Context, nodeID string) protocol.NodePay
// PrepareHeartbeatPayload constructs the heartbeat payload with buffered observability records and returns the window timestamps to acknowledge. // PrepareHeartbeatPayload constructs the heartbeat payload with buffered observability records and returns the window timestamps to acknowledge.
func (c *Cycle) PrepareHeartbeatPayload(ctx context.Context, nodeID string) (protocol.NodePayload, []int64) { func (c *Cycle) PrepareHeartbeatPayload(ctx context.Context, nodeID string) (protocol.NodePayload, []int64) {
payload := c.NodePayload(ctx, nodeID) payload := c.NodePayload(ctx, nodeID)
if c.ObservabilityBuffer == nil || (payload.Snapshot == nil && payload.TrafficReport == nil && len(payload.AccessLogs) == 0) { if c.ObservabilityBuffer == nil || (payload.HostMetrics == nil && payload.EdgeHealth == nil && len(payload.AccessLogs) == 0) {
return payload, nil return payload, nil
} }
now := time.Now().UTC() now := time.Now().UTC()
retainAfterUnix := now.Add(-time.Duration(c.Config.ObservabilityReplayMinutes) * time.Minute).Unix() retainAfterUnix := now.Add(-time.Duration(c.Config.ObservabilityReplayMinutes) * time.Minute).Unix()
windowStartedAtUnix := state.ObservabilityWindowStartedAt(payload.Snapshot, payload.OpenrestyObservation, payload.TrafficReport) windowStartedAtUnix := state.ObservabilityWindowStartedAt(payload.HostMetrics, payload.EdgeHealth)
if windowStartedAtUnix <= 0 { if windowStartedAtUnix <= 0 {
return payload, nil return payload, nil
} }
record := state.ObservabilityBufferRecord{ record := state.ObservabilityBufferRecord{
WindowStartedAtUnix: windowStartedAtUnix, WindowStartedAtUnix: windowStartedAtUnix,
Snapshot: payload.Snapshot, HostMetrics: payload.HostMetrics,
OpenrestyObservation: payload.OpenrestyObservation, EdgeHealth: payload.EdgeHealth,
TrafficReport: payload.TrafficReport, AccessLogs: payload.AccessLogs,
AccessLogs: payload.AccessLogs, QueuedAtUnix: now.Unix(),
QueuedAtUnix: now.Unix(),
} }
if err := c.ObservabilityBuffer.Upsert(record, retainAfterUnix); err != nil { if err := c.ObservabilityBuffer.Upsert(record, retainAfterUnix); err != nil {
slog.Error("upsert observability buffer failed", "error", err) slog.Error("upsert observability buffer failed", "error", err)
@@ -171,15 +167,14 @@ func (c *Cycle) PrepareHeartbeatPayload(ctx context.Context, nodeID string) (pro
continue continue
} }
buffered = append(buffered, protocol.BufferedObservabilityRecord{ buffered = append(buffered, protocol.BufferedObservabilityRecord{
WindowStartedAtUnix: item.WindowStartedAtUnix, CapturedAtUnix: item.WindowStartedAtUnix,
Snapshot: item.Snapshot, HostMetrics: item.HostMetrics,
OpenrestyObservation: item.OpenrestyObservation, EdgeHealth: item.EdgeHealth,
TrafficReport: item.TrafficReport, AccessLogs: item.AccessLogs,
AccessLogs: item.AccessLogs,
}) })
ackWindows = append(ackWindows, item.WindowStartedAtUnix) ackWindows = append(ackWindows, item.WindowStartedAtUnix)
} }
payload.BufferedObservability = buffered payload.Buffered = buffered
ackWindows = append(ackWindows, windowStartedAtUnix) ackWindows = append(ackWindows, windowStartedAtUnix)
return payload, ackWindows return payload, ackWindows
} }
+32 -127
View File
@@ -2,147 +2,52 @@ package nginx
import "github.com/Rain-kl/Wavelet/internal/apps/agent/protocol" import "github.com/Rain-kl/Wavelet/internal/apps/agent/protocol"
const ( // Local OpenResty observability endpoint (target model):
openRestyObservabilityWindowTTL = "7200" // GET /openflare/observability returns instantaneous health/connections only.
openRestyObservabilityWindowSize = "60" // Business traffic is collected exclusively from access.log.
)
const openRestyObservabilityInitLua = `local dict = ngx.shared.openflare_observability const openRestyObservabilityInitLua = `return
if not dict then
return
end
return
` `
const openRestyObservabilityLogLua = `local dict = ngx.shared.openflare_observability // log.lua no longer accumulates business counters (access.log is the authority).
if not dict then const openRestyObservabilityLogLua = `return
return
end
local request_uri = tostring(ngx.var.uri or "")
if request_uri == "/openflare/observability" or request_uri == "/openflare/stub_status" then
return
end
local ttl = ` + openRestyObservabilityWindowTTL + `
local now = ngx.time()
local window_size = ` + openRestyObservabilityWindowSize + `
local window_start = now - (now % window_size)
local function ensure_counter(key)
dict:add(key, 0, ttl)
end
local function incr(key, delta)
ensure_counter(key)
local value, err = dict:incr(key, delta)
if not value and err == "not found" then
dict:set(key, delta, ttl)
end
end
local function remember_value(list_key, marker_key, value)
if value == "" then
return
end
if not dict:add(marker_key, 1, ttl) then
return
end
local existing = dict:get(list_key)
if not existing or existing == "" then
dict:set(list_key, value, ttl)
return
end
dict:set(list_key, existing .. "\n" .. value, ttl)
end
local window_prefix = tostring(window_start)
incr("request_count:" .. window_prefix, 1)
local status = tostring(ngx.status or 0)
if status ~= "0" then
incr("status:" .. window_prefix .. ":" .. status, 1)
remember_value(
"status_keys:" .. window_prefix,
"status_marker:" .. window_prefix .. ":" .. status,
status
)
if tonumber(status) and tonumber(status) >= 500 then
incr("error_count:" .. window_prefix, 1)
end
end
local host = tostring(ngx.var.host or "")
if host ~= "" then
incr("domain:" .. window_prefix .. ":" .. host, 1)
remember_value(
"domain_keys:" .. window_prefix,
"domain_marker:" .. window_prefix .. ":" .. host,
host
)
end
local remote_addr = tostring(ngx.var.binary_remote_addr or ngx.var.remote_addr or "")
if remote_addr ~= "" and dict:add("visitor:" .. window_prefix .. ":" .. remote_addr, 1, ttl) then
incr("unique_visitor_count:" .. window_prefix, 1)
end
local request_length = tonumber(ngx.var.request_length) or 0
if request_length > 0 then
incr("openresty_rx_bytes:" .. window_prefix, request_length)
end
local bytes_sent = tonumber(ngx.var.bytes_sent) or tonumber(ngx.var.body_bytes_sent) or 0
if bytes_sent > 0 then
incr("openresty_tx_bytes:" .. window_prefix, bytes_sent)
end
` `
// read.lua exposes stub_status-style connection gauges as JSON.
const openRestyObservabilityReadLua = `local cjson = require "cjson.safe" const openRestyObservabilityReadLua = `local cjson = require "cjson.safe"
local dict = ngx.shared.openflare_observability local function read_stub_status()
if not dict then local res = ngx.location.capture("/openflare/stub_status")
ngx.status = ngx.HTTP_SERVICE_UNAVAILABLE if not res or res.status ~= 200 or not res.body then
ngx.say(cjson.encode({ message = "shared dict unavailable" })) return nil
return
end
local now = ngx.time()
local window_size = ` + openRestyObservabilityWindowSize + `
local window_start = now - (now % window_size)
local current_window = tostring(window_start)
local function read_counter(key)
return tonumber(dict:get(key) or 0) or 0
end
local function read_map(window_id, prefix, list_key)
local result = {}
local raw = dict:get(list_key .. ":" .. window_id)
if not raw or raw == "" then
return result
end end
for value in string.gmatch(raw, "[^\n]+") do local body = res.body
result[value] = read_counter(prefix .. ":" .. window_id .. ":" .. value) local active = tonumber(string.match(body, "Active connections:%s*(%d+)")) or 0
end local reading = tonumber(string.match(body, "Reading:%s*(%d+)")) or 0
return result local writing = tonumber(string.match(body, "Writing:%s*(%d+)")) or 0
local waiting = tonumber(string.match(body, "Waiting:%s*(%d+)")) or 0
return {
active = active,
reading = reading,
writing = writing,
waiting = waiting
}
end end
local connections = read_stub_status()
local payload = { local payload = {
window_started_at_unix = window_start, ok = connections ~= nil,
window_ended_at_unix = now, captured_at_unix = ngx.time(),
request_count = read_counter("request_count:" .. current_window), connections = connections or {
error_count = read_counter("error_count:" .. current_window), active = 0,
unique_visitor_count = read_counter("unique_visitor_count:" .. current_window), reading = 0,
status_codes = read_map(current_window, "status", "status_keys"), writing = 0,
top_domains = read_map(current_window, "domain", "domain_keys"), waiting = 0
source_countries = {}, }
openresty_rx_bytes = read_counter("openresty_rx_bytes:" .. current_window),
openresty_tx_bytes = read_counter("openresty_tx_bytes:" .. current_window)
} }
ngx.header.content_type = "application/json" ngx.header.content_type = "application/json"
ngx.status = ngx.HTTP_OK
ngx.say(cjson.encode(payload)) ngx.say(cjson.encode(payload))
` `
@@ -0,0 +1,35 @@
package nginx
import (
"strings"
"testing"
)
func TestManagedObservabilityLuaIsHealthOnly(t *testing.T) {
t.Parallel()
files := ManagedObservabilityLuaFiles()
var logLua, readLua string
for _, file := range files {
switch file.Path {
case "log.lua":
logLua = file.Content
case "read.lua":
readLua = file.Content
}
}
if logLua == "" || readLua == "" {
t.Fatal("expected log.lua and read.lua")
}
// Business counters must not be written in log phase.
if strings.Contains(logLua, "openresty_rx_bytes") ||
strings.Contains(logLua, "request_count") {
t.Fatal("log.lua must not accumulate business counters")
}
if !strings.Contains(readLua, "connections") || !strings.Contains(readLua, "ok") {
t.Fatal("read.lua must expose ok + connections health snapshot")
}
if strings.Contains(readLua, "top_domains") || strings.Contains(readLua, "request_count") {
t.Fatal("read.lua must not expose business traffic aggregates")
}
}
+30 -9
View File
@@ -84,16 +84,37 @@ func BuildSnapshot(cfg *config.Config, stateStore *state.Store) *protocol.NodeMe
return metric return metric
} }
// BuildOpenrestyObservation builds the OpenResty observation protocol model from the managed metrics. // BuildEdgeHealth builds the edge_health payload from a local probe and node status.
func BuildOpenrestyObservation(managed *ManagedOpenRestyMetrics) *protocol.NodeOpenrestyObservation { func BuildEdgeHealth(probe *EdgeHealthSnapshot, openrestyStatus, openrestyMessage string) *protocol.NodeEdgeHealth {
if managed == nil { status := strings.TrimSpace(openrestyStatus)
return nil message := strings.TrimSpace(openrestyMessage)
if probe == nil {
if status == "" {
return nil
}
return &protocol.NodeEdgeHealth{
CapturedAtUnix: time.Now().UTC().Unix(),
Status: status,
Message: message,
Connections: 0,
}
} }
return &protocol.NodeOpenrestyObservation{
CapturedAtUnix: time.Now().UTC().Unix(), captured := probe.CapturedAtUnix
OpenrestyRxBytes: managed.OpenrestyRxBytes, if captured <= 0 {
OpenrestyTxBytes: managed.OpenrestyTxBytes, captured = time.Now().UTC().Unix()
OpenrestyConnections: managed.OpenrestyConnections, }
if status == "" {
status = protocol.OpenrestyStatusUnknown
if probe.OK {
status = protocol.OpenrestyStatusHealthy
}
}
return &protocol.NodeEdgeHealth{
CapturedAtUnix: captured,
Status: status,
Message: message,
Connections: probe.Connections,
} }
} }
@@ -4,48 +4,37 @@ import (
"context" "context"
"encoding/json" "encoding/json"
"fmt" "fmt"
"io"
"net/http" "net/http"
"regexp"
"strconv"
"strings"
"time" "time"
"github.com/Rain-kl/Wavelet/internal/apps/agent/config" "github.com/Rain-kl/Wavelet/internal/apps/agent/config"
"github.com/Rain-kl/Wavelet/internal/apps/agent/protocol"
) )
const ( const openRestyObservabilityPath = "/openflare/observability"
openRestyObservabilityPath = "/openflare/observability"
openRestyStubStatusPath = "/openflare/stub_status"
stubStatusActiveMatchGroupCount = 2
)
var stubStatusActivePattern = regexp.MustCompile(`Active connections:\s+(\d+)`) // EdgeHealthSnapshot is the L2 OpenResty health probe result.
type EdgeHealthSnapshot struct {
// ManagedOpenRestyMetrics holds metrics collected from the local OpenResty instance. OK bool
type ManagedOpenRestyMetrics struct { CapturedAtUnix int64
TrafficReport *protocol.NodeTrafficReport Connections int64
OpenrestyRxBytes int64 Reading int64
OpenrestyTxBytes int64 Writing int64
OpenrestyConnections int64 Waiting int64
} }
type openRestyObservabilityResponse struct { type openRestyObservabilityResponse struct {
WindowStartedAtUnix int64 `json:"window_started_at_unix"` OK bool `json:"ok"`
WindowEndedAtUnix int64 `json:"window_ended_at_unix"` CapturedAtUnix int64 `json:"captured_at_unix"`
RequestCount int64 `json:"request_count"` Connections struct {
ErrorCount int64 `json:"error_count"` Active int64 `json:"active"`
UniqueVisitorCount int64 `json:"unique_visitor_count"` Reading int64 `json:"reading"`
StatusCodes map[string]int64 `json:"status_codes"` Writing int64 `json:"writing"`
TopDomains map[string]int64 `json:"top_domains"` Waiting int64 `json:"waiting"`
SourceCountries map[string]int64 `json:"source_countries"` } `json:"connections"`
OpenrestyRxBytes int64 `json:"openresty_rx_bytes"`
OpenrestyTxBytes int64 `json:"openresty_tx_bytes"`
} }
// CollectManagedOpenRestyMetrics collects metrics from the local OpenResty observability endpoints. // CollectEdgeHealth probes the local OpenResty observability JSON endpoint.
func CollectManagedOpenRestyMetrics(ctx context.Context, cfg *config.Config) *ManagedOpenRestyMetrics { func CollectEdgeHealth(ctx context.Context, cfg *config.Config) *EdgeHealthSnapshot {
if cfg == nil || cfg.OpenrestyObservabilityPort <= 0 { if cfg == nil || cfg.OpenrestyObservabilityPort <= 0 {
return nil return nil
} }
@@ -53,31 +42,23 @@ func CollectManagedOpenRestyMetrics(ctx context.Context, cfg *config.Config) *Ma
baseURL := fmt.Sprintf("http://127.0.0.1:%d", cfg.OpenrestyObservabilityPort) baseURL := fmt.Sprintf("http://127.0.0.1:%d", cfg.OpenrestyObservabilityPort)
client := &http.Client{Timeout: 1500 * time.Millisecond} client := &http.Client{Timeout: 1500 * time.Millisecond}
observabilityResp := openRestyObservabilityResponse{} var resp openRestyObservabilityResponse
if err := fetchLocalJSON(ctx, client, baseURL+openRestyObservabilityPath, &observabilityResp); err != nil { if err := fetchLocalJSON(ctx, client, baseURL+openRestyObservabilityPath, &resp); err != nil {
return nil return nil
} }
result := &ManagedOpenRestyMetrics{ captured := resp.CapturedAtUnix
TrafficReport: &protocol.NodeTrafficReport{ if captured <= 0 {
WindowStartedAtUnix: observabilityResp.WindowStartedAtUnix, captured = time.Now().UTC().Unix()
WindowEndedAtUnix: observabilityResp.WindowEndedAtUnix,
RequestCount: observabilityResp.RequestCount,
ErrorCount: observabilityResp.ErrorCount,
UniqueVisitorCount: observabilityResp.UniqueVisitorCount,
StatusCodes: normalizeCountMap(observabilityResp.StatusCodes),
TopDomains: normalizeCountMap(observabilityResp.TopDomains),
SourceCountries: normalizeCountMap(observabilityResp.SourceCountries),
},
OpenrestyRxBytes: observabilityResp.OpenrestyRxBytes,
OpenrestyTxBytes: observabilityResp.OpenrestyTxBytes,
} }
return &EdgeHealthSnapshot{
if text, err := fetchLocalText(ctx, client, baseURL+openRestyStubStatusPath); err == nil { OK: resp.OK,
result.OpenrestyConnections = parseStubStatusActiveConnections(text) CapturedAtUnix: captured,
Connections: resp.Connections.Active,
Reading: resp.Connections.Reading,
Writing: resp.Connections.Writing,
Waiting: resp.Connections.Waiting,
} }
return result
} }
func fetchLocalJSON(ctx context.Context, client *http.Client, url string, target any) error { func fetchLocalJSON(ctx context.Context, client *http.Client, url string, target any) error {
@@ -95,50 +76,3 @@ func fetchLocalJSON(ctx context.Context, client *http.Client, url string, target
} }
return json.NewDecoder(resp.Body).Decode(target) return json.NewDecoder(resp.Body).Decode(target)
} }
func fetchLocalText(ctx context.Context, client *http.Client, url string) (string, error) {
req, err := http.NewRequestWithContext(ctx, "GET", url, nil)
if err != nil {
return "", err
}
resp, err := client.Do(req)
if err != nil {
return "", err
}
defer func() { _ = resp.Body.Close() }()
if resp.StatusCode != http.StatusOK {
return "", fmt.Errorf("unexpected local stub status: %s", resp.Status)
}
data, err := io.ReadAll(resp.Body)
if err != nil {
return "", err
}
return string(data), nil
}
func parseStubStatusActiveConnections(raw string) int64 {
matches := stubStatusActivePattern.FindStringSubmatch(raw)
if len(matches) != stubStatusActiveMatchGroupCount {
return 0
}
value, err := strconv.ParseInt(matches[1], 10, 64)
if err != nil {
return 0
}
return value
}
func normalizeCountMap(values map[string]int64) map[string]int64 {
if len(values) == 0 {
return map[string]int64{}
}
result := make(map[string]int64, len(values))
for key, value := range values {
key = strings.TrimSpace(key)
if key == "" || value <= 0 {
continue
}
result[key] = value
}
return result
}
@@ -2,82 +2,63 @@ package observability
import ( import (
"context" "context"
"net"
"net/http" "net/http"
"net/http/httptest" "net/http/httptest"
"strings"
"testing" "testing"
"github.com/Rain-kl/Wavelet/internal/apps/agent/config" "github.com/Rain-kl/Wavelet/internal/apps/agent/config"
) )
func TestCollectManagedOpenRestyMetrics(t *testing.T) { func TestCollectEdgeHealth(t *testing.T) {
listener, err := net.Listen("tcp", "127.0.0.1:0") server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
if err != nil { if r.URL.Path != "/openflare/observability" {
t.Fatalf("Listen failed: %v", err) http.NotFound(w, r)
} return
port := listener.Addr().(*net.TCPAddr).Port }
_, _ = w.Write([]byte(`{"ok":true,"captured_at_unix":1710403200,"connections":{"active":12,"reading":1,"writing":2,"waiting":9}}`))
mux := http.NewServeMux() }))
mux.HandleFunc(openRestyObservabilityPath, func(writer http.ResponseWriter, request *http.Request) {
writer.Header().Set("Content-Type", "application/json")
_, _ = writer.Write([]byte(`{"window_started_at_unix":1710403200,"window_ended_at_unix":1710403210,"request_count":12,"error_count":2,"unique_visitor_count":5,"status_codes":{"200":10,"502":2},"top_domains":{"app.example.com":9,"api.example.com":3},"source_countries":{},"openresty_rx_bytes":4096,"openresty_tx_bytes":8192}`))
})
mux.HandleFunc(openRestyStubStatusPath, func(writer http.ResponseWriter, request *http.Request) {
_, _ = writer.Write([]byte("Active connections: 7 \nserver accepts handled requests\n 10 10 12 \nReading: 1 Writing: 2 Waiting: 4 \n"))
})
server := httptest.NewUnstartedServer(mux)
server.Listener = listener
server.Start()
defer server.Close() defer server.Close()
metrics := CollectManagedOpenRestyMetrics(context.Background(), &config.Config{ health := CollectEdgeHealth(context.Background(), &config.Config{
OpenrestyObservabilityPort: port, OpenrestyObservabilityPort: mustPort(server.URL),
}) })
if metrics == nil || metrics.TrafficReport == nil { if health == nil {
t.Fatalf("expected managed openresty metrics, got %+v", metrics) t.Fatal("expected edge health")
} }
if metrics.TrafficReport.RequestCount != 12 || metrics.TrafficReport.ErrorCount != 2 { if !health.OK || health.Connections != 12 {
t.Fatalf("unexpected traffic report: %+v", metrics.TrafficReport) t.Fatalf("unexpected health: %+v", health)
}
if metrics.OpenrestyRxBytes != 4096 || metrics.OpenrestyTxBytes != 8192 {
t.Fatalf("unexpected openresty byte counters: %+v", metrics)
}
if metrics.OpenrestyConnections != 7 {
t.Fatalf("unexpected openresty connections: %+v", metrics)
} }
} }
func TestParseStubStatusActiveConnections(t *testing.T) { func mustPort(rawURL string) int {
if value := parseStubStatusActiveConnections("Active connections: 19\n"); value != 19 { u := rawURL
t.Fatalf("unexpected active connections: %d", value) idx := stringsLastColon(u)
if idx < 0 {
return 0
} }
var port int
for _, ch := range u[idx+1:] {
if ch < '0' || ch > '9' {
break
}
port = port*10 + int(ch-'0')
}
return port
} }
func TestNormalizeCountMapDropsEmptyKeys(t *testing.T) { func stringsLastColon(s string) int {
normalized := normalizeCountMap(map[string]int64{ for i := len(s) - 1; i >= 0; i-- {
"": 4, if s[i] == ':' {
" 200 ": 3, return i
"app.example.com": 0, }
})
if len(normalized) != 1 || normalized["200"] != 3 {
t.Fatalf("unexpected normalized map: %+v", normalized)
} }
return -1
} }
func TestCollectManagedOpenRestyMetricsHandlesUnavailableEndpoint(t *testing.T) { func TestCollectEdgeHealthHandlesUnavailableEndpoint(t *testing.T) {
cfg := &config.Config{OpenrestyObservabilityPort: 1} if health := CollectEdgeHealth(context.Background(), &config.Config{
if metrics := CollectManagedOpenRestyMetrics(context.Background(), cfg); metrics != nil { OpenrestyObservabilityPort: 1,
t.Fatalf("expected nil metrics for unavailable endpoint, got %+v", metrics) }); health != nil {
} t.Fatalf("expected nil health, got %+v", health)
}
func TestOpenRestyObservabilityPathsAreStable(t *testing.T) {
if !strings.HasPrefix(openRestyObservabilityPath, "/openflare/") {
t.Fatalf("unexpected observability path: %s", openRestyObservabilityPath)
}
if !strings.HasPrefix(openRestyStubStatusPath, "/openflare/") {
t.Fatalf("unexpected stub status path: %s", openRestyStubStatusPath)
} }
} }
+35 -145
View File
@@ -6,10 +6,8 @@ import (
"errors" "errors"
"io" "io"
"log/slog" "log/slog"
"net/http"
"os" "os"
"regexp" "regexp"
"sort"
"strconv" "strconv"
"strings" "strings"
"time" "time"
@@ -20,63 +18,42 @@ import (
) )
type accessLogRecord struct { type accessLogRecord struct {
Timestamp string `json:"ts"` Timestamp string `json:"ts"`
Host string `json:"host"` Host string `json:"host"`
RemoteAddr string `json:"remote_addr"` RemoteAddr string `json:"remote_addr"`
Path string `json:"path"` Path string `json:"path"`
Status int `json:"status"` Status int `json:"status"`
BytesSent int64 `json:"bytes_sent"` BytesSent int64 `json:"bytes_sent"`
RequestLength int64 `json:"request_length"` RequestLength int64 `json:"request_length"`
RequestTime float64 `json:"request_time"`
} }
const ( const (
combinedAccessLogMatchGroupCount = 5 combinedAccessLogMatchGroupCount = 5
trafficTopDomainsLimit = 8 // requestTimeSecondsToMs converts OpenResty $request_time (seconds float) to ms.
requestTimeSecondsToMs = 1000.0
// roundHalfUp is added before int64 truncate to round to nearest millisecond.
roundHalfUp = 0.5
) )
var combinedAccessLogPattern = regexp.MustCompile(`^(\S+)\s+\S+\s+\S+\s+\[([^]]+)]\s+"\S+\s+(\S+)(?:\s+[^"]*)?"\s+(\d{3})\s+\S+`) var combinedAccessLogPattern = regexp.MustCompile(`^(\S+)\s+\S+\s+\S+\s+\[([^]]+)]\s+"\S+\s+(\S+)(?:\s+[^"]*)?"\s+(\d{3})\s+\S+`)
// trafficAggregate collects access-log facts for the current heartbeat window.
// Pre-aggregation (UV/TopN/TrafficReport) is intentionally not built.
type trafficAggregate struct { type trafficAggregate struct {
windowStartedAt time.Time logs []protocol.NodeAccessLog
windowEndedAt time.Time
requestCount int64
errorCount int64
openrestyRxBytes int64
openrestyTxBytes int64
statusCodes map[string]int64
topDomains map[string]int64
visitors map[string]struct{}
logs []protocol.NodeAccessLog
} }
// BuildTrafficReport generates a traffic report using access logs or falling back to managed metrics. // CollectAccessLogs tails access.log and returns L1 fact rows for the current heartbeat.
func BuildTrafficReport(cfg *config.Config, stateStore *state.Store, managed *ManagedOpenRestyMetrics) *protocol.NodeTrafficReport { func CollectAccessLogs(cfg *config.Config, stateStore *state.Store) []protocol.NodeAccessLog {
report, _, _ := BuildTrafficObservability(cfg, stateStore, managed)
return report
}
// BuildTrafficObservability returns the traffic report, parsed access logs, and managed metrics.
func BuildTrafficObservability(cfg *config.Config, stateStore *state.Store, managed *ManagedOpenRestyMetrics) (*protocol.NodeTrafficReport, []protocol.NodeAccessLog, *ManagedOpenRestyMetrics) {
if cfg == nil || stateStore == nil { if cfg == nil || stateStore == nil {
if managed != nil && managed.TrafficReport != nil { return nil
return managed.TrafficReport, nil, managed
}
return nil, nil, managed
} }
aggregate := readAccessLogDelta(cfg, stateStore) aggregate := readAccessLogDelta(cfg, stateStore)
var accessLogs []protocol.NodeAccessLog
if aggregate != nil {
accessLogs = aggregate.accessLogs()
}
if managed != nil && managed.TrafficReport != nil {
return managed.TrafficReport, accessLogs, managed
}
if aggregate == nil { if aggregate == nil {
return nil, accessLogs, managed return nil
} }
fallbackManaged := aggregate.managedMetrics() return aggregate.accessLogs()
return aggregate.report(), accessLogs, fallbackManaged
} }
func readAccessLogDelta(cfg *config.Config, stateStore *state.Store) *trafficAggregate { func readAccessLogDelta(cfg *config.Config, stateStore *state.Store) *trafficAggregate {
@@ -149,11 +126,7 @@ func managedAccessLogPath(cfg *config.Config) string {
} }
func newTrafficAggregate() *trafficAggregate { func newTrafficAggregate() *trafficAggregate {
return &trafficAggregate{ return &trafficAggregate{}
statusCodes: make(map[string]int64),
topDomains: make(map[string]int64),
visitors: make(map[string]struct{}),
}
} }
func (aggregate *trafficAggregate) consume(line []byte) { func (aggregate *trafficAggregate) consume(line []byte) {
@@ -167,39 +140,15 @@ func (aggregate *trafficAggregate) consume(line []byte) {
return return
} }
if aggregate.windowStartedAt.IsZero() || record.Timestamp.Before(aggregate.windowStartedAt) {
aggregate.windowStartedAt = record.Timestamp
}
if aggregate.windowEndedAt.IsZero() || record.Timestamp.After(aggregate.windowEndedAt) {
aggregate.windowEndedAt = record.Timestamp
}
aggregate.requestCount++
if record.Status >= http.StatusInternalServerError {
aggregate.errorCount++
}
if record.Status > 0 {
aggregate.statusCodes[strconv.Itoa(record.Status)]++
}
if record.RequestLength > 0 {
aggregate.openrestyRxBytes += record.RequestLength
}
if record.BytesSent > 0 {
aggregate.openrestyTxBytes += record.BytesSent
}
if host := strings.TrimSpace(record.Host); host != "" {
aggregate.topDomains[host]++
}
if remoteAddr := strings.TrimSpace(record.RemoteAddr); remoteAddr != "" {
aggregate.visitors[remoteAddr] = struct{}{}
}
aggregate.logs = append(aggregate.logs, protocol.NodeAccessLog{ aggregate.logs = append(aggregate.logs, protocol.NodeAccessLog{
LoggedAtUnix: record.Timestamp.Unix(), LoggedAtUnix: record.Timestamp.Unix(),
RemoteAddr: strings.TrimSpace(record.RemoteAddr), RemoteAddr: strings.TrimSpace(record.RemoteAddr),
Host: strings.TrimSpace(record.Host), Host: strings.TrimSpace(record.Host),
Path: normalizeAccessLogPath(record.Path), Path: normalizeAccessLogPath(record.Path),
StatusCode: record.Status, StatusCode: record.Status,
BytesSent: record.BytesSent, BytesSent: record.BytesSent,
RequestLength: record.RequestLength,
RequestTimeMs: record.RequestTimeMs,
}) })
} }
@@ -211,6 +160,7 @@ type parsedAccessLogRecord struct {
Status int Status int
BytesSent int64 BytesSent int64
RequestLength int64 RequestLength int64
RequestTimeMs int64
} }
func parseAccessLogRecord(raw string) (parsedAccessLogRecord, bool) { func parseAccessLogRecord(raw string) (parsedAccessLogRecord, bool) {
@@ -230,6 +180,10 @@ func parseJSONAccessLogRecord(raw string) (parsedAccessLogRecord, bool) {
if err != nil { if err != nil {
return parsedAccessLogRecord{}, false return parsedAccessLogRecord{}, false
} }
requestTimeMs := int64(0)
if record.RequestTime > 0 {
requestTimeMs = int64(record.RequestTime*requestTimeSecondsToMs + roundHalfUp)
}
return parsedAccessLogRecord{ return parsedAccessLogRecord{
Timestamp: timestamp, Timestamp: timestamp,
Host: strings.TrimSpace(record.Host), Host: strings.TrimSpace(record.Host),
@@ -238,6 +192,7 @@ func parseJSONAccessLogRecord(raw string) (parsedAccessLogRecord, bool) {
Status: record.Status, Status: record.Status,
BytesSent: record.BytesSent, BytesSent: record.BytesSent,
RequestLength: record.RequestLength, RequestLength: record.RequestLength,
RequestTimeMs: requestTimeMs,
}, true }, true
} }
@@ -262,23 +217,6 @@ func parseCombinedAccessLogRecord(raw string) (parsedAccessLogRecord, bool) {
}, true }, true
} }
func (aggregate *trafficAggregate) report() *protocol.NodeTrafficReport {
if aggregate.requestCount == 0 || aggregate.windowStartedAt.IsZero() || aggregate.windowEndedAt.IsZero() {
return nil
}
return &protocol.NodeTrafficReport{
WindowStartedAtUnix: aggregate.windowStartedAt.Unix(),
WindowEndedAtUnix: aggregate.windowEndedAt.Unix(),
RequestCount: aggregate.requestCount,
ErrorCount: aggregate.errorCount,
UniqueVisitorCount: int64(len(aggregate.visitors)),
StatusCodes: cloneTrafficCounts(aggregate.statusCodes, 0),
TopDomains: topCounts(aggregate.topDomains, trafficTopDomainsLimit),
SourceCountries: map[string]int64{},
}
}
func (aggregate *trafficAggregate) accessLogs() []protocol.NodeAccessLog { func (aggregate *trafficAggregate) accessLogs() []protocol.NodeAccessLog {
if aggregate == nil || len(aggregate.logs) == 0 { if aggregate == nil || len(aggregate.logs) == 0 {
return []protocol.NodeAccessLog{} return []protocol.NodeAccessLog{}
@@ -286,21 +224,6 @@ func (aggregate *trafficAggregate) accessLogs() []protocol.NodeAccessLog {
return append([]protocol.NodeAccessLog(nil), aggregate.logs...) return append([]protocol.NodeAccessLog(nil), aggregate.logs...)
} }
func (aggregate *trafficAggregate) managedMetrics() *ManagedOpenRestyMetrics {
if aggregate == nil {
return nil
}
report := aggregate.report()
if report == nil && aggregate.openrestyRxBytes <= 0 && aggregate.openrestyTxBytes <= 0 {
return nil
}
return &ManagedOpenRestyMetrics{
TrafficReport: report,
OpenrestyRxBytes: aggregate.openrestyRxBytes,
OpenrestyTxBytes: aggregate.openrestyTxBytes,
}
}
func parseAccessLogTime(value string) (time.Time, error) { func parseAccessLogTime(value string) (time.Time, error) {
trimmed := strings.TrimSpace(value) trimmed := strings.TrimSpace(value)
if trimmed == "" { if trimmed == "" {
@@ -313,35 +236,6 @@ func parseAccessLogTime(value string) (time.Time, error) {
return time.Parse("02/Jan/2006:15:04:05 -0700", trimmed) return time.Parse("02/Jan/2006:15:04:05 -0700", trimmed)
} }
func cloneTrafficCounts(values map[string]int64, limit int) map[string]int64 {
if len(values) == 0 {
return map[string]int64{}
}
items := make([]trafficCountItem, 0, len(values))
for key, value := range values {
items = append(items, trafficCountItem{key: key, value: value})
}
sort.Slice(items, func(i int, j int) bool {
if items[i].value == items[j].value {
return items[i].key < items[j].key
}
return items[i].value > items[j].value
})
if limit > 0 && len(items) > limit {
items = items[:limit]
}
result := make(map[string]int64, len(items))
for _, item := range items {
result[item.key] = item.value
}
return result
}
type trafficCountItem struct {
key string
value int64
}
const accessLogPathMaxRunes = 100 const accessLogPathMaxRunes = 100
func normalizeAccessLogPath(value string) string { func normalizeAccessLogPath(value string) string {
@@ -365,7 +259,3 @@ func truncateAccessLogPath(value string) string {
} }
return string(runes[:accessLogPathMaxRunes]) return string(runes[:accessLogPathMaxRunes])
} }
func topCounts(values map[string]int64, limit int) map[string]int64 {
return cloneTrafficCounts(values, limit)
}
+25 -123
View File
@@ -7,39 +7,33 @@ import (
"testing" "testing"
"github.com/Rain-kl/Wavelet/internal/apps/agent/config" "github.com/Rain-kl/Wavelet/internal/apps/agent/config"
"github.com/Rain-kl/Wavelet/internal/apps/agent/protocol"
"github.com/Rain-kl/Wavelet/internal/apps/agent/state" "github.com/Rain-kl/Wavelet/internal/apps/agent/state"
) )
func TestBuildTrafficReportAggregatesManagedAccessLog(t *testing.T) { func TestCollectAccessLogsReturnsFactsOnly(t *testing.T) {
tempDir := t.TempDir() tempDir := t.TempDir()
routeConfigPath := filepath.Join(tempDir, "conf.d", "openflare_routes.conf") logPath := filepath.Join(tempDir, "openflare_access.log")
if err := os.MkdirAll(filepath.Dir(routeConfigPath), 0o755); err != nil {
t.Fatalf("MkdirAll failed: %v", err)
}
logPath := filepath.Join(filepath.Dir(routeConfigPath), "openflare_access.log")
content := []byte( content := []byte(
"{\"ts\":\"2026-03-14T08:00:00Z\",\"host\":\"app.example.com\",\"path\":\"/\",\"remote_addr\":\"10.0.0.1\",\"status\":200}\n" + "{\"ts\":\"2026-03-14T08:00:00Z\",\"host\":\"app.example.com\",\"path\":\"/login\",\"remote_addr\":\"10.0.0.1\",\"status\":200,\"request_length\":128,\"bytes_sent\":512,\"request_time\":0.015}\n" +
"{\"ts\":\"2026-03-14T08:00:05Z\",\"host\":\"app.example.com\",\"path\":\"/healthz\",\"remote_addr\":\"10.0.0.2\",\"status\":503}\n" + "{\"ts\":\"2026-03-14T08:00:05Z\",\"host\":\"api.example.com\",\"path\":\"/v1/ping\",\"remote_addr\":\"10.0.0.2\",\"status\":502,\"request_length\":64,\"bytes_sent\":256,\"request_time\":0.008}\n",
"{\"ts\":\"2026-03-14T08:00:08Z\",\"host\":\"api.example.com\",\"path\":\"/api\",\"remote_addr\":\"10.0.0.1\",\"status\":200}\n",
) )
if err := os.WriteFile(logPath, content, 0o644); err != nil { if err := os.WriteFile(logPath, content, 0o644); err != nil {
t.Fatalf("WriteFile failed: %v", err) t.Fatalf("WriteFile failed: %v", err)
} }
stateStore := state.NewStore(filepath.Join(tempDir, "state.json")) stateStore := state.NewStore(filepath.Join(tempDir, "state.json"))
report := BuildTrafficReport(&config.Config{AccessLogPath: logPath}, stateStore, nil) accessLogs := CollectAccessLogs(&config.Config{AccessLogPath: logPath}, stateStore)
if report == nil { if len(accessLogs) != 2 {
t.Fatal("expected traffic report") t.Fatalf("expected access logs, got %+v", accessLogs)
} }
if report.RequestCount != 3 || report.ErrorCount != 1 || report.UniqueVisitorCount != 2 { if accessLogs[0].BytesSent != 512 || accessLogs[0].RequestLength != 128 {
t.Fatalf("unexpected traffic report counters: %+v", report) t.Fatalf("unexpected first log: %+v", accessLogs[0])
} }
if report.StatusCodes["200"] != 2 || report.StatusCodes["503"] != 1 { if accessLogs[0].RequestTimeMs != 15 {
t.Fatalf("unexpected status codes: %+v", report.StatusCodes) t.Fatalf("request_time_ms = %d, want 15", accessLogs[0].RequestTimeMs)
} }
if report.TopDomains["app.example.com"] != 2 || report.TopDomains["api.example.com"] != 1 { if accessLogs[0].Path != "/login" || accessLogs[1].Path != "/v1/ping" {
t.Fatalf("unexpected top domains: %+v", report.TopDomains) t.Fatalf("unexpected access log paths: %+v", accessLogs)
} }
snapshot, err := stateStore.Load() snapshot, err := stateStore.Load()
@@ -50,20 +44,16 @@ func TestBuildTrafficReportAggregatesManagedAccessLog(t *testing.T) {
t.Fatalf("unexpected access log offset: %d", snapshot.AccessLogOffset) t.Fatalf("unexpected access log offset: %d", snapshot.AccessLogOffset)
} }
secondReport := BuildTrafficReport(&config.Config{AccessLogPath: logPath}, stateStore, nil) moreLogs := CollectAccessLogs(&config.Config{AccessLogPath: logPath}, stateStore)
if secondReport != nil { if len(moreLogs) != 0 {
t.Fatalf("expected no report without appended lines, got %+v", secondReport) t.Fatalf("expected no new logs, got %+v", moreLogs)
} }
} }
func TestBuildTrafficReportResetsOffsetAfterTruncate(t *testing.T) { func TestCollectAccessLogsResetsOffsetAfterTruncate(t *testing.T) {
tempDir := t.TempDir() tempDir := t.TempDir()
routeConfigPath := filepath.Join(tempDir, "conf.d", "openflare_routes.conf") logPath := filepath.Join(tempDir, "openflare_access.log")
if err := os.MkdirAll(filepath.Dir(routeConfigPath), 0o755); err != nil { if err := os.WriteFile(logPath, []byte("{\"ts\":\"2026-03-14T09:00:00Z\",\"host\":\"app.example.com\",\"path\":\"/\",\"remote_addr\":\"10.0.0.3\",\"status\":200,\"bytes_sent\":1}\n"), 0o644); err != nil {
t.Fatalf("MkdirAll failed: %v", err)
}
logPath := filepath.Join(filepath.Dir(routeConfigPath), "openflare_access.log")
if err := os.WriteFile(logPath, []byte("{\"ts\":\"2026-03-14T09:00:00Z\",\"host\":\"app.example.com\",\"path\":\"/\",\"remote_addr\":\"10.0.0.3\",\"status\":200}\n"), 0o644); err != nil {
t.Fatalf("WriteFile failed: %v", err) t.Fatalf("WriteFile failed: %v", err)
} }
@@ -72,50 +62,15 @@ func TestBuildTrafficReportResetsOffsetAfterTruncate(t *testing.T) {
t.Fatalf("Save failed: %v", err) t.Fatalf("Save failed: %v", err)
} }
report := BuildTrafficReport(&config.Config{AccessLogPath: logPath}, stateStore, nil) accessLogs := CollectAccessLogs(&config.Config{AccessLogPath: logPath}, stateStore)
if report == nil || report.RequestCount != 1 { if len(accessLogs) != 1 {
t.Fatalf("expected one request after truncate reset, got %+v", report) t.Fatalf("expected one access log after truncate reset, got %+v", accessLogs)
} }
} }
func TestBuildTrafficObservabilityReturnsAccessLogs(t *testing.T) { func TestCollectAccessLogsTruncatesLongAccessLogPath(t *testing.T) {
tempDir := t.TempDir() tempDir := t.TempDir()
routeConfigPath := filepath.Join(tempDir, "conf.d", "openflare_routes.conf") logPath := filepath.Join(tempDir, "openflare_access.log")
if err := os.MkdirAll(filepath.Dir(routeConfigPath), 0o755); err != nil {
t.Fatalf("MkdirAll failed: %v", err)
}
logPath := filepath.Join(filepath.Dir(routeConfigPath), "openflare_access.log")
content := []byte(
"{\"ts\":\"2026-03-14T08:00:00Z\",\"host\":\"app.example.com\",\"path\":\"/login\",\"remote_addr\":\"10.0.0.1\",\"status\":200,\"request_length\":128,\"bytes_sent\":512}\n" +
"{\"ts\":\"2026-03-14T08:00:05Z\",\"host\":\"api.example.com\",\"path\":\"/v1/ping\",\"remote_addr\":\"10.0.0.2\",\"status\":502,\"request_length\":64,\"bytes_sent\":256}\n",
)
if err := os.WriteFile(logPath, content, 0o644); err != nil {
t.Fatalf("WriteFile failed: %v", err)
}
stateStore := state.NewStore(filepath.Join(tempDir, "state.json"))
report, accessLogs, fallbackMetrics := BuildTrafficObservability(&config.Config{AccessLogPath: logPath}, stateStore, nil)
if report == nil || report.RequestCount != 2 {
t.Fatalf("expected traffic report, got %+v", report)
}
if len(accessLogs) != 2 {
t.Fatalf("expected access logs, got %+v", accessLogs)
}
if fallbackMetrics == nil || fallbackMetrics.OpenrestyRxBytes != 192 || fallbackMetrics.OpenrestyTxBytes != 768 {
t.Fatalf("expected fallback throughput metrics, got %+v", fallbackMetrics)
}
if accessLogs[0].Path != "/login" || accessLogs[1].Path != "/v1/ping" {
t.Fatalf("unexpected access log paths: %+v", accessLogs)
}
}
func TestBuildTrafficObservabilityTruncatesLongAccessLogPath(t *testing.T) {
tempDir := t.TempDir()
routeConfigPath := filepath.Join(tempDir, "conf.d", "openflare_routes.conf")
if err := os.MkdirAll(filepath.Dir(routeConfigPath), 0o755); err != nil {
t.Fatalf("MkdirAll failed: %v", err)
}
logPath := filepath.Join(filepath.Dir(routeConfigPath), "openflare_access.log")
longPath := "/" + strings.Repeat("a", 140) longPath := "/" + strings.Repeat("a", 140)
content := []byte( content := []byte(
"{\"ts\":\"2026-03-14T08:00:00Z\",\"host\":\"app.example.com\",\"path\":\"" + longPath + "\",\"remote_addr\":\"10.0.0.1\",\"status\":200}\n", "{\"ts\":\"2026-03-14T08:00:00Z\",\"host\":\"app.example.com\",\"path\":\"" + longPath + "\",\"remote_addr\":\"10.0.0.1\",\"status\":200}\n",
@@ -125,7 +80,7 @@ func TestBuildTrafficObservabilityTruncatesLongAccessLogPath(t *testing.T) {
} }
stateStore := state.NewStore(filepath.Join(tempDir, "state.json")) stateStore := state.NewStore(filepath.Join(tempDir, "state.json"))
_, accessLogs, _ := BuildTrafficObservability(&config.Config{AccessLogPath: logPath}, stateStore, nil) accessLogs := CollectAccessLogs(&config.Config{AccessLogPath: logPath}, stateStore)
if len(accessLogs) != 1 { if len(accessLogs) != 1 {
t.Fatalf("expected one access log, got %+v", accessLogs) t.Fatalf("expected one access log, got %+v", accessLogs)
} }
@@ -133,56 +88,3 @@ func TestBuildTrafficObservabilityTruncatesLongAccessLogPath(t *testing.T) {
t.Fatalf("expected truncated path length %d, got %d (%q)", accessLogPathMaxRunes, got, accessLogs[0].Path) t.Fatalf("expected truncated path length %d, got %d (%q)", accessLogPathMaxRunes, got, accessLogs[0].Path)
} }
} }
func TestBuildTrafficReportParsesCombinedAccessLog(t *testing.T) {
tempDir := t.TempDir()
routeConfigPath := filepath.Join(tempDir, "conf.d", "openflare_routes.conf")
if err := os.MkdirAll(filepath.Dir(routeConfigPath), 0o755); err != nil {
t.Fatalf("MkdirAll failed: %v", err)
}
logPath := filepath.Join(filepath.Dir(routeConfigPath), "openflare_access.log")
content := []byte(
"10.0.0.1 - - [14/Mar/2026:08:00:00 +0000] \"GET / HTTP/1.1\" 200 123 \"-\" \"curl/8.0\"\n" +
"10.0.0.2 - - [14/Mar/2026:08:00:05 +0000] \"GET /healthz HTTP/1.1\" 502 64 \"-\" \"curl/8.0\"\n" +
"10.0.0.1 - - [14/Mar/2026:08:00:10 +0000] \"GET /api HTTP/1.1\" 200 256 \"-\" \"curl/8.0\"\n",
)
if err := os.WriteFile(logPath, content, 0o644); err != nil {
t.Fatalf("WriteFile failed: %v", err)
}
stateStore := state.NewStore(filepath.Join(tempDir, "state.json"))
report := BuildTrafficReport(&config.Config{AccessLogPath: logPath}, stateStore, nil)
if report == nil {
t.Fatal("expected traffic report from combined access log")
}
if report.RequestCount != 3 || report.ErrorCount != 1 || report.UniqueVisitorCount != 2 {
t.Fatalf("unexpected combined log counters: %+v", report)
}
if report.StatusCodes["200"] != 2 || report.StatusCodes["502"] != 1 {
t.Fatalf("unexpected combined log status codes: %+v", report.StatusCodes)
}
if len(report.TopDomains) != 0 {
t.Fatalf("expected combined access log to omit top domains when host is unavailable, got %+v", report.TopDomains)
}
}
func TestBuildTrafficReportReturnsManagedWindowEvenWhenRequestCountZero(t *testing.T) {
report := BuildTrafficReport(nil, nil, &ManagedOpenRestyMetrics{
TrafficReport: &protocol.NodeTrafficReport{
WindowStartedAtUnix: 1710403200,
WindowEndedAtUnix: 1710403260,
RequestCount: 0,
ErrorCount: 0,
UniqueVisitorCount: 0,
StatusCodes: map[string]int64{},
TopDomains: map[string]int64{},
SourceCountries: map[string]int64{},
},
})
if report == nil {
t.Fatal("expected managed traffic report to be returned even when request count is zero")
}
if report.RequestCount != 0 || report.WindowStartedAtUnix != 1710403200 || report.WindowEndedAtUnix != 1710403260 {
t.Fatalf("unexpected managed traffic report: %+v", report)
}
}
+2 -5
View File
@@ -33,11 +33,8 @@ type NodeSystemProfile = pkgprotocol.NodeSystemProfile
// NodeMetricSnapshot is an alias for pkgprotocol.NodeMetricSnapshot. // NodeMetricSnapshot is an alias for pkgprotocol.NodeMetricSnapshot.
type NodeMetricSnapshot = pkgprotocol.NodeMetricSnapshot type NodeMetricSnapshot = pkgprotocol.NodeMetricSnapshot
// NodeOpenrestyObservation is an alias for pkgprotocol.NodeOpenrestyObservation. // NodeEdgeHealth is an alias for pkgprotocol.NodeEdgeHealth.
type NodeOpenrestyObservation = pkgprotocol.NodeOpenrestyObservation type NodeEdgeHealth = pkgprotocol.NodeEdgeHealth
// NodeTrafficReport is an alias for pkgprotocol.NodeTrafficReport.
type NodeTrafficReport = pkgprotocol.NodeTrafficReport
// NodeAccessLog is an alias for pkgprotocol.NodeAccessLog. // NodeAccessLog is an alias for pkgprotocol.NodeAccessLog.
type NodeAccessLog = pkgprotocol.NodeAccessLog type NodeAccessLog = pkgprotocol.NodeAccessLog
@@ -14,14 +14,13 @@ import (
const observabilityBufferWindowSeconds = 60 const observabilityBufferWindowSeconds = 60
// ObservabilityBufferRecord stores observability data for a single time window. // ObservabilityBufferRecord stores observability facts for a single time window.
type ObservabilityBufferRecord struct { type ObservabilityBufferRecord struct {
WindowStartedAtUnix int64 `json:"window_started_at_unix"` WindowStartedAtUnix int64 `json:"window_started_at_unix"`
Snapshot *protocol.NodeMetricSnapshot `json:"snapshot,omitempty"` HostMetrics *protocol.NodeMetricSnapshot `json:"host_metrics,omitempty"`
OpenrestyObservation *protocol.NodeOpenrestyObservation `json:"openresty_observation,omitempty"` EdgeHealth *protocol.NodeEdgeHealth `json:"edge_health,omitempty"`
TrafficReport *protocol.NodeTrafficReport `json:"traffic_report,omitempty"` AccessLogs []protocol.NodeAccessLog `json:"access_logs,omitempty"`
AccessLogs []protocol.NodeAccessLog `json:"access_logs,omitempty"` QueuedAtUnix int64 `json:"queued_at_unix"`
QueuedAtUnix int64 `json:"queued_at_unix"`
} }
// ObservabilityBufferStore persists observability records to disk for replay on heartbeat. // ObservabilityBufferStore persists observability records to disk for replay on heartbeat.
@@ -39,7 +38,7 @@ func NewObservabilityBufferStore(path string) *ObservabilityBufferStore {
// Upsert inserts or merges an observability record and prunes entries older than retainAfterUnix. // Upsert inserts or merges an observability record and prunes entries older than retainAfterUnix.
func (s *ObservabilityBufferStore) Upsert(record ObservabilityBufferRecord, retainAfterUnix int64) error { func (s *ObservabilityBufferStore) Upsert(record ObservabilityBufferRecord, retainAfterUnix int64) error {
if s == nil || record.WindowStartedAtUnix <= 0 || (record.Snapshot == nil && record.OpenrestyObservation == nil && record.TrafficReport == nil && len(record.AccessLogs) == 0) { if s == nil || record.WindowStartedAtUnix <= 0 || (record.HostMetrics == nil && record.EdgeHealth == nil && len(record.AccessLogs) == 0) {
return nil return nil
} }
s.mu.Lock() s.mu.Lock()
@@ -70,14 +69,11 @@ func (s *ObservabilityBufferStore) Upsert(record ObservabilityBufferRecord, reta
func mergeObservabilityBufferRecord(existing ObservabilityBufferRecord, incoming ObservabilityBufferRecord) ObservabilityBufferRecord { func mergeObservabilityBufferRecord(existing ObservabilityBufferRecord, incoming ObservabilityBufferRecord) ObservabilityBufferRecord {
merged := existing merged := existing
if incoming.Snapshot != nil { if incoming.HostMetrics != nil {
merged.Snapshot = incoming.Snapshot merged.HostMetrics = incoming.HostMetrics
} }
if incoming.OpenrestyObservation != nil { if incoming.EdgeHealth != nil {
merged.OpenrestyObservation = incoming.OpenrestyObservation merged.EdgeHealth = incoming.EdgeHealth
}
if incoming.TrafficReport != nil {
merged.TrafficReport = incoming.TrafficReport
} }
merged.AccessLogs = mergeAccessLogs(existing.AccessLogs, incoming.AccessLogs) merged.AccessLogs = mergeAccessLogs(existing.AccessLogs, incoming.AccessLogs)
if incoming.QueuedAtUnix > 0 { if incoming.QueuedAtUnix > 0 {
@@ -222,18 +218,15 @@ func (s *ObservabilityBufferStore) saveUnlocked(records []ObservabilityBufferRec
return nil return nil
} }
// ObservabilityWindowStartedAt calculates the start of the 60-second window for the given metrics, openresty observation, or traffic report. // ObservabilityWindowStartedAt returns the 60s window start for host metrics or edge health.
func ObservabilityWindowStartedAt(snapshot *protocol.NodeMetricSnapshot, openresty *protocol.NodeOpenrestyObservation, traffic *protocol.NodeTrafficReport) int64 { func ObservabilityWindowStartedAt(hostMetrics *protocol.NodeMetricSnapshot, edgeHealth *protocol.NodeEdgeHealth) int64 {
if traffic != nil && traffic.WindowStartedAtUnix > 0 { if edgeHealth != nil && edgeHealth.CapturedAtUnix > 0 {
return traffic.WindowStartedAtUnix - (traffic.WindowStartedAtUnix % observabilityBufferWindowSeconds) return edgeHealth.CapturedAtUnix - (edgeHealth.CapturedAtUnix % observabilityBufferWindowSeconds)
} }
if openresty != nil && openresty.CapturedAtUnix > 0 { if hostMetrics == nil || hostMetrics.CapturedAtUnix <= 0 {
return openresty.CapturedAtUnix - (openresty.CapturedAtUnix % observabilityBufferWindowSeconds)
}
if snapshot == nil || snapshot.CapturedAtUnix <= 0 {
return 0 return 0
} }
return snapshot.CapturedAtUnix - (snapshot.CapturedAtUnix % observabilityBufferWindowSeconds) return hostMetrics.CapturedAtUnix - (hostMetrics.CapturedAtUnix % observabilityBufferWindowSeconds)
} }
func pruneObservabilityBufferRecords(records []ObservabilityBufferRecord, retainAfterUnix int64) []ObservabilityBufferRecord { func pruneObservabilityBufferRecords(records []ObservabilityBufferRecord, retainAfterUnix int64) []ObservabilityBufferRecord {
@@ -12,24 +12,23 @@ func TestObservabilityBufferStoreUpsertReplayAndAck(t *testing.T) {
if err := store.Upsert(ObservabilityBufferRecord{ if err := store.Upsert(ObservabilityBufferRecord{
WindowStartedAtUnix: 1710403200, WindowStartedAtUnix: 1710403200,
Snapshot: &protocol.NodeMetricSnapshot{CapturedAtUnix: 1710403205}, HostMetrics: &protocol.NodeMetricSnapshot{CapturedAtUnix: 1710403205},
TrafficReport: &protocol.NodeTrafficReport{WindowStartedAtUnix: 1710403200, WindowEndedAtUnix: 1710403260, RequestCount: 5}, EdgeHealth: &protocol.NodeEdgeHealth{CapturedAtUnix: 1710403205, Connections: 5},
QueuedAtUnix: 1710403205, QueuedAtUnix: 1710403205,
}, 1710403000); err != nil { }, 1710403000); err != nil {
t.Fatalf("first upsert failed: %v", err) t.Fatalf("first upsert failed: %v", err)
} }
if err := store.Upsert(ObservabilityBufferRecord{ if err := store.Upsert(ObservabilityBufferRecord{
WindowStartedAtUnix: 1710403200, WindowStartedAtUnix: 1710403200,
Snapshot: &protocol.NodeMetricSnapshot{CapturedAtUnix: 1710403255}, HostMetrics: &protocol.NodeMetricSnapshot{CapturedAtUnix: 1710403255, CPUUsagePercent: 40},
TrafficReport: &protocol.NodeTrafficReport{WindowStartedAtUnix: 1710403200, WindowEndedAtUnix: 1710403260, RequestCount: 12}, EdgeHealth: &protocol.NodeEdgeHealth{CapturedAtUnix: 1710403255, Connections: 12},
QueuedAtUnix: 1710403255, QueuedAtUnix: 1710403255,
}, 1710403000); err != nil { }, 1710403000); err != nil {
t.Fatalf("second upsert failed: %v", err) t.Fatalf("second upsert failed: %v", err)
} }
if err := store.Upsert(ObservabilityBufferRecord{ if err := store.Upsert(ObservabilityBufferRecord{
WindowStartedAtUnix: 1710403260, WindowStartedAtUnix: 1710403260,
Snapshot: &protocol.NodeMetricSnapshot{CapturedAtUnix: 1710403265}, HostMetrics: &protocol.NodeMetricSnapshot{CapturedAtUnix: 1710403265},
TrafficReport: &protocol.NodeTrafficReport{WindowStartedAtUnix: 1710403260, WindowEndedAtUnix: 1710403320, RequestCount: 2},
QueuedAtUnix: 1710403265, QueuedAtUnix: 1710403265,
}, 1710403000); err != nil { }, 1710403000); err != nil {
t.Fatalf("third upsert failed: %v", err) t.Fatalf("third upsert failed: %v", err)
@@ -42,7 +41,7 @@ func TestObservabilityBufferStoreUpsertReplayAndAck(t *testing.T) {
if len(records) != 1 { if len(records) != 1 {
t.Fatalf("expected one replayable record before current window, got %d", len(records)) t.Fatalf("expected one replayable record before current window, got %d", len(records))
} }
if records[0].TrafficReport == nil || records[0].TrafficReport.RequestCount != 12 { if records[0].EdgeHealth == nil || records[0].EdgeHealth.Connections != 12 {
t.Fatalf("expected replayable record to keep latest upsert, got %+v", records[0]) t.Fatalf("expected replayable record to keep latest upsert, got %+v", records[0])
} }
@@ -89,10 +88,10 @@ func TestObservabilityBufferStoreMergesAccessLogsWithinWindow(t *testing.T) {
} }
func TestObservabilityWindowStartedAt(t *testing.T) { func TestObservabilityWindowStartedAt(t *testing.T) {
if value := ObservabilityWindowStartedAt(nil, nil, &protocol.NodeTrafficReport{WindowStartedAtUnix: 1710403200}); value != 1710403200 { if value := ObservabilityWindowStartedAt(nil, &protocol.NodeEdgeHealth{CapturedAtUnix: 1710403259}); value != 1710403200 {
t.Fatalf("unexpected traffic window start: %d", value) t.Fatalf("unexpected edge-health window start: %d", value)
} }
if value := ObservabilityWindowStartedAt(&protocol.NodeMetricSnapshot{CapturedAtUnix: 1710403259}, nil, nil); value != 1710403200 { if value := ObservabilityWindowStartedAt(&protocol.NodeMetricSnapshot{CapturedAtUnix: 1710403259}, nil); value != 1710403200 {
t.Fatalf("unexpected snapshot-derived window start: %d", value) t.Fatalf("unexpected host-metrics window start: %d", value)
} }
} }
+1 -1
View File
@@ -24,7 +24,7 @@ const (
randomTokenBytes = 16 randomTokenBytes = 16
maxDatabaseTextLength = 16000 maxDatabaseTextLength = 16000
defaultAgentHeartbeatInterval = 10000 // 默认心跳间隔 10 秒(毫秒) defaultAgentHeartbeatInterval = 3000 // 默认心跳间隔 3 秒(毫秒)
defaultAgentUpdateRepo = "Rain-kl/OpenFlare" defaultAgentUpdateRepo = "Rain-kl/OpenFlare"
) )
+52 -61
View File
@@ -6,7 +6,6 @@ package agent
import ( import (
"context" "context"
"encoding/json" "encoding/json"
"errors"
"log/slog" "log/slog"
"strings" "strings"
"time" "time"
@@ -28,16 +27,16 @@ const (
healthEventMessageMaxLength = 4096 healthEventMessageMaxLength = 4096
) )
// PersistHeartbeatObservability stores profile, snapshots, traffic, access logs, and health events. // PersistHeartbeatObservability stores profile, host metrics, edge health, and access logs.
func PersistHeartbeatObservability(ctx context.Context, nodeID string, payload NodePayload, reportedAt time.Time) { func PersistHeartbeatObservability(ctx context.Context, nodeID string, payload NodePayload, reportedAt time.Time) {
if strings.TrimSpace(nodeID) == "" { if strings.TrimSpace(nodeID) == "" {
return return
} }
if payload.Profile == nil && if payload.Profile == nil &&
payload.Snapshot == nil && payload.HostMetrics == nil &&
payload.TrafficReport == nil && payload.EdgeHealth == nil &&
len(payload.AccessLogs) == 0 && len(payload.AccessLogs) == 0 &&
len(payload.BufferedObservability) == 0 && len(payload.Buffered) == 0 &&
payload.HealthEvents == nil { payload.HealthEvents == nil {
return return
} }
@@ -47,7 +46,7 @@ func PersistHeartbeatObservability(ctx context.Context, nodeID string, payload N
return return
} }
accessLogRecords, err := buildNodeAccessLogRecords(nodeID, payload.AccessLogs, payload.BufferedObservability, reportedAt) accessLogRecords, err := buildNodeAccessLogRecords(nodeID, payload.AccessLogs, payload.Buffered, reportedAt)
if err != nil { if err != nil {
zap.L().Error("build heartbeat access logs failed", zap.String("node_id", nodeID), zap.Error(err)) zap.L().Error("build heartbeat access logs failed", zap.String("node_id", nodeID), zap.Error(err))
return return
@@ -68,17 +67,14 @@ func PersistHeartbeatObservability(ctx context.Context, nodeID string, payload N
return return
} }
if err := persistBufferedObservability(ctx, nodeID, payload.BufferedObservability, reportedAt); err != nil { if err := persistBufferedObservability(ctx, nodeID, payload.Buffered, reportedAt); err != nil {
zap.L().Error("persist buffered observability failed", zap.String("node_id", nodeID), zap.Error(err)) zap.L().Error("persist buffered observability failed", zap.String("node_id", nodeID), zap.Error(err))
} }
if err := persistNodeMetricSnapshot(ctx, nodeID, payload.Snapshot, reportedAt); err != nil { if err := persistNodeMetricSnapshot(ctx, nodeID, payload.HostMetrics, reportedAt); err != nil {
zap.L().Error("persist metric snapshot failed", zap.String("node_id", nodeID), zap.Error(err)) zap.L().Error("persist metric snapshot failed", zap.String("node_id", nodeID), zap.Error(err))
} }
if err := persistNodeOpenrestyObservation(ctx, nodeID, payload.OpenrestyObservation, reportedAt); err != nil { if err := persistNodeEdgeHealth(ctx, nodeID, payload.EdgeHealth, payload.OpenrestyStatus, reportedAt); err != nil {
zap.L().Error("persist openresty observation failed", zap.String("node_id", nodeID), zap.Error(err)) zap.L().Error("persist edge health failed", zap.String("node_id", nodeID), zap.Error(err))
}
if err := persistNodeTrafficReport(ctx, nodeID, payload.TrafficReport, reportedAt); err != nil {
zap.L().Error("persist traffic report failed", zap.String("node_id", nodeID), zap.Error(err))
} }
if err := persistNodeAccessLogs(ctx, nodeID, accessLogRecords, reportedAt); err != nil { if err := persistNodeAccessLogs(ctx, nodeID, accessLogRecords, reportedAt); err != nil {
@@ -88,19 +84,35 @@ func PersistHeartbeatObservability(ctx context.Context, nodeID string, payload N
func persistBufferedObservability(ctx context.Context, nodeID string, records []BufferedObservabilityRecord, reportedAt time.Time) error { func persistBufferedObservability(ctx context.Context, nodeID string, records []BufferedObservabilityRecord, reportedAt time.Time) error {
for _, record := range records { for _, record := range records {
if err := persistNodeMetricSnapshot(ctx, nodeID, record.Snapshot, reportedAt); err != nil { if err := persistNodeMetricSnapshot(ctx, nodeID, record.HostMetrics, reportedAt); err != nil {
return err return err
} }
if err := persistNodeOpenrestyObservation(ctx, nodeID, record.OpenrestyObservation, reportedAt); err != nil { if err := persistNodeEdgeHealth(ctx, nodeID, record.EdgeHealth, "", reportedAt); err != nil {
return err
}
if err := persistNodeTrafficReport(ctx, nodeID, record.TrafficReport, reportedAt); err != nil {
return err return err
} }
} }
return nil return nil
} }
func persistNodeEdgeHealth(ctx context.Context, nodeID string, health *NodeEdgeHealth, fallbackStatus string, reportedAt time.Time) error {
if health == nil {
return nil
}
status := strings.TrimSpace(health.Status)
if status == "" {
status = strings.TrimSpace(fallbackStatus)
}
if status == "" {
status = openrestyStatusUnknown
}
return model.InsertOpenFlareEdgeHealth(ctx, &model.OpenFlareEdgeHealth{
NodeID: nodeID,
CapturedAt: timeFromUnix(health.CapturedAtUnix, reportedAt),
Status: status,
Connections: health.Connections,
})
}
func persistNodeSystemProfile(tx *gorm.DB, nodeID string, profile *NodeSystemProfile, reportedAt time.Time) error { func persistNodeSystemProfile(tx *gorm.DB, nodeID string, profile *NodeSystemProfile, reportedAt time.Time) error {
if profile == nil { if profile == nil {
return nil return nil
@@ -158,41 +170,6 @@ func persistNodeMetricSnapshot(ctx context.Context, nodeID string, snapshot *Nod
return model.InsertOpenFlareMetricSnapshot(ctx, record) return model.InsertOpenFlareMetricSnapshot(ctx, record)
} }
func persistNodeOpenrestyObservation(ctx context.Context, nodeID string, obs *NodeOpenrestyObservation, reportedAt time.Time) error {
if obs == nil {
return nil
}
record := &model.OpenFlareNodeObservationOpenresty{
NodeID: nodeID,
CapturedAt: timeFromUnix(obs.CapturedAtUnix, reportedAt),
OpenrestyRxBytes: obs.OpenrestyRxBytes,
OpenrestyTxBytes: obs.OpenrestyTxBytes,
OpenrestyConnections: obs.OpenrestyConnections,
}
return model.InsertOpenFlareNodeObservationOpenresty(ctx, record)
}
func persistNodeTrafficReport(ctx context.Context, nodeID string, report *NodeTrafficReport, reportedAt time.Time) error {
if report == nil {
return nil
}
if report.WindowEndedAtUnix > 0 && report.WindowStartedAtUnix > report.WindowEndedAtUnix {
return errors.New("traffic report window_started_at_unix 不能大于 window_ended_at_unix")
}
record := &model.OpenFlareRequestReport{
NodeID: nodeID,
WindowStartedAt: timeFromUnix(report.WindowStartedAtUnix, reportedAt),
WindowEndedAt: timeFromUnix(report.WindowEndedAtUnix, reportedAt),
RequestCount: report.RequestCount,
ErrorCount: report.ErrorCount,
UniqueVisitorCount: report.UniqueVisitorCount,
StatusCodesJSON: marshalJSON(report.StatusCodes),
TopDomainsJSON: marshalJSON(report.TopDomains),
SourceCountriesJSON: marshalJSON(report.SourceCountries),
}
return model.InsertOpenFlareRequestReport(ctx, record)
}
func buildNodeAccessLogRecords(nodeID string, direct []NodeAccessLog, buffered []BufferedObservabilityRecord, reportedAt time.Time) ([]*model.OpenFlareAccessLog, error) { func buildNodeAccessLogRecords(nodeID string, direct []NodeAccessLog, buffered []BufferedObservabilityRecord, reportedAt time.Time) ([]*model.OpenFlareAccessLog, error) {
total := len(direct) total := len(direct)
for _, record := range buffered { for _, record := range buffered {
@@ -213,15 +190,29 @@ func buildNodeAccessLogRecords(nodeID string, direct []NodeAccessLog, buffered [
records := make([]*model.OpenFlareAccessLog, 0, total) records := make([]*model.OpenFlareAccessLog, 0, total)
appendLogs := func(logs []NodeAccessLog) { appendLogs := func(logs []NodeAccessLog) {
for _, item := range logs { for _, item := range logs {
bytesSent := item.BytesSent
if bytesSent < 0 {
bytesSent = 0
}
requestLength := item.RequestLength
if requestLength < 0 {
requestLength = 0
}
requestTimeMs := item.RequestTimeMs
if requestTimeMs < 0 {
requestTimeMs = 0
}
record := &model.OpenFlareAccessLog{ record := &model.OpenFlareAccessLog{
NodeID: nodeID, NodeID: nodeID,
LoggedAt: timeFromUnix(item.LoggedAtUnix, reportedAt), LoggedAt: timeFromUnix(item.LoggedAtUnix, reportedAt),
RemoteAddr: strings.TrimSpace(item.RemoteAddr), RemoteAddr: strings.TrimSpace(item.RemoteAddr),
Region: "", Region: "",
Host: strings.TrimSpace(item.Host), Host: strings.TrimSpace(item.Host),
Path: truncateForDatabase(strings.TrimSpace(item.Path), accessLogPathMaxLength), Path: truncateForDatabase(strings.TrimSpace(item.Path), accessLogPathMaxLength),
StatusCode: item.StatusCode, StatusCode: item.StatusCode,
BytesSent: item.BytesSent, BytesSent: bytesSent,
RequestLength: requestLength,
RequestTimeMs: requestTimeMs,
} }
if resolver != nil { if resolver != nil {
record.Region = resolver.Resolve(record.RemoteAddr) record.Region = resolver.Resolve(record.RemoteAddr)
@@ -14,11 +14,8 @@ type NodeSystemProfile = pkgprotocol.NodeSystemProfile
// NodeMetricSnapshot holds a point-in-time resource-usage sample from an agent. // NodeMetricSnapshot holds a point-in-time resource-usage sample from an agent.
type NodeMetricSnapshot = pkgprotocol.NodeMetricSnapshot type NodeMetricSnapshot = pkgprotocol.NodeMetricSnapshot
// NodeOpenrestyObservation reports the OpenResty process health observed by an agent. // NodeEdgeHealth is L2 OpenResty health + connections.
type NodeOpenrestyObservation = pkgprotocol.NodeOpenrestyObservation type NodeEdgeHealth = pkgprotocol.NodeEdgeHealth
// NodeTrafficReport aggregates traffic counters collected by an agent.
type NodeTrafficReport = pkgprotocol.NodeTrafficReport
// NodeAccessLog is a single access-log record forwarded by an agent. // NodeAccessLog is a single access-log record forwarded by an agent.
type NodeAccessLog = pkgprotocol.NodeAccessLog type NodeAccessLog = pkgprotocol.NodeAccessLog
+20 -47
View File
@@ -44,15 +44,13 @@ var (
initOnce sync.Once initOnce sync.Once
metricSnapshotWriter *batchwriter.Writer[analyticsmodel.NodeMetricSnapshot] metricSnapshotWriter *batchwriter.Writer[analyticsmodel.NodeMetricSnapshot]
requestReportWriter *batchwriter.Writer[analyticsmodel.NodeRequestReport] edgeHealthWriter *batchwriter.Writer[analyticsmodel.NodeEdgeHealth]
openrestyWriter *batchwriter.Writer[analyticsmodel.NodeObsOpenresty]
frpsWriter *batchwriter.Writer[analyticsmodel.NodeObsFrps] frpsWriter *batchwriter.Writer[analyticsmodel.NodeObsFrps]
frpcWriter *batchwriter.Writer[analyticsmodel.NodeObsFrpc] frpcWriter *batchwriter.Writer[analyticsmodel.NodeObsFrpc]
nodeAccessLogWriter *batchwriter.Writer[analyticsmodel.NodeAccessLog] nodeAccessLogWriter *batchwriter.Writer[analyticsmodel.NodeAccessLog]
metricSnapshotDedup *dedupSet metricSnapshotDedup *dedupSet
requestReportDedup *dedupSet edgeHealthDedup *dedupSet
openrestyDedup *dedupSet
frpsDedup *dedupSet frpsDedup *dedupSet
frpcDedup *dedupSet frpcDedup *dedupSet
) )
@@ -65,8 +63,7 @@ func Init(ctx context.Context) {
initOnce.Do(func() { initOnce.Do(func() {
metricSnapshotDedup = newDedupSet() metricSnapshotDedup = newDedupSet()
requestReportDedup = newDedupSet() edgeHealthDedup = newDedupSet()
openrestyDedup = newDedupSet()
frpsDedup = newDedupSet() frpsDedup = newDedupSet()
frpcDedup = newDedupSet() frpcDedup = newDedupSet()
@@ -76,17 +73,11 @@ func Init(ctx context.Context) {
metricSnapshotDedup, metricSnapshotDedup,
metricSnapshotKey, metricSnapshotKey,
) )
requestReportWriter = mustNewObservabilityWriter( edgeHealthWriter = mustNewObservabilityWriter(
"request_reports", "edge_health",
withFlushRetries(analyticsrepo.BatchInsertNodeRequestReports), withFlushRetries(analyticsrepo.BatchInsertNodeEdgeHealth),
requestReportDedup, edgeHealthDedup,
requestReportKey, edgeHealthKey,
)
openrestyWriter = mustNewObservabilityWriter(
"openresty_obs",
withFlushRetries(analyticsrepo.BatchInsertNodeObsOpenresty),
openrestyDedup,
openrestyKey,
) )
frpsWriter = mustNewObservabilityWriter( frpsWriter = mustNewObservabilityWriter(
"frps_obs", "frps_obs",
@@ -103,8 +94,7 @@ func Init(ctx context.Context) {
nodeAccessLogWriter = mustNewNodeAccessLogWriter() nodeAccessLogWriter = mustNewNodeAccessLogWriter()
metricSnapshotWriter.Start(ctx) metricSnapshotWriter.Start(ctx)
requestReportWriter.Start(ctx) edgeHealthWriter.Start(ctx)
openrestyWriter.Start(ctx)
frpsWriter.Start(ctx) frpsWriter.Start(ctx)
frpcWriter.Start(ctx) frpcWriter.Start(ctx)
nodeAccessLogWriter.Start(ctx) nodeAccessLogWriter.Start(ctx)
@@ -123,8 +113,7 @@ func Stop(ctx context.Context) error {
var firstErr error var firstErr error
for _, writer := range []batchStopper{ for _, writer := range []batchStopper{
metricSnapshotWriter, metricSnapshotWriter,
requestReportWriter, edgeHealthWriter,
openrestyWriter,
frpsWriter, frpsWriter,
frpcWriter, frpcWriter,
nodeAccessLogWriter, nodeAccessLogWriter,
@@ -143,8 +132,7 @@ func Stop(ctx context.Context) error {
func WriterStats() []batchwriter.Stats { func WriterStats() []batchwriter.Stats {
writers := []statsProvider{ writers := []statsProvider{
metricSnapshotWriter, metricSnapshotWriter,
requestReportWriter, edgeHealthWriter,
openrestyWriter,
frpsWriter, frpsWriter,
frpcWriter, frpcWriter,
nodeAccessLogWriter, nodeAccessLogWriter,
@@ -164,14 +152,9 @@ func QueueMetricSnapshot(snapshot analyticsmodel.NodeMetricSnapshot) {
queueWithDedup(metricSnapshotWriter, metricSnapshotDedup, metricSnapshotKey(snapshot), snapshot) queueWithDedup(metricSnapshotWriter, metricSnapshotDedup, metricSnapshotKey(snapshot), snapshot)
} }
// QueueRequestReport enqueues a request report for asynchronous flush. // QueueEdgeHealth enqueues an L2 edge health snapshot for asynchronous flush.
func QueueRequestReport(report analyticsmodel.NodeRequestReport) { func QueueEdgeHealth(row analyticsmodel.NodeEdgeHealth) {
queueWithDedup(requestReportWriter, requestReportDedup, requestReportKey(report), report) queueWithDedup(edgeHealthWriter, edgeHealthDedup, edgeHealthKey(row), row)
}
// QueueOpenrestyObservation enqueues an OpenResty observation for asynchronous flush.
func QueueOpenrestyObservation(observation analyticsmodel.NodeObsOpenresty) {
queueWithDedup(openrestyWriter, openrestyDedup, openrestyKey(observation), observation)
} }
// QueueFrpsObservation enqueues an FRPS observation for asynchronous flush. // QueueFrpsObservation enqueues an FRPS observation for asynchronous flush.
@@ -297,11 +280,10 @@ func withFlushRetries[T any](flush batchwriter.FlushFunc[T]) batchwriter.FlushFu
func wireModelInsertHooks() { func wireModelInsertHooks() {
model.SetObservabilityInsertHooks(model.ObservabilityInsertHooks{ model.SetObservabilityInsertHooks(model.ObservabilityInsertHooks{
QueueMetricSnapshot: QueueMetricSnapshot, QueueMetricSnapshot: QueueMetricSnapshot,
QueueRequestReport: QueueRequestReport, QueueEdgeHealth: QueueEdgeHealth,
QueueOpenrestyObservation: QueueOpenrestyObservation, QueueFrpsObservation: QueueFrpsObservation,
QueueFrpsObservation: QueueFrpsObservation, QueueFrpcObservation: QueueFrpcObservation,
QueueFrpcObservation: QueueFrpcObservation,
}) })
model.SetAccessLogInsertHooks(model.AccessLogInsertHooks{ model.SetAccessLogInsertHooks(model.AccessLogInsertHooks{
QueueNodeAccessLogs: QueueNodeAccessLogs, QueueNodeAccessLogs: QueueNodeAccessLogs,
@@ -312,17 +294,8 @@ func metricSnapshotKey(snapshot analyticsmodel.NodeMetricSnapshot) string {
return fmt.Sprintf("%s|%d", snapshot.NodeID, snapshot.CapturedAt.UTC().UnixNano()) return fmt.Sprintf("%s|%d", snapshot.NodeID, snapshot.CapturedAt.UTC().UnixNano())
} }
func requestReportKey(report analyticsmodel.NodeRequestReport) string { func edgeHealthKey(row analyticsmodel.NodeEdgeHealth) string {
return fmt.Sprintf( return fmt.Sprintf("%s|%d", row.NodeID, row.CapturedAt.UTC().UnixNano())
"%s|%d|%d",
report.NodeID,
report.WindowStartedAt.UTC().UnixNano(),
report.WindowEndedAt.UTC().UnixNano(),
)
}
func openrestyKey(observation analyticsmodel.NodeObsOpenresty) string {
return fmt.Sprintf("%s|%d", observation.NodeID, observation.CapturedAt.UTC().UnixNano())
} }
func frpsKey(observation analyticsmodel.NodeObsFrps) string { func frpsKey(observation analyticsmodel.NodeObsFrps) string {
+2 -2
View File
@@ -31,8 +31,8 @@ func computeNodeStatus(node *model.OpenFlareNode) string {
if node.LastSeenAt == nil || node.LastSeenAt.IsZero() { if node.LastSeenAt == nil || node.LastSeenAt.IsZero() {
return nodeStatusPending return nodeStatusPending
} }
// 使用默认阈值 2 分钟 // 默认离线阈值 60 秒(与 node_offline_threshold 默认一致)
threshold := 2 * time.Minute threshold := 60 * time.Second
if time.Since(*node.LastSeenAt) > threshold { if time.Since(*node.LastSeenAt) > threshold {
return nodeStatusOffline return nodeStatusOffline
} }
+48 -52
View File
@@ -122,19 +122,10 @@ func buildOverviewView(ctx context.Context) (*OverviewView, error) {
if err != nil { if err != nil {
return nil, err return nil, err
} }
latestTrafficRows, err := model.ListOpenFlareLatestRequestReportsSince(ctx, "", since)
if err != nil {
return nil, err
}
// Bounded raw windows remain for distributions and trend fallbacks; trends prefer hourly rollups.
snapshots, err := model.ListOpenFlareMetricSnapshotsSince(ctx, "", since, dashboardOverviewSnapshotLimit) snapshots, err := model.ListOpenFlareMetricSnapshotsSince(ctx, "", since, dashboardOverviewSnapshotLimit)
if err != nil { if err != nil {
return nil, err return nil, err
} }
reports, err := model.ListOpenFlareRequestReportsSince(ctx, "", since, dashboardOverviewSnapshotLimit)
if err != nil {
return nil, err
}
accessLogRegions, err := model.ListOpenFlareAccessLogRegionCounts(ctx, "", since, dashboardDistributionLimit) accessLogRegions, err := model.ListOpenFlareAccessLogRegionCounts(ctx, "", since, dashboardDistributionLimit)
if err != nil { if err != nil {
return nil, err return nil, err
@@ -143,21 +134,48 @@ func buildOverviewView(ctx context.Context) (*OverviewView, error) {
if err != nil { if err != nil {
return nil, err return nil, err
} }
openrestySnapshots, err := model.ListOpenFlareNodeObservationOpenresty(ctx, "", since, dashboardOverviewSnapshotLimit)
if err != nil { // L1 business: trends + distributions + totals from access logs only.
return nil, err
}
view := &OverviewView{ view := &OverviewView{
GeneratedAt: now, GeneratedAt: now,
Nodes: make([]NodeHealth, 0, len(nodes)), Nodes: make([]NodeHealth, 0, len(nodes)),
Distributions: observability.BuildTrafficDistributions(reports, accessLogRegions, dashboardDistributionLimit), Distributions: observability.BuildTrafficDistributionsFromAccessLogs(
Trends: observability.BuildNodeTrends(ctx, now, "", snapshots, openrestySnapshots, reports), ctx, since, now, dashboardDistributionLimit, accessLogRegions,
),
Trends: observability.BuildNodeTrends(ctx, now, "", snapshots),
}
// Global traffic summary uses true window uniqExact for UV (not sum of hourly uniques).
if summary, sumErr := model.TrafficSummaryOpenFlareAccessLogs(ctx, model.OpenFlareAccessLogQuery{
Since: since,
Until: now,
}); sumErr == nil {
view.Traffic.RequestCount = summary.RequestCount
view.Traffic.ErrorCount = summary.ErrorCount
view.Traffic.UniqueVisitors = summary.UniqueIPCount
view.Traffic.ReportedNodes = int(summary.NodeCount)
if summary.RequestCount > 0 {
// Average QPS over the 24h window.
view.Traffic.EstimatedQPS = float64(summary.RequestCount) / (24 * 3600)
}
} else {
// Fallback: sum hourly request/error buckets only (UV left from summary path).
applyTrafficTotalsFromTrend(&view.Traffic, view.Trends.Traffic24h)
}
nodeTraffic := map[string]model.OpenFlareAccessLogNodeAggregate{}
if aggregates, aggErr := model.NodeAggregatesOpenFlareAccessLogs(ctx, model.OpenFlareAccessLogQuery{
Since: since,
Until: now,
}); aggErr == nil {
for _, row := range aggregates {
nodeTraffic[row.NodeID] = row
}
} }
var cpuNodeCount int var cpuNodeCount int
var memoryNodeCount int var memoryNodeCount int
latestSnapshots := observability.LatestMetricSnapshotsByNode(latestSnapshotRows) latestSnapshots := observability.LatestMetricSnapshotsByNode(latestSnapshotRows)
latestTrafficReports := observability.LatestTrafficReportsByNode(latestTrafficRows)
activeEventsByNode := observability.ActiveHealthEventsByNode(activeEvents) activeEventsByNode := observability.ActiveHealthEventsByNode(activeEvents)
for _, node := range nodes { for _, node := range nodes {
@@ -175,7 +193,6 @@ func buildOverviewView(ctx context.Context) (*OverviewView, error) {
} }
latestSnapshot := latestSnapshots[node.NodeID] latestSnapshot := latestSnapshots[node.NodeID]
latestTraffic := latestTrafficReports[node.NodeID]
nodeActiveEvents := activeEventsByNode[node.NodeID] nodeActiveEvents := activeEventsByNode[node.NodeID]
nodeHealth := NodeHealth{ nodeHealth := NodeHealth{
@@ -193,14 +210,15 @@ func buildOverviewView(ctx context.Context) (*OverviewView, error) {
} }
cpuNodeCount, memoryNodeCount = applyNodeSnapshotMetrics(&nodeHealth, latestSnapshot, view, cpuNodeCount, memoryNodeCount) cpuNodeCount, memoryNodeCount = applyNodeSnapshotMetrics(&nodeHealth, latestSnapshot, view, cpuNodeCount, memoryNodeCount)
applyNodeTrafficMetrics(&nodeHealth, latestTraffic) if agg, ok := nodeTraffic[node.NodeID]; ok {
nodeHealth.RequestCount = agg.RequestCount
nodeHealth.ErrorCount = agg.ErrorCount
nodeHealth.UniqueVisitorCount = agg.UniqueIPCount
}
view.Nodes = append(view.Nodes, nodeHealth) view.Nodes = append(view.Nodes, nodeHealth)
} }
applyTrafficTotalsFromTrend(&view.Traffic, view.Trends.Traffic24h)
applyTrafficRuntimeMetrics(&view.Traffic, latestTrafficReports)
view.Summary.TotalNodes = len(nodes) view.Summary.TotalNodes = len(nodes)
if cpuNodeCount > 0 { if cpuNodeCount > 0 {
view.Capacity.AverageCPUUsagePercent /= float64(cpuNodeCount) view.Capacity.AverageCPUUsagePercent /= float64(cpuNodeCount)
@@ -246,43 +264,19 @@ func applyNodeSnapshotMetrics(nodeHealth *NodeHealth, snapshot *model.OpenFlareM
return cpuNodeCount, memoryNodeCount return cpuNodeCount, memoryNodeCount
} }
func applyNodeTrafficMetrics(nodeHealth *NodeHealth, traffic *model.OpenFlareRequestReport) {
if traffic == nil {
return
}
nodeHealth.RequestCount = traffic.RequestCount
nodeHealth.ErrorCount = traffic.ErrorCount
nodeHealth.UniqueVisitorCount = traffic.UniqueVisitorCount
}
func applyTrafficTotalsFromTrend(traffic *Traffic, points []observability.TrafficTrendPoint) { func applyTrafficTotalsFromTrend(traffic *Traffic, points []observability.TrafficTrendPoint) {
if traffic == nil { if traffic == nil {
return return
} }
traffic.RequestCount = 0 traffic.RequestCount = 0
traffic.ErrorCount = 0 traffic.ErrorCount = 0
// Do not sum hourly unique visitors — that overcounts. UV must come from TrafficSummary.
for _, point := range points { for _, point := range points {
traffic.RequestCount += point.RequestCount traffic.RequestCount += point.RequestCount
traffic.ErrorCount += point.ErrorCount traffic.ErrorCount += point.ErrorCount
} }
} if traffic.RequestCount > 0 && traffic.ReportedNodes == 0 {
traffic.ReportedNodes = 1
func applyTrafficRuntimeMetrics(traffic *Traffic, latestReports map[string]*model.OpenFlareRequestReport) {
if traffic == nil {
return
}
traffic.UniqueVisitors = 0
traffic.EstimatedQPS = 0
traffic.ReportedNodes = 0
for _, report := range latestReports {
if report == nil {
continue
}
traffic.UniqueVisitors += report.UniqueVisitorCount
traffic.ReportedNodes++
if duration := report.WindowEndedAt.Sub(report.WindowStartedAt).Seconds(); duration > 0 {
traffic.EstimatedQPS += float64(report.RequestCount) / duration
}
} }
} }
@@ -360,12 +354,14 @@ func compressCapacityTrendPoints(points []observability.CapacityTrendPoint) [][]
func compressNetworkTrendPoints(points []observability.NetworkTrendPoint) [][]any { func compressNetworkTrendPoints(points []observability.NetworkTrendPoint) [][]any {
rows := make([][]any, 0, len(points)) rows := make([][]any, 0, len(points))
for _, point := range points { for _, point := range points {
// Compact layout (stable positions):
// [0] bucket, [1] host_rx, [2] host_tx, [3] bytes_received, [4] bytes_provided, [5] reported_nodes
rows = append(rows, []any{ rows = append(rows, []any{
point.BucketStartedAt, point.BucketStartedAt,
point.NetworkRxBytes, point.NetworkRxBytes,
point.NetworkTxBytes, point.NetworkTxBytes,
point.OpenrestyRxBytes, point.BytesReceived,
point.OpenrestyTxBytes, point.BytesProvided,
point.ReportedNodes, point.ReportedNodes,
}) })
} }
@@ -39,7 +39,7 @@ func TestGetOverviewStructure(t *testing.T) {
ctx := context.Background() ctx := context.Background()
now := time.Now().UTC() now := time.Now().UTC()
lastSeen := now.Add(-time.Minute) lastSeen := now.Add(-15 * time.Second) // within default 60s offline threshold
require.NoError(t, db.DB(ctx).Create(&model.OpenFlareNode{ require.NoError(t, db.DB(ctx).Create(&model.OpenFlareNode{
NodeID: "node-dashboard-1", NodeID: "node-dashboard-1",
@@ -75,14 +75,33 @@ func TestGetOverviewStructure(t *testing.T) {
StorageUsedBytes: 2, StorageUsedBytes: 2,
StorageTotalBytes: 10, StorageTotalBytes: 10,
})) }))
require.NoError(t, model.InsertOpenFlareRequestReport(ctx, &model.OpenFlareRequestReport{ // Business traffic from access logs (L1 authority): 12 requests, 1 server error, 4 unique IPs.
NodeID: "node-dashboard-1", logs := make([]*model.OpenFlareAccessLog, 0, 12)
WindowStartedAt: now.Add(-2 * time.Minute), for i := 0; i < 11; i++ {
WindowEndedAt: now.Add(-time.Minute), logs = append(logs, &model.OpenFlareAccessLog{
RequestCount: 12, NodeID: "node-dashboard-1",
ErrorCount: 1, LoggedAt: now.Add(-time.Minute),
UniqueVisitorCount: 4, RemoteAddr: "10.0.0." + string(rune('1'+i%4)), // rough; fixed below
})) Host: "app.example.com",
Path: "/",
StatusCode: 200,
BytesSent: 100,
})
}
ips := []string{"10.0.0.10", "10.0.0.11", "10.0.0.12", "10.0.0.13"}
for i := 0; i < 11; i++ {
logs[i].RemoteAddr = ips[i%4]
}
logs = append(logs, &model.OpenFlareAccessLog{
NodeID: "node-dashboard-1",
LoggedAt: now.Add(-time.Minute),
RemoteAddr: ips[0],
Host: "app.example.com",
Path: "/err",
StatusCode: 502,
BytesSent: 10,
})
require.NoError(t, model.InsertOpenFlareAccessLogsBatch(ctx, logs))
overview, err := GetOverview(ctx) overview, err := GetOverview(ctx)
require.NoError(t, err) require.NoError(t, err)
@@ -98,8 +117,12 @@ func TestGetOverviewStructure(t *testing.T) {
assert.Equal(t, int64(12), overview.Traffic.RequestCount) assert.Equal(t, int64(12), overview.Traffic.RequestCount)
assert.Equal(t, int64(4), overview.Traffic.UniqueVisitors) assert.Equal(t, int64(4), overview.Traffic.UniqueVisitors)
assert.Equal(t, int64(1), overview.Traffic.ErrorCount) assert.Equal(t, int64(1), overview.Traffic.ErrorCount)
assert.InDelta(t, 0.2, overview.Traffic.EstimatedQPS, 0.0001) // QPS over 24h window
assert.InDelta(t, 12.0/(24*3600.0), overview.Traffic.EstimatedQPS, 0.0001)
assert.Equal(t, 1, overview.Traffic.ReportedNodes) assert.Equal(t, 1, overview.Traffic.ReportedNodes)
// Node-level traffic from access log aggregates
onlineNodeCheck := overview.Nodes
require.NotEmpty(t, onlineNodeCheck)
assert.Equal(t, 55.0, overview.Capacity.AverageCPUUsagePercent) assert.Equal(t, 55.0, overview.Capacity.AverageCPUUsagePercent)
assert.Equal(t, 50.0, overview.Capacity.AverageMemoryUsagePercent) assert.Equal(t, 50.0, overview.Capacity.AverageMemoryUsagePercent)
@@ -110,8 +133,9 @@ func TestGetOverviewStructure(t *testing.T) {
require.NotNil(t, overview.Distributions.StatusCodes) require.NotNil(t, overview.Distributions.StatusCodes)
require.NotNil(t, overview.Distributions.TopDomains) require.NotNil(t, overview.Distributions.TopDomains)
require.NotNil(t, overview.Distributions.SourceCountries) require.NotNil(t, overview.Distributions.SourceCountries)
assert.Empty(t, overview.Distributions.StatusCodes) // Status/top domains come from access logs.
assert.Empty(t, overview.Distributions.TopDomains) assert.NotEmpty(t, overview.Distributions.StatusCodes)
assert.NotEmpty(t, overview.Distributions.TopDomains)
assert.Empty(t, overview.Distributions.SourceCountries) assert.Empty(t, overview.Distributions.SourceCountries)
require.Len(t, overview.Trends.Traffic24h, 24) require.Len(t, overview.Trends.Traffic24h, 24)
@@ -147,11 +171,11 @@ func TestGetOverviewStructure(t *testing.T) {
assert.Equal(t, "online", onlineNode[6]) assert.Equal(t, "online", onlineNode[6])
assert.Equal(t, "healthy", onlineNode[7]) assert.Equal(t, "healthy", onlineNode[7])
// Latest-per-node health fields (indexes match compressDashboardNodes). // Latest-per-node health fields (indexes match compressDashboardNodes).
assert.Equal(t, 55.0, onlineNode[11]) // cpu_usage_percent from latest snapshot assert.Equal(t, 55.0, onlineNode[11]) // cpu_usage_percent from latest snapshot
assert.Equal(t, 50.0, onlineNode[12]) // memory_usage_percent assert.Equal(t, 50.0, onlineNode[12]) // memory_usage_percent
assert.Equal(t, int64(12), onlineNode[14]) assert.Equal(t, int64(12), onlineNode[14]) // request_count from access logs
assert.Equal(t, int64(1), onlineNode[15]) assert.Equal(t, int64(1), onlineNode[15]) // error_count
assert.Equal(t, int64(4), onlineNode[16]) assert.Equal(t, int64(4), onlineNode[16]) // unique visitors
pendingNode := nodeByID["node-dashboard-2"] pendingNode := nodeByID["node-dashboard-2"]
require.NotNil(t, pendingNode) require.NotNil(t, pendingNode)
@@ -161,5 +185,4 @@ func TestGetOverviewStructure(t *testing.T) {
assert.Equal(t, 55.0, overview.Capacity.AverageCPUUsagePercent) assert.Equal(t, 55.0, overview.Capacity.AverageCPUUsagePercent)
assert.Equal(t, 1, overview.Traffic.ReportedNodes) assert.Equal(t, 1, overview.Traffic.ReportedNodes)
assert.Equal(t, int64(4), overview.Traffic.UniqueVisitors)
} }
+2 -2
View File
@@ -194,9 +194,9 @@ func computeNodeStatus(node *model.OpenFlareNode) string {
if node.LastSeenAt == nil || node.LastSeenAt.IsZero() { if node.LastSeenAt == nil || node.LastSeenAt.IsZero() {
return nodeStatusPending return nodeStatusPending
} }
// 使用默认阈值 2 分钟,避免在这里读取配置 // 默认离线阈值 60 秒(与 node_offline_threshold 默认一致),避免在这里读取配置
// 实际阈值会在需要精确判断的地方通过 getNodeOfflineThreshold 读取 // 实际阈值会在需要精确判断的地方通过 getNodeOfflineThreshold 读取
threshold := 2 * time.Minute threshold := 60 * time.Second
if time.Since(*node.LastSeenAt) > threshold { if time.Since(*node.LastSeenAt) > threshold {
return nodeStatusOffline return nodeStatusOffline
} }
+2 -2
View File
@@ -322,8 +322,8 @@ func TestComputeNodeStatus(t *testing.T) {
online := &model.OpenFlareNode{LastSeenAt: &now} online := &model.OpenFlareNode{LastSeenAt: &now}
assert.Equal(t, nodeStatusOnline, computeNodeStatus(online)) assert.Equal(t, nodeStatusOnline, computeNodeStatus(online))
// computeNodeStatus 使用默认阈值 2 分钟 // computeNodeStatus 使用默认阈值 60 秒
offlineAt := now.Add(-2*time.Minute - time.Minute) offlineAt := now.Add(-61 * time.Second)
offline := &model.OpenFlareNode{LastSeenAt: &offlineAt} offline := &model.OpenFlareNode{LastSeenAt: &offlineAt}
assert.Equal(t, nodeStatusOffline, computeNodeStatus(offline)) assert.Equal(t, nodeStatusOffline, computeNodeStatus(offline))
} }
@@ -16,6 +16,7 @@ const (
maxAccessLogPageSize = 200 maxAccessLogPageSize = 200
defaultAccessLogSortBy = "logged_at" defaultAccessLogSortBy = "logged_at"
defaultAccessLogSortOrder = "desc" defaultAccessLogSortOrder = "desc"
accessLogSortOrderAsc = "asc"
defaultAccessLogFoldMinute = 3 defaultAccessLogFoldMinute = 3
defaultIPTrendHours = 24 defaultIPTrendHours = 24
defaultIPTrendBucketMinute = 30 defaultIPTrendBucketMinute = 30
@@ -605,8 +606,8 @@ func normalizeAccessLogSortBy(sortBy string) string {
} }
func normalizeAccessLogSortOrder(sortOrder string) string { func normalizeAccessLogSortOrder(sortOrder string) string {
if strings.EqualFold(strings.TrimSpace(sortOrder), "asc") { if strings.EqualFold(strings.TrimSpace(sortOrder), accessLogSortOrderAsc) {
return "asc" return accessLogSortOrderAsc
} }
return defaultAccessLogSortOrder return defaultAccessLogSortOrder
} }
+215 -197
View File
@@ -5,7 +5,6 @@ package observability
import ( import (
"context" "context"
"encoding/json"
"sort" "sort"
"strings" "strings"
"time" "time"
@@ -22,6 +21,7 @@ const (
healthSeverityCritical = "critical" healthSeverityCritical = "critical"
healthSeverityWarning = "warning" healthSeverityWarning = "warning"
percentageMultiplier = 100 percentageMultiplier = 100
sortOrderAsc = "asc"
) )
// DistributionItem is a key/value distribution entry. // DistributionItem is a key/value distribution entry.
@@ -37,9 +37,9 @@ type TrafficDistributions struct {
SourceCountries []DistributionItem `json:"source_countries"` SourceCountries []DistributionItem `json:"source_countries"`
} }
const metricSnapshotOpenrestyMatchWindow = 2 * time.Minute const metricSnapshotEdgeHealthMatchWindow = 2 * time.Minute
// NodeMetricSnapshotView is a metric snapshot enriched with OpenResty observations. // NodeMetricSnapshotView is a metric snapshot enriched with edge health connections.
type NodeMetricSnapshotView struct { type NodeMetricSnapshotView struct {
ID uint `json:"id,omitempty"` ID uint `json:"id,omitempty"`
NodeID string `json:"node_id,omitempty"` NodeID string `json:"node_id,omitempty"`
@@ -53,8 +53,6 @@ type NodeMetricSnapshotView struct {
DiskWriteBytes int64 `json:"disk_write_bytes"` DiskWriteBytes int64 `json:"disk_write_bytes"`
NetworkRxBytes int64 `json:"network_rx_bytes"` NetworkRxBytes int64 `json:"network_rx_bytes"`
NetworkTxBytes int64 `json:"network_tx_bytes"` NetworkTxBytes int64 `json:"network_tx_bytes"`
OpenrestyRxBytes int64 `json:"openresty_rx_bytes"`
OpenrestyTxBytes int64 `json:"openresty_tx_bytes"`
OpenrestyConnections int64 `json:"openresty_connections"` OpenrestyConnections int64 `json:"openresty_connections"`
} }
@@ -98,13 +96,15 @@ type CapacityTrendPoint struct {
} }
// NetworkTrendPoint is a network trend bucket. // NetworkTrendPoint is a network trend bucket.
// Host network_* is L3 (宿主机网卡).
// bytes_received/provided are L1 business bytes from access logs.
type NetworkTrendPoint struct { type NetworkTrendPoint struct {
BucketStartedAt time.Time `json:"bucket_started_at"` BucketStartedAt time.Time `json:"bucket_started_at"`
NetworkRxBytes int64 `json:"network_rx_bytes"` NetworkRxBytes int64 `json:"network_rx_bytes"`
NetworkTxBytes int64 `json:"network_tx_bytes"` NetworkTxBytes int64 `json:"network_tx_bytes"`
OpenrestyRxBytes int64 `json:"openresty_rx_bytes"` BytesReceived int64 `json:"bytes_received"` // sum(request_length)
OpenrestyTxBytes int64 `json:"openresty_tx_bytes"` BytesProvided int64 `json:"bytes_provided"` // sum(bytes_sent)
ReportedNodes int `json:"reported_nodes"` ReportedNodes int `json:"reported_nodes"`
} }
// DiskIOTrendPoint is a disk IO trend bucket. // DiskIOTrendPoint is a disk IO trend bucket.
@@ -141,30 +141,39 @@ type networkCounterState struct {
seen bool seen bool
} }
func buildTrafficWindowSummary(report *model.OpenFlareRequestReport) *TrafficWindowSummary { func buildTrafficWindowSummaryFromAccessLogs(
if report == nil { ctx context.Context,
nodeID string,
since, until time.Time,
) *TrafficWindowSummary {
row, err := model.TrafficSummaryOpenFlareAccessLogs(ctx, model.OpenFlareAccessLogQuery{
NodeID: nodeID,
Since: since,
Until: until,
})
if err != nil || row.RequestCount <= 0 {
return nil return nil
} }
summary := TrafficWindowSummary{ summary := &TrafficWindowSummary{
WindowStartedAt: report.WindowStartedAt, WindowStartedAt: since.UTC(),
WindowEndedAt: report.WindowEndedAt, WindowEndedAt: until.UTC(),
RequestCount: report.RequestCount, RequestCount: row.RequestCount,
UniqueVisitorCount: report.UniqueVisitorCount, UniqueVisitorCount: row.UniqueIPCount,
ErrorCount: report.ErrorCount, ErrorCount: row.ErrorCount,
} }
if duration := report.WindowEndedAt.Sub(report.WindowStartedAt).Seconds(); duration > 0 { if duration := until.Sub(since).Seconds(); duration > 0 {
summary.EstimatedQPS = float64(report.RequestCount) / duration summary.EstimatedQPS = float64(row.RequestCount) / duration
} }
if report.RequestCount > 0 { if row.RequestCount > 0 {
summary.ErrorRatePercent = (float64(report.ErrorCount) / float64(report.RequestCount)) * 100 summary.ErrorRatePercent = (float64(row.ErrorCount) / float64(row.RequestCount)) * 100
} }
return &summary return summary
} }
// BuildMetricSnapshotViews merges metric snapshots with OpenResty observations for API responses. // BuildMetricSnapshotViews merges metric snapshots with edge health connections for API responses.
func BuildMetricSnapshotViews( func BuildMetricSnapshotViews(
snapshots []*model.OpenFlareMetricSnapshot, snapshots []*model.OpenFlareMetricSnapshot,
openrestyObs []*model.OpenFlareNodeObservationOpenresty, edgeHealth []*model.OpenFlareEdgeHealth,
) []*NodeMetricSnapshotView { ) []*NodeMetricSnapshotView {
if len(snapshots) == 0 { if len(snapshots) == 0 {
return []*NodeMetricSnapshotView{} return []*NodeMetricSnapshotView{}
@@ -188,40 +197,49 @@ func BuildMetricSnapshotViews(
NetworkRxBytes: snapshot.NetworkRxBytes, NetworkRxBytes: snapshot.NetworkRxBytes,
NetworkTxBytes: snapshot.NetworkTxBytes, NetworkTxBytes: snapshot.NetworkTxBytes,
} }
if matched := matchOpenrestyObservation(snapshot.CapturedAt, openrestyObs); matched != nil { if matched := matchEdgeHealth(snapshot.CapturedAt, edgeHealth); matched != nil {
view.OpenrestyRxBytes = matched.OpenrestyRxBytes view.OpenrestyConnections = matched.Connections
view.OpenrestyTxBytes = matched.OpenrestyTxBytes
view.OpenrestyConnections = matched.OpenrestyConnections
} }
views = append(views, view) views = append(views, view)
} }
return views return views
} }
// BuildTrafficDistributions aggregates traffic distribution charts. // BuildTrafficDistributionsFromAccessLogs builds distributions from access logs (L1).
func BuildTrafficDistributions( func BuildTrafficDistributionsFromAccessLogs(
reports []*model.OpenFlareRequestReport, ctx context.Context,
accessLogRegions []*model.OpenFlareAccessLogRegionCount, since, until time.Time,
limit int, limit int,
accessLogRegions []*model.OpenFlareAccessLogRegionCount,
) TrafficDistributions { ) TrafficDistributions {
statusCodes := make(distributionAccumulator) statusCodes := make(distributionAccumulator)
topDomains := make(distributionAccumulator) topDomains := make(distributionAccumulator)
reportSourceCountries := make(distributionAccumulator)
for _, report := range reports { query := model.OpenFlareAccessLogQuery{Since: since, Until: until}
mergeJSONCounts(statusCodes, report.StatusCodesJSON) if statusRows, err := model.ValueCountsOpenFlareAccessLogs(ctx, query, "status_code", limit); err == nil {
mergeJSONCounts(topDomains, report.TopDomainsJSON) for _, row := range statusRows {
mergeJSONCounts(reportSourceCountries, report.SourceCountriesJSON) if strings.TrimSpace(row.Value) == "" || row.Count <= 0 {
}
sourceCountries := reportSourceCountries
if len(accessLogRegions) > 0 {
sourceCountries = make(distributionAccumulator, len(accessLogRegions))
for _, item := range accessLogRegions {
if item == nil || strings.TrimSpace(item.Region) == "" || item.Count <= 0 {
continue continue
} }
sourceCountries[item.Region] = item.Count statusCodes[row.Value] = row.Count
} }
} }
if hostRows, err := model.ValueCountsOpenFlareAccessLogs(ctx, query, "host", limit); err == nil {
for _, row := range hostRows {
if strings.TrimSpace(row.Value) == "" || row.Count <= 0 {
continue
}
topDomains[row.Value] = row.Count
}
}
sourceCountries := make(distributionAccumulator)
for _, item := range accessLogRegions {
if item == nil || strings.TrimSpace(item.Region) == "" || item.Count <= 0 {
continue
}
sourceCountries[item.Region] = item.Count
}
return TrafficDistributions{ return TrafficDistributions{
StatusCodes: toDistributionItems(statusCodes, limit), StatusCodes: toDistributionItems(statusCodes, limit),
TopDomains: toDistributionItems(topDomains, limit), TopDomains: toDistributionItems(topDomains, limit),
@@ -231,7 +249,7 @@ func BuildTrafficDistributions(
func buildHealthSummary( func buildHealthSummary(
snapshot *model.OpenFlareMetricSnapshot, snapshot *model.OpenFlareMetricSnapshot,
report *model.OpenFlareRequestReport, traffic *TrafficWindowSummary,
events []*model.OpenFlareHealthEvent, events []*model.OpenFlareHealthEvent,
) HealthSummary { ) HealthSummary {
summary := HealthSummary{} summary := HealthSummary{}
@@ -258,41 +276,37 @@ func buildHealthSummary(
storageUsage := Percentage(snapshot.StorageUsedBytes, snapshot.StorageTotalBytes) storageUsage := Percentage(snapshot.StorageUsedBytes, snapshot.StorageTotalBytes)
summary.HasCapacityRisk = snapshot.CPUUsagePercent >= 80 || memoryUsage >= 85 || storageUsage >= 85 summary.HasCapacityRisk = snapshot.CPUUsagePercent >= 80 || memoryUsage >= 85 || storageUsage >= 85
} }
if report != nil && report.RequestCount >= 100 { if traffic != nil && traffic.RequestCount >= 100 {
summary.HasTrafficRisk = (float64(report.ErrorCount) / float64(report.RequestCount)) >= 0.05 summary.HasTrafficRisk = (float64(traffic.ErrorCount) / float64(traffic.RequestCount)) >= 0.05
} }
summary.HasRuntimeRisk = summary.ActiveAlerts > 0 || summary.HasCapacityRisk || summary.HasTrafficRisk summary.HasRuntimeRisk = summary.ActiveAlerts > 0 || summary.HasCapacityRisk || summary.HasTrafficRisk
return summary return summary
} }
// BuildNodeTrends builds 24h trend series, preferring ClickHouse hourly aggregates // BuildNodeTrends builds 24h trend series.
// over limited raw snapshot windows so capacity/network/disk charts stay complete. // Business traffic (requests/errors/UV and provided/received bytes) comes from access logs.
// Host capacity/disk/network come from metric snapshots (hourly when available).
func BuildNodeTrends( func BuildNodeTrends(
ctx context.Context, ctx context.Context,
now time.Time, now time.Time,
nodeID string, nodeID string,
snapshots []*model.OpenFlareMetricSnapshot, snapshots []*model.OpenFlareMetricSnapshot,
openrestyObs []*model.OpenFlareNodeObservationOpenresty,
reports []*model.OpenFlareRequestReport,
) NodeTrends { ) NodeTrends {
trendSince := now.Add(-24 * time.Hour) trendSince := now.Add(-24 * time.Hour)
trafficTrend := BuildTrafficTrendPoints(now, reports)
if trafficHourly, err := model.ListOpenFlareTrafficHourlySince(ctx, nodeID, trendSince); err == nil && len(trafficHourly) > 0 {
trafficTrend = BuildTrafficTrendPointsFromHourly(now, trafficHourly)
}
trafficTrend := BuildTrafficTrendPointsFromAccessLogs(ctx, now, nodeID, trendSince)
capacityTrend := BuildCapacityTrendPoints(now, snapshots) capacityTrend := BuildCapacityTrendPoints(now, snapshots)
networkTrend := BuildNetworkTrendPoints(now, snapshots, openrestyObs) networkTrend := BuildNetworkTrendPoints(now, snapshots)
// Overlay L1 business bytes onto network points.
applyAccessLogBytesToNetworkTrend(ctx, now, nodeID, trendSince, networkTrend)
diskIOTrend := BuildDiskIOTrendPoints(now, snapshots) diskIOTrend := BuildDiskIOTrendPoints(now, snapshots)
metricHourly, metricErr := model.ListOpenFlareMetricHourlySince(ctx, nodeID, trendSince) metricHourly, metricErr := model.ListOpenFlareMetricHourlySince(ctx, nodeID, trendSince)
if metricErr == nil && len(metricHourly) > 0 { if metricErr == nil && len(metricHourly) > 0 {
capacityTrend = BuildCapacityTrendPointsFromHourly(now, metricHourly) capacityTrend = BuildCapacityTrendPointsFromHourly(now, metricHourly)
diskIOTrend = BuildDiskIOTrendPointsFromHourly(now, metricHourly) diskIOTrend = BuildDiskIOTrendPointsFromHourly(now, metricHourly)
} networkTrend = BuildNetworkTrendPointsFromHourly(now, metricHourly)
openrestyHourly, openrestyErr := model.ListOpenFlareOpenrestyHourlySince(ctx, nodeID, trendSince) applyAccessLogBytesToNetworkTrend(ctx, now, nodeID, trendSince, networkTrend)
if metricErr == nil && openrestyErr == nil && (len(metricHourly) > 0 || len(openrestyHourly) > 0) {
networkTrend = BuildNetworkTrendPointsFromHourly(now, metricHourly, openrestyHourly)
} }
return NodeTrends{ return NodeTrends{
@@ -303,7 +317,131 @@ func BuildNodeTrends(
} }
} }
// BuildTrafficTrendPointsFromAccessLogs builds 24h request/error buckets from access logs.
// Prefers of_access_log_hourly when available; falls back to raw bucket aggregates.
// UniqueVisitorCount on hourly path is 0 (use TrafficSummary for exact UV).
func BuildTrafficTrendPointsFromAccessLogs(ctx context.Context, now time.Time, nodeID string, since time.Time) []TrafficTrendPoint {
start := trendWindowStart(now)
points := make([]TrafficTrendPoint, observabilityTrendBuckets)
for index := range points {
points[index].BucketStartedAt = start.Add(time.Duration(index) * time.Hour)
}
if hourly, err := model.ListOpenFlareTrafficHourlySince(ctx, nodeID, since); err == nil && len(hourly) > 0 {
for _, row := range hourly {
if row == nil {
continue
}
index, ok := trendBucketIndex(row.Hour, start)
if !ok {
continue
}
points[index].RequestCount += row.RequestCount
points[index].ErrorCount += row.ErrorCount
// UniqueVisitorCount intentionally not summed from hourly rollup (always 0 / overcounts).
}
return points
}
buckets, err := model.ListOpenFlareAccessLogBuckets(ctx, model.OpenFlareAccessLogBucketQuery{
NodeID: nodeID,
Since: since,
Until: now,
FoldMinutes: 60,
SortBy: "logged_at",
SortOrder: sortOrderAsc,
})
if err != nil || len(buckets) == 0 {
return points
}
byEpoch := make(map[int64]*model.OpenFlareAccessLogBucketRow, len(buckets))
for _, row := range buckets {
if row == nil {
continue
}
byEpoch[row.BucketEpoch] = row
}
for index := range points {
epoch := points[index].BucketStartedAt.Unix()
if row, ok := byEpoch[epoch]; ok {
points[index].RequestCount = row.RequestCount
points[index].ErrorCount = row.ServerErrorCount
points[index].UniqueVisitorCount = row.UniqueIPCount
}
}
return points
}
func applyAccessLogBytesToNetworkTrend(ctx context.Context, now time.Time, nodeID string, since time.Time, points []NetworkTrendPoint) {
if len(points) == 0 {
return
}
// Prefer of_access_log_hourly (summed across hosts).
if hourly, err := analyticsListAccessLogHourlyBytes(ctx, nodeID, since); err == nil && len(hourly) > 0 {
for hourUnix, totals := range hourly {
for index := range points {
if points[index].BucketStartedAt.Unix() == hourUnix {
points[index].BytesProvided = totals.provided
points[index].BytesReceived = totals.received
}
}
}
return
}
buckets, err := model.ListOpenFlareAccessLogBuckets(ctx, model.OpenFlareAccessLogBucketQuery{
NodeID: nodeID,
Since: since,
Until: now,
FoldMinutes: 60,
SortBy: "logged_at",
SortOrder: sortOrderAsc,
})
if err != nil || len(buckets) == 0 {
return
}
byEpoch := make(map[int64]*model.OpenFlareAccessLogBucketRow, len(buckets))
for _, row := range buckets {
if row == nil {
continue
}
byEpoch[row.BucketEpoch] = row
}
for index := range points {
epoch := points[index].BucketStartedAt.Unix()
if row, ok := byEpoch[epoch]; ok {
points[index].BytesProvided = row.BytesSent
points[index].BytesReceived = row.RequestLength
}
}
}
type accessLogHourBytes struct {
provided int64
received int64
}
func analyticsListAccessLogHourlyBytes(ctx context.Context, nodeID string, since time.Time) (map[int64]accessLogHourBytes, error) {
rows, err := model.ListOpenFlareAccessLogHourlySince(ctx, nodeID, since)
if err != nil {
return nil, err
}
out := make(map[int64]accessLogHourBytes)
for _, row := range rows {
if row == nil {
continue
}
key := row.Hour.UTC().Truncate(time.Hour).Unix()
cur := out[key]
cur.provided += row.BytesSent
cur.received += row.RequestLength
out[key] = cur
}
return out, nil
}
// BuildTrafficTrendPointsFromHourly builds 24h traffic trend buckets from hourly rollups. // BuildTrafficTrendPointsFromHourly builds 24h traffic trend buckets from hourly rollups.
// UniqueVisitorCount is left at 0: hourly UV is not summed (use TrafficSummary for exact UV).
func BuildTrafficTrendPointsFromHourly(now time.Time, hourly []*model.OpenFlareTrafficHourly) []TrafficTrendPoint { func BuildTrafficTrendPointsFromHourly(now time.Time, hourly []*model.OpenFlareTrafficHourly) []TrafficTrendPoint {
start := trendWindowStart(now) start := trendWindowStart(now)
points := make([]TrafficTrendPoint, observabilityTrendBuckets) points := make([]TrafficTrendPoint, observabilityTrendBuckets)
@@ -320,26 +458,6 @@ func BuildTrafficTrendPointsFromHourly(now time.Time, hourly []*model.OpenFlareT
} }
points[index].RequestCount += row.RequestCount points[index].RequestCount += row.RequestCount
points[index].ErrorCount += row.ErrorCount points[index].ErrorCount += row.ErrorCount
points[index].UniqueVisitorCount += row.UniqueVisitorCount
}
return points
}
// BuildTrafficTrendPoints builds 24h traffic trend buckets.
func BuildTrafficTrendPoints(now time.Time, reports []*model.OpenFlareRequestReport) []TrafficTrendPoint {
start := trendWindowStart(now)
points := make([]TrafficTrendPoint, observabilityTrendBuckets)
for index := range points {
points[index].BucketStartedAt = start.Add(time.Duration(index) * time.Hour)
}
for _, report := range reports {
index, ok := trendBucketIndex(report.WindowEndedAt, start)
if !ok {
continue
}
points[index].RequestCount += report.RequestCount
points[index].ErrorCount += report.ErrorCount
points[index].UniqueVisitorCount += report.UniqueVisitorCount
} }
return points return points
} }
@@ -404,12 +522,12 @@ func BuildCapacityTrendPointsFromHourly(now time.Time, hourly []*model.OpenFlare
return points return points
} }
// BuildNetworkTrendPoints builds 24h network trend buckets. // BuildNetworkTrendPoints builds 24h host-network trend buckets.
// Host and OpenResty counters are cumulative; values are consecutive deltas. // Host network counters must be process-lifetime cumulative values; this function
// converts consecutive samples into deltas.
func BuildNetworkTrendPoints( func BuildNetworkTrendPoints(
now time.Time, now time.Time,
snapshots []*model.OpenFlareMetricSnapshot, snapshots []*model.OpenFlareMetricSnapshot,
openrestyObs []*model.OpenFlareNodeObservationOpenresty,
) []NetworkTrendPoint { ) []NetworkTrendPoint {
start := trendWindowStart(now) start := trendWindowStart(now)
points := make([]NetworkTrendPoint, observabilityTrendBuckets) points := make([]NetworkTrendPoint, observabilityTrendBuckets)
@@ -452,51 +570,17 @@ func BuildNetworkTrendPoints(
accumulators[index].nodes[snapshot.NodeID] = struct{}{} accumulators[index].nodes[snapshot.NodeID] = struct{}{}
} }
} }
sort.Slice(openrestyObs, func(i int, j int) bool {
if openrestyObs[i].CapturedAt.Equal(openrestyObs[j].CapturedAt) {
return openrestyObs[i].NodeID < openrestyObs[j].NodeID
}
return openrestyObs[i].CapturedAt.Before(openrestyObs[j].CapturedAt)
})
previousOpenrestyByNode := make(map[string]networkCounterState, len(openrestyObs))
for _, obs := range openrestyObs {
if obs == nil {
continue
}
nodeKey := obs.NodeID
if nodeKey == "" {
nodeKey = unknownTrendNodeKey
}
previous := previousOpenrestyByNode[nodeKey]
previousOpenrestyByNode[nodeKey] = networkCounterState{
rx: obs.OpenrestyRxBytes,
tx: obs.OpenrestyTxBytes,
seen: true,
}
if !previous.seen {
continue
}
index, ok := trendBucketIndex(obs.CapturedAt, start)
if !ok {
continue
}
points[index].OpenrestyRxBytes += nonNegativeDelta(obs.OpenrestyRxBytes, previous.rx)
points[index].OpenrestyTxBytes += nonNegativeDelta(obs.OpenrestyTxBytes, previous.tx)
if obs.NodeID != "" {
accumulators[index].nodes[obs.NodeID] = struct{}{}
}
}
for index := range points { for index := range points {
points[index].ReportedNodes = len(accumulators[index].nodes) points[index].ReportedNodes = len(accumulators[index].nodes)
} }
return points return points
} }
// BuildNetworkTrendPointsFromHourly builds 24h network trend buckets from hourly aggregates. // BuildNetworkTrendPointsFromHourly builds 24h host-network trend buckets from metric hourly aggregates.
// Business bytes (已提供/接收) are applied separately via applyAccessLogBytesToNetworkTrend.
func BuildNetworkTrendPointsFromHourly( func BuildNetworkTrendPointsFromHourly(
now time.Time, now time.Time,
metricHourly []*model.OpenFlareMetricHourly, metricHourly []*model.OpenFlareMetricHourly,
openrestyHourly []*model.OpenFlareOpenrestyHourly,
) []NetworkTrendPoint { ) []NetworkTrendPoint {
start := trendWindowStart(now) start := trendWindowStart(now)
points := make([]NetworkTrendPoint, observabilityTrendBuckets) points := make([]NetworkTrendPoint, observabilityTrendBuckets)
@@ -517,20 +601,6 @@ func BuildNetworkTrendPointsFromHourly(
points[index].ReportedNodes = row.ReportedNodes points[index].ReportedNodes = row.ReportedNodes
} }
} }
for _, row := range openrestyHourly {
if row == nil {
continue
}
index, ok := trendBucketIndex(row.Hour, start)
if !ok {
continue
}
points[index].OpenrestyRxBytes += row.OpenrestyRxBytes
points[index].OpenrestyTxBytes += row.OpenrestyTxBytes
if row.ReportedNodes > points[index].ReportedNodes {
points[index].ReportedNodes = row.ReportedNodes
}
}
return points return points
} }
@@ -623,38 +693,25 @@ func latestMetricSnapshot(snapshots []*model.OpenFlareMetricSnapshot) *model.Ope
return latest return latest
} }
func latestTrafficReport(reports []*model.OpenFlareRequestReport) *model.OpenFlareRequestReport { func matchEdgeHealth(
var latest *model.OpenFlareRequestReport
for _, report := range reports {
if report == nil {
continue
}
if latest == nil || report.WindowEndedAt.After(latest.WindowEndedAt) {
latest = report
}
}
return latest
}
func matchOpenrestyObservation(
capturedAt time.Time, capturedAt time.Time,
observations []*model.OpenFlareNodeObservationOpenresty, health []*model.OpenFlareEdgeHealth,
) *model.OpenFlareNodeObservationOpenresty { ) *model.OpenFlareEdgeHealth {
var matched *model.OpenFlareNodeObservationOpenresty var matched *model.OpenFlareEdgeHealth
bestDelta := metricSnapshotOpenrestyMatchWindow + time.Second bestDelta := metricSnapshotEdgeHealthMatchWindow + time.Second
for _, observation := range observations { for _, row := range health {
if observation == nil { if row == nil {
continue continue
} }
delta := capturedAt.Sub(observation.CapturedAt) delta := capturedAt.Sub(row.CapturedAt)
if delta < 0 { if delta < 0 {
delta = -delta delta = -delta
} }
if delta > metricSnapshotOpenrestyMatchWindow { if delta > metricSnapshotEdgeHealthMatchWindow {
continue continue
} }
if matched == nil || delta < bestDelta { if matched == nil || delta < bestDelta {
matched = observation matched = row
bestDelta = delta bestDelta = delta
} }
} }
@@ -676,21 +733,6 @@ func LatestMetricSnapshotsByNode(snapshots []*model.OpenFlareMetricSnapshot) map
return result return result
} }
// LatestTrafficReportsByNode returns the latest traffic report per node.
func LatestTrafficReportsByNode(reports []*model.OpenFlareRequestReport) map[string]*model.OpenFlareRequestReport {
result := make(map[string]*model.OpenFlareRequestReport, len(reports))
for _, report := range reports {
if report == nil || report.NodeID == "" {
continue
}
if existing, ok := result[report.NodeID]; ok && !report.WindowEndedAt.After(existing.WindowEndedAt) {
continue
}
result[report.NodeID] = report
}
return result
}
// ActiveHealthEventsByNode groups active health events by node id. // ActiveHealthEventsByNode groups active health events by node id.
func ActiveHealthEventsByNode(events []*model.OpenFlareHealthEvent) map[string][]*model.OpenFlareHealthEvent { func ActiveHealthEventsByNode(events []*model.OpenFlareHealthEvent) map[string][]*model.OpenFlareHealthEvent {
result := make(map[string][]*model.OpenFlareHealthEvent) result := make(map[string][]*model.OpenFlareHealthEvent)
@@ -711,30 +753,6 @@ func Percentage(used int64, total int64) float64 {
return (float64(used) / float64(total)) * percentageMultiplier return (float64(used) / float64(total)) * percentageMultiplier
} }
func mergeJSONCounts(target distributionAccumulator, raw string) {
if len(target) == 0 && strings.TrimSpace(raw) == "" {
return
}
values := parseJSONCounts(raw)
for key, value := range values {
if strings.TrimSpace(key) == "" || value <= 0 {
continue
}
target[key] += value
}
}
func parseJSONCounts(raw string) map[string]int64 {
if strings.TrimSpace(raw) == "" {
return nil
}
values := make(map[string]int64)
if err := json.Unmarshal([]byte(raw), &values); err != nil {
return nil
}
return values
}
func toDistributionItems(values distributionAccumulator, limit int) []DistributionItem { func toDistributionItems(values distributionAccumulator, limit int) []DistributionItem {
if len(values) == 0 { if len(values) == 0 {
return []DistributionItem{} return []DistributionItem{}
@@ -25,52 +25,25 @@ func TestBuildTrafficTrendPointsFromHourlyBucketsByHour(t *testing.T) {
if len(points) != observabilityTrendBuckets { if len(points) != observabilityTrendBuckets {
t.Fatalf("BuildTrafficTrendPointsFromHourly() len = %d, want %d", len(points), observabilityTrendBuckets) t.Fatalf("BuildTrafficTrendPointsFromHourly() len = %d, want %d", len(points), observabilityTrendBuckets)
} }
} // Hourly UV must not be summed into trend points.
func TestBuildTrafficTrendPointsBucketsByHour(t *testing.T) {
t.Parallel()
now := time.Date(2026, 6, 19, 17, 30, 0, 0, time.UTC)
reports := []*model.OpenFlareRequestReport{
{
NodeID: "node-a",
WindowStartedAt: now.Add(-3 * time.Hour),
WindowEndedAt: now.Add(-3*time.Hour + time.Minute),
RequestCount: 10,
ErrorCount: 1,
},
{
NodeID: "node-a",
WindowStartedAt: now.Add(-30 * time.Minute),
WindowEndedAt: now.Add(-29 * time.Minute),
RequestCount: 6,
ErrorCount: 0,
},
}
points := BuildTrafficTrendPoints(now, reports)
if len(points) != observabilityTrendBuckets {
t.Fatalf("BuildTrafficTrendPoints() len = %d, want %d", len(points), observabilityTrendBuckets)
}
var totalRequests int64
for _, point := range points { for _, point := range points {
totalRequests += point.RequestCount if point.UniqueVisitorCount != 0 {
t.Fatalf("UniqueVisitorCount = %d, want 0 on hourly path", point.UniqueVisitorCount)
}
} }
if totalRequests != 16 { index, ok := trendBucketIndex(now.Add(-2*time.Hour).Truncate(time.Hour), trendWindowStart(now))
t.Fatalf("total request_count = %d, want 16", totalRequests) if !ok {
t.Fatal("expected valid bucket index")
} }
if points[index].RequestCount != 12 {
currentHour := points[len(points)-1] t.Fatalf("request_count = %d, want 12", points[index].RequestCount)
if currentHour.RequestCount != 6 {
t.Fatalf("current hour request_count = %d, want 6", currentHour.RequestCount)
} }
if currentHour.ErrorCount != 0 { if points[index].ErrorCount != 1 {
t.Fatalf("current hour error_count = %d, want 0", currentHour.ErrorCount) t.Fatalf("error_count = %d, want 1", points[index].ErrorCount)
} }
} }
func TestBuildMetricSnapshotViewsMergesOpenrestyObservation(t *testing.T) { func TestBuildMetricSnapshotViewsMergesEdgeHealthConnections(t *testing.T) {
t.Parallel() t.Parallel()
capturedAt := time.Date(2026, 6, 19, 12, 0, 0, 0, time.UTC) capturedAt := time.Date(2026, 6, 19, 12, 0, 0, 0, time.UTC)
@@ -82,54 +55,30 @@ func TestBuildMetricSnapshotViewsMergesOpenrestyObservation(t *testing.T) {
CPUUsagePercent: 12.5, CPUUsagePercent: 12.5,
}, },
} }
openrestyObs := []*model.OpenFlareNodeObservationOpenresty{ edgeHealth := []*model.OpenFlareEdgeHealth{
{ {
NodeID: "node-a", NodeID: "node-a",
CapturedAt: capturedAt.Add(5 * time.Second), CapturedAt: capturedAt.Add(5 * time.Second),
OpenrestyRxBytes: 4096, Status: "healthy",
OpenrestyTxBytes: 8192, Connections: 7,
OpenrestyConnections: 7,
}, },
} }
views := BuildMetricSnapshotViews(snapshots, openrestyObs) views := BuildMetricSnapshotViews(snapshots, edgeHealth)
if len(views) != 1 { if len(views) != 1 {
t.Fatalf("BuildMetricSnapshotViews() len = %d, want 1", len(views)) t.Fatalf("BuildMetricSnapshotViews() len = %d, want 1", len(views))
} }
if views[0].OpenrestyRxBytes != 4096 {
t.Fatalf("OpenrestyRxBytes = %d, want 4096", views[0].OpenrestyRxBytes)
}
if views[0].OpenrestyTxBytes != 8192 {
t.Fatalf("OpenrestyTxBytes = %d, want 8192", views[0].OpenrestyTxBytes)
}
if views[0].OpenrestyConnections != 7 { if views[0].OpenrestyConnections != 7 {
t.Fatalf("OpenrestyConnections = %d, want 7", views[0].OpenrestyConnections) t.Fatalf("OpenrestyConnections = %d, want 7", views[0].OpenrestyConnections)
} }
} }
func TestLatestTrafficReportUsesLatestWindowEndedAt(t *testing.T) { func TestBuildTrafficWindowSummaryFromAccessLogsNilWithoutData(t *testing.T) {
t.Parallel() t.Parallel()
older := &model.OpenFlareRequestReport{ // Without an access-log store / data, summary is nil.
WindowEndedAt: time.Date(2026, 6, 19, 10, 0, 0, 0, time.UTC), if summary := buildTrafficWindowSummaryFromAccessLogs(t.Context(), "missing", time.Now().Add(-time.Hour), time.Now()); summary != nil {
RequestCount: 3, t.Fatalf("buildTrafficWindowSummaryFromAccessLogs() = %#v, want nil", summary)
}
newer := &model.OpenFlareRequestReport{
WindowEndedAt: time.Date(2026, 6, 19, 11, 0, 0, 0, time.UTC),
RequestCount: 9,
}
latest := latestTrafficReport([]*model.OpenFlareRequestReport{older, newer})
if latest == nil || latest.RequestCount != 9 {
t.Fatalf("latestTrafficReport() = %#v, want newer report with request_count 9", latest)
}
}
func TestBuildTrafficWindowSummaryNilWithoutReport(t *testing.T) {
t.Parallel()
if summary := buildTrafficWindowSummary(nil); summary != nil {
t.Fatalf("buildTrafficWindowSummary(nil) = %#v, want nil", summary)
} }
} }
@@ -173,12 +122,8 @@ func TestBuildNetworkTrendPointsUsesCounterDeltas(t *testing.T) {
{NodeID: "n1", CapturedAt: base.Add(10 * time.Minute), NetworkRxBytes: 1000, NetworkTxBytes: 2000}, {NodeID: "n1", CapturedAt: base.Add(10 * time.Minute), NetworkRxBytes: 1000, NetworkTxBytes: 2000},
{NodeID: "n1", CapturedAt: base.Add(20 * time.Minute), NetworkRxBytes: 1500, NetworkTxBytes: 2600}, {NodeID: "n1", CapturedAt: base.Add(20 * time.Minute), NetworkRxBytes: 1500, NetworkTxBytes: 2600},
} }
openrestyObs := []*model.OpenFlareNodeObservationOpenresty{
{NodeID: "n1", CapturedAt: base.Add(10 * time.Minute), OpenrestyRxBytes: 100, OpenrestyTxBytes: 200},
{NodeID: "n1", CapturedAt: base.Add(20 * time.Minute), OpenrestyRxBytes: 180, OpenrestyTxBytes: 250},
}
points := BuildNetworkTrendPoints(now, snapshots, openrestyObs) points := BuildNetworkTrendPoints(now, snapshots)
current := points[len(points)-1] current := points[len(points)-1]
if current.NetworkRxBytes != 500 { if current.NetworkRxBytes != 500 {
t.Fatalf("network_rx_bytes = %d, want 500", current.NetworkRxBytes) t.Fatalf("network_rx_bytes = %d, want 500", current.NetworkRxBytes)
@@ -186,11 +131,30 @@ func TestBuildNetworkTrendPointsUsesCounterDeltas(t *testing.T) {
if current.NetworkTxBytes != 600 { if current.NetworkTxBytes != 600 {
t.Fatalf("network_tx_bytes = %d, want 600", current.NetworkTxBytes) t.Fatalf("network_tx_bytes = %d, want 600", current.NetworkTxBytes)
} }
if current.OpenrestyRxBytes != 80 { if current.BytesReceived != 0 || current.BytesProvided != 0 {
t.Fatalf("openresty_rx_bytes = %d, want 80", current.OpenrestyRxBytes) t.Fatalf("business bytes should be 0 without access logs overlay, got received=%d provided=%d",
current.BytesReceived, current.BytesProvided)
} }
if current.OpenrestyTxBytes != 50 { }
t.Fatalf("openresty_tx_bytes = %d, want 50", current.OpenrestyTxBytes)
func TestBuildHealthSummaryUsesTrafficSummary(t *testing.T) {
t.Parallel()
snapshot := &model.OpenFlareMetricSnapshot{
CPUUsagePercent: 10,
MemoryUsedBytes: 1,
MemoryTotalBytes: 10,
}
traffic := &TrafficWindowSummary{
RequestCount: 200,
ErrorCount: 20, // 10% error rate
}
summary := buildHealthSummary(snapshot, traffic, nil)
if !summary.HasTrafficRisk {
t.Fatal("HasTrafficRisk = false, want true for 10% error rate with >=100 requests")
}
if summary.HasCapacityRisk {
t.Fatal("HasCapacityRisk = true, want false")
} }
} }
@@ -79,7 +79,6 @@ type NodeView struct {
NodeID string `json:"node_id"` NodeID string `json:"node_id"`
Profile *model.OpenFlareNodeSystemProfile `json:"profile"` Profile *model.OpenFlareNodeSystemProfile `json:"profile"`
MetricSnapshots []*NodeMetricSnapshotView `json:"metric_snapshots"` MetricSnapshots []*NodeMetricSnapshotView `json:"metric_snapshots"`
TrafficReports []*model.OpenFlareRequestReport `json:"traffic_reports"`
HealthEvents []*model.OpenFlareHealthEvent `json:"health_events"` HealthEvents []*model.OpenFlareHealthEvent `json:"health_events"`
Analytics NodeAnalytics `json:"analytics"` Analytics NodeAnalytics `json:"analytics"`
Trends NodeTrends `json:"trends"` Trends NodeTrends `json:"trends"`
@@ -118,11 +117,7 @@ func GetNodeObservability(ctx context.Context, id uint, query NodeQuery) (*NodeV
if err != nil { if err != nil {
return nil, err return nil, err
} }
openrestyObs, err := model.ListOpenFlareNodeObservationOpenresty(ctx, node.NodeID, since, limit) edgeHealth, err := model.ListOpenFlareEdgeHealth(ctx, node.NodeID, since, limit)
if err != nil {
return nil, err
}
reports, err := model.ListOpenFlareRequestReportsSince(ctx, node.NodeID, since, limit)
if err != nil { if err != nil {
return nil, err return nil, err
} }
@@ -134,18 +129,21 @@ func GetNodeObservability(ctx context.Context, id uint, query NodeQuery) (*NodeV
if err != nil { if err != nil {
return nil, err return nil, err
} }
distributions := BuildTrafficDistributionsFromAccessLogs(
ctx, since, now, defaultTrafficDistributionLimit, accessLogRegions,
)
trafficSummary := buildTrafficWindowSummaryFromAccessLogs(ctx, node.NodeID, since, now)
view := &NodeView{ view := &NodeView{
NodeID: node.NodeID, NodeID: node.NodeID,
Profile: profile, Profile: profile,
MetricSnapshots: BuildMetricSnapshotViews(snapshots, openrestyObs), MetricSnapshots: BuildMetricSnapshotViews(snapshots, edgeHealth),
TrafficReports: reports,
HealthEvents: events, HealthEvents: events,
Analytics: NodeAnalytics{ Analytics: NodeAnalytics{
Traffic: buildTrafficWindowSummary(latestTrafficReport(reports)), Traffic: trafficSummary,
Distributions: BuildTrafficDistributions(reports, accessLogRegions, defaultTrafficDistributionLimit), Distributions: distributions,
Health: buildHealthSummary(latestMetricSnapshot(snapshots), latestTrafficReport(reports), events), Health: buildHealthSummary(latestMetricSnapshot(snapshots), trafficSummary, events),
}, },
Trends: BuildNodeTrends(ctx, now, node.NodeID, snapshots, openrestyObs, reports), Trends: BuildNodeTrends(ctx, now, node.NodeID, snapshots),
} }
if node.NodeType == "tunnel_relay" { if node.NodeType == "tunnel_relay" {
frpsObs, frpsErr := model.ListOpenFlareNodeObservationFrps(ctx, node.NodeID, time.Time{}, 1) frpsObs, frpsErr := model.ListOpenFlareNodeObservationFrps(ctx, node.NodeID, time.Time{}, 1)
+1 -1
View File
@@ -16,7 +16,7 @@ const (
relayStatusUnhealthy = "unhealthy" relayStatusUnhealthy = "unhealthy"
releaseChannelStable = "stable" releaseChannelStable = "stable"
defaultAgentHeartbeatInterval = 10000 // 默认心跳间隔 10 秒(毫秒) defaultAgentHeartbeatInterval = 3000 // 默认心跳间隔 3 秒(毫秒)
defaultAgentUpdateRepo = "Rain-kl/OpenFlare" defaultAgentUpdateRepo = "Rain-kl/OpenFlare"
) )
@@ -47,7 +47,7 @@ func reconcileRelayHealthEvents(ctx context.Context, nodeID string, relayStatus
func persistRelayHeartbeatObservability(ctx context.Context, nodeID string, payload HeartbeatPayload, reportedAt time.Time) { func persistRelayHeartbeatObservability(ctx context.Context, nodeID string, payload HeartbeatPayload, reportedAt time.Time) {
agent.PersistHeartbeatObservability(ctx, nodeID, agent.NodePayload{ agent.PersistHeartbeatObservability(ctx, nodeID, agent.NodePayload{
Profile: payload.Profile, Profile: payload.Profile,
Snapshot: payload.Snapshot, HostMetrics: payload.Snapshot,
HealthEvents: payload.HealthEvents, HealthEvents: payload.HealthEvents,
}, reportedAt) }, reportedAt)
@@ -20,10 +20,8 @@ const (
DatabaseCleanupTargetAccessLogs = "node_access_logs" DatabaseCleanupTargetAccessLogs = "node_access_logs"
// DatabaseCleanupTargetMetricSnapshots is the API cleanup target for metric snapshots. // DatabaseCleanupTargetMetricSnapshots is the API cleanup target for metric snapshots.
DatabaseCleanupTargetMetricSnapshots = "node_metric_snapshots" DatabaseCleanupTargetMetricSnapshots = "node_metric_snapshots"
// DatabaseCleanupTargetRequestReports is the API cleanup target for request reports. // DatabaseCleanupTargetEdgeHealth is the API cleanup target for OpenResty edge health (connections).
DatabaseCleanupTargetRequestReports = "node_request_reports" DatabaseCleanupTargetEdgeHealth = "node_edge_health"
// DatabaseCleanupTargetObsOpenresty is the API cleanup target for OpenResty observations.
DatabaseCleanupTargetObsOpenresty = "node_obs_openresty"
// DatabaseCleanupTargetObsFrps is the API cleanup target for FRPS observations. // DatabaseCleanupTargetObsFrps is the API cleanup target for FRPS observations.
DatabaseCleanupTargetObsFrps = "node_obs_frps" DatabaseCleanupTargetObsFrps = "node_obs_frps"
// DatabaseCleanupTargetObsFrpc is the API cleanup target for FRPC observations. // DatabaseCleanupTargetObsFrpc is the API cleanup target for FRPC observations.
@@ -33,8 +31,7 @@ const (
var databaseCleanupTargets = map[string]string{ var databaseCleanupTargets = map[string]string{
DatabaseCleanupTargetAccessLogs: "访问日志", DatabaseCleanupTargetAccessLogs: "访问日志",
DatabaseCleanupTargetMetricSnapshots: "性能快照", DatabaseCleanupTargetMetricSnapshots: "性能快照",
DatabaseCleanupTargetRequestReports: "请求聚合", DatabaseCleanupTargetEdgeHealth: "OpenResty 健康(连接)",
DatabaseCleanupTargetObsOpenresty: "OpenResty 观测",
DatabaseCleanupTargetObsFrps: "FRPS 观测", DatabaseCleanupTargetObsFrps: "FRPS 观测",
DatabaseCleanupTargetObsFrpc: "FRPC 观测", DatabaseCleanupTargetObsFrpc: "FRPC 观测",
} }
@@ -43,8 +40,7 @@ var databaseCleanupTargets = map[string]string{
var databaseCleanupTableTTLDays = map[string]int{ var databaseCleanupTableTTLDays = map[string]int{
DatabaseCleanupTargetAccessLogs: analyticsrepo.TableTTLDaysNodeAccessLogs, DatabaseCleanupTargetAccessLogs: analyticsrepo.TableTTLDaysNodeAccessLogs,
DatabaseCleanupTargetMetricSnapshots: analyticsrepo.TableTTLDaysNodeMetricSnapshots, DatabaseCleanupTargetMetricSnapshots: analyticsrepo.TableTTLDaysNodeMetricSnapshots,
DatabaseCleanupTargetRequestReports: analyticsrepo.TableTTLDaysNodeRequestReports, DatabaseCleanupTargetEdgeHealth: analyticsrepo.TableTTLDaysNodeObs,
DatabaseCleanupTargetObsOpenresty: analyticsrepo.TableTTLDaysNodeObs,
DatabaseCleanupTargetObsFrps: analyticsrepo.TableTTLDaysNodeObs, DatabaseCleanupTargetObsFrps: analyticsrepo.TableTTLDaysNodeObs,
DatabaseCleanupTargetObsFrpc: analyticsrepo.TableTTLDaysNodeObs, DatabaseCleanupTargetObsFrpc: analyticsrepo.TableTTLDaysNodeObs,
} }
@@ -165,8 +161,7 @@ func RunDatabaseAutoCleanupOnce(ctx context.Context, now time.Time) (*DatabaseAu
for _, target := range []string{ for _, target := range []string{
DatabaseCleanupTargetAccessLogs, DatabaseCleanupTargetAccessLogs,
DatabaseCleanupTargetMetricSnapshots, DatabaseCleanupTargetMetricSnapshots,
DatabaseCleanupTargetRequestReports, DatabaseCleanupTargetEdgeHealth,
DatabaseCleanupTargetObsOpenresty,
DatabaseCleanupTargetObsFrps, DatabaseCleanupTargetObsFrps,
DatabaseCleanupTargetObsFrpc, DatabaseCleanupTargetObsFrpc,
} { } {
@@ -201,10 +196,8 @@ func deleteAllObservabilityRows(ctx context.Context, target string) (int64, stri
deleted, err = model.DeleteAllOpenFlareAccessLogs(ctx) deleted, err = model.DeleteAllOpenFlareAccessLogs(ctx)
case DatabaseCleanupTargetMetricSnapshots: case DatabaseCleanupTargetMetricSnapshots:
deleted, err = model.DeleteAllOpenFlareMetricSnapshots(ctx) deleted, err = model.DeleteAllOpenFlareMetricSnapshots(ctx)
case DatabaseCleanupTargetRequestReports: case DatabaseCleanupTargetEdgeHealth:
deleted, err = model.DeleteAllOpenFlareRequestReports(ctx) deleted, err = model.DeleteAllOpenFlareEdgeHealth(ctx)
case DatabaseCleanupTargetObsOpenresty:
deleted, err = model.DeleteAllOpenFlareNodeObservationOpenresty(ctx)
case DatabaseCleanupTargetObsFrps: case DatabaseCleanupTargetObsFrps:
deleted, err = model.DeleteAllOpenFlareNodeObservationFrps(ctx) deleted, err = model.DeleteAllOpenFlareNodeObservationFrps(ctx)
case DatabaseCleanupTargetObsFrpc: case DatabaseCleanupTargetObsFrpc:
@@ -236,10 +229,8 @@ func materializeObservabilityTableTTL(ctx context.Context, target string) (int64
eligible, err = model.DeleteOpenFlareAccessLogsBefore(ctx, cutoff) eligible, err = model.DeleteOpenFlareAccessLogsBefore(ctx, cutoff)
case DatabaseCleanupTargetMetricSnapshots: case DatabaseCleanupTargetMetricSnapshots:
eligible, err = model.DeleteOpenFlareMetricSnapshotsBefore(ctx, cutoff) eligible, err = model.DeleteOpenFlareMetricSnapshotsBefore(ctx, cutoff)
case DatabaseCleanupTargetRequestReports: case DatabaseCleanupTargetEdgeHealth:
eligible, err = model.DeleteOpenFlareRequestReportsBefore(ctx, cutoff) eligible, err = model.DeleteOpenFlareEdgeHealthBefore(ctx, cutoff)
case DatabaseCleanupTargetObsOpenresty:
eligible, err = model.DeleteOpenFlareNodeObservationOpenrestyBefore(ctx, cutoff)
case DatabaseCleanupTargetObsFrps: case DatabaseCleanupTargetObsFrps:
eligible, err = model.DeleteOpenFlareNodeObservationFrpsBefore(ctx, cutoff) eligible, err = model.DeleteOpenFlareNodeObservationFrpsBefore(ctx, cutoff)
case DatabaseCleanupTargetObsFrpc: case DatabaseCleanupTargetObsFrpc:
@@ -157,11 +157,11 @@ func TestRunDatabaseAutoCleanupOnceClampsRetentionToTableTTL(t *testing.T) {
CapturedAt: now.Add(-40 * 24 * time.Hour), CapturedAt: now.Add(-40 * 24 * time.Hour),
CPUUsagePercent: 10, CPUUsagePercent: 10,
})) }))
require.NoError(t, model.InsertOpenFlareRequestReport(ctx, &model.OpenFlareRequestReport{ require.NoError(t, model.InsertOpenFlareEdgeHealth(ctx, &model.OpenFlareEdgeHealth{
NodeID: "node-a", NodeID: "node-a",
WindowStartedAt: now.Add(-41 * 24 * time.Hour), CapturedAt: now.Add(-40 * 24 * time.Hour),
WindowEndedAt: now.Add(-40 * 24 * time.Hour), Status: "healthy",
RequestCount: 15, Connections: 2,
})) }))
require.NoError(t, repository.SaveOrUpdateSystemConfig(ctx, model.ConfigKeyDatabaseAutoCleanupEnabled, "true")) require.NoError(t, repository.SaveOrUpdateSystemConfig(ctx, model.ConfigKeyDatabaseAutoCleanupEnabled, "true"))
@@ -170,7 +170,7 @@ func TestRunDatabaseAutoCleanupOnceClampsRetentionToTableTTL(t *testing.T) {
summary, err := RunDatabaseAutoCleanupOnce(ctx, now) summary, err := RunDatabaseAutoCleanupOnce(ctx, now)
require.NoError(t, err) require.NoError(t, err)
require.NotNil(t, summary) require.NotNil(t, summary)
require.Len(t, summary.Results, 6) require.Len(t, summary.Results, 5)
assert.Equal(t, 1, summary.RetentionDays) assert.Equal(t, 1, summary.RetentionDays)
for _, result := range summary.Results { for _, result := range summary.Results {
@@ -189,9 +189,9 @@ func TestRunDatabaseAutoCleanupOnceClampsRetentionToTableTTL(t *testing.T) {
require.NoError(t, err) require.NoError(t, err)
assert.Empty(t, metricSnapshots) assert.Empty(t, metricSnapshots)
requestReports, err := model.ListOpenFlareRequestReportsSince(ctx, "", time.Time{}, 0) edgeHealth, err := model.ListOpenFlareEdgeHealth(ctx, "", time.Time{}, 0)
require.NoError(t, err) require.NoError(t, err)
assert.Empty(t, requestReports) assert.Empty(t, edgeHealth)
} }
func TestTableTTLDaysForCleanupTarget(t *testing.T) { func TestTableTTLDaysForCleanupTarget(t *testing.T) {
@@ -0,0 +1,11 @@
-- +goose Up
-- L1 access log: request body/header length (接收数据) and optional request duration.
ALTER TABLE of_node_access_logs
ADD COLUMN IF NOT EXISTS request_length UInt64 DEFAULT 0;
ALTER TABLE of_node_access_logs
ADD COLUMN IF NOT EXISTS request_time_ms UInt32 DEFAULT 0;
-- +goose Down
ALTER TABLE of_node_access_logs DROP COLUMN IF EXISTS request_time_ms;
ALTER TABLE of_node_access_logs DROP COLUMN IF EXISTS request_length;
@@ -0,0 +1,149 @@
-- +goose Up
-- M5: L2 edge health table, L1 access-log hourly rollup, drop deprecated pre-aggregation tables.
CREATE TABLE IF NOT EXISTS of_node_edge_health
(
id UInt64,
node_id String,
captured_at DateTime64(3, 'UTC'),
status LowCardinality(String),
connections Int64,
created_at DateTime64(3, 'UTC')
)
ENGINE = MergeTree()
PARTITION BY toYYYYMM(captured_at)
ORDER BY (node_id, captured_at, id)
TTL toDateTime(captured_at) + INTERVAL 30 DAY
SETTINGS index_granularity = 8192;
-- Hourly business traffic from access logs (Server-side only; Agent never writes this).
-- SummingMergeTree merges request/error/bytes; UV is not stored (query raw for exact UV).
CREATE TABLE IF NOT EXISTS of_access_log_hourly
(
node_id String,
hour DateTime('UTC'),
host String,
request_count UInt64,
error_count UInt64,
bytes_sent UInt64,
request_length UInt64
)
ENGINE = SummingMergeTree()
PARTITION BY toYYYYMM(hour)
ORDER BY (node_id, hour, host)
TTL hour + INTERVAL 90 DAY
SETTINGS index_granularity = 8192;
CREATE MATERIALIZED VIEW IF NOT EXISTS of_access_log_hourly_mv
TO of_access_log_hourly
AS
SELECT
node_id,
toStartOfHour(logged_at) AS hour,
host,
toUInt64(count()) AS request_count,
toUInt64(countIf(status_code >= 500)) AS error_count,
sum(bytes_sent) AS bytes_sent,
sum(request_length) AS request_length
FROM of_node_access_logs
GROUP BY node_id, hour, host;
-- Deprecated pre-aggregation paths (business traffic is access logs; OR throughput is not authoritative).
DROP VIEW IF EXISTS of_node_traffic_hourly_mv;
DROP TABLE IF EXISTS of_node_traffic_hourly;
DROP VIEW IF EXISTS of_node_openresty_hourly_mv;
DROP TABLE IF EXISTS of_node_openresty_hourly;
DROP TABLE IF EXISTS of_node_request_reports;
DROP TABLE IF EXISTS of_node_obs_openresty;
-- +goose Down
CREATE TABLE IF NOT EXISTS of_node_obs_openresty
(
id UInt64,
node_id String,
captured_at DateTime64(3, 'UTC'),
openresty_rx_bytes Int64,
openresty_tx_bytes Int64,
openresty_connections Int64,
created_at DateTime64(3, 'UTC')
)
ENGINE = MergeTree()
PARTITION BY toYYYYMM(captured_at)
ORDER BY (node_id, captured_at, id)
TTL toDateTime(captured_at) + INTERVAL 30 DAY
SETTINGS index_granularity = 8192;
CREATE TABLE IF NOT EXISTS of_node_request_reports
(
id UInt64,
node_id String,
window_started_at DateTime64(3, 'UTC'),
window_ended_at DateTime64(3, 'UTC'),
request_count Int64,
error_count Int64,
unique_visitor_count Int64,
status_codes_json String,
top_domains_json String,
source_countries_json String,
created_at DateTime64(3, 'UTC')
)
ENGINE = MergeTree()
PARTITION BY toYYYYMM(window_ended_at)
ORDER BY (node_id, window_ended_at, window_started_at, id)
TTL toDateTime(window_ended_at) + INTERVAL 30 DAY
SETTINGS index_granularity = 8192;
CREATE TABLE IF NOT EXISTS of_node_traffic_hourly
(
node_id String,
hour DateTime,
request_count UInt64,
error_count UInt64,
unique_visitor_count UInt64
)
ENGINE = SummingMergeTree()
PARTITION BY toYYYYMM(hour)
ORDER BY (node_id, hour);
CREATE MATERIALIZED VIEW IF NOT EXISTS of_node_traffic_hourly_mv
TO of_node_traffic_hourly
AS
SELECT
node_id,
toStartOfHour(window_ended_at) AS hour,
sum(request_count) AS request_count,
sum(error_count) AS error_count,
max(unique_visitor_count) AS unique_visitor_count
FROM of_node_request_reports
GROUP BY node_id, hour;
CREATE TABLE IF NOT EXISTS of_node_openresty_hourly
(
node_id String,
hour DateTime,
openresty_rx_min SimpleAggregateFunction(min, Int64),
openresty_rx_max SimpleAggregateFunction(max, Int64),
openresty_tx_min SimpleAggregateFunction(min, Int64),
openresty_tx_max SimpleAggregateFunction(max, Int64)
)
ENGINE = AggregatingMergeTree()
PARTITION BY toYYYYMM(hour)
ORDER BY (node_id, hour)
TTL hour + INTERVAL 30 DAY;
CREATE MATERIALIZED VIEW IF NOT EXISTS of_node_openresty_hourly_mv
TO of_node_openresty_hourly
AS
SELECT
node_id,
toStartOfHour(captured_at) AS hour,
min(openresty_rx_bytes) AS openresty_rx_min,
max(openresty_rx_bytes) AS openresty_rx_max,
min(openresty_tx_bytes) AS openresty_tx_min,
max(openresty_tx_bytes) AS openresty_tx_max
FROM of_node_obs_openresty
GROUP BY node_id, hour;
DROP VIEW IF EXISTS of_access_log_hourly_mv;
DROP TABLE IF EXISTS of_access_log_hourly;
DROP TABLE IF EXISTS of_node_edge_health;
@@ -0,0 +1,40 @@
-- +goose Up
-- One-time historical backfill for of_access_log_hourly.
-- MV only ingests rows after creation; without this, 24h charts fall back to raw access logs.
-- ANTI JOIN avoids double-counting (node_id, hour, host) already filled by the live MV.
-- UV is intentionally NOT stored in of_access_log_hourly (SummingMergeTree counts only).
INSERT INTO of_access_log_hourly
SELECT
s.node_id,
toStartOfHour(s.logged_at) AS hour,
s.host,
toUInt64(count()) AS request_count,
toUInt64(countIf(s.status_code >= 500)) AS error_count,
sum(s.bytes_sent) AS bytes_sent,
sum(s.request_length) AS request_length
FROM of_node_access_logs AS s
ANTI JOIN
(
SELECT
node_id,
hour,
host
FROM of_access_log_hourly
GROUP BY
node_id,
hour,
host
) AS existing
ON s.node_id = existing.node_id
AND toStartOfHour(s.logged_at) = existing.hour
AND s.host = existing.host
WHERE s.logged_at >= now() - INTERVAL 90 DAY
GROUP BY
s.node_id,
hour,
s.host;
-- +goose Down
-- Backfill is additive; down does not remove historical rollup rows (TTL still applies).
SELECT 1;
@@ -0,0 +1,27 @@
-- +goose Up
-- 将仍为历史默认值的心跳/离线配置升级为新默认:心跳 3s、离线 60s。
-- 已由管理员改成其他值的配置不受影响。
UPDATE w_system_configs
SET value = '3000',
updated_at = CURRENT_TIMESTAMP
WHERE key = 'agent_heartbeat_interval'
AND value = '10000';
UPDATE w_system_configs
SET value = '60000',
updated_at = CURRENT_TIMESTAMP
WHERE key = 'node_offline_threshold'
AND value = '120000';
-- +goose Down
UPDATE w_system_configs
SET value = '10000',
updated_at = CURRENT_TIMESTAMP
WHERE key = 'agent_heartbeat_interval'
AND value = '3000';
UPDATE w_system_configs
SET value = '120000',
updated_at = CURRENT_TIMESTAMP
WHERE key = 'node_offline_threshold'
AND value = '60000';
@@ -0,0 +1,27 @@
-- +goose Up
-- 将仍为历史默认值的心跳/离线配置升级为新默认:心跳 3s、离线 60s。
-- 已由管理员改成其他值的配置不受影响。
UPDATE w_system_configs
SET value = '3000',
updated_at = CURRENT_TIMESTAMP
WHERE key = 'agent_heartbeat_interval'
AND value = '10000';
UPDATE w_system_configs
SET value = '60000',
updated_at = CURRENT_TIMESTAMP
WHERE key = 'node_offline_threshold'
AND value = '120000';
-- +goose Down
UPDATE w_system_configs
SET value = '10000',
updated_at = CURRENT_TIMESTAMP
WHERE key = 'agent_heartbeat_interval'
AND value = '3000';
UPDATE w_system_configs
SET value = '120000',
updated_at = CURRENT_TIMESTAMP
WHERE key = 'node_offline_threshold'
AND value = '60000';
+13 -11
View File
@@ -10,21 +10,23 @@ import (
const ( const (
nodeAccessLogTableName = "of_node_access_logs" nodeAccessLogTableName = "of_node_access_logs"
nodeAccessLogInsertColumns = "id, node_id, logged_at, remote_addr, region, host, path, status_code, bytes_sent, created_at" nodeAccessLogInsertColumns = "id, node_id, logged_at, remote_addr, region, host, path, status_code, bytes_sent, request_length, request_time_ms, created_at"
) )
// NodeAccessLog stores OpenFlare edge node access records in ClickHouse. // NodeAccessLog stores OpenFlare edge node access records in ClickHouse.
type NodeAccessLog struct { type NodeAccessLog struct {
ID uint64 `gorm:"column:id"` ID uint64 `gorm:"column:id"`
NodeID string `gorm:"column:node_id"` NodeID string `gorm:"column:node_id"`
LoggedAt time.Time `gorm:"column:logged_at"` LoggedAt time.Time `gorm:"column:logged_at"`
RemoteAddr string `gorm:"column:remote_addr"` RemoteAddr string `gorm:"column:remote_addr"`
Region string `gorm:"column:region"` Region string `gorm:"column:region"`
Host string `gorm:"column:host"` Host string `gorm:"column:host"`
Path string `gorm:"column:path"` Path string `gorm:"column:path"`
StatusCode int32 `gorm:"column:status_code"` StatusCode int32 `gorm:"column:status_code"`
BytesSent uint64 `gorm:"column:bytes_sent"` BytesSent uint64 `gorm:"column:bytes_sent"`
CreatedAt time.Time `gorm:"column:created_at"` RequestLength uint64 `gorm:"column:request_length"`
RequestTimeMs uint32 `gorm:"column:request_time_ms"`
CreatedAt time.Time `gorm:"column:created_at"`
} }
// TableName returns the ClickHouse table name. // TableName returns the ClickHouse table name.
@@ -13,6 +13,7 @@ type NodeAccessLogBucketAggregate struct {
UniqueIPCount int64 `gorm:"column:unique_ip_count"` UniqueIPCount int64 `gorm:"column:unique_ip_count"`
UniqueHostCount int64 `gorm:"column:unique_host_count"` UniqueHostCount int64 `gorm:"column:unique_host_count"`
BytesSent int64 `gorm:"column:bytes_sent"` BytesSent int64 `gorm:"column:bytes_sent"`
RequestLength int64 `gorm:"column:request_length"`
} }
// NodeAccessLogWAFIPAggregate is a per-IP aggregate row for WAF automatic rules. // NodeAccessLogWAFIPAggregate is a per-IP aggregate row for WAF automatic rules.
+25 -48
View File
@@ -12,11 +12,8 @@ const (
nodeMetricSnapshotTableName = "of_node_metric_snapshots" nodeMetricSnapshotTableName = "of_node_metric_snapshots"
nodeMetricSnapshotInsertColumns = "id, node_id, captured_at, cpu_usage_percent, memory_used_bytes, memory_total_bytes, storage_used_bytes, storage_total_bytes, disk_read_bytes, disk_write_bytes, network_rx_bytes, network_tx_bytes, created_at" nodeMetricSnapshotInsertColumns = "id, node_id, captured_at, cpu_usage_percent, memory_used_bytes, memory_total_bytes, storage_used_bytes, storage_total_bytes, disk_read_bytes, disk_write_bytes, network_rx_bytes, network_tx_bytes, created_at"
nodeRequestReportTableName = "of_node_request_reports" nodeEdgeHealthTableName = "of_node_edge_health"
nodeRequestReportInsertColumns = "id, node_id, window_started_at, window_ended_at, request_count, error_count, unique_visitor_count, status_codes_json, top_domains_json, source_countries_json, created_at" nodeEdgeHealthInsertColumns = "id, node_id, captured_at, status, connections, created_at"
nodeObsOpenrestyTableName = "of_node_obs_openresty"
nodeObsOpenrestyInsertColumns = "id, node_id, captured_at, openresty_rx_bytes, openresty_tx_bytes, openresty_connections, created_at"
nodeObsFrpsTableName = "of_node_obs_frps" nodeObsFrpsTableName = "of_node_obs_frps"
nodeObsFrpsInsertColumns = "id, node_id, captured_at, frps_connections, frps_proxy_count, frps_client_count, frps_proxies, created_at" nodeObsFrpsInsertColumns = "id, node_id, captured_at, frps_connections, frps_proxy_count, frps_client_count, frps_proxies, created_at"
@@ -57,60 +54,40 @@ func (NodeMetricSnapshot) BatchInsertSQL() string {
return fmt.Sprintf("INSERT INTO %s (%s)", nodeMetricSnapshotTableName, nodeMetricSnapshotInsertColumns) return fmt.Sprintf("INSERT INTO %s (%s)", nodeMetricSnapshotTableName, nodeMetricSnapshotInsertColumns)
} }
// NodeRequestReport stores aggregated node request window reports in ClickHouse. // NodeEdgeHealth stores L2 OpenResty health snapshots (connections + status).
type NodeRequestReport struct { type NodeEdgeHealth struct {
ID uint64 `gorm:"column:id"` ID uint64 `gorm:"column:id"`
NodeID string `gorm:"column:node_id"` NodeID string `gorm:"column:node_id"`
WindowStartedAt time.Time `gorm:"column:window_started_at"` CapturedAt time.Time `gorm:"column:captured_at"`
WindowEndedAt time.Time `gorm:"column:window_ended_at"` Status string `gorm:"column:status"`
RequestCount int64 `gorm:"column:request_count"` Connections int64 `gorm:"column:connections"`
ErrorCount int64 `gorm:"column:error_count"` CreatedAt time.Time `gorm:"column:created_at"`
UniqueVisitorCount int64 `gorm:"column:unique_visitor_count"`
StatusCodesJSON string `gorm:"column:status_codes_json"`
TopDomainsJSON string `gorm:"column:top_domains_json"`
SourceCountriesJSON string `gorm:"column:source_countries_json"`
CreatedAt time.Time `gorm:"column:created_at"`
} }
// TableName returns the ClickHouse table name. // TableName returns the ClickHouse table name.
func (NodeRequestReport) TableName() string { func (NodeEdgeHealth) TableName() string {
return nodeRequestReportTableName return nodeEdgeHealthTableName
} }
// InsertColumns returns comma-separated column names for batch insert. // InsertColumns returns comma-separated column names for batch insert.
func (NodeRequestReport) InsertColumns() string { func (NodeEdgeHealth) InsertColumns() string {
return nodeRequestReportInsertColumns return nodeEdgeHealthInsertColumns
} }
// BatchInsertSQL returns the INSERT prefix used by native batch writers. // BatchInsertSQL returns the INSERT prefix used by native batch writers.
func (NodeRequestReport) BatchInsertSQL() string { func (NodeEdgeHealth) BatchInsertSQL() string {
return fmt.Sprintf("INSERT INTO %s (%s)", nodeRequestReportTableName, nodeRequestReportInsertColumns) return fmt.Sprintf("INSERT INTO %s (%s)", nodeEdgeHealthTableName, nodeEdgeHealthInsertColumns)
} }
// NodeObsOpenresty stores OpenResty observability snapshots in ClickHouse. // AccessLogHourly is a Server-side hourly rollup of access logs.
type NodeObsOpenresty struct { type AccessLogHourly struct {
ID uint64 `gorm:"column:id"` NodeID string `gorm:"column:node_id"`
NodeID string `gorm:"column:node_id"` Hour time.Time `gorm:"column:hour"`
CapturedAt time.Time `gorm:"column:captured_at"` Host string `gorm:"column:host"`
OpenrestyRxBytes int64 `gorm:"column:openresty_rx_bytes"` RequestCount int64 `gorm:"column:request_count"`
OpenrestyTxBytes int64 `gorm:"column:openresty_tx_bytes"` ErrorCount int64 `gorm:"column:error_count"`
OpenrestyConnections int64 `gorm:"column:openresty_connections"` BytesSent int64 `gorm:"column:bytes_sent"`
CreatedAt time.Time `gorm:"column:created_at"` RequestLength int64 `gorm:"column:request_length"`
}
// TableName returns the ClickHouse table name.
func (NodeObsOpenresty) TableName() string {
return nodeObsOpenrestyTableName
}
// InsertColumns returns comma-separated column names for batch insert.
func (NodeObsOpenresty) InsertColumns() string {
return nodeObsOpenrestyInsertColumns
}
// BatchInsertSQL returns the INSERT prefix used by native batch writers.
func (NodeObsOpenresty) BatchInsertSQL() string {
return fmt.Sprintf("INSERT INTO %s (%s)", nodeObsOpenrestyTableName, nodeObsOpenrestyInsertColumns)
} }
// NodeObsFrps stores FRPS observability snapshots in ClickHouse. // NodeObsFrps stores FRPS observability snapshots in ClickHouse.
+16
View File
@@ -72,6 +72,21 @@ func CountOpenFlareAccessLogs(ctx context.Context, query OpenFlareAccessLogQuery
return currentAccessLogStore().Count(ctx, query) return currentAccessLogStore().Count(ctx, query)
} }
// TrafficSummaryOpenFlareAccessLogs returns window-level request/error/UV/bytes summary.
func TrafficSummaryOpenFlareAccessLogs(ctx context.Context, query OpenFlareAccessLogQuery) (OpenFlareAccessLogTrafficSummary, error) {
return currentAccessLogStore().TrafficSummary(ctx, query)
}
// ValueCountsOpenFlareAccessLogs groups logs by status_code or host.
func ValueCountsOpenFlareAccessLogs(ctx context.Context, query OpenFlareAccessLogQuery, column string, limit int) ([]OpenFlareAccessLogValueCount, error) {
return currentAccessLogStore().ValueCounts(ctx, query, column, limit)
}
// NodeAggregatesOpenFlareAccessLogs returns per-node request/error/UV for the window.
func NodeAggregatesOpenFlareAccessLogs(ctx context.Context, query OpenFlareAccessLogQuery) ([]OpenFlareAccessLogNodeAggregate, error) {
return currentAccessLogStore().NodeAggregates(ctx, query)
}
// ListOpenFlareAccessLogRegionCounts returns region counts for access logs. // ListOpenFlareAccessLogRegionCounts returns region counts for access logs.
func ListOpenFlareAccessLogRegionCounts(ctx context.Context, nodeID string, since time.Time, limit int) ([]*OpenFlareAccessLogRegionCount, error) { func ListOpenFlareAccessLogRegionCounts(ctx context.Context, nodeID string, since time.Time, limit int) ([]*OpenFlareAccessLogRegionCount, error) {
return currentAccessLogStore().RegionCounts(ctx, nodeID, since, limit) return currentAccessLogStore().RegionCounts(ctx, nodeID, since, limit)
@@ -192,6 +207,7 @@ func buildOpenFlareAccessLogBucketRows(ctx context.Context, query OpenFlareAcces
ClientErrorCount: partial.ClientErrorCount, ClientErrorCount: partial.ClientErrorCount,
ServerErrorCount: partial.ServerErrorCount, ServerErrorCount: partial.ServerErrorCount,
BytesSent: partial.BytesSent, BytesSent: partial.BytesSent,
RequestLength: partial.RequestLength,
}) })
} }
return rows, nil return rows, nil
+109 -20
View File
@@ -50,11 +50,38 @@ type accessLogStore interface {
IPSummaries(ctx context.Context, filter OpenFlareAccessLogQuery, recentSince time.Time) ([]openFlareAccessLogIPSummaryRow, error) IPSummaries(ctx context.Context, filter OpenFlareAccessLogQuery, recentSince time.Time) ([]openFlareAccessLogIPSummaryRow, error)
CountIPSummaries(ctx context.Context, filter OpenFlareAccessLogQuery) (int64, error) CountIPSummaries(ctx context.Context, filter OpenFlareAccessLogQuery) (int64, error)
IPTrend(ctx context.Context, filter OpenFlareAccessLogQuery, bucketSeconds int64) ([]openFlareAccessLogIPTrendRow, error) IPTrend(ctx context.Context, filter OpenFlareAccessLogQuery, bucketSeconds int64) ([]openFlareAccessLogIPTrendRow, error)
TrafficSummary(ctx context.Context, filter OpenFlareAccessLogQuery) (OpenFlareAccessLogTrafficSummary, error)
ValueCounts(ctx context.Context, filter OpenFlareAccessLogQuery, column string, limit int) ([]OpenFlareAccessLogValueCount, error)
NodeAggregates(ctx context.Context, filter OpenFlareAccessLogQuery) ([]OpenFlareAccessLogNodeAggregate, error)
DeleteAll(ctx context.Context) (int64, error) DeleteAll(ctx context.Context) (int64, error)
DeleteBefore(ctx context.Context, cutoff time.Time) (int64, error) DeleteBefore(ctx context.Context, cutoff time.Time) (int64, error)
DeleteByNodeBefore(ctx context.Context, nodeID string, before time.Time) (int64, error) DeleteByNodeBefore(ctx context.Context, nodeID string, before time.Time) (int64, error)
} }
// OpenFlareAccessLogTrafficSummary is a window-level traffic summary from access logs.
type OpenFlareAccessLogTrafficSummary struct {
RequestCount int64
ErrorCount int64
UniqueIPCount int64
BytesSent int64
RequestLength int64
NodeCount int64
}
// OpenFlareAccessLogValueCount is a dimension value count.
type OpenFlareAccessLogValueCount struct {
Value string
Count int64
}
// OpenFlareAccessLogNodeAggregate is per-node traffic over a window.
type OpenFlareAccessLogNodeAggregate struct {
NodeID string
RequestCount int64
ErrorCount int64
UniqueIPCount int64
}
var ( var (
accessLogStoreMu sync.RWMutex accessLogStoreMu sync.RWMutex
accessLogStoreHolder accessLogStore accessLogStoreHolder accessLogStore
@@ -176,6 +203,50 @@ func (clickhouseAccessLogStore) DeleteByNodeBefore(ctx context.Context, nodeID s
return analyticsrepo.DeleteNodeAccessLogsByNodeBefore(ctx, nodeID, before) return analyticsrepo.DeleteNodeAccessLogsByNodeBefore(ctx, nodeID, before)
} }
func (clickhouseAccessLogStore) TrafficSummary(ctx context.Context, filter OpenFlareAccessLogQuery) (OpenFlareAccessLogTrafficSummary, error) {
row, err := analyticsrepo.TrafficSummaryNodeAccessLogs(ctx, toNodeAccessLogFilter(filter))
if err != nil {
return OpenFlareAccessLogTrafficSummary{}, err
}
return OpenFlareAccessLogTrafficSummary{
RequestCount: row.RequestCount,
ErrorCount: row.ErrorCount,
UniqueIPCount: row.UniqueIPCount,
BytesSent: row.BytesSent,
RequestLength: row.RequestLength,
NodeCount: row.NodeCount,
}, nil
}
func (clickhouseAccessLogStore) ValueCounts(ctx context.Context, filter OpenFlareAccessLogQuery, column string, limit int) ([]OpenFlareAccessLogValueCount, error) {
rows, err := analyticsrepo.ValueCountsNodeAccessLogs(ctx, toNodeAccessLogFilter(filter), column, limit)
if err != nil {
return nil, err
}
result := make([]OpenFlareAccessLogValueCount, len(rows))
for i, row := range rows {
result[i] = OpenFlareAccessLogValueCount{Value: row.Value, Count: row.Count}
}
return result, nil
}
func (clickhouseAccessLogStore) NodeAggregates(ctx context.Context, filter OpenFlareAccessLogQuery) ([]OpenFlareAccessLogNodeAggregate, error) {
rows, err := analyticsrepo.NodeAggregatesNodeAccessLogs(ctx, toNodeAccessLogFilter(filter))
if err != nil {
return nil, err
}
result := make([]OpenFlareAccessLogNodeAggregate, len(rows))
for i, row := range rows {
result[i] = OpenFlareAccessLogNodeAggregate{
NodeID: row.NodeID,
RequestCount: row.RequestCount,
ErrorCount: row.ErrorCount,
UniqueIPCount: row.UniqueIPCount,
}
}
return result, nil
}
func toNodeAccessLogFilter(query OpenFlareAccessLogQuery) analyticsrepo.NodeAccessLogFilter { func toNodeAccessLogFilter(query OpenFlareAccessLogQuery) analyticsrepo.NodeAccessLogFilter {
return analyticsrepo.NodeAccessLogFilter{ return analyticsrepo.NodeAccessLogFilter{
NodeID: query.NodeID, NodeID: query.NodeID,
@@ -197,17 +268,27 @@ func toAnalyticsNodeAccessLog(record *OpenFlareAccessLog) analyticsmodel.NodeAcc
if record.BytesSent > 0 { if record.BytesSent > 0 {
bytesSent = uint64(record.BytesSent) bytesSent = uint64(record.BytesSent)
} }
var requestLength uint64
if record.RequestLength > 0 {
requestLength = uint64(record.RequestLength)
}
var requestTimeMs uint32
if record.RequestTimeMs > 0 && record.RequestTimeMs <= int64(math.MaxUint32) {
requestTimeMs = uint32(record.RequestTimeMs)
}
return analyticsmodel.NodeAccessLog{ return analyticsmodel.NodeAccessLog{
ID: record.ID, ID: record.ID,
NodeID: record.NodeID, NodeID: record.NodeID,
LoggedAt: record.LoggedAt, LoggedAt: record.LoggedAt,
RemoteAddr: record.RemoteAddr, RemoteAddr: record.RemoteAddr,
Region: record.Region, Region: record.Region,
Host: record.Host, Host: record.Host,
Path: record.Path, Path: record.Path,
StatusCode: openFlareAccessLogStatusCodeToInt32(record.StatusCode), StatusCode: openFlareAccessLogStatusCodeToInt32(record.StatusCode),
BytesSent: bytesSent, BytesSent: bytesSent,
CreatedAt: record.CreatedAt, RequestLength: requestLength,
RequestTimeMs: requestTimeMs,
CreatedAt: record.CreatedAt,
} }
} }
@@ -220,17 +301,25 @@ func fromAnalyticsNodeAccessLogs(rows []analyticsmodel.NodeAccessLog) []*OpenFla
} else { } else {
bytesSent = math.MaxInt64 bytesSent = math.MaxInt64
} }
var requestLength int64
if row.RequestLength <= math.MaxInt64 {
requestLength = int64(row.RequestLength)
} else {
requestLength = math.MaxInt64
}
result[index] = &OpenFlareAccessLog{ result[index] = &OpenFlareAccessLog{
ID: row.ID, ID: row.ID,
NodeID: row.NodeID, NodeID: row.NodeID,
LoggedAt: row.LoggedAt, LoggedAt: row.LoggedAt,
RemoteAddr: row.RemoteAddr, RemoteAddr: row.RemoteAddr,
Region: row.Region, Region: row.Region,
Host: row.Host, Host: row.Host,
Path: row.Path, Path: row.Path,
StatusCode: int(row.StatusCode), StatusCode: int(row.StatusCode),
BytesSent: bytesSent, BytesSent: bytesSent,
CreatedAt: row.CreatedAt, RequestLength: requestLength,
RequestTimeMs: int64(row.RequestTimeMs),
CreatedAt: row.CreatedAt,
} }
} }
return result return result
@@ -9,6 +9,7 @@ import (
"net/http" "net/http"
"net/netip" "net/netip"
"sort" "sort"
"strconv"
"strings" "strings"
"sync" "sync"
"time" "time"
@@ -16,6 +17,11 @@ import (
"github.com/Rain-kl/Wavelet/internal/db/idgen" "github.com/Rain-kl/Wavelet/internal/db/idgen"
) )
const (
accessLogColumnStatusCode = "status_code"
accessLogColumnHost = "host"
)
type memoryAccessLogStore struct { type memoryAccessLogStore struct {
mu sync.RWMutex mu sync.RWMutex
records []*OpenFlareAccessLog records []*OpenFlareAccessLog
@@ -123,6 +129,7 @@ func (s *memoryAccessLogStore) BucketAggregates(_ context.Context, filter OpenFl
} }
item.RequestCount++ item.RequestCount++
item.BytesSent += row.BytesSent item.BytesSent += row.BytesSent
item.RequestLength += row.RequestLength
switch { switch {
case row.StatusCode < 400: case row.StatusCode < 400:
item.SuccessCount++ item.SuccessCount++
@@ -155,6 +162,7 @@ func (s *memoryAccessLogStore) BucketAggregates(_ context.Context, filter OpenFl
ClientErrorCount: result[index].ClientErrorCount, ClientErrorCount: result[index].ClientErrorCount,
ServerErrorCount: result[index].ServerErrorCount, ServerErrorCount: result[index].ServerErrorCount,
BytesSent: result[index].BytesSent, BytesSent: result[index].BytesSent,
RequestLength: result[index].RequestLength,
} }
} }
sortOpenFlareAccessLogBucketRows(bucketRows, filter.SortBy, filter.SortOrder) sortOpenFlareAccessLogBucketRows(bucketRows, filter.SortBy, filter.SortOrder)
@@ -168,6 +176,7 @@ func (s *memoryAccessLogStore) BucketAggregates(_ context.Context, filter OpenFl
ClientErrorCount: bucketRows[index].ClientErrorCount, ClientErrorCount: bucketRows[index].ClientErrorCount,
ServerErrorCount: bucketRows[index].ServerErrorCount, ServerErrorCount: bucketRows[index].ServerErrorCount,
BytesSent: bucketRows[index].BytesSent, BytesSent: bucketRows[index].BytesSent,
RequestLength: bucketRows[index].RequestLength,
} }
} }
if filter.PageSize > 0 { if filter.PageSize > 0 {
@@ -434,6 +443,113 @@ func (s *memoryAccessLogStore) DeleteByNodeBefore(_ context.Context, nodeID stri
return deleted, nil return deleted, nil
} }
func (s *memoryAccessLogStore) TrafficSummary(_ context.Context, filter OpenFlareAccessLogQuery) (OpenFlareAccessLogTrafficSummary, error) {
s.mu.RLock()
defer s.mu.RUnlock()
rows := s.filterRecords(filter)
ips := make(map[string]struct{})
nodes := make(map[string]struct{})
var summary OpenFlareAccessLogTrafficSummary
for _, row := range rows {
summary.RequestCount++
summary.BytesSent += row.BytesSent
summary.RequestLength += row.RequestLength
if row.StatusCode >= http.StatusInternalServerError {
summary.ErrorCount++
}
if ip := strings.TrimSpace(row.RemoteAddr); ip != "" {
ips[ip] = struct{}{}
}
if id := strings.TrimSpace(row.NodeID); id != "" {
nodes[id] = struct{}{}
}
}
summary.UniqueIPCount = int64(len(ips))
summary.NodeCount = int64(len(nodes))
return summary, nil
}
func (s *memoryAccessLogStore) ValueCounts(_ context.Context, filter OpenFlareAccessLogQuery, column string, limit int) ([]OpenFlareAccessLogValueCount, error) {
s.mu.RLock()
defer s.mu.RUnlock()
col := strings.TrimSpace(strings.ToLower(column))
if col != accessLogColumnStatusCode && col != accessLogColumnHost {
return nil, nil
}
rows := s.filterRecords(filter)
counts := make(map[string]int64)
for _, row := range rows {
var value string
if col == accessLogColumnStatusCode {
value = strconv.Itoa(row.StatusCode)
} else {
value = strings.TrimSpace(row.Host)
}
if value == "" {
continue
}
counts[value]++
}
result := make([]OpenFlareAccessLogValueCount, 0, len(counts))
for value, count := range counts {
result = append(result, OpenFlareAccessLogValueCount{Value: value, Count: count})
}
sort.Slice(result, func(i, j int) bool {
if result[i].Count == result[j].Count {
return result[i].Value < result[j].Value
}
return result[i].Count > result[j].Count
})
if limit > 0 && len(result) > limit {
result = result[:limit]
}
return result, nil
}
func (s *memoryAccessLogStore) NodeAggregates(_ context.Context, filter OpenFlareAccessLogQuery) ([]OpenFlareAccessLogNodeAggregate, error) {
s.mu.RLock()
defer s.mu.RUnlock()
rows := s.filterRecords(filter)
type acc struct {
OpenFlareAccessLogNodeAggregate
ips map[string]struct{}
}
byNode := make(map[string]*acc)
for _, row := range rows {
id := strings.TrimSpace(row.NodeID)
if id == "" {
continue
}
item := byNode[id]
if item == nil {
item = &acc{
OpenFlareAccessLogNodeAggregate: OpenFlareAccessLogNodeAggregate{NodeID: id},
ips: make(map[string]struct{}),
}
byNode[id] = item
}
item.RequestCount++
if row.StatusCode >= http.StatusInternalServerError {
item.ErrorCount++
}
if ip := strings.TrimSpace(row.RemoteAddr); ip != "" {
item.ips[ip] = struct{}{}
}
}
result := make([]OpenFlareAccessLogNodeAggregate, 0, len(byNode))
for _, item := range byNode {
item.UniqueIPCount = int64(len(item.ips))
result = append(result, item.OpenFlareAccessLogNodeAggregate)
}
sort.Slice(result, func(i, j int) bool {
if result[i].RequestCount == result[j].RequestCount {
return result[i].NodeID < result[j].NodeID
}
return result[i].RequestCount > result[j].RequestCount
})
return result, nil
}
func (s *memoryAccessLogStore) filterRecords(query OpenFlareAccessLogQuery) []*OpenFlareAccessLog { func (s *memoryAccessLogStore) filterRecords(query OpenFlareAccessLogQuery) []*OpenFlareAccessLog {
result := make([]*OpenFlareAccessLog, 0, len(s.records)) result := make([]*OpenFlareAccessLog, 0, len(s.records))
for _, row := range s.records { for _, row := range s.records {
+72 -140
View File
@@ -37,40 +37,21 @@ func (OpenFlareMetricSnapshot) TableName() string {
return "of_node_metric_snapshots" return "of_node_metric_snapshots"
} }
// OpenFlareRequestReport stores aggregated traffic windows per node in ClickHouse (database: openflare, table: of_node_request_reports).
// ClickHouse DDL is managed by goose; reads/writes go through internal/repository/analytics.
type OpenFlareRequestReport struct {
ID uint `json:"id" gorm:"primaryKey;autoIncrement"`
NodeID string `json:"node_id" gorm:"index;size:64;not null"`
WindowStartedAt time.Time `json:"window_started_at" gorm:"index"`
WindowEndedAt time.Time `json:"window_ended_at" gorm:"index"`
RequestCount int64 `json:"request_count"`
ErrorCount int64 `json:"error_count"`
UniqueVisitorCount int64 `json:"unique_visitor_count"`
StatusCodesJSON string `json:"status_codes_json" gorm:"type:text"`
TopDomainsJSON string `json:"top_domains_json" gorm:"type:text"`
SourceCountriesJSON string `json:"source_countries_json" gorm:"type:text"`
CreatedAt time.Time `json:"created_at" gorm:"autoCreateTime"`
}
// TableName returns the GORM table name.
func (OpenFlareRequestReport) TableName() string {
return "of_node_request_reports"
}
// OpenFlareAccessLog stores a single access log row in ClickHouse (database: openflare, table: of_node_access_logs). // OpenFlareAccessLog stores a single access log row in ClickHouse (database: openflare, table: of_node_access_logs).
// ClickHouse DDL is managed by goose; reads/writes go through internal/repository/analytics. // ClickHouse DDL is managed by goose; reads/writes go through internal/repository/analytics.
type OpenFlareAccessLog struct { type OpenFlareAccessLog struct {
ID uint64 `json:"id,string" gorm:"column:id"` ID uint64 `json:"id,string" gorm:"column:id"`
NodeID string `json:"node_id" gorm:"index;size:64;not null"` NodeID string `json:"node_id" gorm:"index;size:64;not null"`
LoggedAt time.Time `json:"logged_at" gorm:"index"` LoggedAt time.Time `json:"logged_at" gorm:"index"`
RemoteAddr string `json:"remote_addr" gorm:"index;size:128"` RemoteAddr string `json:"remote_addr" gorm:"index;size:128"`
Region string `json:"region" gorm:"size:128"` Region string `json:"region" gorm:"size:128"`
Host string `json:"host" gorm:"index;size:255"` Host string `json:"host" gorm:"index;size:255"`
Path string `json:"path" gorm:"size:2048"` Path string `json:"path" gorm:"size:2048"`
StatusCode int `json:"status_code" gorm:"index"` StatusCode int `json:"status_code" gorm:"index"`
BytesSent int64 `json:"bytes_sent" gorm:"column:bytes_sent;not null;default:0"` BytesSent int64 `json:"bytes_sent" gorm:"column:bytes_sent;not null;default:0"`
CreatedAt time.Time `json:"created_at" gorm:"autoCreateTime"` RequestLength int64 `json:"request_length" gorm:"column:request_length;not null;default:0"`
RequestTimeMs int64 `json:"request_time_ms" gorm:"column:request_time_ms;not null;default:0"`
CreatedAt time.Time `json:"created_at" gorm:"autoCreateTime"`
} }
// TableName returns the GORM table name. // TableName returns the GORM table name.
@@ -130,21 +111,19 @@ func (OpenFlareNodeSystemProfile) TableName() string {
return "of_node_system_profiles" return "of_node_system_profiles"
} }
// OpenFlareNodeObservationOpenresty stores openresty network observations in ClickHouse (database: openflare, table: of_node_obs_openresty). // OpenFlareEdgeHealth is L2 OpenResty health (of_node_edge_health).
// ClickHouse DDL is managed by goose; reads/writes go through internal/repository/analytics. type OpenFlareEdgeHealth struct {
type OpenFlareNodeObservationOpenresty struct { ID uint `json:"id"`
ID uint `json:"id" gorm:"primaryKey;autoIncrement"` NodeID string `json:"node_id"`
NodeID string `json:"node_id" gorm:"index;size:64;not null"` CapturedAt time.Time `json:"captured_at"`
CapturedAt time.Time `json:"captured_at" gorm:"index"` Status string `json:"status"`
OpenrestyRxBytes int64 `json:"openresty_rx_bytes"` Connections int64 `json:"connections"`
OpenrestyTxBytes int64 `json:"openresty_tx_bytes"` CreatedAt time.Time `json:"created_at"`
OpenrestyConnections int64 `json:"openresty_connections"`
CreatedAt time.Time `json:"created_at" gorm:"autoCreateTime"`
} }
// TableName returns the GORM table name. // TableName returns the ClickHouse table name.
func (OpenFlareNodeObservationOpenresty) TableName() string { func (OpenFlareEdgeHealth) TableName() string {
return "of_node_obs_openresty" return "of_node_edge_health"
} }
// OpenFlareNodeObservationFrpc stores tunnel client frpc observations in ClickHouse (database: openflare, table: of_node_obs_frpc). // OpenFlareNodeObservationFrpc stores tunnel client frpc observations in ClickHouse (database: openflare, table: of_node_obs_frpc).
@@ -223,6 +202,7 @@ type OpenFlareAccessLogBucketRow struct {
ClientErrorCount int64 `json:"client_error_count"` ClientErrorCount int64 `json:"client_error_count"`
ServerErrorCount int64 `json:"server_error_count"` ServerErrorCount int64 `json:"server_error_count"`
BytesSent int64 `json:"bytes_sent"` BytesSent int64 `json:"bytes_sent"`
RequestLength int64 `json:"request_length"`
} }
// OpenFlareAccessLogBucketIPQuery filters folded IP summary queries (v1 stub). // OpenFlareAccessLogBucketIPQuery filters folded IP summary queries (v1 stub).
@@ -314,14 +294,9 @@ func InsertOpenFlareMetricSnapshot(ctx context.Context, record *OpenFlareMetricS
return currentObservabilityStore().InsertMetricSnapshot(ctx, record) return currentObservabilityStore().InsertMetricSnapshot(ctx, record)
} }
// InsertOpenFlareRequestReport inserts a request report into ClickHouse. // InsertOpenFlareEdgeHealth inserts an L2 edge health snapshot into ClickHouse.
func InsertOpenFlareRequestReport(ctx context.Context, record *OpenFlareRequestReport) error { func InsertOpenFlareEdgeHealth(ctx context.Context, record *OpenFlareEdgeHealth) error {
return currentObservabilityStore().InsertRequestReport(ctx, record) return currentObservabilityStore().InsertEdgeHealth(ctx, record)
}
// InsertOpenFlareNodeObservationOpenresty inserts an OpenResty observation into ClickHouse.
func InsertOpenFlareNodeObservationOpenresty(ctx context.Context, record *OpenFlareNodeObservationOpenresty) error {
return currentObservabilityStore().InsertNodeObservationOpenresty(ctx, record)
} }
// InsertOpenFlareNodeObservationFrps inserts an FRPS observation into ClickHouse. // InsertOpenFlareNodeObservationFrps inserts an FRPS observation into ClickHouse.
@@ -357,28 +332,6 @@ func ListOpenFlareLatestMetricSnapshotsSince(ctx context.Context, nodeID string,
return openFlareLatestMetricSnapshots(all), nil return openFlareLatestMetricSnapshots(all), nil
} }
// ListOpenFlareRequestReportsSince returns request reports since the given time.
func ListOpenFlareRequestReportsSince(ctx context.Context, nodeID string, since time.Time, limit int) ([]*OpenFlareRequestReport, error) {
return currentObservabilityStore().ListRequestReports(ctx, nodeID, since, limit)
}
// ListOpenFlareLatestRequestReportsSince returns the latest request report per node.
// Prefer ClickHouse LIMIT 1 BY; on CH unavailability fall back to store list + reduce.
func ListOpenFlareLatestRequestReportsSince(ctx context.Context, nodeID string, since time.Time) ([]*OpenFlareRequestReport, error) {
rows, err := analyticsrepo.ListLatestNodeRequestReports(ctx, analyticsrepo.NodeObservabilityFilter{
NodeID: nodeID,
Since: since,
})
if err == nil {
return fromAnalyticsNodeRequestReports(rows), nil
}
all, listErr := ListOpenFlareRequestReportsSince(ctx, nodeID, since, 0)
if listErr != nil {
return nil, err
}
return openFlareLatestRequestReports(all), nil
}
func openFlareLatestMetricSnapshots(snapshots []*OpenFlareMetricSnapshot) []*OpenFlareMetricSnapshot { func openFlareLatestMetricSnapshots(snapshots []*OpenFlareMetricSnapshot) []*OpenFlareMetricSnapshot {
latestByNode := make(map[string]*OpenFlareMetricSnapshot, len(snapshots)) latestByNode := make(map[string]*OpenFlareMetricSnapshot, len(snapshots))
for _, snapshot := range snapshots { for _, snapshot := range snapshots {
@@ -397,24 +350,6 @@ func openFlareLatestMetricSnapshots(snapshots []*OpenFlareMetricSnapshot) []*Ope
return result return result
} }
func openFlareLatestRequestReports(reports []*OpenFlareRequestReport) []*OpenFlareRequestReport {
latestByNode := make(map[string]*OpenFlareRequestReport, len(reports))
for _, report := range reports {
if report == nil || report.NodeID == "" {
continue
}
if existing, ok := latestByNode[report.NodeID]; ok && !report.WindowEndedAt.After(existing.WindowEndedAt) {
continue
}
latestByNode[report.NodeID] = report
}
result := make([]*OpenFlareRequestReport, 0, len(latestByNode))
for _, report := range latestByNode {
result = append(result, report)
}
return result
}
// OpenFlareTrafficHourly is an hourly traffic rollup row. // OpenFlareTrafficHourly is an hourly traffic rollup row.
type OpenFlareTrafficHourly struct { type OpenFlareTrafficHourly struct {
NodeID string `json:"node_id"` NodeID string `json:"node_id"`
@@ -425,6 +360,7 @@ type OpenFlareTrafficHourly struct {
} }
// ListOpenFlareTrafficHourlySince returns hourly traffic rollup rows since the given time. // ListOpenFlareTrafficHourlySince returns hourly traffic rollup rows since the given time.
// Source: of_access_log_hourly (M5).
func ListOpenFlareTrafficHourlySince(ctx context.Context, nodeID string, since time.Time) ([]*OpenFlareTrafficHourly, error) { func ListOpenFlareTrafficHourlySince(ctx context.Context, nodeID string, since time.Time) ([]*OpenFlareTrafficHourly, error) {
rows, err := analyticsrepo.ListNodeTrafficHourly(ctx, analyticsrepo.NodeObservabilityFilter{ rows, err := analyticsrepo.ListNodeTrafficHourly(ctx, analyticsrepo.NodeObservabilityFilter{
NodeID: nodeID, NodeID: nodeID,
@@ -446,6 +382,41 @@ func ListOpenFlareTrafficHourlySince(ctx context.Context, nodeID string, since t
return result, nil return result, nil
} }
// OpenFlareAccessLogHourly is a per-node/host hourly access log rollup.
type OpenFlareAccessLogHourly struct {
NodeID string `json:"node_id"`
Hour time.Time `json:"hour"`
Host string `json:"host"`
RequestCount int64 `json:"request_count"`
ErrorCount int64 `json:"error_count"`
BytesSent int64 `json:"bytes_sent"`
RequestLength int64 `json:"request_length"`
}
// ListOpenFlareAccessLogHourlySince returns of_access_log_hourly rows since the given time.
func ListOpenFlareAccessLogHourlySince(ctx context.Context, nodeID string, since time.Time) ([]*OpenFlareAccessLogHourly, error) {
rows, err := analyticsrepo.ListAccessLogHourly(ctx, analyticsrepo.NodeObservabilityFilter{
NodeID: nodeID,
Since: since,
})
if err != nil {
return nil, err
}
result := make([]*OpenFlareAccessLogHourly, len(rows))
for index, row := range rows {
result[index] = &OpenFlareAccessLogHourly{
NodeID: row.NodeID,
Hour: row.Hour,
Host: row.Host,
RequestCount: row.RequestCount,
ErrorCount: row.ErrorCount,
BytesSent: row.BytesSent,
RequestLength: row.RequestLength,
}
}
return result, nil
}
// OpenFlareMetricHourly is an hourly metric snapshot aggregation row. // OpenFlareMetricHourly is an hourly metric snapshot aggregation row.
type OpenFlareMetricHourly struct { type OpenFlareMetricHourly struct {
Hour time.Time `json:"hour"` Hour time.Time `json:"hour"`
@@ -458,14 +429,6 @@ type OpenFlareMetricHourly struct {
ReportedNodes int `json:"reported_nodes"` ReportedNodes int `json:"reported_nodes"`
} }
// OpenFlareOpenrestyHourly is an hourly OpenResty observation aggregation row.
type OpenFlareOpenrestyHourly struct {
Hour time.Time `json:"hour"`
OpenrestyRxBytes int64 `json:"openresty_rx_bytes"`
OpenrestyTxBytes int64 `json:"openresty_tx_bytes"`
ReportedNodes int `json:"reported_nodes"`
}
// ListOpenFlareMetricHourlySince returns hourly metric aggregates since the given time. // ListOpenFlareMetricHourlySince returns hourly metric aggregates since the given time.
func ListOpenFlareMetricHourlySince(ctx context.Context, nodeID string, since time.Time) ([]*OpenFlareMetricHourly, error) { func ListOpenFlareMetricHourlySince(ctx context.Context, nodeID string, since time.Time) ([]*OpenFlareMetricHourly, error) {
rows, err := analyticsrepo.ListNodeMetricHourly(ctx, analyticsrepo.NodeObservabilityFilter{ rows, err := analyticsrepo.ListNodeMetricHourly(ctx, analyticsrepo.NodeObservabilityFilter{
@@ -491,27 +454,6 @@ func ListOpenFlareMetricHourlySince(ctx context.Context, nodeID string, since ti
return result, nil return result, nil
} }
// ListOpenFlareOpenrestyHourlySince returns hourly OpenResty aggregates since the given time.
func ListOpenFlareOpenrestyHourlySince(ctx context.Context, nodeID string, since time.Time) ([]*OpenFlareOpenrestyHourly, error) {
rows, err := analyticsrepo.ListNodeOpenrestyHourly(ctx, analyticsrepo.NodeObservabilityFilter{
NodeID: nodeID,
Since: since,
})
if err != nil {
return nil, err
}
result := make([]*OpenFlareOpenrestyHourly, len(rows))
for index, row := range rows {
result[index] = &OpenFlareOpenrestyHourly{
Hour: row.Hour,
OpenrestyRxBytes: row.OpenrestyRxBytes,
OpenrestyTxBytes: row.OpenrestyTxBytes,
ReportedNodes: row.ReportedNodes,
}
}
return result, nil
}
// ListOpenFlareActiveHealthEvents returns active health events across all nodes. // ListOpenFlareActiveHealthEvents returns active health events across all nodes.
func ListOpenFlareActiveHealthEvents(ctx context.Context) ([]*OpenFlareHealthEvent, error) { func ListOpenFlareActiveHealthEvents(ctx context.Context) ([]*OpenFlareHealthEvent, error) {
conn := db.DB(ctx) conn := db.DB(ctx)
@@ -561,24 +503,14 @@ func DeleteAllOpenFlareMetricSnapshots(ctx context.Context) (int64, error) {
return currentObservabilityStore().DeleteAllMetricSnapshots(ctx) return currentObservabilityStore().DeleteAllMetricSnapshots(ctx)
} }
// DeleteOpenFlareRequestReportsBefore deletes request reports ending before cutoff. // DeleteOpenFlareEdgeHealthBefore deletes edge health rows captured before cutoff.
func DeleteOpenFlareRequestReportsBefore(ctx context.Context, cutoff time.Time) (int64, error) { func DeleteOpenFlareEdgeHealthBefore(ctx context.Context, cutoff time.Time) (int64, error) {
return currentObservabilityStore().DeleteRequestReportsBefore(ctx, cutoff) return currentObservabilityStore().DeleteEdgeHealthBefore(ctx, cutoff)
} }
// DeleteAllOpenFlareRequestReports deletes all request reports. // DeleteAllOpenFlareEdgeHealth deletes all edge health snapshots.
func DeleteAllOpenFlareRequestReports(ctx context.Context) (int64, error) { func DeleteAllOpenFlareEdgeHealth(ctx context.Context) (int64, error) {
return currentObservabilityStore().DeleteAllRequestReports(ctx) return currentObservabilityStore().DeleteAllEdgeHealth(ctx)
}
// DeleteOpenFlareNodeObservationOpenrestyBefore deletes OpenResty observations captured before cutoff.
func DeleteOpenFlareNodeObservationOpenrestyBefore(ctx context.Context, cutoff time.Time) (int64, error) {
return currentObservabilityStore().DeleteNodeObservationOpenrestyBefore(ctx, cutoff)
}
// DeleteAllOpenFlareNodeObservationOpenresty deletes all OpenResty observations.
func DeleteAllOpenFlareNodeObservationOpenresty(ctx context.Context) (int64, error) {
return currentObservabilityStore().DeleteAllNodeObservationOpenresty(ctx)
} }
// DeleteOpenFlareNodeObservationFrpsBefore deletes FRPS observations captured before cutoff. // DeleteOpenFlareNodeObservationFrpsBefore deletes FRPS observations captured before cutoff.
@@ -633,9 +565,9 @@ func GetOpenFlareNodeSystemProfile(ctx context.Context, nodeID string) (*OpenFla
return &profile, nil return &profile, nil
} }
// ListOpenFlareNodeObservationOpenresty returns openresty observations. // ListOpenFlareEdgeHealth returns L2 edge health snapshots.
func ListOpenFlareNodeObservationOpenresty(ctx context.Context, nodeID string, since time.Time, limit int) ([]*OpenFlareNodeObservationOpenresty, error) { func ListOpenFlareEdgeHealth(ctx context.Context, nodeID string, since time.Time, limit int) ([]*OpenFlareEdgeHealth, error) {
return currentObservabilityStore().ListNodeObservationOpenresty(ctx, nodeID, since, limit) return currentObservabilityStore().ListEdgeHealth(ctx, nodeID, since, limit)
} }
// ListOpenFlareNodeObservationFrpc returns frpc observations. // ListOpenFlareNodeObservationFrpc returns frpc observations.
+46 -105
View File
@@ -6,6 +6,7 @@ package model
import ( import (
"context" "context"
"math" "math"
"strings"
"sync" "sync"
"time" "time"
@@ -16,11 +17,10 @@ import (
// ObservabilityInsertHooks queues observability rows for async ClickHouse write. // ObservabilityInsertHooks queues observability rows for async ClickHouse write.
// Wired from openflare/chwriter.Init so model never imports the apps layer. // Wired from openflare/chwriter.Init so model never imports the apps layer.
type ObservabilityInsertHooks struct { type ObservabilityInsertHooks struct {
QueueMetricSnapshot func(analyticsmodel.NodeMetricSnapshot) QueueMetricSnapshot func(analyticsmodel.NodeMetricSnapshot)
QueueRequestReport func(analyticsmodel.NodeRequestReport) QueueEdgeHealth func(analyticsmodel.NodeEdgeHealth)
QueueOpenrestyObservation func(analyticsmodel.NodeObsOpenresty) QueueFrpsObservation func(analyticsmodel.NodeObsFrps)
QueueFrpsObservation func(analyticsmodel.NodeObsFrps) QueueFrpcObservation func(analyticsmodel.NodeObsFrpc)
QueueFrpcObservation func(analyticsmodel.NodeObsFrpc)
} }
var ( var (
@@ -47,15 +47,10 @@ type observabilityStore interface {
DeleteAllMetricSnapshots(ctx context.Context) (int64, error) DeleteAllMetricSnapshots(ctx context.Context) (int64, error)
DeleteMetricSnapshotsBefore(ctx context.Context, cutoff time.Time) (int64, error) DeleteMetricSnapshotsBefore(ctx context.Context, cutoff time.Time) (int64, error)
InsertRequestReport(ctx context.Context, record *OpenFlareRequestReport) error InsertEdgeHealth(ctx context.Context, record *OpenFlareEdgeHealth) error
ListRequestReports(ctx context.Context, nodeID string, since time.Time, limit int) ([]*OpenFlareRequestReport, error) ListEdgeHealth(ctx context.Context, nodeID string, since time.Time, limit int) ([]*OpenFlareEdgeHealth, error)
DeleteAllRequestReports(ctx context.Context) (int64, error) DeleteAllEdgeHealth(ctx context.Context) (int64, error)
DeleteRequestReportsBefore(ctx context.Context, cutoff time.Time) (int64, error) DeleteEdgeHealthBefore(ctx context.Context, cutoff time.Time) (int64, error)
InsertNodeObservationOpenresty(ctx context.Context, record *OpenFlareNodeObservationOpenresty) error
ListNodeObservationOpenresty(ctx context.Context, nodeID string, since time.Time, limit int) ([]*OpenFlareNodeObservationOpenresty, error)
DeleteAllNodeObservationOpenresty(ctx context.Context) (int64, error)
DeleteNodeObservationOpenrestyBefore(ctx context.Context, cutoff time.Time) (int64, error)
InsertNodeObservationFrps(ctx context.Context, record *OpenFlareNodeObservationFrps) error InsertNodeObservationFrps(ctx context.Context, record *OpenFlareNodeObservationFrps) error
ListNodeObservationFrps(ctx context.Context, nodeID string, since time.Time, limit int) ([]*OpenFlareNodeObservationFrps, error) ListNodeObservationFrps(ctx context.Context, nodeID string, since time.Time, limit int) ([]*OpenFlareNodeObservationFrps, error)
@@ -128,56 +123,40 @@ func (clickhouseObservabilityStore) DeleteMetricSnapshotsBefore(ctx context.Cont
return analyticsrepo.DeleteNodeMetricSnapshotsBefore(ctx, cutoff) return analyticsrepo.DeleteNodeMetricSnapshotsBefore(ctx, cutoff)
} }
func (clickhouseObservabilityStore) InsertRequestReport(_ context.Context, record *OpenFlareRequestReport) error { const edgeHealthStatusUnknown = "unknown"
func normalizeEdgeHealthStatus(status string) string {
status = strings.TrimSpace(status)
if status == "" {
return edgeHealthStatusUnknown
}
return status
}
func (clickhouseObservabilityStore) InsertEdgeHealth(_ context.Context, record *OpenFlareEdgeHealth) error {
if record == nil { if record == nil {
return nil return nil
} }
if hook := currentObservabilityInsertHooks().QueueRequestReport; hook != nil { if hook := currentObservabilityInsertHooks().QueueEdgeHealth; hook != nil {
hook(toAnalyticsNodeRequestReport(record)) hook(toAnalyticsNodeEdgeHealth(record))
} }
return nil return nil
} }
func (clickhouseObservabilityStore) ListRequestReports(ctx context.Context, nodeID string, since time.Time, limit int) ([]*OpenFlareRequestReport, error) { func (clickhouseObservabilityStore) ListEdgeHealth(ctx context.Context, nodeID string, since time.Time, limit int) ([]*OpenFlareEdgeHealth, error) {
rows, err := analyticsrepo.ListNodeRequestReports(ctx, toNodeObservabilityFilter(nodeID, since, limit)) rows, err := analyticsrepo.ListNodeEdgeHealth(ctx, toNodeObservabilityFilter(nodeID, since, limit))
if err != nil { if err != nil {
return nil, err return nil, err
} }
return fromAnalyticsNodeRequestReports(rows), nil return fromAnalyticsNodeEdgeHealth(rows), nil
} }
func (clickhouseObservabilityStore) DeleteAllRequestReports(ctx context.Context) (int64, error) { func (clickhouseObservabilityStore) DeleteAllEdgeHealth(ctx context.Context) (int64, error) {
return analyticsrepo.DeleteAllNodeRequestReports(ctx) return analyticsrepo.DeleteAllNodeEdgeHealth(ctx)
} }
func (clickhouseObservabilityStore) DeleteRequestReportsBefore(ctx context.Context, cutoff time.Time) (int64, error) { func (clickhouseObservabilityStore) DeleteEdgeHealthBefore(ctx context.Context, cutoff time.Time) (int64, error) {
return analyticsrepo.DeleteNodeRequestReportsBefore(ctx, cutoff) return analyticsrepo.DeleteNodeEdgeHealthBefore(ctx, cutoff)
}
func (clickhouseObservabilityStore) InsertNodeObservationOpenresty(_ context.Context, record *OpenFlareNodeObservationOpenresty) error {
if record == nil {
return nil
}
if hook := currentObservabilityInsertHooks().QueueOpenrestyObservation; hook != nil {
hook(toAnalyticsNodeObsOpenresty(record))
}
return nil
}
func (clickhouseObservabilityStore) ListNodeObservationOpenresty(ctx context.Context, nodeID string, since time.Time, limit int) ([]*OpenFlareNodeObservationOpenresty, error) {
rows, err := analyticsrepo.ListNodeObsOpenresty(ctx, toNodeObservabilityFilter(nodeID, since, limit))
if err != nil {
return nil, err
}
return fromAnalyticsNodeObsOpenresty(rows), nil
}
func (clickhouseObservabilityStore) DeleteAllNodeObservationOpenresty(ctx context.Context) (int64, error) {
return analyticsrepo.DeleteAllNodeObsOpenresty(ctx)
}
func (clickhouseObservabilityStore) DeleteNodeObservationOpenrestyBefore(ctx context.Context, cutoff time.Time) (int64, error) {
return analyticsrepo.DeleteNodeObsOpenrestyBefore(ctx, cutoff)
} }
func (clickhouseObservabilityStore) InsertNodeObservationFrps(_ context.Context, record *OpenFlareNodeObservationFrps) error { func (clickhouseObservabilityStore) InsertNodeObservationFrps(_ context.Context, record *OpenFlareNodeObservationFrps) error {
@@ -280,65 +259,27 @@ func fromAnalyticsNodeMetricSnapshots(rows []analyticsmodel.NodeMetricSnapshot)
return result return result
} }
func toAnalyticsNodeRequestReport(record *OpenFlareRequestReport) analyticsmodel.NodeRequestReport { func toAnalyticsNodeEdgeHealth(record *OpenFlareEdgeHealth) analyticsmodel.NodeEdgeHealth {
return analyticsmodel.NodeRequestReport{ return analyticsmodel.NodeEdgeHealth{
ID: uint64(record.ID), ID: uint64(record.ID),
NodeID: record.NodeID, NodeID: record.NodeID,
WindowStartedAt: record.WindowStartedAt, CapturedAt: record.CapturedAt,
WindowEndedAt: record.WindowEndedAt, Status: normalizeEdgeHealthStatus(record.Status),
RequestCount: record.RequestCount, Connections: record.Connections,
ErrorCount: record.ErrorCount, CreatedAt: record.CreatedAt,
UniqueVisitorCount: record.UniqueVisitorCount,
StatusCodesJSON: record.StatusCodesJSON,
TopDomainsJSON: record.TopDomainsJSON,
SourceCountriesJSON: record.SourceCountriesJSON,
CreatedAt: record.CreatedAt,
} }
} }
func fromAnalyticsNodeRequestReports(rows []analyticsmodel.NodeRequestReport) []*OpenFlareRequestReport { func fromAnalyticsNodeEdgeHealth(rows []analyticsmodel.NodeEdgeHealth) []*OpenFlareEdgeHealth {
result := make([]*OpenFlareRequestReport, len(rows)) result := make([]*OpenFlareEdgeHealth, len(rows))
for index, row := range rows { for index, row := range rows {
result[index] = &OpenFlareRequestReport{ result[index] = &OpenFlareEdgeHealth{
ID: uint(row.ID), ID: uint(row.ID),
NodeID: row.NodeID, NodeID: row.NodeID,
WindowStartedAt: row.WindowStartedAt, CapturedAt: row.CapturedAt,
WindowEndedAt: row.WindowEndedAt, Status: normalizeEdgeHealthStatus(row.Status),
RequestCount: row.RequestCount, Connections: row.Connections,
ErrorCount: row.ErrorCount, CreatedAt: row.CreatedAt,
UniqueVisitorCount: row.UniqueVisitorCount,
StatusCodesJSON: row.StatusCodesJSON,
TopDomainsJSON: row.TopDomainsJSON,
SourceCountriesJSON: row.SourceCountriesJSON,
CreatedAt: row.CreatedAt,
}
}
return result
}
func toAnalyticsNodeObsOpenresty(record *OpenFlareNodeObservationOpenresty) analyticsmodel.NodeObsOpenresty {
return analyticsmodel.NodeObsOpenresty{
ID: uint64(record.ID),
NodeID: record.NodeID,
CapturedAt: record.CapturedAt,
OpenrestyRxBytes: record.OpenrestyRxBytes,
OpenrestyTxBytes: record.OpenrestyTxBytes,
OpenrestyConnections: record.OpenrestyConnections,
CreatedAt: record.CreatedAt,
}
}
func fromAnalyticsNodeObsOpenresty(rows []analyticsmodel.NodeObsOpenresty) []*OpenFlareNodeObservationOpenresty {
result := make([]*OpenFlareNodeObservationOpenresty, len(rows))
for index, row := range rows {
result[index] = &OpenFlareNodeObservationOpenresty{
ID: uint(row.ID),
NodeID: row.NodeID,
CapturedAt: row.CapturedAt,
OpenrestyRxBytes: row.OpenrestyRxBytes,
OpenrestyTxBytes: row.OpenrestyTxBytes,
OpenrestyConnections: row.OpenrestyConnections,
CreatedAt: row.CreatedAt,
} }
} }
return result return result
@@ -16,8 +16,7 @@ import (
type memoryObservabilityStore struct { type memoryObservabilityStore struct {
mu sync.RWMutex mu sync.RWMutex
metricSnapshots []*OpenFlareMetricSnapshot metricSnapshots []*OpenFlareMetricSnapshot
requestReports []*OpenFlareRequestReport edgeHealth []*OpenFlareEdgeHealth
openrestyObs []*OpenFlareNodeObservationOpenresty
frpsObs []*OpenFlareNodeObservationFrps frpsObs []*OpenFlareNodeObservationFrps
frpcObs []*OpenFlareNodeObservationFrpc frpcObs []*OpenFlareNodeObservationFrpc
} }
@@ -69,93 +68,46 @@ func (s *memoryObservabilityStore) DeleteMetricSnapshotsBefore(_ context.Context
return deleted, nil return deleted, nil
} }
func (s *memoryObservabilityStore) InsertRequestReport(_ context.Context, record *OpenFlareRequestReport) error { func (s *memoryObservabilityStore) InsertEdgeHealth(_ context.Context, record *OpenFlareEdgeHealth) error {
if record == nil { if record == nil {
return nil return nil
} }
s.mu.Lock() s.mu.Lock()
defer s.mu.Unlock() defer s.mu.Unlock()
copyRecord := cloneOpenFlareRequestReport(record) s.edgeHealth = append(s.edgeHealth, cloneOpenFlareEdgeHealth(record))
if memoryRequestReportExists(s.requestReports, copyRecord.NodeID, copyRecord.WindowStartedAt, copyRecord.WindowEndedAt) {
return nil
}
s.requestReports = append(s.requestReports, copyRecord)
return nil return nil
} }
func (s *memoryObservabilityStore) ListRequestReports(_ context.Context, nodeID string, since time.Time, limit int) ([]*OpenFlareRequestReport, error) { func (s *memoryObservabilityStore) ListEdgeHealth(_ context.Context, nodeID string, since time.Time, limit int) ([]*OpenFlareEdgeHealth, error) {
s.mu.RLock() s.mu.RLock()
defer s.mu.RUnlock() defer s.mu.RUnlock()
rows := memoryFilterRequestReports(s.requestReports, nodeID, since) rows := memoryFilterEdgeHealth(s.edgeHealth, nodeID, since)
sortOpenFlareRequestReports(rows) sortOpenFlareEdgeHealth(rows)
return memoryLimitObservabilityRows(rows, limit), nil return memoryLimitObservabilityRows(rows, limit), nil
} }
func (s *memoryObservabilityStore) DeleteAllRequestReports(_ context.Context) (int64, error) { func (s *memoryObservabilityStore) DeleteAllEdgeHealth(_ context.Context) (int64, error) {
s.mu.Lock() s.mu.Lock()
defer s.mu.Unlock() defer s.mu.Unlock()
count := int64(len(s.requestReports)) count := int64(len(s.edgeHealth))
s.requestReports = nil s.edgeHealth = nil
return count, nil return count, nil
} }
func (s *memoryObservabilityStore) DeleteRequestReportsBefore(_ context.Context, cutoff time.Time) (int64, error) { func (s *memoryObservabilityStore) DeleteEdgeHealthBefore(_ context.Context, cutoff time.Time) (int64, error) {
s.mu.Lock() s.mu.Lock()
defer s.mu.Unlock() defer s.mu.Unlock()
cutoff = cutoff.UTC() cutoff = cutoff.UTC()
remaining := make([]*OpenFlareRequestReport, 0, len(s.requestReports)) remaining := make([]*OpenFlareEdgeHealth, 0, len(s.edgeHealth))
var deleted int64 var deleted int64
for _, row := range s.requestReports { for _, row := range s.edgeHealth {
if row.WindowEndedAt.Before(cutoff) {
deleted++
continue
}
remaining = append(remaining, row)
}
s.requestReports = remaining
return deleted, nil
}
func (s *memoryObservabilityStore) InsertNodeObservationOpenresty(_ context.Context, record *OpenFlareNodeObservationOpenresty) error {
if record == nil {
return nil
}
s.mu.Lock()
defer s.mu.Unlock()
s.openrestyObs = append(s.openrestyObs, cloneOpenFlareNodeObservationOpenresty(record))
return nil
}
func (s *memoryObservabilityStore) ListNodeObservationOpenresty(_ context.Context, nodeID string, since time.Time, limit int) ([]*OpenFlareNodeObservationOpenresty, error) {
s.mu.RLock()
defer s.mu.RUnlock()
rows := memoryFilterOpenrestyObservations(s.openrestyObs, nodeID, since)
sortOpenFlareNodeObservationOpenresty(rows)
return memoryLimitObservabilityRows(rows, limit), nil
}
func (s *memoryObservabilityStore) DeleteAllNodeObservationOpenresty(_ context.Context) (int64, error) {
s.mu.Lock()
defer s.mu.Unlock()
count := int64(len(s.openrestyObs))
s.openrestyObs = nil
return count, nil
}
func (s *memoryObservabilityStore) DeleteNodeObservationOpenrestyBefore(_ context.Context, cutoff time.Time) (int64, error) {
s.mu.Lock()
defer s.mu.Unlock()
cutoff = cutoff.UTC()
remaining := make([]*OpenFlareNodeObservationOpenresty, 0, len(s.openrestyObs))
var deleted int64
for _, row := range s.openrestyObs {
if row.CapturedAt.Before(cutoff) { if row.CapturedAt.Before(cutoff) {
deleted++ deleted++
continue continue
} }
remaining = append(remaining, row) remaining = append(remaining, row)
} }
s.openrestyObs = remaining s.edgeHealth = remaining
return deleted, nil return deleted, nil
} }
@@ -259,22 +211,8 @@ func memoryFilterMetricSnapshots(rows []*OpenFlareMetricSnapshot, nodeID string,
return result return result
} }
func memoryFilterRequestReports(rows []*OpenFlareRequestReport, nodeID string, since time.Time) []*OpenFlareRequestReport { func memoryFilterEdgeHealth(rows []*OpenFlareEdgeHealth, nodeID string, since time.Time) []*OpenFlareEdgeHealth {
result := make([]*OpenFlareRequestReport, 0, len(rows)) result := make([]*OpenFlareEdgeHealth, 0, len(rows))
for _, row := range rows {
if !memoryObservabilityMatchesNodeID(row.NodeID, nodeID) {
continue
}
if !since.IsZero() && row.WindowEndedAt.Before(since) {
continue
}
result = append(result, row)
}
return result
}
func memoryFilterOpenrestyObservations(rows []*OpenFlareNodeObservationOpenresty, nodeID string, since time.Time) []*OpenFlareNodeObservationOpenresty {
result := make([]*OpenFlareNodeObservationOpenresty, 0, len(rows))
for _, row := range rows { for _, row := range rows {
if !memoryObservabilityMatchesNodeID(row.NodeID, nodeID) { if !memoryObservabilityMatchesNodeID(row.NodeID, nodeID) {
continue continue
@@ -333,19 +271,6 @@ func memoryMetricSnapshotExists(rows []*OpenFlareMetricSnapshot, nodeID string,
return false return false
} }
func memoryRequestReportExists(rows []*OpenFlareRequestReport, nodeID string, windowStartedAt, windowEndedAt time.Time) bool {
windowStartedAt = windowStartedAt.UTC()
windowEndedAt = windowEndedAt.UTC()
for _, row := range rows {
if row.NodeID == nodeID &&
row.WindowStartedAt.UTC().Equal(windowStartedAt) &&
row.WindowEndedAt.UTC().Equal(windowEndedAt) {
return true
}
}
return false
}
func sortOpenFlareMetricSnapshots(items []*OpenFlareMetricSnapshot) { func sortOpenFlareMetricSnapshots(items []*OpenFlareMetricSnapshot) {
sort.Slice(items, func(i, j int) bool { sort.Slice(items, func(i, j int) bool {
left := items[i] left := items[i]
@@ -360,21 +285,7 @@ func sortOpenFlareMetricSnapshots(items []*OpenFlareMetricSnapshot) {
}) })
} }
func sortOpenFlareRequestReports(items []*OpenFlareRequestReport) { func sortOpenFlareEdgeHealth(items []*OpenFlareEdgeHealth) {
sort.Slice(items, func(i, j int) bool {
left := items[i]
right := items[j]
if left == nil || right == nil {
return left != nil
}
if compare := openFlareAccessLogCompareInt64(left.WindowEndedAt.Unix(), right.WindowEndedAt.Unix()); compare != 0 {
return compare > 0
}
return openFlareAccessLogCompareInt64(openFlareAccessLogUintToInt64(uint64(left.ID)), openFlareAccessLogUintToInt64(uint64(right.ID))) > 0
})
}
func sortOpenFlareNodeObservationOpenresty(items []*OpenFlareNodeObservationOpenresty) {
sort.Slice(items, func(i, j int) bool { sort.Slice(items, func(i, j int) bool {
left := items[i] left := items[i]
right := items[j] right := items[j]
@@ -441,22 +352,7 @@ func cloneOpenFlareMetricSnapshot(record *OpenFlareMetricSnapshot) *OpenFlareMet
return &copyRecord return &copyRecord
} }
func cloneOpenFlareRequestReport(record *OpenFlareRequestReport) *OpenFlareRequestReport { func cloneOpenFlareEdgeHealth(record *OpenFlareEdgeHealth) *OpenFlareEdgeHealth {
copyRecord := *record
if copyRecord.ID == 0 {
copyRecord.ID = uint(idgen.NextUint64ID())
}
now := time.Now().UTC()
if copyRecord.CreatedAt.IsZero() {
copyRecord.CreatedAt = now
}
copyRecord.WindowStartedAt = copyRecord.WindowStartedAt.UTC()
copyRecord.WindowEndedAt = copyRecord.WindowEndedAt.UTC()
copyRecord.CreatedAt = copyRecord.CreatedAt.UTC()
return &copyRecord
}
func cloneOpenFlareNodeObservationOpenresty(record *OpenFlareNodeObservationOpenresty) *OpenFlareNodeObservationOpenresty {
copyRecord := *record copyRecord := *record
if copyRecord.ID == 0 { if copyRecord.ID == 0 {
copyRecord.ID = uint(idgen.NextUint64ID()) copyRecord.ID = uint(idgen.NextUint64ID())
@@ -468,6 +364,9 @@ func cloneOpenFlareNodeObservationOpenresty(record *OpenFlareNodeObservationOpen
if copyRecord.CapturedAt.IsZero() { if copyRecord.CapturedAt.IsZero() {
copyRecord.CapturedAt = now copyRecord.CapturedAt = now
} }
if strings.TrimSpace(copyRecord.Status) == "" {
copyRecord.Status = edgeHealthStatusUnknown
}
copyRecord.CapturedAt = copyRecord.CapturedAt.UTC() copyRecord.CapturedAt = copyRecord.CapturedAt.UTC()
copyRecord.CreatedAt = copyRecord.CreatedAt.UTC() copyRecord.CreatedAt = copyRecord.CreatedAt.UTC()
return &copyRecord return &copyRecord
@@ -17,9 +17,7 @@ const (
TableTTLDaysNodeAccessLogs = 90 TableTTLDaysNodeAccessLogs = 90
// TableTTLDaysNodeMetricSnapshots is the of_node_metric_snapshots TTL (30 days). // TableTTLDaysNodeMetricSnapshots is the of_node_metric_snapshots TTL (30 days).
TableTTLDaysNodeMetricSnapshots = 30 TableTTLDaysNodeMetricSnapshots = 30
// TableTTLDaysNodeRequestReports is the of_node_request_reports TTL (30 days). // TableTTLDaysNodeObs is the of_node_edge_health / of_node_obs_frps / of_node_obs_frpc TTL (30 days).
TableTTLDaysNodeRequestReports = 30
// TableTTLDaysNodeObs is the of_node_obs_* TTL (30 days).
TableTTLDaysNodeObs = 30 TableTTLDaysNodeObs = 30
// TableTTLDaysUserAccessLogs is the w_user_access_logs TTL (180 days). // TableTTLDaysUserAccessLogs is the w_user_access_logs TTL (180 days).
TableTTLDaysUserAccessLogs = 180 TableTTLDaysUserAccessLogs = 180
@@ -31,7 +31,6 @@ func TestCleanupModeConstants(t *testing.T) {
func TestTableTTLDaysMatchDDL(t *testing.T) { func TestTableTTLDaysMatchDDL(t *testing.T) {
assert.Equal(t, 90, TableTTLDaysNodeAccessLogs) assert.Equal(t, 90, TableTTLDaysNodeAccessLogs)
assert.Equal(t, 30, TableTTLDaysNodeMetricSnapshots) assert.Equal(t, 30, TableTTLDaysNodeMetricSnapshots)
assert.Equal(t, 30, TableTTLDaysNodeRequestReports)
assert.Equal(t, 30, TableTTLDaysNodeObs) assert.Equal(t, 30, TableTTLDaysNodeObs)
assert.Equal(t, 180, TableTTLDaysUserAccessLogs) assert.Equal(t, 180, TableTTLDaysUserAccessLogs)
} }
@@ -6,6 +6,7 @@ package analytics
import ( import (
"context" "context"
"fmt" "fmt"
"strings"
"time" "time"
"github.com/ClickHouse/clickhouse-go/v2/lib/driver" "github.com/ClickHouse/clickhouse-go/v2/lib/driver"
@@ -35,7 +36,7 @@ func ListNodeAccessLogs(ctx context.Context, filter NodeAccessLogFilter) ([]anal
clause, args := buildNodeAccessLogFilterClause(filter) clause, args := buildNodeAccessLogFilterClause(filter)
tableName := nodeAccessLogTableName() tableName := nodeAccessLogTableName()
sql := fmt.Sprintf(` sql := fmt.Sprintf(`
SELECT id, node_id, logged_at, remote_addr, region, host, path, status_code, bytes_sent, created_at SELECT id, node_id, logged_at, remote_addr, region, host, path, status_code, bytes_sent, request_length, request_time_ms, created_at
FROM %s FROM %s
WHERE %s WHERE %s
ORDER BY %s`, tableName, clause, nodeAccessLogOrderClause(filter.SortBy, filter.SortOrder)) ORDER BY %s`, tableName, clause, nodeAccessLogOrderClause(filter.SortBy, filter.SortOrder))
@@ -68,6 +69,8 @@ func scanNodeAccessLogRows(rows driver.Rows) ([]analyticsmodel.NodeAccessLog, er
&item.Path, &item.Path,
&item.StatusCode, &item.StatusCode,
&item.BytesSent, &item.BytesSent,
&item.RequestLength,
&item.RequestTimeMs,
&item.CreatedAt, &item.CreatedAt,
); err != nil { ); err != nil {
return nil, fmt.Errorf("scan node access log row: %w", err) return nil, fmt.Errorf("scan node access log row: %w", err)
@@ -143,3 +146,157 @@ ORDER BY count DESC, trimmed_region ASC`, tableName, clause)
} }
return result, nil return result, nil
} }
// NodeAccessLogTrafficSummary is a window-level access log traffic summary.
type NodeAccessLogTrafficSummary struct {
RequestCount int64
ErrorCount int64
UniqueIPCount int64
BytesSent int64
RequestLength int64
NodeCount int64
}
// NodeAccessLogValueCount is a grouped value count (status_code, host, ...).
type NodeAccessLogValueCount struct {
Value string
Count int64
}
// NodeAccessLogNodeAggregate is per-node traffic over a window.
type NodeAccessLogNodeAggregate struct {
NodeID string
RequestCount int64
ErrorCount int64
UniqueIPCount int64
}
// TrafficSummaryNodeAccessLogs returns request/error/UV/bytes/node counts for the filter.
func TrafficSummaryNodeAccessLogs(ctx context.Context, filter NodeAccessLogFilter) (NodeAccessLogTrafficSummary, error) {
conn, err := nodeAccessLogConn()
if err != nil {
return NodeAccessLogTrafficSummary{}, err
}
clause, args := buildNodeAccessLogFilterClause(filter)
tableName := nodeAccessLogTableName()
sql := fmt.Sprintf(`
SELECT
count() AS request_count,
countIf(status_code >= 500) AS error_count,
uniqExactIf(remote_addr, remote_addr != '') AS unique_ips,
sum(bytes_sent) AS bytes_sent,
sum(request_length) AS request_length,
uniqExactIf(node_id, node_id != '') AS node_count
FROM %s
WHERE %s`, tableName, clause)
var requestCount, errorCount, uniqueIPs, bytesSent, requestLength, nodeCount uint64
if err := conn.QueryRow(ctx, sql, args...).Scan(
&requestCount, &errorCount, &uniqueIPs, &bytesSent, &requestLength, &nodeCount,
); err != nil {
return NodeAccessLogTrafficSummary{}, fmt.Errorf("traffic summary node access logs: %w", err)
}
return NodeAccessLogTrafficSummary{
RequestCount: safeInt64Count(requestCount),
ErrorCount: safeInt64Count(errorCount),
UniqueIPCount: safeInt64Count(uniqueIPs),
BytesSent: safeInt64Count(bytesSent),
RequestLength: safeInt64Count(requestLength),
NodeCount: safeInt64Count(nodeCount),
}, nil
}
// ValueCountsNodeAccessLogs groups logs by a single dimension column.
// Allowed columns: status_code, host.
func ValueCountsNodeAccessLogs(ctx context.Context, filter NodeAccessLogFilter, column string, limit int) ([]NodeAccessLogValueCount, error) {
conn, err := nodeAccessLogConn()
if err != nil {
return nil, err
}
col := strings.TrimSpace(strings.ToLower(column))
switch col {
case nodeAccessLogColumnStatusCode, nodeAccessLogColumnHost:
default:
return nil, fmt.Errorf("unsupported value count column: %s", column)
}
clause, args := buildNodeAccessLogFilterClause(filter)
tableName := nodeAccessLogTableName()
// status_code is numeric; cast to string for a uniform Value field.
var valueExpr string
if col == nodeAccessLogColumnStatusCode {
valueExpr = "toString(" + nodeAccessLogColumnStatusCode + ")"
} else {
valueExpr = "trim(" + nodeAccessLogColumnHost + ")"
}
sql := fmt.Sprintf(`
SELECT %s AS value, count() AS count
FROM %s
WHERE %s AND %s != ''
GROUP BY value
ORDER BY count DESC, value ASC`, valueExpr, tableName, clause, valueExpr)
if limit > 0 {
sql += clickHouseLimitClause
args = append(args, limit)
}
rows, err := conn.Query(ctx, sql, args...)
if err != nil {
return nil, fmt.Errorf("value counts node access logs: %w", err)
}
defer func() { _ = rows.Close() }()
var result []NodeAccessLogValueCount
for rows.Next() {
var (
value string
count uint64
)
if err := rows.Scan(&value, &count); err != nil {
return nil, fmt.Errorf("scan value count row: %w", err)
}
result = append(result, NodeAccessLogValueCount{
Value: value,
Count: safeInt64Count(count),
})
}
return result, nil
}
// NodeAggregatesNodeAccessLogs returns per-node request/error/UV aggregates.
func NodeAggregatesNodeAccessLogs(ctx context.Context, filter NodeAccessLogFilter) ([]NodeAccessLogNodeAggregate, error) {
conn, err := nodeAccessLogConn()
if err != nil {
return nil, err
}
clause, args := buildNodeAccessLogFilterClause(filter)
tableName := nodeAccessLogTableName()
sql := fmt.Sprintf(`
SELECT
node_id,
count() AS request_count,
countIf(status_code >= 500) AS error_count,
uniqExactIf(remote_addr, remote_addr != '') AS unique_ips
FROM %s
WHERE %s AND node_id != ''
GROUP BY node_id
ORDER BY request_count DESC, node_id ASC`, tableName, clause)
rows, err := conn.Query(ctx, sql, args...)
if err != nil {
return nil, fmt.Errorf("node aggregates node access logs: %w", err)
}
defer func() { _ = rows.Close() }()
var result []NodeAccessLogNodeAggregate
for rows.Next() {
var (
nodeID string
requestCount, errorCount, uniqueIPs uint64
)
if err := rows.Scan(&nodeID, &requestCount, &errorCount, &uniqueIPs); err != nil {
return nil, fmt.Errorf("scan node aggregate row: %w", err)
}
result = append(result, NodeAccessLogNodeAggregate{
NodeID: nodeID,
RequestCount: safeInt64Count(requestCount),
ErrorCount: safeInt64Count(errorCount),
UniqueIPCount: safeInt64Count(uniqueIPs),
})
}
return result, nil
}
@@ -17,6 +17,10 @@ const (
nodeAccessLogSortAscInput = "asc" nodeAccessLogSortAscInput = "asc"
nodeAccessLogColumnRemoteAddr = "remote_addr" nodeAccessLogColumnRemoteAddr = "remote_addr"
nodeAccessLogColumnStatusCode = "status_code"
nodeAccessLogColumnHost = "host"
nodeAccessLogColumnPath = "path"
nodeAccessLogColumnLoggedAt = "logged_at"
) )
// NodeAccessLogFilter scopes ClickHouse node access log queries. // NodeAccessLogFilter scopes ClickHouse node access log queries.
@@ -88,21 +92,21 @@ func nodeAccessLogOrderClause(sortBy string, sortOrder string) string {
if normalizeNodeAccessLogSortOrder(sortOrder) == nodeAccessLogSortAscInput { if normalizeNodeAccessLogSortOrder(sortOrder) == nodeAccessLogSortAscInput {
direction = nodeAccessLogSortAsc direction = nodeAccessLogSortAsc
} }
column := "logged_at" column := nodeAccessLogColumnLoggedAt
switch strings.TrimSpace(sortBy) { switch strings.TrimSpace(sortBy) {
case "status_code": case nodeAccessLogColumnStatusCode:
column = "status_code" column = nodeAccessLogColumnStatusCode
case nodeAccessLogColumnRemoteAddr: case nodeAccessLogColumnRemoteAddr:
column = nodeAccessLogColumnRemoteAddr column = nodeAccessLogColumnRemoteAddr
case "host": case nodeAccessLogColumnHost:
column = "host" column = nodeAccessLogColumnHost
case "path": case nodeAccessLogColumnPath:
column = "path" column = nodeAccessLogColumnPath
} }
if column == "logged_at" { if column == nodeAccessLogColumnLoggedAt {
return column + " " + direction + ", id " + direction return column + " " + direction + ", id " + direction
} }
return column + " " + direction + ", logged_at " + direction + ", id " + direction return column + " " + direction + ", " + nodeAccessLogColumnLoggedAt + " " + direction + ", id " + direction
} }
func normalizeNodeAccessLogRemoteAddr(value string) string { func normalizeNodeAccessLogRemoteAddr(value string) string {
@@ -48,7 +48,8 @@ SELECT
countIf(status_code >= 500) AS server_error_count, countIf(status_code >= 500) AS server_error_count,
uniqExactIf(remote_addr, remote_addr != '') AS unique_ip_count, uniqExactIf(remote_addr, remote_addr != '') AS unique_ip_count,
uniqExactIf(host, host != '') AS unique_host_count, uniqExactIf(host, host != '') AS unique_host_count,
sum(bytes_sent) AS bytes_sent sum(bytes_sent) AS bytes_sent,
sum(request_length) AS request_length
FROM %s FROM %s
WHERE %s WHERE %s
GROUP BY bucket_epoch GROUP BY bucket_epoch
@@ -69,10 +70,10 @@ ORDER BY %s`, bucketExpr, tableName, clause, nodeAccessLogBucketOrderClause(filt
var result []NodeAccessLogBucketAggregate var result []NodeAccessLogBucketAggregate
for rows.Next() { for rows.Next() {
var ( var (
bucketEpoch int64 bucketEpoch int64
requestCount, successCount, clientErrorCount, serverErrorCount, uniqueIPCount, uniqueHostCount, bytesSent uint64 requestCount, successCount, clientErrorCount, serverErrorCount, uniqueIPCount, uniqueHostCount, bytesSent, requestLength uint64
) )
if err := rows.Scan(&bucketEpoch, &requestCount, &successCount, &clientErrorCount, &serverErrorCount, &uniqueIPCount, &uniqueHostCount, &bytesSent); err != nil { if err := rows.Scan(&bucketEpoch, &requestCount, &successCount, &clientErrorCount, &serverErrorCount, &uniqueIPCount, &uniqueHostCount, &bytesSent, &requestLength); err != nil {
return nil, fmt.Errorf("scan bucket aggregate row: %w", err) return nil, fmt.Errorf("scan bucket aggregate row: %w", err)
} }
result = append(result, NodeAccessLogBucketAggregate{ result = append(result, NodeAccessLogBucketAggregate{
@@ -84,6 +85,7 @@ ORDER BY %s`, bucketExpr, tableName, clause, nodeAccessLogBucketOrderClause(filt
UniqueIPCount: safeInt64Count(uniqueIPCount), UniqueIPCount: safeInt64Count(uniqueIPCount),
UniqueHostCount: safeInt64Count(uniqueHostCount), UniqueHostCount: safeInt64Count(uniqueHostCount),
BytesSent: safeInt64Count(bytesSent), BytesSent: safeInt64Count(bytesSent),
RequestLength: safeInt64Count(requestLength),
}) })
} }
return result, nil return result, nil
@@ -49,6 +49,8 @@ func TestBatchInsertNodeAccessLogs_UsesModelBatchSQL(t *testing.T) {
assert.True(t, mockBatch.sendCalled) assert.True(t, mockBatch.sendCalled)
require.Len(t, mockBatch.rows, 1) require.Len(t, mockBatch.rows, 1)
assert.Equal(t, "node-a", mockBatch.rows[0][1]) assert.Equal(t, "node-a", mockBatch.rows[0][1])
require.Len(t, mockBatch.rows[0], 10) require.Len(t, mockBatch.rows[0], 12)
assert.Equal(t, uint64(2048), mockBatch.rows[0][8]) assert.Equal(t, uint64(2048), mockBatch.rows[0][8]) // bytes_sent
assert.Equal(t, uint64(0), mockBatch.rows[0][9]) // request_length
assert.Equal(t, uint32(0), mockBatch.rows[0][10]) // request_time_ms
} }
@@ -48,6 +48,8 @@ func BatchInsertNodeAccessLogs(ctx context.Context, logs []analyticsmodel.NodeAc
logItem.Path, logItem.Path,
logItem.StatusCode, logItem.StatusCode,
logItem.BytesSent, logItem.BytesSent,
logItem.RequestLength,
logItem.RequestTimeMs,
createdAt.UTC(), createdAt.UTC(),
); err != nil { ); err != nil {
return fmt.Errorf("append node access log to batch: %w", err) return fmt.Errorf("append node access log to batch: %w", err)
@@ -67,62 +67,16 @@ ORDER BY %s%s`, nodeMetricSnapshotTableName(), clause, nodeObservabilityCaptured
return scanNodeMetricSnapshotRows(rows) return scanNodeMetricSnapshotRows(rows)
} }
// ListNodeRequestReports returns request reports matching filter. // ListNodeEdgeHealth returns L2 OpenResty health snapshots.
func ListNodeRequestReports(ctx context.Context, filter NodeObservabilityFilter) ([]analyticsmodel.NodeRequestReport, error) { func ListNodeEdgeHealth(ctx context.Context, filter NodeObservabilityFilter) ([]analyticsmodel.NodeEdgeHealth, error) {
conn, err := observabilityConn()
if err != nil {
return nil, err
}
clause, args := buildNodeObservabilityFilterClause(filter, "window_ended_at")
tableName := nodeRequestReportTableName()
sql := fmt.Sprintf(`
SELECT id, node_id, window_started_at, window_ended_at, request_count, error_count, unique_visitor_count, status_codes_json, top_domains_json, source_countries_json, created_at
FROM %s
WHERE %s
ORDER BY %s`, tableName, clause, nodeObservabilityWindowEndedAtOrderClause())
if filter.Limit > 0 {
sql += clickHouseLimitClause
args = append(args, filter.Limit)
}
rows, err := conn.Query(ctx, sql, args...)
if err != nil {
return nil, fmt.Errorf("list node request reports: %w", err)
}
defer func() { _ = rows.Close() }()
return scanNodeRequestReportRows(rows)
}
// ListLatestNodeRequestReports returns the latest request report per node_id.
// Uses ClickHouse LIMIT 1 BY so dashboard traffic health is not skewed by a global raw LIMIT.
func ListLatestNodeRequestReports(ctx context.Context, filter NodeObservabilityFilter) ([]analyticsmodel.NodeRequestReport, error) {
conn, err := observabilityConn()
if err != nil {
return nil, err
}
clause, args := buildNodeObservabilityFilterClause(filter, "window_ended_at")
sql := fmt.Sprintf(`
SELECT id, node_id, window_started_at, window_ended_at, request_count, error_count, unique_visitor_count, status_codes_json, top_domains_json, source_countries_json, created_at
FROM %s
WHERE %s
ORDER BY %s%s`, nodeRequestReportTableName(), clause, nodeObservabilityWindowEndedAtOrderClause(), clickHouseLimit1ByNodeIDClause)
rows, err := conn.Query(ctx, sql, args...)
if err != nil {
return nil, fmt.Errorf("list latest node request reports: %w", err)
}
defer func() { _ = rows.Close() }()
return scanNodeRequestReportRows(rows)
}
// ListNodeObsOpenresty returns OpenResty observations matching filter.
func ListNodeObsOpenresty(ctx context.Context, filter NodeObservabilityFilter) ([]analyticsmodel.NodeObsOpenresty, error) {
conn, err := observabilityConn() conn, err := observabilityConn()
if err != nil { if err != nil {
return nil, err return nil, err
} }
clause, args := buildNodeObservabilityFilterClause(filter, "captured_at") clause, args := buildNodeObservabilityFilterClause(filter, "captured_at")
tableName := nodeObsOpenrestyTableName() tableName := nodeEdgeHealthTableName()
sql := fmt.Sprintf(` sql := fmt.Sprintf(`
SELECT id, node_id, captured_at, openresty_rx_bytes, openresty_tx_bytes, openresty_connections, created_at SELECT id, node_id, captured_at, status, connections, created_at
FROM %s FROM %s
WHERE %s WHERE %s
ORDER BY %s`, tableName, clause, nodeObservabilityCapturedAtOrderClause()) ORDER BY %s`, tableName, clause, nodeObservabilityCapturedAtOrderClause())
@@ -132,10 +86,27 @@ ORDER BY %s`, tableName, clause, nodeObservabilityCapturedAtOrderClause())
} }
rows, err := conn.Query(ctx, sql, args...) rows, err := conn.Query(ctx, sql, args...)
if err != nil { if err != nil {
return nil, fmt.Errorf("list node openresty observations: %w", err) return nil, fmt.Errorf("list node edge health: %w", err)
} }
defer func() { _ = rows.Close() }() defer func() { _ = rows.Close() }()
return scanNodeObsOpenrestyRows(rows) var result []analyticsmodel.NodeEdgeHealth
for rows.Next() {
var item analyticsmodel.NodeEdgeHealth
if err := rows.Scan(
&item.ID,
&item.NodeID,
&item.CapturedAt,
&item.Status,
&item.Connections,
&item.CreatedAt,
); err != nil {
return nil, fmt.Errorf("scan node edge health row: %w", err)
}
item.CapturedAt = item.CapturedAt.UTC()
item.CreatedAt = item.CreatedAt.UTC()
result = append(result, item)
}
return result, nil
} }
// ListNodeObsFrps returns FRPS observations matching filter. // ListNodeObsFrps returns FRPS observations matching filter.
@@ -216,55 +187,6 @@ func scanNodeMetricSnapshotRows(rows driver.Rows) ([]analyticsmodel.NodeMetricSn
return result, nil return result, nil
} }
func scanNodeRequestReportRows(rows driver.Rows) ([]analyticsmodel.NodeRequestReport, error) {
var result []analyticsmodel.NodeRequestReport
for rows.Next() {
var item analyticsmodel.NodeRequestReport
if err := rows.Scan(
&item.ID,
&item.NodeID,
&item.WindowStartedAt,
&item.WindowEndedAt,
&item.RequestCount,
&item.ErrorCount,
&item.UniqueVisitorCount,
&item.StatusCodesJSON,
&item.TopDomainsJSON,
&item.SourceCountriesJSON,
&item.CreatedAt,
); err != nil {
return nil, fmt.Errorf("scan node request report row: %w", err)
}
item.WindowStartedAt = item.WindowStartedAt.UTC()
item.WindowEndedAt = item.WindowEndedAt.UTC()
item.CreatedAt = item.CreatedAt.UTC()
result = append(result, item)
}
return result, nil
}
func scanNodeObsOpenrestyRows(rows driver.Rows) ([]analyticsmodel.NodeObsOpenresty, error) {
var result []analyticsmodel.NodeObsOpenresty
for rows.Next() {
var item analyticsmodel.NodeObsOpenresty
if err := rows.Scan(
&item.ID,
&item.NodeID,
&item.CapturedAt,
&item.OpenrestyRxBytes,
&item.OpenrestyTxBytes,
&item.OpenrestyConnections,
&item.CreatedAt,
); err != nil {
return nil, fmt.Errorf("scan node openresty observation row: %w", err)
}
item.CapturedAt = item.CapturedAt.UTC()
item.CreatedAt = item.CreatedAt.UTC()
result = append(result, item)
}
return result, nil
}
func scanNodeObsFrpsRows(rows driver.Rows) ([]analyticsmodel.NodeObsFrps, error) { func scanNodeObsFrpsRows(rows driver.Rows) ([]analyticsmodel.NodeObsFrps, error) {
var result []analyticsmodel.NodeObsFrps var result []analyticsmodel.NodeObsFrps
for rows.Next() { for rows.Next() {
@@ -288,13 +210,10 @@ func scanNodeObsFrpsRows(rows driver.Rows) ([]analyticsmodel.NodeObsFrps, error)
return result, nil return result, nil
} }
const nodeTrafficHourlyTableName = "of_node_traffic_hourly"
// NodeTrafficHourly is an hourly traffic rollup row. // NodeTrafficHourly is an hourly traffic rollup row.
// //
// UniqueVisitorCount is a peak per-window estimate from short request reports // UniqueVisitorCount is always 0 when sourced from of_access_log_hourly
// (MV uses max()), not true distinct visitors across the hour. SummingMergeTree // (true UV requires raw uniqExact on access logs).
// may still inflate residual unmerged parts; do not present as exact UV.
type NodeTrafficHourly struct { type NodeTrafficHourly struct {
NodeID string NodeID string
Hour time.Time Hour time.Time
@@ -319,16 +238,40 @@ type NodeMetricHourly struct {
ReportedNodes int ReportedNodes int
} }
// NodeOpenrestyHourly is an hourly OpenResty observation aggregation row. // ListNodeTrafficHourly returns hourly traffic from of_access_log_hourly (M5).
type NodeOpenrestyHourly struct { // UniqueVisitorCount is always 0 here (UV requires raw uniqExact on access logs).
Hour time.Time func ListNodeTrafficHourly(ctx context.Context, filter NodeObservabilityFilter) ([]NodeTrafficHourly, error) {
OpenrestyRxBytes int64 rows, err := ListAccessLogHourly(ctx, filter)
OpenrestyTxBytes int64 if err != nil {
ReportedNodes int return nil, err
}
// Aggregate across hosts per node/hour.
type key struct {
node string
hour int64
}
merged := make(map[key]*NodeTrafficHourly)
order := make([]key, 0)
for _, row := range rows {
k := key{node: row.NodeID, hour: row.Hour.UTC().Unix()}
item := merged[k]
if item == nil {
item = &NodeTrafficHourly{NodeID: row.NodeID, Hour: row.Hour.UTC()}
merged[k] = item
order = append(order, k)
}
item.RequestCount += row.RequestCount
item.ErrorCount += row.ErrorCount
}
result := make([]NodeTrafficHourly, 0, len(order))
for _, k := range order {
result = append(result, *merged[k])
}
return result, nil
} }
// ListNodeTrafficHourly returns hourly traffic rollup rows matching filter. // ListAccessLogHourly returns Server-side access log hourly rollups.
func ListNodeTrafficHourly(ctx context.Context, filter NodeObservabilityFilter) ([]NodeTrafficHourly, error) { func ListAccessLogHourly(ctx context.Context, filter NodeObservabilityFilter) ([]analyticsmodel.AccessLogHourly, error) {
conn, err := observabilityConn() conn, err := observabilityConn()
if err != nil { if err != nil {
return nil, err return nil, err
@@ -338,32 +281,42 @@ func ListNodeTrafficHourly(ctx context.Context, filter NodeObservabilityFilter)
SELECT SELECT
node_id, node_id,
hour, hour,
host,
sum(request_count) AS request_count, sum(request_count) AS request_count,
sum(error_count) AS error_count, sum(error_count) AS error_count,
max(unique_visitor_count) AS unique_visitor_count sum(bytes_sent) AS bytes_sent,
sum(request_length) AS request_length
FROM %s FROM %s
WHERE %s WHERE %s
GROUP BY node_id, hour GROUP BY node_id, hour, host
ORDER BY hour ASC`, nodeTrafficHourlyTableName, clause) ORDER BY hour ASC, node_id ASC, host ASC`, accessLogHourlyTableName(), clause)
rows, err := conn.Query(ctx, sql, args...) rows, err := conn.Query(ctx, sql, args...)
if err != nil { if err != nil {
return nil, fmt.Errorf("list node traffic hourly: %w", err) return nil, fmt.Errorf("list access log hourly: %w", err)
} }
defer func() { _ = rows.Close() }() defer func() { _ = rows.Close() }()
var result []analyticsmodel.AccessLogHourly
result := make([]NodeTrafficHourly, 0)
for rows.Next() { for rows.Next() {
var ( var (
item NodeTrafficHourly item analyticsmodel.AccessLogHourly
requestCount, errorCount, uniqueVisitorCount uint64 requestCount, errorCount, bytesSent, requestLength uint64
) )
if err := rows.Scan(&item.NodeID, &item.Hour, &requestCount, &errorCount, &uniqueVisitorCount); err != nil { if err := rows.Scan(
return nil, fmt.Errorf("scan node traffic hourly row: %w", err) &item.NodeID,
&item.Hour,
&item.Host,
&requestCount,
&errorCount,
&bytesSent,
&requestLength,
); err != nil {
return nil, fmt.Errorf("scan access log hourly row: %w", err)
} }
item.Hour = item.Hour.UTC() item.Hour = item.Hour.UTC()
item.RequestCount = safeInt64Count(requestCount) item.RequestCount = safeInt64Count(requestCount)
item.ErrorCount = safeInt64Count(errorCount) item.ErrorCount = safeInt64Count(errorCount)
item.UniqueVisitorCount = safeInt64Count(uniqueVisitorCount) item.BytesSent = safeInt64Count(bytesSent)
item.RequestLength = safeInt64Count(requestLength)
result = append(result, item) result = append(result, item)
} }
return result, nil return result, nil
@@ -556,138 +509,6 @@ func scanNodeMetricHourlyRows(rows driver.Rows) ([]NodeMetricHourly, error) {
return result, nil return result, nil
} }
// ListNodeOpenrestyHourly returns hourly OpenResty observation aggregates matching filter.
// Same rollup-first / per-hour merge strategy as ListNodeMetricHourly.
func ListNodeOpenrestyHourly(ctx context.Context, filter NodeObservabilityFilter) ([]NodeOpenrestyHourly, error) {
rollup, rollupErr := listNodeOpenrestyHourlyFromRollup(ctx, filter)
if rollupErr == nil && len(rollup) > 0 && hourlyRollupCoversWindow(rollup[0].Hour, filter.Since) {
return rollup, nil
}
raw, rawErr := listNodeOpenrestyHourlyFromRaw(ctx, filter)
if rawErr != nil {
if rollupErr == nil && len(rollup) > 0 {
return rollup, nil
}
return nil, rawErr
}
if len(rollup) == 0 {
return raw, nil
}
return mergeNodeOpenrestyHourlyPreferRollup(rollup, raw), nil
}
func mergeNodeOpenrestyHourlyPreferRollup(rollup, raw []NodeOpenrestyHourly) []NodeOpenrestyHourly {
byHour := make(map[int64]NodeOpenrestyHourly, len(raw)+len(rollup))
order := make([]int64, 0, len(raw)+len(rollup))
add := func(row NodeOpenrestyHourly, overwrite bool) {
key := row.Hour.UTC().Truncate(time.Hour).Unix()
if _, exists := byHour[key]; !exists {
order = append(order, key)
byHour[key] = row
return
}
if overwrite {
byHour[key] = row
}
}
for _, row := range raw {
add(row, false)
}
for _, row := range rollup {
add(row, true)
}
sort.Slice(order, func(i, j int) bool { return order[i] < order[j] })
result := make([]NodeOpenrestyHourly, 0, len(order))
for _, key := range order {
result = append(result, byHour[key])
}
return result
}
func listNodeOpenrestyHourlyFromRollup(ctx context.Context, filter NodeObservabilityFilter) ([]NodeOpenrestyHourly, error) {
conn, err := observabilityConn()
if err != nil {
return nil, err
}
clause, args := buildNodeObservabilityFilterClause(filter, "hour")
sql := fmt.Sprintf(`
SELECT
hour,
sum(greatest(openresty_rx_max - openresty_rx_min, 0)) AS openresty_rx_bytes,
sum(greatest(openresty_tx_max - openresty_tx_min, 0)) AS openresty_tx_bytes,
toUInt64(uniqExact(node_id)) AS reported_nodes
FROM %s
WHERE %s
GROUP BY hour
ORDER BY hour ASC`, nodeOpenrestyHourlyTableName(), clause)
rows, err := conn.Query(ctx, sql, args...)
if err != nil {
return nil, fmt.Errorf("list node openresty hourly from rollup: %w", err)
}
defer func() { _ = rows.Close() }()
return scanNodeOpenrestyHourlyRows(rows)
}
func listNodeOpenrestyHourlyFromRaw(ctx context.Context, filter NodeObservabilityFilter) ([]NodeOpenrestyHourly, error) {
conn, err := observabilityConn()
if err != nil {
return nil, err
}
clause, args := buildNodeObservabilityFilterClause(filter, "captured_at")
tableName := nodeObsOpenrestyTableName()
sql := fmt.Sprintf(`
SELECT
hour,
sum(if(openresty_rx_delta >= 0, openresty_rx_delta, 0)) AS openresty_rx_bytes,
sum(if(openresty_tx_delta >= 0, openresty_tx_delta, 0)) AS openresty_tx_bytes,
toUInt64(uniqExact(node_id)) AS reported_nodes
FROM (
SELECT
node_id,
toStartOfHour(captured_at) AS hour,
openresty_rx_bytes - lagInFrame(openresty_rx_bytes, 1, openresty_rx_bytes) OVER (
PARTITION BY node_id ORDER BY captured_at, id
ROWS BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW
) AS openresty_rx_delta,
openresty_tx_bytes - lagInFrame(openresty_tx_bytes, 1, openresty_tx_bytes) OVER (
PARTITION BY node_id ORDER BY captured_at, id
ROWS BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW
) AS openresty_tx_delta
FROM %s
WHERE %s
)
GROUP BY hour
ORDER BY hour ASC`, tableName, clause)
rows, err := conn.Query(ctx, sql, args...)
if err != nil {
return nil, fmt.Errorf("list node openresty hourly: %w", err)
}
defer func() { _ = rows.Close() }()
return scanNodeOpenrestyHourlyRows(rows)
}
func scanNodeOpenrestyHourlyRows(rows driver.Rows) ([]NodeOpenrestyHourly, error) {
result := make([]NodeOpenrestyHourly, 0)
for rows.Next() {
var (
item NodeOpenrestyHourly
reportedNodes uint64
rx int64
tx int64
)
if err := rows.Scan(&item.Hour, &rx, &tx, &reportedNodes); err != nil {
return nil, fmt.Errorf("scan node openresty hourly row: %w", err)
}
item.Hour = item.Hour.UTC()
item.OpenrestyRxBytes = rx
item.OpenrestyTxBytes = tx
item.ReportedNodes = int(safeInt64Count(reportedNodes))
result = append(result, item)
}
return result, nil
}
func scanNodeObsFrpcRows(rows driver.Rows) ([]analyticsmodel.NodeObsFrpc, error) { func scanNodeObsFrpcRows(rows driver.Rows) ([]analyticsmodel.NodeObsFrpc, error) {
var result []analyticsmodel.NodeObsFrpc var result []analyticsmodel.NodeObsFrpc
for rows.Next() { for rows.Next() {
@@ -51,78 +51,35 @@ func MaterializeNodeMetricSnapshotsTTL(ctx context.Context) (CleanupOutcome, err
) )
} }
// DeleteAllNodeRequestReports hard-deletes all node request reports via TRUNCATE. // DeleteAllNodeEdgeHealth truncates of_node_edge_health.
func DeleteAllNodeRequestReports(ctx context.Context) (int64, error) { func DeleteAllNodeEdgeHealth(ctx context.Context) (int64, error) {
conn, err := observabilityConn() conn, err := observabilityConn()
if err != nil { if err != nil {
return 0, err return 0, err
} }
outcome, err := truncateClickHouseTable(ctx, conn, nodeRequestReportTableName()) outcome, err := truncateClickHouseTable(ctx, conn, nodeEdgeHealthTableName())
if err != nil { if err != nil {
return 0, err return 0, err
} }
return outcome.DeletedCount, nil return outcome.DeletedCount, nil
} }
// DeleteNodeRequestReportsBefore force-materializes of_node_request_reports table TTL. // DeleteNodeEdgeHealthBefore force-materializes of_node_edge_health TTL.
// cutoff is ignored; see MaterializeNodeRequestReportsTTL. func DeleteNodeEdgeHealthBefore(ctx context.Context, _ time.Time) (int64, error) {
func DeleteNodeRequestReportsBefore(ctx context.Context, _ time.Time) (int64, error) { outcome, err := MaterializeNodeEdgeHealthTTL(ctx)
outcome, err := MaterializeNodeRequestReportsTTL(ctx)
if err != nil { if err != nil {
return 0, err return 0, err
} }
return outcome.EligibleCount, nil return outcome.EligibleCount, nil
} }
// MaterializeNodeRequestReportsTTL force-materializes table TTL and reports an honest outcome. // MaterializeNodeEdgeHealthTTL force-materializes of_node_edge_health table TTL.
func MaterializeNodeRequestReportsTTL(ctx context.Context) (CleanupOutcome, error) { func MaterializeNodeEdgeHealthTTL(ctx context.Context) (CleanupOutcome, error) {
conn, err := observabilityConn() conn, err := observabilityConn()
if err != nil { if err != nil {
return CleanupOutcome{}, err return CleanupOutcome{}, err
} }
tableName := nodeRequestReportTableName() tableName := nodeEdgeHealthTableName()
ttlDays := TableTTLDaysNodeRequestReports
cutoff := tableTTLCutoff(ttlDays, time.Now())
return materializeExpiredByTableTTL(
ctx,
conn,
tableName,
ttlDays,
fmt.Sprintf("SELECT count() FROM %s WHERE window_ended_at < ?", tableName),
[]any{cutoff},
)
}
// DeleteAllNodeObsOpenresty hard-deletes all OpenResty observations via TRUNCATE.
func DeleteAllNodeObsOpenresty(ctx context.Context) (int64, error) {
conn, err := observabilityConn()
if err != nil {
return 0, err
}
outcome, err := truncateClickHouseTable(ctx, conn, nodeObsOpenrestyTableName())
if err != nil {
return 0, err
}
return outcome.DeletedCount, nil
}
// DeleteNodeObsOpenrestyBefore force-materializes of_node_obs_openresty table TTL.
// cutoff is ignored; see MaterializeNodeObsOpenrestyTTL.
func DeleteNodeObsOpenrestyBefore(ctx context.Context, _ time.Time) (int64, error) {
outcome, err := MaterializeNodeObsOpenrestyTTL(ctx)
if err != nil {
return 0, err
}
return outcome.EligibleCount, nil
}
// MaterializeNodeObsOpenrestyTTL force-materializes table TTL and reports an honest outcome.
func MaterializeNodeObsOpenrestyTTL(ctx context.Context) (CleanupOutcome, error) {
conn, err := observabilityConn()
if err != nil {
return CleanupOutcome{}, err
}
tableName := nodeObsOpenrestyTableName()
ttlDays := TableTTLDaysNodeObs ttlDays := TableTTLDaysNodeObs
cutoff := tableTTLCutoff(ttlDays, time.Now()) cutoff := tableTTLCutoff(ttlDays, time.Now())
return materializeExpiredByTableTTL( return materializeExpiredByTableTTL(
@@ -38,20 +38,16 @@ func nodeObservabilityCapturedAtOrderClause() string {
return "captured_at DESC, id DESC" return "captured_at DESC, id DESC"
} }
func nodeObservabilityWindowEndedAtOrderClause() string {
return "window_ended_at DESC, id DESC"
}
func nodeMetricSnapshotTableName() string { func nodeMetricSnapshotTableName() string {
return "of_node_metric_snapshots" return "of_node_metric_snapshots"
} }
func nodeRequestReportTableName() string { func nodeEdgeHealthTableName() string {
return "of_node_request_reports" return "of_node_edge_health"
} }
func nodeObsOpenrestyTableName() string { func accessLogHourlyTableName() string {
return "of_node_obs_openresty" return "of_access_log_hourly"
} }
func nodeObsFrpsTableName() string { func nodeObsFrpsTableName() string {
@@ -66,9 +62,5 @@ func nodeMetricCapacityHourlyTableName() string {
return "of_node_metric_capacity_hourly" return "of_node_metric_capacity_hourly"
} }
func nodeOpenrestyHourlyTableName() string {
return "of_node_openresty_hourly"
}
// clickHouseLimit1ByNodeIDClause selects the first row per node_id after ORDER BY. // clickHouseLimit1ByNodeIDClause selects the first row per node_id after ORDER BY.
const clickHouseLimit1ByNodeIDClause = " LIMIT 1 BY node_id" const clickHouseLimit1ByNodeIDClause = " LIMIT 1 BY node_id"
@@ -35,23 +35,6 @@ func TestListLatestNodeMetricSnapshots_UsesLimit1ByNodeID(t *testing.T) {
assert.Equal(t, since, mock.queryArgs[0][0]) assert.Equal(t, since, mock.queryArgs[0][0])
} }
func TestListLatestNodeRequestReports_UsesLimit1ByNodeID(t *testing.T) {
ctx := context.Background()
mock := &mockConn{}
db.SetChConnForTest(mock)
t.Cleanup(func() { db.SetChConnForTest(nil) })
_, err := ListLatestNodeRequestReports(ctx, NodeObservabilityFilter{NodeID: "node-a"})
require.NoError(t, err)
require.Len(t, mock.queries, 1)
assert.Contains(t, mock.queries[0], "LIMIT 1 BY node_id")
assert.Contains(t, mock.queries[0], nodeRequestReportTableName())
assert.Contains(t, mock.queries[0], "window_ended_at DESC")
require.Len(t, mock.queryArgs, 1)
require.Len(t, mock.queryArgs[0], 1)
assert.Equal(t, "node-a", mock.queryArgs[0][0])
}
func TestListNodeMetricHourly_PrefersRollup(t *testing.T) { func TestListNodeMetricHourly_PrefersRollup(t *testing.T) {
ctx := context.Background() ctx := context.Background()
hour := time.Date(2026, 7, 10, 12, 0, 0, 0, time.UTC) hour := time.Date(2026, 7, 10, 12, 0, 0, 0, time.UTC)
@@ -170,55 +153,3 @@ func TestListNodeMetricHourly_FallsBackToRawOnRollupError(t *testing.T) {
assert.Contains(t, mock.queries[0], nodeMetricCapacityHourlyTableName()) assert.Contains(t, mock.queries[0], nodeMetricCapacityHourlyTableName())
assert.Contains(t, mock.queries[1], "lagInFrame") assert.Contains(t, mock.queries[1], "lagInFrame")
} }
func TestListNodeOpenrestyHourly_PrefersRollup(t *testing.T) {
ctx := context.Background()
hour := time.Date(2026, 7, 10, 14, 0, 0, 0, time.UTC)
since := hour.Add(-1 * time.Hour)
mock := &mockConn{
queryFn: func(_ context.Context, query string, _ ...any) (driver.Rows, error) {
if strings.Contains(query, nodeOpenrestyHourlyTableName()) {
return &mockRows{data: [][]any{{
hour, int64(50), int64(70), uint64(3),
}}}, nil
}
return nil, errors.New("raw path should not be used when rollup covers the window")
},
}
db.SetChConnForTest(mock)
t.Cleanup(func() { db.SetChConnForTest(nil) })
rows, err := ListNodeOpenrestyHourly(ctx, NodeObservabilityFilter{Since: since})
require.NoError(t, err)
require.Len(t, rows, 1)
assert.Equal(t, int64(50), rows[0].OpenrestyRxBytes)
assert.Equal(t, int64(70), rows[0].OpenrestyTxBytes)
assert.Equal(t, 3, rows[0].ReportedNodes)
}
func TestListNodeOpenrestyHourly_FallsBackToRawOnEmptyRollup(t *testing.T) {
ctx := context.Background()
hour := time.Date(2026, 7, 10, 15, 0, 0, 0, time.UTC)
mock := &mockConn{
queryFn: func(_ context.Context, query string, _ ...any) (driver.Rows, error) {
if strings.Contains(query, nodeOpenrestyHourlyTableName()) {
return &mockRows{}, nil
}
if strings.Contains(query, nodeObsOpenrestyTableName()) {
return &mockRows{data: [][]any{{
hour, int64(9), int64(8), uint64(1),
}}}, nil
}
return &mockRows{}, nil
},
}
db.SetChConnForTest(mock)
t.Cleanup(func() { db.SetChConnForTest(nil) })
rows, err := ListNodeOpenrestyHourly(ctx, NodeObservabilityFilter{})
require.NoError(t, err)
require.Len(t, rows, 1)
assert.Equal(t, int64(9), rows[0].OpenrestyRxBytes)
require.GreaterOrEqual(t, len(mock.queries), 2)
assert.Contains(t, mock.queries[1], "lagInFrame")
}
@@ -14,34 +14,35 @@ import (
"github.com/stretchr/testify/require" "github.com/stretchr/testify/require"
) )
func TestInsertNodeObsOpenresty_EmptyNodeID(t *testing.T) { func TestInsertNodeEdgeHealth_EmptyNodeID(t *testing.T) {
err := InsertNodeObsOpenresty(context.Background(), analyticsmodel.NodeObsOpenresty{}) err := InsertNodeEdgeHealth(context.Background(), analyticsmodel.NodeEdgeHealth{})
require.NoError(t, err) require.NoError(t, err)
} }
func TestInsertNodeObsOpenresty_UsesModelBatchSQL(t *testing.T) { func TestInsertNodeEdgeHealth_UsesEdgeHealthBatchSQL(t *testing.T) {
ctx := context.Background() ctx := context.Background()
mockBatch := &mockBatch{} mockBatch := &mockBatch{}
mockConn := &mockConn{ mockConn := &mockConn{
batch: mockBatch, batch: mockBatch,
batchQuery: analyticsmodel.NodeObsOpenresty{}.BatchInsertSQL(), batchQuery: analyticsmodel.NodeEdgeHealth{}.BatchInsertSQL(),
} }
db.SetChConnForTest(mockConn) db.SetChConnForTest(mockConn)
t.Cleanup(func() { db.SetChConnForTest(nil) }) t.Cleanup(func() { db.SetChConnForTest(nil) })
capturedAt := time.Now().UTC() capturedAt := time.Now().UTC()
err := InsertNodeObsOpenresty(ctx, analyticsmodel.NodeObsOpenresty{ err := InsertNodeEdgeHealth(ctx, analyticsmodel.NodeEdgeHealth{
NodeID: "node-a", NodeID: "node-a",
CapturedAt: capturedAt, CapturedAt: capturedAt,
OpenrestyRxBytes: 100, Status: "",
OpenrestyTxBytes: 200, Connections: 3,
OpenrestyConnections: 3, CreatedAt: capturedAt,
CreatedAt: capturedAt,
}) })
require.NoError(t, err) require.NoError(t, err)
assert.True(t, mockConn.prepareCalled) assert.True(t, mockConn.prepareCalled)
assert.Equal(t, analyticsmodel.NodeObsOpenresty{}.BatchInsertSQL(), mockConn.preparedQuery) assert.Equal(t, analyticsmodel.NodeEdgeHealth{}.BatchInsertSQL(), mockConn.preparedQuery)
assert.True(t, mockBatch.sendCalled) assert.True(t, mockBatch.sendCalled)
require.Len(t, mockBatch.rows, 1) require.Len(t, mockBatch.rows, 1)
assert.Equal(t, "node-a", mockBatch.rows[0][1]) assert.Equal(t, "node-a", mockBatch.rows[0][1])
assert.Equal(t, "unknown", mockBatch.rows[0][3]) // status default
assert.Equal(t, int64(3), mockBatch.rows[0][4]) // connections
} }
@@ -14,6 +14,8 @@ import (
analyticsmodel "github.com/Rain-kl/Wavelet/internal/model/analytics" analyticsmodel "github.com/Rain-kl/Wavelet/internal/model/analytics"
) )
const edgeHealthStatusUnknown = "unknown"
// InsertNodeMetricSnapshot writes a single metric snapshot via the batch API. // InsertNodeMetricSnapshot writes a single metric snapshot via the batch API.
func InsertNodeMetricSnapshot(ctx context.Context, snapshot analyticsmodel.NodeMetricSnapshot) error { func InsertNodeMetricSnapshot(ctx context.Context, snapshot analyticsmodel.NodeMetricSnapshot) error {
if strings.TrimSpace(snapshot.NodeID) == "" { if strings.TrimSpace(snapshot.NodeID) == "" {
@@ -78,105 +80,49 @@ func BatchInsertNodeMetricSnapshots(ctx context.Context, snapshots []analyticsmo
return nil return nil
} }
// InsertNodeRequestReport writes a single request report via the batch API. func normalizeEdgeHealthStatus(status string) string {
func InsertNodeRequestReport(ctx context.Context, report analyticsmodel.NodeRequestReport) error { status = strings.TrimSpace(status)
if strings.TrimSpace(report.NodeID) == "" { if status == "" {
return nil return edgeHealthStatusUnknown
} }
return BatchInsertNodeRequestReports(ctx, []analyticsmodel.NodeRequestReport{report}) return status
} }
// BatchInsertNodeRequestReports writes request reports to ClickHouse. // InsertNodeEdgeHealth writes a single edge health snapshot.
func BatchInsertNodeRequestReports(ctx context.Context, reports []analyticsmodel.NodeRequestReport) error { func InsertNodeEdgeHealth(ctx context.Context, row analyticsmodel.NodeEdgeHealth) error {
if len(reports) == 0 { if strings.TrimSpace(row.NodeID) == "" {
return nil
}
return BatchInsertNodeEdgeHealth(ctx, []analyticsmodel.NodeEdgeHealth{row})
}
// BatchInsertNodeEdgeHealth writes L2 OpenResty health snapshots to ClickHouse.
func BatchInsertNodeEdgeHealth(ctx context.Context, rows []analyticsmodel.NodeEdgeHealth) error {
if len(rows) == 0 {
return nil return nil
} }
if db.ChConn == nil { if db.ChConn == nil {
return fmt.Errorf("clickhouse connection is not initialized") return fmt.Errorf("clickhouse connection is not initialized")
} }
batch, err := db.ChConn.PrepareBatch(ctx, analyticsmodel.NodeEdgeHealth{}.BatchInsertSQL())
batch, err := db.ChConn.PrepareBatch(ctx, analyticsmodel.NodeRequestReport{}.BatchInsertSQL())
if err != nil { if err != nil {
return fmt.Errorf("prepare clickhouse batch: %w", err) return fmt.Errorf("prepare clickhouse batch: %w", err)
} }
now := time.Now().UTC() now := time.Now().UTC()
for _, report := range reports { for _, row := range rows {
nodeID := strings.TrimSpace(report.NodeID) nodeID := strings.TrimSpace(row.NodeID)
if nodeID == "" { if nodeID == "" {
continue continue
} }
id := report.ID id := row.ID
if id == 0 { if id == 0 {
id = idgen.NextUint64ID() id = idgen.NextUint64ID()
} }
createdAt := report.CreatedAt createdAt := row.CreatedAt
if createdAt.IsZero() { if createdAt.IsZero() {
createdAt = now createdAt = now
} }
if err := batch.Append( capturedAt := row.CapturedAt.UTC()
id,
nodeID,
report.WindowStartedAt.UTC(),
report.WindowEndedAt.UTC(),
report.RequestCount,
report.ErrorCount,
report.UniqueVisitorCount,
report.StatusCodesJSON,
report.TopDomainsJSON,
report.SourceCountriesJSON,
createdAt.UTC(),
); err != nil {
return fmt.Errorf("append node request report to batch: %w", err)
}
}
if batch.Rows() == 0 {
return nil
}
if err := batch.Send(); err != nil {
return fmt.Errorf("send clickhouse batch: %w", err)
}
return nil
}
// InsertNodeObsOpenresty writes a single OpenResty observation via the batch API.
func InsertNodeObsOpenresty(ctx context.Context, obs analyticsmodel.NodeObsOpenresty) error {
if strings.TrimSpace(obs.NodeID) == "" {
return nil
}
return BatchInsertNodeObsOpenresty(ctx, []analyticsmodel.NodeObsOpenresty{obs})
}
// BatchInsertNodeObsOpenresty writes OpenResty observations to ClickHouse.
func BatchInsertNodeObsOpenresty(ctx context.Context, observations []analyticsmodel.NodeObsOpenresty) error {
if len(observations) == 0 {
return nil
}
if db.ChConn == nil {
return fmt.Errorf("clickhouse connection is not initialized")
}
batch, err := db.ChConn.PrepareBatch(ctx, analyticsmodel.NodeObsOpenresty{}.BatchInsertSQL())
if err != nil {
return fmt.Errorf("prepare clickhouse batch: %w", err)
}
now := time.Now().UTC()
for _, obs := range observations {
nodeID := strings.TrimSpace(obs.NodeID)
if nodeID == "" {
continue
}
id := obs.ID
if id == 0 {
id = idgen.NextUint64ID()
}
createdAt := obs.CreatedAt
if createdAt.IsZero() {
createdAt = now
}
capturedAt := obs.CapturedAt.UTC()
if capturedAt.IsZero() { if capturedAt.IsZero() {
capturedAt = now capturedAt = now
} }
@@ -184,15 +130,13 @@ func BatchInsertNodeObsOpenresty(ctx context.Context, observations []analyticsmo
id, id,
nodeID, nodeID,
capturedAt, capturedAt,
obs.OpenrestyRxBytes, normalizeEdgeHealthStatus(row.Status),
obs.OpenrestyTxBytes, row.Connections,
obs.OpenrestyConnections,
createdAt.UTC(), createdAt.UTC(),
); err != nil { ); err != nil {
return fmt.Errorf("append node openresty observation to batch: %w", err) return fmt.Errorf("append node edge health to batch: %w", err)
} }
} }
if batch.Rows() == 0 { if batch.Rows() == 0 {
return nil return nil
} }
+4 -1
View File
@@ -11,6 +11,9 @@ import (
"path/filepath" "path/filepath"
) )
// formatDetectHeadBytes is the sniff window used for archive format detection.
const formatDetectHeadBytes = 512
// ExtractOptions controls package extraction. // ExtractOptions controls package extraction.
type ExtractOptions struct { type ExtractOptions struct {
// Limits bounds files and sizes during extraction when EnforceLimits is true. // Limits bounds files and sizes during extraction when EnforceLimits is true.
@@ -55,7 +58,7 @@ func ExtractFile(filePath string, format Format, destDir string, opts ExtractOpt
return err return err
} }
if format == "" { if format == "" {
head := make([]byte, 512) head := make([]byte, formatDetectHeadBytes)
n, readErr := file.ReadAt(head, 0) n, readErr := file.ReadAt(head, 0)
if readErr != nil && readErr != io.EOF { if readErr != nil && readErr != io.EOF {
return readErr return readErr
+1 -1
View File
@@ -42,7 +42,7 @@ func InspectFile(filePath string, format Format, opts InspectOptions) (*Manifest
return nil, err return nil, err
} }
if format == "" { if format == "" {
head := make([]byte, 512) head := make([]byte, formatDetectHeadBytes)
n, readErr := file.ReadAt(head, 0) n, readErr := file.ReadAt(head, 0)
if readErr != nil && readErr != io.EOF { if readErr != nil && readErr != io.EOF {
return nil, readErr return nil, readErr
+42 -51
View File
@@ -74,25 +74,35 @@ const (
OpenrestyStatusUnknown = "unknown" OpenrestyStatusUnknown = "unknown"
) )
// NodePayload is the agent node registration payload. // NodePayload is the agent node registration / heartbeat payload.
// schema_version 2: host_metrics + edge_health + access_logs facts only (no business pre-aggregation).
// Agents are destroy/rebuild upgraded; no wire-level compatibility aliases.
type NodePayload struct { type NodePayload struct {
NodeID string `json:"node_id"` SchemaVersion int `json:"schema_version,omitempty"`
Name string `json:"name"` NodeID string `json:"node_id"`
IP string `json:"ip"` Name string `json:"name"`
Version string `json:"version"` IP string `json:"ip"`
ExtVersion string `json:"ext_version"` Version string `json:"version"`
CurrentVersion string `json:"current_version"` ExtVersion string `json:"ext_version"`
LastError string `json:"last_error"` CurrentVersion string `json:"current_version"`
OpenrestyStatus string `json:"openresty_status"` LastError string `json:"last_error"`
OpenrestyMessage string `json:"openresty_message"` OpenrestyStatus string `json:"openresty_status"`
Profile *NodeSystemProfile `json:"profile,omitempty"` OpenrestyMessage string `json:"openresty_message"`
Snapshot *NodeMetricSnapshot `json:"snapshot,omitempty"` Profile *NodeSystemProfile `json:"profile,omitempty"`
OpenrestyObservation *NodeOpenrestyObservation `json:"openresty_observation,omitempty"` HostMetrics *NodeMetricSnapshot `json:"host_metrics,omitempty"`
TrafficReport *NodeTrafficReport `json:"traffic_report,omitempty"` EdgeHealth *NodeEdgeHealth `json:"edge_health,omitempty"`
AccessLogs []NodeAccessLog `json:"access_logs,omitempty"` AccessLogs []NodeAccessLog `json:"access_logs,omitempty"`
BufferedObservability []BufferedObservabilityRecord `json:"buffered_observability,omitempty"` Buffered []BufferedObservabilityRecord `json:"buffered,omitempty"`
HealthEvents []NodeHealthEvent `json:"health_events"` HealthEvents []NodeHealthEvent `json:"health_events"`
WAFIPGroupChecksums map[string]string `json:"waf_ip_group_checksums,omitempty"` WAFIPGroupChecksums map[string]string `json:"waf_ip_group_checksums,omitempty"`
}
// NodeEdgeHealth is an instantaneous OpenResty health snapshot (L2).
type NodeEdgeHealth struct {
CapturedAtUnix int64 `json:"captured_at_unix"`
Status string `json:"status"`
Message string `json:"message"`
Connections int64 `json:"connections"`
} }
// NodeSystemProfile describes the system profile of a node. // NodeSystemProfile describes the system profile of a node.
@@ -124,43 +134,24 @@ type NodeMetricSnapshot struct {
NetworkTxBytes int64 `json:"network_tx_bytes"` NetworkTxBytes int64 `json:"network_tx_bytes"`
} }
// NodeOpenrestyObservation holds OpenResty observation data. // NodeAccessLog is an access log entry from agent (L1 business fact).
type NodeOpenrestyObservation struct {
CapturedAtUnix int64 `json:"captured_at_unix"`
OpenrestyRxBytes int64 `json:"openresty_rx_bytes"`
OpenrestyTxBytes int64 `json:"openresty_tx_bytes"`
OpenrestyConnections int64 `json:"openresty_connections"`
}
// NodeTrafficReport is a traffic report from agent.
type NodeTrafficReport struct {
WindowStartedAtUnix int64 `json:"window_started_at_unix"`
WindowEndedAtUnix int64 `json:"window_ended_at_unix"`
RequestCount int64 `json:"request_count"`
ErrorCount int64 `json:"error_count"`
UniqueVisitorCount int64 `json:"unique_visitor_count"`
StatusCodes map[string]int64 `json:"status_codes"`
TopDomains map[string]int64 `json:"top_domains"`
SourceCountries map[string]int64 `json:"source_countries"`
}
// NodeAccessLog is an access log entry from agent.
type NodeAccessLog struct { type NodeAccessLog struct {
LoggedAtUnix int64 `json:"logged_at_unix"` LoggedAtUnix int64 `json:"logged_at_unix"`
RemoteAddr string `json:"remote_addr"` RemoteAddr string `json:"remote_addr"`
Host string `json:"host"` Host string `json:"host"`
Path string `json:"path"` Path string `json:"path"`
StatusCode int `json:"status_code"` StatusCode int `json:"status_code"`
BytesSent int64 `json:"bytes_sent"` BytesSent int64 `json:"bytes_sent"` // body bytes = 已提供数据
RequestLength int64 `json:"request_length"` // 接收数据
RequestTimeMs int64 `json:"request_time_ms"` // optional
} }
// BufferedObservabilityRecord is a buffered observability record. // BufferedObservabilityRecord is a buffered observability record (facts only).
type BufferedObservabilityRecord struct { type BufferedObservabilityRecord struct {
WindowStartedAtUnix int64 `json:"window_started_at_unix"` CapturedAtUnix int64 `json:"captured_at_unix,omitempty"`
Snapshot *NodeMetricSnapshot `json:"snapshot,omitempty"` HostMetrics *NodeMetricSnapshot `json:"host_metrics,omitempty"`
OpenrestyObservation *NodeOpenrestyObservation `json:"openresty_observation,omitempty"` EdgeHealth *NodeEdgeHealth `json:"edge_health,omitempty"`
TrafficReport *NodeTrafficReport `json:"traffic_report,omitempty"` AccessLogs []NodeAccessLog `json:"access_logs,omitempty"`
AccessLogs []NodeAccessLog `json:"access_logs,omitempty"`
} }
// NodeHealthEvent represents a node health event. // NodeHealthEvent represents a node health event.