Compare commits

...

8 Commits

Author SHA1 Message Date
sagit b314192621 feat(monitoring): add node/tunnel metrics, service monitors, and health checks (#331)
## Summary
- Add comprehensive monitoring system with
NodeMetric/TunnelMetric/ServiceMonitor models
- Implement metrics ingestion service with per-minute bucket aggregation
and upsert support
- Add health checker for node connectivity monitoring with configurable
intervals
- Wire node metrics from WebSocket SystemInfo messages to metrics
service
- Add tunnel metrics ingestion from flow upload endpoint with
transaction support
- Create monitoring REST API endpoints for nodes, tunnels, and services
- Implement service monitor CRUD and execution (TCP/ICMP health checks)
- Add MonitorPermission model for non-admin access control to monitoring
features
- Create frontend monitor page with node/tunnel/service views
- Include schema migration (v6) for tunnel_metric unique index and
deduplication
- Fix tunnel entry port conflict validation to use transaction (Tx
variants)
2026-03-18 15:12:22 +08:00
sagit 1e5f9bfb04 Merge branch 'main' into opencode/shiny-falcon 2026-03-18 14:17:06 +08:00
sagitchu 5972378897 fix: resolve merge conflicts and fix monitoring bugs
- Add missing 'uptime' field to NodeMetricApiItem type definition
- Fix WS message handling: non-UpgradeProgress typed messages now
  broadcast via broadcastInfo instead of being silently dropped
- Strengthen looksLikeSystemInfoMessage heuristic to require ≥3
  matching keys to avoid false positives
- Fix tab/space indentation inconsistency in admin.tsx useEffect
- Remove duplicate method declarations from merge (repository_control,
  mutations)
- Update tunnel_entry_sqlite_test to use renamed Tx suffix function
2026-03-18 14:12:47 +08:00
sagitchu 455900ba41 Merge branch 'main' into opencode/shiny-falcon
# Conflicts:
#	go-backend/internal/http/handler/mutations.go
#	go-backend/tests/contract/issue313_entry_port_conflict_contract_test.go
2026-03-18 14:09:00 +08:00
sagit 85e57213ee chore: update knowledge base metadata for release 2.1.8 (#337)
Updating AGENTS.md with new release version and current commit hash.
2026-03-18 13:52:28 +08:00
sagitchu 1377061234 chore: update knowledge base metadata for release 2.1.8 2026-03-18 13:49:41 +08:00
sagitchu 46a60376c4 Merge remote-tracking branch 'origin/main' into opencode/shiny-falcon
# Conflicts:
#	vite-frontend/src/pages/node.tsx
#	vite-frontend/src/pages/user.tsx
2026-03-17 15:18:25 +08:00
sagitchu 9de240f034 feat(monitoring): add node/tunnel metrics, service monitors, and health checks
- Add NodeMetric/TunnelMetric/ServiceMonitor models and repository methods
- Implement metrics ingestion service with per-minute bucket aggregation
- Add health checker for node connectivity monitoring
- Wire node metrics from WebSocket SystemInfo messages
- Add tunnel metrics ingestion from flow upload endpoint
- Create monitoring REST API endpoints for nodes, tunnels, services
- Implement service monitor CRUD and execution (TCP/ICMP checks)
- Add MonitorPermission for non-admin access control
- Create frontend monitor page with node/tunnel/service views
- Add tunnel metrics ingestion from agent flow reports
- Include schema migration for tunnel_metric unique index
- Fix tunnel entry port conflict validation to use transaction

Entire-Checkpoint: 030821a7c8e3
2026-03-17 14:59:09 +08:00
44 changed files with 8504 additions and 75 deletions
+3
View File
@@ -62,6 +62,9 @@ go-gost/ss/
.classpath
.project
.settings/
# OpenCode session metadata
.entire/
bin/
tmp/
*.swp
+3 -3
View File
@@ -1,9 +1,9 @@
# PROJECT KNOWLEDGE BASE
**Generated:** Thu Feb 26 2026
**Commit:** 21008cc
**Generated:** Wed Mar 18 2026
**Commit:** ea21a7d
**Branch:** main
**Tag:** 2.1.5-rc15
**Tag:** 2.1.8
## OVERVIEW
FLVX (formerly Flux Panel) is a traffic forwarding management system built on a forked GOST v3 stack. It ships as a Go-based admin API (SQLite/PostgreSQL) + Vite/React UI + Go forwarding agent, with optional mobile WebView wrappers.
+360
View File
@@ -0,0 +1,360 @@
package health
import (
"context"
"errors"
"fmt"
"log"
"net"
"strings"
"sync"
"time"
"go-backend/internal/monitoring"
"go-backend/internal/store/model"
"go-backend/internal/store/repo"
"go-backend/internal/ws"
)
type nodeCommander interface {
SendCommand(nodeID int64, cmdType string, data interface{}, timeout time.Duration) (ws.CommandResult, error)
}
type Checker struct {
repo *repo.Repository
commander nodeCommander
lastRun map[int64]int64
inFlight map[int64]struct{}
mu sync.RWMutex
cancel context.CancelFunc
wg sync.WaitGroup
}
func NewChecker(repo *repo.Repository, commander nodeCommander) *Checker {
return &Checker{
repo: repo,
commander: commander,
lastRun: make(map[int64]int64),
inFlight: make(map[int64]struct{}),
}
}
func (c *Checker) Start(ctx context.Context) {
c.mu.Lock()
ctx, cancel := context.WithCancel(ctx)
c.cancel = cancel
c.mu.Unlock()
c.runChecks(ctx)
for {
limits := c.loadServiceMonitorLimits()
scanInterval := time.Duration(limits.CheckerScanIntervalSec) * time.Second
if scanInterval <= 0 {
scanInterval = 30 * time.Second
}
timer := time.NewTimer(scanInterval)
select {
case <-ctx.Done():
timer.Stop()
return
case <-timer.C:
c.runChecks(ctx)
}
}
}
func (c *Checker) Stop() {
c.mu.Lock()
if c.cancel != nil {
c.cancel()
}
c.mu.Unlock()
c.wg.Wait()
}
func (c *Checker) RunOnce(m *model.ServiceMonitor) (*model.ServiceMonitorResult, error) {
if c == nil {
return nil, errors.New("checker not initialized")
}
if m == nil {
return nil, errors.New("monitor is nil")
}
limits := c.loadServiceMonitorLimits()
return c.executeCheck(m, time.Now().UnixMilli(), limits), nil
}
func (c *Checker) runChecks(ctx context.Context) {
if c == nil || c.repo == nil {
return
}
limits := c.loadServiceMonitorLimits()
monitors, err := c.repo.ListEnabledServiceMonitors()
if err != nil {
log.Printf("service monitor scheduler failed op=list_enabled err=%v", err)
return
}
if len(monitors) == 0 {
return
}
// Use persisted result timestamps to avoid restart bursts.
latest, err := c.repo.GetLatestServiceMonitorResults()
if err != nil {
log.Printf("service monitor scheduler failed op=get_latest_results err=%v", err)
latest = nil
}
persistedLast := make(map[int64]int64, len(latest))
for _, r := range latest {
if r.MonitorID <= 0 || r.Timestamp <= 0 {
continue
}
persistedLast[r.MonitorID] = r.Timestamp
}
now := time.Now().UnixMilli()
due := make([]model.ServiceMonitor, 0, len(monitors))
for _, m := range monitors {
select {
case <-ctx.Done():
return
default:
}
intervalSec := m.IntervalSec
if intervalSec <= 0 {
intervalSec = limits.DefaultIntervalSec
}
if intervalSec < limits.MinIntervalSec {
intervalSec = limits.MinIntervalSec
}
intervalMs := int64(intervalSec) * 1000
c.mu.Lock()
if _, ok := c.inFlight[m.ID]; ok {
c.mu.Unlock()
continue
}
lastSeen := persistedLast[m.ID]
if v := c.lastRun[m.ID]; v > lastSeen {
lastSeen = v
}
if lastSeen > 0 && intervalMs > 0 && now-lastSeen < intervalMs {
c.mu.Unlock()
continue
}
c.inFlight[m.ID] = struct{}{}
// Use now as a best-effort guard against overlapping scans; the final
// timestamp is updated again when the result is persisted.
c.lastRun[m.ID] = now
c.mu.Unlock()
due = append(due, m)
}
if len(due) == 0 {
return
}
workerLimit := limits.WorkerLimit
if workerLimit <= 0 {
workerLimit = 1
}
if workerLimit > len(due) {
workerLimit = len(due)
}
jobs := make(chan model.ServiceMonitor, len(due))
for _, m := range due {
jobs <- m
}
close(jobs)
for i := 0; i < workerLimit; i++ {
c.wg.Add(1)
go func() {
defer c.wg.Done()
for {
select {
case <-ctx.Done():
return
case m, ok := <-jobs:
if !ok {
return
}
ts := time.Now().UnixMilli()
result := c.executeCheck(&m, ts, limits)
if err := c.repo.InsertServiceMonitorResult(result); err != nil {
log.Printf("monitoring write failed op=service_monitor_result.insert monitor_id=%d err=%v", result.MonitorID, err)
}
c.mu.Lock()
c.lastRun[m.ID] = result.Timestamp
delete(c.inFlight, m.ID)
c.mu.Unlock()
}
}
}()
}
}
func (c *Checker) executeCheck(m *model.ServiceMonitor, timestamp int64, limits monitoring.ServiceMonitorLimits) *model.ServiceMonitorResult {
result := &model.ServiceMonitorResult{
MonitorID: m.ID,
NodeID: m.NodeID,
Timestamp: timestamp,
}
timeoutSec := m.TimeoutSec
if timeoutSec <= 0 {
timeoutSec = limits.DefaultTimeoutSec
}
if timeoutSec < limits.MinTimeoutSec {
timeoutSec = limits.MinTimeoutSec
}
if timeoutSec > limits.MaxTimeoutSec {
timeoutSec = limits.MaxTimeoutSec
}
timeout := time.Duration(timeoutSec) * time.Second
// When nodeId is set, run checks on the specified node.
if m.NodeID > 0 {
c.checkOnNode(m, timeoutSec, timeout, result)
return result
}
switch strings.ToLower(strings.TrimSpace(m.Type)) {
case "tcp":
c.checkTCP(m.Target, timeout, result)
case "icmp":
result.Success = 0
result.ErrorMessage = "ICMP 监控必须指定执行节点"
default:
result.Success = 0
result.ErrorMessage = fmt.Sprintf("不支持的检查类型: %s", m.Type)
}
return result
}
func (c *Checker) loadServiceMonitorLimits() monitoring.ServiceMonitorLimits {
defaults := monitoring.DefaultServiceMonitorLimits()
if c == nil || c.repo == nil {
return defaults
}
cfg, err := c.repo.GetConfigsByNames([]string{
monitoring.ConfigServiceMonitorCheckerScanIntervalSec,
monitoring.ConfigServiceMonitorWorkerLimit,
monitoring.ConfigServiceMonitorMinIntervalSec,
monitoring.ConfigServiceMonitorDefaultIntervalSec,
monitoring.ConfigServiceMonitorMinTimeoutSec,
monitoring.ConfigServiceMonitorDefaultTimeoutSec,
monitoring.ConfigServiceMonitorMaxTimeoutSec,
})
if err != nil {
return defaults
}
return monitoring.ServiceMonitorLimitsFromConfigMap(cfg)
}
type serviceMonitorCheckRequest struct {
MonitorID int64 `json:"monitorId"`
Type string `json:"type"`
Target string `json:"target"`
TimeoutSec int `json:"timeoutSec"`
}
func (c *Checker) checkOnNode(m *model.ServiceMonitor, timeoutSec int, timeout time.Duration, result *model.ServiceMonitorResult) {
if c == nil || m == nil || result == nil {
return
}
if c.commander == nil {
result.Success = 0
result.ErrorMessage = "节点检查不可用"
return
}
checkType := strings.ToLower(strings.TrimSpace(m.Type))
if checkType != "tcp" && checkType != "icmp" {
result.Success = 0
result.ErrorMessage = fmt.Sprintf("不支持的检查类型: %s", m.Type)
return
}
if strings.TrimSpace(m.Target) == "" {
result.Success = 0
result.ErrorMessage = "检查目标为空"
return
}
req := serviceMonitorCheckRequest{
MonitorID: m.ID,
Type: checkType,
Target: m.Target,
TimeoutSec: timeoutSec,
}
cmdTimeout := timeout
if cmdTimeout < 2*time.Second {
cmdTimeout = 2 * time.Second
}
cmdTimeout = cmdTimeout + 2*time.Second
cmdRes, err := c.commander.SendCommand(m.NodeID, "ServiceMonitorCheck", req, cmdTimeout)
if err != nil {
result.Success = 0
result.ErrorMessage = err.Error()
return
}
if cmdRes.Data == nil {
result.Success = 0
result.ErrorMessage = "节点返回为空"
return
}
if v, ok := cmdRes.Data["success"]; ok {
if b, ok := v.(bool); ok {
if b {
result.Success = 1
} else {
result.Success = 0
}
}
}
if v, ok := cmdRes.Data["latencyMs"]; ok {
if f, ok := v.(float64); ok {
result.LatencyMs = f
}
}
if v, ok := cmdRes.Data["statusCode"]; ok {
if f, ok := v.(float64); ok {
result.StatusCode = int(f)
}
}
if v, ok := cmdRes.Data["errorMessage"]; ok {
if s, ok := v.(string); ok {
result.ErrorMessage = s
}
}
}
func (c *Checker) checkTCP(target string, timeout time.Duration, result *model.ServiceMonitorResult) {
start := time.Now()
conn, err := net.DialTimeout("tcp", target, timeout)
latency := time.Since(start)
result.LatencyMs = float64(latency.Milliseconds())
if err != nil {
result.Success = 0
result.ErrorMessage = err.Error()
return
}
_ = conn.Close()
result.Success = 1
}
+477
View File
@@ -0,0 +1,477 @@
package health
import (
"context"
"net"
"testing"
"time"
"go-backend/internal/monitoring"
"go-backend/internal/store/model"
"go-backend/internal/store/repo"
"go-backend/internal/ws"
)
type fakeCommander struct {
lastNodeID int64
lastType string
lastData interface{}
res ws.CommandResult
err error
}
type delayedCommander struct {
delayByMonitorID map[int64]time.Duration
}
func (d *delayedCommander) SendCommand(nodeID int64, cmdType string, data interface{}, _ time.Duration) (ws.CommandResult, error) {
_ = nodeID
_ = cmdType
if req, ok := data.(serviceMonitorCheckRequest); ok {
if delay := d.delayByMonitorID[req.MonitorID]; delay > 0 {
time.Sleep(delay)
}
}
return ws.CommandResult{
Success: true,
Data: map[string]interface{}{
"success": true,
"latencyMs": float64(1),
},
}, nil
}
func (f *fakeCommander) SendCommand(nodeID int64, cmdType string, data interface{}, _ time.Duration) (ws.CommandResult, error) {
f.lastNodeID = nodeID
f.lastType = cmdType
f.lastData = data
return f.res, f.err
}
func TestTCPHealthCheckViaMonitor(t *testing.T) {
listener, err := net.Listen("tcp", "127.0.0.1:0")
if err != nil {
t.Fatalf("listen: %v", err)
}
defer listener.Close()
addr := listener.Addr().String()
go func() {
for {
conn, err := listener.Accept()
if err != nil {
return
}
conn.Close()
}
}()
t.Run("successful tcp check", func(t *testing.T) {
checker := NewChecker(nil, nil)
limits := checker.loadServiceMonitorLimits()
now := time.Now().UnixMilli()
monitor := &model.ServiceMonitor{
Type: "tcp",
Target: addr,
TimeoutSec: 5,
}
result := checker.executeCheck(monitor, now, limits)
if result.Success != 1 {
t.Fatalf("expected success, got error: %s", result.ErrorMessage)
}
if result.LatencyMs < 0 {
t.Fatalf("expected non-negative latency, got %f", result.LatencyMs)
}
})
t.Run("failed tcp check - connection refused", func(t *testing.T) {
checker := NewChecker(nil, nil)
limits := checker.loadServiceMonitorLimits()
now := time.Now().UnixMilli()
monitor := &model.ServiceMonitor{
Type: "tcp",
Target: "127.0.0.1:1",
TimeoutSec: 1,
}
result := checker.executeCheck(monitor, now, limits)
if result.Success == 1 {
t.Fatalf("expected failure for connection refused")
}
if result.ErrorMessage == "" {
t.Fatalf("expected error message")
}
})
}
func TestCheckerRunChecks(t *testing.T) {
r, err := repo.Open(":memory:")
if err != nil {
t.Fatalf("open repo: %v", err)
}
defer r.Close()
listener, err := net.Listen("tcp", "127.0.0.1:0")
if err != nil {
t.Fatalf("listen: %v", err)
}
defer listener.Close()
tcpAddr := listener.Addr().String()
go func() {
for {
conn, err := listener.Accept()
if err != nil {
return
}
conn.Close()
}
}()
now := time.Now().UnixMilli()
monitors := []*model.ServiceMonitor{
{
Name: "TCP Monitor",
Type: "tcp",
Target: tcpAddr,
IntervalSec: 60,
TimeoutSec: 5,
NodeID: 0,
Enabled: 1,
CreatedTime: now,
UpdatedTime: now,
},
{
Name: "TCP Monitor 2",
Type: "tcp",
Target: tcpAddr,
IntervalSec: 60,
TimeoutSec: 5,
NodeID: 0,
Enabled: 1,
CreatedTime: now,
UpdatedTime: now,
},
{
Name: "Disabled Monitor",
Type: "tcp",
Target: "127.0.0.1:1",
IntervalSec: 60,
TimeoutSec: 5,
NodeID: 0,
Enabled: 0,
CreatedTime: now,
UpdatedTime: now,
},
}
for _, m := range monitors {
if err := r.CreateServiceMonitor(m); err != nil {
t.Fatalf("create monitor: %v", err)
}
}
monitors[2].Enabled = 0
if err := r.UpdateServiceMonitor(monitors[2]); err != nil {
t.Fatalf("update disabled monitor: %v", err)
}
enabledMonitors, err := r.ListEnabledServiceMonitors()
if err != nil {
t.Fatalf("list enabled monitors: %v", err)
}
if len(enabledMonitors) != 2 {
t.Fatalf("expected 2 enabled monitors, got %d", len(enabledMonitors))
}
checker := NewChecker(r, nil)
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Second)
defer cancel()
go checker.Start(ctx)
time.Sleep(500 * time.Millisecond)
results, err := r.GetServiceMonitorResults(monitors[0].ID, 10)
if err != nil {
t.Fatalf("get tcp results: %v", err)
}
if len(results) == 0 {
t.Fatalf("expected at least one result for tcp monitor")
}
for _, res := range results {
if res.Success != 1 {
t.Fatalf("expected success for tcp monitor, got failure: %s", res.ErrorMessage)
}
}
results2, err := r.GetServiceMonitorResults(monitors[1].ID, 10)
if err != nil {
t.Fatalf("get tcp results 2: %v", err)
}
if len(results2) == 0 {
t.Fatalf("expected at least one result for tcp monitor 2")
}
for _, res := range results2 {
if res.Success != 1 {
t.Fatalf("expected success for tcp monitor 2, got failure: %s", res.ErrorMessage)
}
}
disabledResults, err := r.GetServiceMonitorResults(monitors[2].ID, 10)
if err != nil {
t.Fatalf("get disabled results: %v", err)
}
if len(disabledResults) != 0 {
t.Fatalf("expected no results for disabled monitor, got %d", len(disabledResults))
}
}
func TestCheckerUnsupportedType(t *testing.T) {
checker := NewChecker(nil, nil)
limits := checker.loadServiceMonitorLimits()
now := time.Now().UnixMilli()
monitor := &model.ServiceMonitor{
Type: "http",
Target: "https://example.com",
TimeoutSec: 5,
}
result := checker.executeCheck(monitor, now, limits)
if result.Success == 1 {
t.Fatalf("expected failure for unsupported type")
}
if result.ErrorMessage == "" {
t.Fatalf("expected error message for unsupported type")
}
}
func TestCheckerDefaultTimeout(t *testing.T) {
listener, err := net.Listen("tcp", "127.0.0.1:0")
if err != nil {
t.Fatalf("listen: %v", err)
}
defer listener.Close()
addr := listener.Addr().String()
go func() {
for {
conn, err := listener.Accept()
if err != nil {
return
}
conn.Close()
}
}()
checker := NewChecker(nil, nil)
limits := checker.loadServiceMonitorLimits()
now := time.Now().UnixMilli()
monitor := &model.ServiceMonitor{
Type: "tcp",
Target: addr,
TimeoutSec: 0,
}
result := checker.executeCheck(monitor, now, limits)
if result.Success != 1 {
t.Fatalf("expected success with default timeout, got error: %s", result.ErrorMessage)
}
}
func TestCheckerStop(t *testing.T) {
r, err := repo.Open(":memory:")
if err != nil {
t.Fatalf("open repo: %v", err)
}
defer r.Close()
listener, err := net.Listen("tcp", "127.0.0.1:0")
if err != nil {
t.Fatalf("listen: %v", err)
}
defer listener.Close()
go func() {
for {
conn, err := listener.Accept()
if err != nil {
return
}
conn.Close()
}
}()
now := time.Now().UnixMilli()
monitor := &model.ServiceMonitor{
Name: "Test Monitor",
Type: "tcp",
Target: listener.Addr().String(),
IntervalSec: 60,
TimeoutSec: 5,
NodeID: 0,
Enabled: 1,
CreatedTime: now,
UpdatedTime: now,
}
if err := r.CreateServiceMonitor(monitor); err != nil {
t.Fatalf("create monitor: %v", err)
}
checker := NewChecker(r, nil)
ctx := context.Background()
go checker.Start(ctx)
time.Sleep(100 * time.Millisecond)
checker.Stop()
results, err := r.GetServiceMonitorResults(monitor.ID, 10)
if err != nil {
t.Fatalf("get results: %v", err)
}
if len(results) == 0 {
t.Fatalf("expected at least one result before stop")
}
}
func TestCheckerRunsOnNodeWhenNodeIDSet(t *testing.T) {
fake := &fakeCommander{
res: ws.CommandResult{
Success: true,
Data: map[string]interface{}{
"success": false,
"latencyMs": float64(12),
"errorMessage": "unreachable",
},
},
}
checker := NewChecker(nil, fake)
limits := checker.loadServiceMonitorLimits()
now := time.Now().UnixMilli()
monitor := &model.ServiceMonitor{
ID: 99,
Type: "icmp",
Target: "8.8.8.8",
TimeoutSec: 2,
NodeID: 123,
}
res := checker.executeCheck(monitor, now, limits)
if fake.lastNodeID != 123 {
t.Fatalf("expected command to be sent to node 123, got %d", fake.lastNodeID)
}
if fake.lastType != "ServiceMonitorCheck" {
t.Fatalf("expected ServiceMonitorCheck command, got %s", fake.lastType)
}
if res.Success != 0 {
t.Fatalf("expected failed result from node check")
}
if res.ErrorMessage != "unreachable" {
t.Fatalf("expected errorMessage unreachable, got %q", res.ErrorMessage)
}
}
func TestCheckerDoesNotBurstOnRestartWhenRecentResultsExist(t *testing.T) {
r, err := repo.Open(":memory:")
if err != nil {
t.Fatalf("open repo: %v", err)
}
defer r.Close()
now := time.Now().UnixMilli()
monitor := &model.ServiceMonitor{
Name: "recent-monitor",
Type: "tcp",
Target: "127.0.0.1:1",
IntervalSec: 60,
TimeoutSec: 1,
NodeID: 0,
Enabled: 1,
CreatedTime: now,
UpdatedTime: now,
}
if err := r.CreateServiceMonitor(monitor); err != nil {
t.Fatalf("create monitor: %v", err)
}
if err := r.InsertServiceMonitorResult(&model.ServiceMonitorResult{
MonitorID: monitor.ID,
NodeID: 0,
Timestamp: now - 10_000,
Success: 1,
}); err != nil {
t.Fatalf("seed recent result: %v", err)
}
checker := NewChecker(r, nil)
ctx, cancel := context.WithCancel(context.Background())
go checker.Start(ctx)
// Give the initial scan a chance to run.
time.Sleep(200 * time.Millisecond)
cancel()
checker.Stop()
results, err := r.GetServiceMonitorResults(monitor.ID, 10)
if err != nil {
t.Fatalf("get results: %v", err)
}
if len(results) != 1 {
t.Fatalf("expected no immediate rerun (1 result), got %d", len(results))
}
}
func TestCheckerConcurrencyPreventsSlowMonitorBlockingOthers(t *testing.T) {
r, err := repo.Open(":memory:")
if err != nil {
t.Fatalf("open repo: %v", err)
}
defer r.Close()
now := time.Now().UnixMilli()
// Force worker limit to at least 2 for this test.
_ = r.UpsertConfig(monitoring.ConfigServiceMonitorWorkerLimit, "2", now)
slow := &model.ServiceMonitor{
Name: "slow",
Type: "icmp",
Target: "8.8.8.8",
IntervalSec: 60,
TimeoutSec: 1,
NodeID: 123,
Enabled: 1,
CreatedTime: now,
UpdatedTime: now,
}
if err := r.CreateServiceMonitor(slow); err != nil {
t.Fatalf("create slow monitor: %v", err)
}
fast := &model.ServiceMonitor{
Name: "fast",
Type: "icmp",
Target: "1.1.1.1",
IntervalSec: 60,
TimeoutSec: 1,
NodeID: 123,
Enabled: 1,
CreatedTime: now,
UpdatedTime: now,
}
if err := r.CreateServiceMonitor(fast); err != nil {
t.Fatalf("create fast monitor: %v", err)
}
cmd := &delayedCommander{delayByMonitorID: map[int64]time.Duration{slow.ID: 800 * time.Millisecond}}
checker := NewChecker(r, cmd)
ctx, cancel := context.WithCancel(context.Background())
go checker.Start(ctx)
// Fast monitor should complete even while slow one is still running.
time.Sleep(250 * time.Millisecond)
results, err := r.GetServiceMonitorResults(fast.ID, 10)
if err != nil {
t.Fatalf("get fast results: %v", err)
}
if len(results) == 0 {
t.Fatalf("expected fast monitor to have results without waiting for slow")
}
cancel()
checker.Stop()
}
+47 -3
View File
@@ -16,17 +16,21 @@ import (
"time"
"go-backend/internal/auth"
"go-backend/internal/health"
"go-backend/internal/http/middleware"
"go-backend/internal/http/response"
"go-backend/internal/metrics"
"go-backend/internal/security"
"go-backend/internal/store/repo"
"go-backend/internal/ws"
)
type Handler struct {
repo *repo.Repository
jwtSecret string
wsServer *ws.Server
repo *repo.Repository
jwtSecret string
wsServer *ws.Server
metrics *metrics.IngestionService
healthCheck *health.Checker
captchaMu sync.Mutex
captchaTokens map[string]int64
@@ -83,10 +87,31 @@ func New(repo *repo.Repository, jwtSecret string) *Handler {
repo: repo,
jwtSecret: jwtSecret,
wsServer: ws.NewServer(repo, jwtSecret),
metrics: metrics.NewIngestionService(repo),
healthCheck: nil,
captchaTokens: make(map[string]int64),
pendingUpgradeRedeploy: make(map[int64]struct{}),
}
h.healthCheck = health.NewChecker(repo, h.wsServer)
h.wsServer.SetNodeOnlineHook(h.onNodeOnline)
h.wsServer.SetNodeMetricHook(func(nodeID int64, info ws.SystemInfo) {
metricInfo := metrics.SystemInfo{
Uptime: info.Uptime,
BytesReceived: info.BytesReceived,
BytesTransmitted: info.BytesTransmitted,
CPUUsage: info.CPUUsage,
MemoryUsage: info.MemoryUsage,
DiskUsage: info.DiskUsage,
Load1: info.Load1,
Load5: info.Load5,
Load15: info.Load15,
TCPConns: info.TCPConns,
UDPConns: info.UDPConns,
NetInSpeed: info.NetInSpeed,
NetOutSpeed: info.NetOutSpeed,
}
h.metrics.RecordNodeMetric(nodeID, metricInfo)
})
return h
}
@@ -200,6 +225,23 @@ func (h *Handler) Register(mux *http.ServeMux) {
mux.HandleFunc("/api/v1/announcement/get", h.getAnnouncement)
mux.HandleFunc("/api/v1/announcement/update", h.updateAnnouncement)
mux.HandleFunc("/api/v1/monitor/access", h.monitorAccessHandler)
mux.HandleFunc("/api/v1/monitor/nodes/", h.monitorNodeMetricsHandler)
mux.HandleFunc("/api/v1/monitor/nodes", h.monitorNodeListHandler)
mux.HandleFunc("/api/v1/monitor/tunnels", h.monitorTunnelListHandler)
mux.HandleFunc("/api/v1/monitor/tunnels/", h.monitorTunnelMetrics)
mux.HandleFunc("/api/v1/monitor/services", h.monitorServiceListHandler)
mux.HandleFunc("/api/v1/monitor/services/create", h.monitorServiceCreate)
mux.HandleFunc("/api/v1/monitor/services/update", h.monitorServiceUpdate)
mux.HandleFunc("/api/v1/monitor/services/delete", h.monitorServiceDelete)
mux.HandleFunc("/api/v1/monitor/services/run", h.monitorServiceRun)
mux.HandleFunc("/api/v1/monitor/services/latest-results", h.monitorServiceLatestResultsHandler)
mux.HandleFunc("/api/v1/monitor/services/limits", h.monitorServiceLimitsHandler)
mux.HandleFunc("/api/v1/monitor/services/", h.monitorServiceResultsHandler)
mux.HandleFunc("/api/v1/monitor/permission/list", h.monitorPermissionList)
mux.HandleFunc("/api/v1/monitor/permission/assign", h.monitorPermissionAssign)
mux.HandleFunc("/api/v1/monitor/permission/remove", h.monitorPermissionRemove)
mux.HandleFunc("/flow/test", h.flowTest)
mux.HandleFunc("/flow/config", h.flowConfig)
mux.HandleFunc("/flow/upload", h.flowUpload)
@@ -728,6 +770,8 @@ func (h *Handler) flowUpload(w http.ResponseWriter, r *http.Request) {
if err == nil && strings.TrimSpace(raw) != "" {
var items []flowItem
if json.Unmarshal([]byte(raw), &items) == nil {
nowMs := time.Now().UnixMilli()
h.recordTunnelMetricsFromFlowItems(node.ID, items, nowMs)
for _, item := range items {
h.processFlowItem(node.ID, item)
}
+17 -1
View File
@@ -18,12 +18,14 @@ func (h *Handler) StartBackgroundJobs() {
ctx, cancel := context.WithCancel(context.Background())
h.jobsCancel = cancel
h.jobsStarted = true
h.jobsWG.Add(3)
h.jobsWG.Add(5)
h.jobsMu.Unlock()
go h.runHourlyStatsLoop(ctx)
go h.runDailyMaintenanceLoop(ctx)
go h.runNodeRenewalCycleLoop(ctx)
go h.runMetricsIngestion(ctx)
go h.runHealthChecks(ctx)
}
func (h *Handler) StopBackgroundJobs() {
@@ -47,6 +49,20 @@ func (h *Handler) StopBackgroundJobs() {
h.jobsWG.Wait()
}
func (h *Handler) runMetricsIngestion(ctx context.Context) {
defer h.jobsWG.Done()
if h.metrics != nil {
h.metrics.Start(ctx)
}
}
func (h *Handler) runHealthChecks(ctx context.Context) {
defer h.jobsWG.Done()
if h.healthCheck != nil {
h.healthCheck.Start(ctx)
}
}
func (h *Handler) runHourlyStatsLoop(ctx context.Context) {
defer h.jobsWG.Done()
@@ -0,0 +1,795 @@
package handler
import (
"log"
"net/http"
"strconv"
"strings"
"time"
"go-backend/internal/http/response"
"go-backend/internal/monitoring"
"go-backend/internal/store/model"
)
const (
defaultMetricsRangeMs = int64(60 * 60 * 1000) // 1h
maxMetricsRangeMs = int64(24 * 60 * 60 * 1000) // 24h
)
func (h *Handler) resolveServiceMonitorLimits() monitoring.ServiceMonitorLimits {
defaults := monitoring.DefaultServiceMonitorLimits()
if h == nil || h.repo == nil {
return defaults
}
cfg, err := h.repo.GetConfigsByNames([]string{
monitoring.ConfigServiceMonitorCheckerScanIntervalSec,
monitoring.ConfigServiceMonitorWorkerLimit,
monitoring.ConfigServiceMonitorMinIntervalSec,
monitoring.ConfigServiceMonitorDefaultIntervalSec,
monitoring.ConfigServiceMonitorMinTimeoutSec,
monitoring.ConfigServiceMonitorDefaultTimeoutSec,
monitoring.ConfigServiceMonitorMaxTimeoutSec,
})
if err != nil {
return defaults
}
return monitoring.ServiceMonitorLimitsFromConfigMap(cfg)
}
func (h *Handler) monitorNodeMetricsHandler(w http.ResponseWriter, r *http.Request) {
if r.Method != http.MethodGet {
response.WriteJSON(w, response.ErrDefault("请求失败"))
return
}
if !h.ensureMonitoringAccess(w, r) {
return
}
path := r.URL.Path
prefix := "/api/v1/monitor/nodes/"
if !strings.HasPrefix(path, prefix) {
response.WriteJSON(w, response.ErrDefault("无效的路径"))
return
}
rest := strings.TrimPrefix(path, prefix)
if strings.HasSuffix(rest, "/metrics/latest") {
h.handleNodeMetricsLatest(w, r, strings.TrimSuffix(rest, "/metrics/latest"))
return
}
if strings.HasSuffix(rest, "/metrics") {
h.handleNodeMetrics(w, r, strings.TrimSuffix(rest, "/metrics"))
return
}
response.WriteJSON(w, response.ErrDefault("无效的路径"))
}
type monitorNodeListItem struct {
ID int64 `json:"id"`
Inx int `json:"inx"`
Name string `json:"name"`
Status int `json:"status"`
UpdatedTime int64 `json:"updatedTime"`
}
func (h *Handler) monitorNodeListHandler(w http.ResponseWriter, r *http.Request) {
if r.Method != http.MethodGet {
response.WriteJSON(w, response.ErrDefault("请求失败"))
return
}
if !h.ensureMonitoringAccess(w, r) {
return
}
nodes, err := h.repo.ListMonitorNodes()
if err != nil {
response.WriteJSON(w, response.Err(-2, err.Error()))
return
}
items := make([]monitorNodeListItem, 0, len(nodes))
for _, n := range nodes {
updated := int64(0)
if n.UpdatedTime.Valid {
updated = n.UpdatedTime.Int64
}
items = append(items, monitorNodeListItem{
ID: n.ID,
Inx: n.Inx,
Name: n.Name,
Status: n.Status,
UpdatedTime: updated,
})
}
response.WriteJSON(w, response.OK(items))
}
type monitorTunnelListItem struct {
ID int64 `json:"id"`
Inx int `json:"inx"`
Name string `json:"name"`
Status int `json:"status"`
UpdatedTime int64 `json:"updatedTime"`
}
func (h *Handler) monitorTunnelListHandler(w http.ResponseWriter, r *http.Request) {
if r.Method != http.MethodGet {
response.WriteJSON(w, response.ErrDefault("请求失败"))
return
}
if !h.ensureMonitoringAccess(w, r) {
return
}
tunnels, err := h.repo.ListMonitorTunnels()
if err != nil {
response.WriteJSON(w, response.Err(-2, err.Error()))
return
}
items := make([]monitorTunnelListItem, 0, len(tunnels))
for _, t := range tunnels {
items = append(items, monitorTunnelListItem{
ID: t.ID,
Inx: t.Inx,
Name: t.Name,
Status: t.Status,
UpdatedTime: t.UpdatedTime,
})
}
response.WriteJSON(w, response.OK(items))
}
func (h *Handler) handleNodeMetrics(w http.ResponseWriter, r *http.Request, nodeIDStr string) {
nodeID, err := strconv.ParseInt(nodeIDStr, 10, 64)
if err != nil || nodeID <= 0 {
response.WriteJSON(w, response.ErrDefault("无效的节点ID"))
return
}
now := time.Now().UnixMilli()
startMs := now - defaultMetricsRangeMs
endMs := now
if s := r.URL.Query().Get("start"); s != "" {
if v, err := strconv.ParseInt(s, 10, 64); err == nil {
startMs = v
}
}
if e := r.URL.Query().Get("end"); e != "" {
if v, err := strconv.ParseInt(e, 10, 64); err == nil {
endMs = v
}
}
if startMs <= 0 || endMs <= 0 {
response.WriteJSON(w, response.ErrDefault("无效的时间范围"))
return
}
if endMs < startMs {
response.WriteJSON(w, response.ErrDefault("无效的时间范围"))
return
}
if endMs-startMs > maxMetricsRangeMs {
response.WriteJSON(w, response.ErrDefault("时间范围过大"))
return
}
metrics, err := h.repo.GetNodeMetrics(nodeID, startMs, endMs)
if err != nil {
response.WriteJSON(w, response.Err(-2, err.Error()))
return
}
response.WriteJSON(w, response.OK(metrics))
}
func (h *Handler) handleNodeMetricsLatest(w http.ResponseWriter, _ *http.Request, nodeIDStr string) {
nodeID, err := strconv.ParseInt(nodeIDStr, 10, 64)
if err != nil || nodeID <= 0 {
response.WriteJSON(w, response.ErrDefault("无效的节点ID"))
return
}
metric, err := h.repo.GetLatestNodeMetric(nodeID)
if err != nil {
response.WriteJSON(w, response.Err(-2, err.Error()))
return
}
if metric == nil {
response.WriteJSON(w, response.OK(nil))
return
}
response.WriteJSON(w, response.OK(metric))
}
func (h *Handler) monitorTunnelMetrics(w http.ResponseWriter, r *http.Request) {
if r.Method != http.MethodGet {
response.WriteJSON(w, response.ErrDefault("请求失败"))
return
}
if !h.ensureMonitoringAccess(w, r) {
return
}
tunnelIDStr := extractPathParam(r.URL.Path, "/api/v1/monitor/tunnels/", "/metrics")
tunnelID, err := strconv.ParseInt(tunnelIDStr, 10, 64)
if err != nil || tunnelID <= 0 {
response.WriteJSON(w, response.ErrDefault("无效的隧道ID"))
return
}
now := time.Now().UnixMilli()
startMs := now - defaultMetricsRangeMs
endMs := now
if s := r.URL.Query().Get("start"); s != "" {
if v, err := strconv.ParseInt(s, 10, 64); err == nil {
startMs = v
}
}
if e := r.URL.Query().Get("end"); e != "" {
if v, err := strconv.ParseInt(e, 10, 64); err == nil {
endMs = v
}
}
if startMs <= 0 || endMs <= 0 {
response.WriteJSON(w, response.ErrDefault("无效的时间范围"))
return
}
if endMs < startMs {
response.WriteJSON(w, response.ErrDefault("无效的时间范围"))
return
}
if endMs-startMs > maxMetricsRangeMs {
response.WriteJSON(w, response.ErrDefault("时间范围过大"))
return
}
metrics, err := h.repo.GetTunnelMetricsAggregated(tunnelID, startMs, endMs)
if err != nil {
response.WriteJSON(w, response.Err(-2, err.Error()))
return
}
response.WriteJSON(w, response.OK(metrics))
}
func (h *Handler) monitorServiceListHandler(w http.ResponseWriter, r *http.Request) {
if r.Method != http.MethodGet {
response.WriteJSON(w, response.ErrDefault("请求失败"))
return
}
if !h.ensureMonitoringAccess(w, r) {
return
}
monitors, err := h.repo.ListServiceMonitors()
if err != nil {
response.WriteJSON(w, response.Err(-2, err.Error()))
return
}
response.WriteJSON(w, response.OK(monitors))
}
type createServiceMonitorRequest struct {
Name string `json:"name"`
Type string `json:"type"`
Target string `json:"target"`
IntervalSec int `json:"intervalSec"`
TimeoutSec int `json:"timeoutSec"`
NodeID int64 `json:"nodeId"`
Enabled *int `json:"enabled"`
}
func (h *Handler) monitorServiceCreate(w http.ResponseWriter, r *http.Request) {
if r.Method != http.MethodPost {
response.WriteJSON(w, response.ErrDefault("请求失败"))
return
}
if !h.ensureMonitoringAccess(w, r) {
return
}
var req createServiceMonitorRequest
if err := decodeJSON(r.Body, &req); err != nil {
response.WriteJSON(w, response.ErrDefault("请求参数错误"))
return
}
name := strings.TrimSpace(req.Name)
if name == "" {
response.WriteJSON(w, response.ErrDefault("名称不能为空"))
return
}
monitorType := strings.ToLower(strings.TrimSpace(req.Type))
if monitorType != "tcp" && monitorType != "icmp" {
response.WriteJSON(w, response.ErrDefault("类型必须是 tcp 或 icmp"))
return
}
target := strings.TrimSpace(req.Target)
if target == "" {
response.WriteJSON(w, response.ErrDefault("目标地址不能为空"))
return
}
limits := h.resolveServiceMonitorLimits()
intervalSec := req.IntervalSec
if intervalSec <= 0 {
intervalSec = limits.DefaultIntervalSec
}
if intervalSec < limits.MinIntervalSec {
intervalSec = limits.MinIntervalSec
}
timeoutSec := req.TimeoutSec
if timeoutSec <= 0 {
timeoutSec = limits.DefaultTimeoutSec
}
if timeoutSec < limits.MinTimeoutSec {
timeoutSec = limits.MinTimeoutSec
}
if timeoutSec > limits.MaxTimeoutSec {
timeoutSec = limits.MaxTimeoutSec
}
enabled := 1
if req.Enabled != nil {
if *req.Enabled == 0 || *req.Enabled == 1 {
enabled = *req.Enabled
}
}
now := time.Now().UnixMilli()
if req.NodeID > 0 {
n, err := h.repo.GetNodeByID(req.NodeID)
if err != nil {
response.WriteJSON(w, response.Err(-2, err.Error()))
return
}
if n == nil {
response.WriteJSON(w, response.ErrDefault("节点不存在"))
return
}
}
m := &model.ServiceMonitor{
Name: name,
Type: monitorType,
Target: target,
IntervalSec: intervalSec,
TimeoutSec: timeoutSec,
NodeID: req.NodeID,
Enabled: enabled,
CreatedTime: now,
UpdatedTime: now,
}
if m.Type == "icmp" && m.NodeID <= 0 {
response.WriteJSON(w, response.ErrDefault("ICMP 监控必须选择执行节点"))
return
}
// enabled is already normalized above.
if err := h.repo.CreateServiceMonitor(m); err != nil {
response.WriteJSON(w, response.Err(-2, err.Error()))
return
}
response.WriteJSON(w, response.OK(m))
}
type updateServiceMonitorRequest struct {
ID int64 `json:"id"`
Name string `json:"name"`
Type string `json:"type"`
Target string `json:"target"`
IntervalSec int `json:"intervalSec"`
TimeoutSec int `json:"timeoutSec"`
NodeID *int64 `json:"nodeId"`
Enabled *int `json:"enabled"`
}
func (h *Handler) monitorServiceUpdate(w http.ResponseWriter, r *http.Request) {
if r.Method != http.MethodPost {
response.WriteJSON(w, response.ErrDefault("请求失败"))
return
}
if !h.ensureMonitoringAccess(w, r) {
return
}
var req updateServiceMonitorRequest
if err := decodeJSON(r.Body, &req); err != nil {
response.WriteJSON(w, response.ErrDefault("请求参数错误"))
return
}
if req.ID <= 0 {
response.WriteJSON(w, response.ErrDefault("无效的监控ID"))
return
}
existing, err := h.repo.GetServiceMonitor(req.ID)
if err != nil {
response.WriteJSON(w, response.Err(-2, err.Error()))
return
}
if existing == nil {
response.WriteJSON(w, response.ErrDefault("监控不存在"))
return
}
name := strings.TrimSpace(req.Name)
if name != "" {
existing.Name = name
}
monitorType := strings.ToLower(strings.TrimSpace(req.Type))
if monitorType == "tcp" || monitorType == "icmp" {
existing.Type = monitorType
}
target := strings.TrimSpace(req.Target)
if target != "" {
existing.Target = target
}
limits := h.resolveServiceMonitorLimits()
if req.IntervalSec > 0 {
intervalSec := req.IntervalSec
if intervalSec < limits.MinIntervalSec {
intervalSec = limits.MinIntervalSec
}
existing.IntervalSec = intervalSec
}
if req.TimeoutSec > 0 {
timeoutSec := req.TimeoutSec
if timeoutSec < limits.MinTimeoutSec {
timeoutSec = limits.MinTimeoutSec
}
if timeoutSec > limits.MaxTimeoutSec {
timeoutSec = limits.MaxTimeoutSec
}
existing.TimeoutSec = timeoutSec
}
if req.NodeID != nil {
existing.NodeID = *req.NodeID
}
if req.Enabled != nil {
if *req.Enabled == 0 || *req.Enabled == 1 {
existing.Enabled = *req.Enabled
}
}
existing.UpdatedTime = time.Now().UnixMilli()
if existing.Type == "icmp" && existing.NodeID <= 0 {
response.WriteJSON(w, response.ErrDefault("ICMP 监控必须选择执行节点"))
return
}
if existing.NodeID > 0 {
n, err := h.repo.GetNodeByID(existing.NodeID)
if err != nil {
response.WriteJSON(w, response.Err(-2, err.Error()))
return
}
if n == nil {
response.WriteJSON(w, response.ErrDefault("节点不存在"))
return
}
}
if err := h.repo.UpdateServiceMonitor(existing); err != nil {
response.WriteJSON(w, response.Err(-2, err.Error()))
return
}
response.WriteJSON(w, response.OK(existing))
}
type deleteServiceMonitorRequest struct {
ID int64 `json:"id"`
}
func (h *Handler) monitorServiceDelete(w http.ResponseWriter, r *http.Request) {
if r.Method != http.MethodPost {
response.WriteJSON(w, response.ErrDefault("请求失败"))
return
}
if !h.ensureMonitoringAccess(w, r) {
return
}
var req deleteServiceMonitorRequest
if err := decodeJSON(r.Body, &req); err != nil {
response.WriteJSON(w, response.ErrDefault("请求参数错误"))
return
}
if req.ID <= 0 {
response.WriteJSON(w, response.ErrDefault("无效的监控ID"))
return
}
if err := h.repo.DeleteServiceMonitor(req.ID); err != nil {
response.WriteJSON(w, response.Err(-2, err.Error()))
return
}
response.WriteJSON(w, response.OKEmpty())
}
func (h *Handler) monitorServiceRun(w http.ResponseWriter, r *http.Request) {
if r.Method != http.MethodPost {
response.WriteJSON(w, response.ErrDefault("请求失败"))
return
}
if !h.ensureMonitoringAccess(w, r) {
return
}
if h.healthCheck == nil {
response.WriteJSON(w, response.ErrDefault("监控服务不可用"))
return
}
var req deleteServiceMonitorRequest
if err := decodeJSON(r.Body, &req); err != nil {
response.WriteJSON(w, response.ErrDefault("请求参数错误"))
return
}
if req.ID <= 0 {
response.WriteJSON(w, response.ErrDefault("无效的监控ID"))
return
}
m, err := h.repo.GetServiceMonitor(req.ID)
if err != nil {
response.WriteJSON(w, response.Err(-2, err.Error()))
return
}
if m == nil {
response.WriteJSON(w, response.ErrDefault("监控不存在"))
return
}
res, err := h.healthCheck.RunOnce(m)
if err != nil {
response.WriteJSON(w, response.Err(-2, err.Error()))
return
}
if err := h.repo.InsertServiceMonitorResult(res); err != nil {
log.Printf("monitoring write failed op=service_monitor_result.manual_insert monitor_id=%d err=%v", res.MonitorID, err)
}
response.WriteJSON(w, response.OK(res))
}
func (h *Handler) monitorServiceResultsHandler(w http.ResponseWriter, r *http.Request) {
if r.Method != http.MethodGet {
response.WriteJSON(w, response.ErrDefault("请求失败"))
return
}
if !h.ensureMonitoringAccess(w, r) {
return
}
monitorIDStr := extractPathParam(r.URL.Path, "/api/v1/monitor/services/", "/results")
monitorID, err := strconv.ParseInt(monitorIDStr, 10, 64)
if err != nil || monitorID <= 0 {
response.WriteJSON(w, response.ErrDefault("无效的监控ID"))
return
}
limit := 100
if l := r.URL.Query().Get("limit"); l != "" {
if v, err := strconv.Atoi(l); err == nil && v > 0 && v <= 1000 {
limit = v
}
}
results, err := h.repo.GetServiceMonitorResults(monitorID, limit)
if err != nil {
response.WriteJSON(w, response.Err(-2, err.Error()))
return
}
response.WriteJSON(w, response.OK(results))
}
func (h *Handler) monitorServiceLatestResultsHandler(w http.ResponseWriter, r *http.Request) {
if r.Method != http.MethodGet {
response.WriteJSON(w, response.ErrDefault("请求失败"))
return
}
if !h.ensureMonitoringAccess(w, r) {
return
}
results, err := h.repo.GetLatestServiceMonitorResults()
if err != nil {
response.WriteJSON(w, response.Err(-2, err.Error()))
return
}
response.WriteJSON(w, response.OK(results))
}
func (h *Handler) monitorServiceLimitsHandler(w http.ResponseWriter, r *http.Request) {
if r.Method != http.MethodGet {
response.WriteJSON(w, response.ErrDefault("请求失败"))
return
}
if !h.ensureMonitoringAccess(w, r) {
return
}
response.WriteJSON(w, response.OK(h.resolveServiceMonitorLimits()))
}
func extractPathParam(path, prefix, suffix string) string {
if !strings.HasPrefix(path, prefix) {
return ""
}
rest := strings.TrimPrefix(path, prefix)
if suffix != "" {
rest = strings.TrimSuffix(rest, suffix)
}
return rest
}
type monitorAccessData struct {
Allowed bool `json:"allowed"`
Reason string `json:"reason,omitempty"`
}
// monitorAccessHandler is a lightweight capability check for frontend navigation.
// It does NOT replace authorization on the actual monitoring endpoints.
func (h *Handler) monitorAccessHandler(w http.ResponseWriter, r *http.Request) {
if r.Method != http.MethodGet {
response.WriteJSON(w, response.ErrDefault("请求失败"))
return
}
userID, roleID, err := userRoleFromRequest(r)
if err != nil {
response.WriteJSON(w, response.Err(401, "未登录或token已过期"))
return
}
if roleID == 0 {
response.WriteJSON(w, response.OK(monitorAccessData{Allowed: true}))
return
}
allowed, err := h.repo.HasMonitorPermission(userID)
if err != nil {
response.WriteJSON(w, response.Err(-2, err.Error()))
return
}
data := monitorAccessData{Allowed: allowed}
if !allowed {
data.Reason = "need_admin_grant"
}
response.WriteJSON(w, response.OK(data))
}
func (h *Handler) ensureAdminAccess(w http.ResponseWriter, r *http.Request) bool {
_, roleID, err := userRoleFromRequest(r)
if err != nil {
response.WriteJSON(w, response.Err(401, "未登录或token已过期"))
return false
}
if roleID != 0 {
response.WriteJSON(w, response.Err(403, "权限不足,仅管理员可操作"))
return false
}
return true
}
func (h *Handler) ensureMonitoringAccess(w http.ResponseWriter, r *http.Request) bool {
userID, roleID, err := userRoleFromRequest(r)
if err != nil {
response.WriteJSON(w, response.Err(401, "未登录或token已过期"))
return false
}
if roleID == 0 {
return true
}
allowed, err := h.repo.HasMonitorPermission(userID)
if err != nil {
response.WriteJSON(w, response.Err(-2, err.Error()))
return false
}
if !allowed {
response.WriteJSON(w, response.Err(403, "权限不足:当前账户非管理员,且未被授予监控权限。请联系管理员在用户管理中授权监控权限。"))
return false
}
return true
}
func (h *Handler) monitorPermissionList(w http.ResponseWriter, r *http.Request) {
if r.Method != http.MethodGet {
response.WriteJSON(w, response.ErrDefault("请求失败"))
return
}
if !h.ensureAdminAccess(w, r) {
return
}
items, err := h.repo.ListMonitorPermissions()
if err != nil {
response.WriteJSON(w, response.Err(-2, err.Error()))
return
}
response.WriteJSON(w, response.OK(items))
}
type monitorPermissionMutationRequest struct {
UserID int64 `json:"userId"`
}
func (h *Handler) monitorPermissionAssign(w http.ResponseWriter, r *http.Request) {
if r.Method != http.MethodPost {
response.WriteJSON(w, response.ErrDefault("请求失败"))
return
}
if !h.ensureAdminAccess(w, r) {
return
}
var req monitorPermissionMutationRequest
if err := decodeJSON(r.Body, &req); err != nil {
response.WriteJSON(w, response.ErrDefault("请求参数错误"))
return
}
if req.UserID <= 0 {
response.WriteJSON(w, response.ErrDefault("无效的用户ID"))
return
}
u, err := h.repo.GetUserByID(req.UserID)
if err != nil {
response.WriteJSON(w, response.Err(-2, err.Error()))
return
}
if u == nil {
response.WriteJSON(w, response.ErrDefault("用户不存在"))
return
}
if err := h.repo.InsertMonitorPermission(req.UserID, time.Now().UnixMilli()); err != nil {
response.WriteJSON(w, response.Err(-2, err.Error()))
return
}
response.WriteJSON(w, response.OKEmpty())
}
func (h *Handler) monitorPermissionRemove(w http.ResponseWriter, r *http.Request) {
if r.Method != http.MethodPost {
response.WriteJSON(w, response.ErrDefault("请求失败"))
return
}
if !h.ensureAdminAccess(w, r) {
return
}
var req monitorPermissionMutationRequest
if err := decodeJSON(r.Body, &req); err != nil {
response.WriteJSON(w, response.ErrDefault("请求参数错误"))
return
}
if req.UserID <= 0 {
response.WriteJSON(w, response.ErrDefault("无效的用户ID"))
return
}
if err := h.repo.DeleteMonitorPermission(req.UserID); err != nil {
response.WriteJSON(w, response.Err(-2, err.Error()))
return
}
response.WriteJSON(w, response.OKEmpty())
}
+17 -19
View File
@@ -832,7 +832,7 @@ func (h *Handler) tunnelUpdate(w http.ResponseWriter, r *http.Request) {
newEntryNodeIDs = append(newEntryNodeIDs, in.NodeID)
}
}
if err := h.validateTunnelEntryPortConflictsForNewEntries(tx, id, oldEntryNodeIDs, newEntryNodeIDs); err != nil {
if err := h.validateTunnelEntryPortConflictsForNewEntriesTx(tx, id, oldEntryNodeIDs, newEntryNodeIDs); err != nil {
response.WriteJSON(w, response.ErrDefault(err.Error()))
return
}
@@ -1020,7 +1020,21 @@ func (h *Handler) cleanupTunnelForwardRuntimesOnRemovedEntryNodes(tunnelID int64
}
}
func (h *Handler) validateTunnelEntryPortConflictsForNewEntries(tx *gorm.DB, tunnelID int64, oldEntryNodeIDs, newEntryNodeIDs []int64) error {
func (h *Handler) validateForwardPortAvailabilityTx(tx *gorm.DB, node *nodeRecord, port int, currentForwardID int64) error {
if h == nil || h.repo == nil || tx == nil || node == nil || port <= 0 {
return nil
}
occupied, err := h.repo.HasOtherForwardOnNodePortTx(tx, node.ID, port, currentForwardID)
if err != nil {
return err
}
if occupied {
return fmt.Errorf("节点 %s 端口 %d 已被其他转发占用", node.Name, port)
}
return nil
}
func (h *Handler) validateTunnelEntryPortConflictsForNewEntriesTx(tx *gorm.DB, tunnelID int64, oldEntryNodeIDs, newEntryNodeIDs []int64) error {
if h == nil || h.repo == nil || tx == nil || tunnelID <= 0 {
return nil
}
@@ -1054,9 +1068,7 @@ func (h *Handler) validateTunnelEntryPortConflictsForNewEntries(tx *gorm.DB, tun
if nodeErr != nil {
continue
}
if err := validateLocalNodePort(node, port); err != nil {
return fmt.Errorf("转发 %s 入口端口冲突: %w", f.Name, err)
}
if err := h.validateForwardPortAvailabilityTx(tx, node, port, f.ID); err != nil {
return fmt.Errorf("转发 %s 入口端口冲突: %w", f.Name, err)
}
@@ -4153,20 +4165,6 @@ func (h *Handler) validateForwardPortAvailability(node *nodeRecord, port int, cu
return nil
}
func (h *Handler) validateForwardPortAvailabilityTx(tx *gorm.DB, node *nodeRecord, port int, currentForwardID int64) error {
if h == nil || h.repo == nil || tx == nil || node == nil || port <= 0 {
return nil
}
occupied, err := h.repo.HasOtherForwardOnNodePortTx(tx, node.ID, port, currentForwardID)
if err != nil {
return err
}
if occupied {
return fmt.Errorf("节点 %s 端口 %d 已被其他转发占用", node.Name, port)
}
return nil
}
func parsePortRangeMinMax(input string) (int, int) {
input = strings.TrimSpace(input)
if input == "" {
@@ -83,7 +83,7 @@ func TestValidateTunnelEntryPortConflictsForNewEntriesDoesNotBlockOnSQLiteTx(t *
doneCh := make(chan struct{})
go func() {
defer close(doneCh)
errCh <- h.validateTunnelEntryPortConflictsForNewEntries(tx, tunnelID, []int64{oldEntryID}, []int64{oldEntryID, newEntryID})
errCh <- h.validateTunnelEntryPortConflictsForNewEntriesTx(tx, tunnelID, []int64{oldEntryID}, []int64{oldEntryID, newEntryID})
}()
select {
@@ -0,0 +1,111 @@
package handler
import (
"log"
"strings"
"time"
"go-backend/internal/store/model"
)
type tunnelTrafficDelta struct {
bytesIn int64
bytesOut int64
}
func unixMilliBucketMinute(nowMs int64) int64 {
if nowMs <= 0 {
return 0
}
const minuteMs = int64(time.Minute / time.Millisecond)
return nowMs - (nowMs % minuteMs)
}
func (h *Handler) recordTunnelMetricsFromFlowItems(nodeID int64, items []flowItem, nowMs int64) {
if h == nil || h.repo == nil {
return
}
if nodeID <= 0 || len(items) == 0 {
return
}
bucketTs := unixMilliBucketMinute(nowMs)
if bucketTs <= 0 {
return
}
forwardDeltas := make(map[int64]tunnelTrafficDelta)
for _, item := range items {
name := strings.TrimSpace(item.N)
if name == "" || name == "web_api" {
continue
}
forwardID, _, _, ok := parseFlowServiceIDs(name)
if !ok {
continue
}
if item.D == 0 && item.U == 0 {
continue
}
d := forwardDeltas[forwardID]
d.bytesIn += item.D
d.bytesOut += item.U
forwardDeltas[forwardID] = d
}
if len(forwardDeltas) == 0 {
return
}
forwardIDs := make([]int64, 0, len(forwardDeltas))
for id := range forwardDeltas {
forwardIDs = append(forwardIDs, id)
}
forwardTunnelMap, err := h.repo.MapForwardIDsToTunnelIDs(forwardIDs)
if err != nil {
log.Printf("monitoring write skipped op=tunnel_metric.map_forward_to_tunnel node_id=%d err=%v", nodeID, err)
return
}
if len(forwardTunnelMap) == 0 {
return
}
tunnelAgg := make(map[int64]tunnelTrafficDelta)
for forwardID, delta := range forwardDeltas {
tunnelID := forwardTunnelMap[forwardID]
if tunnelID <= 0 {
continue
}
a := tunnelAgg[tunnelID]
a.bytesIn += delta.bytesIn
a.bytesOut += delta.bytesOut
tunnelAgg[tunnelID] = a
}
if len(tunnelAgg) == 0 {
return
}
metrics := make([]*model.TunnelMetric, 0, len(tunnelAgg))
for tunnelID, delta := range tunnelAgg {
if delta.bytesIn == 0 && delta.bytesOut == 0 {
continue
}
metrics = append(metrics, &model.TunnelMetric{
TunnelID: tunnelID,
NodeID: nodeID,
Timestamp: bucketTs,
BytesIn: delta.bytesIn,
BytesOut: delta.bytesOut,
Connections: 0,
Errors: 0,
AvgLatencyMs: 0,
})
}
if len(metrics) == 0 {
return
}
if err := h.repo.UpsertTunnelMetricBuckets(metrics); err != nil {
log.Printf("monitoring write failed op=tunnel_metric.upsert_buckets node_id=%d bucket_ts=%d count=%d err=%v", nodeID, bucketTs, len(metrics), err)
}
}
@@ -101,6 +101,10 @@ func shouldSkip(path string) bool {
}
func requiresAdmin(path string) bool {
if strings.HasPrefix(path, "/api/v1/monitor/permission/") {
return true
}
if strings.HasPrefix(path, "/api/v1/group/") {
return true
}
+135
View File
@@ -0,0 +1,135 @@
package metrics
import (
"context"
"log"
"sync"
"time"
"go-backend/internal/store/model"
"go-backend/internal/store/repo"
)
type SystemInfo struct {
Uptime uint64 `json:"uptime"`
BytesReceived uint64 `json:"bytes_received"`
BytesTransmitted uint64 `json:"bytes_transmitted"`
CPUUsage float64 `json:"cpu_usage"`
MemoryUsage float64 `json:"memory_usage"`
DiskUsage float64 `json:"disk_usage"`
Load1 float64 `json:"load1"`
Load5 float64 `json:"load5"`
Load15 float64 `json:"load15"`
TCPConns int64 `json:"tcp_conns"`
UDPConns int64 `json:"udp_conns"`
NetInSpeed int64 `json:"net_in_speed"`
NetOutSpeed int64 `json:"net_out_speed"`
}
type IngestionService struct {
repo *repo.Repository
nodeBuffer []*model.NodeMetric
nodeBufferMu sync.Mutex
flushInterval time.Duration
retentionDays int
}
func NewIngestionService(repo *repo.Repository) *IngestionService {
return &IngestionService{
repo: repo,
nodeBuffer: make([]*model.NodeMetric, 0, 500),
flushInterval: 30 * time.Second,
retentionDays: 7,
}
}
func (s *IngestionService) Start(ctx context.Context) {
flushTicker := time.NewTicker(s.flushInterval)
defer flushTicker.Stop()
pruneTicker := time.NewTicker(1 * time.Hour)
defer pruneTicker.Stop()
for {
select {
case <-ctx.Done():
s.flushNodeMetrics()
return
case <-flushTicker.C:
s.flushNodeMetrics()
case <-pruneTicker.C:
s.pruneMetrics()
}
}
}
func (s *IngestionService) RecordNodeMetric(nodeID int64, info SystemInfo) {
m := &model.NodeMetric{
NodeID: nodeID,
Timestamp: time.Now().UnixMilli(),
CPUUsage: info.CPUUsage,
MemUsage: info.MemoryUsage,
DiskUsage: info.DiskUsage,
NetInBytes: int64(info.BytesReceived),
NetOutBytes: int64(info.BytesTransmitted),
NetInSpeed: info.NetInSpeed,
NetOutSpeed: info.NetOutSpeed,
Load1: info.Load1,
Load5: info.Load5,
Load15: info.Load15,
TCPConns: info.TCPConns,
UDPConns: info.UDPConns,
Uptime: int64(info.Uptime),
}
s.nodeBufferMu.Lock()
s.nodeBuffer = append(s.nodeBuffer, m)
shouldFlush := len(s.nodeBuffer) >= 200
s.nodeBufferMu.Unlock()
if shouldFlush {
go s.flushNodeMetrics()
}
}
func (s *IngestionService) flushNodeMetrics() {
s.nodeBufferMu.Lock()
if len(s.nodeBuffer) == 0 {
s.nodeBufferMu.Unlock()
return
}
buffer := s.nodeBuffer
s.nodeBuffer = make([]*model.NodeMetric, 0, 500)
s.nodeBufferMu.Unlock()
if s.repo == nil {
return
}
if err := s.repo.InsertNodeMetricBatch(buffer); err != nil {
log.Printf("monitoring write failed op=node_metric.flush count=%d err=%v", len(buffer), err)
}
}
func (s *IngestionService) pruneMetrics() {
cutoff := time.Now().Add(-time.Duration(s.retentionDays) * 24 * time.Hour).UnixMilli()
if s.repo == nil {
return
}
if err := s.repo.PruneNodeMetrics(cutoff); err != nil {
log.Printf("monitoring prune failed op=node_metric cutoff=%d err=%v", cutoff, err)
}
if err := s.repo.PruneTunnelMetrics(cutoff); err != nil {
log.Printf("monitoring prune failed op=tunnel_metric cutoff=%d err=%v", cutoff, err)
}
if err := s.repo.PruneServiceMonitorResults(cutoff); err != nil {
log.Printf("monitoring prune failed op=service_monitor_result cutoff=%d err=%v", cutoff, err)
}
}
func (s *IngestionService) GetLatestMetric(nodeID int64) (*model.NodeMetric, error) {
return s.repo.GetLatestNodeMetric(nodeID)
}
func (s *IngestionService) GetMetrics(nodeID int64, startMs, endMs int64) ([]model.NodeMetric, error) {
return s.repo.GetNodeMetrics(nodeID, startMs, endMs)
}
@@ -0,0 +1,294 @@
package metrics
import (
"context"
"testing"
"time"
"go-backend/internal/store/repo"
)
func TestRecordNodeMetric(t *testing.T) {
r, err := repo.Open(":memory:")
if err != nil {
t.Fatalf("open repo: %v", err)
}
defer r.Close()
svc := NewIngestionService(r)
info := SystemInfo{
Uptime: 86400,
BytesReceived: 1024000,
BytesTransmitted: 2048000,
CPUUsage: 45.5,
MemoryUsage: 60.2,
DiskUsage: 30.1,
Load1: 1.5,
Load5: 1.2,
Load15: 0.9,
TCPConns: 100,
UDPConns: 50,
NetInSpeed: 51200,
NetOutSpeed: 102400,
}
svc.RecordNodeMetric(1, info)
svc.flushNodeMetrics()
metrics, err := r.GetNodeMetrics(1, 0, time.Now().UnixMilli()+1000)
if err != nil {
t.Fatalf("get metrics: %v", err)
}
if len(metrics) != 1 {
t.Fatalf("expected 1 metric, got %d", len(metrics))
}
m := metrics[0]
if m.CPUUsage != 45.5 {
t.Fatalf("expected CPUUsage 45.5, got %f", m.CPUUsage)
}
if m.MemUsage != 60.2 {
t.Fatalf("expected MemUsage 60.2, got %f", m.MemUsage)
}
if m.DiskUsage != 30.1 {
t.Fatalf("expected DiskUsage 30.1, got %f", m.DiskUsage)
}
if m.Load1 != 1.5 {
t.Fatalf("expected Load1 1.5, got %f", m.Load1)
}
if m.TCPConns != 100 {
t.Fatalf("expected TCPConns 100, got %d", m.TCPConns)
}
if m.UDPConns != 50 {
t.Fatalf("expected UDPConns 50, got %d", m.UDPConns)
}
}
func TestRecordNodeMetricAutoFlush(t *testing.T) {
r, err := repo.Open(":memory:")
if err != nil {
t.Fatalf("open repo: %v", err)
}
defer r.Close()
svc := NewIngestionService(r)
info := SystemInfo{
CPUUsage: 50.0,
MemoryUsage: 60.0,
DiskUsage: 30.0,
}
for i := 0; i < 250; i++ {
svc.RecordNodeMetric(1, info)
}
time.Sleep(100 * time.Millisecond)
metrics, err := r.GetNodeMetrics(1, 0, time.Now().UnixMilli()+1000)
if err != nil {
t.Fatalf("get metrics: %v", err)
}
if len(metrics) < 200 {
t.Fatalf("expected at least 200 metrics after auto-flush, got %d", len(metrics))
}
}
func TestIngestionServiceStart(t *testing.T) {
r, err := repo.Open(":memory:")
if err != nil {
t.Fatalf("open repo: %v", err)
}
defer r.Close()
svc := NewIngestionService(r)
svc.flushInterval = 100 * time.Millisecond
ctx, cancel := context.WithTimeout(context.Background(), 500*time.Millisecond)
defer cancel()
info := SystemInfo{
CPUUsage: 45.0,
MemoryUsage: 55.0,
DiskUsage: 35.0,
}
go svc.Start(ctx)
for i := 0; i < 10; i++ {
svc.RecordNodeMetric(1, info)
time.Sleep(50 * time.Millisecond)
}
<-ctx.Done()
metrics, err := r.GetNodeMetrics(1, 0, time.Now().UnixMilli()+1000)
if err != nil {
t.Fatalf("get metrics: %v", err)
}
if len(metrics) == 0 {
t.Fatalf("expected metrics after service run")
}
}
func TestGetLatestMetric(t *testing.T) {
r, err := repo.Open(":memory:")
if err != nil {
t.Fatalf("open repo: %v", err)
}
defer r.Close()
svc := NewIngestionService(r)
now := time.Now().UnixMilli()
info1 := SystemInfo{CPUUsage: 40.0, MemoryUsage: 50.0, DiskUsage: 30.0}
svc.RecordNodeMetric(1, info1)
time.Sleep(5 * time.Millisecond)
info2 := SystemInfo{CPUUsage: 60.0, MemoryUsage: 70.0, DiskUsage: 40.0}
svc.RecordNodeMetric(1, info2)
svc.flushNodeMetrics()
latest, err := svc.GetLatestMetric(1)
if err != nil {
t.Fatalf("get latest: %v", err)
}
if latest == nil {
t.Fatalf("expected latest metric")
}
if latest.CPUUsage != 60.0 {
t.Fatalf("expected latest CPUUsage 60.0, got %f", latest.CPUUsage)
}
_ = now
latestNone, err := svc.GetLatestMetric(999)
if err != nil {
t.Fatalf("get latest for non-existent: %v", err)
}
if latestNone != nil {
t.Fatalf("expected nil for non-existent node")
}
}
func TestGetMetricsWithTimeRange(t *testing.T) {
r, err := repo.Open(":memory:")
if err != nil {
t.Fatalf("open repo: %v", err)
}
defer r.Close()
svc := NewIngestionService(r)
now := time.Now().UnixMilli()
for i := 0; i < 5; i++ {
info := SystemInfo{
CPUUsage: float64(40 + i*5),
MemoryUsage: 50.0,
DiskUsage: 30.0,
}
svc.RecordNodeMetric(1, info)
time.Sleep(10 * time.Millisecond)
}
svc.flushNodeMetrics()
metrics, err := svc.GetMetrics(1, 0, now+1000)
if err != nil {
t.Fatalf("get metrics: %v", err)
}
if len(metrics) != 5 {
t.Fatalf("expected 5 metrics, got %d", len(metrics))
}
}
func TestPruneMetrics(t *testing.T) {
r, err := repo.Open(":memory:")
if err != nil {
t.Fatalf("open repo: %v", err)
}
defer r.Close()
svc := NewIngestionService(r)
svc.retentionDays = 1
info := SystemInfo{CPUUsage: 50.0, MemoryUsage: 60.0, DiskUsage: 30.0}
svc.RecordNodeMetric(1, info)
svc.flushNodeMetrics()
svc.pruneMetrics()
metrics, err := r.GetNodeMetrics(1, 0, time.Now().UnixMilli()+1000)
if err != nil {
t.Fatalf("get metrics: %v", err)
}
if len(metrics) != 1 {
t.Fatalf("expected 1 metric (not pruned), got %d", len(metrics))
}
}
func TestMultipleNodes(t *testing.T) {
r, err := repo.Open(":memory:")
if err != nil {
t.Fatalf("open repo: %v", err)
}
defer r.Close()
svc := NewIngestionService(r)
info := SystemInfo{
CPUUsage: 50.0,
MemoryUsage: 60.0,
DiskUsage: 30.0,
}
svc.RecordNodeMetric(1, info)
svc.RecordNodeMetric(2, info)
svc.RecordNodeMetric(3, info)
svc.flushNodeMetrics()
for nodeID := int64(1); nodeID <= 3; nodeID++ {
metrics, err := r.GetNodeMetrics(nodeID, 0, time.Now().UnixMilli()+1000)
if err != nil {
t.Fatalf("get metrics for node %d: %v", nodeID, err)
}
if len(metrics) != 1 {
t.Fatalf("expected 1 metric for node %d, got %d", nodeID, len(metrics))
}
}
}
func TestZeroValues(t *testing.T) {
r, err := repo.Open(":memory:")
if err != nil {
t.Fatalf("open repo: %v", err)
}
defer r.Close()
svc := NewIngestionService(r)
info := SystemInfo{}
svc.RecordNodeMetric(1, info)
svc.flushNodeMetrics()
metrics, err := r.GetNodeMetrics(1, 0, time.Now().UnixMilli()+1000)
if err != nil {
t.Fatalf("get metrics: %v", err)
}
if len(metrics) != 1 {
t.Fatalf("expected 1 metric, got %d", len(metrics))
}
m := metrics[0]
if m.CPUUsage != 0 || m.MemUsage != 0 || m.DiskUsage != 0 {
t.Fatalf("expected zero values, got CPU=%f Mem=%f Disk=%f", m.CPUUsage, m.MemUsage, m.DiskUsage)
}
}
+114
View File
@@ -0,0 +1,114 @@
package monitoring
import (
"strconv"
"strings"
)
type ServiceMonitorLimits struct {
CheckerScanIntervalSec int `json:"checkerScanIntervalSec"`
WorkerLimit int `json:"workerLimit"`
MinIntervalSec int `json:"minIntervalSec"`
DefaultIntervalSec int `json:"defaultIntervalSec"`
MinTimeoutSec int `json:"minTimeoutSec"`
DefaultTimeoutSec int `json:"defaultTimeoutSec"`
MaxTimeoutSec int `json:"maxTimeoutSec"`
}
const (
ConfigServiceMonitorCheckerScanIntervalSec = "service_monitor_checker_scan_interval_sec"
ConfigServiceMonitorWorkerLimit = "service_monitor_worker_limit"
ConfigServiceMonitorMinIntervalSec = "service_monitor_min_interval_sec"
ConfigServiceMonitorDefaultIntervalSec = "service_monitor_default_interval_sec"
ConfigServiceMonitorMinTimeoutSec = "service_monitor_min_timeout_sec"
ConfigServiceMonitorDefaultTimeoutSec = "service_monitor_default_timeout_sec"
ConfigServiceMonitorMaxTimeoutSec = "service_monitor_max_timeout_sec"
)
func DefaultServiceMonitorLimits() ServiceMonitorLimits {
return ServiceMonitorLimits{
CheckerScanIntervalSec: 30,
WorkerLimit: 5,
MinIntervalSec: 30,
DefaultIntervalSec: 60,
MinTimeoutSec: 1,
DefaultTimeoutSec: 5,
MaxTimeoutSec: 60,
}
}
// ServiceMonitorLimitsFromConfigMap parses limits from vite_config values.
// Missing/invalid values fall back to defaults.
func ServiceMonitorLimitsFromConfigMap(cfg map[string]string) ServiceMonitorLimits {
limits := DefaultServiceMonitorLimits()
if cfg == nil {
return limits
}
limits.CheckerScanIntervalSec = parseConfigInt(cfg, ConfigServiceMonitorCheckerScanIntervalSec, limits.CheckerScanIntervalSec)
limits.WorkerLimit = parseConfigInt(cfg, ConfigServiceMonitorWorkerLimit, limits.WorkerLimit)
limits.MinIntervalSec = parseConfigInt(cfg, ConfigServiceMonitorMinIntervalSec, limits.MinIntervalSec)
limits.DefaultIntervalSec = parseConfigInt(cfg, ConfigServiceMonitorDefaultIntervalSec, limits.DefaultIntervalSec)
limits.MinTimeoutSec = parseConfigInt(cfg, ConfigServiceMonitorMinTimeoutSec, limits.MinTimeoutSec)
limits.DefaultTimeoutSec = parseConfigInt(cfg, ConfigServiceMonitorDefaultTimeoutSec, limits.DefaultTimeoutSec)
limits.MaxTimeoutSec = parseConfigInt(cfg, ConfigServiceMonitorMaxTimeoutSec, limits.MaxTimeoutSec)
return normalizeServiceMonitorLimits(limits)
}
func normalizeServiceMonitorLimits(limits ServiceMonitorLimits) ServiceMonitorLimits {
if limits.CheckerScanIntervalSec <= 0 {
limits.CheckerScanIntervalSec = 30
}
if limits.WorkerLimit <= 0 {
limits.WorkerLimit = 5
}
if limits.WorkerLimit > 50 {
limits.WorkerLimit = 50
}
if limits.MinIntervalSec <= 0 {
limits.MinIntervalSec = limits.CheckerScanIntervalSec
}
if limits.MinIntervalSec < limits.CheckerScanIntervalSec {
limits.MinIntervalSec = limits.CheckerScanIntervalSec
}
if limits.DefaultIntervalSec <= 0 {
limits.DefaultIntervalSec = 60
}
if limits.DefaultIntervalSec < limits.MinIntervalSec {
limits.DefaultIntervalSec = limits.MinIntervalSec
}
if limits.MinTimeoutSec <= 0 {
limits.MinTimeoutSec = 1
}
if limits.DefaultTimeoutSec <= 0 {
limits.DefaultTimeoutSec = 5
}
if limits.DefaultTimeoutSec < limits.MinTimeoutSec {
limits.DefaultTimeoutSec = limits.MinTimeoutSec
}
if limits.MaxTimeoutSec <= 0 {
limits.MaxTimeoutSec = 60
}
if limits.MaxTimeoutSec < limits.DefaultTimeoutSec {
limits.MaxTimeoutSec = limits.DefaultTimeoutSec
}
return limits
}
func parseConfigInt(cfg map[string]string, key string, fallback int) int {
v := strings.TrimSpace(cfg[key])
if v == "" {
return fallback
}
n, err := strconv.Atoi(v)
if err != nil {
return fallback
}
return n
}
+73
View File
@@ -235,6 +235,16 @@ type GroupPermissionGrant struct {
func (GroupPermissionGrant) TableName() string { return "group_permission_grant" }
// MonitorPermission grants a non-admin user access to monitoring endpoints.
// One row per user_id.
type MonitorPermission struct {
ID int64 `gorm:"primaryKey;autoIncrement" json:"id"`
UserID int64 `gorm:"column:user_id;not null;uniqueIndex:idx_monitor_permission_user" json:"userId"`
CreatedTime int64 `gorm:"column:created_time;not null" json:"createdTime"`
}
func (MonitorPermission) TableName() string { return "monitor_permission" }
type ViteConfig struct {
ID int64 `gorm:"primaryKey;autoIncrement" json:"id"`
Name string `gorm:"type:varchar(200);not null;uniqueIndex" json:"name"`
@@ -641,3 +651,66 @@ type UserForwardDetail struct {
Status int
CreatedAt int64
}
type NodeMetric struct {
ID int64 `gorm:"primaryKey;autoIncrement" json:"id"`
NodeID int64 `gorm:"column:node_id;not null;index:idx_node_metric_node_time,priority:1" json:"nodeId"`
Timestamp int64 `gorm:"not null;index:idx_node_metric_node_time,priority:2;index:idx_node_metric_time" json:"timestamp"`
CPUUsage float64 `gorm:"column:cpu_usage" json:"cpuUsage"`
MemUsage float64 `gorm:"column:mem_usage" json:"memoryUsage"`
DiskUsage float64 `gorm:"column:disk_usage" json:"diskUsage"`
NetInBytes int64 `gorm:"column:net_in_bytes" json:"netInBytes"`
NetOutBytes int64 `gorm:"column:net_out_bytes" json:"netOutBytes"`
NetInSpeed int64 `gorm:"column:net_in_speed" json:"netInSpeed"`
NetOutSpeed int64 `gorm:"column:net_out_speed" json:"netOutSpeed"`
Load1 float64 `gorm:"column:load1" json:"load1"`
Load5 float64 `gorm:"column:load5" json:"load5"`
Load15 float64 `gorm:"column:load15" json:"load15"`
TCPConns int64 `gorm:"column:tcp_conns" json:"tcpConns"`
UDPConns int64 `gorm:"column:udp_conns" json:"udpConns"`
Uptime int64 `gorm:"column:uptime" json:"uptime"`
}
func (NodeMetric) TableName() string { return "node_metric" }
type TunnelMetric struct {
ID int64 `gorm:"primaryKey;autoIncrement" json:"id"`
TunnelID int64 `gorm:"column:tunnel_id;not null;index:idx_tunnel_metric_tunnel_time,priority:1" json:"tunnelId"`
NodeID int64 `gorm:"column:node_id;not null;index:idx_tunnel_metric_tunnel_time,priority:2" json:"nodeId"`
Timestamp int64 `gorm:"not null;index:idx_tunnel_metric_tunnel_time,priority:3;index:idx_tunnel_metric_time" json:"timestamp"`
BytesIn int64 `gorm:"column:bytes_in" json:"bytesIn"`
BytesOut int64 `gorm:"column:bytes_out" json:"bytesOut"`
Connections int64 `gorm:"column:connections" json:"connections"`
Errors int64 `gorm:"column:errors" json:"errors"`
AvgLatencyMs float64 `gorm:"column:avg_latency_ms" json:"avgLatencyMs"`
}
func (TunnelMetric) TableName() string { return "tunnel_metric" }
type ServiceMonitor struct {
ID int64 `gorm:"primaryKey;autoIncrement" json:"id"`
Name string `gorm:"type:varchar(100);not null" json:"name"`
Type string `gorm:"type:varchar(20);not null" json:"type"`
Target string `gorm:"type:text;not null" json:"target"`
IntervalSec int `gorm:"column:interval_sec;not null;default:60" json:"intervalSec"`
TimeoutSec int `gorm:"column:timeout_sec;not null;default:5" json:"timeoutSec"`
NodeID int64 `gorm:"column:node_id;index" json:"nodeId"`
Enabled int `gorm:"not null;default:1" json:"enabled"`
CreatedTime int64 `gorm:"column:created_time;not null" json:"createdTime"`
UpdatedTime int64 `gorm:"column:updated_time;not null" json:"updatedTime"`
}
func (ServiceMonitor) TableName() string { return "service_monitor" }
type ServiceMonitorResult struct {
ID int64 `gorm:"primaryKey;autoIncrement" json:"id"`
MonitorID int64 `gorm:"column:monitor_id;not null;index:idx_monitor_result_monitor_time,priority:1" json:"monitorId"`
NodeID int64 `gorm:"column:node_id;not null;index" json:"nodeId"`
Timestamp int64 `gorm:"not null;index:idx_monitor_result_monitor_time,priority:2" json:"timestamp"`
Success int `gorm:"not null" json:"success"`
LatencyMs float64 `gorm:"column:latency_ms" json:"latencyMs"`
StatusCode int `gorm:"column:status_code" json:"statusCode"`
ErrorMessage string `gorm:"column:error_message;type:text" json:"errorMessage"`
}
func (ServiceMonitorResult) TableName() string { return "service_monitor_result" }
+487 -1
View File
@@ -47,6 +47,10 @@ type UserGroupBackup = model.UserGroupBackup
type PermissionBackup = model.PermissionBackup
type PermissionGrantBackup = model.PermissionGrantBackup
type ImportResult = model.ImportResult
type NodeMetric = model.NodeMetric
type TunnelMetric = model.TunnelMetric
type ServiceMonitor = model.ServiceMonitor
type ServiceMonitorResult = model.ServiceMonitorResult
// ─── Repository ──────────────────────────────────────────────────────
@@ -176,12 +180,17 @@ func autoMigrateAll(db *gorm.DB) error {
&model.UserGroupUser{},
&model.GroupPermission{},
&model.GroupPermissionGrant{},
&model.MonitorPermission{},
&model.ViteConfig{},
&model.PeerShare{},
&model.PeerShareRuntime{},
&model.FederationTunnelBinding{},
&model.Announcement{},
&model.SchemaVersion{},
&model.NodeMetric{},
&model.TunnelMetric{},
&model.ServiceMonitor{},
&model.ServiceMonitorResult{},
}
if db.Dialector.Name() != "sqlite" {
@@ -394,6 +403,24 @@ func (r *Repository) ListConfigs() (map[string]string, error) {
return result, nil
}
func (r *Repository) GetConfigsByNames(names []string) (map[string]string, error) {
if r == nil || r.db == nil {
return nil, errors.New("repository not initialized")
}
if len(names) == 0 {
return map[string]string{}, nil
}
var configs []model.ViteConfig
if err := r.db.Select("name", "value").Where("name IN ?", names).Find(&configs).Error; err != nil {
return nil, err
}
result := make(map[string]string, len(configs))
for _, c := range configs {
result[c.Name] = c.Value
}
return result, nil
}
func (r *Repository) UpsertConfig(name, value string, now int64) error {
if r == nil || r.db == nil {
return errors.New("repository not initialized")
@@ -2689,12 +2716,13 @@ func (r *Repository) GetUserTunnelByID(id int64) (*model.UserTunnel, error) {
// ─── Migration ───────────────────────────────────────────────────────
const currentSchemaVersion = 5
const currentSchemaVersion = 6
var ensurePostgresIDDefaultsFn = ensurePostgresIDDefaults
var migrateViteConfigValueColumnTypeFn = migrateViteConfigValueColumnType
var migrateSpeedLimitTunnelBindingFn = migrateSpeedLimitTunnelBinding
var migratePostgresTrafficInt64ColumnsFn = migratePostgresTrafficInt64Columns
var migrateTunnelMetricBucketUniqueIndexFn = migrateTunnelMetricBucketUniqueIndex
func getSchemaVersion(db *gorm.DB) int {
var v model.SchemaVersion
@@ -2764,6 +2792,12 @@ func migrateSchema(db *gorm.DB) error {
}
}
if ver < 6 {
if err := migrateTunnelMetricBucketUniqueIndexFn(db); err != nil {
return err
}
}
setSchemaVersion(db, currentSchemaVersion)
return nil
}
@@ -2867,6 +2901,130 @@ func migratePostgresTrafficInt64Columns(db *gorm.DB) error {
return nil
}
func migrateTunnelMetricBucketUniqueIndex(db *gorm.DB) error {
if db == nil {
return errors.New("nil db")
}
if !db.Migrator().HasTable(&model.TunnelMetric{}) {
return nil
}
return db.Transaction(func(tx *gorm.DB) error {
// Only do the heavier dedupe work when needed.
var dupGroups int64
q := `
SELECT COUNT(1) AS cnt
FROM (
SELECT 1
FROM tunnel_metric
GROUP BY tunnel_id, node_id, timestamp
HAVING COUNT(*) > 1
) t
`
if err := tx.Raw(q).Scan(&dupGroups).Error; err != nil {
return fmt.Errorf("inspect tunnel_metric duplicates: %w", err)
}
if dupGroups > 0 {
switch tx.Dialector.Name() {
case "postgres":
sql := `
WITH agg AS (
SELECT MIN(id) AS keep_id,
tunnel_id,
node_id,
timestamp,
SUM(bytes_in) AS bytes_in,
SUM(bytes_out) AS bytes_out,
SUM(connections) AS connections,
SUM(errors) AS errors,
AVG(avg_latency_ms) AS avg_latency_ms
FROM tunnel_metric
GROUP BY tunnel_id, node_id, timestamp
HAVING COUNT(*) > 1
), updated AS (
UPDATE tunnel_metric tm
SET bytes_in = agg.bytes_in,
bytes_out = agg.bytes_out,
connections = agg.connections,
errors = agg.errors,
avg_latency_ms = agg.avg_latency_ms
FROM agg
WHERE tm.id = agg.keep_id
RETURNING tm.id
)
DELETE FROM tunnel_metric tm
USING agg
WHERE tm.tunnel_id = agg.tunnel_id
AND tm.node_id = agg.node_id
AND tm.timestamp = agg.timestamp
AND tm.id <> agg.keep_id
`
if err := tx.Exec(sql).Error; err != nil {
return fmt.Errorf("dedupe tunnel_metric buckets: %w", err)
}
default:
// SQLite (and other) path.
if err := tx.Exec(`DROP TABLE IF EXISTS tunnel_metric_dedupe`).Error; err != nil {
return fmt.Errorf("prepare tunnel_metric dedupe table: %w", err)
}
if err := tx.Exec(`
CREATE TEMP TABLE tunnel_metric_dedupe AS
SELECT MIN(id) AS keep_id,
tunnel_id,
node_id,
timestamp,
SUM(bytes_in) AS bytes_in,
SUM(bytes_out) AS bytes_out,
SUM(connections) AS connections,
SUM(errors) AS errors,
AVG(avg_latency_ms) AS avg_latency_ms
FROM tunnel_metric
GROUP BY tunnel_id, node_id, timestamp
HAVING COUNT(*) > 1
`).Error; err != nil {
return fmt.Errorf("build tunnel_metric dedupe table: %w", err)
}
if err := tx.Exec(`
UPDATE tunnel_metric
SET bytes_in = (SELECT bytes_in FROM tunnel_metric_dedupe d WHERE d.keep_id = tunnel_metric.id),
bytes_out = (SELECT bytes_out FROM tunnel_metric_dedupe d WHERE d.keep_id = tunnel_metric.id),
connections = (SELECT connections FROM tunnel_metric_dedupe d WHERE d.keep_id = tunnel_metric.id),
errors = (SELECT errors FROM tunnel_metric_dedupe d WHERE d.keep_id = tunnel_metric.id),
avg_latency_ms = (SELECT avg_latency_ms FROM tunnel_metric_dedupe d WHERE d.keep_id = tunnel_metric.id)
WHERE id IN (SELECT keep_id FROM tunnel_metric_dedupe)
`).Error; err != nil {
return fmt.Errorf("update tunnel_metric deduped rows: %w", err)
}
if err := tx.Exec(`
DELETE FROM tunnel_metric
WHERE id IN (
SELECT tm.id
FROM tunnel_metric tm
JOIN tunnel_metric_dedupe d
ON tm.tunnel_id = d.tunnel_id
AND tm.node_id = d.node_id
AND tm.timestamp = d.timestamp
WHERE tm.id <> d.keep_id
)
`).Error; err != nil {
return fmt.Errorf("delete tunnel_metric duplicates: %w", err)
}
_ = tx.Exec(`DROP TABLE IF EXISTS tunnel_metric_dedupe`).Error
}
}
// Uniqueness is required for safe upsert on (tunnel_id, node_id, timestamp).
if err := tx.Exec(
`CREATE UNIQUE INDEX IF NOT EXISTS uidx_tunnel_metric_bucket ON tunnel_metric(tunnel_id, node_id, timestamp)`,
).Error; err != nil {
return fmt.Errorf("create tunnel_metric unique index: %w", err)
}
return nil
})
}
func alterPostgresColumnToBigIntIfNeeded(db *gorm.DB, tableName, columnName string) error {
if db == nil {
return errors.New("nil db")
@@ -3140,3 +3298,331 @@ var osMkdirAll = func(path string) error {
// Suppress unused import warning for log
var _ = log.Printf
func (r *Repository) InsertNodeMetric(m *model.NodeMetric) error {
if r == nil || r.db == nil {
return nil
}
return r.db.Create(m).Error
}
func (r *Repository) InsertNodeMetricBatch(metrics []*model.NodeMetric) error {
if r == nil || r.db == nil || len(metrics) == 0 {
return nil
}
return r.db.CreateInBatches(metrics, 100).Error
}
func (r *Repository) GetNodeMetrics(nodeID int64, startMs, endMs int64) ([]model.NodeMetric, error) {
if r == nil || r.db == nil {
return nil, nil
}
var metrics []model.NodeMetric
err := r.db.Where("node_id = ? AND timestamp >= ? AND timestamp <= ?", nodeID, startMs, endMs).
Order("timestamp DESC").
Limit(5000).
Find(&metrics).Error
if len(metrics) > 1 {
for i, j := 0, len(metrics)-1; i < j; i, j = i+1, j-1 {
metrics[i], metrics[j] = metrics[j], metrics[i]
}
}
return metrics, err
}
func (r *Repository) GetLatestNodeMetric(nodeID int64) (*model.NodeMetric, error) {
if r == nil || r.db == nil {
return nil, nil
}
var m model.NodeMetric
err := r.db.Where("node_id = ?", nodeID).Order("timestamp DESC").First(&m).Error
if err != nil {
if errors.Is(err, gorm.ErrRecordNotFound) {
return nil, nil
}
return nil, err
}
return &m, nil
}
func (r *Repository) PruneNodeMetrics(olderThanMs int64) error {
if r == nil || r.db == nil {
return nil
}
return r.db.Where("timestamp < ?", olderThanMs).Delete(&model.NodeMetric{}).Error
}
func (r *Repository) InsertTunnelMetric(m *model.TunnelMetric) error {
if r == nil || r.db == nil {
return nil
}
return r.db.Create(m).Error
}
func (r *Repository) InsertTunnelMetricBatch(metrics []*model.TunnelMetric) error {
if r == nil || r.db == nil || len(metrics) == 0 {
return nil
}
return r.db.CreateInBatches(metrics, 100).Error
}
// UpsertTunnelMetricBuckets adds the provided metric deltas into per-minute buckets.
// Requires a unique index on (tunnel_id, node_id, timestamp) for safe upserts.
func (r *Repository) UpsertTunnelMetricBuckets(metrics []*model.TunnelMetric) error {
if r == nil || r.db == nil || len(metrics) == 0 {
return nil
}
// Postgres rejects a single INSERT ... ON CONFLICT when the input contains
// duplicate conflict keys. Pre-aggregate within this batch to keep inserts safe.
type bucketKey struct {
tunnelID int64
nodeID int64
timestamp int64
}
agg := make(map[bucketKey]*model.TunnelMetric, len(metrics))
for _, m := range metrics {
if m == nil {
continue
}
if m.TunnelID <= 0 || m.NodeID <= 0 || m.Timestamp <= 0 {
continue
}
if m.BytesIn == 0 && m.BytesOut == 0 && m.Connections == 0 && m.Errors == 0 {
continue
}
k := bucketKey{tunnelID: m.TunnelID, nodeID: m.NodeID, timestamp: m.Timestamp}
if existing, ok := agg[k]; ok {
existing.BytesIn += m.BytesIn
existing.BytesOut += m.BytesOut
existing.Connections += m.Connections
existing.Errors += m.Errors
if existing.AvgLatencyMs == 0 && m.AvgLatencyMs != 0 {
existing.AvgLatencyMs = m.AvgLatencyMs
}
continue
}
cp := *m
agg[k] = &cp
}
if len(agg) == 0 {
return nil
}
rows := make([]*model.TunnelMetric, 0, len(agg))
for _, v := range agg {
rows = append(rows, v)
}
return r.db.Clauses(clause.OnConflict{
Columns: []clause.Column{{Name: "tunnel_id"}, {Name: "node_id"}, {Name: "timestamp"}},
DoUpdates: clause.Assignments(map[string]interface{}{
"bytes_in": gorm.Expr("bytes_in + excluded.bytes_in"),
"bytes_out": gorm.Expr("bytes_out + excluded.bytes_out"),
"connections": gorm.Expr("connections + excluded.connections"),
"errors": gorm.Expr("errors + excluded.errors"),
// avg_latency_ms is not additive; keep the existing bucket value.
}),
}).CreateInBatches(rows, 100).Error
}
func (r *Repository) GetTunnelMetrics(tunnelID int64, startMs, endMs int64) ([]model.TunnelMetric, error) {
if r == nil || r.db == nil {
return nil, nil
}
var metrics []model.TunnelMetric
err := r.db.Where("tunnel_id = ? AND timestamp >= ? AND timestamp <= ?", tunnelID, startMs, endMs).
Order("timestamp DESC").
Limit(5000).
Find(&metrics).Error
if len(metrics) > 1 {
for i, j := 0, len(metrics)-1; i < j; i, j = i+1, j-1 {
metrics[i], metrics[j] = metrics[j], metrics[i]
}
}
return metrics, err
}
// GetTunnelMetricsAggregated returns tunnel-level aggregated series (one point per timestamp).
// Storage remains per (tunnel_id, node_id, timestamp) for future drill-down.
func (r *Repository) GetTunnelMetricsAggregated(tunnelID int64, startMs, endMs int64) ([]model.TunnelMetric, error) {
if r == nil || r.db == nil {
return nil, nil
}
var metrics []model.TunnelMetric
err := r.db.Model(&model.TunnelMetric{}).
Select(
"tunnel_id, 0 AS node_id, timestamp, "+
"SUM(bytes_in) AS bytes_in, "+
"SUM(bytes_out) AS bytes_out, "+
"SUM(connections) AS connections, "+
"SUM(errors) AS errors, "+
"AVG(avg_latency_ms) AS avg_latency_ms",
).
Where("tunnel_id = ? AND timestamp >= ? AND timestamp <= ?", tunnelID, startMs, endMs).
Group("tunnel_id, timestamp").
Order("timestamp ASC").
Limit(5000).
Scan(&metrics).Error
if metrics == nil {
metrics = make([]model.TunnelMetric, 0)
}
return metrics, err
}
func (r *Repository) PruneTunnelMetrics(olderThanMs int64) error {
if r == nil || r.db == nil {
return nil
}
return r.db.Where("timestamp < ?", olderThanMs).Delete(&model.TunnelMetric{}).Error
}
func (r *Repository) ListServiceMonitors() ([]model.ServiceMonitor, error) {
if r == nil || r.db == nil {
return nil, nil
}
var monitors []model.ServiceMonitor
err := r.db.Order("id ASC").Find(&monitors).Error
return monitors, err
}
func (r *Repository) ListEnabledServiceMonitors() ([]model.ServiceMonitor, error) {
if r == nil || r.db == nil {
return nil, nil
}
var monitors []model.ServiceMonitor
err := r.db.Where("enabled = 1 AND type IN (?)", []string{"tcp", "icmp"}).Order("id ASC").Find(&monitors).Error
return monitors, err
}
func (r *Repository) GetServiceMonitor(id int64) (*model.ServiceMonitor, error) {
if r == nil || r.db == nil {
return nil, nil
}
var m model.ServiceMonitor
err := r.db.First(&m, id).Error
if err != nil {
if errors.Is(err, gorm.ErrRecordNotFound) {
return nil, nil
}
return nil, err
}
return &m, nil
}
func (r *Repository) CreateServiceMonitor(m *model.ServiceMonitor) error {
if r == nil || r.db == nil {
return nil
}
return r.db.Create(m).Error
}
func (r *Repository) UpdateServiceMonitor(m *model.ServiceMonitor) error {
if r == nil || r.db == nil {
return nil
}
return r.db.Save(m).Error
}
func (r *Repository) DeleteServiceMonitor(id int64) error {
if r == nil || r.db == nil {
return nil
}
if id <= 0 {
return nil
}
// Keep API/UI semantics simple: deleting a monitor also deletes its history.
return r.db.Transaction(func(tx *gorm.DB) error {
if err := tx.Where("monitor_id = ?", id).Delete(&model.ServiceMonitorResult{}).Error; err != nil {
return err
}
return tx.Delete(&model.ServiceMonitor{}, id).Error
})
}
func (r *Repository) InsertServiceMonitorResult(result *model.ServiceMonitorResult) error {
if r == nil || r.db == nil {
return nil
}
return r.db.Create(result).Error
}
func (r *Repository) GetServiceMonitorResults(monitorID int64, limit int) ([]model.ServiceMonitorResult, error) {
if r == nil || r.db == nil {
return nil, nil
}
if limit <= 0 {
limit = 100
}
var results []model.ServiceMonitorResult
err := r.db.Where("monitor_id = ?", monitorID).
Order("timestamp DESC").
Limit(limit).
Find(&results).Error
return results, err
}
// GetLatestServiceMonitorResults returns the newest result per monitor_id.
// This is intended for list rendering (avoid N+1 queries).
func (r *Repository) GetLatestServiceMonitorResults() ([]model.ServiceMonitorResult, error) {
if r == nil || r.db == nil {
return nil, nil
}
var results []model.ServiceMonitorResult
// Prefer a window-function query (works on modern SQLite + Postgres).
q1 := `
SELECT id, monitor_id, node_id, timestamp, success, latency_ms, status_code, error_message
FROM (
SELECT *, ROW_NUMBER() OVER (PARTITION BY monitor_id ORDER BY timestamp DESC, id DESC) AS rn
FROM service_monitor_result
) t
WHERE rn = 1
ORDER BY monitor_id ASC
`
if err := r.db.Raw(q1).Scan(&results).Error; err == nil {
return results, nil
}
// Fallback: just return newest rows (best-effort). This avoids hard failure on older SQLite builds.
// Note: This may not include all monitors if the table is extremely large and skewed.
results = nil
q2 := `
SELECT id, monitor_id, node_id, timestamp, success, latency_ms, status_code, error_message
FROM service_monitor_result
ORDER BY timestamp DESC, id DESC
LIMIT 5000
`
err := r.db.Raw(q2).Scan(&results).Error
if err != nil {
return nil, err
}
seen := make(map[int64]struct{}, len(results))
out := make([]model.ServiceMonitorResult, 0, len(results))
for _, row := range results {
if row.MonitorID <= 0 {
continue
}
if _, ok := seen[row.MonitorID]; ok {
continue
}
seen[row.MonitorID] = struct{}{}
out = append(out, row)
}
// Keep response stable for the frontend.
sort.Slice(out, func(i, j int) bool { return out[i].MonitorID < out[j].MonitorID })
return out, nil
}
func (r *Repository) PruneServiceMonitorResults(olderThanMs int64) error {
if r == nil || r.db == nil {
return nil
}
return r.db.Where("timestamp < ?", olderThanMs).Delete(&model.ServiceMonitorResult{}).Error
}
@@ -64,6 +64,7 @@ func (r *Repository) ListForwardsByTunnelTx(tx *gorm.DB, tunnelID int64) ([]mode
return rows, nil
}
func (r *Repository) ListActiveTunnelIDsByNode(nodeID int64) ([]int64, error) {
if r == nil || r.db == nil {
return nil, errors.New("repository not initialized")
@@ -125,6 +126,7 @@ func (r *Repository) ListForwardPortsTx(tx *gorm.DB, forwardID int64) ([]model.F
return rows, nil
}
func (r *Repository) HasOtherForwardOnNodePort(nodeID int64, port int, currentForwardID int64) (bool, error) {
if r == nil || r.db == nil {
return false, errors.New("repository not initialized")
@@ -151,6 +153,7 @@ func (r *Repository) HasOtherForwardOnNodePortTx(tx *gorm.DB, nodeID int64, port
return count > 0, nil
}
func (r *Repository) GetTunnelOutProtocol(tunnelID int64) (string, error) {
if r == nil || r.db == nil {
return "", errors.New("repository not initialized")
@@ -161,6 +161,64 @@ func (r *Repository) ForwardExists(forwardID int64) (bool, error) {
return count > 0, nil
}
// MapForwardIDsToTunnelIDs returns a mapping from forward.id to forward.tunnel_id.
// Missing forward IDs are omitted from the returned map.
func (r *Repository) MapForwardIDsToTunnelIDs(forwardIDs []int64) (map[int64]int64, error) {
if r == nil || r.db == nil {
return nil, errors.New("repository not initialized")
}
if len(forwardIDs) == 0 {
return map[int64]int64{}, nil
}
// Deduplicate and filter invalid IDs.
ids := make([]int64, 0, len(forwardIDs))
seen := make(map[int64]struct{}, len(forwardIDs))
for _, id := range forwardIDs {
if id <= 0 {
continue
}
if _, ok := seen[id]; ok {
continue
}
seen[id] = struct{}{}
ids = append(ids, id)
}
if len(ids) == 0 {
return map[int64]int64{}, nil
}
type row struct {
ID int64 `gorm:"column:id"`
TunnelID int64 `gorm:"column:tunnel_id"`
}
out := make(map[int64]int64, len(ids))
const chunkSize = 500
for start := 0; start < len(ids); start += chunkSize {
end := start + chunkSize
if end > len(ids) {
end = len(ids)
}
var rows []row
if err := r.db.Model(&model.Forward{}).
Select("id", "tunnel_id").
Where("id IN ?", ids[start:end]).
Find(&rows).Error; err != nil {
return nil, err
}
for _, r := range rows {
if r.ID <= 0 || r.TunnelID <= 0 {
continue
}
out[r.ID] = r.TunnelID
}
}
return out, nil
}
func (r *Repository) SpeedLimitExists(id int64) (bool, error) {
if r == nil || r.db == nil {
return false, errors.New("repository not initialized")
@@ -0,0 +1,18 @@
package repo
import (
"errors"
"go-backend/internal/store/model"
)
func (r *Repository) ListMonitorNodes() ([]model.Node, error) {
if r == nil || r.db == nil {
return nil, errors.New("repository not initialized")
}
var nodes []model.Node
err := r.db.Select("id", "inx", "name", "status", "updated_time").
Order("inx ASC, id ASC").
Find(&nodes).Error
return nodes, err
}
@@ -0,0 +1,54 @@
package repo
import (
"errors"
"go-backend/internal/store/model"
"gorm.io/gorm/clause"
)
func (r *Repository) InsertMonitorPermission(userID int64, now int64) error {
if r == nil || r.db == nil {
return errors.New("repository not initialized")
}
if userID <= 0 {
return nil
}
row := model.MonitorPermission{UserID: userID, CreatedTime: now}
return r.db.Clauses(clause.OnConflict{DoNothing: true}).Create(&row).Error
}
func (r *Repository) DeleteMonitorPermission(userID int64) error {
if r == nil || r.db == nil {
return errors.New("repository not initialized")
}
if userID <= 0 {
return nil
}
return r.db.Where("user_id = ?", userID).Delete(&model.MonitorPermission{}).Error
}
func (r *Repository) HasMonitorPermission(userID int64) (bool, error) {
if r == nil || r.db == nil {
return false, errors.New("repository not initialized")
}
if userID <= 0 {
return false, nil
}
var count int64
err := r.db.Model(&model.MonitorPermission{}).Where("user_id = ?", userID).Count(&count).Error
if err != nil {
return false, err
}
return count > 0, nil
}
func (r *Repository) ListMonitorPermissions() ([]model.MonitorPermission, error) {
if r == nil || r.db == nil {
return nil, errors.New("repository not initialized")
}
var items []model.MonitorPermission
err := r.db.Order("id ASC").Find(&items).Error
return items, err
}
@@ -0,0 +1,18 @@
package repo
import (
"errors"
"go-backend/internal/store/model"
)
func (r *Repository) ListMonitorTunnels() ([]model.Tunnel, error) {
if r == nil || r.db == nil {
return nil, errors.New("repository not initialized")
}
var tunnels []model.Tunnel
err := r.db.Select("id", "inx", "name", "status", "updated_time").
Order("inx ASC, id ASC").
Find(&tunnels).Error
return tunnels, err
}
@@ -0,0 +1,134 @@
package repo
import (
"sync"
"testing"
"time"
"go-backend/internal/store/model"
)
func TestGetTunnelMetricsAggregatedSumsAcrossNodes(t *testing.T) {
r, err := Open(":memory:")
if err != nil {
t.Fatalf("open repo: %v", err)
}
defer r.Close()
ts := time.Now().UnixMilli()
if err := r.InsertTunnelMetric(&model.TunnelMetric{
TunnelID: 1,
NodeID: 1,
Timestamp: ts,
BytesIn: 100,
BytesOut: 200,
}); err != nil {
t.Fatalf("insert tunnel metric n1: %v", err)
}
if err := r.InsertTunnelMetric(&model.TunnelMetric{
TunnelID: 1,
NodeID: 2,
Timestamp: ts,
BytesIn: 300,
BytesOut: 400,
}); err != nil {
t.Fatalf("insert tunnel metric n2: %v", err)
}
metrics, err := r.GetTunnelMetricsAggregated(1, ts-1000, ts+1000)
if err != nil {
t.Fatalf("get aggregated tunnel metrics: %v", err)
}
if len(metrics) != 1 {
t.Fatalf("expected 1 aggregated point, got %d", len(metrics))
}
if metrics[0].Timestamp != ts {
t.Fatalf("expected timestamp %d, got %d", ts, metrics[0].Timestamp)
}
if metrics[0].BytesIn != 400 {
t.Fatalf("expected bytesIn 400, got %d", metrics[0].BytesIn)
}
if metrics[0].BytesOut != 600 {
t.Fatalf("expected bytesOut 600, got %d", metrics[0].BytesOut)
}
}
func TestUpsertTunnelMetricBucketsAggregatesDuplicateKeysInBatch(t *testing.T) {
r, err := Open(":memory:")
if err != nil {
t.Fatalf("open repo: %v", err)
}
defer r.Close()
ts := time.Now().UnixMilli()
items := []*model.TunnelMetric{
{TunnelID: 1, NodeID: 1, Timestamp: ts, BytesIn: 10, BytesOut: 20},
{TunnelID: 1, NodeID: 1, Timestamp: ts, BytesIn: 30, BytesOut: 40},
}
if err := r.UpsertTunnelMetricBuckets(items); err != nil {
t.Fatalf("upsert buckets: %v", err)
}
rows, err := r.GetTunnelMetrics(1, ts-1000, ts+1000)
if err != nil {
t.Fatalf("get tunnel metrics: %v", err)
}
if len(rows) != 1 {
t.Fatalf("expected 1 stored row, got %d", len(rows))
}
if rows[0].BytesIn != 40 {
t.Fatalf("expected bytesIn 40, got %d", rows[0].BytesIn)
}
if rows[0].BytesOut != 60 {
t.Fatalf("expected bytesOut 60, got %d", rows[0].BytesOut)
}
}
func TestUpsertTunnelMetricBucketsIsSafeUnderConcurrency(t *testing.T) {
r, err := Open(":memory:")
if err != nil {
t.Fatalf("open repo: %v", err)
}
defer r.Close()
ts := time.Now().UnixMilli()
const workers = 20
const perWorkerIn = int64(5)
const perWorkerOut = int64(7)
var wg sync.WaitGroup
wg.Add(workers)
for i := 0; i < workers; i++ {
go func() {
defer wg.Done()
_ = r.UpsertTunnelMetricBuckets([]*model.TunnelMetric{{
TunnelID: 1,
NodeID: 1,
Timestamp: ts,
BytesIn: perWorkerIn,
BytesOut: perWorkerOut,
}})
}()
}
wg.Wait()
rows, err := r.GetTunnelMetrics(1, ts-1000, ts+1000)
if err != nil {
t.Fatalf("get tunnel metrics: %v", err)
}
if len(rows) != 1 {
t.Fatalf("expected 1 stored row, got %d", len(rows))
}
wantIn := int64(workers) * perWorkerIn
wantOut := int64(workers) * perWorkerOut
if rows[0].BytesIn != wantIn {
t.Fatalf("expected bytesIn %d, got %d", wantIn, rows[0].BytesIn)
}
if rows[0].BytesOut != wantOut {
t.Fatalf("expected bytesOut %d, got %d", wantOut, rows[0].BytesOut)
}
}
+91 -4
View File
@@ -72,6 +72,7 @@ type Server struct {
jwtSecret string
upgrader websocket.Upgrader
onNodeOnline func(nodeID int64)
onNodeMetric func(nodeID int64, info SystemInfo)
mu sync.RWMutex
admins map[*connWrap]struct{}
@@ -80,6 +81,22 @@ type Server struct {
pending map[string]pendingRequest
}
type SystemInfo struct {
Uptime uint64 `json:"uptime"`
BytesReceived uint64 `json:"bytes_received"`
BytesTransmitted uint64 `json:"bytes_transmitted"`
CPUUsage float64 `json:"cpu_usage"`
MemoryUsage float64 `json:"memory_usage"`
DiskUsage float64 `json:"disk_usage"`
Load1 float64 `json:"load1"`
Load5 float64 `json:"load5"`
Load15 float64 `json:"load15"`
TCPConns int64 `json:"tcp_conns"`
UDPConns int64 `json:"udp_conns"`
NetInSpeed int64 `json:"net_in_speed"`
NetOutSpeed int64 `json:"net_out_speed"`
}
func (s *Server) SetNodeOnlineHook(fn func(nodeID int64)) {
if s == nil {
return
@@ -89,6 +106,15 @@ func (s *Server) SetNodeOnlineHook(fn func(nodeID int64)) {
s.mu.Unlock()
}
func (s *Server) SetNodeMetricHook(fn func(nodeID int64, info SystemInfo)) {
if s == nil {
return
}
s.mu.Lock()
s.onNodeMetric = fn
s.mu.Unlock()
}
func NewServer(repo *repo.Repository, jwtSecret string) *Server {
return &Server{
repo: repo,
@@ -231,12 +257,73 @@ func (s *Server) handleNode(w http.ResponseWriter, r *http.Request, nodeID int64
var parsed struct {
Type string `json:"type"`
}
if json.Unmarshal([]byte(msg), &parsed) == nil && parsed.Type == "UpgradeProgress" {
s.broadcastTyped(nodeID, "upgrade_progress", msg)
} else {
s.broadcastInfo(nodeID, msg)
if json.Unmarshal([]byte(msg), &parsed) == nil && parsed.Type != "" {
switch parsed.Type {
case "UpgradeProgress":
s.broadcastTyped(nodeID, "upgrade_progress", msg)
continue
default:
// Unknown typed messages still get broadcast so future
// agent message types are not silently lost.
s.broadcastInfo(nodeID, msg)
continue
}
}
if looksLikeSystemInfoMessage(msg) {
var sysInfo SystemInfo
if err := json.Unmarshal([]byte(msg), &sysInfo); err == nil {
s.mu.RLock()
onMetric := s.onNodeMetric
s.mu.RUnlock()
if onMetric != nil {
go onMetric(nodeID, sysInfo)
}
s.broadcastTyped(nodeID, "metric", msg)
continue
}
}
s.broadcastInfo(nodeID, msg)
}
}
func looksLikeSystemInfoMessage(msg string) bool {
// Keep this as a cheap heuristic so that arbitrary JSON objects don't get
// misclassified as metrics (SystemInfo unmarshal would otherwise succeed with
// all-zero values).
if strings.TrimSpace(msg) == "" {
return false
}
if !strings.Contains(msg, "{") {
return false
}
keys := []string{
"\"uptime\"",
"\"cpu_usage\"",
"\"memory_usage\"",
"\"disk_usage\"",
"\"bytes_received\"",
"\"bytes_transmitted\"",
"\"net_in_speed\"",
"\"net_out_speed\"",
"\"tcp_conns\"",
"\"udp_conns\"",
"\"load1\"",
"\"load5\"",
"\"load15\"",
}
matched := 0
for _, k := range keys {
if strings.Contains(msg, k) {
matched++
if matched >= 3 {
return true
}
}
}
return false
}
func (s *Server) SendCommand(nodeID int64, cmdType string, data interface{}, timeout time.Duration) (CommandResult, error) {
@@ -112,6 +112,12 @@ func TestIssue313_EntryPortCrossTunnelConflictContract(t *testing.T) {
t.Fatalf("insert forward_port a: %v", err)
}
// Simulate legacy dirty data: tunnel A already occupies port 2000 on entryB2.
// When tunnel B adds entryB2, the inherited forward port should conflict cross-tunnel.
if err := repo.DB().Exec(`INSERT INTO forward_port(forward_id, node_id, port) VALUES(?, ?, ?)`, forwardAID, entryB2, 2000).Error; err != nil {
t.Fatalf("insert forward_port a on entryB2: %v", err)
}
if err := repo.DB().Exec(`
INSERT INTO user_tunnel(id, user_id, tunnel_id, speed_id, num, flow, in_flow, out_flow, flow_reset_time, exp_time, status)
VALUES(3132, 1, ?, NULL, 999, 99999, 0, 0, 1, 2727251700000, 1)
@@ -170,7 +176,8 @@ func TestIssue313_EntryPortCrossTunnelConflictContract(t *testing.T) {
t.Fatalf("expected update failure due to cross-tunnel port conflict, got success with code 0")
}
if !bytes.Contains([]byte(out.Msg), []byte("端口")) && !bytes.Contains([]byte(out.Msg), []byte("占用")) {
msgBytes := []byte(out.Msg)
if !bytes.Contains(msgBytes, []byte("端口")) && !bytes.Contains(msgBytes, []byte("占用")) {
t.Fatalf("expected port conflict error message, got %q", out.Msg)
}
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,97 @@
package contract_test
import (
"bytes"
"encoding/json"
"net/http"
"net/http/httptest"
"testing"
"time"
"go-backend/internal/store/model"
)
func TestFlowUploadInsertsTunnelMetrics(t *testing.T) {
secret := "monitoring-jwt-secret"
router, repo := setupContractRouter(t, secret)
now := time.Now().UnixMilli()
node := &model.Node{
Name: "node-1",
Secret: "node-secret",
ServerIP: "127.0.0.1",
Port: "10000-10010",
TCPListenAddr: "[::]",
UDPListenAddr: "[::]",
CreatedTime: now,
Status: 1,
}
if err := repo.DB().Create(node).Error; err != nil {
t.Fatalf("seed node: %v", err)
}
tunnel := &model.Tunnel{
Name: "tunnel-1",
TrafficRatio: 1.0,
Type: 1,
Protocol: "tls",
Flow: 1,
CreatedTime: now,
UpdatedTime: now,
Status: 1,
}
if err := repo.DB().Create(tunnel).Error; err != nil {
t.Fatalf("seed tunnel: %v", err)
}
forward := &model.Forward{
UserID: 123,
UserName: "user-123",
Name: "forward-1",
TunnelID: tunnel.ID,
RemoteAddr: "1.1.1.1:80",
CreatedTime: now,
UpdatedTime: now,
Status: 1,
}
if err := repo.DB().Create(forward).Error; err != nil {
t.Fatalf("seed forward: %v", err)
}
serviceName := jsonNumber(forward.ID) + "_123_0"
body, _ := json.Marshal([]map[string]interface{}{{
"n": serviceName,
"u": 200,
"d": 100,
}})
req := httptest.NewRequest(http.MethodPost, "/flow/upload?secret="+node.Secret, bytes.NewReader(body))
res := httptest.NewRecorder()
router.ServeHTTP(res, req)
if res.Code != http.StatusOK {
t.Fatalf("expected status 200, got %d", res.Code)
}
metrics, err := repo.GetTunnelMetrics(tunnel.ID, 0, now+60_000)
if err != nil {
t.Fatalf("get tunnel metrics: %v", err)
}
if len(metrics) != 1 {
t.Fatalf("expected 1 tunnel metric row, got %d", len(metrics))
}
if metrics[0].TunnelID != tunnel.ID {
t.Fatalf("expected tunnelId %d, got %d", tunnel.ID, metrics[0].TunnelID)
}
if metrics[0].NodeID != node.ID {
t.Fatalf("expected nodeId %d, got %d", node.ID, metrics[0].NodeID)
}
if metrics[0].BytesIn != 100 {
t.Fatalf("expected bytesIn 100, got %d", metrics[0].BytesIn)
}
if metrics[0].BytesOut != 200 {
t.Fatalf("expected bytesOut 200, got %d", metrics[0].BytesOut)
}
}
+3 -6
View File
@@ -7,7 +7,6 @@ import (
"errors"
"fmt"
"io"
"log"
"net"
"os"
"os/exec"
@@ -63,11 +62,9 @@ func SetProtocolBlock(httpOn int, tlsOn int, socksOn int) {
type Option func(opts *options)
func init() {
_, err := LoadConfig("config.json")
fmt.Println("config.json loaded")
if err != nil {
log.Fatal(err)
}
// NOTE: This package can be imported by tests/tools that don't have a local
// config.json. Missing config should not crash the process.
_, _ = LoadConfig("config.json")
needWrap = isTls+isSocks+isHttp > 0
}
+380 -18
View File
@@ -17,7 +17,7 @@ import (
"runtime"
"strconv"
"strings"
"sync" // 新增:用于管理连接状态的互斥锁
"sync"
"time"
"github.com/go-gost/x/config"
@@ -25,34 +25,67 @@ import (
"github.com/go-gost/x/service"
"github.com/gorilla/websocket"
"github.com/shirou/gopsutil/v3/cpu"
"github.com/shirou/gopsutil/v3/disk"
"github.com/shirou/gopsutil/v3/host"
"github.com/shirou/gopsutil/v3/load"
"github.com/shirou/gopsutil/v3/mem"
psnet "github.com/shirou/gopsutil/v3/net"
"golang.org/x/net/icmp"
"golang.org/x/net/ipv4"
"golang.org/x/net/ipv6"
)
// SystemInfo 系统信息结构体
type SystemInfo struct {
Uptime uint64 `json:"uptime"` // 开机时间 (秒)
BytesReceived uint64 `json:"bytes_received"` // 接收字节数
BytesTransmitted uint64 `json:"bytes_transmitted"` // 发送字节数
CPUUsage float64 `json:"cpu_usage"` // CPU使用率(百分比)
MemoryUsage float64 `json:"memory_usage"` // 内存使用率(百分比)
Uptime uint64 `json:"uptime"`
BytesReceived uint64 `json:"bytes_received"`
BytesTransmitted uint64 `json:"bytes_transmitted"`
CPUUsage float64 `json:"cpu_usage"`
MemoryUsage float64 `json:"memory_usage"`
DiskUsage float64 `json:"disk_usage"`
Load1 float64 `json:"load1"`
Load5 float64 `json:"load5"`
Load15 float64 `json:"load15"`
TCPConns int64 `json:"tcp_conns"`
UDPConns int64 `json:"udp_conns"`
NetInSpeed int64 `json:"net_in_speed"`
NetOutSpeed int64 `json:"net_out_speed"`
}
// NetworkStats 网络统计信息
type NetworkStats struct {
BytesReceived uint64 `json:"bytes_received"` // 接收字节数
BytesTransmitted uint64 `json:"bytes_transmitted"` // 发送字节数
BytesReceived uint64 `json:"bytes_received"`
BytesTransmitted uint64 `json:"bytes_transmitted"`
BytesRecvDelta uint64 `json:"bytes_recv_delta"`
BytesSentDelta uint64 `json:"bytes_sent_delta"`
}
// CPUInfo CPU信息
type CPUInfo struct {
Usage float64 `json:"usage"` // CPU使用率(百分比)
Usage float64 `json:"usage"`
}
// MemoryInfo 内存信息
type MemoryInfo struct {
Usage float64 `json:"usage"` // 内存使用率(百分比)
Usage float64 `json:"usage"`
}
// DiskInfo 磁盘信息
type DiskInfo struct {
Usage float64 `json:"usage"`
}
// LoadInfo 负载信息
type LoadInfo struct {
Load1 float64 `json:"load1"`
Load5 float64 `json:"load5"`
Load15 float64 `json:"load15"`
}
// ConnectionInfo 连接信息
type ConnectionInfo struct {
TCPConns int64 `json:"tcp_conns"`
UDPConns int64 `json:"udp_conns"`
}
// CommandMessage 命令消息结构体
@@ -91,6 +124,25 @@ type TcpPingResponse struct {
RequestId string `json:"requestId,omitempty"`
}
// ServiceMonitorCheckRequest service monitor check request.
type ServiceMonitorCheckRequest struct {
MonitorID int64 `json:"monitorId"`
Type string `json:"type"` // tcp|icmp
Target string `json:"target"`
TimeoutSec int `json:"timeoutSec"`
}
// ServiceMonitorCheckResult node-executed check output.
// CommandResponse.Success indicates command execution status.
// Actual check success is represented by this struct.
type ServiceMonitorCheckResult struct {
MonitorID int64 `json:"monitorId"`
Success bool `json:"success"`
LatencyMs float64 `json:"latencyMs"`
StatusCode int `json:"statusCode,omitempty"`
ErrorMessage string `json:"errorMessage,omitempty"`
}
const (
reporterReadWait = 60 * time.Second
reporterWriteWait = 5 * time.Second
@@ -134,7 +186,7 @@ func NewWebSocketReporter(serverURL string, secret string) *WebSocketReporter {
return &WebSocketReporter{
url: serverURL,
reconnectTime: 5 * time.Second, // 重连间隔
pingInterval: 2 * time.Second, // 发送间隔改为2秒
pingInterval: 5 * time.Second, // 指标上报间隔
configInterval: 10 * time.Minute, // 配置上报间隔
ctx: ctx,
cancel: cancel,
@@ -449,11 +501,35 @@ func (w *WebSocketReporter) handleConnection() {
}
}
var lastNetBytesReceived uint64
var lastNetBytesTransmitted uint64
var lastNetTime int64
var connInfoCached ConnectionInfo
var connInfoCachedAt int64
var connInfoCachedMu sync.Mutex
// collectSystemInfo 收集系统信息
func (w *WebSocketReporter) collectSystemInfo() SystemInfo {
networkStats := getNetworkStats()
cpuInfo := getCPUInfo()
memoryInfo := getMemoryInfo()
diskInfo := getDiskInfo()
loadInfo := getLoadInfo()
connInfo := getConnectionInfo()
now := time.Now().UnixMilli()
var netInSpeed, netOutSpeed int64
if lastNetTime > 0 {
deltaMs := now - lastNetTime
if deltaMs > 0 {
netInSpeed = int64(float64(networkStats.BytesRecvDelta) * 1000 / float64(deltaMs))
netOutSpeed = int64(float64(networkStats.BytesSentDelta) * 1000 / float64(deltaMs))
}
}
lastNetBytesReceived = networkStats.BytesReceived
lastNetBytesTransmitted = networkStats.BytesTransmitted
lastNetTime = now
return SystemInfo{
Uptime: getUptime(),
@@ -461,6 +537,14 @@ func (w *WebSocketReporter) collectSystemInfo() SystemInfo {
BytesTransmitted: networkStats.BytesTransmitted,
CPUUsage: cpuInfo.Usage,
MemoryUsage: memoryInfo.Usage,
DiskUsage: diskInfo.Usage,
Load1: loadInfo.Load1,
Load5: loadInfo.Load5,
Load15: loadInfo.Load15,
TCPConns: connInfo.TCPConns,
UDPConns: connInfo.UDPConns,
NetInSpeed: netInSpeed,
NetOutSpeed: netOutSpeed,
}
}
@@ -622,7 +706,7 @@ func (w *WebSocketReporter) handleReceivedMessage(messageType int, message []byt
if cmdMsg.Type != "call" {
// 其他状态变更命令保持同步,确保顺序执行
if cmdMsg.Type == "TcpPing" || cmdMsg.Type == "UpgradeAgent" || cmdMsg.Type == "RollbackAgent" {
if cmdMsg.Type == "TcpPing" || cmdMsg.Type == "ServiceMonitorCheck" || cmdMsg.Type == "UpgradeAgent" || cmdMsg.Type == "RollbackAgent" {
go w.routeCommand(cmdMsg)
} else {
w.routeCommand(cmdMsg)
@@ -638,7 +722,7 @@ func (w *WebSocketReporter) handleReceivedMessage(messageType int, message []byt
}
if cmdMsg.Type != "call" {
// 其他状态变更命令保持同步,确保顺序执行
if cmdMsg.Type == "TcpPing" || cmdMsg.Type == "UpgradeAgent" || cmdMsg.Type == "RollbackAgent" {
if cmdMsg.Type == "TcpPing" || cmdMsg.Type == "ServiceMonitorCheck" || cmdMsg.Type == "UpgradeAgent" || cmdMsg.Type == "RollbackAgent" {
go w.routeCommand(cmdMsg)
} else {
w.routeCommand(cmdMsg)
@@ -726,6 +810,13 @@ func (w *WebSocketReporter) routeCommand(cmd CommandMessage) {
response.Data = tcpPingResult
// needSaveConfig = false (默认值)
// Service monitor check (read-only)
case "ServiceMonitorCheck":
var checkResult ServiceMonitorCheckResult
checkResult, err = w.handleServiceMonitorCheck(cmd.Data)
response.Type = "ServiceMonitorCheckResponse"
response.Data = checkResult
// Protocol blocking switches
case "SetProtocol":
err = w.handleSetProtocol(cmd.Data)
@@ -1381,17 +1472,21 @@ func getNetworkStats() NetworkStats {
return stats
}
// 汇总所有非回环接口的流量
for _, io := range ioCounters {
// 跳过回环接口
if io.Name == "lo" || strings.HasPrefix(io.Name, "lo") {
continue
}
stats.BytesReceived += io.BytesRecv
stats.BytesTransmitted += io.BytesSent
}
if lastNetBytesReceived > 0 && stats.BytesReceived >= lastNetBytesReceived {
stats.BytesRecvDelta = stats.BytesReceived - lastNetBytesReceived
}
if lastNetBytesTransmitted > 0 && stats.BytesTransmitted >= lastNetBytesTransmitted {
stats.BytesSentDelta = stats.BytesTransmitted - lastNetBytesTransmitted
}
return stats
}
@@ -1399,8 +1494,8 @@ func getNetworkStats() NetworkStats {
func getCPUInfo() CPUInfo {
var cpuInfo CPUInfo
// 获取CPU使用率
percentages, err := cpu.Percent(time.Second, false)
// 获取CPU使用率 (non-blocking)
percentages, err := cpu.Percent(0, false)
if err == nil && len(percentages) > 0 {
cpuInfo.Usage = percentages[0]
}
@@ -1422,6 +1517,69 @@ func getMemoryInfo() MemoryInfo {
return memInfo
}
// getDiskInfo 获取磁盘信息
func getDiskInfo() DiskInfo {
var diskInfo DiskInfo
usage, err := disk.Usage("/")
if err != nil {
return diskInfo
}
diskInfo.Usage = usage.UsedPercent
return diskInfo
}
// getLoadInfo 获取负载信息
func getLoadInfo() LoadInfo {
var loadInfo LoadInfo
avg, err := load.Avg()
if err != nil {
return loadInfo
}
loadInfo.Load1 = avg.Load1
loadInfo.Load5 = avg.Load5
loadInfo.Load15 = avg.Load15
return loadInfo
}
// getConnectionInfo 获取连接信息
func getConnectionInfo() ConnectionInfo {
now := time.Now().UnixMilli()
const refreshEveryMs = int64((15 * time.Second) / time.Millisecond)
connInfoCachedMu.Lock()
if connInfoCachedAt > 0 && now-connInfoCachedAt < refreshEveryMs {
v := connInfoCached
connInfoCachedMu.Unlock()
return v
}
connInfoCachedMu.Unlock()
var connInfo ConnectionInfo
connStats, err := psnet.Connections("tcp")
if err == nil {
connInfo.TCPConns = int64(len(connStats))
}
udpStats, err := psnet.Connections("udp")
if err == nil {
connInfo.UDPConns = int64(len(udpStats))
}
connInfoCachedMu.Lock()
connInfoCached = connInfo
connInfoCachedAt = now
connInfoCachedMu.Unlock()
return connInfo
}
// StartWebSocketReporterWithConfig 使用配置字段启动WebSocket报告器
func StartWebSocketReporterWithConfig(addr string, secret string, http int, tls int, socks int, version string) *WebSocketReporter {
@@ -1503,6 +1661,210 @@ func (w *WebSocketReporter) handleTcpPing(data interface{}) (TcpPingResponse, er
return response, nil
}
// handleServiceMonitorCheck executes a service monitor check on this node.
// It always returns a result (command execution is considered successful even if the check fails).
func (w *WebSocketReporter) handleServiceMonitorCheck(data interface{}) (ServiceMonitorCheckResult, error) {
jsonData, err := json.Marshal(data)
if err != nil {
return ServiceMonitorCheckResult{}, fmt.Errorf("序列化检查数据失败: %v", err)
}
var req ServiceMonitorCheckRequest
if err := json.Unmarshal(jsonData, &req); err != nil {
return ServiceMonitorCheckResult{}, fmt.Errorf("解析检查请求失败: %v", err)
}
checkType := strings.ToLower(strings.TrimSpace(req.Type))
target := strings.TrimSpace(req.Target)
res := ServiceMonitorCheckResult{MonitorID: req.MonitorID}
if checkType != "tcp" && checkType != "icmp" {
res.Success = false
res.ErrorMessage = "不支持的检查类型"
return res, nil
}
if target == "" {
res.Success = false
res.ErrorMessage = "检查目标为空"
return res, nil
}
timeoutSec := req.TimeoutSec
if timeoutSec <= 0 {
timeoutSec = 5
}
timeout := time.Duration(timeoutSec) * time.Second
start := time.Now()
switch checkType {
case "tcp":
// Validate and normalize host:port.
_, _, splitErr := net.SplitHostPort(target)
if splitErr != nil {
res.Success = false
res.ErrorMessage = "无效的TCP目标"
res.LatencyMs = float64(time.Since(start).Milliseconds())
return res, nil
}
conn, dialErr := net.DialTimeout("tcp", target, timeout)
res.LatencyMs = float64(time.Since(start).Milliseconds())
if dialErr != nil {
res.Success = false
res.ErrorMessage = dialErr.Error()
return res, nil
}
_ = conn.Close()
res.Success = true
return res, nil
case "icmp":
rtt, pingErr := icmpPing(target, timeout)
res.LatencyMs = float64(rtt.Milliseconds())
if pingErr != nil {
res.Success = false
res.ErrorMessage = pingErr.Error()
return res, nil
}
res.Success = true
return res, nil
}
res.Success = false
res.ErrorMessage = "未知错误"
res.LatencyMs = float64(time.Since(start).Milliseconds())
return res, nil
}
func icmpPing(target string, timeout time.Duration) (time.Duration, error) {
start := time.Now()
target = strings.TrimSpace(target)
if target == "" {
return time.Since(start), fmt.Errorf("无效的ICMP目标")
}
// Avoid accepting URL-like targets.
if strings.Contains(target, "://") {
return time.Since(start), fmt.Errorf("无效的ICMP目标")
}
if strings.HasPrefix(target, "[") && strings.HasSuffix(target, "]") {
target = strings.TrimSuffix(strings.TrimPrefix(target, "["), "]")
}
ipAddr, err := net.ResolveIPAddr("ip", target)
if err != nil || ipAddr == nil || ipAddr.IP == nil {
if err == nil {
err = fmt.Errorf("unknown address")
}
return time.Since(start), fmt.Errorf("解析目标失败: %v", err)
}
isV4 := ipAddr.IP.To4() != nil
listenAddr := "0.0.0.0"
proto := 1
var echoType icmp.Type = ipv4.ICMPTypeEcho
var echoReplyType icmp.Type = ipv4.ICMPTypeEchoReply
networks := []string{"udp4", "ip4:icmp"}
if !isV4 {
listenAddr = "::"
proto = 58
echoType = ipv6.ICMPTypeEchoRequest
echoReplyType = ipv6.ICMPTypeEchoReply
networks = []string{"udp6", "ip6:ipv6-icmp"}
}
var conn *icmp.PacketConn
selectedNetwork := ""
var lastErr error
for _, nw := range networks {
c, err := icmp.ListenPacket(nw, listenAddr)
if err == nil {
conn = c
selectedNetwork = nw
break
}
lastErr = err
}
if conn == nil {
if lastErr != nil {
return time.Since(start), fmt.Errorf("创建ICMP连接失败: %v", lastErr)
}
return time.Since(start), fmt.Errorf("创建ICMP连接失败")
}
defer conn.Close()
id := os.Getpid() & 0xffff
seq := 1
wm := icmp.Message{
Type: echoType,
Code: 0,
Body: &icmp.Echo{
ID: id,
Seq: seq,
Data: []byte("FLVX-PING"),
},
}
wb, err := wm.Marshal(nil)
if err != nil {
return time.Since(start), err
}
_ = conn.SetDeadline(time.Now().Add(timeout))
var dst net.Addr
if strings.HasPrefix(selectedNetwork, "udp") {
dst = &net.UDPAddr{IP: ipAddr.IP, Zone: ipAddr.Zone}
} else {
dst = &net.IPAddr{IP: ipAddr.IP, Zone: ipAddr.Zone}
}
if _, err := conn.WriteTo(wb, dst); err != nil {
return time.Since(start), err
}
addrIP := func(a net.Addr) net.IP {
switch v := a.(type) {
case *net.IPAddr:
return v.IP
case *net.UDPAddr:
return v.IP
default:
return nil
}
}
rb := make([]byte, 1500)
for {
n, peer, err := conn.ReadFrom(rb)
if err != nil {
return time.Since(start), err
}
if p := addrIP(peer); p != nil && !p.Equal(ipAddr.IP) {
continue
}
rm, err := icmp.ParseMessage(proto, rb[:n])
if err != nil {
continue
}
if rm.Type != echoReplyType {
continue
}
echo, ok := rm.Body.(*icmp.Echo)
if !ok {
continue
}
if echo.Seq != seq {
continue
}
// For non-privileged endpoints, the kernel may choose the ID.
if !strings.HasPrefix(selectedNetwork, "udp") && echo.ID != id {
continue
}
return time.Since(start), nil
}
}
// tcpPingHost 执行TCP连接测试,返回平均连接时间和失败率
func tcpPingHost(ip string, port int, count int, timeoutMs int) (float64, float64, error) {
var totalTime float64
+104
View File
@@ -0,0 +1,104 @@
# 037 - Monitoring: Node Metrics + Service Health Checks
## Context
This worktree introduces a monitoring feature set:
- Node runtime metrics streamed via WebSocket (agent -> panel -> admin clients)
- Metrics ingestion + retention in panel DB
- Service monitoring (TCP/ICMP checks only) + result storage
- Frontend monitor view (charts + monitor CRUD + run + results)
- Dedicated monitor page (`/monitor`) that works for authorized non-admin users
The initial implementation landed without a plan doc and had several correctness issues (API JSON shape mismatch, wrong time units, contract test hangs under SQLite single-connection mode, etc.). This plan documents what exists, what was fixed, and what is still incomplete/needs decisions.
## Goals
- Metrics endpoints return stable JSON fields matching frontend types.
- Contract tests cover metrics + monitor CRUD and are deterministic.
- WebSocket metric messages update node cards correctly.
- Monitoring view queries the correct time range and renders timestamps correctly.
- go-gost/x unit tests do not depend on a local config.json.
## Non-goals (for this plan)
- A full monitor scheduling system (jitter/backoff/concurrency budgets/per-monitor next-run) beyond the current simple loop.
- Building a full alerting pipeline (notifications, thresholds, paging).
## Current Status (as of this worktree)
- Backend models updated with JSON tags for monitoring structs.
- Handler endpoints for metrics + service monitors added.
- Metrics ingestion service implemented with buffering + retention pruning.
- Health checker implemented (panel-side when `nodeId == 0`; node-executed via WS when `nodeId > 0`) and background jobs wired.
- Frontend monitor view added; build passes.
- Contract tests for monitoring added.
- Monitoring endpoints are accessible by admin users and non-admin users explicitly authorized by admin (via `monitor_permission`).
- Frontend exposes monitoring via a dedicated `/monitor` page; admin can grant/revoke monitoring permission from the User permissions modal.
- Frontend includes tunnel metrics charts (backed by `/api/v1/monitor/tunnels` list + `/api/v1/monitor/tunnels/:id/metrics`).
## Known Semantics Gaps (need decisions)
- `service_monitor.intervalSec` is best-effort (checker ticks every 30s; intervals shorter than that won't run faster).
- `service_monitor_result.success` is stored as int (0/1). Frontend currently treats it as number; decide if API should expose boolean.
## Admin Authorization API
Monitoring permission management (admin-only):
- `GET /api/v1/monitor/permission/list`
- `POST /api/v1/monitor/permission/assign` body: `{ "userId": 123 }`
- `POST /api/v1/monitor/permission/remove` body: `{ "userId": 123 }`
## Checklist
### Phase 1: Correctness + Contracts
- [x] Align monitoring JSON response fields with frontend/contract expectations (add json tags or DTO mapping).
- [x] Fix frontend monitor time range query (use ms start/end; avoid `start=60`).
- [x] Fix frontend timestamp rendering (treat timestamp as UnixMilli).
- [x] Fix node realtime metric speed field compatibility (support snake_case speed fields).
- [x] Fix SQLite contract hang by ensuring tunnel-entry precheck uses tx-safe DB reads (no nested connection acquisition).
- [x] Ensure monitoring contract tests pass.
### Phase 2: Semantics Alignment (Decide + Implement)
- [x] Decide "service monitors run where":
- Option B: node-executed when `nodeId > 0` (chosen)
- [ ] Define interval semantics:
- Per-monitor next-run scheduling vs global scan loop
- Backoff on failures
- Maximum monitors + runtime cost guardrails
- [ ] Standardize API type for `success`:
- Keep int for backward compatibility, or
- Return boolean in API responses (DTO) while storing int in DB
### Phase 2.1: Partial Implementation (No Semantics Decision Yet)
- [x] Honor `intervalSec` best-effort in panel-side checker (min cadence still bound by global loop).
### Phase 2.2: Node-Executed Checks
- [x] Add a WebSocket command for node-executed monitor checks (`ServiceMonitorCheck`).
- [x] Panel health checker dispatches checks to the specified node when `nodeId > 0`.
- [x] Allow unrestricted targets by policy; restrict monitoring endpoints to admin + explicitly authorized users.
- [x] Remove HTTP checks; service monitoring supports only `tcp` and `icmp`.
### Phase 3: Hardening + Performance
- [x] Add query limits/guards for metrics endpoints (max range, max rows) to avoid accidental full-history pulls.
- [ ] Consider indexing review and retention configurability (env or config table).
- [ ] Review concurrency: ingestion buffer flush goroutine spawning and DB write pressure.
- [x] Add minimal UI affordances: time range selector, empty/error states, and service monitor run/results UI.
- [x] Ensure monitoring UI works for authorized non-admin users (dedicated `/monitor` page; no reliance on admin-only `/node/*`).
### Phase 4: Hygiene
- [x] Add `.entire/metadata/` to `.gitignore` (should never be committed).
- [ ] Add a short developer note in docs/README if needed (API endpoints + semantics).
## Test Plan
Backend:
```bash
cd go-backend && go test ./... -count=1
cd go-backend && go test ./tests/contract -count=1 -timeout 120s
```
Agent fork:
```bash
cd go-gost/x && go test ./... -count=1
```
Frontend:
```bash
cd vite-frontend && npm run build
```
## Notes
- Node-executed checks can be used for internal probing by design; access is restricted to administrators.
@@ -0,0 +1,59 @@
# 038 - Monitoring Bug Fixes + Optimizations
## Context
Monitoring in FLVX currently spans:
- Agent -> panel WebSocket realtime system metrics (CPU/mem/disk/net/load/conns)
- Panel-side ingestion + retention pruning (`node_metric`)
- Service monitors (TCP/ICMP) with scheduled checks + stored results
- Frontend monitor page (`/monitor`) with charts + monitor CRUD/run/results
While the feature set works end-to-end, there are a few correctness footguns and a couple of obvious performance hot spots (agent-side sampling cost and frontend N+1 polling patterns).
## Goals
- Service monitor updates do not accidentally clear `nodeId` / `enabled` when fields are omitted.
- Checker cadence is explicit (intervals below the scan cadence are clamped / best-effort).
- Reduce frontend requests for service monitor status (avoid per-monitor polling).
- Reduce agent sampling overhead and DB write volume without breaking UI expectations.
- Avoid misclassifying arbitrary JSON as a metric message on the WS channel.
## Non-goals
- A full scheduler (per-monitor next-run queue, jitter/backoff, concurrency budgets).
- Alerting/notifications.
- Implementing full tunnel-metrics ingestion (connections/errors/latency) beyond current endpoints.
## Checklist
### Phase 1: Backend Correctness + Hardening
- [x] Make `/api/v1/monitor/services/update` treat `nodeId` and `enabled` as optional fields (no accidental zeroing).
- [x] Clamp `intervalSec` to a minimum that matches the checker scan cadence (and apply the same clamp in the checker).
- [x] Add `GET /api/v1/monitor/services/latest-results` returning the latest result per monitor (for frontend list rendering).
- [x] WS metric parsing: only treat messages as metrics when they look like a system-metric payload.
### Phase 2: Frontend UX + Request Reduction
- [x] Fix “立即检查” toast severity (failure should be an error toast).
- [x] Use `latest-results` endpoint to render service monitor status without N+1 polling.
- [x] Add a small hint when chart data is truncated by backend row limits.
### Phase 3: Agent Sampling Optimizations
- [x] Reduce default WS metric send interval (2s -> 5s).
- [x] Make CPU sampling non-blocking and cache heavy metrics (e.g. connection counts) to reduce per-sample cost.
## Test Plan
Backend:
```bash
cd go-backend && go test ./... -count=1
```
Agent fork:
```bash
cd go-gost/x && go test ./... -count=1
```
Frontend (best-effort in this environment):
```bash
cd vite-frontend && npm run build
```
## Rollout Notes
- Agent sampling interval change reduces metric resolution and DB growth; charts remain usable and realtime UI remains responsive.
- Existing monitors with very small `intervalSec` are best-effort; effective cadence remains bounded by the checker scan loop.
@@ -0,0 +1,41 @@
# 039 - Monitoring: Tunnel Metrics Ingestion
## Context
The `/monitor` UI includes tunnel metric charts backed by:
- `GET /api/v1/monitor/tunnels` (list)
- `GET /api/v1/monitor/tunnels/:id/metrics` (timeseries)
The backend has the `tunnel_metric` table + query endpoints, but there is no production code path that writes tunnel metrics. As a result, tunnel charts are typically empty.
## Goal
Persist tunnel traffic timeseries based on agent flow uploads (`POST /flow/upload`).
## Scope
- Write `tunnel_metric` rows from flow uploads.
- Keep write volume bounded (aggregate per minute).
- Provide contract coverage that a flow upload creates tunnel metrics.
## Non-goals
- Populate connections/errors/latency for tunnel metrics (remain 0 for now).
- A full aggregation pipeline across multiple nodes per tunnel at query time.
UI note:
- The tunnel chart only exposes the Traffic view for now; other tabs are hidden.
## Design
- Agent reports per-service traffic deltas via `/flow/upload` with items `{n,u,d}`.
- Backend derives `forward_id` from service name (`<forwardID>_<userID>_<userTunnelID>[...suffix]`).
- Map `forward_id -> tunnel_id` in batch.
- Aggregate per `(node_id, tunnel_id, minute_bucket)` and upsert into `tunnel_metric` using an UPDATE-then-INSERT fallback.
## Checklist
- [x] Add repository helper: map forward IDs to tunnel IDs.
- [x] Add repository helper: upsert per-minute tunnel metric buckets.
- [x] Extend `/flow/upload` handler to record tunnel metrics from incoming items.
- [x] Add contract test verifying flow upload produces tunnel metrics.
- [x] Run backend tests.
## Test Plan
```bash
cd go-backend && go test ./... -count=1
```
@@ -0,0 +1,40 @@
# 040 - Service Monitor Limits Config + UI Hints
## Goal
Make service monitor interval/timeout constraints configurable (instead of hard-coded clamps) and make the UI clearly communicate the effective limits.
## Current Pain
- Backend clamps `intervalSec` and `timeoutSec` with hard-coded constants.
- Checker scan cadence is also hard-coded, so users can set values that will never be honored.
- Frontend form does not explain allowed ranges or why values may change.
## Approach
- Add frontend-configurable limits stored in `vite_config` (with safe defaults matching current behavior).
- Backend always normalizes using the configured limits.
- Expose the current limits via a monitoring endpoint so the UI can render accurate hints.
- Frontend shows min/max and validates before submit.
- Admin can edit the limits on `/config`.
## Config Keys (vite_config)
- `service_monitor_checker_scan_interval_sec` (default: 30)
- `service_monitor_min_interval_sec` (default: 30; auto-raised to at least scan interval)
- `service_monitor_default_interval_sec` (default: 60)
- `service_monitor_min_timeout_sec` (default: 1)
- `service_monitor_default_timeout_sec` (default: 5)
- `service_monitor_max_timeout_sec` (default: 60)
## Checklist
Backend:
- [x] Introduce shared `ServiceMonitorLimits` config loader.
- [x] Use limits for create/update normalization.
- [x] Use limits in checker (scan interval + timeout clamp).
- [x] Add `GET /api/v1/monitor/services/limits` to return current limits.
Frontend:
- [x] Fetch limits once and render input descriptions.
- [x] Validate interval/timeout client-side and show inline errors.
- [x] Add `/config` items to edit the `vite_config` keys.
Verification:
- [x] `cd go-backend && go test ./... -count=1`
- [x] `cd vite-frontend && npm run lint && npm run build`
@@ -0,0 +1,224 @@
# 041 - Monitoring Reliability, Realtime, and UX Hardening
## Context
Current monitoring support in FLVX already covers three major areas:
- Node runtime metrics from agent WebSocket telemetry, buffered into `node_metric`, exposed by `/api/v1/monitor/nodes*`, and rendered on `/monitor`.
- Tunnel metrics derived from `/flow/upload`, stored in `tunnel_metric`, exposed by `/api/v1/monitor/tunnels*`, and rendered on `/monitor`.
- Service monitoring for `tcp` and `icmp`, including CRUD, scheduled checks, manual run, history, and non-admin authorization via `monitor_permission`.
The feature set is usable, but the audit found several correctness, reliability, and UX gaps:
- The monitor page is not truly realtime and can lag DB ingestion by tens of seconds.
- Tunnel metrics are only partially implemented and are not aggregated correctly for multi-node tunnels.
- Some monitoring writes fail silently, which can hide data-loss and retention issues.
- Service monitor scheduling is functional but too naive for larger monitor sets and restart scenarios.
- The monitoring UI exposes incomplete semantics, weak freshness cues, and inconsistent permission/error affordances.
- Several monitoring endpoints and edge cases still lack direct automated coverage.
This plan collects all currently known monitoring follow-up work into one implementation document.
## Goals
- Make node monitoring data freshness explicit and reduce stale or misleading chart behavior.
- Make tunnel metrics correct for multi-node tunnels and align schema/query/UI semantics.
- Harden service monitor scheduling, persistence, and cleanup behavior.
- Improve observability so monitoring ingestion and result writes never fail silently.
- Upgrade the monitoring UI so operators can understand status, freshness, scope, and failures at a glance.
- Expand automated coverage for all monitoring APIs and the highest-risk aggregation/scheduler cases.
## Non-goals
- Add a full alerting or notification pipeline.
- Add brand-new monitor protocols beyond the current `tcp` and `icmp` scope.
- Build a large analytics dashboard outside the existing monitoring page structure.
- Introduce frontend test infrastructure for broad component/unit testing unless required by an implementation step.
## Audit Findings To Address
- Node metrics on `/monitor` are DB-polled rather than realtime-streamed.
- Node metrics are buffered for 30s, so charts can lag behind observed node state.
- Tunnel metrics are stored per `(tunnel_id, node_id, timestamp)` but queried and rendered as if they were already tunnel-level aggregates.
- Tunnel metric minute-bucket upsert uses update-then-insert without uniqueness guarantees.
- Tunnel metrics only populate `bytesIn` and `bytesOut`; `connections`, `errors`, and `avgLatencyMs` are placeholder values.
- Node/tunnel/service-monitor writes can fail silently due to ignored errors.
- Service monitor scheduler is serial and uses in-memory `lastRun`, causing restart skew and slow-monitor head-of-line blocking.
- Deleting a service monitor does not clean up related historical results.
- `expectedCode` exists on the model but is not implemented in behavior or UX.
- The monitoring page/menu is exposed before permission is known, leading to avoidable denied-entry UX.
- Service monitor UI does not clearly show latest result freshness, last check time, or whether a displayed row is stale.
- Chart labels and units are not operator-friendly for long time windows and network-heavy views.
- Monitoring API coverage is incomplete for list, permission, limits, latest-results, multi-node tunnel aggregation, and concurrency paths.
## Design
### 1. Node Monitoring Freshness and Realtime Model
- Keep the existing WebSocket node telemetry stream as the source of live state.
- Preserve DB-backed metrics queries for historical charts, but explicitly separate them from live cards/status.
- On `/monitor`, add a lightweight realtime subscription path reusing the existing admin WebSocket feed already used by the node page.
- Use realtime events for:
- node online/offline state,
- a small “latest value” strip or summary above charts,
- freshness timestamp display.
- Keep charts historical and DB-backed by default, but add a visible freshness hint such as:
- `历史图表,最近落库延迟约 0-30s`, or
- `最近入库时间: ...`.
- Do not remove buffered ingestion immediately; first make lag transparent in UI and observable in logs/metrics.
- Optional second-step optimization: reduce flush interval or add a bounded flush-on-latest-view mode if DB pressure remains acceptable.
### 2. Tunnel Metrics Data Model and Query Semantics
- Decide and document one API contract:
- `GET /api/v1/monitor/tunnels/:id/metrics` must return tunnel-level aggregated series for the selected time range, not raw per-node rows.
- Keep storage per `(tunnel_id, node_id, timestamp)` because it is useful for future drill-down.
- Change query behavior so the tunnel metrics endpoint aggregates rows by timestamp across all nodes for the tunnel:
- `SUM(bytes_in)`,
- `SUM(bytes_out)`,
- `SUM(connections)`,
- `SUM(errors)`,
- `AVG` or weighted-average strategy for latency, if latency is later implemented.
- Return a single point per timestamp to the frontend.
- If future node drill-down is needed, add a separate endpoint rather than mixing per-node rows into the current chart API.
### 3. Tunnel Metric Upsert Safety
- Replace the current update-then-insert fallback with a uniqueness-backed upsert strategy.
- Add a unique index on `(tunnel_id, node_id, timestamp)`.
- Implement DB-safe upsert behavior compatible with SQLite and PostgreSQL via GORM clauses or equivalent dialect-safe SQL.
- Preserve additive semantics for traffic counters inside the bucket.
- Add concurrency coverage proving that parallel uploads for the same bucket do not create duplicate rows.
### 4. Tunnel Metric Scope Clarification
- Short term: make the UI and API explicitly traffic-only where the backend only has traffic truth.
- Remove or hide unsupported tunnel metric modes from the current UX until real data exists.
- Do not expose zero-filled placeholders as if they were valid telemetry.
- Keep schema fields if future support is planned, but label them as unimplemented in code comments and avoid rendering them as live features.
### 5. Service Monitor Scheduler Hardening
- Replace the current fully serial best-effort loop with bounded concurrency:
- retain a global scan loop or next-run calculation,
- collect monitors due for execution,
- execute them with a configurable worker limit,
- avoid one slow node/target delaying all others.
- Move scheduling semantics from pure in-memory `lastRun` toward persisted or history-derived next-run safety:
- on restart, do not fire an uncontrolled burst for all monitors if they just ran;
- use latest persisted result timestamp or a persisted scheduler state to calculate due-ness.
- Keep interval clamping behavior aligned with configured limits.
- Continue supporting local panel execution for `tcp` and node execution for `tcp`/`icmp`.
### 6. Service Monitor Data Lifecycle
- Define monitor deletion semantics explicitly:
- either cascade-delete historical `service_monitor_result` rows when a monitor is deleted, or
- retain them intentionally and exclude orphan rows from latest/list endpoints.
- Preferred approach: delete associated results with the monitor so the UI/API model stays simple.
- Either implement `expectedCode` fully or remove it from the model/API surface for now.
- Because service monitoring currently supports only `tcp` and `icmp`, and no HTTP checks are implemented, `expectedCode` should likely be removed from the data model/API until a real HTTP monitor exists.
### 7. Observability and Failure Handling
- Stop swallowing monitoring persistence errors.
- For all node/tunnel/service-monitor writes:
- log structured errors with entity identifiers and operation names,
- increment internal counters if an existing metrics/logging primitive exists,
- keep request/loop behavior resilient, but make failure visible.
- Apply this to:
- node metric batch flush,
- tunnel metric bucket writes,
- scheduled service monitor result writes,
- manual service monitor result writes,
- pruning failures.
- Avoid user-facing hard failures for background ingestion, but surface operational diagnostics in logs.
### 8. Monitoring UI Semantics and Navigation
- Keep `/monitor` accessible only to authenticated users, but improve pre-entry affordances:
- hide or disable the navigation item for users without monitor permission when role/permission data is known,
- or show a locked state with explanation instead of allowing a full denied page transition.
- Preserve the backend permission check as the source of truth.
- On the page itself, upgrade semantics:
- distinguish `enabled/disabled` from `healthy/unhealthy` in the service monitor table,
- show `last checked at`,
- show `latest result` separately from monitor switch state,
- show whether the latest displayed result is stale.
- Add an at-a-glance monitoring summary near the top:
- online/offline node counts,
- monitors healthy/unhealthy/disabled counts,
- latest data freshness text.
### 9. Monitoring UI Readability Improvements
- Improve chart axis labeling for long ranges:
- use date + time formatting for 24h windows,
- keep shorter labels for short ranges.
- Format bytes and rates into human-readable units (`KB/s`, `MB/s`, `GB`) instead of raw integers.
- Expose clear empty states and fetch-error states instead of silent failures.
- Make tunnel charts explicitly say `流量趋势` if only traffic is supported.
- Show `statusCode` in the results modal only if the corresponding monitor type ever uses it; otherwise omit it.
### 10. API and Test Coverage Expansion
- Add contract coverage for:
- `GET /api/v1/monitor/nodes`,
- `GET /api/v1/monitor/tunnels`,
- `GET /api/v1/monitor/services/latest-results`,
- `GET /api/v1/monitor/services/limits`,
- monitor permission list/assign/remove endpoints.
- Add backend tests for:
- multi-node tunnel aggregation returning one point per timestamp,
- tunnel upsert concurrency safety,
- service monitor restart/due scheduling semantics,
- service monitor delete cleanup behavior,
- background write failure logging where practical.
- Keep existing build/test targets green for backend, agent, and frontend.
## Checklist
### Phase 1: Correctness Fixes
- [x] Aggregate `GET /api/v1/monitor/tunnels/:id/metrics` by timestamp across all node rows for the selected tunnel.
- [x] Add a unique index for tunnel metric minute buckets and replace the race-prone update-then-insert flow with safe upsert logic.
- [x] Stop exposing unsupported tunnel metric dimensions (`connections`, `errors`, `latency`) as active UI features while the backend still stores placeholders.
- [x] Define and implement service monitor deletion cleanup so history does not leave orphaned result rows.
- [x] Remove or fully implement `expectedCode`; do not keep dead monitoring fields in the live API/model contract.
### Phase 2: Reliability and Scheduling
- [x] Replace serial service monitor execution with bounded-concurrency execution for due monitors.
- [x] Persist or derive service monitor next-run behavior so process restarts do not trigger uncontrolled immediate reruns.
- [x] Ensure monitor scheduler semantics remain aligned with configured min interval and checker scan cadence.
- [x] Add structured logging for all monitoring persistence failures and prune failures.
- [x] Audit all ignored monitoring write errors and convert them into visible operational diagnostics.
### Phase 3: Realtime and Freshness UX
- [x] Reuse the existing admin WebSocket stream on `/monitor` for live node status and latest-value freshness indicators.
- [x] Add visible chart freshness metadata so users know historical charts are DB-backed and may lag ingestion.
- [x] Decide whether to reduce node metric flush interval after instrumentation confirms acceptable DB impact (decision: keep 30s default for now; revisit after observing DB write rate and UI staleness in production).
- [x] Add a monitoring summary strip showing online nodes, unhealthy monitors, and latest data time.
### Phase 4: Monitoring Page UX Cleanup
- [x] Separate service monitor switch state (`启用/禁用`) from probe health (`成功/失败`).
- [x] Add `last checked at` to the service monitor list and results modal context.
- [x] Mark stale results clearly when the latest result is older than the configured interval budget.
- [x] Format traffic and speed values in human-readable units instead of raw bytes.
- [x] Improve chart time labels for 24h windows to include date context.
- [x] Replace silent frontend fetch failures with explicit inline error or toast handling.
- [x] Rename or relabel tunnel chart UI to make its current scope unambiguous.
### Phase 5: Permission and Navigation UX
- [x] Avoid showing a fully interactive monitor nav entry to users who lack monitoring permission once permission state is known.
- [x] Preserve backend authorization as the final gate and keep denied responses intact.
- [x] Improve denied-state copy so users understand whether they need admin grant vs role change.
### Phase 6: Automated Coverage
- [x] Add contract tests for monitor node list, tunnel list, latest service monitor results, limits, and permission endpoints.
- [x] Add contract or repository tests for multi-node tunnel aggregation correctness.
- [x] Add concurrency tests for tunnel metric upsert safety.
- [x] Add scheduler tests covering restart behavior, due monitor selection, and slow-monitor isolation.
- [x] Keep existing monitoring contract tests passing after all changes.
## Implementation Notes
- Prefer backward-compatible API changes where possible, but favor correctness over preserving misleading tunnel metric semantics.
- Do not introduce a fake realtime chart if the data source remains DB-backed; label it honestly.
- If permission visibility requires an extra frontend capability call, keep it lightweight and cacheable.
- If schema/index changes are introduced, they must remain compatible with both SQLite and PostgreSQL.
## Final Verification Targets
- [x] `GET /api/v1/monitor/tunnels/:id/metrics` returns one aggregated point per timestamp even when multiple nodes report the same tunnel bucket.
- [x] Parallel `/flow/upload` calls for the same tunnel/node/minute do not create duplicate bucket rows.
- [x] `/monitor` clearly distinguishes live state from historical persisted charts and surfaces data freshness to the operator.
- [x] Service monitor list shows enabled state, latest health result, latest check time, and stale-state semantics correctly.
- [x] Deleting a service monitor no longer leaves dangling historical data in list-facing APIs.
- [x] Monitoring ingestion/result write failures are visible in logs and no longer fail silently.
- [x] Non-admin users without monitoring permission do not get a confusing monitor-entry experience, while granted users continue to access monitoring successfully.
- [x] Backend monitoring contract tests pass.
- [x] New repository/scheduler tests pass.
- [x] `cd go-backend && go test ./... -count=1` passes.
- [x] `cd go-gost/x && go test ./socket/... -count=1` passes.
- [x] `cd vite-frontend && npm run build` passes.
+10
View File
@@ -0,0 +1,10 @@
# PLAN: Release 2.1.8
## Overview
Synchronize with the main branch and release a new tag `2.1.8`.
## Tasks
- [x] Pull latest changes from `main`
- [x] Update `AGENTS.md` with new tag and current commit hash (PR #337 created and set to auto-merge)
- [x] Create git tag `2.1.8`
- [x] Push git tag `2.1.8` to origin
+9
View File
@@ -4,6 +4,7 @@ import { useEffect } from "react";
import IndexPage from "@/pages/index";
import ChangePasswordPage from "@/pages/change-password";
import DashboardPage from "@/pages/dashboard";
import MonitorPage from "@/pages/monitor";
import ForwardPage from "@/pages/forward";
import TunnelPage from "@/pages/tunnel";
import NodePage from "@/pages/node";
@@ -122,6 +123,14 @@ function App() {
}
path="/dashboard"
/>
<Route
element={
<ProtectedRoute>
<MonitorPage />
</ProtectedRoute>
}
path="/monitor"
/>
<Route
element={
<ProtectedRoute>
+92
View File
@@ -30,6 +30,16 @@ import type {
SpeedLimitMutationPayload,
UpdatePasswordPayload,
BackupImportPayload,
NodeMetricApiItem,
TunnelMetricApiItem,
ServiceMonitorApiItem,
ServiceMonitorResultApiItem,
ServiceMonitorLimitsApiData,
ServiceMonitorMutationPayload,
MonitorNodeApiItem,
MonitorTunnelApiItem,
MonitorPermissionApiItem,
MonitorAccessApiData,
} from "./types";
import axios from "axios";
@@ -402,3 +412,85 @@ export const getAnnouncement = () =>
Network.get<AnnouncementData>("/announcement/get");
export const updateAnnouncement = (data: AnnouncementData) =>
Network.post("/announcement/update", data);
export const getNodeMetrics = (
nodeId: number,
start?: number,
end?: number,
) => {
const params: Record<string, string> = {};
if (start) params.start = String(start);
if (end) params.end = String(end);
return Network.get<NodeMetricApiItem[]>(
`/monitor/nodes/${nodeId}/metrics`,
params,
);
};
export const getNodeMetricsLatest = (nodeId: number) =>
Network.get<NodeMetricApiItem>(`/monitor/nodes/${nodeId}/metrics/latest`);
export const getTunnelMetrics = (
tunnelId: number,
start?: number,
end?: number,
) => {
const params: Record<string, string> = {};
if (start) params.start = String(start);
if (end) params.end = String(end);
return Network.get<TunnelMetricApiItem[]>(
`/monitor/tunnels/${tunnelId}/metrics`,
params,
);
};
export const getMonitorTunnels = () =>
Network.get<MonitorTunnelApiItem[]>("/monitor/tunnels");
export const getServiceMonitorList = () =>
Network.get<ServiceMonitorApiItem[]>("/monitor/services");
export const getServiceMonitorLimits = () =>
Network.get<ServiceMonitorLimitsApiData>("/monitor/services/limits");
export const createServiceMonitor = (data: ServiceMonitorMutationPayload) =>
Network.post<ServiceMonitorApiItem>("/monitor/services/create", data);
export const updateServiceMonitor = (data: ServiceMonitorMutationPayload) =>
Network.post<ServiceMonitorApiItem>("/monitor/services/update", data);
export const deleteServiceMonitor = (id: number) =>
Network.post("/monitor/services/delete", { id });
export const getServiceMonitorResults = (monitorId: number, limit = 100) =>
Network.get<ServiceMonitorResultApiItem[]>(
`/monitor/services/${monitorId}/results`,
{ limit: String(limit) },
);
export const getServiceMonitorLatestResults = () =>
Network.get<ServiceMonitorResultApiItem[]>(
"/monitor/services/latest-results",
);
export const runServiceMonitor = (id: number) =>
Network.post<ServiceMonitorResultApiItem>("/monitor/services/run", { id });
export const getMonitorNodes = () =>
Network.get<MonitorNodeApiItem[]>("/monitor/nodes");
export const getMonitorAccess = () =>
Network.get<MonitorAccessApiData>("/monitor/access");
export const getMonitorPermissionList = () =>
Network.get<MonitorPermissionApiItem[]>("/monitor/permission/list");
export const assignMonitorPermission = (userId: number) =>
Network.post("/monitor/permission/assign", { userId });
export const removeMonitorPermission = (userId: number) =>
Network.post("/monitor/permission/remove", { userId });
+102
View File
@@ -385,3 +385,105 @@ export interface BackupImportPayload {
types: string[];
[key: string]: unknown;
}
export interface NodeMetricApiItem {
id: number;
nodeId: number;
timestamp: number;
cpuUsage: number;
memoryUsage: number;
diskUsage: number;
netInBytes: number;
netOutBytes: number;
netInSpeed: number;
netOutSpeed: number;
load1: number;
load5: number;
load15: number;
tcpConns: number;
udpConns: number;
uptime: number;
}
export interface TunnelMetricApiItem {
id: number;
tunnelId: number;
nodeId: number;
timestamp: number;
bytesIn: number;
bytesOut: number;
connections: number;
errors: number;
avgLatencyMs: number;
}
export interface ServiceMonitorApiItem {
id: number;
name: string;
// Keep as string for forward-compatibility.
type: string;
target: string;
intervalSec: number;
timeoutSec: number;
nodeId: number;
enabled: number;
createdTime: number;
updatedTime: number;
}
export interface ServiceMonitorResultApiItem {
id: number;
monitorId: number;
timestamp: number;
success: number;
latencyMs: number;
statusCode: number;
errorMessage: string;
}
export interface ServiceMonitorMutationPayload {
id?: number;
name: string;
type: "tcp" | "icmp";
target: string;
intervalSec?: number;
timeoutSec?: number;
nodeId?: number;
enabled?: number;
}
export interface ServiceMonitorLimitsApiData {
checkerScanIntervalSec: number;
minIntervalSec: number;
defaultIntervalSec: number;
minTimeoutSec: number;
defaultTimeoutSec: number;
maxTimeoutSec: number;
}
export interface MonitorNodeApiItem {
id: number;
inx: number;
name: string;
status: number;
updatedTime: number;
}
export interface MonitorTunnelApiItem {
id: number;
inx: number;
name: string;
status: number;
updatedTime: number;
}
export interface MonitorPermissionApiItem {
id: number;
userId: number;
createdTime: number;
}
export interface MonitorAccessApiData {
allowed: boolean;
reason?: string;
}
+84 -3
View File
@@ -21,7 +21,7 @@ import {
import { Input } from "@/shadcn-bridge/heroui/input";
import { BrandLogo } from "@/components/brand-logo";
import { VersionFooter } from "@/components/version-footer";
import { updatePassword } from "@/api";
import { getMonitorAccess, updatePassword } from "@/api";
import { safeLogout } from "@/utils/logout";
import { siteConfig } from "@/config/site";
import { useMobileBreakpoint } from "@/hooks/useMobileBreakpoint";
@@ -56,6 +56,10 @@ export default function AdminLayout({
);
const [username, setUsername] = useState("");
const [isAdmin, setIsAdmin] = useState(false);
const [monitorAllowed, setMonitorAllowed] = useState<boolean | null>(null);
const [monitorAccessReason, setMonitorAccessReason] = useState<string | null>(
null,
);
const [passwordLoading, setPasswordLoading] = useState(false);
const [passwordForm, setPasswordForm] = useState<PasswordForm>({
newUsername: "",
@@ -117,6 +121,19 @@ export default function AdminLayout({
),
adminOnly: true,
},
{
path: "/monitor",
label: "监控",
icon: (
<svg className="w-5 h-5" fill="currentColor" viewBox="0 0 20 20">
<path
clipRule="evenodd"
d="M3 3a1 1 0 000 2v11a1 1 0 001 1h13a1 1 0 100-2H5V5a1 1 0 00-1-1H3zm13.707 4.293a1 1 0 00-1.414 0L12 10.586 10.707 9.293a1 1 0 00-1.414 0L7 11.586l-1.293-1.293a1 1 0 10-1.414 1.414l2 2a1 1 0 001.414 0L10 11.414l1.293 1.293a1 1 0 001.414 0l3-3a1 1 0 000-1.414z"
fillRule="evenodd"
/>
</svg>
),
},
{
path: "/limit",
label: "限速",
@@ -184,6 +201,41 @@ export default function AdminLayout({
setUsername(name);
setIsAdmin(adminFlag);
// Monitor permission is not strictly role-based; non-admin users may be
// granted access explicitly. Fetch a lightweight capability flag so we can
// avoid a confusing 403 navigation.
if (adminFlag) {
setMonitorAllowed(true);
setMonitorAccessReason(null);
return;
}
let cancelled = false;
(async () => {
try {
const res = await getMonitorAccess();
if (cancelled) return;
if (res.code === 0 && res.data) {
setMonitorAllowed(Boolean(res.data.allowed));
setMonitorAccessReason(
res.data.allowed ? null : (res.data.reason || null),
);
return;
}
// Fail open to preserve legacy navigation behavior.
setMonitorAllowed(true);
setMonitorAccessReason(null);
} catch {
if (cancelled) return;
setMonitorAllowed(true);
setMonitorAccessReason(null);
}
})();
return () => {
cancelled = true;
};
}, []);
useEffect(() => {
@@ -218,6 +270,23 @@ export default function AdminLayout({
// 菜单点击处理
const handleMenuClick = (path: string) => {
if (path === "/monitor" && monitorAllowed !== true) {
if (monitorAllowed == null) {
toast("正在检查监控权限,请稍后重试");
return;
}
const hint =
monitorAccessReason === "need_admin_grant"
? "暂无监控权限,请联系管理员在用户页面授予监控权限"
: "暂无监控权限,请联系管理员授权";
toast.error(hint);
return;
}
navigate(path);
if (isMobile) {
hideMobileMenu();
@@ -346,6 +415,8 @@ export default function AdminLayout({
<ul className="space-y-1">
{filteredMenuItems.map((item) => {
const isActive = location.pathname === item.path;
const isMonitor = item.path === "/monitor";
const isMonitorBlocked = isMonitor && monitorAllowed !== true;
return (
<li key={item.path}>
@@ -353,13 +424,23 @@ export default function AdminLayout({
className={`
w-full flex items-center p-2 rounded-lg text-left
relative min-h-[44px] overflow-hidden transition-colors
${isMonitorBlocked ? "opacity-60" : ""}
${
isActive
? "text-primary-600 dark:text-primary-300"
: "text-gray-700 dark:text-gray-200"
: isMonitorBlocked
? "text-gray-500 dark:text-gray-400"
: "text-gray-700 dark:text-gray-200"
}
`}
title={isCollapsed ? item.label : undefined}
aria-disabled={isMonitorBlocked}
title={
isCollapsed
? isMonitorBlocked
? `${item.label} (无权限)`
: item.label
: undefined
}
transition={{ duration: 0.15 }}
onClick={() => handleMenuClick(item.path)}
>
+75 -2
View File
@@ -1,8 +1,10 @@
import React, { useState, useEffect } from "react";
import { useNavigate, useLocation } from "react-router-dom";
import toast from "react-hot-toast";
import { BrandLogo } from "@/components/brand-logo";
import { siteConfig } from "@/config/site";
import { getMonitorAccess } from "@/api";
import { getAdminFlag } from "@/utils/session";
import { useScrollTopOnPathChange } from "@/hooks/useScrollTopOnPathChange";
@@ -17,6 +19,10 @@ export default function H5Layout({ children }: { children: React.ReactNode }) {
const navigate = useNavigate();
const location = useLocation();
const [isAdmin, setIsAdmin] = useState(false);
const [monitorAllowed, setMonitorAllowed] = useState<boolean | null>(null);
const [monitorAccessReason, setMonitorAccessReason] = useState<string | null>(
null,
);
useScrollTopOnPathChange();
@@ -72,6 +78,19 @@ export default function H5Layout({ children }: { children: React.ReactNode }) {
),
adminOnly: true,
},
{
path: "/monitor",
label: "监控",
icon: (
<svg className="w-6 h-6" fill="currentColor" viewBox="0 0 20 20">
<path
clipRule="evenodd"
d="M3 3a1 1 0 000 2v11a1 1 0 001 1h13a1 1 0 100-2H5V5a1 1 0 00-1-1H3zm13.707 4.293a1 1 0 00-1.414 0L12 10.586 10.707 9.293a1 1 0 00-1.414 0L7 11.586l-1.293-1.293a1 1 0 10-1.414 1.414l2 2a1 1 0 001.414 0L10 11.414l1.293 1.293a1 1 0 001.414 0l3-3a1 1 0 000-1.414z"
fillRule="evenodd"
/>
</svg>
),
},
{
path: "/profile",
label: "我的",
@@ -84,11 +103,60 @@ export default function H5Layout({ children }: { children: React.ReactNode }) {
];
useEffect(() => {
setIsAdmin(getAdminFlag());
const adminFlag = getAdminFlag();
setIsAdmin(adminFlag);
if (adminFlag) {
setMonitorAllowed(true);
setMonitorAccessReason(null);
return;
}
let cancelled = false;
(async () => {
try {
const res = await getMonitorAccess();
if (cancelled) return;
if (res.code === 0 && res.data) {
setMonitorAllowed(Boolean(res.data.allowed));
setMonitorAccessReason(
res.data.allowed ? null : (res.data.reason || null),
);
return;
}
setMonitorAllowed(true);
setMonitorAccessReason(null);
} catch {
if (cancelled) return;
setMonitorAllowed(true);
setMonitorAccessReason(null);
}
})();
return () => {
cancelled = true;
};
}, []);
// Tab点击处理
const handleTabClick = (path: string) => {
if (path === "/monitor" && monitorAllowed !== true) {
if (monitorAllowed == null) {
toast("正在检查监控权限,请稍后重试");
return;
}
const hint =
monitorAccessReason === "need_admin_grant"
? "暂无监控权限,请联系管理员授权"
: "暂无监控权限";
toast.error(hint);
return;
}
navigate(path);
};
@@ -121,6 +189,8 @@ export default function H5Layout({ children }: { children: React.ReactNode }) {
<nav className="bg-white dark:bg-black border-t border-gray-200 dark:border-gray-600 h-[calc(4rem+var(--safe-area-bottom))] flex-shrink-0 flex items-center justify-around px-2 fixed bottom-0 left-0 right-0 z-30">
{filteredTabItems.map((item) => {
const isActive = location.pathname === item.path;
const isMonitor = item.path === "/monitor";
const isMonitorBlocked = isMonitor && monitorAllowed !== true;
return (
<button
@@ -128,10 +198,13 @@ export default function H5Layout({ children }: { children: React.ReactNode }) {
className={`
flex flex-col items-center justify-center flex-1 h-full pb-[var(--safe-area-bottom)]
transition-colors duration-200 min-h-[44px]
${isMonitorBlocked ? "opacity-60" : ""}
${
isActive
? "text-primary-600 dark:text-primary-400"
: "text-gray-500 dark:text-gray-400 hover:text-gray-700 dark:hover:text-gray-200"
: isMonitorBlocked
? "text-gray-500 dark:text-gray-400"
: "text-gray-500 dark:text-gray-400 hover:text-gray-700 dark:hover:text-gray-200"
}
`}
onClick={() => handleTabClick(item.path)}
+111
View File
@@ -0,0 +1,111 @@
import type { MonitorNodeApiItem } from "@/api/types";
import { useCallback, useEffect, useMemo, useState } from "react";
import toast from "react-hot-toast";
import { RefreshCw } from "lucide-react";
import { AnimatedPage } from "@/components/animated-page";
import { Button } from "@/shadcn-bridge/heroui/button";
import { Card, CardBody, CardHeader } from "@/shadcn-bridge/heroui/card";
import { getMonitorNodes } from "@/api";
import { MonitorView } from "@/pages/node/monitor-view";
type MonitorNode = {
id: number;
name: string;
connectionStatus: "online" | "offline";
};
export default function MonitorPage() {
const [nodes, setNodes] = useState<MonitorNodeApiItem[]>([]);
const [nodesLoading, setNodesLoading] = useState(false);
const [nodesError, setNodesError] = useState<string | null>(null);
const loadNodes = useCallback(async () => {
setNodesLoading(true);
try {
const response = await getMonitorNodes();
if (response.code === 0 && Array.isArray(response.data)) {
setNodesError(null);
setNodes(response.data);
return;
}
if (response.code === 403) {
setNodes([]);
setNodesError(response.msg || "暂无监控权限,请联系管理员授权");
return;
}
toast.error(response.msg || "加载节点失败");
} catch {
toast.error("加载节点失败");
} finally {
setNodesLoading(false);
}
}, []);
useEffect(() => {
void loadNodes();
}, [loadNodes]);
useEffect(() => {
const timer = window.setInterval(() => {
void loadNodes();
}, 30_000);
return () => window.clearInterval(timer);
}, [loadNodes]);
const nodeMap = useMemo(() => {
const list: MonitorNode[] = nodes
.filter((n) => Number(n.id) > 0)
.map((n) => ({
id: Number(n.id),
name: String(n.name ?? ""),
connectionStatus: n.status === 1 ? "online" : "offline",
}));
return new Map<number, MonitorNode>(list.map((n) => [n.id, n]));
}, [nodes]);
return (
<AnimatedPage className="px-3 lg:px-6 py-8">
<div className="mb-6 space-y-3">
<div className="flex items-center justify-between gap-3">
<div className="min-w-0">
<h2 className="text-xl font-semibold truncate">监控</h2>
<div className="text-xs text-default-500 truncate">
实时节点状态 + 历史指标图表 + 隧道流量 + 服务监控(TCP/ICMP)
</div>
</div>
<Button
isLoading={nodesLoading}
size="sm"
variant="flat"
onPress={loadNodes}
>
<RefreshCw className="w-4 h-4 mr-1" />
刷新节点
</Button>
</div>
{nodesError ? (
<Card>
<CardHeader>
<h3 className="text-sm font-semibold">节点列表</h3>
</CardHeader>
<CardBody>
<div className="text-sm text-default-600">{nodesError}</div>
</CardBody>
</Card>
) : null}
</div>
<MonitorView nodeMap={nodeMap} />
</AnimatedPage>
);
}
+154 -11
View File
@@ -67,7 +67,10 @@ import {
getNodeRenewalCycleLabel,
type NodeRenewalCycle,
} from "@/pages/node/renewal";
import { buildNodeSystemInfo } from "@/pages/node/system-info";
import {
buildNodeSystemInfo,
type NodeSystemInfo,
} from "@/pages/node/system-info";
import { useNodeOfflineTimers } from "@/pages/node/use-node-offline-timers";
import { useNodeRealtime } from "@/pages/node/use-node-realtime";
import { useLocalStorageState } from "@/hooks/use-local-storage-state";
@@ -100,15 +103,7 @@ interface Node {
remoteUrl?: string;
syncError?: string;
connectionStatus: "online" | "offline";
systemInfo?: {
cpuUsage: number;
memoryUsage: number;
uploadTraffic: number;
downloadTraffic: number;
uploadSpeed: number;
downloadSpeed: number;
uptime: number;
} | null;
systemInfo?: NodeSystemInfo | null;
copyLoading?: boolean;
upgradeLoading?: boolean;
rollbackLoading?: boolean;
@@ -297,6 +292,13 @@ export default function NodePage() {
"node-active-tab",
"local",
);
// Backward-compat: older versions stored extra tab values.
useEffect(() => {
if (activeTab !== "local" && activeTab !== "remote") {
setActiveTab("local");
}
}, [activeTab, setActiveTab]);
const [remoteUsageMap, setRemoteUsageMap] = useState<
Record<number, RemoteUsageNode>
>({});
@@ -581,6 +583,43 @@ export default function NodePage() {
} catch {
// ignore parse errors
}
} else if (type === "metric") {
clearOfflineTimer(nodeId);
setNodeList((prev) =>
prev.map((node) => {
if (node.id !== nodeId) return node;
const metric =
typeof messageData === "string"
? JSON.parse(messageData)
: messageData;
if (!metric || typeof metric !== "object") return node;
return {
...node,
connectionStatus: "online",
systemInfo: {
cpuUsage: metric.cpuUsage ?? metric.cpu_usage ?? 0,
memoryUsage: metric.memoryUsage ?? metric.memory_usage ?? 0,
uploadTraffic:
metric.netOutBytes ?? metric.bytes_transmitted ?? 0,
downloadTraffic: metric.netInBytes ?? metric.bytes_received ?? 0,
uploadSpeed: metric.netOutSpeed ?? metric.net_out_speed ?? 0,
downloadSpeed: metric.netInSpeed ?? metric.net_in_speed ?? 0,
uptime: metric.uptime ?? 0,
diskUsage: metric.diskUsage ?? metric.disk_usage,
load1: metric.load1,
load5: metric.load5,
load15: metric.load15,
tcpConns: metric.tcpConns ?? metric.tcp_conns,
udpConns: metric.udpConns ?? metric.udp_conns,
netInSpeed: metric.netInSpeed ?? metric.net_in_speed,
netOutSpeed: metric.netOutSpeed ?? metric.net_out_speed,
},
};
}),
);
}
};
@@ -2244,6 +2283,111 @@ export default function NodePage() {
</div>
</div>
</div>
{/* 扩展指标:磁盘/负载/连接 */}
{node.connectionStatus === "online" &&
node.systemInfo && (
<div className="space-y-2">
{node.systemInfo.diskUsage !==
undefined && (
<div>
<div className="flex justify-between text-xs mb-1">
<span>磁盘</span>
<span className="font-mono">
{node.systemInfo.diskUsage.toFixed(
1,
)}
%
</span>
</div>
<Progress
aria-label="磁盘使用率"
color={getProgressColor(
node.systemInfo.diskUsage,
false,
)}
size="sm"
value={node.systemInfo.diskUsage}
/>
</div>
)}
{(node.systemInfo.load1 !== undefined ||
node.systemInfo.load5 !== undefined ||
node.systemInfo.load15 !==
undefined) && (
<div className="grid grid-cols-3 gap-1 text-xs">
{node.systemInfo.load1 !==
undefined && (
<div className="text-center p-1.5 bg-default-50 dark:bg-default-100 rounded">
<div className="text-default-500 text-[10px]">
负载 1m
</div>
<div className="font-mono">
{node.systemInfo.load1.toFixed(
2,
)}
</div>
</div>
)}
{node.systemInfo.load5 !==
undefined && (
<div className="text-center p-1.5 bg-default-50 dark:bg-default-100 rounded">
<div className="text-default-500 text-[10px]">
负载 5m
</div>
<div className="font-mono">
{node.systemInfo.load5.toFixed(
2,
)}
</div>
</div>
)}
{node.systemInfo.load15 !==
undefined && (
<div className="text-center p-1.5 bg-default-50 dark:bg-default-100 rounded">
<div className="text-default-500 text-[10px]">
负载 15m
</div>
<div className="font-mono">
{node.systemInfo.load15.toFixed(
2,
)}
</div>
</div>
)}
</div>
)}
{(node.systemInfo.tcpConns !==
undefined ||
node.systemInfo.udpConns !==
undefined) && (
<div className="grid grid-cols-2 gap-2 text-xs">
{node.systemInfo.tcpConns !==
undefined && (
<div className="text-center p-1.5 bg-default-50 dark:bg-default-100 rounded">
<div className="text-default-500 text-[10px]">
TCP 连接
</div>
<div className="font-mono">
{node.systemInfo.tcpConns}
</div>
</div>
)}
{node.systemInfo.udpConns !==
undefined && (
<div className="text-center p-1.5 bg-default-50 dark:bg-default-100 rounded">
<div className="text-default-500 text-[10px]">
UDP 连接
</div>
<div className="font-mono">
{node.systemInfo.udpConns}
</div>
</div>
)}
</div>
)}
</div>
)}
</div>
</>
)}
@@ -2329,7 +2473,6 @@ export default function NodePage() {
</SortableContext>
</DndContext>
)}
{/* 新增/编辑节点对话框 */}
<Modal
backdrop="blur"
File diff suppressed because it is too large Load Diff
@@ -6,6 +6,14 @@ export interface NodeSystemInfo {
uploadSpeed: number;
downloadSpeed: number;
uptime: number;
diskUsage?: number;
load1?: number;
load5?: number;
load15?: number;
tcpConns?: number;
udpConns?: number;
netInSpeed?: number;
netOutSpeed?: number;
}
type RawSystemInfo = Record<string, string | number | undefined>;
@@ -82,5 +90,13 @@ export const buildNodeSystemInfo = (
uploadSpeed,
downloadSpeed,
uptime,
diskUsage: toFloat(raw.disk_usage),
load1: toFloat(raw.load1),
load5: toFloat(raw.load5),
load15: toFloat(raw.load15),
tcpConns: toInteger(raw.tcp_conns),
udpConns: toInteger(raw.udp_conns),
netInSpeed: toInteger(raw.net_in_speed),
netOutSpeed: toInteger(raw.net_out_speed),
};
};
+151 -2
View File
@@ -30,6 +30,7 @@ import { Chip } from "@/shadcn-bridge/heroui/chip";
import { Select, SelectItem } from "@/shadcn-bridge/heroui/select";
import { RadioGroup, Radio } from "@/shadcn-bridge/heroui/radio";
import { Checkbox } from "@/shadcn-bridge/heroui/checkbox";
import { Switch } from "@/shadcn-bridge/heroui/switch";
import { DatePicker } from "@/shadcn-bridge/heroui/date-picker";
import { Spinner } from "@/shadcn-bridge/heroui/spinner";
import { Progress } from "@/shadcn-bridge/heroui/progress";
@@ -58,6 +59,9 @@ import {
resetUserQuota,
getUserGroupList,
getUserGroups,
getMonitorPermissionList,
assignMonitorPermission,
removeMonitorPermission,
} from "@/api";
import {
EditIcon,
@@ -229,6 +233,13 @@ export default function UserPage() {
onClose: onTunnelModalClose,
} = useDisclosure();
const [currentUser, setCurrentUser] = useState<User | null>(null);
const [monitorPermissionUserIds, setMonitorPermissionUserIds] = useState<
Set<number>
>(new Set());
const [monitorPermissionLoading, setMonitorPermissionLoading] =
useState(false);
const [monitorPermissionMutatingUserId, setMonitorPermissionMutatingUserId] =
useState<number | null>(null);
const [userTunnels, setUserTunnels] = useState<UserTunnel[]>([]);
const [tunnelListLoading, setTunnelListLoading] = useState(false);
@@ -396,6 +407,32 @@ export default function UserPage() {
} catch {}
}, []);
const loadMonitorPermissions = useCallback(async () => {
setMonitorPermissionLoading(true);
try {
const response = await getMonitorPermissionList();
if (response.code === 0) {
const ids = new Set<number>();
if (Array.isArray(response.data)) {
response.data.forEach((item: any) => {
const id = Number(item?.userId ?? 0);
if (id > 0) ids.add(id);
});
}
setMonitorPermissionUserIds(ids);
} else if (response.code !== 403) {
toast.error(response.msg || "获取监控权限失败");
}
} catch {
// ignore
} finally {
setMonitorPermissionLoading(false);
}
}, []);
const loadUserTunnels = useCallback(async (userId: number) => {
setTunnelListLoading(true);
try {
@@ -422,7 +459,8 @@ export default function UserPage() {
void loadTunnels();
void loadSpeedLimits();
void loadUserGroups();
}, [loadSpeedLimits, loadTunnels, loadUserGroups]);
void loadMonitorPermissions();
}, [loadMonitorPermissions, loadSpeedLimits, loadTunnels, loadUserGroups]);
useEffect(() => {
void loadUsers();
@@ -598,6 +636,63 @@ export default function UserPage() {
}
};
const setUserMonitorPermission = useCallback(
async (userId: number, enabled: boolean) => {
if (userId <= 0) return;
if (monitorPermissionMutatingUserId === userId) return;
const prevEnabled = monitorPermissionUserIds.has(userId);
if (prevEnabled === enabled) return;
setMonitorPermissionMutatingUserId(userId);
// Optimistic update for better UX.
setMonitorPermissionUserIds((prev) => {
const next = new Set(prev);
if (enabled) {
next.add(userId);
} else {
next.delete(userId);
}
return next;
});
try {
const response = enabled
? await assignMonitorPermission(userId)
: await removeMonitorPermission(userId);
if (response.code === 0) {
toast.success(enabled ? "已授权监控" : "已撤销监控");
return;
}
toast.error(response.msg || "操作失败");
throw new Error("mutation failed");
} catch {
// Revert optimistic update on failure.
setMonitorPermissionUserIds((prev) => {
const next = new Set(prev);
if (prevEnabled) {
next.add(userId);
} else {
next.delete(userId);
}
return next;
});
} finally {
setMonitorPermissionMutatingUserId(null);
}
},
[monitorPermissionMutatingUserId, monitorPermissionUserIds],
);
// 隧道权限管理操作
const handleManageTunnels = (user: User) => {
setCurrentUser(user);
@@ -1050,6 +1145,26 @@ export default function UserPage() {
{/* 其他信息 */}
<div className="space-y-1.5 pt-2 border-t border-divider">
{(user.dailyQuotaGB ?? 0) > 0 ||
(user.monthlyQuotaGB ?? 0) > 0 ||
(user.disabledByQuota ?? 0) > 0 ? (
<>
<div className="flex justify-between text-sm">
<span className="text-default-600">每日配额</span>
<span className="font-medium text-xs">
{formatFlow(Number(user.dailyUsedBytes ?? 0))} /{" "}
{formatQuotaLimit(user.dailyQuotaGB)}
</span>
</div>
<div className="flex justify-between text-sm">
<span className="text-default-600">每月配额</span>
<span className="font-medium text-xs">
{formatFlow(Number(user.monthlyUsedBytes ?? 0))}{" "}
/ {formatQuotaLimit(user.monthlyQuotaGB)}
</span>
</div>
</>
) : null}
<div className="flex justify-between text-sm">
<span className="text-default-600">规则数量</span>
<span className="font-medium text-xs">
@@ -1414,9 +1529,43 @@ export default function UserPage() {
onClose={onTunnelModalClose}
>
<ModalContent>
<ModalHeader>用户 {currentUser?.user} 的隧道权限管理</ModalHeader>
<ModalHeader>用户 {currentUser?.user} 的权限管理</ModalHeader>
<ModalBody>
<div className="space-y-6">
{/* 监控权限部分 */}
<div>
<h3 className="text-lg font-semibold mb-4">监控权限</h3>
<div className="flex items-center justify-between gap-4 bg-default-100 dark:bg-default-50 p-4 rounded-lg border border-default-200 dark:border-default-100/30">
<div className="min-w-0">
<div className="text-sm font-medium text-foreground">
允许访问监控功能
</div>
<div className="text-xs text-default-500 mt-1">
授予后,该用户可以访问监控页面并管理服务监控(TCP/ICMP)。
</div>
</div>
<div className="flex items-center gap-2 shrink-0">
{monitorPermissionLoading ? <Spinner size="sm" /> : null}
<Switch
isDisabled={
!currentUser ||
monitorPermissionLoading ||
monitorPermissionMutatingUserId === currentUser.id
}
isSelected={
currentUser
? monitorPermissionUserIds.has(currentUser.id)
: false
}
onValueChange={(v) =>
currentUser &&
void setUserMonitorPermission(currentUser.id, v)
}
/>
</div>
</div>
</div>
{/* 分配新权限部分 */}
<div>
<h3 className="text-lg font-semibold mb-4">分配新权限</h3>