mirror of
https://github.com/Rain-kl/OpenFlare.git
synced 2026-09-28 13:46:38 +08:00
Compare commits
32 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 13c5073bf8 | |||
| da1dd92404 | |||
| 4b11279662 | |||
| bbadcca294 | |||
| 44ce6497a1 | |||
| 4b83f91b31 | |||
| 9d2fac5d4c | |||
| b4b93ff4ed | |||
| 160e63558f | |||
| 9b3555c569 | |||
| b928928958 | |||
| b312460ddf | |||
| 50f7257d93 | |||
| 336185f01c | |||
| 44bba0f19a | |||
| f0eca028f9 | |||
| 58624db397 | |||
| 38946d1af5 | |||
| caf2ffcff4 | |||
| 0e86fe3547 | |||
| 6525bef15d | |||
| 3e910f1961 | |||
| 28c14eb054 | |||
| ae618905a3 | |||
| 5ad151469c | |||
| 6467b32d8e | |||
| cf72420815 | |||
| 34225cb88a | |||
| 389f02b6b0 | |||
| 2fcbb945fb | |||
| 23501259b2 | |||
| c561e65cd3 |
+21
-7
@@ -1,7 +1,8 @@
|
||||
# ──────────────────────────────────────────────────────────────────────────────
|
||||
# openflare — 环境变量配置模板
|
||||
# 复制此文件为 .env 并填入实际值: cp .env.example .env
|
||||
# 环境变量优先级高于 config.yaml / config.docker.yaml
|
||||
# 环境变量优先级高于 config.yaml
|
||||
# docker compose 会读取本文件(env_file: .env)并替换 compose 中的 ${VAR}
|
||||
# ──────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
# ─── 时区 ─────────────────────────────────────────────────────────────────────
|
||||
@@ -22,15 +23,16 @@ APP_SESSION_HTTP_ONLY=true
|
||||
# HTTPS 部署时设为 true,HTTP 环境必须为 false
|
||||
APP_SESSION_SECURE=true
|
||||
|
||||
# ─── 数据库 ────────────────────────────────────────────────────────────────────
|
||||
# ─── 数据库(PostgreSQL)──────────────────────────────────────────────────────
|
||||
# 设置 DB_HOST 后自动启用 PostgreSQL,也可通过 DB_ENABLED 显式控制
|
||||
# DB_ENABLED=false 时使用 SQLite 作为后备数据库
|
||||
DB_ENABLED=true
|
||||
# SQLITE_PATH=./data/openflare.db
|
||||
# compose 内应用连服务名;本机直连 Docker 映射端口时用 127.0.0.1
|
||||
DB_HOST=postgres
|
||||
DB_PORT=5432
|
||||
DB_USERNAME=postgres
|
||||
DB_PASSWORD=postgres
|
||||
DB_USERNAME=openflare
|
||||
DB_PASSWORD=replace-with-strong-password
|
||||
DB_NAME=openflare
|
||||
DB_SSL_MODE=disable
|
||||
DB_TIMEZONE=Asia/Shanghai
|
||||
@@ -38,7 +40,7 @@ DB_TIMEZONE=Asia/Shanghai
|
||||
# DB_MAX_IDLE_CONN=16
|
||||
# DB_MAX_OPEN_CONN=128
|
||||
|
||||
# ─── Redis ─────────────────────────────────────────────────────────────────────
|
||||
# ─── Redis / Valkey ────────────────────────────────────────────────────────────
|
||||
# 设置 REDIS_ADDR 后自动启用,也可通过 REDIS_ENABLED 显式控制
|
||||
REDIS_ENABLED=true
|
||||
REDIS_ADDR=redis:6379
|
||||
@@ -47,13 +49,20 @@ REDIS_ADDR=redis:6379
|
||||
# REDIS_DB=0
|
||||
REDIS_KEY_PREFIX=openflare:
|
||||
# REDIS_POOL_SIZE=100
|
||||
# compose 宿主机映射端口(仅 docker-compose 使用)
|
||||
# REDIS_PORT=6379
|
||||
|
||||
# ─── ClickHouse(必需)────────────────────────────────────────────────────
|
||||
# ─── ClickHouse(必需)────────────────────────────────────────────────────────
|
||||
# CLICKHOUSE_HOST 设置后会自动启用;测试环境可显式 CLICKHOUSE_ENABLED=true 做 live 联调
|
||||
CLICKHOUSE_ENABLED=true
|
||||
# compose 内:clickhouse:9000;本机连映射端口:127.0.0.1:9000
|
||||
CLICKHOUSE_HOST=clickhouse:9000
|
||||
CLICKHOUSE_USERNAME=default
|
||||
CLICKHOUSE_PASSWORD=123456
|
||||
# 须与 compose clickhouse 服务密码一致(首次初始化后改密码需清 data/clickhouse_data)
|
||||
CLICKHOUSE_PASSWORD=replace-with-clickhouse-password
|
||||
CLICKHOUSE_NAME=openflare
|
||||
|
||||
|
||||
# ─── 日志 ──────────────────────────────────────────────────────────────────────
|
||||
LOG_LEVEL=info
|
||||
LOG_FORMAT=console
|
||||
@@ -67,6 +76,11 @@ OTEL_EXPORTER_OTLP_INSECURE=true
|
||||
OTEL_SAMPLING_RATE=0.0
|
||||
# 全局 Tracer 命名空间,默认为 github.com/Rain-kl/OpenFlare
|
||||
# OTEL_TRACER_NAME=github.com/Rain-kl/OpenFlare
|
||||
# compose 可选端口覆盖
|
||||
# JAEGER_VERSION=2.19.0
|
||||
# JAEGER_UI_PORT=16686
|
||||
# JAEGER_OTLP_GRPC_PORT=4317
|
||||
# JAEGER_OTLP_HTTP_PORT=4318
|
||||
|
||||
# ─── Worker ────────────────────────────────────────────────────────────────────
|
||||
# WORKER_CONCURRENCY=20
|
||||
|
||||
+19
-13
@@ -56,7 +56,7 @@ Quick links:
|
||||
```yaml
|
||||
services:
|
||||
openflare:
|
||||
image: ghcr.io/rain-kl/openflare-server:latest
|
||||
image: ghcr.io/rain-kl/openflare:latest
|
||||
restart: unless-stopped
|
||||
env_file: .env
|
||||
environment:
|
||||
@@ -64,7 +64,7 @@ services:
|
||||
ports:
|
||||
- "3000:3000"
|
||||
volumes:
|
||||
- ./uploads:/app/uploads
|
||||
- openflare_uploads:/app/uploads
|
||||
depends_on:
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
@@ -77,13 +77,13 @@ services:
|
||||
image: postgres:17-alpine
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
POSTGRES_DB: openflare
|
||||
POSTGRES_USER: openflare
|
||||
POSTGRES_PASSWORD: replace-with-strong-password
|
||||
POSTGRES_DB: ${DB_NAME:-openflare}
|
||||
POSTGRES_USER: ${DB_USERNAME:-openflare}
|
||||
POSTGRES_PASSWORD: ${DB_PASSWORD:-replace-with-strong-password}
|
||||
volumes:
|
||||
- ./data/postgres_data:/var/lib/postgresql/data
|
||||
- openflare_postgres_data:/var/lib/postgresql/data
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pg_isready -U openflare -d openflare"]
|
||||
test: ["CMD-SHELL", "pg_isready -U ${DB_USERNAME:-openflare} -d ${DB_NAME:-openflare}"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
@@ -93,7 +93,7 @@ services:
|
||||
restart: unless-stopped
|
||||
command: ["valkey-server", "--appendonly", "yes"]
|
||||
volumes:
|
||||
- ./data/valkey:/data
|
||||
- openflare_redis_data:/data
|
||||
healthcheck:
|
||||
test: ["CMD", "valkey-cli", "ping"]
|
||||
interval: 10s
|
||||
@@ -105,19 +105,25 @@ services:
|
||||
image: clickhouse/clickhouse-server:25.3-alpine
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
CLICKHOUSE_DB: openflare
|
||||
CLICKHOUSE_USER: default
|
||||
CLICKHOUSE_PASSWORD: 123456
|
||||
CLICKHOUSE_DB: ${CLICKHOUSE_NAME:-openflare}
|
||||
CLICKHOUSE_USER: ${CLICKHOUSE_USERNAME:-default}
|
||||
CLICKHOUSE_PASSWORD: ${CLICKHOUSE_PASSWORD:-replace-with-clickhouse-password}
|
||||
CLICKHOUSE_DEFAULT_ACCESS_MANAGEMENT: 1
|
||||
TZ: ${TZ:-Asia/Shanghai}
|
||||
volumes:
|
||||
- ./data/clickhouse_data:/var/lib/clickhouse
|
||||
- openflare_clickhouse_data:/var/lib/clickhouse
|
||||
healthcheck:
|
||||
test: ["CMD", "clickhouse-client", "--query", "SELECT 1"]
|
||||
test: ["CMD", "clickhouse-client", "--user", "${CLICKHOUSE_USERNAME:-default}", "--password", "${CLICKHOUSE_PASSWORD:-replace-with-clickhouse-password}", "--query", "SELECT 1"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
start_period: 15s
|
||||
|
||||
volumes:
|
||||
openflare_uploads:
|
||||
openflare_postgres_data:
|
||||
openflare_redis_data:
|
||||
openflare_clickhouse_data:
|
||||
```
|
||||
|
||||
```bash
|
||||
|
||||
@@ -48,16 +48,41 @@ OpenFlare 是开源 CDN 编排与边缘安全平台。它支持反向代理、
|
||||
* **SSO 单点登录**:支持 GitHub OAuth 与标准 OIDC 协议,无缝接入企业身份提供商实现统一登录。
|
||||
* **统一观测**:聚合节点请求指标、实时访问日志明细、宿主机与 Nginx 资源快照、健康事件以及网络波动补传缓冲。
|
||||
|
||||
## 界面预览
|
||||
|
||||
### 仪表盘总览
|
||||
|
||||

|
||||
|
||||
### 节点详情
|
||||
|
||||

|
||||
|
||||
### 配置新增
|
||||
|
||||

|
||||
|
||||
## 快速开始
|
||||
|
||||
### 1. 启动 Server
|
||||
|
||||
使用 docker-compose
|
||||
|
||||
```bash
|
||||
# 下载环境变量模板并创建 .env 文件
|
||||
curl -o .env.example https://raw.githubusercontent.com/Rain-kl/OpenFlare/refs/heads/main/.env.example
|
||||
cp .env.example .env
|
||||
|
||||
# ClickHouse 服务端:curl performance.xml 到 ./config/clickhouse,整目录挂载到 config.d(不要放 listen 配置)
|
||||
mkdir -p ./config/clickhouse
|
||||
curl -fsSL -o ./config/clickhouse/performance.xml \
|
||||
https://raw.githubusercontent.com/Rain-kl/OpenFlare/refs/heads/main/config/clickhouse/performance.xml
|
||||
```
|
||||
|
||||
```yaml
|
||||
services:
|
||||
openflare:
|
||||
image: ghcr.io/rain-kl/openflare-server:latest
|
||||
image: ghcr.io/rain-kl/openflare:latest
|
||||
restart: unless-stopped
|
||||
env_file: .env
|
||||
environment:
|
||||
@@ -65,7 +90,7 @@ services:
|
||||
ports:
|
||||
- "3000:3000"
|
||||
volumes:
|
||||
- ./uploads:/app/uploads
|
||||
- openflare_uploads:/app/uploads
|
||||
depends_on:
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
@@ -78,13 +103,13 @@ services:
|
||||
image: postgres:17-alpine
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
POSTGRES_DB: openflare
|
||||
POSTGRES_USER: openflare
|
||||
POSTGRES_PASSWORD: replace-with-strong-password
|
||||
POSTGRES_DB: ${DB_NAME:-openflare}
|
||||
POSTGRES_USER: ${DB_USERNAME:-openflare}
|
||||
POSTGRES_PASSWORD: ${DB_PASSWORD:-replace-with-strong-password}
|
||||
volumes:
|
||||
- ./data/postgres_data:/var/lib/postgresql/data
|
||||
- openflare_postgres_data:/var/lib/postgresql/data
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pg_isready -U openflare -d openflare"]
|
||||
test: ["CMD-SHELL", "pg_isready -U ${DB_USERNAME:-openflare} -d ${DB_NAME:-openflare}"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
@@ -94,7 +119,7 @@ services:
|
||||
restart: unless-stopped
|
||||
command: ["valkey-server", "--appendonly", "yes"]
|
||||
volumes:
|
||||
- ./data/valkey:/data
|
||||
- openflare_redis_data:/data
|
||||
healthcheck:
|
||||
test: ["CMD", "valkey-cli", "ping"]
|
||||
interval: 10s
|
||||
@@ -106,19 +131,30 @@ services:
|
||||
image: clickhouse/clickhouse-server:25.3-alpine
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
CLICKHOUSE_DB: openflare
|
||||
CLICKHOUSE_USER: default
|
||||
CLICKHOUSE_PASSWORD: 123456
|
||||
CLICKHOUSE_DB: ${CLICKHOUSE_NAME:-openflare}
|
||||
CLICKHOUSE_USER: ${CLICKHOUSE_USERNAME:-default}
|
||||
CLICKHOUSE_PASSWORD: ${CLICKHOUSE_PASSWORD:-replace-with-clickhouse-password}
|
||||
CLICKHOUSE_DEFAULT_ACCESS_MANAGEMENT: 1
|
||||
TZ: ${TZ:-Asia/Shanghai}
|
||||
ulimits:
|
||||
nofile:
|
||||
soft: 262144
|
||||
hard: 262144
|
||||
volumes:
|
||||
- ./data/clickhouse_data:/var/lib/clickhouse
|
||||
- openflare_clickhouse_data:/var/lib/clickhouse
|
||||
- ./config/clickhouse:/etc/clickhouse-server/config.d:ro
|
||||
healthcheck:
|
||||
test: ["CMD", "clickhouse-client", "--query", "SELECT 1"]
|
||||
test: ["CMD", "clickhouse-client", "--user", "${CLICKHOUSE_USERNAME:-default}", "--password", "${CLICKHOUSE_PASSWORD:-replace-with-clickhouse-password}", "--query", "SELECT 1"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
start_period: 15s
|
||||
|
||||
volumes:
|
||||
openflare_uploads:
|
||||
openflare_postgres_data:
|
||||
openflare_redis_data:
|
||||
openflare_clickhouse_data:
|
||||
```
|
||||
|
||||
详细部署说明见 [部署文档](https://open-flare.pages.dev/deployment/deployment)。
|
||||
@@ -150,20 +186,6 @@ docker run -d --name openflare-agent --restart unless-stopped \
|
||||
ghcr.io/rain-kl/openflare-agent:latest
|
||||
```
|
||||
|
||||
## 界面预览
|
||||
|
||||
### 仪表盘总览
|
||||
|
||||

|
||||
|
||||
### 节点详情
|
||||
|
||||

|
||||
|
||||
### 配置新增
|
||||
|
||||

|
||||
|
||||
## 开源协议
|
||||
|
||||
本项目采用 [Apache License 2.0](./LICENSE) 开源。
|
||||
|
||||
+8
-5
@@ -99,15 +99,18 @@ otel:
|
||||
|
||||
|
||||
# ─── ClickHouse (required) ──────────────────────────────────────────────────────
|
||||
# Analytics / observability OLAP store. Telemetry writes are best-effort (async batch).
|
||||
clickhouse:
|
||||
enabled: true
|
||||
hosts:
|
||||
- "127.0.0.1:9000"
|
||||
- "127.0.0.1:9000" # compose 内应用可用 clickhouse:9000(经 CLICKHOUSE_HOST)
|
||||
username: "default"
|
||||
password: "123456"
|
||||
password: "replace-with-clickhouse-password" # 与 .env / compose CLICKHOUSE_PASSWORD 一致
|
||||
database: "openflare"
|
||||
max_idle_conn: 10
|
||||
max_open_conn: 100
|
||||
max_idle_conn: 8 # keep warm sockets low to save client + server RAM
|
||||
max_open_conn: 16 # cap concurrent native sessions on modest CH boxes
|
||||
conn_max_lifetime: 3600
|
||||
dial_timeout: 5
|
||||
block_buffer_size: 10
|
||||
block_buffer_size: 32 # rows buffered per block; 32 is enough for our batch sizes
|
||||
# Runtime client also enables async_insert (wait_for_async_insert=1, busy_timeout≈2s)
|
||||
# in internal/db/clickhouse.go — not configured via YAML.
|
||||
|
||||
@@ -0,0 +1,25 @@
|
||||
<?xml version="1.0"?>
|
||||
<!--
|
||||
Tuned for small control-plane hosts (e.g. 3c6g).
|
||||
|
||||
background_pool_size * background_merges_mutations_concurrency_ratio must stay
|
||||
greater than merge_tree number_of_free_entries_in_pool_to_execute_mutation
|
||||
(ClickHouse 25.x refuses to start otherwise). Keep the merge free-entry
|
||||
thresholds low so a small pool remains valid.
|
||||
-->
|
||||
<clickhouse>
|
||||
<max_concurrent_queries>20</max_concurrent_queries>
|
||||
<background_pool_size>4</background_pool_size>
|
||||
<background_merges_mutations_concurrency_ratio>2</background_merges_mutations_concurrency_ratio>
|
||||
<background_schedule_pool_size>4</background_schedule_pool_size>
|
||||
<background_common_pool_size>2</background_common_pool_size>
|
||||
<background_fetches_pool_size>2</background_fetches_pool_size>
|
||||
<background_move_pool_size>1</background_move_pool_size>
|
||||
<mark_cache_size>268435456</mark_cache_size>
|
||||
<uncompressed_cache_size>0</uncompressed_cache_size>
|
||||
<merge_tree>
|
||||
<number_of_free_entries_in_pool_to_execute_mutation>2</number_of_free_entries_in_pool_to_execute_mutation>
|
||||
<number_of_free_entries_in_pool_to_lower_max_size_of_merge>2</number_of_free_entries_in_pool_to_lower_max_size_of_merge>
|
||||
<number_of_free_entries_in_pool_to_execute_optimize_entire_partition>2</number_of_free_entries_in_pool_to_execute_optimize_entire_partition>
|
||||
</merge_tree>
|
||||
</clickhouse>
|
||||
+16
-11
@@ -5,7 +5,7 @@ services:
|
||||
dockerfile: docker/Dockerfile
|
||||
args:
|
||||
VERSION: v0.9.9
|
||||
# image: ghcr.io/rain-kl/openflare-server:latest
|
||||
# image: ghcr.io/rain-kl/openflare:latest
|
||||
restart: unless-stopped
|
||||
env_file: .env
|
||||
environment:
|
||||
@@ -34,13 +34,13 @@ services:
|
||||
ports:
|
||||
- "5432:5432"
|
||||
environment:
|
||||
POSTGRES_DB: openflare
|
||||
POSTGRES_USER: openflare
|
||||
POSTGRES_PASSWORD: replace-with-strong-password
|
||||
POSTGRES_DB: ${DB_NAME:-openflare}
|
||||
POSTGRES_USER: ${DB_USERNAME:-openflare}
|
||||
POSTGRES_PASSWORD: ${DB_PASSWORD:-replace-with-strong-password}
|
||||
volumes:
|
||||
- ./data/postgres_data:/var/lib/postgresql/data
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pg_isready -U openflare -d openflare"]
|
||||
test: ["CMD-SHELL", "pg_isready -U ${DB_USERNAME:-openflare} -d ${DB_NAME:-openflare}"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
@@ -74,18 +74,23 @@ services:
|
||||
image: clickhouse/clickhouse-server:25.3-alpine
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
CLICKHOUSE_DB: openflare
|
||||
CLICKHOUSE_USER: default
|
||||
CLICKHOUSE_PASSWORD: 123456
|
||||
CLICKHOUSE_DB: ${CLICKHOUSE_NAME:-openflare}
|
||||
CLICKHOUSE_USER: ${CLICKHOUSE_USERNAME:-default}
|
||||
CLICKHOUSE_PASSWORD: ${CLICKHOUSE_PASSWORD:-replace-with-clickhouse-password}
|
||||
CLICKHOUSE_DEFAULT_ACCESS_MANAGEMENT: 1
|
||||
TZ: ${TZ:-Asia/Shanghai}
|
||||
ulimits:
|
||||
nofile:
|
||||
soft: 262144
|
||||
hard: 262144
|
||||
ports:
|
||||
- "${CLICKHOUSE_HTTP_PORT:-8123}:8123"
|
||||
- "${CLICKHOUSE_NATIVE_PORT:-9000}:9000"
|
||||
- "8123:8123"
|
||||
- "9000:9000"
|
||||
volumes:
|
||||
- ./data/clickhouse_data:/var/lib/clickhouse
|
||||
- ./config/clickhouse:/etc/clickhouse-server/config.d:ro
|
||||
healthcheck:
|
||||
test: ["CMD", "clickhouse-client", "--query", "SELECT 1"]
|
||||
test: ["CMD", "clickhouse-client", "--user", "${CLICKHOUSE_USERNAME:-default}", "--password", "${CLICKHOUSE_PASSWORD:-replace-with-clickhouse-password}", "--query", "SELECT 1"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
|
||||
@@ -11,6 +11,9 @@ sidebar: false
|
||||
## 重大变更
|
||||
|
||||
> [!IMPORTANT]
|
||||
>
|
||||
> 3.1.2 版本更新了 CLickHouse 部署配置。
|
||||
>
|
||||
> 3.0.0 版本为 Wavelet 平台迁移与架构重构版本,涉及数据库表结构、环境变量以及前后端底层架构的重大变更。请务必在升级前备份数据库,并且更新到 V2.3.4。
|
||||
> 目前已知的兼容性问题:
|
||||
> - Pages 无法迁移, 升级前请先手动下载并备份 Pages 静态站点的 ZIP 包,升级后重新创建。
|
||||
@@ -18,6 +21,50 @@ sidebar: false
|
||||
|
||||
## [unreleased]
|
||||
|
||||
## [v3.1.2] - 2026-07-10
|
||||
|
||||
### 修复
|
||||
|
||||
- 修复节点/仪表盘 24 小时容量、网络、磁盘 IO 趋势在 ClickHouse 限流查询下几乎为空的问题:改为基于小时级聚合与计数器 delta 统计,避免仅依赖最近有限条原始快照导致历史时段全空。
|
||||
- 降低静置时 ClickHouse CPU:可观测/访问日志 batchwriter 启用 `MinBatchSize` 与 `MaxFlushWait`,减少心跳小 part 写入;Docker `performance.xml` 收紧小规格后台 merge 池。
|
||||
- ClickHouse 清理语义:按保留天数仅 `MATERIALIZE` 表 DDL TTL,`deleted_count` 不再伪报删除;短于表 TTL 的保留请求被拒绝。
|
||||
- 可观测 dedup 仅在入队成功后保留,flush 失败释放键并短重试;审计 writer 增加 `MaxFlushWait`;`/admin/status/clickhouse` 暴露 batch writer 队列深度/丢弃/flush 错误。
|
||||
- model 层通过 hooks 写入 CH,去除对 `chwriter` 的直接依赖。
|
||||
- Dashboard 每节点最新指标改为 `LIMIT 1 BY node_id`;新增 metric/openresty 小时预聚合表;读路径按小时 merge(rollup 窗口完整时仅走预聚合,不足时用 raw 补洞),并提供历史 backfill 迁移。
|
||||
- 小规格默认连接池下调;`async_insert_busy_timeout` 调至 2s;`of_node_traffic_hourly` 增加 30 天 TTL,UV 改为峰值窗口估计并修正前端文案。
|
||||
- Docker ClickHouse:`performance.xml` 下调 merge free-entry 阈值以兼容小 `background_pool`(避免 25.x 启动 Code 36)。
|
||||
|
||||
### 文档
|
||||
|
||||
- 同步 `.env.example` 与 `config.example.yaml`;ClickHouse 服务端配置改为 curl `performance.xml` 到 `./config/clickhouse` 后整目录挂载至 `config.d`(不要放入 listen 配置)。
|
||||
|
||||
## [v3.1.1] - 2026-07-06
|
||||
|
||||
### 修改
|
||||
|
||||
- 将 `cap_login_enabled` 默认值由 `true` 变更为 `false`,默认关闭登录界面 PoW 人机验证。
|
||||
|
||||
## [v3.1.0] - 2026-07-04
|
||||
|
||||
### 修改
|
||||
|
||||
- 修复 ClickHouse TTL 迁移:`DateTime64` 时间列通过 `toDateTime()` 转换后再设置 TTL,避免 goose 启动报错;移除 `MODIFY ORDER BY`(ClickHouse 不允许将排序键缩短至短于隐式主键前缀)。
|
||||
- ClickHouse 遗留治理 Phase 2:保留期清理改为 TTL `MATERIALIZE TTL`(全量清理使用 `TRUNCATE`),消除定时 `ALTER DELETE` mutation;移除 GORM 双连接池并统一 `ChConn` 读路径;查询侧去除 `trim(remote_addr)`;`wait_for_async_insert` 调整为 1;新增 `/admin/status/clickhouse` 运维指标与 `of_node_traffic_hourly` 预聚合 MV。
|
||||
- ClickHouse 写入路径优化:移除 Agent 心跳路径中的同步 `ALTER DELETE` 保留清理;`batchwriter` 新增 `MinBatchSize` 抑制过小批次定时 flush;可观测 writer 批次提升至 500、flush 间隔 5s,并为 OpenResty/FRPS/FRPC 补全去重。
|
||||
- ClickHouse 客户端启用 `async_insert` 异步写入缓冲,并调高 `block_buffer_size` 与连接池默认值,降低小 part 与连接争用。
|
||||
- Dashboard 与节点可观测 API 消除无 `LIMIT` 全表扫描、增加短 TTL 内存缓存,前端轮询间隔分别调整为 60s/30s。
|
||||
- 访问日志与 WAF IP 组同步改为 ClickHouse 侧聚合与 SQL 分页,默认查询窗口限制为近 7 天,浏览器分布查询增加 Top 100 限制。
|
||||
- ClickHouse 分析表新增 TTL 自动过期策略:`w_user_access_logs` 180 天、`of_node_access_logs` 90 天,其余节点观测与聚合表 30 天。
|
||||
- 节点访问日志写入 ClickHouse 时对 `remote_addr` 执行 `TrimSpace` 规范化,避免首尾空白影响 IP 汇总统计。
|
||||
- 数据库自动清理任务新增 OpenResty、FRPS、FRPC 观测表清理目标。
|
||||
- Docker 部署为 ClickHouse 服务增加 `nofile` ulimits 与 `docker/clickhouse/config.d/performance.xml` 性能配置挂载,限制 `max_concurrent_queries`、`background_pool_size` 与 `background_merges_mutations_concurrency_ratio`,降低高负载下的合并与查询争用。
|
||||
- 审计访问日志写入 ClickHouse 时仅保留安全相关请求头(Authorization、Cookie、X-Forwarded-For、X-Real-IP、User-Agent、Content-Type),敏感头字段以 SHA-256 摘要脱敏,并将序列化后的 headers 载荷上限收紧至 2KB,减小 `w_user_access_logs` 行宽与 merge CPU 开销。
|
||||
- 隐藏侧边栏“文档库”分组中的“规范示例”与“接口文档”,并将“使用文档”及其他相关页面的文档链接统一跳转至外部文档 https://open-flare.pages.dev/
|
||||
- 修复全局搜索数据源覆盖不全的问题,补全了所有核心业务控制台页面(节点、规则、域名、证书、DNS、源站、WAF、IP组、Pages、版本发布、访问日志、应用记录和性能调优)及缺失的管理员专有页面(存储、数据、推送、日志)的搜索检索支持。
|
||||
- 修复系统自更新(Updater)检测上游 GitHub Action Release 时,因资产包名称前缀(`openflare-server`)与仓库名不完全一致导致匹配失败并报错“未找到兼容的 Release”的问题。
|
||||
- 修复系统设置页面(`/admin/settings`)基于 URL `tab` 参数的定位逻辑,补全缺失的 `openflare-ops` (OpenFlare) Tab,且在不带参数时默认选中 OpenFlare 选项卡。
|
||||
- 移除系统设置中 OpenFlare 标签页下的“版本信息”卡片及对应的升级管理弹窗逻辑。
|
||||
|
||||
## [v3.0.2] - 2026-06-30
|
||||
|
||||
### 修复
|
||||
|
||||
@@ -81,6 +81,7 @@ Agent:
|
||||
仓库根目录已提供完整 `docker-compose.yaml`(含 PostgreSQL、Redis、ClickHouse、Jaeger)。
|
||||
|
||||
```bash
|
||||
curl -o .env.example https://raw.githubusercontent.com/Rain-kl/OpenFlare/refs/heads/main/.env.example
|
||||
cp .env.example .env
|
||||
# 编辑 .env,至少修改 APP_SESSION_SECRET 与数据库密码
|
||||
docker compose up -d
|
||||
|
||||
+91
-29
@@ -8,6 +8,30 @@ OpenFlare Server 是 Gin + GORM 单体控制面,负责管理端 UI、管理 AP
|
||||
> **关于外部依赖**:
|
||||
> OpenFlare 系统内建了对后台异步任务(Asynq 框架)及海量节点日志分析与度量指标(观测面板)的支持。因此,**无论采用何种部署模式,系统都必须依赖 Redis(或 Valkey)与 ClickHouse 的运行**。各个部署方案的主要差异在于主关系型数据库的选择(SQLite vs PostgreSQL)以及是否启用链路追踪服务(Jaeger)。
|
||||
|
||||
> [!TIP]
|
||||
> **ClickHouse 服务端性能配置(推荐挂载)**
|
||||
> 控制面常见为小规格主机(如 3c6g)。仓库提供的 `performance.xml` 会收紧后台 merge/mutation 线程池,避免默认配置在小机器上静置 CPU 偏高或 ClickHouse 25.x 启动校验失败。
|
||||
> 将本地目录 `./config/clickhouse` 挂载到容器 `/etc/clickhouse-server/config.d`。
|
||||
> **目录内只放 `performance.xml`,不要放入任何 listen 相关配置**(监听地址沿用官方镜像默认即可)。
|
||||
|
||||
部署前将配置拉到本地:
|
||||
|
||||
```bash
|
||||
mkdir -p ./config/clickhouse
|
||||
curl -fsSL -o ./config/clickhouse/performance.xml \
|
||||
https://raw.githubusercontent.com/Rain-kl/OpenFlare/refs/heads/main/config/clickhouse/performance.xml
|
||||
```
|
||||
|
||||
在 ClickHouse 服务的 `volumes` 中增加(与数据卷并列):
|
||||
|
||||
```yaml
|
||||
volumes:
|
||||
- ./data/clickhouse_data:/var/lib/clickhouse # 或 named volume
|
||||
- ./config/clickhouse:/etc/clickhouse-server/config.d:ro
|
||||
```
|
||||
|
||||
修改 `performance.xml` 后需 `docker compose restart clickhouse` 才生效。
|
||||
|
||||
---
|
||||
|
||||
## 方式一:Docker 部署 (推荐)
|
||||
@@ -27,7 +51,7 @@ version: '3.8'
|
||||
|
||||
services:
|
||||
openflare:
|
||||
image: ghcr.io/rain-kl/openflare-server:latest
|
||||
image: ghcr.io/rain-kl/openflare:latest
|
||||
container_name: openflare-server
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
@@ -71,10 +95,15 @@ services:
|
||||
CLICKHOUSE_PASSWORD: 123456
|
||||
CLICKHOUSE_DEFAULT_ACCESS_MANAGEMENT: 1
|
||||
TZ: Asia/Shanghai
|
||||
ulimits:
|
||||
nofile:
|
||||
soft: 262144
|
||||
hard: 262144
|
||||
volumes:
|
||||
- ./data/clickhouse_data:/var/lib/clickhouse
|
||||
- ./config/clickhouse:/etc/clickhouse-server/config.d:ro
|
||||
healthcheck:
|
||||
test: ["CMD", "clickhouse-client", "--query", "SELECT 1"]
|
||||
test: ["CMD", "clickhouse-client", "--user", "default", "--password", "123456", "--query", "SELECT 1"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
@@ -84,6 +113,9 @@ services:
|
||||
运行启动命令:
|
||||
|
||||
```bash
|
||||
mkdir -p ./config/clickhouse
|
||||
curl -fsSL -o ./config/clickhouse/performance.xml \
|
||||
https://raw.githubusercontent.com/Rain-kl/OpenFlare/refs/heads/main/config/clickhouse/performance.xml
|
||||
docker compose up -d
|
||||
```
|
||||
|
||||
@@ -100,7 +132,7 @@ docker compose up -d
|
||||
```yaml
|
||||
services:
|
||||
openflare:
|
||||
image: ghcr.io/rain-kl/openflare-server:latest
|
||||
image: ghcr.io/rain-kl/openflare:latest
|
||||
restart: unless-stopped
|
||||
env_file: .env
|
||||
environment:
|
||||
@@ -108,7 +140,7 @@ services:
|
||||
ports:
|
||||
- "3000:3000"
|
||||
volumes:
|
||||
- ./uploads:/app/uploads
|
||||
- openflare_uploads:/app/uploads
|
||||
depends_on:
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
@@ -121,13 +153,13 @@ services:
|
||||
image: postgres:17-alpine
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
POSTGRES_DB: openflare
|
||||
POSTGRES_USER: openflare
|
||||
POSTGRES_PASSWORD: replace-with-strong-password
|
||||
POSTGRES_DB: ${DB_NAME:-openflare}
|
||||
POSTGRES_USER: ${DB_USERNAME:-openflare}
|
||||
POSTGRES_PASSWORD: ${DB_PASSWORD:-replace-with-strong-password}
|
||||
volumes:
|
||||
- ./data/postgres_data:/var/lib/postgresql/data
|
||||
- openflare_postgres_data:/var/lib/postgresql/data
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pg_isready -U openflare -d openflare"]
|
||||
test: ["CMD-SHELL", "pg_isready -U ${DB_USERNAME:-openflare} -d ${DB_NAME:-openflare}"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
@@ -137,7 +169,7 @@ services:
|
||||
restart: unless-stopped
|
||||
command: ["valkey-server", "--appendonly", "yes"]
|
||||
volumes:
|
||||
- ./data/valkey:/data
|
||||
- openflare_redis_data:/data
|
||||
healthcheck:
|
||||
test: ["CMD", "valkey-cli", "ping"]
|
||||
interval: 10s
|
||||
@@ -149,24 +181,39 @@ services:
|
||||
image: clickhouse/clickhouse-server:25.3-alpine
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
CLICKHOUSE_DB: openflare
|
||||
CLICKHOUSE_USER: default
|
||||
CLICKHOUSE_PASSWORD: replace-with-clickhouse-password
|
||||
CLICKHOUSE_DB: ${CLICKHOUSE_NAME:-openflare}
|
||||
CLICKHOUSE_USER: ${CLICKHOUSE_USERNAME:-default}
|
||||
CLICKHOUSE_PASSWORD: ${CLICKHOUSE_PASSWORD:-replace-with-clickhouse-password}
|
||||
CLICKHOUSE_DEFAULT_ACCESS_MANAGEMENT: 1
|
||||
TZ: ${TZ:-Asia/Shanghai}
|
||||
ulimits:
|
||||
nofile:
|
||||
soft: 262144
|
||||
hard: 262144
|
||||
volumes:
|
||||
- ./data/clickhouse_data:/var/lib/clickhouse
|
||||
- openflare_clickhouse_data:/var/lib/clickhouse
|
||||
- ./config/clickhouse:/etc/clickhouse-server/config.d:ro
|
||||
healthcheck:
|
||||
test: ["CMD", "clickhouse-client", "--query", "SELECT 1"]
|
||||
test: ["CMD", "clickhouse-client", "--user", "${CLICKHOUSE_USERNAME:-default}", "--password", "${CLICKHOUSE_PASSWORD:-replace-with-clickhouse-password}", "--query", "SELECT 1"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
start_period: 15s
|
||||
|
||||
volumes:
|
||||
openflare_uploads:
|
||||
openflare_postgres_data:
|
||||
openflare_redis_data:
|
||||
openflare_clickhouse_data:
|
||||
```
|
||||
|
||||
创建对应的 `.env` 文件来配置系统环境变量(可复制并修改根目录下的 `.env.example`):
|
||||
|
||||
```bash
|
||||
mkdir -p ./config/clickhouse
|
||||
curl -fsSL -o ./config/clickhouse/performance.xml \
|
||||
https://raw.githubusercontent.com/Rain-kl/OpenFlare/refs/heads/main/config/clickhouse/performance.xml
|
||||
curl -o .env.example https://raw.githubusercontent.com/Rain-kl/OpenFlare/refs/heads/main/.env.example
|
||||
cp .env.example .env
|
||||
# 编辑 .env 文件,填入对应的数据库、Redis、ClickHouse 连接地址、密码与 APP_SESSION_SECRET
|
||||
|
||||
@@ -188,7 +235,7 @@ version: '3.8'
|
||||
|
||||
services:
|
||||
openflare:
|
||||
image: ghcr.io/rain-kl/openflare-server:latest
|
||||
image: ghcr.io/rain-kl/openflare:latest
|
||||
restart: unless-stopped
|
||||
env_file: .env
|
||||
environment:
|
||||
@@ -199,7 +246,7 @@ services:
|
||||
ports:
|
||||
- "3000:3000"
|
||||
volumes:
|
||||
- ./uploads:/app/uploads
|
||||
- openflare_uploads:/app/uploads
|
||||
depends_on:
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
@@ -214,13 +261,13 @@ services:
|
||||
image: postgres:17-alpine
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
POSTGRES_DB: openflare
|
||||
POSTGRES_USER: openflare
|
||||
POSTGRES_PASSWORD: replace-with-strong-password
|
||||
POSTGRES_DB: ${DB_NAME:-openflare}
|
||||
POSTGRES_USER: ${DB_USERNAME:-openflare}
|
||||
POSTGRES_PASSWORD: ${DB_PASSWORD:-replace-with-strong-password}
|
||||
volumes:
|
||||
- ./data/postgres_data:/var/lib/postgresql/data
|
||||
- openflare_postgres_data:/var/lib/postgresql/data
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pg_isready -U openflare -d openflare"]
|
||||
test: ["CMD-SHELL", "pg_isready -U ${DB_USERNAME:-openflare} -d ${DB_NAME:-openflare}"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
@@ -230,7 +277,7 @@ services:
|
||||
restart: unless-stopped
|
||||
command: ["valkey-server", "--appendonly", "yes"]
|
||||
volumes:
|
||||
- ./data/valkey:/data
|
||||
- openflare_redis_data:/data
|
||||
healthcheck:
|
||||
test: ["CMD", "valkey-cli", "ping"]
|
||||
interval: 10s
|
||||
@@ -252,24 +299,39 @@ services:
|
||||
image: clickhouse/clickhouse-server:25.3-alpine
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
CLICKHOUSE_DB: openflare
|
||||
CLICKHOUSE_USER: default
|
||||
CLICKHOUSE_PASSWORD: replace-with-clickhouse-password
|
||||
CLICKHOUSE_DB: ${CLICKHOUSE_NAME:-openflare}
|
||||
CLICKHOUSE_USER: ${CLICKHOUSE_USERNAME:-default}
|
||||
CLICKHOUSE_PASSWORD: ${CLICKHOUSE_PASSWORD:-replace-with-clickhouse-password}
|
||||
CLICKHOUSE_DEFAULT_ACCESS_MANAGEMENT: 1
|
||||
TZ: ${TZ:-Asia/Shanghai}
|
||||
ulimits:
|
||||
nofile:
|
||||
soft: 262144
|
||||
hard: 262144
|
||||
volumes:
|
||||
- ./data/clickhouse_data:/var/lib/clickhouse
|
||||
- openflare_clickhouse_data:/var/lib/clickhouse
|
||||
- ./config/clickhouse:/etc/clickhouse-server/config.d:ro
|
||||
healthcheck:
|
||||
test: ["CMD", "clickhouse-client", "--query", "SELECT 1"]
|
||||
test: ["CMD", "clickhouse-client", "--user", "${CLICKHOUSE_USERNAME:-default}", "--password", "${CLICKHOUSE_PASSWORD:-replace-with-clickhouse-password}", "--query", "SELECT 1"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
start_period: 15s
|
||||
|
||||
volumes:
|
||||
openflare_uploads:
|
||||
openflare_postgres_data:
|
||||
openflare_redis_data:
|
||||
openflare_clickhouse_data:
|
||||
```
|
||||
|
||||
启动并验证:
|
||||
|
||||
```bash
|
||||
mkdir -p ./config/clickhouse
|
||||
curl -fsSL -o ./config/clickhouse/performance.xml \
|
||||
https://raw.githubusercontent.com/Rain-kl/OpenFlare/refs/heads/main/config/clickhouse/performance.xml
|
||||
curl -o .env.example https://raw.githubusercontent.com/Rain-kl/OpenFlare/refs/heads/main/.env.example
|
||||
cp .env.example .env
|
||||
# 编辑 .env 文件并确保设置好 APP_SESSION_SECRET 密码
|
||||
|
||||
@@ -290,7 +352,7 @@ docker compose up -d
|
||||
| Go | `1.25+` |
|
||||
| Node.js | `18+` |
|
||||
| pnpm | 推荐通过 `corepack enable` 使用项目声明的 pnpm |
|
||||
| 外部服务 | 必须在本地或远端运行 Redis (Valkey) 和 ClickHouse 实例 |
|
||||
| 外部服务 | 必须在本地或远端运行 Redis (Valkey) 和 ClickHouse 实例;ClickHouse 建议挂载仓库提供的 `performance.xml`(见上文「ClickHouse 服务端性能配置」) |
|
||||
|
||||
### 1. 构建管理端前端
|
||||
|
||||
|
||||
+219
-5
@@ -309,6 +309,7 @@ const docTemplate = `{
|
||||
"parameters": [
|
||||
{
|
||||
"type": "integer",
|
||||
"format": "int64",
|
||||
"description": "认证源 ID 或名称",
|
||||
"name": "id",
|
||||
"in": "path",
|
||||
@@ -386,6 +387,7 @@ const docTemplate = `{
|
||||
"parameters": [
|
||||
{
|
||||
"type": "integer",
|
||||
"format": "int64",
|
||||
"description": "认证源 ID 或名称",
|
||||
"name": "id",
|
||||
"in": "path",
|
||||
@@ -453,6 +455,7 @@ const docTemplate = `{
|
||||
"parameters": [
|
||||
{
|
||||
"type": "integer",
|
||||
"format": "int64",
|
||||
"description": "认证源 ID 或名称",
|
||||
"name": "id",
|
||||
"in": "path",
|
||||
@@ -1363,6 +1366,7 @@ const docTemplate = `{
|
||||
"parameters": [
|
||||
{
|
||||
"type": "integer",
|
||||
"format": "int64",
|
||||
"description": "通道ID",
|
||||
"name": "id",
|
||||
"in": "path",
|
||||
@@ -1416,6 +1420,7 @@ const docTemplate = `{
|
||||
"parameters": [
|
||||
{
|
||||
"type": "integer",
|
||||
"format": "int64",
|
||||
"description": "通道ID",
|
||||
"name": "id",
|
||||
"in": "path",
|
||||
@@ -1872,6 +1877,67 @@ const docTemplate = `{
|
||||
}
|
||||
}
|
||||
},
|
||||
"/api/v1/admin/status/clickhouse": {
|
||||
"get": {
|
||||
"security": [
|
||||
{
|
||||
"SessionCookie": []
|
||||
}
|
||||
],
|
||||
"description": "返回 ClickHouse parts、mutation、async_insert 队列等运维指标,需要管理员权限",
|
||||
"produces": [
|
||||
"application/json"
|
||||
],
|
||||
"tags": [
|
||||
"admin"
|
||||
],
|
||||
"summary": "获取 ClickHouse 运行指标",
|
||||
"responses": {
|
||||
"200": {
|
||||
"description": "获取成功",
|
||||
"schema": {
|
||||
"allOf": [
|
||||
{
|
||||
"$ref": "#/definitions/response.Any"
|
||||
},
|
||||
{
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"data": {
|
||||
"$ref": "#/definitions/analytics.ClickHouseOperationalStats"
|
||||
}
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
},
|
||||
"400": {
|
||||
"description": "ClickHouse 未启用",
|
||||
"schema": {
|
||||
"$ref": "#/definitions/response.Any"
|
||||
}
|
||||
},
|
||||
"401": {
|
||||
"description": "未登录",
|
||||
"schema": {
|
||||
"$ref": "#/definitions/response.Any"
|
||||
}
|
||||
},
|
||||
"403": {
|
||||
"description": "无管理员权限",
|
||||
"schema": {
|
||||
"$ref": "#/definitions/response.Any"
|
||||
}
|
||||
},
|
||||
"500": {
|
||||
"description": "内部错误",
|
||||
"schema": {
|
||||
"$ref": "#/definitions/response.Any"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"/api/v1/admin/system-configs": {
|
||||
"get": {
|
||||
"security": [
|
||||
@@ -3368,6 +3434,7 @@ const docTemplate = `{
|
||||
},
|
||||
{
|
||||
"type": "integer",
|
||||
"format": "int64",
|
||||
"description": "上传用户 ID",
|
||||
"name": "user_id",
|
||||
"in": "query"
|
||||
@@ -3691,6 +3758,11 @@ const docTemplate = `{
|
||||
],
|
||||
"summary": "获取用户列表",
|
||||
"parameters": [
|
||||
{
|
||||
"type": "string",
|
||||
"name": "email",
|
||||
"in": "query"
|
||||
},
|
||||
{
|
||||
"minimum": 1,
|
||||
"type": "integer",
|
||||
@@ -3909,6 +3981,92 @@ const docTemplate = `{
|
||||
}
|
||||
}
|
||||
},
|
||||
"put": {
|
||||
"security": [
|
||||
{
|
||||
"SessionCookie": []
|
||||
}
|
||||
],
|
||||
"description": "更新指定用户的昵称、邮箱、管理员权限,并可选重置密码,需要管理员权限",
|
||||
"consumes": [
|
||||
"application/json"
|
||||
],
|
||||
"produces": [
|
||||
"application/json"
|
||||
],
|
||||
"tags": [
|
||||
"admin"
|
||||
],
|
||||
"summary": "更新用户信息",
|
||||
"parameters": [
|
||||
{
|
||||
"type": "integer",
|
||||
"description": "用户 ID",
|
||||
"name": "id",
|
||||
"in": "path",
|
||||
"required": true
|
||||
},
|
||||
{
|
||||
"description": "更新参数",
|
||||
"name": "request",
|
||||
"in": "body",
|
||||
"required": true,
|
||||
"schema": {
|
||||
"$ref": "#/definitions/user.updateUserRequest"
|
||||
}
|
||||
}
|
||||
],
|
||||
"responses": {
|
||||
"200": {
|
||||
"description": "更新成功",
|
||||
"schema": {
|
||||
"allOf": [
|
||||
{
|
||||
"$ref": "#/definitions/response.Any"
|
||||
},
|
||||
{
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"data": {
|
||||
"type": "string"
|
||||
}
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
},
|
||||
"400": {
|
||||
"description": "参数错误",
|
||||
"schema": {
|
||||
"$ref": "#/definitions/response.Any"
|
||||
}
|
||||
},
|
||||
"401": {
|
||||
"description": "未登录",
|
||||
"schema": {
|
||||
"$ref": "#/definitions/response.Any"
|
||||
}
|
||||
},
|
||||
"403": {
|
||||
"description": "无管理员权限或尝试修改自身权限",
|
||||
"schema": {
|
||||
"$ref": "#/definitions/response.Any"
|
||||
}
|
||||
},
|
||||
"404": {
|
||||
"description": "用户不存在",
|
||||
"schema": {
|
||||
"$ref": "#/definitions/response.Any"
|
||||
}
|
||||
},
|
||||
"500": {
|
||||
"description": "内部错误",
|
||||
"schema": {
|
||||
"$ref": "#/definitions/response.Any"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"delete": {
|
||||
"security": [
|
||||
{
|
||||
@@ -11144,6 +11302,7 @@ const docTemplate = `{
|
||||
"parameters": [
|
||||
{
|
||||
"type": "integer",
|
||||
"format": "int64",
|
||||
"description": "外部帐号绑定记录 ID",
|
||||
"name": "id",
|
||||
"in": "path",
|
||||
@@ -12948,6 +13107,29 @@ const docTemplate = `{
|
||||
}
|
||||
}
|
||||
},
|
||||
"analytics.ClickHouseOperationalStats": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"active_parts": {
|
||||
"type": "integer"
|
||||
},
|
||||
"async_insert_bytes": {
|
||||
"type": "integer"
|
||||
},
|
||||
"async_insert_queue": {
|
||||
"type": "integer"
|
||||
},
|
||||
"database": {
|
||||
"type": "string"
|
||||
},
|
||||
"pending_mutations": {
|
||||
"type": "integer"
|
||||
},
|
||||
"total_rows": {
|
||||
"type": "integer"
|
||||
}
|
||||
}
|
||||
},
|
||||
"apply_log.CleanupInput": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
@@ -13775,19 +13957,22 @@ const docTemplate = `{
|
||||
"source_countries": {
|
||||
"type": "object",
|
||||
"additionalProperties": {
|
||||
"type": "integer"
|
||||
"type": "integer",
|
||||
"format": "int64"
|
||||
}
|
||||
},
|
||||
"status_codes": {
|
||||
"type": "object",
|
||||
"additionalProperties": {
|
||||
"type": "integer"
|
||||
"type": "integer",
|
||||
"format": "int64"
|
||||
}
|
||||
},
|
||||
"top_domains": {
|
||||
"type": "object",
|
||||
"additionalProperties": {
|
||||
"type": "integer"
|
||||
"type": "integer",
|
||||
"format": "int64"
|
||||
}
|
||||
},
|
||||
"unique_visitor_count": {
|
||||
@@ -14217,7 +14402,7 @@ const docTemplate = `{
|
||||
"type": "string"
|
||||
},
|
||||
"id": {
|
||||
"type": "integer"
|
||||
"type": "string"
|
||||
},
|
||||
"is_active": {
|
||||
"type": "boolean"
|
||||
@@ -14252,7 +14437,7 @@ const docTemplate = `{
|
||||
"type": "string"
|
||||
},
|
||||
"id": {
|
||||
"type": "integer"
|
||||
"type": "string"
|
||||
},
|
||||
"is_active": {
|
||||
"type": "boolean"
|
||||
@@ -15092,6 +15277,11 @@ const docTemplate = `{
|
||||
"UploadStatusPending": "待使用",
|
||||
"UploadStatusUsed": "已使用"
|
||||
},
|
||||
"x-enum-descriptions": [
|
||||
"待使用",
|
||||
"已使用",
|
||||
"已删除"
|
||||
],
|
||||
"x-enum-varnames": [
|
||||
"UploadStatusPending",
|
||||
"UploadStatusUsed",
|
||||
@@ -18085,6 +18275,30 @@ const docTemplate = `{
|
||||
}
|
||||
}
|
||||
},
|
||||
"user.updateUserRequest": {
|
||||
"type": "object",
|
||||
"required": [
|
||||
"email"
|
||||
],
|
||||
"properties": {
|
||||
"email": {
|
||||
"type": "string",
|
||||
"maxLength": 255
|
||||
},
|
||||
"is_admin": {
|
||||
"type": "boolean"
|
||||
},
|
||||
"nickname": {
|
||||
"type": "string",
|
||||
"maxLength": 64
|
||||
},
|
||||
"password": {
|
||||
"type": "string",
|
||||
"maxLength": 64,
|
||||
"minLength": 8
|
||||
}
|
||||
}
|
||||
},
|
||||
"user.updateUserStatusRequest": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
|
||||
+35
-16
@@ -30,6 +30,14 @@ Agent 统一通过 OpenResty 二进制控制运行时。本地部署需要节点
|
||||
|
||||
为了保证异步任务队列(Asynq 框架)及可观测流量看板功能完整运行,快速开始推荐采用 **PostgreSQL + Redis + ClickHouse** 经典单机版编排。
|
||||
|
||||
先拉取 ClickHouse 服务端性能配置到 `./config/clickhouse`(目录内**仅**放 `performance.xml`,不要放 listen 配置):
|
||||
|
||||
```bash
|
||||
mkdir -p ./config/clickhouse
|
||||
curl -fsSL -o ./config/clickhouse/performance.xml \
|
||||
https://raw.githubusercontent.com/Rain-kl/OpenFlare/refs/heads/main/config/clickhouse/performance.xml
|
||||
```
|
||||
|
||||
在空目录中创建 `docker-compose.yaml`:
|
||||
|
||||
```yaml
|
||||
@@ -37,26 +45,26 @@ version: '3.8'
|
||||
|
||||
services:
|
||||
openflare:
|
||||
image: ghcr.io/rain-kl/openflare-server:latest
|
||||
image: ghcr.io/rain-kl/openflare:latest
|
||||
container_name: openflare-server
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
- "3000:3000"
|
||||
volumes:
|
||||
- ./uploads:/app/uploads
|
||||
- openflare_uploads:/app/uploads
|
||||
environment:
|
||||
TZ: Asia/Shanghai
|
||||
APP_SESSION_SECRET: 'replace-with-a-long-random-string' # 生产环境请替换为长随机字符串
|
||||
DB_ENABLED: "true"
|
||||
DB_HOST: "postgres"
|
||||
DB_PORT: "5432"
|
||||
DB_USERNAME: "openflare"
|
||||
DB_PASSWORD: "replace-with-strong-password"
|
||||
DB_NAME: "openflare"
|
||||
DB_USERNAME: "${DB_USERNAME:-openflare}"
|
||||
DB_PASSWORD: "${DB_PASSWORD:-replace-with-strong-password}"
|
||||
DB_NAME: "${DB_NAME:-openflare}"
|
||||
REDIS_ENABLED: "true"
|
||||
REDIS_ADDRS: "redis:6379"
|
||||
REDIS_ADDR: "redis:6379"
|
||||
CLICKHOUSE_ENABLED: "true"
|
||||
CLICKHOUSE_HOSTS: "clickhouse:9000"
|
||||
CLICKHOUSE_HOST: "clickhouse:9000"
|
||||
depends_on:
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
@@ -69,13 +77,13 @@ services:
|
||||
image: postgres:17-alpine
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
POSTGRES_DB: openflare
|
||||
POSTGRES_USER: openflare
|
||||
POSTGRES_PASSWORD: replace-with-strong-password
|
||||
POSTGRES_DB: ${DB_NAME:-openflare}
|
||||
POSTGRES_USER: ${DB_USERNAME:-openflare}
|
||||
POSTGRES_PASSWORD: ${DB_PASSWORD:-replace-with-strong-password}
|
||||
volumes:
|
||||
- ./data/postgres_data:/var/lib/postgresql/data
|
||||
- openflare_postgres_data:/var/lib/postgresql/data
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pg_isready -U openflare -d openflare"]
|
||||
test: ["CMD-SHELL", "pg_isready -U ${DB_USERNAME:-openflare} -d ${DB_NAME:-openflare}"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
@@ -85,7 +93,7 @@ services:
|
||||
restart: unless-stopped
|
||||
command: ["valkey-server", "--appendonly", "yes"]
|
||||
volumes:
|
||||
- ./data/valkey:/data
|
||||
- openflare_redis_data:/data
|
||||
healthcheck:
|
||||
test: ["CMD", "valkey-cli", "ping"]
|
||||
interval: 10s
|
||||
@@ -98,17 +106,28 @@ services:
|
||||
environment:
|
||||
CLICKHOUSE_DB: openflare
|
||||
CLICKHOUSE_USER: default
|
||||
CLICKHOUSE_PASSWORD: 123456
|
||||
CLICKHOUSE_PASSWORD: ${CLICKHOUSE_PASSWORD:-replace-with-clickhouse-password}
|
||||
CLICKHOUSE_DEFAULT_ACCESS_MANAGEMENT: 1
|
||||
TZ: Asia/Shanghai
|
||||
ulimits:
|
||||
nofile:
|
||||
soft: 262144
|
||||
hard: 262144
|
||||
volumes:
|
||||
- ./data/clickhouse_data:/var/lib/clickhouse
|
||||
- openflare_clickhouse_data:/var/lib/clickhouse
|
||||
- ./config/clickhouse:/etc/clickhouse-server/config.d:ro
|
||||
healthcheck:
|
||||
test: ["CMD", "clickhouse-client", "--query", "SELECT 1"]
|
||||
test: ["CMD", "clickhouse-client", "--user", "default", "--password", "${CLICKHOUSE_PASSWORD:-replace-with-clickhouse-password}", "--query", "SELECT 1"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
start_period: 15s
|
||||
|
||||
volumes:
|
||||
openflare_uploads:
|
||||
openflare_postgres_data:
|
||||
openflare_redis_data:
|
||||
openflare_clickhouse_data:
|
||||
```
|
||||
|
||||
启动服务:
|
||||
|
||||
@@ -0,0 +1,22 @@
|
||||
# ClickHouse P0–P3 修复计划
|
||||
|
||||
> 状态: 已完成(已合并主工作区,`make code-check` 通过)
|
||||
> 策略: 4 个互不干扰 worktree 并行,最后由主代理合并
|
||||
|
||||
## 任务拆分
|
||||
|
||||
| ID | Worktree 主题 | 范围 | 禁止改动 |
|
||||
|----|---------------|------|----------|
|
||||
| WT1 | P0 清理语义 C1 | cleanup maintenance / delete / tasks | chwriter、dashboard、DDL 新 MV |
|
||||
| WT2 | 写路径 C2+H1+H2+H3 | chwriter、batchwriter、risk_control、model store 分层、status 指标 | goose 迁移、dashboard 读逻辑 |
|
||||
| WT3 | 读路径 H4+H5 | 最新快照查询、metric/openresty 小时 MV + 读路径 | chwriter、cleanup |
|
||||
| WT4 | P3 打磨 | 连接池/async_insert、traffic hourly TTL、UV 语义 | model store 分层、cleanup |
|
||||
|
||||
## 合并顺序
|
||||
|
||||
1. WT1 → 2. WT2 → 3. WT3 → 4. WT4
|
||||
(迁移文件时间戳已错开,changelog 由主代理统一写)
|
||||
|
||||
## 验收
|
||||
|
||||
各 worktree: 相关 `go test` + 可运行部分;合并后 `make code-check`。
|
||||
@@ -0,0 +1,47 @@
|
||||
# ClickHouse CPU 性能优化计划
|
||||
|
||||
> PLAN_ID: `63ba981b`
|
||||
> 状态: 已完成(含 Phase 2 遗留治理)
|
||||
> 目标: 完成 P0–P2 优化,降低 ClickHouse CPU 占用
|
||||
|
||||
## 背景
|
||||
|
||||
ClickHouse CPU 偏高由写入侧(小 part 频繁 flush、心跳同步 DELETE mutation)与查询侧(无 LIMIT 全表扫、高频轮询、WAF 全量拉日志)叠加导致。
|
||||
|
||||
## PR Plan
|
||||
|
||||
### PR 1: 写入路径 P0 优化
|
||||
|
||||
- **Description:** 移除心跳路径同步 `ALTER DELETE`;为 `batchwriter` 增加 `MinBatchSize`;调大可观测 writer 批次与 flush 间隔;为 openresty/frps/frpc 补全去重。
|
||||
- **Files/components affected:** `internal/apps/openflare/agent/observability.go`, `internal/db/batchwriter/`, `internal/apps/openflare/chwriter/`, `internal/db/batchwriter/*_test.go`
|
||||
- **Dependencies:** None
|
||||
|
||||
### PR 2: ClickHouse 客户端与配置 P1
|
||||
|
||||
- **Description:** 启用 `async_insert` 等写入优化 settings;提高 `block_buffer_size` 默认值;更新 `config.example.yaml` 与配置模型注释。
|
||||
- **Files/components affected:** `internal/db/clickhouse.go`, `internal/config/model.go`, `internal/config/config.go`, `config.example.yaml`
|
||||
- **Dependencies:** None
|
||||
|
||||
### PR 3: Dashboard 与可观测查询 P0
|
||||
|
||||
- **Description:** 消除 `limit=0` 无界查询;复用已有限制数据构建趋势;增加服务端短 TTL 缓存;降低前端轮询频率。
|
||||
- **Files/components affected:** `internal/apps/openflare/dashboard/logics.go`, `internal/apps/openflare/observability/node_logics.go`, `frontend/app/(main)/page.tsx`, `frontend/app/(main)/nodes/components/node-observability.tsx`
|
||||
- **Dependencies:** None
|
||||
|
||||
### PR 4: 访问日志与 WAF 查询 P0/P1
|
||||
|
||||
- **Description:** WAF IP 同步改为 ClickHouse 侧聚合;IP 汇总与折叠日志 SQL 分页;消除 count 重复全量扫描;列表 API 强制默认时间窗口。
|
||||
- **Files/components affected:** `internal/apps/openflare/waf/ip_group_sync.go`, `internal/repository/analytics/node_access_log_stats.go`, `internal/model/openflare_access_log.go`, `internal/apps/openflare/observability/access_log_logics.go`, `internal/repository/analytics/access_log_stats.go`
|
||||
- **Dependencies:** None
|
||||
|
||||
### PR 5: ClickHouse DDL 与数据规范化 P1
|
||||
|
||||
- **Description:** 为 7 张分析表添加 TTL;收窄 `of_node_access_logs` ORDER BY;插入时规范化 `remote_addr`(去 trim 查询);将可观测 obs 三表纳入自动清理。
|
||||
- **Files/components affected:** `internal/db/migrator/goose/clickhouse/`, `internal/repository/analytics/node_access_log_writer.go`, `internal/apps/openflare/tasks/database_cleanup.go`, `internal/model/analytics/`
|
||||
- **Dependencies:** PR 1
|
||||
|
||||
### PR 6: 基础设施与审计减负 P2
|
||||
|
||||
- **Description:** Docker ClickHouse 服务端基础调优;审计日志 headers 截断/精简;更新 changelog。
|
||||
- **Files/components affected:** `docker-compose.yaml`, `docker/clickhouse/` (if needed), `internal/apps/risk_control/middleware.go`, `docs/changelog/index.md`
|
||||
- **Dependencies:** None
|
||||
@@ -99,13 +99,21 @@ Server 的所有核心基础配置定义在 `config.yaml` 中,且均支持环
|
||||
| `redis.pool_size` | `REDIS_POOL_SIZE` | Redis 连接池大小 | `100` |
|
||||
|
||||
### 4. ClickHouse 配置 (`clickhouse:`)
|
||||
|
||||
> **说明**:下列为 OpenFlare **客户端**连接参数。ClickHouse **服务端**小规格调优:将 `performance.xml` curl 到 `./config/clickhouse/`,compose 挂载 `./config/clickhouse:/etc/clickhouse-server/config.d:ro`(目录内不要放 listen 配置),详见 [启动 Server](../deployment/server.md)。
|
||||
|
||||
| 配置文件 YAML 路径 | 对应覆盖环境变量 | 作用说明 | 默认值 |
|
||||
| --- | --- | --- | --- |
|
||||
| `clickhouse.enabled` | `CLICKHOUSE_ENABLED` | 是否启用 ClickHouse。**系统节点指标与访问日志在此进行海量写入** | `true` |
|
||||
| `clickhouse.hosts` | `CLICKHOUSE_HOST` | ClickHouse 集群连接地址数组(环境变量仅设置单地址) | `["127.0.0.1:9000"]` |
|
||||
| `clickhouse.username` | `CLICKHOUSE_USERNAME` | ClickHouse 账号用户名 | `default` |
|
||||
| `clickhouse.password` | `CLICKHOUSE_PASSWORD` | ClickHouse 密码 | `123456` |
|
||||
| `clickhouse.password` | `CLICKHOUSE_PASSWORD` | ClickHouse 密码 | `replace-with-clickhouse-password` |
|
||||
| `clickhouse.database` | `CLICKHOUSE_NAME` | ClickHouse 存储的数据库名称 | `openflare` |
|
||||
| `clickhouse.max_idle_conn` | - | 客户端空闲连接数(小规格默认偏低) | `8` |
|
||||
| `clickhouse.max_open_conn` | - | 客户端最大打开连接数 | `16` |
|
||||
| `clickhouse.conn_max_lifetime` | - | 连接最大存活时间(秒) | `3600` |
|
||||
| `clickhouse.dial_timeout` | - | 建连超时(秒) | `5` |
|
||||
| `clickhouse.block_buffer_size` | - | 原生协议 block 缓冲行数 | `32` |
|
||||
|
||||
### 5. 系统日志配置 (`log:`)
|
||||
| 配置文件 YAML 路径 | 对应覆盖环境变量 | 作用说明 | 默认值 |
|
||||
@@ -161,7 +169,7 @@ Server 的所有核心基础配置定义在 `config.yaml` 中,且均支持环
|
||||
### 2. 人机安全校验 (PoW Captcha)
|
||||
| 配置键 (Key) | 数据类型 | 作用说明 | 默认值 |
|
||||
| --- | --- | --- | --- |
|
||||
| `cap_login_enabled` | `bool` | 是否在登录界面强制要求进行本地 PoW 算力防爆破人机验证 | `true` |
|
||||
| `cap_login_enabled` | `bool` | 是否在登录界面强制要求进行本地 PoW 算力防爆破人机验证 | `false` |
|
||||
| `cap_auto_solve` | `bool` | 打开页面后是否由浏览器自动开始后台背景计算算力(无需用户手动点击)| `true` |
|
||||
| `cap_challenge_count` | `int` | 人机验证所需的计算难题数。数量越大,计算要求时间越长(推荐 1~5) | `1` |
|
||||
| `cap_challenge_difficulty`| `int`| 每次计算所需的 PoW 哈希前缀匹配难度。推荐数值在 3-5 之间 | `4` |
|
||||
|
||||
+219
-5
@@ -302,6 +302,7 @@
|
||||
"parameters": [
|
||||
{
|
||||
"type": "integer",
|
||||
"format": "int64",
|
||||
"description": "认证源 ID 或名称",
|
||||
"name": "id",
|
||||
"in": "path",
|
||||
@@ -379,6 +380,7 @@
|
||||
"parameters": [
|
||||
{
|
||||
"type": "integer",
|
||||
"format": "int64",
|
||||
"description": "认证源 ID 或名称",
|
||||
"name": "id",
|
||||
"in": "path",
|
||||
@@ -446,6 +448,7 @@
|
||||
"parameters": [
|
||||
{
|
||||
"type": "integer",
|
||||
"format": "int64",
|
||||
"description": "认证源 ID 或名称",
|
||||
"name": "id",
|
||||
"in": "path",
|
||||
@@ -1356,6 +1359,7 @@
|
||||
"parameters": [
|
||||
{
|
||||
"type": "integer",
|
||||
"format": "int64",
|
||||
"description": "通道ID",
|
||||
"name": "id",
|
||||
"in": "path",
|
||||
@@ -1409,6 +1413,7 @@
|
||||
"parameters": [
|
||||
{
|
||||
"type": "integer",
|
||||
"format": "int64",
|
||||
"description": "通道ID",
|
||||
"name": "id",
|
||||
"in": "path",
|
||||
@@ -1865,6 +1870,67 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"/api/v1/admin/status/clickhouse": {
|
||||
"get": {
|
||||
"security": [
|
||||
{
|
||||
"SessionCookie": []
|
||||
}
|
||||
],
|
||||
"description": "返回 ClickHouse parts、mutation、async_insert 队列等运维指标,需要管理员权限",
|
||||
"produces": [
|
||||
"application/json"
|
||||
],
|
||||
"tags": [
|
||||
"admin"
|
||||
],
|
||||
"summary": "获取 ClickHouse 运行指标",
|
||||
"responses": {
|
||||
"200": {
|
||||
"description": "获取成功",
|
||||
"schema": {
|
||||
"allOf": [
|
||||
{
|
||||
"$ref": "#/definitions/response.Any"
|
||||
},
|
||||
{
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"data": {
|
||||
"$ref": "#/definitions/analytics.ClickHouseOperationalStats"
|
||||
}
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
},
|
||||
"400": {
|
||||
"description": "ClickHouse 未启用",
|
||||
"schema": {
|
||||
"$ref": "#/definitions/response.Any"
|
||||
}
|
||||
},
|
||||
"401": {
|
||||
"description": "未登录",
|
||||
"schema": {
|
||||
"$ref": "#/definitions/response.Any"
|
||||
}
|
||||
},
|
||||
"403": {
|
||||
"description": "无管理员权限",
|
||||
"schema": {
|
||||
"$ref": "#/definitions/response.Any"
|
||||
}
|
||||
},
|
||||
"500": {
|
||||
"description": "内部错误",
|
||||
"schema": {
|
||||
"$ref": "#/definitions/response.Any"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"/api/v1/admin/system-configs": {
|
||||
"get": {
|
||||
"security": [
|
||||
@@ -3361,6 +3427,7 @@
|
||||
},
|
||||
{
|
||||
"type": "integer",
|
||||
"format": "int64",
|
||||
"description": "上传用户 ID",
|
||||
"name": "user_id",
|
||||
"in": "query"
|
||||
@@ -3684,6 +3751,11 @@
|
||||
],
|
||||
"summary": "获取用户列表",
|
||||
"parameters": [
|
||||
{
|
||||
"type": "string",
|
||||
"name": "email",
|
||||
"in": "query"
|
||||
},
|
||||
{
|
||||
"minimum": 1,
|
||||
"type": "integer",
|
||||
@@ -3902,6 +3974,92 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"put": {
|
||||
"security": [
|
||||
{
|
||||
"SessionCookie": []
|
||||
}
|
||||
],
|
||||
"description": "更新指定用户的昵称、邮箱、管理员权限,并可选重置密码,需要管理员权限",
|
||||
"consumes": [
|
||||
"application/json"
|
||||
],
|
||||
"produces": [
|
||||
"application/json"
|
||||
],
|
||||
"tags": [
|
||||
"admin"
|
||||
],
|
||||
"summary": "更新用户信息",
|
||||
"parameters": [
|
||||
{
|
||||
"type": "integer",
|
||||
"description": "用户 ID",
|
||||
"name": "id",
|
||||
"in": "path",
|
||||
"required": true
|
||||
},
|
||||
{
|
||||
"description": "更新参数",
|
||||
"name": "request",
|
||||
"in": "body",
|
||||
"required": true,
|
||||
"schema": {
|
||||
"$ref": "#/definitions/user.updateUserRequest"
|
||||
}
|
||||
}
|
||||
],
|
||||
"responses": {
|
||||
"200": {
|
||||
"description": "更新成功",
|
||||
"schema": {
|
||||
"allOf": [
|
||||
{
|
||||
"$ref": "#/definitions/response.Any"
|
||||
},
|
||||
{
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"data": {
|
||||
"type": "string"
|
||||
}
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
},
|
||||
"400": {
|
||||
"description": "参数错误",
|
||||
"schema": {
|
||||
"$ref": "#/definitions/response.Any"
|
||||
}
|
||||
},
|
||||
"401": {
|
||||
"description": "未登录",
|
||||
"schema": {
|
||||
"$ref": "#/definitions/response.Any"
|
||||
}
|
||||
},
|
||||
"403": {
|
||||
"description": "无管理员权限或尝试修改自身权限",
|
||||
"schema": {
|
||||
"$ref": "#/definitions/response.Any"
|
||||
}
|
||||
},
|
||||
"404": {
|
||||
"description": "用户不存在",
|
||||
"schema": {
|
||||
"$ref": "#/definitions/response.Any"
|
||||
}
|
||||
},
|
||||
"500": {
|
||||
"description": "内部错误",
|
||||
"schema": {
|
||||
"$ref": "#/definitions/response.Any"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"delete": {
|
||||
"security": [
|
||||
{
|
||||
@@ -11137,6 +11295,7 @@
|
||||
"parameters": [
|
||||
{
|
||||
"type": "integer",
|
||||
"format": "int64",
|
||||
"description": "外部帐号绑定记录 ID",
|
||||
"name": "id",
|
||||
"in": "path",
|
||||
@@ -12941,6 +13100,29 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"analytics.ClickHouseOperationalStats": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"active_parts": {
|
||||
"type": "integer"
|
||||
},
|
||||
"async_insert_bytes": {
|
||||
"type": "integer"
|
||||
},
|
||||
"async_insert_queue": {
|
||||
"type": "integer"
|
||||
},
|
||||
"database": {
|
||||
"type": "string"
|
||||
},
|
||||
"pending_mutations": {
|
||||
"type": "integer"
|
||||
},
|
||||
"total_rows": {
|
||||
"type": "integer"
|
||||
}
|
||||
}
|
||||
},
|
||||
"apply_log.CleanupInput": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
@@ -13768,19 +13950,22 @@
|
||||
"source_countries": {
|
||||
"type": "object",
|
||||
"additionalProperties": {
|
||||
"type": "integer"
|
||||
"type": "integer",
|
||||
"format": "int64"
|
||||
}
|
||||
},
|
||||
"status_codes": {
|
||||
"type": "object",
|
||||
"additionalProperties": {
|
||||
"type": "integer"
|
||||
"type": "integer",
|
||||
"format": "int64"
|
||||
}
|
||||
},
|
||||
"top_domains": {
|
||||
"type": "object",
|
||||
"additionalProperties": {
|
||||
"type": "integer"
|
||||
"type": "integer",
|
||||
"format": "int64"
|
||||
}
|
||||
},
|
||||
"unique_visitor_count": {
|
||||
@@ -14210,7 +14395,7 @@
|
||||
"type": "string"
|
||||
},
|
||||
"id": {
|
||||
"type": "integer"
|
||||
"type": "string"
|
||||
},
|
||||
"is_active": {
|
||||
"type": "boolean"
|
||||
@@ -14245,7 +14430,7 @@
|
||||
"type": "string"
|
||||
},
|
||||
"id": {
|
||||
"type": "integer"
|
||||
"type": "string"
|
||||
},
|
||||
"is_active": {
|
||||
"type": "boolean"
|
||||
@@ -15085,6 +15270,11 @@
|
||||
"UploadStatusPending": "待使用",
|
||||
"UploadStatusUsed": "已使用"
|
||||
},
|
||||
"x-enum-descriptions": [
|
||||
"待使用",
|
||||
"已使用",
|
||||
"已删除"
|
||||
],
|
||||
"x-enum-varnames": [
|
||||
"UploadStatusPending",
|
||||
"UploadStatusUsed",
|
||||
@@ -18078,6 +18268,30 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"user.updateUserRequest": {
|
||||
"type": "object",
|
||||
"required": [
|
||||
"email"
|
||||
],
|
||||
"properties": {
|
||||
"email": {
|
||||
"type": "string",
|
||||
"maxLength": 255
|
||||
},
|
||||
"is_admin": {
|
||||
"type": "boolean"
|
||||
},
|
||||
"nickname": {
|
||||
"type": "string",
|
||||
"maxLength": 64
|
||||
},
|
||||
"password": {
|
||||
"type": "string",
|
||||
"maxLength": 64,
|
||||
"minLength": 8
|
||||
}
|
||||
}
|
||||
},
|
||||
"user.updateUserStatusRequest": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
|
||||
+140
-2
@@ -169,6 +169,21 @@ definitions:
|
||||
$ref: '#/definitions/github_com_Rain-kl_Wavelet_pkg_protocol.WAFIPGroup'
|
||||
type: array
|
||||
type: object
|
||||
analytics.ClickHouseOperationalStats:
|
||||
properties:
|
||||
active_parts:
|
||||
type: integer
|
||||
async_insert_bytes:
|
||||
type: integer
|
||||
async_insert_queue:
|
||||
type: integer
|
||||
database:
|
||||
type: string
|
||||
pending_mutations:
|
||||
type: integer
|
||||
total_rows:
|
||||
type: integer
|
||||
type: object
|
||||
apply_log.CleanupInput:
|
||||
properties:
|
||||
delete_all:
|
||||
@@ -713,14 +728,17 @@ definitions:
|
||||
type: integer
|
||||
source_countries:
|
||||
additionalProperties:
|
||||
format: int64
|
||||
type: integer
|
||||
type: object
|
||||
status_codes:
|
||||
additionalProperties:
|
||||
format: int64
|
||||
type: integer
|
||||
type: object
|
||||
top_domains:
|
||||
additionalProperties:
|
||||
format: int64
|
||||
type: integer
|
||||
type: object
|
||||
unique_visitor_count:
|
||||
@@ -1004,7 +1022,7 @@ definitions:
|
||||
created_by:
|
||||
type: string
|
||||
id:
|
||||
type: integer
|
||||
type: string
|
||||
is_active:
|
||||
type: boolean
|
||||
main_config:
|
||||
@@ -1027,7 +1045,7 @@ definitions:
|
||||
created_by:
|
||||
type: string
|
||||
id:
|
||||
type: integer
|
||||
type: string
|
||||
is_active:
|
||||
type: boolean
|
||||
version:
|
||||
@@ -1593,6 +1611,10 @@ definitions:
|
||||
UploadStatusDeleted: 已删除
|
||||
UploadStatusPending: 待使用
|
||||
UploadStatusUsed: 已使用
|
||||
x-enum-descriptions:
|
||||
- 待使用
|
||||
- 已使用
|
||||
- 已删除
|
||||
x-enum-varnames:
|
||||
- UploadStatusPending
|
||||
- UploadStatusUsed
|
||||
@@ -3577,6 +3599,23 @@ definitions:
|
||||
website:
|
||||
type: string
|
||||
type: object
|
||||
user.updateUserRequest:
|
||||
properties:
|
||||
email:
|
||||
maxLength: 255
|
||||
type: string
|
||||
is_admin:
|
||||
type: boolean
|
||||
nickname:
|
||||
maxLength: 64
|
||||
type: string
|
||||
password:
|
||||
maxLength: 64
|
||||
minLength: 8
|
||||
type: string
|
||||
required:
|
||||
- email
|
||||
type: object
|
||||
user.updateUserStatusRequest:
|
||||
properties:
|
||||
is_active:
|
||||
@@ -4085,6 +4124,7 @@ paths:
|
||||
description: 删除指定认证源及其关联的所有外部帐号绑定记录,警告:删除后相关用户将无法通过该源登录,需要管理员权限
|
||||
parameters:
|
||||
- description: 认证源 ID 或名称
|
||||
format: int64
|
||||
in: path
|
||||
name: id
|
||||
required: true
|
||||
@@ -4124,6 +4164,7 @@ paths:
|
||||
description: 更新指定 ID 的认证源配置。若 client_secret 字段为空,则保留原有密钥不变,需要管理员权限
|
||||
parameters:
|
||||
- description: 认证源 ID 或名称
|
||||
format: int64
|
||||
in: path
|
||||
name: id
|
||||
required: true
|
||||
@@ -4174,6 +4215,7 @@ paths:
|
||||
description: 启用或禁用指定认证源。尝试启用时将验证 Client ID 和 Client Secret 是否已配置,需要管理员权限
|
||||
parameters:
|
||||
- description: 认证源 ID 或名称
|
||||
format: int64
|
||||
in: path
|
||||
name: id
|
||||
required: true
|
||||
@@ -4670,6 +4712,7 @@ paths:
|
||||
description: 根据ID删除消息通道,需要管理员权限
|
||||
parameters:
|
||||
- description: 通道ID
|
||||
format: int64
|
||||
in: path
|
||||
name: id
|
||||
required: true
|
||||
@@ -4692,6 +4735,7 @@ paths:
|
||||
description: 修改消息通道配置,需要管理员权限
|
||||
parameters:
|
||||
- description: 通道ID
|
||||
format: int64
|
||||
in: path
|
||||
name: id
|
||||
required: true
|
||||
@@ -5016,6 +5060,42 @@ paths:
|
||||
summary: 获取系统状态信息
|
||||
tags:
|
||||
- admin
|
||||
/api/v1/admin/status/clickhouse:
|
||||
get:
|
||||
description: 返回 ClickHouse parts、mutation、async_insert 队列等运维指标,需要管理员权限
|
||||
produces:
|
||||
- application/json
|
||||
responses:
|
||||
"200":
|
||||
description: 获取成功
|
||||
schema:
|
||||
allOf:
|
||||
- $ref: '#/definitions/response.Any'
|
||||
- properties:
|
||||
data:
|
||||
$ref: '#/definitions/analytics.ClickHouseOperationalStats'
|
||||
type: object
|
||||
"400":
|
||||
description: ClickHouse 未启用
|
||||
schema:
|
||||
$ref: '#/definitions/response.Any'
|
||||
"401":
|
||||
description: 未登录
|
||||
schema:
|
||||
$ref: '#/definitions/response.Any'
|
||||
"403":
|
||||
description: 无管理员权限
|
||||
schema:
|
||||
$ref: '#/definitions/response.Any'
|
||||
"500":
|
||||
description: 内部错误
|
||||
schema:
|
||||
$ref: '#/definitions/response.Any'
|
||||
security:
|
||||
- SessionCookie: []
|
||||
summary: 获取 ClickHouse 运行指标
|
||||
tags:
|
||||
- admin
|
||||
/api/v1/admin/system-configs:
|
||||
get:
|
||||
description: 返回所有系统配置列表,支持按配置类型(system/business)过滤,需要管理员权限
|
||||
@@ -5912,6 +5992,7 @@ paths:
|
||||
name: extension
|
||||
type: string
|
||||
- description: 上传用户 ID
|
||||
format: int64
|
||||
in: query
|
||||
name: user_id
|
||||
type: integer
|
||||
@@ -6108,6 +6189,9 @@ paths:
|
||||
get:
|
||||
description: 分页返回用户列表,支持按用户 ID 和用户名筛选,需要管理员权限
|
||||
parameters:
|
||||
- in: query
|
||||
name: email
|
||||
type: string
|
||||
- in: query
|
||||
minimum: 1
|
||||
name: page
|
||||
@@ -6291,6 +6375,59 @@ paths:
|
||||
summary: 获取用户详情
|
||||
tags:
|
||||
- admin
|
||||
put:
|
||||
consumes:
|
||||
- application/json
|
||||
description: 更新指定用户的昵称、邮箱、管理员权限,并可选重置密码,需要管理员权限
|
||||
parameters:
|
||||
- description: 用户 ID
|
||||
in: path
|
||||
name: id
|
||||
required: true
|
||||
type: integer
|
||||
- description: 更新参数
|
||||
in: body
|
||||
name: request
|
||||
required: true
|
||||
schema:
|
||||
$ref: '#/definitions/user.updateUserRequest'
|
||||
produces:
|
||||
- application/json
|
||||
responses:
|
||||
"200":
|
||||
description: 更新成功
|
||||
schema:
|
||||
allOf:
|
||||
- $ref: '#/definitions/response.Any'
|
||||
- properties:
|
||||
data:
|
||||
type: string
|
||||
type: object
|
||||
"400":
|
||||
description: 参数错误
|
||||
schema:
|
||||
$ref: '#/definitions/response.Any'
|
||||
"401":
|
||||
description: 未登录
|
||||
schema:
|
||||
$ref: '#/definitions/response.Any'
|
||||
"403":
|
||||
description: 无管理员权限或尝试修改自身权限
|
||||
schema:
|
||||
$ref: '#/definitions/response.Any'
|
||||
"404":
|
||||
description: 用户不存在
|
||||
schema:
|
||||
$ref: '#/definitions/response.Any'
|
||||
"500":
|
||||
description: 内部错误
|
||||
schema:
|
||||
$ref: '#/definitions/response.Any'
|
||||
security:
|
||||
- SessionCookie: []
|
||||
summary: 更新用户信息
|
||||
tags:
|
||||
- admin
|
||||
/api/v1/admin/users/{id}/status:
|
||||
put:
|
||||
consumes:
|
||||
@@ -10642,6 +10779,7 @@ paths:
|
||||
description: 解除当前登录用户与指定外部帐号的绑定关系,需要登录
|
||||
parameters:
|
||||
- description: 外部帐号绑定记录 ID
|
||||
format: int64
|
||||
in: path
|
||||
name: id
|
||||
required: true
|
||||
|
||||
@@ -16,7 +16,6 @@ import {
|
||||
AlertDialogHeader,
|
||||
AlertDialogTitle,
|
||||
} from "@/components/ui/alert-dialog"
|
||||
import {Badge} from "@/components/ui/badge"
|
||||
import {Button} from "@/components/ui/button"
|
||||
import {Card, CardContent, CardDescription, CardHeader, CardTitle} from "@/components/ui/card"
|
||||
import {Input} from "@/components/ui/input"
|
||||
@@ -28,9 +27,6 @@ import {ErrorInline} from "@/components/layout/error"
|
||||
import {LoadingStateWithBorder} from "@/components/layout/loading"
|
||||
import type {DatabaseCleanupTarget} from "@/lib/services/openflare"
|
||||
import {NodeService, OptionService, StatusService, UptimeKumaService,} from "@/lib/services/openflare"
|
||||
import {AdminStatusService} from "@/lib/services/admin"
|
||||
import {VersionUpgradeDialog} from "@/app/(main)/components/version-upgrade-dialog"
|
||||
import {adminUpdateStatusQueryKey, openflarePublicStatusQueryKey,} from "@/lib/hooks/use-openflare-server-upgrade"
|
||||
|
||||
import {
|
||||
agentOptionEntries,
|
||||
@@ -47,6 +43,7 @@ import {
|
||||
import {UptimeKumaSiteSelectModal} from "./uptimekuma-site-modal"
|
||||
|
||||
const optionsQueryKey = ["openflare", "options"] as const
|
||||
const openflarePublicStatusQueryKey = ["openflare", "public-status"] as const
|
||||
|
||||
const cleanupTargets: Array<{
|
||||
target: DatabaseCleanupTarget
|
||||
@@ -85,7 +82,6 @@ export function OpenFlareOpsSettings() {
|
||||
label: string
|
||||
} | null>(null)
|
||||
const [cleanupRetentionDays, setCleanupRetentionDays] = useState("")
|
||||
const [versionDialogOpen, setVersionDialogOpen] = useState(false)
|
||||
|
||||
const optionsQuery = useQuery({
|
||||
queryKey: optionsQueryKey,
|
||||
@@ -102,10 +98,6 @@ export function OpenFlareOpsSettings() {
|
||||
queryFn: () => NodeService.getBootstrapToken(),
|
||||
})
|
||||
|
||||
const releaseQuery = useQuery({
|
||||
queryKey: adminUpdateStatusQueryKey,
|
||||
queryFn: () => AdminStatusService.getUpdateStatus(),
|
||||
})
|
||||
|
||||
useEffect(() => {
|
||||
if (!optionsQuery.data) return
|
||||
@@ -605,50 +597,6 @@ export function OpenFlareOpsSettings() {
|
||||
</Card>
|
||||
</div>
|
||||
|
||||
<Card className="border-dashed shadow-none">
|
||||
<CardHeader className="flex flex-row items-center justify-between gap-4">
|
||||
<div>
|
||||
<CardTitle className="text-base">版本信息</CardTitle>
|
||||
<CardDescription>
|
||||
检查上游 GitHub Release 并升级当前服务。
|
||||
</CardDescription>
|
||||
</div>
|
||||
<Button type="button" size="sm" onClick={() => setVersionDialogOpen(true)}>
|
||||
管理升级
|
||||
</Button>
|
||||
</CardHeader>
|
||||
<CardContent className="grid gap-3 sm:grid-cols-2 lg:grid-cols-4">
|
||||
<InfoCell label="当前版本" value={statusQuery.data?.version ?? releaseQuery.data?.current_version ?? "—"} />
|
||||
<InfoCell
|
||||
label="最新 Release"
|
||||
value={releaseQuery.data?.latest_version ?? "—"}
|
||||
/>
|
||||
<div className="rounded-lg border border-dashed px-3 py-2">
|
||||
<p className="text-[10px] uppercase tracking-wider text-muted-foreground">更新状态</p>
|
||||
<div className="mt-2">
|
||||
{releaseQuery.data?.update_available ? (
|
||||
<Badge variant="secondary">有新版本</Badge>
|
||||
) : (
|
||||
<Badge variant="outline">已是最新</Badge>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
<InfoCell
|
||||
label="启动时间"
|
||||
value={
|
||||
statusQuery.data?.start_time
|
||||
? new Date(statusQuery.data.start_time * 1000).toLocaleString()
|
||||
: "—"
|
||||
}
|
||||
/>
|
||||
</CardContent>
|
||||
</Card>
|
||||
|
||||
<VersionUpgradeDialog
|
||||
open={versionDialogOpen}
|
||||
onOpenChange={setVersionDialogOpen}
|
||||
canUpgrade
|
||||
/>
|
||||
|
||||
<UptimeKumaSiteSelectModal
|
||||
open={uptimeKumaModalOpen}
|
||||
|
||||
@@ -7,7 +7,7 @@ import dynamic from "next/dynamic"
|
||||
import {useEffect, useMemo} from "react"
|
||||
import {useQuery} from "@tanstack/react-query"
|
||||
import {Loader2, Settings} from "lucide-react"
|
||||
import {useRouter} from "next/navigation"
|
||||
import {useRouter, useSearchParams} from "next/navigation"
|
||||
import {motion} from "motion/react"
|
||||
|
||||
import {Tabs, TabsContent, TabsList, TabsTrigger} from "@/components/ui/tabs"
|
||||
@@ -64,6 +64,17 @@ function systemConfigMap(configs: SystemConfig[]) {
|
||||
export function AdminSettingsPageClient() {
|
||||
const { user, loading } = useAuth()
|
||||
const router = useRouter()
|
||||
const searchParams = useSearchParams()
|
||||
|
||||
const activeTab = useMemo(() => {
|
||||
const rawTab = searchParams.get("tab")
|
||||
const validTabs = ["openflare-ops", "security", "operation", "system", "other", "status", "info"]
|
||||
return rawTab && validTabs.includes(rawTab) ? rawTab : "openflare-ops"
|
||||
}, [searchParams])
|
||||
|
||||
const handleTabChange = (value: string) => {
|
||||
router.push(`/admin/settings?tab=${value}`)
|
||||
}
|
||||
|
||||
const systemConfigsQuery = useQuery({
|
||||
queryKey: ["admin", "system-configs"],
|
||||
@@ -103,7 +114,7 @@ export function AdminSettingsPageClient() {
|
||||
<h1 className="text-2xl font-semibold tracking-tight">系统设置</h1>
|
||||
</div>
|
||||
</div>
|
||||
<Tabs defaultValue="security" className="w-full">
|
||||
<Tabs value={activeTab} onValueChange={handleTabChange} className="w-full">
|
||||
<TabsList variant="line" className="w-fit inline-flex gap-8 mb-6">
|
||||
<TabsTrigger value="openflare-ops" className="px-0 pb-2 text-xs font-semibold">
|
||||
OpenFlare
|
||||
|
||||
@@ -1,8 +1,20 @@
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
import {Suspense} from "react"
|
||||
import {Loader2} from "lucide-react"
|
||||
import {AdminSettingsPageClient} from "./page-client"
|
||||
|
||||
export default function AdminSettingsPage() {
|
||||
return <AdminSettingsPageClient />
|
||||
}
|
||||
return (
|
||||
<Suspense
|
||||
fallback={
|
||||
<div className="flex items-center justify-center min-h-[400px]">
|
||||
<Loader2 className="size-6 animate-spin text-primary" />
|
||||
</div>
|
||||
}
|
||||
>
|
||||
<AdminSettingsPageClient />
|
||||
</Suspense>
|
||||
)
|
||||
}
|
||||
|
||||
@@ -34,7 +34,7 @@ export function DashboardStatCards({
|
||||
{formatCompactNumber(traffic.request_count)}
|
||||
</div>
|
||||
<p className="text-[10px] text-muted-foreground">
|
||||
独立访客 {formatCompactNumber(traffic.unique_visitors)} · 错误{' '}
|
||||
窗口UV(估) {formatCompactNumber(traffic.unique_visitors)} · 错误{' '}
|
||||
{formatCompactNumber(traffic.error_count)} · 估算 QPS{' '}
|
||||
{traffic.estimated_qps.toFixed(2)}
|
||||
</p>
|
||||
|
||||
@@ -1,223 +0,0 @@
|
||||
'use client';
|
||||
|
||||
import {useEffect} from 'react';
|
||||
import {ExternalLink, Loader2} from 'lucide-react';
|
||||
import ReactMarkdown from 'react-markdown';
|
||||
import remarkGfm from 'remark-gfm';
|
||||
|
||||
import {Badge} from '@/components/ui/badge';
|
||||
import {Button} from '@/components/ui/button';
|
||||
import {Card, CardContent, CardDescription, CardHeader, CardTitle} from '@/components/ui/card';
|
||||
import {
|
||||
AlertDialog,
|
||||
AlertDialogAction,
|
||||
AlertDialogCancel,
|
||||
AlertDialogContent,
|
||||
AlertDialogDescription,
|
||||
AlertDialogFooter,
|
||||
AlertDialogHeader,
|
||||
AlertDialogTitle,
|
||||
AlertDialogTrigger,
|
||||
} from '@/components/ui/alert-dialog';
|
||||
import {Dialog, DialogContent, DialogDescription, DialogHeader, DialogTitle} from '@/components/ui/dialog';
|
||||
import {useOpenFlareServerUpgrade} from '@/lib/hooks/use-openflare-server-upgrade';
|
||||
import type {AppUpdateStatus} from '@/lib/services/admin/types';
|
||||
import {formatDateTime} from '@/lib/utils';
|
||||
import {formatRelativeTime} from '@/app/(main)/nodes/components/node-utils';
|
||||
|
||||
function getUpgradeBadge(update: AppUpdateStatus | null | undefined) {
|
||||
if (!update) {
|
||||
return { label: '未检查', variant: 'outline' as const };
|
||||
}
|
||||
if (update.update_available) {
|
||||
return { label: '可升级', variant: 'secondary' as const };
|
||||
}
|
||||
return { label: '最新', variant: 'default' as const };
|
||||
}
|
||||
|
||||
export function VersionUpgradeDialog({
|
||||
open,
|
||||
onOpenChange,
|
||||
canUpgrade = true,
|
||||
}: {
|
||||
open: boolean;
|
||||
onOpenChange: (open: boolean) => void;
|
||||
canUpgrade?: boolean;
|
||||
}) {
|
||||
const {
|
||||
currentVersion,
|
||||
update,
|
||||
releaseErrorMessage,
|
||||
isInitialLoading,
|
||||
isChecking,
|
||||
isUpgrading,
|
||||
handleOpen,
|
||||
handleCheckRelease,
|
||||
handleUpgrade,
|
||||
} = useOpenFlareServerUpgrade({ open, canUpgrade });
|
||||
|
||||
useEffect(() => {
|
||||
if (open) {
|
||||
handleOpen();
|
||||
}
|
||||
}, [open, handleOpen]);
|
||||
|
||||
const upgradeBadge = getUpgradeBadge(update);
|
||||
const isBusy = isChecking || isUpgrading;
|
||||
|
||||
return (
|
||||
<Dialog open={open} onOpenChange={onOpenChange}>
|
||||
<DialogContent className="sm:max-w-3xl max-h-[90vh] overflow-y-auto">
|
||||
<DialogHeader>
|
||||
<DialogTitle>服务端版本</DialogTitle>
|
||||
<DialogDescription>
|
||||
检查上游 GitHub Release 并升级当前服务。升级开始后服务会短暂重启。
|
||||
</DialogDescription>
|
||||
</DialogHeader>
|
||||
|
||||
<div className="space-y-4">
|
||||
<div className="grid gap-4 md:grid-cols-2">
|
||||
<Card className="border-dashed shadow-none py-4 gap-3">
|
||||
<CardHeader className="px-4 pb-0">
|
||||
<CardTitle className="text-sm">当前版本</CardTitle>
|
||||
</CardHeader>
|
||||
<CardContent className="px-4">
|
||||
<div className="flex flex-wrap items-center gap-2">
|
||||
<p className="text-sm font-medium">{currentVersion}</p>
|
||||
<Badge variant={upgradeBadge.variant}>{upgradeBadge.label}</Badge>
|
||||
</div>
|
||||
</CardContent>
|
||||
</Card>
|
||||
|
||||
<Card className="border-dashed shadow-none py-4 gap-3">
|
||||
<CardHeader className="px-4 pb-0">
|
||||
<CardTitle className="text-sm">最新版本</CardTitle>
|
||||
</CardHeader>
|
||||
<CardContent className="px-4 space-y-3">
|
||||
<p className="text-sm font-medium">{update?.latest_version || '未检查'}</p>
|
||||
{canUpgrade ? (
|
||||
<Button
|
||||
type="button"
|
||||
variant="outline"
|
||||
size="sm"
|
||||
disabled={isBusy}
|
||||
onClick={handleCheckRelease}
|
||||
>
|
||||
{isChecking ? '检查中...' : '检查更新'}
|
||||
</Button>
|
||||
) : null}
|
||||
</CardContent>
|
||||
</Card>
|
||||
</div>
|
||||
|
||||
{isInitialLoading ? (
|
||||
<div className="flex items-center justify-center py-8 text-sm text-muted-foreground">
|
||||
<Loader2 className="size-4 mr-2 animate-spin" />
|
||||
加载版本信息...
|
||||
</div>
|
||||
) : null}
|
||||
|
||||
{!isInitialLoading && releaseErrorMessage ? (
|
||||
<div className="rounded-lg border border-destructive/30 bg-destructive/5 px-4 py-3 text-sm text-destructive">
|
||||
{releaseErrorMessage}
|
||||
</div>
|
||||
) : null}
|
||||
|
||||
{!isInitialLoading && !releaseErrorMessage && !update ? (
|
||||
<div className="rounded-lg border border-dashed px-4 py-8 text-center text-sm text-muted-foreground">
|
||||
尚未检查更新,点击「检查更新」后展示 GitHub Release 信息。
|
||||
</div>
|
||||
) : null}
|
||||
|
||||
{update ? (
|
||||
<Card className="border-dashed shadow-none py-4 gap-3">
|
||||
<CardHeader className="px-4 pb-0">
|
||||
<CardTitle className="text-sm">GitHub Release · {update.latest_version}</CardTitle>
|
||||
<CardDescription>
|
||||
{update.published_at
|
||||
? `发布时间:${formatRelativeTime(update.published_at)} · ${formatDateTime(update.published_at)}`
|
||||
: '未提供发布时间'}
|
||||
</CardDescription>
|
||||
</CardHeader>
|
||||
<CardContent className="px-4 space-y-4">
|
||||
<div className="flex flex-wrap items-center gap-2">
|
||||
<Badge variant={update.update_available ? 'secondary' : 'default'}>
|
||||
{update.update_available ? '发现新版本' : '已经是最新版本'}
|
||||
</Badge>
|
||||
{update.prerelease ? (
|
||||
<Badge variant="secondary">Preview 发布</Badge>
|
||||
) : (
|
||||
<Badge variant="outline">正式发布</Badge>
|
||||
)}
|
||||
{!update.can_upgrade ? (
|
||||
<Badge variant="destructive">当前平台不支持自动升级</Badge>
|
||||
) : null}
|
||||
</div>
|
||||
|
||||
<div className="prose prose-sm dark:prose-invert max-w-none text-sm">
|
||||
<ReactMarkdown remarkPlugins={[remarkGfm]}>
|
||||
{update.release_notes || '暂无更新说明'}
|
||||
</ReactMarkdown>
|
||||
</div>
|
||||
|
||||
{update.release_url ? (
|
||||
<a
|
||||
href={update.release_url}
|
||||
target="_blank"
|
||||
rel="noreferrer"
|
||||
className="inline-flex items-center text-sm text-primary hover:underline"
|
||||
>
|
||||
查看发布详情
|
||||
<ExternalLink className="size-3 ml-1" />
|
||||
</a>
|
||||
) : null}
|
||||
|
||||
{canUpgrade ? (
|
||||
<div className="flex justify-end">
|
||||
<AlertDialog>
|
||||
<AlertDialogTrigger asChild>
|
||||
<Button
|
||||
type="button"
|
||||
disabled={
|
||||
!update.update_available ||
|
||||
isUpgrading ||
|
||||
!update.can_upgrade ||
|
||||
isBusy
|
||||
}
|
||||
>
|
||||
{isUpgrading ? '升级中...' : '立即升级'}
|
||||
</Button>
|
||||
</AlertDialogTrigger>
|
||||
<AlertDialogContent>
|
||||
<AlertDialogHeader>
|
||||
<AlertDialogTitle>升级到 {update.latest_version}?</AlertDialogTitle>
|
||||
<AlertDialogDescription>
|
||||
服务将下载并校验 {update.asset_name},随后替换当前二进制并重启。请确保安装目录可写,且服务允许原地重启。
|
||||
</AlertDialogDescription>
|
||||
</AlertDialogHeader>
|
||||
<AlertDialogFooter>
|
||||
<AlertDialogCancel>取消</AlertDialogCancel>
|
||||
<AlertDialogAction onClick={handleUpgrade}>确认升级</AlertDialogAction>
|
||||
</AlertDialogFooter>
|
||||
</AlertDialogContent>
|
||||
</AlertDialog>
|
||||
</div>
|
||||
) : null}
|
||||
|
||||
{!update.can_upgrade ? (
|
||||
<p className="text-sm text-muted-foreground">
|
||||
{update.current_version === 'dev'
|
||||
? '开发构建没有可比较的 Release 版本,不能执行自动升级。'
|
||||
: update.update_available
|
||||
? '当前平台暂不支持自动替换二进制,请从 Release 页面手动升级。'
|
||||
: '当前版本无需升级。'}
|
||||
</p>
|
||||
) : null}
|
||||
</CardContent>
|
||||
</Card>
|
||||
) : null}
|
||||
</div>
|
||||
</DialogContent>
|
||||
</Dialog>
|
||||
);
|
||||
}
|
||||
@@ -289,7 +289,7 @@ export function NodeObservability({
|
||||
const observabilityQuery = useQuery({
|
||||
queryKey: ['openflare', 'node-observability', nodeId],
|
||||
queryFn: () => NodeService.getObservability(nodeId, { hours: 24, limit: 48 }),
|
||||
refetchInterval: 10000,
|
||||
refetchInterval: 30000,
|
||||
});
|
||||
|
||||
const cleanupMutation = useMutation({
|
||||
@@ -661,7 +661,7 @@ export function NodeObservability({
|
||||
</p>
|
||||
<p className="mt-2 text-sm text-muted-foreground">
|
||||
{trafficSummary
|
||||
? `近 60 秒 · UV ${formatMetricCount(trafficSummary.unique_visitor_count)}`
|
||||
? `近 60 秒 · 窗口UV ${formatMetricCount(trafficSummary.unique_visitor_count)}`
|
||||
: '暂无窗口流量摘要'}
|
||||
</p>
|
||||
</div>
|
||||
|
||||
@@ -28,7 +28,7 @@ export default function OpenFlareDashboardPage() {
|
||||
const overviewQuery = useQuery({
|
||||
queryKey: dashboardQueryKey,
|
||||
queryFn: () => DashboardService.getOverview(),
|
||||
refetchInterval: 30_000,
|
||||
refetchInterval: 60_000,
|
||||
});
|
||||
|
||||
const overview = overviewQuery.data;
|
||||
|
||||
@@ -73,7 +73,7 @@ export function LoginForm({ onOTPStateChange }: { onOTPStateChange?: (show: bool
|
||||
queryFn: () => AuthService.getAuthSources(),
|
||||
})
|
||||
|
||||
const capEnabled = configBool(publicConfigQuery.data?.cap_login_enabled, true)
|
||||
const capEnabled = configBool(publicConfigQuery.data?.cap_login_enabled, false)
|
||||
const capAutoSolve = configBool(publicConfigQuery.data?.cap_auto_solve, true)
|
||||
|
||||
const loginMutation = useMutation({
|
||||
|
||||
@@ -68,7 +68,7 @@ export function RegisterForm() {
|
||||
|
||||
const emailRegisterEnabled = configBool(publicConfigQuery.data?.email_register_verification_enabled, false)
|
||||
|
||||
const capEnabled = configBool(publicConfigQuery.data?.cap_login_enabled, true)
|
||||
const capEnabled = configBool(publicConfigQuery.data?.cap_login_enabled, false)
|
||||
const capAutoSolve = configBool(publicConfigQuery.data?.cap_auto_solve, true)
|
||||
|
||||
const [capScope, setCapScope] = useState<'send_email_code' | 'register'>('send_email_code')
|
||||
|
||||
@@ -245,7 +245,7 @@ export const apiSections: PolicySection[] = [
|
||||
"registration_enabled": "false",
|
||||
"password_login_enabled": "true",
|
||||
"password_register_enabled": "false",
|
||||
"cap_login_enabled": "true",
|
||||
"cap_login_enabled": "false",
|
||||
"oidc_login_enabled": "true"
|
||||
}
|
||||
}`}
|
||||
|
||||
@@ -5,7 +5,7 @@ import {AnimatePresence, motion} from "motion/react"
|
||||
import {useUser} from "@/contexts/user-context"
|
||||
import {Card, CardHeader, CardTitle} from "@/components/ui/card"
|
||||
import {Button} from "@/components/ui/button"
|
||||
import {ArrowRight, ExternalLink, FileText, HelpCircle, Layers, Shield, ShieldCheck, Terminal, User} from "lucide-react"
|
||||
import {ArrowRight, ExternalLink, HelpCircle, Layers, Shield, ShieldCheck, Terminal, User} from "lucide-react"
|
||||
import Link from "next/link"
|
||||
|
||||
export function HomeMain() {
|
||||
@@ -22,21 +22,11 @@ export function HomeMain() {
|
||||
bgColor: "bg-blue-500/10",
|
||||
borderColor: "hover:border-blue-500/30",
|
||||
},
|
||||
{
|
||||
title: "开发接口文档",
|
||||
description: "查看开放平台的 RESTful 接口规格说明",
|
||||
icon: FileText,
|
||||
url: "/docs/api",
|
||||
color: "text-emerald-500",
|
||||
bgColor: "bg-emerald-500/10",
|
||||
borderColor: "hover:border-emerald-500/30",
|
||||
external: true,
|
||||
},
|
||||
{
|
||||
title: "使用文档",
|
||||
description: "学习如何集成 API 及日常操作帮助指南",
|
||||
icon: HelpCircle,
|
||||
url: "/docs/how-to-use",
|
||||
url: "https://open-flare.pages.dev/",
|
||||
color: "text-purple-500",
|
||||
bgColor: "bg-purple-500/10",
|
||||
borderColor: "hover:border-purple-500/30",
|
||||
|
||||
@@ -4,8 +4,6 @@ import {ComponentType, useMemo} from "react"
|
||||
import {useMutation, useQueryClient} from "@tanstack/react-query"
|
||||
import {
|
||||
Bell,
|
||||
Code,
|
||||
CreditCard,
|
||||
Database,
|
||||
FileText,
|
||||
FolderOpen,
|
||||
@@ -61,9 +59,7 @@ const MENU_GROUPS: MenuGroup[] = [
|
||||
{
|
||||
name: "文档菜单",
|
||||
items: [
|
||||
{ path: "/admin/demo", label: "规范示例", description: "内置 UI 组件与设计规范的展示、调试与参考", icon: Code },
|
||||
{ path: "/docs/api", label: "接口文档", description: "系统 Swagger 交互式 API 接口文档", icon: CreditCard },
|
||||
{ path: "/docs/how-to-use", label: "使用文档", description: "面向开发与运营的部署使用指南", icon: FileText },
|
||||
{ path: "https://open-flare.pages.dev/", label: "使用文档", description: "面向开发与运营的部署使用指南", icon: FileText },
|
||||
]
|
||||
}
|
||||
]
|
||||
|
||||
@@ -126,10 +126,10 @@ curl -X POST https://api.example.com/api/v1/auth/register \\
|
||||
</ul>
|
||||
|
||||
<div className="flex flex-wrap gap-4">
|
||||
<Link href="/docs/api">
|
||||
<Link href="https://open-flare.pages.dev/" target="_blank" rel="noopener noreferrer">
|
||||
<Button variant="secondary" className="rounded-full text-xs hover:bg-muted-foreground/10">
|
||||
<Book className="w-3 h-3" />
|
||||
API 文档
|
||||
使用文档
|
||||
</Button>
|
||||
</Link>
|
||||
</div>
|
||||
|
||||
@@ -43,8 +43,7 @@ export const FooterSection = React.memo(function FooterSection({ className }: Fo
|
||||
<div className="lg:col-span-1">
|
||||
<h3 className="font-semibold text-foreground mb-6">开发</h3>
|
||||
<ul className="space-y-4 text-sm text-muted-foreground">
|
||||
<li><FooterLink href="/docs/how-to-use">快速开始</FooterLink></li>
|
||||
<li><FooterLink href="/docs/api">API 文档</FooterLink></li>
|
||||
<li><FooterLink href="https://open-flare.pages.dev/">使用文档</FooterLink></li>
|
||||
<li><FooterLink href="https://github.com/Rain-kl/OpenFlare">源代码</FooterLink></li>
|
||||
</ul>
|
||||
</div>
|
||||
|
||||
@@ -72,7 +72,7 @@ export const HeroSection = React.memo(function HeroSection({ className }: HeroSe
|
||||
</Button>
|
||||
</Link>
|
||||
|
||||
<Link href="/docs/how-to-use" className="w-full sm:w-auto">
|
||||
<Link href="https://open-flare.pages.dev/" target="_blank" rel="noopener noreferrer" className="w-full sm:w-auto">
|
||||
<Button
|
||||
variant="secondary"
|
||||
size="lg"
|
||||
|
||||
@@ -47,8 +47,6 @@ import {
|
||||
ArrowUpRight,
|
||||
Bell,
|
||||
ChevronDown,
|
||||
Code,
|
||||
CreditCard,
|
||||
Database,
|
||||
FileQuestionMark,
|
||||
FileText,
|
||||
@@ -79,9 +77,7 @@ const data = {
|
||||
{ title: "系统设置", url: "/admin/settings", icon: Settings },
|
||||
],
|
||||
document: [
|
||||
{ title: "规范示例", url: "/admin/demo", icon: Code },
|
||||
{ title: "接口文档", url: "/docs/api", icon: CreditCard, external: true },
|
||||
{ title: "使用文档", url: "/docs/how-to-use", icon: FileText, external: true },
|
||||
{ title: "使用文档", url: "https://open-flare.pages.dev/", icon: FileText, external: true },
|
||||
],
|
||||
}
|
||||
|
||||
@@ -281,7 +277,7 @@ export function AppSidebar({ ...props }: React.ComponentProps<typeof Sidebar>) {
|
||||
</DropdownMenuItem>
|
||||
<DropdownMenuSeparator className="my-2" />
|
||||
<DropdownMenuItem onClick={() => {
|
||||
router.push("/docs/how-to-use")
|
||||
window.open("https://open-flare.pages.dev/", "_blank", "noopener,noreferrer")
|
||||
handleCloseSidebar()
|
||||
}}>
|
||||
<FileQuestionMark className="mr-2 size-4" />
|
||||
|
||||
@@ -1,127 +0,0 @@
|
||||
'use client';
|
||||
|
||||
import {useMutation, useQuery} from '@tanstack/react-query';
|
||||
import {useCallback, useEffect, useRef, useState} from 'react';
|
||||
|
||||
import {AdminStatusService} from '@/lib/services/admin';
|
||||
import type {AppUpdateStatus} from '@/lib/services/admin/types';
|
||||
import {StatusService} from '@/lib/services/openflare';
|
||||
|
||||
export const openflarePublicStatusQueryKey = ['openflare', 'public-status'] as const;
|
||||
|
||||
export const adminUpdateStatusQueryKey = ['admin', 'update'] as const;
|
||||
|
||||
export function useOpenFlareServerUpgrade({
|
||||
open,
|
||||
canUpgrade,
|
||||
}: {
|
||||
open: boolean;
|
||||
canUpgrade: boolean;
|
||||
}) {
|
||||
const [feedback, setFeedback] = useState<string | null>(null);
|
||||
|
||||
const upgradeReloadStartedRef = useRef(false);
|
||||
const upgradeReloadTimerRef = useRef<number | null>(null);
|
||||
|
||||
const statusQuery = useQuery({
|
||||
queryKey: openflarePublicStatusQueryKey,
|
||||
queryFn: () => StatusService.getPublicStatus(),
|
||||
enabled: open,
|
||||
});
|
||||
|
||||
const updateQuery = useQuery({
|
||||
queryKey: adminUpdateStatusQueryKey,
|
||||
queryFn: () => AdminStatusService.getUpdateStatus(),
|
||||
enabled: open && canUpgrade,
|
||||
staleTime: 5 * 60 * 1000,
|
||||
});
|
||||
|
||||
const scheduleUpgradePageReload = useCallback(() => {
|
||||
if (upgradeReloadStartedRef.current) {
|
||||
return;
|
||||
}
|
||||
|
||||
upgradeReloadStartedRef.current = true;
|
||||
setFeedback('服务升级已进入重启阶段,页面将在服务恢复后自动刷新。');
|
||||
|
||||
const reloadWhenServerReady = async () => {
|
||||
try {
|
||||
await StatusService.getPublicStatus();
|
||||
window.location.reload();
|
||||
} catch {
|
||||
upgradeReloadTimerRef.current = window.setTimeout(reloadWhenServerReady, 1500);
|
||||
}
|
||||
};
|
||||
|
||||
upgradeReloadTimerRef.current = window.setTimeout(reloadWhenServerReady, 1200);
|
||||
}, []);
|
||||
|
||||
useEffect(() => {
|
||||
return () => {
|
||||
if (upgradeReloadTimerRef.current !== null) {
|
||||
window.clearTimeout(upgradeReloadTimerRef.current);
|
||||
}
|
||||
};
|
||||
}, []);
|
||||
|
||||
const upgradeMutation = useMutation({
|
||||
mutationFn: () => AdminStatusService.applyUpdate(),
|
||||
onSuccess: () => {
|
||||
scheduleUpgradePageReload();
|
||||
setFeedback('升级包已校验完成,服务正在重启。');
|
||||
},
|
||||
onError: (error) => {
|
||||
setFeedback(error instanceof Error ? error.message : '升级失败,请稍后重试。');
|
||||
},
|
||||
});
|
||||
|
||||
const resetTransientState = useCallback(() => {
|
||||
setFeedback(null);
|
||||
upgradeReloadStartedRef.current = false;
|
||||
}, []);
|
||||
|
||||
const handleOpen = useCallback(() => {
|
||||
resetTransientState();
|
||||
if (canUpgrade) {
|
||||
void updateQuery.refetch();
|
||||
}
|
||||
}, [canUpgrade, resetTransientState, updateQuery]);
|
||||
|
||||
const handleCheckRelease = useCallback(() => {
|
||||
setFeedback(null);
|
||||
if (!canUpgrade) {
|
||||
return;
|
||||
}
|
||||
void updateQuery.refetch();
|
||||
}, [canUpgrade, updateQuery]);
|
||||
|
||||
const handleUpgrade = useCallback(() => {
|
||||
setFeedback(null);
|
||||
upgradeMutation.mutate();
|
||||
}, [upgradeMutation]);
|
||||
|
||||
const update = updateQuery.data;
|
||||
const releaseErrorMessage =
|
||||
feedback ||
|
||||
(updateQuery.isError
|
||||
? updateQuery.error instanceof Error
|
||||
? updateQuery.error.message
|
||||
: '版本检查失败,请稍后重试。'
|
||||
: undefined);
|
||||
|
||||
const currentVersion = statusQuery.data?.version || update?.current_version || 'unknown';
|
||||
|
||||
return {
|
||||
currentVersion,
|
||||
update,
|
||||
releaseErrorMessage,
|
||||
isInitialLoading: updateQuery.isLoading && !updateQuery.data && canUpgrade,
|
||||
isChecking: updateQuery.isFetching,
|
||||
isUpgrading: upgradeMutation.isPending,
|
||||
handleOpen,
|
||||
handleCheckRelease,
|
||||
handleUpgrade,
|
||||
};
|
||||
}
|
||||
|
||||
export type {AppUpdateStatus};
|
||||
@@ -56,23 +56,121 @@ export const searchData: SearchItem[] = [
|
||||
},
|
||||
|
||||
// ==================== 文档库 ====================
|
||||
{
|
||||
id: 'docs-api',
|
||||
title: '开发接口文档',
|
||||
description: '查看 RESTful API 接口规格定义',
|
||||
url: '/docs/api',
|
||||
category: 'page',
|
||||
keywords: ['api', 'docs', '文档', '接口', 'specification'],
|
||||
},
|
||||
{
|
||||
id: 'docs-how-to-use',
|
||||
title: '使用帮助文档',
|
||||
description: '查看新手教程和集成示例',
|
||||
url: '/docs/how-to-use',
|
||||
url: 'https://open-flare.pages.dev/',
|
||||
category: 'page',
|
||||
keywords: ['docs', '文档', '使用', 'how to', 'tutorial', '教程', 'help'],
|
||||
},
|
||||
|
||||
// ==================== 业务控制台 ====================
|
||||
{
|
||||
id: 'console-nodes',
|
||||
title: '节点管理',
|
||||
description: '管理边缘节点、中继节点与内网穿透通道',
|
||||
url: '/nodes',
|
||||
category: 'page',
|
||||
keywords: ['node', '节点', '边缘节点', '中继', '内网穿透', 'tunnel', '服务器'],
|
||||
},
|
||||
{
|
||||
id: 'console-proxy-routes',
|
||||
title: '规则管理',
|
||||
description: '配置反向代理、路由匹配规则、WAF 策略与缓存设置',
|
||||
url: '/proxy-routes',
|
||||
category: 'page',
|
||||
keywords: ['route', '规则', '路由', '代理', '反向代理', 'proxy'],
|
||||
},
|
||||
{
|
||||
id: 'console-websites',
|
||||
title: '域名列表',
|
||||
description: '管理托管域名及证书绑定与监听配置',
|
||||
url: '/websites',
|
||||
category: 'page',
|
||||
keywords: ['website', 'domain', '网站', '域名', '站点'],
|
||||
},
|
||||
{
|
||||
id: 'console-certificates',
|
||||
title: 'TLS 证书',
|
||||
description: '申请与管理 SSL/TLS 证书,支持自动续期',
|
||||
url: '/certificates',
|
||||
category: 'page',
|
||||
keywords: ['certificate', 'ssl', 'tls', '证书', 'https', '加密'],
|
||||
},
|
||||
{
|
||||
id: 'console-dns-accounts',
|
||||
title: 'DNS 账号',
|
||||
description: '配置 DNS 服务商 API 凭证以自动申请证书及管理解析',
|
||||
url: '/dns-accounts',
|
||||
category: 'page',
|
||||
keywords: ['dns', 'dns account', '账号', '域名解析', 'cloudflare', 'aliyun', 'tencent'],
|
||||
},
|
||||
{
|
||||
id: 'console-origins',
|
||||
title: '源站地址',
|
||||
description: '管理反向代理的目标后端服务器与负载均衡组',
|
||||
url: '/origins',
|
||||
category: 'page',
|
||||
keywords: ['origin', '源站', '后端', 'backend', '服务器', '负载均衡'],
|
||||
},
|
||||
{
|
||||
id: 'console-waf',
|
||||
title: 'WAF 防火墙',
|
||||
description: '配置 Web 应用防火墙规则,阻断恶意请求',
|
||||
url: '/waf',
|
||||
category: 'page',
|
||||
keywords: ['waf', '防火墙', '安全', 'security', '拦截', '规则'],
|
||||
},
|
||||
{
|
||||
id: 'console-ip-groups',
|
||||
title: 'IP 组',
|
||||
description: '定义 IP 地址列表以在 WAF 或路由中实现黑白名单控制',
|
||||
url: '/ip-groups',
|
||||
category: 'page',
|
||||
keywords: ['ip', 'ip group', 'ip组', '黑名单', '白名单', '访问控制'],
|
||||
},
|
||||
{
|
||||
id: 'console-pages',
|
||||
title: 'Pages 静态托管',
|
||||
description: '上传或部署静态网页,提供全球 CDN 加速托管',
|
||||
url: '/pages',
|
||||
category: 'page',
|
||||
keywords: ['pages', '静态托管', 'cdn', '网站', '部署', 'static'],
|
||||
},
|
||||
{
|
||||
id: 'console-config-versions',
|
||||
title: '版本发布',
|
||||
description: '查看、对比、发布与回滚系统配置版本',
|
||||
url: '/config-versions',
|
||||
category: 'page',
|
||||
keywords: ['version', 'config', '版本', '发布', '回滚', '对比', '部署'],
|
||||
},
|
||||
{
|
||||
id: 'console-access-logs',
|
||||
title: '访问日志',
|
||||
description: '查看并检索全量网站访问请求日志与网络分析数据',
|
||||
url: '/access-logs',
|
||||
category: 'page',
|
||||
keywords: ['log', 'logs', '访问日志', '分析', '流量', '请求'],
|
||||
},
|
||||
{
|
||||
id: 'console-apply-logs',
|
||||
title: '应用记录',
|
||||
description: '查看节点配置下发、同步与生效的历史记录',
|
||||
url: '/apply-logs',
|
||||
category: 'page',
|
||||
keywords: ['apply', 'log', 'logs', '应用记录', '配置下发', '同步', '部署历史'],
|
||||
},
|
||||
{
|
||||
id: 'console-performance',
|
||||
title: '性能调优',
|
||||
description: '调优网络连接、代理超时与核心系统性能参数',
|
||||
url: '/performance',
|
||||
category: 'page',
|
||||
keywords: ['performance', '性能', '调优', '优化', '参数', '连接', '超时'],
|
||||
},
|
||||
|
||||
// ==================== 个人设置 ====================
|
||||
{
|
||||
id: 'settings',
|
||||
@@ -98,7 +196,6 @@ export const searchData: SearchItem[] = [
|
||||
category: 'setting',
|
||||
keywords: ['appearance', '外观', '主题', 'theme', 'dark', 'light'],
|
||||
},
|
||||
// ==================== 管理员 ====================
|
||||
{
|
||||
id: 'admin-settings',
|
||||
title: '系统设置',
|
||||
@@ -131,6 +228,38 @@ export const searchData: SearchItem[] = [
|
||||
category: 'admin',
|
||||
keywords: ['admin', '管理员', '任务', '异步', 'tasks', 'scheduler', 'worker'],
|
||||
},
|
||||
{
|
||||
id: 'admin-files',
|
||||
title: '存储管理',
|
||||
description: '查看、检索与清理上传到对象存储中的文件 (管理员专属)',
|
||||
url: '/admin/files',
|
||||
category: 'admin',
|
||||
keywords: ['admin', '管理员', '存储', '文件', 'files', 'upload', 's3'],
|
||||
},
|
||||
{
|
||||
id: 'admin-database',
|
||||
title: '数据管理',
|
||||
description: '监控数据库表大小、分页浏览物理表内容并支持交互式 SQL (管理员专属)',
|
||||
url: '/admin/database',
|
||||
category: 'admin',
|
||||
keywords: ['admin', '管理员', '数据库', 'database', 'sql', 'query', 'gorm'],
|
||||
},
|
||||
{
|
||||
id: 'admin-push',
|
||||
title: '通知推送',
|
||||
description: '配置与下发邮件、Lark 和 Telegram 渠道通知推送 (管理员专属)',
|
||||
url: '/admin/push',
|
||||
category: 'admin',
|
||||
keywords: ['admin', '管理员', '推送', '通知', 'push', 'mail', 'telegram', 'lark'],
|
||||
},
|
||||
{
|
||||
id: 'admin-logs',
|
||||
title: '系统日志',
|
||||
description: '查看系统日志与后台异步任务执行日志 (管理员专属)',
|
||||
url: '/admin/logs',
|
||||
category: 'admin',
|
||||
keywords: ['admin', '管理员', '日志', 'logs', 'system log', 'terminal'],
|
||||
},
|
||||
]
|
||||
|
||||
/**
|
||||
|
||||
@@ -248,7 +248,7 @@ func enrichAccessLogsWithUsers(ctx context.Context, list []accessLogItem) {
|
||||
// @Router /api/v1/admin/logs/access [get]
|
||||
func GetAccessLogs(c *gin.Context) {
|
||||
ctx := c.Request.Context()
|
||||
if !config.Config.ClickHouse.Enabled || db.ChDB(ctx) == nil {
|
||||
if !config.Config.ClickHouse.Enabled || !db.ChConnReady() {
|
||||
response.AbortWithError(c, http.StatusBadRequest, "ClickHouse 存储服务未启用,无法检索访问日志")
|
||||
return
|
||||
}
|
||||
@@ -348,7 +348,7 @@ type logsAnalyticsResponse struct {
|
||||
// @Router /api/v1/admin/logs/analytics [get]
|
||||
func GetLogsAnalytics(c *gin.Context) {
|
||||
ctx := c.Request.Context()
|
||||
if !config.Config.ClickHouse.Enabled || db.ChDB(ctx) == nil {
|
||||
if !config.Config.ClickHouse.Enabled || !db.ChConnReady() {
|
||||
response.AbortWithError(c, http.StatusBadRequest, "ClickHouse 存储服务未启用,无法获取分析数据")
|
||||
return
|
||||
}
|
||||
|
||||
@@ -0,0 +1,53 @@
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
package status
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
|
||||
"github.com/Rain-kl/Wavelet/internal/apps/openflare/chwriter"
|
||||
"github.com/Rain-kl/Wavelet/internal/apps/risk_control"
|
||||
"github.com/Rain-kl/Wavelet/internal/common/response"
|
||||
"github.com/Rain-kl/Wavelet/internal/config"
|
||||
"github.com/Rain-kl/Wavelet/internal/db"
|
||||
"github.com/Rain-kl/Wavelet/internal/db/batchwriter"
|
||||
analyticsrepo "github.com/Rain-kl/Wavelet/internal/repository/analytics"
|
||||
"github.com/gin-gonic/gin"
|
||||
)
|
||||
|
||||
// GetClickHouseStatus returns ClickHouse operational metrics for administrators.
|
||||
// @Summary 获取 ClickHouse 运行指标
|
||||
// @Description 返回 ClickHouse parts、mutation、async_insert 队列及进程内 batch writer 指标,需要管理员权限
|
||||
// @Tags admin
|
||||
// @Produce json
|
||||
// @Security SessionCookie
|
||||
// @Success 200 {object} response.Any{data=analyticsrepo.ClickHouseOperationalStats} "获取成功"
|
||||
// @Failure 400 {object} response.Any "ClickHouse 未启用"
|
||||
// @Failure 401 {object} response.Any "未登录"
|
||||
// @Failure 403 {object} response.Any "无管理员权限"
|
||||
// @Failure 500 {object} response.Any "内部错误"
|
||||
// @Router /api/v1/admin/status/clickhouse [get]
|
||||
func GetClickHouseStatus(c *gin.Context) {
|
||||
if !config.Config.ClickHouse.Enabled || !db.ChConnReady() {
|
||||
response.AbortWithError(c, http.StatusBadRequest, "ClickHouse 存储服务未启用")
|
||||
return
|
||||
}
|
||||
|
||||
stats, err := analyticsrepo.GetClickHouseOperationalStats(c.Request.Context())
|
||||
if err != nil {
|
||||
response.AbortInternal(c, "获取 ClickHouse 运行指标失败")
|
||||
return
|
||||
}
|
||||
stats.BatchWriters = collectBatchWriterStats()
|
||||
c.JSON(http.StatusOK, response.OK(stats))
|
||||
}
|
||||
|
||||
func collectBatchWriterStats() []batchwriter.Stats {
|
||||
out := chwriter.WriterStats()
|
||||
if out == nil {
|
||||
out = make([]batchwriter.Stats, 0, 1)
|
||||
}
|
||||
out = append(out, risk_control.LogWriterStats())
|
||||
return out
|
||||
}
|
||||
@@ -144,6 +144,8 @@ func expectedAssetNames(repository, tag string) []string {
|
||||
extension = "zip"
|
||||
}
|
||||
names = append(names, fmt.Sprintf("%s_%s_%s_%s.%s", repoName, tag, runtime.GOOS, runtime.GOARCH, extension))
|
||||
names = append(names, fmt.Sprintf("%s_%s_%s_%s.%s", strings.ToLower(repoName), tag, runtime.GOOS, runtime.GOARCH, extension))
|
||||
names = append(names, fmt.Sprintf("%s-server_%s_%s_%s.%s", strings.ToLower(repoName), tag, runtime.GOOS, runtime.GOARCH, extension))
|
||||
}
|
||||
}
|
||||
return names
|
||||
|
||||
@@ -56,13 +56,13 @@ func TestProtectionEnabledReflectsLoginSwitch(t *testing.T) {
|
||||
|
||||
ResetRuntimeSettingsForTest()
|
||||
|
||||
if !ProtectionEnabled(ctx) {
|
||||
t.Fatal("ProtectionEnabled() = false, want true from seed defaults")
|
||||
if ProtectionEnabled(ctx) {
|
||||
t.Fatal("ProtectionEnabled() = true, want false from seed defaults")
|
||||
}
|
||||
|
||||
if err := db.DB(ctx).Model(&model.SystemConfig{}).
|
||||
Where("key = ?", model.ConfigKeyCapLoginEnabled).
|
||||
Update("value", "false").Error; err != nil {
|
||||
Update("value", "true").Error; err != nil {
|
||||
t.Fatalf("Update(cap_login_enabled) error = %v", err)
|
||||
}
|
||||
if err := repository.InvalidateSystemConfigCache(ctx, model.ConfigKeyCapLoginEnabled); err != nil {
|
||||
@@ -70,8 +70,8 @@ func TestProtectionEnabledReflectsLoginSwitch(t *testing.T) {
|
||||
}
|
||||
InvalidateRuntimeSettings()
|
||||
|
||||
if ProtectionEnabled(ctx) {
|
||||
t.Fatal("ProtectionEnabled() = true, want false after config update")
|
||||
if !ProtectionEnabled(ctx) {
|
||||
t.Fatal("ProtectionEnabled() = false, want true after config update")
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -24,8 +24,6 @@ const (
|
||||
healthSeverityInfo = "info"
|
||||
healthSeverityWarning = "warning"
|
||||
healthSeverityCritical = "critical"
|
||||
nodeAccessLogRetentionDays = 90
|
||||
nodeAccessLogRetentionWindow = nodeAccessLogRetentionDays * 24 * time.Hour
|
||||
accessLogPathMaxLength = 100
|
||||
healthEventMessageMaxLength = 4096
|
||||
)
|
||||
@@ -237,15 +235,11 @@ func buildNodeAccessLogRecords(nodeID string, direct []NodeAccessLog, buffered [
|
||||
return records, nil
|
||||
}
|
||||
|
||||
func persistNodeAccessLogs(ctx context.Context, nodeID string, records []*model.OpenFlareAccessLog, reportedAt time.Time) error {
|
||||
func persistNodeAccessLogs(ctx context.Context, _ string, records []*model.OpenFlareAccessLog, _ time.Time) error {
|
||||
if len(records) == 0 {
|
||||
return nil
|
||||
}
|
||||
if err := model.InsertOpenFlareAccessLogsBatch(ctx, records); err != nil {
|
||||
return err
|
||||
}
|
||||
_, err := model.DeleteOpenFlareAccessLogsByNodeBefore(ctx, nodeID, reportedAt.Add(-nodeAccessLogRetentionWindow))
|
||||
return err
|
||||
return model.InsertOpenFlareAccessLogsBatch(ctx, records)
|
||||
}
|
||||
|
||||
func reconcileNodeHealthEvents(tx *gorm.DB, nodeID string, events []NodeHealthEvent, reportedAt time.Time) error {
|
||||
|
||||
@@ -25,7 +25,7 @@ func newDedupSet() *dedupSet {
|
||||
|
||||
// markIfNew records key when it has not been seen within dedupTTL.
|
||||
func (s *dedupSet) markIfNew(key string) bool {
|
||||
if key == "" {
|
||||
if s == nil || key == "" {
|
||||
return false
|
||||
}
|
||||
|
||||
@@ -33,19 +33,33 @@ func (s *dedupSet) markIfNew(key string) bool {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
|
||||
// Periodically clean up all expired keys (e.g., every 30 seconds)
|
||||
if now.Sub(s.lastCleanup) >= 30*time.Second {
|
||||
for existing, expiresAt := range s.keys {
|
||||
if now.After(expiresAt) {
|
||||
delete(s.keys, existing)
|
||||
}
|
||||
}
|
||||
s.lastCleanup = now
|
||||
}
|
||||
s.cleanupExpiredLocked(now)
|
||||
|
||||
if expiresAt, exists := s.keys[key]; exists && now.Before(expiresAt) {
|
||||
return false
|
||||
}
|
||||
s.keys[key] = now.Add(dedupTTL)
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
// unmark removes a key so a later enqueue or flush retry may accept it again.
|
||||
func (s *dedupSet) unmark(key string) {
|
||||
if s == nil || key == "" {
|
||||
return
|
||||
}
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
delete(s.keys, key)
|
||||
}
|
||||
|
||||
func (s *dedupSet) cleanupExpiredLocked(now time.Time) {
|
||||
if now.Sub(s.lastCleanup) < 30*time.Second {
|
||||
return
|
||||
}
|
||||
for existing, expiresAt := range s.keys {
|
||||
if now.After(expiresAt) {
|
||||
delete(s.keys, existing)
|
||||
}
|
||||
}
|
||||
s.lastCleanup = now
|
||||
}
|
||||
|
||||
@@ -3,7 +3,16 @@
|
||||
|
||||
package chwriter
|
||||
|
||||
import "testing"
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/Rain-kl/Wavelet/internal/db/batchwriter"
|
||||
analyticsmodel "github.com/Rain-kl/Wavelet/internal/model/analytics"
|
||||
)
|
||||
|
||||
func TestDedupSetMarkIfNew(t *testing.T) {
|
||||
t.Parallel()
|
||||
@@ -21,4 +30,157 @@ func TestDedupSetMarkIfNew(t *testing.T) {
|
||||
if set.markIfNew("") {
|
||||
t.Fatal("markIfNew() = true, want false on empty key")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestDedupSetUnmarkAllowsRetry(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
set := newDedupSet()
|
||||
if !set.markIfNew("k") {
|
||||
t.Fatal("markIfNew() = false, want true")
|
||||
}
|
||||
set.unmark("k")
|
||||
if !set.markIfNew("k") {
|
||||
t.Fatal("markIfNew() after unmark = false, want true")
|
||||
}
|
||||
}
|
||||
|
||||
func TestQueueWithDedupDoesNotMarkWhenEnqueueFails(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
cfg := batchwriter.DefaultConfig()
|
||||
cfg.QueueSize = 1
|
||||
cfg.MaxBatchSize = 10
|
||||
cfg.FlushInterval = time.Hour
|
||||
|
||||
// Block the worker so the queue stays full after one enqueue.
|
||||
block := make(chan struct{})
|
||||
writer, err := batchwriter.New[int](cfg, func(context.Context, []int) error {
|
||||
<-block
|
||||
return nil
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("New() error = %v", err)
|
||||
}
|
||||
writer.Start(context.Background())
|
||||
t.Cleanup(func() {
|
||||
close(block)
|
||||
stopCtx, cancel := context.WithTimeout(context.Background(), time.Second)
|
||||
defer cancel()
|
||||
_ = writer.Stop(stopCtx)
|
||||
})
|
||||
|
||||
// Fill the channel buffer (and the worker's current receive slot may empty one).
|
||||
// Keep enqueueing until full so subsequent queueWithDedup fails.
|
||||
for i := 0; i < cfg.QueueSize+2; i++ {
|
||||
_ = writer.TryEnqueue(i)
|
||||
if writer.IsFull() {
|
||||
break
|
||||
}
|
||||
}
|
||||
if !writer.IsFull() {
|
||||
t.Fatal("writer not full after filling; cannot test enqueue failure path")
|
||||
}
|
||||
|
||||
dedup := newDedupSet()
|
||||
queueWithDedup(writer, dedup, "dedup-key", 99)
|
||||
// Key must not remain marked after failed enqueue.
|
||||
if !dedup.markIfNew("dedup-key") {
|
||||
t.Fatal("dedup key still marked after failed enqueue; want unmark")
|
||||
}
|
||||
}
|
||||
|
||||
func TestQueueWithDedupMarksOnlyOnSuccess(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
cfg := batchwriter.DefaultConfig()
|
||||
cfg.MaxBatchSize = 100
|
||||
cfg.FlushInterval = time.Hour
|
||||
|
||||
writer, err := batchwriter.New[int](cfg, func(context.Context, []int) error { return nil })
|
||||
if err != nil {
|
||||
t.Fatalf("New() error = %v", err)
|
||||
}
|
||||
writer.Start(context.Background())
|
||||
t.Cleanup(func() {
|
||||
stopCtx, cancel := context.WithTimeout(context.Background(), time.Second)
|
||||
defer cancel()
|
||||
_ = writer.Stop(stopCtx)
|
||||
})
|
||||
|
||||
dedup := newDedupSet()
|
||||
queueWithDedup(writer, dedup, "ok-key", 1)
|
||||
if dedup.markIfNew("ok-key") {
|
||||
t.Fatal("markIfNew() = true after successful enqueue, want false (key marked)")
|
||||
}
|
||||
}
|
||||
|
||||
func TestFlushErrorHandlerUnmarksKeys(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dedup := newDedupSet()
|
||||
flushErr := errors.New("ch down")
|
||||
|
||||
var (
|
||||
mu sync.Mutex
|
||||
errCount int
|
||||
)
|
||||
|
||||
cfg := batchwriter.Config{
|
||||
Name: "test_obs",
|
||||
QueueSize: 10,
|
||||
MaxBatchSize: 1,
|
||||
FlushInterval: time.Hour,
|
||||
}
|
||||
keyFn := func(s analyticsmodel.NodeMetricSnapshot) string {
|
||||
return metricSnapshotKey(s)
|
||||
}
|
||||
writer, err := batchwriter.New(
|
||||
cfg,
|
||||
func(context.Context, []analyticsmodel.NodeMetricSnapshot) error { return flushErr },
|
||||
batchwriter.WithFlushErrorHandler[analyticsmodel.NodeMetricSnapshot](func(_ context.Context, items []analyticsmodel.NodeMetricSnapshot, err error) {
|
||||
mu.Lock()
|
||||
errCount++
|
||||
mu.Unlock()
|
||||
for _, item := range items {
|
||||
dedup.unmark(keyFn(item))
|
||||
}
|
||||
}),
|
||||
)
|
||||
if err != nil {
|
||||
t.Fatalf("New() error = %v", err)
|
||||
}
|
||||
writer.Start(context.Background())
|
||||
t.Cleanup(func() {
|
||||
stopCtx, cancel := context.WithTimeout(context.Background(), time.Second)
|
||||
defer cancel()
|
||||
_ = writer.Stop(stopCtx)
|
||||
})
|
||||
|
||||
item := analyticsmodel.NodeMetricSnapshot{
|
||||
NodeID: "n1",
|
||||
CapturedAt: time.Unix(1, 0).UTC(),
|
||||
}
|
||||
key := keyFn(item)
|
||||
if !dedup.markIfNew(key) {
|
||||
t.Fatal("markIfNew failed")
|
||||
}
|
||||
if !writer.TryEnqueue(item) {
|
||||
t.Fatal("TryEnqueue failed")
|
||||
}
|
||||
|
||||
deadline := time.Now().Add(time.Second)
|
||||
for {
|
||||
mu.Lock()
|
||||
ready := errCount >= 1
|
||||
mu.Unlock()
|
||||
if ready || time.Now().After(deadline) {
|
||||
break
|
||||
}
|
||||
time.Sleep(5 * time.Millisecond)
|
||||
}
|
||||
|
||||
if !dedup.markIfNew(key) {
|
||||
t.Fatal("key still marked after flush error unmark; want available for retry")
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,86 @@
|
||||
//go:build live_ch
|
||||
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
package chwriter_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/Rain-kl/Wavelet/internal/apps/openflare/chwriter"
|
||||
"github.com/Rain-kl/Wavelet/internal/db"
|
||||
"github.com/Rain-kl/Wavelet/internal/model"
|
||||
)
|
||||
|
||||
// Run with Docker ClickHouse + config.yaml:
|
||||
//
|
||||
// go test -tags live_ch ./internal/apps/openflare/chwriter -run TestLiveAppWritePath -count=1 -timeout 2m
|
||||
func TestLiveAppWritePath(t *testing.T) {
|
||||
if !db.ChConnReady() {
|
||||
t.Skip("ClickHouse connection not ready")
|
||||
}
|
||||
ctx := context.Background()
|
||||
chwriter.Init(ctx)
|
||||
|
||||
now := time.Now().UTC()
|
||||
nodeID := "e2e-app-write-" + now.Format("150405")
|
||||
if err := model.InsertOpenFlareMetricSnapshot(ctx, &model.OpenFlareMetricSnapshot{
|
||||
NodeID: nodeID,
|
||||
CapturedAt: now,
|
||||
CPUUsagePercent: 33.3,
|
||||
MemoryUsedBytes: 111,
|
||||
MemoryTotalBytes: 1000,
|
||||
StorageUsedBytes: 222,
|
||||
StorageTotalBytes: 2000,
|
||||
DiskReadBytes: 10,
|
||||
DiskWriteBytes: 20,
|
||||
NetworkRxBytes: 30,
|
||||
NetworkTxBytes: 40,
|
||||
}); err != nil {
|
||||
t.Fatalf("InsertOpenFlareMetricSnapshot: %v", err)
|
||||
}
|
||||
|
||||
deadline := time.Now().Add(45 * time.Second)
|
||||
var found bool
|
||||
for time.Now().Before(deadline) {
|
||||
rows, err := model.ListOpenFlareMetricSnapshotsSince(ctx, nodeID, now.Add(-time.Minute), 10)
|
||||
if err != nil {
|
||||
t.Fatalf("ListOpenFlareMetricSnapshotsSince: %v", err)
|
||||
}
|
||||
if len(rows) > 0 {
|
||||
found = true
|
||||
t.Logf("found snapshot id=%d cpu=%.1f after flush", rows[0].ID, rows[0].CPUUsagePercent)
|
||||
break
|
||||
}
|
||||
time.Sleep(2 * time.Second)
|
||||
}
|
||||
if !found {
|
||||
t.Fatal("metric snapshot not visible in ClickHouse after flush wait")
|
||||
}
|
||||
|
||||
latest, err := model.ListOpenFlareLatestMetricSnapshotsSince(ctx, "", now.Add(-time.Hour))
|
||||
if err != nil {
|
||||
t.Fatalf("ListOpenFlareLatestMetricSnapshotsSince: %v", err)
|
||||
}
|
||||
var latestOK bool
|
||||
for _, row := range latest {
|
||||
if row != nil && row.NodeID == nodeID {
|
||||
latestOK = true
|
||||
break
|
||||
}
|
||||
}
|
||||
if !latestOK {
|
||||
t.Fatalf("latest-per-node query missing node %s (rows=%d)", nodeID, len(latest))
|
||||
}
|
||||
|
||||
stats := chwriter.WriterStats()
|
||||
if len(stats) == 0 {
|
||||
t.Fatal("WriterStats empty after Init")
|
||||
}
|
||||
for _, s := range stats {
|
||||
t.Logf("writer %s running=%v depth=%d drops=%d flush_err=%d", s.Name, s.Running, s.Depth, s.Drops, s.FlushErrors)
|
||||
}
|
||||
}
|
||||
@@ -14,19 +14,30 @@ import (
|
||||
"github.com/Rain-kl/Wavelet/internal/config"
|
||||
"github.com/Rain-kl/Wavelet/internal/db/batchwriter"
|
||||
"github.com/Rain-kl/Wavelet/internal/lifecycle"
|
||||
"github.com/Rain-kl/Wavelet/internal/model"
|
||||
analyticsmodel "github.com/Rain-kl/Wavelet/internal/model/analytics"
|
||||
analyticsrepo "github.com/Rain-kl/Wavelet/internal/repository/analytics"
|
||||
"github.com/Rain-kl/Wavelet/pkg/logger"
|
||||
)
|
||||
|
||||
const (
|
||||
// Observability traffic is sparse (heartbeat ~10s/node). Prefer larger batches to
|
||||
// cut ClickHouse parts/merges; MaxFlushWait bounds visibility lag for single-node labs.
|
||||
observabilityQueueSize = 5_000
|
||||
observabilityMaxBatchSize = 200
|
||||
observabilityFlushEvery = 2 * time.Second
|
||||
observabilityMaxBatchSize = 500
|
||||
observabilityMinBatchSize = 20
|
||||
observabilityFlushEvery = 10 * time.Second
|
||||
observabilityMaxFlushWait = 30 * time.Second
|
||||
|
||||
nodeAccessLogQueueSize = 10_000
|
||||
nodeAccessLogMaxBatchSize = 1_000
|
||||
nodeAccessLogFlushEvery = time.Second
|
||||
nodeAccessLogMinBatchSize = 50
|
||||
nodeAccessLogFlushEvery = 2 * time.Second
|
||||
nodeAccessLogMaxFlushWait = 5 * time.Second
|
||||
|
||||
// flushAttempts is total tries (1 initial + short retries) before giving up a batch.
|
||||
flushAttempts = 2
|
||||
flushRetryBackoff = 50 * time.Millisecond
|
||||
)
|
||||
|
||||
var (
|
||||
@@ -41,6 +52,9 @@ var (
|
||||
|
||||
metricSnapshotDedup *dedupSet
|
||||
requestReportDedup *dedupSet
|
||||
openrestyDedup *dedupSet
|
||||
frpsDedup *dedupSet
|
||||
frpcDedup *dedupSet
|
||||
)
|
||||
|
||||
// Init starts OpenFlare ClickHouse batch writers. Safe to call multiple times.
|
||||
@@ -52,12 +66,40 @@ func Init(ctx context.Context) {
|
||||
initOnce.Do(func() {
|
||||
metricSnapshotDedup = newDedupSet()
|
||||
requestReportDedup = newDedupSet()
|
||||
openrestyDedup = newDedupSet()
|
||||
frpsDedup = newDedupSet()
|
||||
frpcDedup = newDedupSet()
|
||||
|
||||
metricSnapshotWriter = mustNewObservabilityWriter("metric_snapshots", analyticsrepo.BatchInsertNodeMetricSnapshots)
|
||||
requestReportWriter = mustNewObservabilityWriter("request_reports", analyticsrepo.BatchInsertNodeRequestReports)
|
||||
openrestyWriter = mustNewObservabilityWriter("openresty_obs", analyticsrepo.BatchInsertNodeObsOpenresty)
|
||||
frpsWriter = mustNewObservabilityWriter("frps_obs", analyticsrepo.BatchInsertNodeObsFrps)
|
||||
frpcWriter = mustNewObservabilityWriter("frpc_obs", analyticsrepo.BatchInsertNodeObsFrpc)
|
||||
metricSnapshotWriter = mustNewObservabilityWriter(
|
||||
"metric_snapshots",
|
||||
withFlushRetries(analyticsrepo.BatchInsertNodeMetricSnapshots),
|
||||
metricSnapshotDedup,
|
||||
metricSnapshotKey,
|
||||
)
|
||||
requestReportWriter = mustNewObservabilityWriter(
|
||||
"request_reports",
|
||||
withFlushRetries(analyticsrepo.BatchInsertNodeRequestReports),
|
||||
requestReportDedup,
|
||||
requestReportKey,
|
||||
)
|
||||
openrestyWriter = mustNewObservabilityWriter(
|
||||
"openresty_obs",
|
||||
withFlushRetries(analyticsrepo.BatchInsertNodeObsOpenresty),
|
||||
openrestyDedup,
|
||||
openrestyKey,
|
||||
)
|
||||
frpsWriter = mustNewObservabilityWriter(
|
||||
"frps_obs",
|
||||
withFlushRetries(analyticsrepo.BatchInsertNodeObsFrps),
|
||||
frpsDedup,
|
||||
frpsKey,
|
||||
)
|
||||
frpcWriter = mustNewObservabilityWriter(
|
||||
"frpc_obs",
|
||||
withFlushRetries(analyticsrepo.BatchInsertNodeObsFrpc),
|
||||
frpcDedup,
|
||||
frpcKey,
|
||||
)
|
||||
nodeAccessLogWriter = mustNewNodeAccessLogWriter()
|
||||
|
||||
metricSnapshotWriter.Start(ctx)
|
||||
@@ -67,6 +109,7 @@ func Init(ctx context.Context) {
|
||||
frpcWriter.Start(ctx)
|
||||
nodeAccessLogWriter.Start(ctx)
|
||||
|
||||
wireModelInsertHooks()
|
||||
lifecycle.OnShutdown("openflare_chwriter", Stop)
|
||||
})
|
||||
}
|
||||
@@ -96,57 +139,49 @@ func Stop(ctx context.Context) error {
|
||||
return firstErr
|
||||
}
|
||||
|
||||
// WriterStats returns queue depth and failure counters for all OpenFlare writers.
|
||||
func WriterStats() []batchwriter.Stats {
|
||||
writers := []statsProvider{
|
||||
metricSnapshotWriter,
|
||||
requestReportWriter,
|
||||
openrestyWriter,
|
||||
frpsWriter,
|
||||
frpcWriter,
|
||||
nodeAccessLogWriter,
|
||||
}
|
||||
out := make([]batchwriter.Stats, 0, len(writers))
|
||||
for _, w := range writers {
|
||||
if w == nil {
|
||||
continue
|
||||
}
|
||||
out = append(out, w.Stats())
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// QueueMetricSnapshot enqueues a metric snapshot for asynchronous flush.
|
||||
func QueueMetricSnapshot(snapshot analyticsmodel.NodeMetricSnapshot) {
|
||||
if metricSnapshotWriter == nil {
|
||||
return
|
||||
}
|
||||
key := fmt.Sprintf("%s|%d", snapshot.NodeID, snapshot.CapturedAt.UTC().UnixNano())
|
||||
if !metricSnapshotDedup.markIfNew(key) {
|
||||
return
|
||||
}
|
||||
metricSnapshotWriter.TryEnqueue(snapshot)
|
||||
queueWithDedup(metricSnapshotWriter, metricSnapshotDedup, metricSnapshotKey(snapshot), snapshot)
|
||||
}
|
||||
|
||||
// QueueRequestReport enqueues a request report for asynchronous flush.
|
||||
func QueueRequestReport(report analyticsmodel.NodeRequestReport) {
|
||||
if requestReportWriter == nil {
|
||||
return
|
||||
}
|
||||
key := fmt.Sprintf(
|
||||
"%s|%d|%d",
|
||||
report.NodeID,
|
||||
report.WindowStartedAt.UTC().UnixNano(),
|
||||
report.WindowEndedAt.UTC().UnixNano(),
|
||||
)
|
||||
if !requestReportDedup.markIfNew(key) {
|
||||
return
|
||||
}
|
||||
requestReportWriter.TryEnqueue(report)
|
||||
queueWithDedup(requestReportWriter, requestReportDedup, requestReportKey(report), report)
|
||||
}
|
||||
|
||||
// QueueOpenrestyObservation enqueues an OpenResty observation for asynchronous flush.
|
||||
func QueueOpenrestyObservation(observation analyticsmodel.NodeObsOpenresty) {
|
||||
if openrestyWriter == nil {
|
||||
return
|
||||
}
|
||||
openrestyWriter.TryEnqueue(observation)
|
||||
queueWithDedup(openrestyWriter, openrestyDedup, openrestyKey(observation), observation)
|
||||
}
|
||||
|
||||
// QueueFrpsObservation enqueues an FRPS observation for asynchronous flush.
|
||||
func QueueFrpsObservation(observation analyticsmodel.NodeObsFrps) {
|
||||
if frpsWriter == nil {
|
||||
return
|
||||
}
|
||||
frpsWriter.TryEnqueue(observation)
|
||||
queueWithDedup(frpsWriter, frpsDedup, frpsKey(observation), observation)
|
||||
}
|
||||
|
||||
// QueueFrpcObservation enqueues an FRPC observation for asynchronous flush.
|
||||
func QueueFrpcObservation(observation analyticsmodel.NodeObsFrpc) {
|
||||
if frpcWriter == nil {
|
||||
return
|
||||
}
|
||||
frpcWriter.TryEnqueue(observation)
|
||||
queueWithDedup(frpcWriter, frpcDedup, frpcKey(observation), observation)
|
||||
}
|
||||
|
||||
// QueueNodeAccessLogs enqueues node access logs for asynchronous flush.
|
||||
@@ -159,19 +194,46 @@ func QueueNodeAccessLogs(logs []analyticsmodel.NodeAccessLog) {
|
||||
}
|
||||
}
|
||||
|
||||
func mustNewObservabilityWriter[T any](name string, flush batchwriter.FlushFunc[T]) *batchwriter.Writer[T] {
|
||||
func queueWithDedup[T any](writer *batchwriter.Writer[T], dedup *dedupSet, key string, item T) {
|
||||
if writer == nil {
|
||||
return
|
||||
}
|
||||
// Mark first so concurrent duplicates still collapse; release on enqueue failure
|
||||
// so a full queue does not permanently suppress the item.
|
||||
if !dedup.markIfNew(key) {
|
||||
return
|
||||
}
|
||||
if !writer.TryEnqueue(item) {
|
||||
dedup.unmark(key)
|
||||
}
|
||||
}
|
||||
|
||||
func mustNewObservabilityWriter[T any](
|
||||
name string,
|
||||
flush batchwriter.FlushFunc[T],
|
||||
dedup *dedupSet,
|
||||
keyFn func(T) string,
|
||||
) *batchwriter.Writer[T] {
|
||||
cfg := batchwriter.Config{
|
||||
Name: name,
|
||||
QueueSize: observabilityQueueSize,
|
||||
MaxBatchSize: observabilityMaxBatchSize,
|
||||
MinBatchSize: observabilityMinBatchSize,
|
||||
FlushInterval: observabilityFlushEvery,
|
||||
MaxFlushWait: observabilityMaxFlushWait,
|
||||
}
|
||||
writer, err := batchwriter.New(
|
||||
cfg,
|
||||
flush,
|
||||
withObservabilityDropHandler[T](name),
|
||||
batchwriter.WithFlushErrorHandler[T](func(ctx context.Context, batchSize int, err error) {
|
||||
logger.ErrorF(ctx, "[OpenFlare] flush %s failed (batch=%d): %v", name, batchSize, err)
|
||||
batchwriter.WithFlushErrorHandler[T](func(ctx context.Context, items []T, err error) {
|
||||
logger.ErrorF(ctx, "[OpenFlare] flush %s failed (batch=%d): %v", name, len(items), err)
|
||||
if dedup == nil || keyFn == nil {
|
||||
return
|
||||
}
|
||||
for _, item := range items {
|
||||
dedup.unmark(keyFn(item))
|
||||
}
|
||||
}),
|
||||
)
|
||||
if err != nil {
|
||||
@@ -185,14 +247,18 @@ func mustNewNodeAccessLogWriter() *batchwriter.Writer[analyticsmodel.NodeAccessL
|
||||
Name: "node_access_logs",
|
||||
QueueSize: nodeAccessLogQueueSize,
|
||||
MaxBatchSize: nodeAccessLogMaxBatchSize,
|
||||
MinBatchSize: nodeAccessLogMinBatchSize,
|
||||
FlushInterval: nodeAccessLogFlushEvery,
|
||||
MaxFlushWait: nodeAccessLogMaxFlushWait,
|
||||
}
|
||||
writer, err := batchwriter.New[analyticsmodel.NodeAccessLog](cfg, analyticsrepo.BatchInsertNodeAccessLogs,
|
||||
writer, err := batchwriter.New[analyticsmodel.NodeAccessLog](
|
||||
cfg,
|
||||
withFlushRetries(analyticsrepo.BatchInsertNodeAccessLogs),
|
||||
batchwriter.WithDropHandler[analyticsmodel.NodeAccessLog](func(item analyticsmodel.NodeAccessLog) {
|
||||
logger.WarnF(context.Background(), "[OpenFlare] node access log queue full, dropping log for node %s path %s", item.NodeID, item.Path)
|
||||
}),
|
||||
batchwriter.WithFlushErrorHandler[analyticsmodel.NodeAccessLog](func(ctx context.Context, batchSize int, err error) {
|
||||
logger.ErrorF(ctx, "[OpenFlare] flush node access logs failed (batch=%d): %v", batchSize, err)
|
||||
batchwriter.WithFlushErrorHandler[analyticsmodel.NodeAccessLog](func(ctx context.Context, items []analyticsmodel.NodeAccessLog, err error) {
|
||||
logger.ErrorF(ctx, "[OpenFlare] flush node access logs failed (batch=%d): %v", len(items), err)
|
||||
}),
|
||||
)
|
||||
if err != nil {
|
||||
@@ -207,10 +273,74 @@ func withObservabilityDropHandler[T any](name string) batchwriter.Option[T] {
|
||||
})
|
||||
}
|
||||
|
||||
// withFlushRetries wraps a flush function with a short retry to ride out brief CH blips.
|
||||
func withFlushRetries[T any](flush batchwriter.FlushFunc[T]) batchwriter.FlushFunc[T] {
|
||||
return func(ctx context.Context, items []T) error {
|
||||
var err error
|
||||
for attempt := 1; attempt <= flushAttempts; attempt++ {
|
||||
err = flush(ctx, items)
|
||||
if err == nil {
|
||||
return nil
|
||||
}
|
||||
if attempt == flushAttempts {
|
||||
break
|
||||
}
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return ctx.Err()
|
||||
case <-time.After(flushRetryBackoff * time.Duration(attempt)):
|
||||
}
|
||||
}
|
||||
return err
|
||||
}
|
||||
}
|
||||
|
||||
func wireModelInsertHooks() {
|
||||
model.SetObservabilityInsertHooks(model.ObservabilityInsertHooks{
|
||||
QueueMetricSnapshot: QueueMetricSnapshot,
|
||||
QueueRequestReport: QueueRequestReport,
|
||||
QueueOpenrestyObservation: QueueOpenrestyObservation,
|
||||
QueueFrpsObservation: QueueFrpsObservation,
|
||||
QueueFrpcObservation: QueueFrpcObservation,
|
||||
})
|
||||
model.SetAccessLogInsertHooks(model.AccessLogInsertHooks{
|
||||
QueueNodeAccessLogs: QueueNodeAccessLogs,
|
||||
})
|
||||
}
|
||||
|
||||
func metricSnapshotKey(snapshot analyticsmodel.NodeMetricSnapshot) string {
|
||||
return fmt.Sprintf("%s|%d", snapshot.NodeID, snapshot.CapturedAt.UTC().UnixNano())
|
||||
}
|
||||
|
||||
func requestReportKey(report analyticsmodel.NodeRequestReport) string {
|
||||
return fmt.Sprintf(
|
||||
"%s|%d|%d",
|
||||
report.NodeID,
|
||||
report.WindowStartedAt.UTC().UnixNano(),
|
||||
report.WindowEndedAt.UTC().UnixNano(),
|
||||
)
|
||||
}
|
||||
|
||||
func openrestyKey(observation analyticsmodel.NodeObsOpenresty) string {
|
||||
return fmt.Sprintf("%s|%d", observation.NodeID, observation.CapturedAt.UTC().UnixNano())
|
||||
}
|
||||
|
||||
func frpsKey(observation analyticsmodel.NodeObsFrps) string {
|
||||
return fmt.Sprintf("%s|%d", observation.NodeID, observation.CapturedAt.UTC().UnixNano())
|
||||
}
|
||||
|
||||
func frpcKey(observation analyticsmodel.NodeObsFrpc) string {
|
||||
return fmt.Sprintf("%s|%d", observation.NodeID, observation.CapturedAt.UTC().UnixNano())
|
||||
}
|
||||
|
||||
type batchStopper interface {
|
||||
Stop(ctx context.Context) error
|
||||
}
|
||||
|
||||
type statsProvider interface {
|
||||
Stats() batchwriter.Stats
|
||||
}
|
||||
|
||||
func running() bool {
|
||||
return metricSnapshotWriter != nil && metricSnapshotWriter.Running()
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,33 @@
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
package dashboard
|
||||
|
||||
import (
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
const overviewCacheTTL = 30 * time.Second
|
||||
|
||||
var overviewCache struct {
|
||||
mu sync.Mutex
|
||||
payload *OverviewPayload
|
||||
expiresAt time.Time
|
||||
}
|
||||
|
||||
func getCachedOverview() (*OverviewPayload, bool) {
|
||||
overviewCache.mu.Lock()
|
||||
defer overviewCache.mu.Unlock()
|
||||
if overviewCache.payload == nil || time.Now().After(overviewCache.expiresAt) {
|
||||
return nil, false
|
||||
}
|
||||
return overviewCache.payload, true
|
||||
}
|
||||
|
||||
func setCachedOverview(payload *OverviewPayload) {
|
||||
overviewCache.mu.Lock()
|
||||
defer overviewCache.mu.Unlock()
|
||||
overviewCache.payload = payload
|
||||
overviewCache.expiresAt = time.Now().Add(overviewCacheTTL)
|
||||
}
|
||||
@@ -18,6 +18,7 @@ const (
|
||||
nodeStatusPending = "pending"
|
||||
|
||||
dashboardDistributionLimit = 8
|
||||
dashboardOverviewSnapshotLimit = 500
|
||||
highCPUUsagePercentThreshold = 80
|
||||
highMemoryUsagePercentThreshold = 85
|
||||
highStorageUsagePercentThreshold = 85
|
||||
|
||||
@@ -97,11 +97,16 @@ type trendsPayload struct {
|
||||
|
||||
// GetOverview aggregates dashboard overview data from nodes and observability tables.
|
||||
func GetOverview(ctx context.Context) (*OverviewPayload, error) {
|
||||
if payload, ok := getCachedOverview(); ok {
|
||||
return payload, nil
|
||||
}
|
||||
view, err := buildOverviewView(ctx)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return compressOverview(view), nil
|
||||
payload := compressOverview(view)
|
||||
setCachedOverview(payload)
|
||||
return payload, nil
|
||||
}
|
||||
|
||||
func buildOverviewView(ctx context.Context) (*OverviewView, error) {
|
||||
@@ -112,11 +117,21 @@ func buildOverviewView(ctx context.Context) (*OverviewView, error) {
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
snapshots, err := model.ListOpenFlareMetricSnapshotsSince(ctx, "", since, 0)
|
||||
// Latest-per-node health: dedicated LIMIT 1 BY queries (not a global raw LIMIT).
|
||||
latestSnapshotRows, err := model.ListOpenFlareLatestMetricSnapshotsSince(ctx, "", since)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
reports, err := model.ListOpenFlareRequestReportsSince(ctx, "", since, 0)
|
||||
latestTrafficRows, err := model.ListOpenFlareLatestRequestReportsSince(ctx, "", since)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
// Bounded raw windows remain for distributions and trend fallbacks; trends prefer hourly rollups.
|
||||
snapshots, err := model.ListOpenFlareMetricSnapshotsSince(ctx, "", since, dashboardOverviewSnapshotLimit)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
reports, err := model.ListOpenFlareRequestReportsSince(ctx, "", since, dashboardOverviewSnapshotLimit)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -128,27 +143,21 @@ func buildOverviewView(ctx context.Context) (*OverviewView, error) {
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
openrestySnapshots, err := model.ListOpenFlareNodeObservationOpenresty(ctx, "", since, 0)
|
||||
openrestySnapshots, err := model.ListOpenFlareNodeObservationOpenresty(ctx, "", since, dashboardOverviewSnapshotLimit)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
view := &OverviewView{
|
||||
GeneratedAt: now,
|
||||
Nodes: make([]NodeHealth, 0, len(nodes)),
|
||||
Distributions: observability.BuildTrafficDistributions(reports, accessLogRegions, dashboardDistributionLimit),
|
||||
Trends: observability.NodeTrends{
|
||||
Traffic24h: observability.BuildTrafficTrendPoints(now, reports),
|
||||
Capacity24h: observability.BuildCapacityTrendPoints(now, snapshots),
|
||||
Network24h: observability.BuildNetworkTrendPoints(now, snapshots, openrestySnapshots),
|
||||
DiskIO24h: observability.BuildDiskIOTrendPoints(now, snapshots),
|
||||
},
|
||||
Trends: observability.BuildNodeTrends(ctx, now, "", snapshots, openrestySnapshots, reports),
|
||||
}
|
||||
|
||||
var cpuNodeCount int
|
||||
var memoryNodeCount int
|
||||
latestSnapshots := observability.LatestMetricSnapshotsByNode(snapshots)
|
||||
latestTrafficReports := observability.LatestTrafficReportsByNode(reports)
|
||||
latestSnapshots := observability.LatestMetricSnapshotsByNode(latestSnapshotRows)
|
||||
latestTrafficReports := observability.LatestTrafficReportsByNode(latestTrafficRows)
|
||||
activeEventsByNode := observability.ActiveHealthEventsByNode(activeEvents)
|
||||
|
||||
for _, node := range nodes {
|
||||
|
||||
@@ -58,6 +58,32 @@ func TestGetOverviewStructure(t *testing.T) {
|
||||
OpenrestyStatus: "unknown",
|
||||
}).Error)
|
||||
|
||||
// Seed older + newer snapshots per node; health must use latest-per-node, not a global raw limit.
|
||||
require.NoError(t, model.InsertOpenFlareMetricSnapshot(ctx, &model.OpenFlareMetricSnapshot{
|
||||
NodeID: "node-dashboard-1",
|
||||
CapturedAt: now.Add(-2 * time.Hour),
|
||||
CPUUsagePercent: 10,
|
||||
MemoryUsedBytes: 1,
|
||||
MemoryTotalBytes: 10,
|
||||
}))
|
||||
require.NoError(t, model.InsertOpenFlareMetricSnapshot(ctx, &model.OpenFlareMetricSnapshot{
|
||||
NodeID: "node-dashboard-1",
|
||||
CapturedAt: now.Add(-time.Minute),
|
||||
CPUUsagePercent: 55,
|
||||
MemoryUsedBytes: 5,
|
||||
MemoryTotalBytes: 10,
|
||||
StorageUsedBytes: 2,
|
||||
StorageTotalBytes: 10,
|
||||
}))
|
||||
require.NoError(t, model.InsertOpenFlareRequestReport(ctx, &model.OpenFlareRequestReport{
|
||||
NodeID: "node-dashboard-1",
|
||||
WindowStartedAt: now.Add(-2 * time.Minute),
|
||||
WindowEndedAt: now.Add(-time.Minute),
|
||||
RequestCount: 12,
|
||||
ErrorCount: 1,
|
||||
UniqueVisitorCount: 4,
|
||||
}))
|
||||
|
||||
overview, err := GetOverview(ctx)
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, overview)
|
||||
@@ -69,14 +95,14 @@ func TestGetOverviewStructure(t *testing.T) {
|
||||
assert.Equal(t, 0, overview.Summary.OfflineNodes)
|
||||
assert.Equal(t, 0, overview.Summary.UnhealthyNodes)
|
||||
|
||||
assert.Equal(t, int64(0), overview.Traffic.RequestCount)
|
||||
assert.Equal(t, int64(0), overview.Traffic.UniqueVisitors)
|
||||
assert.Equal(t, int64(0), overview.Traffic.ErrorCount)
|
||||
assert.Equal(t, float64(0), overview.Traffic.EstimatedQPS)
|
||||
assert.Equal(t, 0, overview.Traffic.ReportedNodes)
|
||||
assert.Equal(t, int64(12), overview.Traffic.RequestCount)
|
||||
assert.Equal(t, int64(4), overview.Traffic.UniqueVisitors)
|
||||
assert.Equal(t, int64(1), overview.Traffic.ErrorCount)
|
||||
assert.InDelta(t, 0.2, overview.Traffic.EstimatedQPS, 0.0001)
|
||||
assert.Equal(t, 1, overview.Traffic.ReportedNodes)
|
||||
|
||||
assert.Equal(t, float64(0), overview.Capacity.AverageCPUUsagePercent)
|
||||
assert.Equal(t, float64(0), overview.Capacity.AverageMemoryUsagePercent)
|
||||
assert.Equal(t, 55.0, overview.Capacity.AverageCPUUsagePercent)
|
||||
assert.Equal(t, 50.0, overview.Capacity.AverageMemoryUsagePercent)
|
||||
assert.Equal(t, 0, overview.Capacity.HighCPUNodes)
|
||||
assert.Equal(t, 0, overview.Capacity.HighMemoryNodes)
|
||||
assert.Equal(t, 0, overview.Capacity.HighStorageNodes)
|
||||
@@ -120,10 +146,20 @@ func TestGetOverviewStructure(t *testing.T) {
|
||||
assert.Equal(t, "Edge 1", onlineNode[2])
|
||||
assert.Equal(t, "online", onlineNode[6])
|
||||
assert.Equal(t, "healthy", onlineNode[7])
|
||||
// Latest-per-node health fields (indexes match compressDashboardNodes).
|
||||
assert.Equal(t, 55.0, onlineNode[11]) // cpu_usage_percent from latest snapshot
|
||||
assert.Equal(t, 50.0, onlineNode[12]) // memory_usage_percent
|
||||
assert.Equal(t, int64(12), onlineNode[14])
|
||||
assert.Equal(t, int64(1), onlineNode[15])
|
||||
assert.Equal(t, int64(4), onlineNode[16])
|
||||
|
||||
pendingNode := nodeByID["node-dashboard-2"]
|
||||
require.NotNil(t, pendingNode)
|
||||
assert.Equal(t, "Edge 2", pendingNode[2])
|
||||
assert.Equal(t, "pending", pendingNode[6])
|
||||
assert.Equal(t, "unknown", pendingNode[7])
|
||||
|
||||
assert.Equal(t, 55.0, overview.Capacity.AverageCPUUsagePercent)
|
||||
assert.Equal(t, 1, overview.Traffic.ReportedNodes)
|
||||
assert.Equal(t, int64(4), overview.Traffic.UniqueVisitors)
|
||||
}
|
||||
|
||||
@@ -21,11 +21,12 @@ const (
|
||||
defaultIPTrendBucketMinute = 30
|
||||
maxIPTrendHours = 168
|
||||
nodeAccessLogRetentionDays = 90
|
||||
defaultAccessLogQueryDays = 7
|
||||
accessLogFieldRemoteAddr = "remote_addr"
|
||||
accessLogFieldRequestCount = "request_count"
|
||||
)
|
||||
|
||||
var nodeAccessLogRetentionWindow = nodeAccessLogRetentionDays * 24 * time.Hour
|
||||
var defaultAccessLogQueryWindow = defaultAccessLogQueryDays * 24 * time.Hour
|
||||
|
||||
// AccessLogQuery filters access log list queries.
|
||||
type AccessLogQuery struct {
|
||||
@@ -346,7 +347,7 @@ func ListFoldedAccessLogIPs(ctx context.Context, input FoldedAccessLogIPQuery) (
|
||||
// ListAccessLogIPSummaries returns paginated IP summaries.
|
||||
func ListAccessLogIPSummaries(ctx context.Context, input AccessLogIPSummaryQuery) (*AccessLogIPSummaryList, error) {
|
||||
normalized := normalizeAccessLogIPSummaryQuery(input)
|
||||
since := time.Now().UTC().Add(-nodeAccessLogRetentionWindow)
|
||||
since := defaultAccessLogSince()
|
||||
recentSince := time.Now().UTC().Add(-3 * time.Hour)
|
||||
query := model.OpenFlareAccessLogIPSummaryQuery{
|
||||
NodeID: strings.TrimSpace(normalized.NodeID),
|
||||
@@ -453,7 +454,7 @@ func buildModelAccessLogQuery(input AccessLogQuery) model.OpenFlareAccessLogQuer
|
||||
RemoteAddr: strings.TrimSpace(input.RemoteAddr),
|
||||
Host: strings.TrimSpace(input.Host),
|
||||
Path: strings.TrimSpace(input.Path),
|
||||
Since: time.Now().UTC().Add(-nodeAccessLogRetentionWindow),
|
||||
Since: defaultAccessLogSince(),
|
||||
Page: input.Page,
|
||||
PageSize: input.PageSize,
|
||||
SortBy: input.SortBy,
|
||||
@@ -461,6 +462,10 @@ func buildModelAccessLogQuery(input AccessLogQuery) model.OpenFlareAccessLogQuer
|
||||
}
|
||||
}
|
||||
|
||||
func defaultAccessLogSince() time.Time {
|
||||
return time.Now().UTC().Add(-defaultAccessLogQueryWindow)
|
||||
}
|
||||
|
||||
func listNodeNameMap(ctx context.Context, logs []*model.OpenFlareAccessLog) (map[string]string, error) {
|
||||
nodeIDs := make([]string, 0, len(logs))
|
||||
seen := make(map[string]struct{}, len(logs))
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
package observability
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"sort"
|
||||
"strings"
|
||||
@@ -13,6 +14,7 @@ import (
|
||||
)
|
||||
|
||||
const observabilityTrendBuckets = 24
|
||||
const unknownTrendNodeKey = "__unknown__"
|
||||
|
||||
const (
|
||||
healthEventStatusActive = "active"
|
||||
@@ -133,6 +135,12 @@ type diskCounterState struct {
|
||||
seen bool
|
||||
}
|
||||
|
||||
type networkCounterState struct {
|
||||
rx int64
|
||||
tx int64
|
||||
seen bool
|
||||
}
|
||||
|
||||
func buildTrafficWindowSummary(report *model.OpenFlareRequestReport) *TrafficWindowSummary {
|
||||
if report == nil {
|
||||
return nil
|
||||
@@ -257,6 +265,66 @@ func buildHealthSummary(
|
||||
return summary
|
||||
}
|
||||
|
||||
// BuildNodeTrends builds 24h trend series, preferring ClickHouse hourly aggregates
|
||||
// over limited raw snapshot windows so capacity/network/disk charts stay complete.
|
||||
func BuildNodeTrends(
|
||||
ctx context.Context,
|
||||
now time.Time,
|
||||
nodeID string,
|
||||
snapshots []*model.OpenFlareMetricSnapshot,
|
||||
openrestyObs []*model.OpenFlareNodeObservationOpenresty,
|
||||
reports []*model.OpenFlareRequestReport,
|
||||
) NodeTrends {
|
||||
trendSince := now.Add(-24 * time.Hour)
|
||||
trafficTrend := BuildTrafficTrendPoints(now, reports)
|
||||
if trafficHourly, err := model.ListOpenFlareTrafficHourlySince(ctx, nodeID, trendSince); err == nil && len(trafficHourly) > 0 {
|
||||
trafficTrend = BuildTrafficTrendPointsFromHourly(now, trafficHourly)
|
||||
}
|
||||
|
||||
capacityTrend := BuildCapacityTrendPoints(now, snapshots)
|
||||
networkTrend := BuildNetworkTrendPoints(now, snapshots, openrestyObs)
|
||||
diskIOTrend := BuildDiskIOTrendPoints(now, snapshots)
|
||||
|
||||
metricHourly, metricErr := model.ListOpenFlareMetricHourlySince(ctx, nodeID, trendSince)
|
||||
if metricErr == nil && len(metricHourly) > 0 {
|
||||
capacityTrend = BuildCapacityTrendPointsFromHourly(now, metricHourly)
|
||||
diskIOTrend = BuildDiskIOTrendPointsFromHourly(now, metricHourly)
|
||||
}
|
||||
openrestyHourly, openrestyErr := model.ListOpenFlareOpenrestyHourlySince(ctx, nodeID, trendSince)
|
||||
if metricErr == nil && openrestyErr == nil && (len(metricHourly) > 0 || len(openrestyHourly) > 0) {
|
||||
networkTrend = BuildNetworkTrendPointsFromHourly(now, metricHourly, openrestyHourly)
|
||||
}
|
||||
|
||||
return NodeTrends{
|
||||
Traffic24h: trafficTrend,
|
||||
Capacity24h: capacityTrend,
|
||||
Network24h: networkTrend,
|
||||
DiskIO24h: diskIOTrend,
|
||||
}
|
||||
}
|
||||
|
||||
// BuildTrafficTrendPointsFromHourly builds 24h traffic trend buckets from hourly rollups.
|
||||
func BuildTrafficTrendPointsFromHourly(now time.Time, hourly []*model.OpenFlareTrafficHourly) []TrafficTrendPoint {
|
||||
start := trendWindowStart(now)
|
||||
points := make([]TrafficTrendPoint, observabilityTrendBuckets)
|
||||
for index := range points {
|
||||
points[index].BucketStartedAt = start.Add(time.Duration(index) * time.Hour)
|
||||
}
|
||||
for _, row := range hourly {
|
||||
if row == nil {
|
||||
continue
|
||||
}
|
||||
index, ok := trendBucketIndex(row.Hour, start)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
points[index].RequestCount += row.RequestCount
|
||||
points[index].ErrorCount += row.ErrorCount
|
||||
points[index].UniqueVisitorCount += row.UniqueVisitorCount
|
||||
}
|
||||
return points
|
||||
}
|
||||
|
||||
// BuildTrafficTrendPoints builds 24h traffic trend buckets.
|
||||
func BuildTrafficTrendPoints(now time.Time, reports []*model.OpenFlareRequestReport) []TrafficTrendPoint {
|
||||
start := trendWindowStart(now)
|
||||
@@ -314,7 +382,30 @@ func BuildCapacityTrendPoints(now time.Time, snapshots []*model.OpenFlareMetricS
|
||||
return points
|
||||
}
|
||||
|
||||
// BuildCapacityTrendPointsFromHourly builds 24h capacity trend buckets from hourly aggregates.
|
||||
func BuildCapacityTrendPointsFromHourly(now time.Time, hourly []*model.OpenFlareMetricHourly) []CapacityTrendPoint {
|
||||
start := trendWindowStart(now)
|
||||
points := make([]CapacityTrendPoint, observabilityTrendBuckets)
|
||||
for index := range points {
|
||||
points[index].BucketStartedAt = start.Add(time.Duration(index) * time.Hour)
|
||||
}
|
||||
for _, row := range hourly {
|
||||
if row == nil {
|
||||
continue
|
||||
}
|
||||
index, ok := trendBucketIndex(row.Hour, start)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
points[index].AverageCPUUsagePercent = row.AverageCPUUsagePercent
|
||||
points[index].AverageMemoryUsagePercent = row.AverageMemoryUsagePercent
|
||||
points[index].ReportedNodes = row.ReportedNodes
|
||||
}
|
||||
return points
|
||||
}
|
||||
|
||||
// BuildNetworkTrendPoints builds 24h network trend buckets.
|
||||
// Host and OpenResty counters are cumulative; values are consecutive deltas.
|
||||
func BuildNetworkTrendPoints(
|
||||
now time.Time,
|
||||
snapshots []*model.OpenFlareMetricSnapshot,
|
||||
@@ -327,24 +418,70 @@ func BuildNetworkTrendPoints(
|
||||
points[index].BucketStartedAt = start.Add(time.Duration(index) * time.Hour)
|
||||
accumulators[index].nodes = make(map[string]struct{})
|
||||
}
|
||||
sort.Slice(snapshots, func(i int, j int) bool {
|
||||
if snapshots[i].CapturedAt.Equal(snapshots[j].CapturedAt) {
|
||||
return snapshots[i].NodeID < snapshots[j].NodeID
|
||||
}
|
||||
return snapshots[i].CapturedAt.Before(snapshots[j].CapturedAt)
|
||||
})
|
||||
previousHostByNode := make(map[string]networkCounterState, len(snapshots))
|
||||
for _, snapshot := range snapshots {
|
||||
if snapshot == nil {
|
||||
continue
|
||||
}
|
||||
nodeKey := snapshot.NodeID
|
||||
if nodeKey == "" {
|
||||
nodeKey = unknownTrendNodeKey
|
||||
}
|
||||
previous := previousHostByNode[nodeKey]
|
||||
previousHostByNode[nodeKey] = networkCounterState{
|
||||
rx: snapshot.NetworkRxBytes,
|
||||
tx: snapshot.NetworkTxBytes,
|
||||
seen: true,
|
||||
}
|
||||
if !previous.seen {
|
||||
continue
|
||||
}
|
||||
index, ok := trendBucketIndex(snapshot.CapturedAt, start)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
points[index].NetworkRxBytes += snapshot.NetworkRxBytes
|
||||
points[index].NetworkTxBytes += snapshot.NetworkTxBytes
|
||||
points[index].NetworkRxBytes += nonNegativeDelta(snapshot.NetworkRxBytes, previous.rx)
|
||||
points[index].NetworkTxBytes += nonNegativeDelta(snapshot.NetworkTxBytes, previous.tx)
|
||||
if snapshot.NodeID != "" {
|
||||
accumulators[index].nodes[snapshot.NodeID] = struct{}{}
|
||||
}
|
||||
}
|
||||
sort.Slice(openrestyObs, func(i int, j int) bool {
|
||||
if openrestyObs[i].CapturedAt.Equal(openrestyObs[j].CapturedAt) {
|
||||
return openrestyObs[i].NodeID < openrestyObs[j].NodeID
|
||||
}
|
||||
return openrestyObs[i].CapturedAt.Before(openrestyObs[j].CapturedAt)
|
||||
})
|
||||
previousOpenrestyByNode := make(map[string]networkCounterState, len(openrestyObs))
|
||||
for _, obs := range openrestyObs {
|
||||
if obs == nil {
|
||||
continue
|
||||
}
|
||||
nodeKey := obs.NodeID
|
||||
if nodeKey == "" {
|
||||
nodeKey = unknownTrendNodeKey
|
||||
}
|
||||
previous := previousOpenrestyByNode[nodeKey]
|
||||
previousOpenrestyByNode[nodeKey] = networkCounterState{
|
||||
rx: obs.OpenrestyRxBytes,
|
||||
tx: obs.OpenrestyTxBytes,
|
||||
seen: true,
|
||||
}
|
||||
if !previous.seen {
|
||||
continue
|
||||
}
|
||||
index, ok := trendBucketIndex(obs.CapturedAt, start)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
points[index].OpenrestyRxBytes += obs.OpenrestyRxBytes
|
||||
points[index].OpenrestyTxBytes += obs.OpenrestyTxBytes
|
||||
points[index].OpenrestyRxBytes += nonNegativeDelta(obs.OpenrestyRxBytes, previous.rx)
|
||||
points[index].OpenrestyTxBytes += nonNegativeDelta(obs.OpenrestyTxBytes, previous.tx)
|
||||
if obs.NodeID != "" {
|
||||
accumulators[index].nodes[obs.NodeID] = struct{}{}
|
||||
}
|
||||
@@ -355,6 +492,48 @@ func BuildNetworkTrendPoints(
|
||||
return points
|
||||
}
|
||||
|
||||
// BuildNetworkTrendPointsFromHourly builds 24h network trend buckets from hourly aggregates.
|
||||
func BuildNetworkTrendPointsFromHourly(
|
||||
now time.Time,
|
||||
metricHourly []*model.OpenFlareMetricHourly,
|
||||
openrestyHourly []*model.OpenFlareOpenrestyHourly,
|
||||
) []NetworkTrendPoint {
|
||||
start := trendWindowStart(now)
|
||||
points := make([]NetworkTrendPoint, observabilityTrendBuckets)
|
||||
for index := range points {
|
||||
points[index].BucketStartedAt = start.Add(time.Duration(index) * time.Hour)
|
||||
}
|
||||
for _, row := range metricHourly {
|
||||
if row == nil {
|
||||
continue
|
||||
}
|
||||
index, ok := trendBucketIndex(row.Hour, start)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
points[index].NetworkRxBytes += row.NetworkRxBytes
|
||||
points[index].NetworkTxBytes += row.NetworkTxBytes
|
||||
if row.ReportedNodes > points[index].ReportedNodes {
|
||||
points[index].ReportedNodes = row.ReportedNodes
|
||||
}
|
||||
}
|
||||
for _, row := range openrestyHourly {
|
||||
if row == nil {
|
||||
continue
|
||||
}
|
||||
index, ok := trendBucketIndex(row.Hour, start)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
points[index].OpenrestyRxBytes += row.OpenrestyRxBytes
|
||||
points[index].OpenrestyTxBytes += row.OpenrestyTxBytes
|
||||
if row.ReportedNodes > points[index].ReportedNodes {
|
||||
points[index].ReportedNodes = row.ReportedNodes
|
||||
}
|
||||
}
|
||||
return points
|
||||
}
|
||||
|
||||
// BuildDiskIOTrendPoints builds 24h disk IO trend buckets.
|
||||
func BuildDiskIOTrendPoints(now time.Time, snapshots []*model.OpenFlareMetricSnapshot) []DiskIOTrendPoint {
|
||||
start := trendWindowStart(now)
|
||||
@@ -374,7 +553,7 @@ func BuildDiskIOTrendPoints(now time.Time, snapshots []*model.OpenFlareMetricSna
|
||||
for _, snapshot := range snapshots {
|
||||
nodeKey := snapshot.NodeID
|
||||
if nodeKey == "" {
|
||||
nodeKey = "__unknown__"
|
||||
nodeKey = unknownTrendNodeKey
|
||||
}
|
||||
previous := previousByNode[nodeKey]
|
||||
previousByNode[nodeKey] = diskCounterState{
|
||||
@@ -389,16 +568,8 @@ func BuildDiskIOTrendPoints(now time.Time, snapshots []*model.OpenFlareMetricSna
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
readDelta := snapshot.DiskReadBytes - previous.read
|
||||
writeDelta := snapshot.DiskWriteBytes - previous.write
|
||||
if readDelta < 0 {
|
||||
readDelta = 0
|
||||
}
|
||||
if writeDelta < 0 {
|
||||
writeDelta = 0
|
||||
}
|
||||
points[index].DiskReadBytes += readDelta
|
||||
points[index].DiskWriteBytes += writeDelta
|
||||
points[index].DiskReadBytes += nonNegativeDelta(snapshot.DiskReadBytes, previous.read)
|
||||
points[index].DiskWriteBytes += nonNegativeDelta(snapshot.DiskWriteBytes, previous.write)
|
||||
if snapshot.NodeID != "" {
|
||||
accumulators[index].nodes[snapshot.NodeID] = struct{}{}
|
||||
}
|
||||
@@ -409,6 +580,36 @@ func BuildDiskIOTrendPoints(now time.Time, snapshots []*model.OpenFlareMetricSna
|
||||
return points
|
||||
}
|
||||
|
||||
// BuildDiskIOTrendPointsFromHourly builds 24h disk IO trend buckets from hourly aggregates.
|
||||
func BuildDiskIOTrendPointsFromHourly(now time.Time, hourly []*model.OpenFlareMetricHourly) []DiskIOTrendPoint {
|
||||
start := trendWindowStart(now)
|
||||
points := make([]DiskIOTrendPoint, observabilityTrendBuckets)
|
||||
for index := range points {
|
||||
points[index].BucketStartedAt = start.Add(time.Duration(index) * time.Hour)
|
||||
}
|
||||
for _, row := range hourly {
|
||||
if row == nil {
|
||||
continue
|
||||
}
|
||||
index, ok := trendBucketIndex(row.Hour, start)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
points[index].DiskReadBytes += row.DiskReadBytes
|
||||
points[index].DiskWriteBytes += row.DiskWriteBytes
|
||||
points[index].ReportedNodes = row.ReportedNodes
|
||||
}
|
||||
return points
|
||||
}
|
||||
|
||||
func nonNegativeDelta(current int64, previous int64) int64 {
|
||||
delta := current - previous
|
||||
if delta < 0 {
|
||||
return 0
|
||||
}
|
||||
return delta
|
||||
}
|
||||
|
||||
func latestMetricSnapshot(snapshots []*model.OpenFlareMetricSnapshot) *model.OpenFlareMetricSnapshot {
|
||||
var latest *model.OpenFlareMetricSnapshot
|
||||
for _, snapshot := range snapshots {
|
||||
|
||||
@@ -10,6 +10,23 @@ import (
|
||||
"github.com/Rain-kl/Wavelet/internal/model"
|
||||
)
|
||||
|
||||
func TestBuildTrafficTrendPointsFromHourlyBucketsByHour(t *testing.T) {
|
||||
now := time.Date(2026, 7, 2, 15, 30, 0, 0, time.UTC)
|
||||
hourly := []*model.OpenFlareTrafficHourly{
|
||||
{
|
||||
NodeID: "node-a",
|
||||
Hour: now.Add(-2 * time.Hour).Truncate(time.Hour),
|
||||
RequestCount: 12,
|
||||
ErrorCount: 1,
|
||||
UniqueVisitorCount: 4,
|
||||
},
|
||||
}
|
||||
points := BuildTrafficTrendPointsFromHourly(now, hourly)
|
||||
if len(points) != observabilityTrendBuckets {
|
||||
t.Fatalf("BuildTrafficTrendPointsFromHourly() len = %d, want %d", len(points), observabilityTrendBuckets)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildTrafficTrendPointsBucketsByHour(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
@@ -115,3 +132,84 @@ func TestBuildTrafficWindowSummaryNilWithoutReport(t *testing.T) {
|
||||
t.Fatalf("buildTrafficWindowSummary(nil) = %#v, want nil", summary)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildCapacityTrendPointsFromHourlyFillsBuckets(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
now := time.Date(2026, 7, 10, 9, 30, 0, 0, time.UTC)
|
||||
hourly := []*model.OpenFlareMetricHourly{
|
||||
{
|
||||
Hour: now.Add(-3 * time.Hour).Truncate(time.Hour),
|
||||
AverageCPUUsagePercent: 42.5,
|
||||
AverageMemoryUsagePercent: 61.2,
|
||||
ReportedNodes: 1,
|
||||
},
|
||||
{
|
||||
Hour: now.Truncate(time.Hour),
|
||||
AverageCPUUsagePercent: 12.0,
|
||||
AverageMemoryUsagePercent: 50.0,
|
||||
ReportedNodes: 2,
|
||||
},
|
||||
}
|
||||
|
||||
points := BuildCapacityTrendPointsFromHourly(now, hourly)
|
||||
if len(points) != observabilityTrendBuckets {
|
||||
t.Fatalf("len = %d, want %d", len(points), observabilityTrendBuckets)
|
||||
}
|
||||
if points[len(points)-4].AverageCPUUsagePercent != 42.5 {
|
||||
t.Fatalf("hour-3 cpu = %v, want 42.5", points[len(points)-4].AverageCPUUsagePercent)
|
||||
}
|
||||
if points[len(points)-1].ReportedNodes != 2 {
|
||||
t.Fatalf("current hour reported_nodes = %d, want 2", points[len(points)-1].ReportedNodes)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildNetworkTrendPointsUsesCounterDeltas(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
now := time.Date(2026, 7, 10, 9, 30, 0, 0, time.UTC)
|
||||
base := now.Truncate(time.Hour)
|
||||
snapshots := []*model.OpenFlareMetricSnapshot{
|
||||
{NodeID: "n1", CapturedAt: base.Add(10 * time.Minute), NetworkRxBytes: 1000, NetworkTxBytes: 2000},
|
||||
{NodeID: "n1", CapturedAt: base.Add(20 * time.Minute), NetworkRxBytes: 1500, NetworkTxBytes: 2600},
|
||||
}
|
||||
openrestyObs := []*model.OpenFlareNodeObservationOpenresty{
|
||||
{NodeID: "n1", CapturedAt: base.Add(10 * time.Minute), OpenrestyRxBytes: 100, OpenrestyTxBytes: 200},
|
||||
{NodeID: "n1", CapturedAt: base.Add(20 * time.Minute), OpenrestyRxBytes: 180, OpenrestyTxBytes: 250},
|
||||
}
|
||||
|
||||
points := BuildNetworkTrendPoints(now, snapshots, openrestyObs)
|
||||
current := points[len(points)-1]
|
||||
if current.NetworkRxBytes != 500 {
|
||||
t.Fatalf("network_rx_bytes = %d, want 500", current.NetworkRxBytes)
|
||||
}
|
||||
if current.NetworkTxBytes != 600 {
|
||||
t.Fatalf("network_tx_bytes = %d, want 600", current.NetworkTxBytes)
|
||||
}
|
||||
if current.OpenrestyRxBytes != 80 {
|
||||
t.Fatalf("openresty_rx_bytes = %d, want 80", current.OpenrestyRxBytes)
|
||||
}
|
||||
if current.OpenrestyTxBytes != 50 {
|
||||
t.Fatalf("openresty_tx_bytes = %d, want 50", current.OpenrestyTxBytes)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildDiskIOTrendPointsFromHourlyFillsBuckets(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
now := time.Date(2026, 7, 10, 9, 30, 0, 0, time.UTC)
|
||||
hourly := []*model.OpenFlareMetricHourly{
|
||||
{
|
||||
Hour: now.Add(-1 * time.Hour).Truncate(time.Hour),
|
||||
DiskReadBytes: 1024,
|
||||
DiskWriteBytes: 2048,
|
||||
ReportedNodes: 1,
|
||||
},
|
||||
}
|
||||
|
||||
points := BuildDiskIOTrendPointsFromHourly(now, hourly)
|
||||
prev := points[len(points)-2]
|
||||
if prev.DiskReadBytes != 1024 || prev.DiskWriteBytes != 2048 {
|
||||
t.Fatalf("previous hour disk io = %#v, want read=1024 write=2048", prev)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -7,6 +7,7 @@ import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/Rain-kl/Wavelet/internal/model"
|
||||
@@ -18,8 +19,19 @@ const (
|
||||
defaultObservabilityLimit = 120
|
||||
maxObservabilityLimit = 500
|
||||
defaultTrafficDistributionLimit = 8
|
||||
nodeObservabilityCacheTTL = 15 * time.Second
|
||||
)
|
||||
|
||||
var nodeObservabilityCache struct {
|
||||
mu sync.Mutex
|
||||
views map[string]cachedNodeObservability
|
||||
}
|
||||
|
||||
type cachedNodeObservability struct {
|
||||
view *NodeView
|
||||
expiresAt time.Time
|
||||
}
|
||||
|
||||
// NodeQuery filters node observability data.
|
||||
type NodeQuery struct {
|
||||
Hours int `json:"hours"`
|
||||
@@ -87,6 +99,9 @@ func GetNodeObservability(ctx context.Context, id uint, query NodeQuery) (*NodeV
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if view, ok := getCachedNodeObservability(node.NodeID); ok {
|
||||
return view, nil
|
||||
}
|
||||
|
||||
limit := normalizeObservabilityLimit(query.Limit)
|
||||
since := now.Add(-normalizeObservabilityWindow(query.Hours))
|
||||
@@ -115,23 +130,10 @@ func GetNodeObservability(ctx context.Context, id uint, query NodeQuery) (*NodeV
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
trendSnapshots, err := model.ListOpenFlareMetricSnapshotsSince(ctx, node.NodeID, now.Add(-24*time.Hour), 0)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
trendOpenresty, err := model.ListOpenFlareNodeObservationOpenresty(ctx, node.NodeID, now.Add(-24*time.Hour), 0)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
trendReports, err := model.ListOpenFlareRequestReportsSince(ctx, node.NodeID, now.Add(-24*time.Hour), 0)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
events, err := model.ListOpenFlareHealthEvents(ctx, node.NodeID, false, limit)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
view := &NodeView{
|
||||
NodeID: node.NodeID,
|
||||
Profile: profile,
|
||||
@@ -143,12 +145,7 @@ func GetNodeObservability(ctx context.Context, id uint, query NodeQuery) (*NodeV
|
||||
Distributions: BuildTrafficDistributions(reports, accessLogRegions, defaultTrafficDistributionLimit),
|
||||
Health: buildHealthSummary(latestMetricSnapshot(snapshots), latestTrafficReport(reports), events),
|
||||
},
|
||||
Trends: NodeTrends{
|
||||
Traffic24h: BuildTrafficTrendPoints(now, trendReports),
|
||||
Capacity24h: BuildCapacityTrendPoints(now, trendSnapshots),
|
||||
Network24h: BuildNetworkTrendPoints(now, trendSnapshots, trendOpenresty),
|
||||
DiskIO24h: BuildDiskIOTrendPoints(now, trendSnapshots),
|
||||
},
|
||||
Trends: BuildNodeTrends(ctx, now, node.NodeID, snapshots, openrestyObs, reports),
|
||||
}
|
||||
if node.NodeType == "tunnel_relay" {
|
||||
frpsObs, frpsErr := model.ListOpenFlareNodeObservationFrps(ctx, node.NodeID, time.Time{}, 1)
|
||||
@@ -161,9 +158,35 @@ func GetNodeObservability(ctx context.Context, id uint, query NodeQuery) (*NodeV
|
||||
}
|
||||
view.RelayDashboard = buildRelayDashboardSnapshot(node, latestFrps)
|
||||
}
|
||||
setCachedNodeObservability(node.NodeID, view)
|
||||
return view, nil
|
||||
}
|
||||
|
||||
func getCachedNodeObservability(nodeID string) (*NodeView, bool) {
|
||||
nodeObservabilityCache.mu.Lock()
|
||||
defer nodeObservabilityCache.mu.Unlock()
|
||||
if nodeObservabilityCache.views == nil {
|
||||
return nil, false
|
||||
}
|
||||
entry, ok := nodeObservabilityCache.views[nodeID]
|
||||
if !ok || time.Now().After(entry.expiresAt) {
|
||||
return nil, false
|
||||
}
|
||||
return entry.view, true
|
||||
}
|
||||
|
||||
func setCachedNodeObservability(nodeID string, view *NodeView) {
|
||||
nodeObservabilityCache.mu.Lock()
|
||||
defer nodeObservabilityCache.mu.Unlock()
|
||||
if nodeObservabilityCache.views == nil {
|
||||
nodeObservabilityCache.views = make(map[string]cachedNodeObservability)
|
||||
}
|
||||
nodeObservabilityCache.views[nodeID] = cachedNodeObservability{
|
||||
view: view,
|
||||
expiresAt: time.Now().Add(nodeObservabilityCacheTTL),
|
||||
}
|
||||
}
|
||||
|
||||
// CleanupHealthEvents removes all health events for a node.
|
||||
func CleanupHealthEvents(ctx context.Context, id uint) (*HealthEventCleanupResult, error) {
|
||||
node, err := model.GetOpenFlareNodeByID(ctx, id)
|
||||
|
||||
@@ -59,6 +59,9 @@ type databaseCleanupResult struct {
|
||||
Target string `json:"target"`
|
||||
TargetLabel string `json:"target_label"`
|
||||
DeletedCount int64 `json:"deleted_count"`
|
||||
EligibleCount int64 `json:"eligible_count,omitempty"`
|
||||
CleanupMode string `json:"cleanup_mode,omitempty"`
|
||||
TableTTLDays int `json:"table_ttl_days,omitempty"`
|
||||
DeleteAll bool `json:"delete_all"`
|
||||
RetentionDays *int `json:"retention_days,omitempty"`
|
||||
}
|
||||
@@ -198,6 +201,9 @@ func cleanupDatabaseObservability(ctx context.Context, input databaseCleanupInpu
|
||||
Target: result.Target,
|
||||
TargetLabel: result.TargetLabel,
|
||||
DeletedCount: result.DeletedCount,
|
||||
EligibleCount: result.EligibleCount,
|
||||
CleanupMode: result.CleanupMode,
|
||||
TableTTLDays: result.TableTTLDays,
|
||||
DeleteAll: result.DeleteAll,
|
||||
RetentionDays: result.RetentionDays,
|
||||
}, nil
|
||||
|
||||
@@ -146,21 +146,26 @@ func TestCleanupDatabaseObservabilityDeletesRows(t *testing.T) {
|
||||
},
|
||||
}))
|
||||
|
||||
retention := 7
|
||||
result, err := cleanupDatabaseObservability(ctx, databaseCleanupInput{
|
||||
// Retention shorter than table TTL (90d for access logs) must be rejected.
|
||||
shortRetention := 7
|
||||
_, err := cleanupDatabaseObservability(ctx, databaseCleanupInput{
|
||||
Target: "node_access_logs",
|
||||
RetentionDays: &retention,
|
||||
RetentionDays: &shortRetention,
|
||||
})
|
||||
require.Error(t, err)
|
||||
|
||||
// Full truncate still hard-deletes all rows.
|
||||
result, err := cleanupDatabaseObservability(ctx, databaseCleanupInput{
|
||||
Target: "node_access_logs",
|
||||
})
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, "node_access_logs", result.Target)
|
||||
assert.Equal(t, "访问日志", result.TargetLabel)
|
||||
assert.Equal(t, int64(1), result.DeletedCount)
|
||||
assert.False(t, result.DeleteAll)
|
||||
require.NotNil(t, result.RetentionDays)
|
||||
assert.Equal(t, 7, *result.RetentionDays)
|
||||
assert.Equal(t, int64(2), result.DeletedCount)
|
||||
assert.True(t, result.DeleteAll)
|
||||
assert.Equal(t, "truncate", result.CleanupMode)
|
||||
|
||||
rows, err := model.ListOpenFlareAccessLogs(ctx, model.OpenFlareAccessLogQuery{Page: 0, PageSize: 10})
|
||||
require.NoError(t, err)
|
||||
require.Len(t, rows, 1)
|
||||
assert.Equal(t, "/recent", rows[0].Path)
|
||||
assert.Empty(t, rows)
|
||||
}
|
||||
|
||||
@@ -12,6 +12,7 @@ import (
|
||||
|
||||
"github.com/Rain-kl/Wavelet/internal/model"
|
||||
"github.com/Rain-kl/Wavelet/internal/repository"
|
||||
analyticsrepo "github.com/Rain-kl/Wavelet/internal/repository/analytics"
|
||||
)
|
||||
|
||||
const (
|
||||
@@ -21,12 +22,31 @@ const (
|
||||
DatabaseCleanupTargetMetricSnapshots = "node_metric_snapshots"
|
||||
// DatabaseCleanupTargetRequestReports is the API cleanup target for request reports.
|
||||
DatabaseCleanupTargetRequestReports = "node_request_reports"
|
||||
// DatabaseCleanupTargetObsOpenresty is the API cleanup target for OpenResty observations.
|
||||
DatabaseCleanupTargetObsOpenresty = "node_obs_openresty"
|
||||
// DatabaseCleanupTargetObsFrps is the API cleanup target for FRPS observations.
|
||||
DatabaseCleanupTargetObsFrps = "node_obs_frps"
|
||||
// DatabaseCleanupTargetObsFrpc is the API cleanup target for FRPC observations.
|
||||
DatabaseCleanupTargetObsFrpc = "node_obs_frpc"
|
||||
)
|
||||
|
||||
var databaseCleanupTargets = map[string]string{
|
||||
DatabaseCleanupTargetAccessLogs: "访问日志",
|
||||
DatabaseCleanupTargetMetricSnapshots: "性能快照",
|
||||
DatabaseCleanupTargetRequestReports: "请求聚合",
|
||||
DatabaseCleanupTargetObsOpenresty: "OpenResty 观测",
|
||||
DatabaseCleanupTargetObsFrps: "FRPS 观测",
|
||||
DatabaseCleanupTargetObsFrpc: "FRPC 观测",
|
||||
}
|
||||
|
||||
// databaseCleanupTableTTLDays maps API targets to ClickHouse DDL TTL days.
|
||||
var databaseCleanupTableTTLDays = map[string]int{
|
||||
DatabaseCleanupTargetAccessLogs: analyticsrepo.TableTTLDaysNodeAccessLogs,
|
||||
DatabaseCleanupTargetMetricSnapshots: analyticsrepo.TableTTLDaysNodeMetricSnapshots,
|
||||
DatabaseCleanupTargetRequestReports: analyticsrepo.TableTTLDaysNodeRequestReports,
|
||||
DatabaseCleanupTargetObsOpenresty: analyticsrepo.TableTTLDaysNodeObs,
|
||||
DatabaseCleanupTargetObsFrps: analyticsrepo.TableTTLDaysNodeObs,
|
||||
DatabaseCleanupTargetObsFrpc: analyticsrepo.TableTTLDaysNodeObs,
|
||||
}
|
||||
|
||||
// DatabaseCleanupInput describes a manual observability cleanup request.
|
||||
@@ -36,13 +56,21 @@ type DatabaseCleanupInput struct {
|
||||
}
|
||||
|
||||
// DatabaseCleanupResult summarizes a manual observability cleanup run.
|
||||
//
|
||||
// Semantics:
|
||||
// - delete_all / cleanup_mode=truncate: DeletedCount is hard-deleted rows (TRUNCATE).
|
||||
// - retention path / cleanup_mode=ttl_materialize: DeletedCount is always 0;
|
||||
// EligibleCount estimates rows past the table DDL TTL (not an arbitrary younger cutoff).
|
||||
type DatabaseCleanupResult struct {
|
||||
Target string `json:"target"`
|
||||
TargetLabel string `json:"target_label"`
|
||||
DeletedCount int64 `json:"deleted_count"`
|
||||
DeleteAll bool `json:"delete_all"`
|
||||
RetentionDays *int `json:"retention_days,omitempty"`
|
||||
Cutoff *time.Time `json:"cutoff,omitempty"`
|
||||
Target string `json:"target"`
|
||||
TargetLabel string `json:"target_label"`
|
||||
DeletedCount int64 `json:"deleted_count"`
|
||||
EligibleCount int64 `json:"eligible_count,omitempty"`
|
||||
CleanupMode string `json:"cleanup_mode,omitempty"`
|
||||
TableTTLDays int `json:"table_ttl_days,omitempty"`
|
||||
DeleteAll bool `json:"delete_all"`
|
||||
RetentionDays *int `json:"retention_days,omitempty"`
|
||||
Cutoff *time.Time `json:"cutoff,omitempty"`
|
||||
}
|
||||
|
||||
// DatabaseAutoCleanupSummary summarizes a scheduled auto-cleanup run.
|
||||
@@ -52,7 +80,17 @@ type DatabaseAutoCleanupSummary struct {
|
||||
Results []DatabaseCleanupResult `json:"results"`
|
||||
}
|
||||
|
||||
// TableTTLDaysForCleanupTarget returns the DDL TTL days for a cleanup target.
|
||||
func TableTTLDaysForCleanupTarget(target string) (int, bool) {
|
||||
days, ok := databaseCleanupTableTTLDays[strings.TrimSpace(target)]
|
||||
return days, ok
|
||||
}
|
||||
|
||||
// CleanupDatabaseObservability deletes observability rows for the given target.
|
||||
//
|
||||
// When RetentionDays is nil, rows are hard-deleted via TRUNCATE.
|
||||
// When RetentionDays is set, ClickHouse only force-materializes the table TTL policy:
|
||||
// retention_days shorter than the table TTL is rejected (do not fake success).
|
||||
func CleanupDatabaseObservability(ctx context.Context, input DatabaseCleanupInput) (*DatabaseCleanupResult, error) {
|
||||
target := strings.TrimSpace(input.Target)
|
||||
targetLabel, ok := databaseCleanupTargets[target]
|
||||
@@ -63,34 +101,51 @@ func CleanupDatabaseObservability(ctx context.Context, input DatabaseCleanupInpu
|
||||
return nil, errors.New("retention_days 必须为大于 0 的整数")
|
||||
}
|
||||
|
||||
tableTTLDays := databaseCleanupTableTTLDays[target]
|
||||
result := &DatabaseCleanupResult{
|
||||
Target: target,
|
||||
TargetLabel: targetLabel,
|
||||
DeleteAll: input.RetentionDays == nil,
|
||||
Target: target,
|
||||
TargetLabel: targetLabel,
|
||||
DeleteAll: input.RetentionDays == nil,
|
||||
TableTTLDays: tableTTLDays,
|
||||
}
|
||||
|
||||
if input.RetentionDays == nil {
|
||||
deleted, err := deleteAllObservabilityRows(ctx, target)
|
||||
deleted, mode, err := deleteAllObservabilityRows(ctx, target)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
result.DeletedCount = deleted
|
||||
result.EligibleCount = deleted
|
||||
result.CleanupMode = mode
|
||||
return result, nil
|
||||
}
|
||||
|
||||
retentionDays := *input.RetentionDays
|
||||
cutoff := time.Now().UTC().Add(-time.Duration(retentionDays) * 24 * time.Hour)
|
||||
deleted, err := deleteObservabilityRowsBefore(ctx, target, cutoff)
|
||||
if retentionDays < tableTTLDays {
|
||||
return nil, fmt.Errorf(
|
||||
"retention_days 不能小于表 TTL(%d 天);ClickHouse 仅支持按表 TTL 物化过期,更短保留请使用清空全部或调整 DDL",
|
||||
tableTTLDays,
|
||||
)
|
||||
}
|
||||
|
||||
// MATERIALIZE TTL only enforces DDL policy; cutoff reported is the table TTL boundary.
|
||||
tableCutoff := time.Now().UTC().Add(-time.Duration(tableTTLDays) * 24 * time.Hour)
|
||||
eligible, mode, err := materializeObservabilityTableTTL(ctx, target)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
result.DeletedCount = deleted
|
||||
result.DeletedCount = 0
|
||||
result.EligibleCount = eligible
|
||||
result.CleanupMode = mode
|
||||
result.RetentionDays = &retentionDays
|
||||
result.Cutoff = &cutoff
|
||||
result.Cutoff = &tableCutoff
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// RunDatabaseAutoCleanupOnce runs retention-based cleanup for all observability targets.
|
||||
//
|
||||
// Configured retention shorter than a target's table TTL is clamped up to the table TTL
|
||||
// so the scheduled job can force-materialize each table policy without failing.
|
||||
func RunDatabaseAutoCleanupOnce(ctx context.Context, now time.Time) (*DatabaseAutoCleanupSummary, error) {
|
||||
enabled, err := repository.GetBoolByKey(ctx, model.ConfigKeyDatabaseAutoCleanupEnabled)
|
||||
if err != nil {
|
||||
@@ -111,10 +166,17 @@ func RunDatabaseAutoCleanupOnce(ctx context.Context, now time.Time) (*DatabaseAu
|
||||
DatabaseCleanupTargetAccessLogs,
|
||||
DatabaseCleanupTargetMetricSnapshots,
|
||||
DatabaseCleanupTargetRequestReports,
|
||||
DatabaseCleanupTargetObsOpenresty,
|
||||
DatabaseCleanupTargetObsFrps,
|
||||
DatabaseCleanupTargetObsFrpc,
|
||||
} {
|
||||
effectiveDays := retentionDays
|
||||
if ttl, ok := databaseCleanupTableTTLDays[target]; ok && effectiveDays < ttl {
|
||||
effectiveDays = ttl
|
||||
}
|
||||
result, err := CleanupDatabaseObservability(ctx, DatabaseCleanupInput{
|
||||
Target: target,
|
||||
RetentionDays: &retentionDays,
|
||||
RetentionDays: &effectiveDays,
|
||||
})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
@@ -129,28 +191,64 @@ func RunDatabaseAutoCleanupOnce(ctx context.Context, now time.Time) (*DatabaseAu
|
||||
}, nil
|
||||
}
|
||||
|
||||
func deleteAllObservabilityRows(ctx context.Context, target string) (int64, error) {
|
||||
func deleteAllObservabilityRows(ctx context.Context, target string) (int64, string, error) {
|
||||
var (
|
||||
deleted int64
|
||||
err error
|
||||
)
|
||||
switch target {
|
||||
case DatabaseCleanupTargetAccessLogs:
|
||||
return model.DeleteAllOpenFlareAccessLogs(ctx)
|
||||
deleted, err = model.DeleteAllOpenFlareAccessLogs(ctx)
|
||||
case DatabaseCleanupTargetMetricSnapshots:
|
||||
return model.DeleteAllOpenFlareMetricSnapshots(ctx)
|
||||
deleted, err = model.DeleteAllOpenFlareMetricSnapshots(ctx)
|
||||
case DatabaseCleanupTargetRequestReports:
|
||||
return model.DeleteAllOpenFlareRequestReports(ctx)
|
||||
deleted, err = model.DeleteAllOpenFlareRequestReports(ctx)
|
||||
case DatabaseCleanupTargetObsOpenresty:
|
||||
deleted, err = model.DeleteAllOpenFlareNodeObservationOpenresty(ctx)
|
||||
case DatabaseCleanupTargetObsFrps:
|
||||
deleted, err = model.DeleteAllOpenFlareNodeObservationFrps(ctx)
|
||||
case DatabaseCleanupTargetObsFrpc:
|
||||
deleted, err = model.DeleteAllOpenFlareNodeObservationFrpc(ctx)
|
||||
default:
|
||||
return 0, errors.New("unsupported cleanup target")
|
||||
return 0, "", errors.New("unsupported cleanup target")
|
||||
}
|
||||
if err != nil {
|
||||
return 0, "", err
|
||||
}
|
||||
return deleted, analyticsrepo.CleanupModeTruncate, nil
|
||||
}
|
||||
|
||||
func deleteObservabilityRowsBefore(ctx context.Context, target string, cutoff time.Time) (int64, error) {
|
||||
// materializeObservabilityTableTTL triggers table-TTL materialize (or memory-store delete-before
|
||||
// with the table TTL cutoff for tests) and returns the eligible/estimate row count.
|
||||
func materializeObservabilityTableTTL(ctx context.Context, target string) (int64, string, error) {
|
||||
ttlDays, ok := databaseCleanupTableTTLDays[target]
|
||||
if !ok {
|
||||
return 0, "", errors.New("unsupported cleanup target")
|
||||
}
|
||||
cutoff := time.Now().UTC().Add(-time.Duration(ttlDays) * 24 * time.Hour)
|
||||
|
||||
var (
|
||||
eligible int64
|
||||
err error
|
||||
)
|
||||
switch target {
|
||||
case DatabaseCleanupTargetAccessLogs:
|
||||
return model.DeleteOpenFlareAccessLogsBefore(ctx, cutoff)
|
||||
eligible, err = model.DeleteOpenFlareAccessLogsBefore(ctx, cutoff)
|
||||
case DatabaseCleanupTargetMetricSnapshots:
|
||||
return model.DeleteOpenFlareMetricSnapshotsBefore(ctx, cutoff)
|
||||
eligible, err = model.DeleteOpenFlareMetricSnapshotsBefore(ctx, cutoff)
|
||||
case DatabaseCleanupTargetRequestReports:
|
||||
return model.DeleteOpenFlareRequestReportsBefore(ctx, cutoff)
|
||||
eligible, err = model.DeleteOpenFlareRequestReportsBefore(ctx, cutoff)
|
||||
case DatabaseCleanupTargetObsOpenresty:
|
||||
eligible, err = model.DeleteOpenFlareNodeObservationOpenrestyBefore(ctx, cutoff)
|
||||
case DatabaseCleanupTargetObsFrps:
|
||||
eligible, err = model.DeleteOpenFlareNodeObservationFrpsBefore(ctx, cutoff)
|
||||
case DatabaseCleanupTargetObsFrpc:
|
||||
eligible, err = model.DeleteOpenFlareNodeObservationFrpcBefore(ctx, cutoff)
|
||||
default:
|
||||
return 0, errors.New("unsupported cleanup target")
|
||||
return 0, "", errors.New("unsupported cleanup target")
|
||||
}
|
||||
if err != nil {
|
||||
return 0, "", err
|
||||
}
|
||||
return eligible, analyticsrepo.CleanupModeTTLMaterialize, nil
|
||||
}
|
||||
|
||||
@@ -11,6 +11,7 @@ import (
|
||||
"github.com/Rain-kl/Wavelet/internal/db"
|
||||
"github.com/Rain-kl/Wavelet/internal/model"
|
||||
"github.com/Rain-kl/Wavelet/internal/repository"
|
||||
analyticsrepo "github.com/Rain-kl/Wavelet/internal/repository/analytics"
|
||||
"github.com/glebarez/sqlite"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
@@ -36,13 +37,41 @@ func setupDatabaseCleanupTestDB(t *testing.T) context.Context {
|
||||
return context.Background()
|
||||
}
|
||||
|
||||
func TestCleanupDatabaseObservabilityDeletesTargetedRows(t *testing.T) {
|
||||
func TestCleanupDatabaseObservabilityRejectsRetentionShorterThanTableTTL(t *testing.T) {
|
||||
ctx := setupDatabaseCleanupTestDB(t)
|
||||
|
||||
retentionDays := 7 // metric snapshots DDL TTL is 30 days
|
||||
result, err := CleanupDatabaseObservability(ctx, DatabaseCleanupInput{
|
||||
Target: DatabaseCleanupTargetMetricSnapshots,
|
||||
RetentionDays: &retentionDays,
|
||||
})
|
||||
require.Error(t, err)
|
||||
assert.Nil(t, result)
|
||||
assert.Contains(t, err.Error(), "不能小于表 TTL")
|
||||
assert.Contains(t, err.Error(), "30")
|
||||
}
|
||||
|
||||
func TestCleanupDatabaseObservabilityRejectsAccessLogRetentionShorterThanTableTTL(t *testing.T) {
|
||||
ctx := setupDatabaseCleanupTestDB(t)
|
||||
|
||||
retentionDays := 30 // access logs DDL TTL is 90 days
|
||||
result, err := CleanupDatabaseObservability(ctx, DatabaseCleanupInput{
|
||||
Target: DatabaseCleanupTargetAccessLogs,
|
||||
RetentionDays: &retentionDays,
|
||||
})
|
||||
require.Error(t, err)
|
||||
assert.Nil(t, result)
|
||||
assert.Contains(t, err.Error(), "90")
|
||||
}
|
||||
|
||||
func TestCleanupDatabaseObservabilityMaterializeDoesNotClaimHardDelete(t *testing.T) {
|
||||
ctx := setupDatabaseCleanupTestDB(t)
|
||||
now := time.Now().UTC()
|
||||
|
||||
// One row past metric table TTL (30d), one still inside the window.
|
||||
require.NoError(t, model.InsertOpenFlareMetricSnapshot(ctx, &model.OpenFlareMetricSnapshot{
|
||||
NodeID: "node-a",
|
||||
CapturedAt: now.Add(-10 * 24 * time.Hour),
|
||||
CapturedAt: now.Add(-40 * 24 * time.Hour),
|
||||
CPUUsagePercent: 10,
|
||||
}))
|
||||
require.NoError(t, model.InsertOpenFlareMetricSnapshot(ctx, &model.OpenFlareMetricSnapshot{
|
||||
@@ -51,15 +80,22 @@ func TestCleanupDatabaseObservabilityDeletesTargetedRows(t *testing.T) {
|
||||
CPUUsagePercent: 20,
|
||||
}))
|
||||
|
||||
retentionDays := 7
|
||||
retentionDays := analyticsrepo.TableTTLDaysNodeMetricSnapshots
|
||||
result, err := CleanupDatabaseObservability(ctx, DatabaseCleanupInput{
|
||||
Target: DatabaseCleanupTargetMetricSnapshots,
|
||||
RetentionDays: &retentionDays,
|
||||
})
|
||||
require.NoError(t, err)
|
||||
assert.False(t, result.DeleteAll)
|
||||
assert.Equal(t, int64(1), result.DeletedCount)
|
||||
assert.Equal(t, analyticsrepo.CleanupModeTTLMaterialize, result.CleanupMode)
|
||||
assert.Equal(t, analyticsrepo.TableTTLDaysNodeMetricSnapshots, result.TableTTLDays)
|
||||
// MATERIALIZE is not a counted hard delete.
|
||||
assert.Equal(t, int64(0), result.DeletedCount)
|
||||
assert.Equal(t, int64(1), result.EligibleCount)
|
||||
require.NotNil(t, result.Cutoff)
|
||||
assert.True(t, result.Cutoff.Before(now.Add(-29*24*time.Hour)))
|
||||
|
||||
// Memory store applies the table-TTL cutoff for tests; only the recent row remains.
|
||||
rows, err := model.ListOpenFlareMetricSnapshotsSince(ctx, "", time.Time{}, 0)
|
||||
require.NoError(t, err)
|
||||
require.Len(t, rows, 1)
|
||||
@@ -94,20 +130,23 @@ func TestCleanupDatabaseObservabilityDeletesAllRowsWhenRetentionMissing(t *testi
|
||||
})
|
||||
require.NoError(t, err)
|
||||
assert.True(t, result.DeleteAll)
|
||||
assert.Equal(t, analyticsrepo.CleanupModeTruncate, result.CleanupMode)
|
||||
assert.Equal(t, int64(2), result.DeletedCount)
|
||||
assert.Equal(t, int64(2), result.EligibleCount)
|
||||
|
||||
rows, err := model.ListOpenFlareAccessLogs(ctx, model.OpenFlareAccessLogQuery{Page: 0, PageSize: 10})
|
||||
require.NoError(t, err)
|
||||
assert.Empty(t, rows)
|
||||
}
|
||||
|
||||
func TestRunDatabaseAutoCleanupOnceDeletesAllObservabilityTargets(t *testing.T) {
|
||||
func TestRunDatabaseAutoCleanupOnceClampsRetentionToTableTTL(t *testing.T) {
|
||||
ctx := setupDatabaseCleanupTestDB(t)
|
||||
now := time.Now().UTC()
|
||||
|
||||
// Access logs TTL=90d, metrics TTL=30d. Config retention=1 must clamp, not reject.
|
||||
require.NoError(t, model.InsertOpenFlareAccessLogsBatch(ctx, []*model.OpenFlareAccessLog{{
|
||||
NodeID: "node-a",
|
||||
LoggedAt: now.Add(-48 * time.Hour),
|
||||
LoggedAt: now.Add(-100 * 24 * time.Hour),
|
||||
RemoteAddr: "203.0.113.10",
|
||||
Host: "example.com",
|
||||
Path: "/access",
|
||||
@@ -115,13 +154,13 @@ func TestRunDatabaseAutoCleanupOnceDeletesAllObservabilityTargets(t *testing.T)
|
||||
}}))
|
||||
require.NoError(t, model.InsertOpenFlareMetricSnapshot(ctx, &model.OpenFlareMetricSnapshot{
|
||||
NodeID: "node-a",
|
||||
CapturedAt: now.Add(-48 * time.Hour),
|
||||
CapturedAt: now.Add(-40 * 24 * time.Hour),
|
||||
CPUUsagePercent: 10,
|
||||
}))
|
||||
require.NoError(t, model.InsertOpenFlareRequestReport(ctx, &model.OpenFlareRequestReport{
|
||||
NodeID: "node-a",
|
||||
WindowStartedAt: now.Add(-49 * time.Hour),
|
||||
WindowEndedAt: now.Add(-48 * time.Hour),
|
||||
WindowStartedAt: now.Add(-41 * 24 * time.Hour),
|
||||
WindowEndedAt: now.Add(-40 * 24 * time.Hour),
|
||||
RequestCount: 15,
|
||||
}))
|
||||
|
||||
@@ -131,7 +170,16 @@ func TestRunDatabaseAutoCleanupOnceDeletesAllObservabilityTargets(t *testing.T)
|
||||
summary, err := RunDatabaseAutoCleanupOnce(ctx, now)
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, summary)
|
||||
require.Len(t, summary.Results, 3)
|
||||
require.Len(t, summary.Results, 6)
|
||||
assert.Equal(t, 1, summary.RetentionDays)
|
||||
|
||||
for _, result := range summary.Results {
|
||||
assert.Equal(t, analyticsrepo.CleanupModeTTLMaterialize, result.CleanupMode)
|
||||
assert.Equal(t, int64(0), result.DeletedCount, "target %s must not claim hard delete", result.Target)
|
||||
assert.GreaterOrEqual(t, result.TableTTLDays, 30)
|
||||
require.NotNil(t, result.RetentionDays)
|
||||
assert.GreaterOrEqual(t, *result.RetentionDays, result.TableTTLDays)
|
||||
}
|
||||
|
||||
accessLogs, err := model.ListOpenFlareAccessLogs(ctx, model.OpenFlareAccessLogQuery{Page: 0, PageSize: 10})
|
||||
require.NoError(t, err)
|
||||
@@ -145,3 +193,16 @@ func TestRunDatabaseAutoCleanupOnceDeletesAllObservabilityTargets(t *testing.T)
|
||||
require.NoError(t, err)
|
||||
assert.Empty(t, requestReports)
|
||||
}
|
||||
|
||||
func TestTableTTLDaysForCleanupTarget(t *testing.T) {
|
||||
days, ok := TableTTLDaysForCleanupTarget(DatabaseCleanupTargetAccessLogs)
|
||||
require.True(t, ok)
|
||||
assert.Equal(t, 90, days)
|
||||
|
||||
days, ok = TableTTLDaysForCleanupTarget(DatabaseCleanupTargetMetricSnapshots)
|
||||
require.True(t, ok)
|
||||
assert.Equal(t, 30, days)
|
||||
|
||||
_, ok = TableTTLDaysForCleanupTarget("unknown")
|
||||
assert.False(t, ok)
|
||||
}
|
||||
|
||||
@@ -234,15 +234,15 @@ func evaluateParsedIPGroupAutoConfig(ctx context.Context, config ipGroupAutoConf
|
||||
}
|
||||
programs = append(programs, program)
|
||||
}
|
||||
logs, err := model.ListOpenFlareAccessLogsForWAFIPGroup(ctx, model.OpenFlareAccessLogQuery{
|
||||
aggregates, err := model.ListOpenFlareAccessLogWAFIPAggregates(ctx, model.OpenFlareAccessLogQuery{
|
||||
Since: now.Add(-time.Duration(config.LookbackMinutes) * time.Minute),
|
||||
Until: now,
|
||||
})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
accumulators := make(map[string]*ipGroupAutoAccumulator)
|
||||
for _, item := range logs {
|
||||
accumulators := make(map[string]*ipGroupAutoAccumulator, len(aggregates))
|
||||
for _, item := range aggregates {
|
||||
if item == nil {
|
||||
continue
|
||||
}
|
||||
@@ -250,30 +250,23 @@ func evaluateParsedIPGroupAutoConfig(ctx context.Context, config ipGroupAutoConf
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
acc := accumulators[ip]
|
||||
if acc == nil {
|
||||
acc = &ipGroupAutoAccumulator{
|
||||
ip: ip,
|
||||
statusCounts: make(map[int]int),
|
||||
}
|
||||
accumulators[ip] = acc
|
||||
lastSeen := time.Time{}
|
||||
if item.LastSeenEpoch > 0 {
|
||||
lastSeen = time.Unix(item.LastSeenEpoch, 0).UTC()
|
||||
}
|
||||
acc.requestCount++
|
||||
acc.statusCounts[item.StatusCode]++
|
||||
if item.StatusCode == http.StatusNotFound {
|
||||
acc.status404Count++
|
||||
statusCounts := make(map[int]int, len(item.StatusCounts))
|
||||
for code, count := range item.StatusCounts {
|
||||
statusCounts[code] = count
|
||||
}
|
||||
if item.StatusCode >= 400 && item.StatusCode < 500 {
|
||||
acc.clientErrorCount++
|
||||
}
|
||||
if item.StatusCode >= http.StatusInternalServerError {
|
||||
acc.serverErrorCount++
|
||||
}
|
||||
if hostIsIPLiteral(item.Host) {
|
||||
acc.ipHostCount++
|
||||
}
|
||||
if item.LoggedAt.After(acc.lastSeen) {
|
||||
acc.lastSeen = item.LoggedAt
|
||||
accumulators[ip] = &ipGroupAutoAccumulator{
|
||||
ip: ip,
|
||||
requestCount: item.RequestCount,
|
||||
status404Count: item.Status404Count,
|
||||
ipHostCount: item.IPHostCount,
|
||||
clientErrorCount: item.ClientErrorCount,
|
||||
serverErrorCount: item.ServerErrorCount,
|
||||
lastSeen: lastSeen,
|
||||
statusCounts: statusCounts,
|
||||
}
|
||||
}
|
||||
matched := make([]string, 0)
|
||||
@@ -336,11 +329,6 @@ func normalizeIPLiteral(value string) (string, bool) {
|
||||
return addr.String(), true
|
||||
}
|
||||
|
||||
func hostIsIPLiteral(value string) bool {
|
||||
_, ok := normalizeIPLiteral(value)
|
||||
return ok
|
||||
}
|
||||
|
||||
func downloadIPGroupSubscription(ctx context.Context, rawURL string) ([]byte, error) {
|
||||
if err := validateSubscriptionURL(rawURL); err != nil {
|
||||
return nil, err
|
||||
|
||||
@@ -6,6 +6,7 @@ package risk_control
|
||||
import (
|
||||
"context"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/Rain-kl/Wavelet/internal/config"
|
||||
"github.com/Rain-kl/Wavelet/internal/db/batchwriter"
|
||||
@@ -15,6 +16,11 @@ import (
|
||||
"github.com/Rain-kl/Wavelet/pkg/logger"
|
||||
)
|
||||
|
||||
const (
|
||||
// Bound visibility lag for sparse access-log traffic when MinBatchSize is not met.
|
||||
accessLogMaxFlushWait = 3 * time.Second
|
||||
)
|
||||
|
||||
var (
|
||||
logWriterMu sync.RWMutex
|
||||
logWriter *batchwriter.Writer[*analytics.UserAccessLog]
|
||||
@@ -33,6 +39,8 @@ func InitLogWriter(ctx context.Context) {
|
||||
}
|
||||
|
||||
cfg := batchwriter.DefaultConfig()
|
||||
cfg.Name = "user_access_logs"
|
||||
cfg.MaxFlushWait = accessLogMaxFlushWait
|
||||
writer, err := batchwriter.New[*analytics.UserAccessLog](cfg, func(ctx context.Context, items []*analytics.UserAccessLog) error {
|
||||
rows := make([]analytics.UserAccessLog, 0, len(items))
|
||||
for _, item := range items {
|
||||
@@ -50,8 +58,8 @@ func InitLogWriter(ctx context.Context) {
|
||||
}
|
||||
logger.WarnF(context.Background(), "[RiskControl] Log queue full, dropping log item for path: %s", path)
|
||||
}),
|
||||
batchwriter.WithFlushErrorHandler[*analytics.UserAccessLog](func(ctx context.Context, batchSize int, err error) {
|
||||
logger.ErrorF(ctx, "[RiskControl] Send ClickHouse batch failed (batch=%d): %v", batchSize, err)
|
||||
batchwriter.WithFlushErrorHandler[*analytics.UserAccessLog](func(ctx context.Context, items []*analytics.UserAccessLog, err error) {
|
||||
logger.ErrorF(ctx, "[RiskControl] Send ClickHouse batch failed (batch=%d): %v", len(items), err)
|
||||
}),
|
||||
)
|
||||
if err != nil {
|
||||
@@ -82,6 +90,16 @@ func IsBufferFull() bool {
|
||||
return writer.IsFull()
|
||||
}
|
||||
|
||||
// LogWriterStats returns queue depth and failure counters for the access-log writer.
|
||||
// When the writer is not initialized, it returns a zero-value Stats with the expected name.
|
||||
func LogWriterStats() batchwriter.Stats {
|
||||
writer := currentLogWriter()
|
||||
if writer == nil {
|
||||
return batchwriter.Stats{Name: "user_access_logs"}
|
||||
}
|
||||
return writer.Stats()
|
||||
}
|
||||
|
||||
// QueueAccessLog enqueues an access log without blocking.
|
||||
func QueueAccessLog(logItem *analytics.UserAccessLog) {
|
||||
writer := currentLogWriter()
|
||||
@@ -108,4 +126,4 @@ func currentLogWriter() *batchwriter.Writer[*analytics.UserAccessLog] {
|
||||
logWriterMu.RLock()
|
||||
defer logWriterMu.RUnlock()
|
||||
return logWriter
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,30 @@
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
package risk_control
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func TestAccessLogMaxFlushWaitInRange(t *testing.T) {
|
||||
t.Parallel()
|
||||
if accessLogMaxFlushWait < 2*time.Second || accessLogMaxFlushWait > 5*time.Second {
|
||||
t.Fatalf("accessLogMaxFlushWait = %v, want in [2s, 5s]", accessLogMaxFlushWait)
|
||||
}
|
||||
}
|
||||
|
||||
func TestLogWriterStatsWhenNil(t *testing.T) {
|
||||
t.Parallel()
|
||||
reset := SetLogWriterForTest(nil)
|
||||
t.Cleanup(reset)
|
||||
|
||||
stats := LogWriterStats()
|
||||
if stats.Name != "user_access_logs" {
|
||||
t.Fatalf("LogWriterStats().Name = %q, want user_access_logs", stats.Name)
|
||||
}
|
||||
if stats.Running {
|
||||
t.Fatal("LogWriterStats().Running = true for nil writer, want false")
|
||||
}
|
||||
}
|
||||
@@ -5,8 +5,11 @@
|
||||
package risk_control
|
||||
|
||||
import (
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"encoding/json"
|
||||
"net/http"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/Rain-kl/Wavelet/internal/apps/oauth"
|
||||
@@ -18,6 +21,61 @@ import (
|
||||
"github.com/gin-gonic/gin"
|
||||
)
|
||||
|
||||
const maxAuditLogHeadersBytes = 2 * 1024
|
||||
|
||||
var auditLogHeaderAllowlist = map[string]struct{}{
|
||||
"Authorization": {},
|
||||
"Cookie": {},
|
||||
"X-Forwarded-For": {},
|
||||
"X-Real-Ip": {},
|
||||
"User-Agent": {},
|
||||
"Content-Type": {},
|
||||
}
|
||||
|
||||
func marshalAuditLogHeaders(headers http.Header) string {
|
||||
if headers == nil {
|
||||
return ""
|
||||
}
|
||||
|
||||
filtered := make(http.Header)
|
||||
for key, values := range headers {
|
||||
if _, ok := auditLogHeaderAllowlist[key]; !ok {
|
||||
continue
|
||||
}
|
||||
filtered[key] = redactAuditLogHeaderValues(key, values)
|
||||
}
|
||||
|
||||
headersBytes, err := json.Marshal(filtered)
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
if len(headersBytes) <= maxAuditLogHeadersBytes {
|
||||
return string(headersBytes)
|
||||
}
|
||||
return string(headersBytes[:maxAuditLogHeadersBytes])
|
||||
}
|
||||
|
||||
func redactAuditLogHeaderValues(key string, values []string) []string {
|
||||
switch key {
|
||||
case "Authorization", "Cookie":
|
||||
redacted := make([]string, len(values))
|
||||
for i, value := range values {
|
||||
redacted[i] = hashAuditLogSensitiveValue(value)
|
||||
}
|
||||
return redacted
|
||||
default:
|
||||
return values
|
||||
}
|
||||
}
|
||||
|
||||
func hashAuditLogSensitiveValue(value string) string {
|
||||
if strings.TrimSpace(value) == "" {
|
||||
return ""
|
||||
}
|
||||
sum := sha256.Sum256([]byte(value))
|
||||
return "sha256:" + hex.EncodeToString(sum[:8])
|
||||
}
|
||||
|
||||
// RiskControlMiddleware 全局日志采集中间件
|
||||
func RiskControlMiddleware() gin.HandlerFunc {
|
||||
return func(c *gin.Context) {
|
||||
@@ -47,19 +105,7 @@ func RiskControlMiddleware() gin.HandlerFunc {
|
||||
// 4. 计算耗时并异步推送到缓冲队列
|
||||
latency := time.Since(start).Milliseconds()
|
||||
|
||||
var headersStr string
|
||||
if c.Request.Header != nil {
|
||||
// 克隆 Header,避免污染原 HTTP 请求的 Header 对象
|
||||
clonedHeaders := make(http.Header)
|
||||
for k, v := range c.Request.Header {
|
||||
clonedHeaders[k] = v
|
||||
}
|
||||
clonedHeaders.Del("Cookie")
|
||||
|
||||
if headersBytes, err := json.Marshal(clonedHeaders); err == nil {
|
||||
headersStr = string(headersBytes)
|
||||
}
|
||||
}
|
||||
headersStr := marshalAuditLogHeaders(c.Request.Header)
|
||||
|
||||
const maxHTTPStatus = 999
|
||||
status := c.Writer.Status()
|
||||
|
||||
@@ -94,6 +94,8 @@ func TestRiskControlMiddleware(t *testing.T) {
|
||||
req, _ := http.NewRequest(http.MethodGet, "/test", nil)
|
||||
req.Header.Set("X-Test-Header", "hello")
|
||||
req.Header.Set("Cookie", "session_id=abcdef123456")
|
||||
req.Header.Set("Authorization", "Bearer secret-token")
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
r.ServeHTTP(w, req)
|
||||
|
||||
assert.Equal(t, http.StatusOK, w.Code)
|
||||
@@ -106,8 +108,11 @@ func TestRiskControlMiddleware(t *testing.T) {
|
||||
assert.Equal(t, http.MethodGet, logItem.Method)
|
||||
assert.Equal(t, int32(http.StatusOK), logItem.Status)
|
||||
assert.NotEmpty(t, logItem.Headers)
|
||||
assert.Contains(t, logItem.Headers, "X-Test-Header")
|
||||
assert.NotContains(t, logItem.Headers, "Cookie")
|
||||
assert.NotContains(t, logItem.Headers, "X-Test-Header")
|
||||
assert.Contains(t, logItem.Headers, "Content-Type")
|
||||
assert.Contains(t, logItem.Headers, "sha256:")
|
||||
assert.NotContains(t, logItem.Headers, "secret-token")
|
||||
assert.NotContains(t, logItem.Headers, "session_id=abcdef123456")
|
||||
case <-time.After(200 * time.Millisecond):
|
||||
t.Fatal("expected flushed log item, but got none")
|
||||
}
|
||||
|
||||
@@ -118,9 +118,17 @@ func applyDefaults(c *configModel) {
|
||||
}
|
||||
|
||||
func applyClickHouseDefaults(c *configModel) {
|
||||
// Tests disable ClickHouse by default to avoid accidental connections.
|
||||
// Opt in with CLICKHOUSE_ENABLED=true for live integration tests (e.g. -tags live_ch).
|
||||
if isTest() {
|
||||
c.ClickHouse.Enabled = false
|
||||
return
|
||||
if v, ok := os.LookupEnv("CLICKHOUSE_ENABLED"); !ok {
|
||||
c.ClickHouse.Enabled = false
|
||||
return
|
||||
} else if b, err := strconv.ParseBool(v); err != nil || !b {
|
||||
c.ClickHouse.Enabled = false
|
||||
return
|
||||
}
|
||||
// Keep Enabled=true from env and continue applying host/pool defaults.
|
||||
}
|
||||
if !c.ClickHouse.Enabled {
|
||||
c.ClickHouse.Enabled = true
|
||||
@@ -134,11 +142,14 @@ func applyClickHouseDefaults(c *configModel) {
|
||||
if c.ClickHouse.Username == "" {
|
||||
c.ClickHouse.Username = "default"
|
||||
}
|
||||
// Pool / buffer defaults target small control-plane hosts (e.g. 3c6g):
|
||||
// oversized open/idle pools waste RAM and amplify concurrent CH pressure;
|
||||
// large block buffers add client memory without helping our small batch inserts.
|
||||
if c.ClickHouse.MaxIdleConn <= 0 {
|
||||
c.ClickHouse.MaxIdleConn = 10
|
||||
c.ClickHouse.MaxIdleConn = 8
|
||||
}
|
||||
if c.ClickHouse.MaxOpenConn <= 0 {
|
||||
c.ClickHouse.MaxOpenConn = 100
|
||||
c.ClickHouse.MaxOpenConn = 16
|
||||
}
|
||||
if c.ClickHouse.ConnMaxLifetime <= 0 {
|
||||
c.ClickHouse.ConnMaxLifetime = 3600
|
||||
@@ -147,7 +158,7 @@ func applyClickHouseDefaults(c *configModel) {
|
||||
c.ClickHouse.DialTimeout = 5
|
||||
}
|
||||
if c.ClickHouse.BlockBufferSize == 0 {
|
||||
c.ClickHouse.BlockBufferSize = 10
|
||||
c.ClickHouse.BlockBufferSize = 32
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -71,18 +71,19 @@ type databaseReplicaConfig struct {
|
||||
Password string `mapstructure:"password"`
|
||||
}
|
||||
|
||||
// clickhouse 配置
|
||||
// clickHouseConfig ClickHouse 原生客户端配置。
|
||||
// 连接池 / block_buffer 默认值按小型控制面主机(如 3c6g)收敛,见 applyClickHouseDefaults。
|
||||
type clickHouseConfig struct {
|
||||
Enabled bool `mapstructure:"enabled"`
|
||||
Hosts []string `mapstructure:"hosts"`
|
||||
Username string `mapstructure:"username"`
|
||||
Password string `mapstructure:"password"`
|
||||
Database string `mapstructure:"database"`
|
||||
MaxIdleConn int `mapstructure:"max_idle_conn"`
|
||||
MaxOpenConn int `mapstructure:"max_open_conn"`
|
||||
ConnMaxLifetime int `mapstructure:"conn_max_lifetime"`
|
||||
DialTimeout int `mapstructure:"dial_timeout"`
|
||||
BlockBufferSize uint8 `mapstructure:"block_buffer_size"`
|
||||
MaxIdleConn int `mapstructure:"max_idle_conn"` // 默认 8
|
||||
MaxOpenConn int `mapstructure:"max_open_conn"` // 默认 16
|
||||
ConnMaxLifetime int `mapstructure:"conn_max_lifetime"` // 秒
|
||||
DialTimeout int `mapstructure:"dial_timeout"` // 秒
|
||||
BlockBufferSize uint8 `mapstructure:"block_buffer_size"` // 默认 32
|
||||
}
|
||||
|
||||
// redisConfig Redis配置
|
||||
|
||||
@@ -9,9 +9,10 @@ import (
|
||||
)
|
||||
|
||||
const (
|
||||
defaultQueueSize = 10_000
|
||||
defaultMaxBatchSize = 1_000
|
||||
defaultFlushEvery = time.Second
|
||||
defaultQueueSize = 10_000
|
||||
defaultMaxBatchSize = 1_000
|
||||
defaultMinBatchSize = 50
|
||||
defaultFlushEvery = time.Second
|
||||
)
|
||||
|
||||
// Config controls queue capacity and flush thresholds for a Writer instance.
|
||||
@@ -25,8 +26,17 @@ type Config struct {
|
||||
// MaxBatchSize triggers a flush when the in-memory batch reaches this count.
|
||||
MaxBatchSize int
|
||||
|
||||
// FlushInterval triggers a time-based flush even when the batch is smaller.
|
||||
// MinBatchSize is the minimum in-memory batch size for time-based flushes.
|
||||
// Zero disables the threshold and preserves legacy interval flush behavior.
|
||||
// When set, interval flushes below this size are skipped unless MaxFlushWait elapses.
|
||||
MinBatchSize int
|
||||
|
||||
// FlushInterval is how often the worker checks whether a time-based flush should run.
|
||||
FlushInterval time.Duration
|
||||
|
||||
// MaxFlushWait forces a flush of any non-empty batch once the oldest item has waited
|
||||
// this long, even if MinBatchSize has not been reached. Zero disables the force path.
|
||||
MaxFlushWait time.Duration
|
||||
}
|
||||
|
||||
// DefaultConfig returns production-friendly defaults aligned with audit log batching.
|
||||
@@ -34,6 +44,7 @@ func DefaultConfig() Config {
|
||||
return Config{
|
||||
QueueSize: defaultQueueSize,
|
||||
MaxBatchSize: defaultMaxBatchSize,
|
||||
MinBatchSize: defaultMinBatchSize,
|
||||
FlushInterval: defaultFlushEvery,
|
||||
}
|
||||
}
|
||||
@@ -45,8 +56,14 @@ func (c Config) validate() error {
|
||||
if c.MaxBatchSize <= 0 {
|
||||
return fmt.Errorf("batchwriter: max batch size must be positive")
|
||||
}
|
||||
if c.MinBatchSize < 0 {
|
||||
return fmt.Errorf("batchwriter: min batch size must be non-negative")
|
||||
}
|
||||
if c.FlushInterval <= 0 {
|
||||
return fmt.Errorf("batchwriter: flush interval must be positive")
|
||||
}
|
||||
if c.MaxFlushWait < 0 {
|
||||
return fmt.Errorf("batchwriter: max flush wait must be non-negative")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -9,22 +9,34 @@ package batchwriter
|
||||
import (
|
||||
"context"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
)
|
||||
|
||||
// FlushFunc persists a batch of queued items. It is invoked from the worker goroutine.
|
||||
type FlushFunc[T any] func(ctx context.Context, items []T) error
|
||||
|
||||
// FlushErrorHandler is called when FlushFunc returns an error. The batch is discarded
|
||||
// after the handler returns; the worker continues processing.
|
||||
type FlushErrorHandler func(ctx context.Context, batchSize int, err error)
|
||||
// FlushErrorHandler is called when FlushFunc returns an error after optional retries.
|
||||
// The batch is discarded after the handler returns; the worker continues processing.
|
||||
// Handlers receive the failed items so callers can release dedup keys or re-queue.
|
||||
type FlushErrorHandler[T any] func(ctx context.Context, items []T, err error)
|
||||
|
||||
// Stats is a point-in-time snapshot of Writer queue and failure counters.
|
||||
type Stats struct {
|
||||
Name string `json:"name"`
|
||||
Depth int `json:"depth"`
|
||||
Cap int `json:"cap"`
|
||||
Drops int64 `json:"drops"`
|
||||
FlushErrors int64 `json:"flush_errors"`
|
||||
Running bool `json:"running"`
|
||||
}
|
||||
|
||||
// Writer buffers items and flushes them by size or interval.
|
||||
type Writer[T any] struct {
|
||||
cfg Config
|
||||
flush FlushFunc[T]
|
||||
|
||||
onFlushError FlushErrorHandler
|
||||
onFlushError FlushErrorHandler[T]
|
||||
onDrop func(T)
|
||||
|
||||
startOnce sync.Once
|
||||
@@ -34,13 +46,16 @@ type Writer[T any] struct {
|
||||
ch chan T
|
||||
workerCtx context.Context
|
||||
done chan struct{}
|
||||
|
||||
drops atomic.Int64
|
||||
flushErrors atomic.Int64
|
||||
}
|
||||
|
||||
// Option configures optional Writer callbacks.
|
||||
type Option[T any] func(*Writer[T])
|
||||
|
||||
// WithFlushErrorHandler registers a callback for flush failures.
|
||||
func WithFlushErrorHandler[T any](handler FlushErrorHandler) Option[T] {
|
||||
func WithFlushErrorHandler[T any](handler FlushErrorHandler[T]) Option[T] {
|
||||
return func(w *Writer[T]) {
|
||||
w.onFlushError = handler
|
||||
}
|
||||
@@ -168,22 +183,37 @@ func (w *Writer[T]) Cap() int {
|
||||
return w.cfg.QueueSize
|
||||
}
|
||||
|
||||
// Stats returns a point-in-time snapshot of queue depth and failure counters.
|
||||
func (w *Writer[T]) Stats() Stats {
|
||||
return Stats{
|
||||
Name: w.cfg.Name,
|
||||
Depth: w.Len(),
|
||||
Cap: w.Cap(),
|
||||
Drops: w.drops.Load(),
|
||||
FlushErrors: w.flushErrors.Load(),
|
||||
Running: w.Running(),
|
||||
}
|
||||
}
|
||||
|
||||
func (w *Writer[T]) run() {
|
||||
ticker := time.NewTicker(w.cfg.FlushInterval)
|
||||
defer ticker.Stop()
|
||||
|
||||
batch := make([]T, 0, w.cfg.MaxBatchSize)
|
||||
var batchStartedAt time.Time
|
||||
flush := func() {
|
||||
if len(batch) == 0 {
|
||||
return
|
||||
}
|
||||
items := append([]T(nil), batch...)
|
||||
if err := w.flush(w.workerCtx, items); err != nil {
|
||||
w.flushErrors.Add(1)
|
||||
if w.onFlushError != nil {
|
||||
w.onFlushError(w.workerCtx, len(items), err)
|
||||
w.onFlushError(w.workerCtx, items, err)
|
||||
}
|
||||
}
|
||||
batch = batch[:0]
|
||||
batchStartedAt = time.Time{}
|
||||
}
|
||||
|
||||
defer func() {
|
||||
@@ -197,19 +227,38 @@ func (w *Writer[T]) run() {
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
if len(batch) == 0 {
|
||||
batchStartedAt = time.Now()
|
||||
}
|
||||
batch = append(batch, item)
|
||||
if len(batch) >= w.cfg.MaxBatchSize {
|
||||
flush()
|
||||
}
|
||||
case <-ticker.C:
|
||||
flush()
|
||||
if w.shouldFlushOnInterval(len(batch), batchStartedAt, time.Now()) {
|
||||
flush()
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (w *Writer[T]) shouldFlushOnInterval(batchLen int, batchStartedAt time.Time, now time.Time) bool {
|
||||
if batchLen == 0 {
|
||||
return false
|
||||
}
|
||||
if w.cfg.MinBatchSize == 0 || batchLen >= w.cfg.MinBatchSize {
|
||||
return true
|
||||
}
|
||||
if w.cfg.MaxFlushWait <= 0 || batchStartedAt.IsZero() {
|
||||
return false
|
||||
}
|
||||
return !now.Before(batchStartedAt.Add(w.cfg.MaxFlushWait))
|
||||
}
|
||||
|
||||
func (w *Writer[T]) notifyDrop(item T) {
|
||||
w.drops.Add(1)
|
||||
if w.onDrop == nil {
|
||||
return
|
||||
}
|
||||
w.onDrop(item)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -36,7 +36,7 @@ func TestWriterFlushesOnMaxBatchSize(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
var (
|
||||
mu sync.Mutex
|
||||
mu sync.Mutex
|
||||
batches [][]int
|
||||
)
|
||||
cfg := DefaultConfig()
|
||||
@@ -98,6 +98,7 @@ func TestWriterFlushesOnInterval(t *testing.T) {
|
||||
)
|
||||
cfg := DefaultConfig()
|
||||
cfg.MaxBatchSize = 100
|
||||
cfg.MinBatchSize = 0
|
||||
cfg.FlushInterval = 20 * time.Millisecond
|
||||
|
||||
writer, err := New[int](cfg, func(_ context.Context, items []int) error {
|
||||
@@ -219,27 +220,189 @@ func TestWriterStopDrainsQueuedItems(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestWriterSkipsIntervalFlushBelowMinBatchSize(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
var (
|
||||
mu sync.Mutex
|
||||
batch []int
|
||||
)
|
||||
cfg := DefaultConfig()
|
||||
cfg.MaxBatchSize = 100
|
||||
cfg.MinBatchSize = 5
|
||||
cfg.FlushInterval = 20 * time.Millisecond
|
||||
|
||||
writer, err := New[int](cfg, func(_ context.Context, items []int) error {
|
||||
mu.Lock()
|
||||
defer mu.Unlock()
|
||||
batch = append([]int(nil), items...)
|
||||
return nil
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("New() error = %v", err)
|
||||
}
|
||||
|
||||
writer.Start(context.Background())
|
||||
t.Cleanup(func() {
|
||||
stopCtx, cancel := context.WithTimeout(context.Background(), time.Second)
|
||||
defer cancel()
|
||||
if err := writer.Stop(stopCtx); err != nil {
|
||||
t.Fatalf("Stop() error = %v", err)
|
||||
}
|
||||
})
|
||||
|
||||
for i := range 3 {
|
||||
if !writer.TryEnqueue(i + 1) {
|
||||
t.Fatalf("TryEnqueue(%d) = false, want true", i+1)
|
||||
}
|
||||
}
|
||||
|
||||
time.Sleep(100 * time.Millisecond)
|
||||
|
||||
mu.Lock()
|
||||
got := batch
|
||||
mu.Unlock()
|
||||
|
||||
if len(got) != 0 {
|
||||
t.Fatalf("interval flush with below-min batch = %v, want no flush", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestWriterFlushesOnIntervalWhenMinBatchSizeReached(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
var (
|
||||
mu sync.Mutex
|
||||
batch []int
|
||||
)
|
||||
cfg := DefaultConfig()
|
||||
cfg.MaxBatchSize = 100
|
||||
cfg.MinBatchSize = 3
|
||||
cfg.FlushInterval = 20 * time.Millisecond
|
||||
|
||||
writer, err := New[int](cfg, func(_ context.Context, items []int) error {
|
||||
mu.Lock()
|
||||
defer mu.Unlock()
|
||||
batch = append([]int(nil), items...)
|
||||
return nil
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("New() error = %v", err)
|
||||
}
|
||||
|
||||
writer.Start(context.Background())
|
||||
t.Cleanup(func() {
|
||||
stopCtx, cancel := context.WithTimeout(context.Background(), time.Second)
|
||||
defer cancel()
|
||||
if err := writer.Stop(stopCtx); err != nil {
|
||||
t.Fatalf("Stop() error = %v", err)
|
||||
}
|
||||
})
|
||||
|
||||
for i := range 3 {
|
||||
if !writer.TryEnqueue(i + 1) {
|
||||
t.Fatalf("TryEnqueue(%d) = false, want true", i+1)
|
||||
}
|
||||
}
|
||||
|
||||
deadline := time.Now().Add(time.Second)
|
||||
for {
|
||||
mu.Lock()
|
||||
ready := len(batch) == 3
|
||||
mu.Unlock()
|
||||
if ready || time.Now().After(deadline) {
|
||||
break
|
||||
}
|
||||
time.Sleep(5 * time.Millisecond)
|
||||
}
|
||||
|
||||
mu.Lock()
|
||||
got := batch
|
||||
mu.Unlock()
|
||||
|
||||
want := []int{1, 2, 3}
|
||||
if diff := cmp.Diff(want, got); diff != "" {
|
||||
t.Fatalf("interval flush at min batch size mismatch (-want +got):\n%s", diff)
|
||||
}
|
||||
}
|
||||
|
||||
func TestWriterForcesFlushAfterMaxFlushWait(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
var (
|
||||
mu sync.Mutex
|
||||
batch []int
|
||||
)
|
||||
cfg := DefaultConfig()
|
||||
cfg.MaxBatchSize = 100
|
||||
cfg.MinBatchSize = 50
|
||||
cfg.FlushInterval = 20 * time.Millisecond
|
||||
cfg.MaxFlushWait = 80 * time.Millisecond
|
||||
|
||||
writer, err := New[int](cfg, func(_ context.Context, items []int) error {
|
||||
mu.Lock()
|
||||
defer mu.Unlock()
|
||||
batch = append([]int(nil), items...)
|
||||
return nil
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("New() error = %v", err)
|
||||
}
|
||||
|
||||
writer.Start(context.Background())
|
||||
t.Cleanup(func() {
|
||||
stopCtx, cancel := context.WithTimeout(context.Background(), time.Second)
|
||||
defer cancel()
|
||||
if err := writer.Stop(stopCtx); err != nil {
|
||||
t.Fatalf("Stop() error = %v", err)
|
||||
}
|
||||
})
|
||||
|
||||
if !writer.TryEnqueue(1) {
|
||||
t.Fatal("TryEnqueue(1) = false, want true")
|
||||
}
|
||||
|
||||
deadline := time.Now().Add(time.Second)
|
||||
for {
|
||||
mu.Lock()
|
||||
ready := len(batch) == 1
|
||||
mu.Unlock()
|
||||
if ready || time.Now().After(deadline) {
|
||||
break
|
||||
}
|
||||
time.Sleep(5 * time.Millisecond)
|
||||
}
|
||||
|
||||
mu.Lock()
|
||||
got := batch
|
||||
mu.Unlock()
|
||||
if diff := cmp.Diff([]int{1}, got); diff != "" {
|
||||
t.Fatalf("max flush wait mismatch (-want +got):\n%s", diff)
|
||||
}
|
||||
}
|
||||
|
||||
func TestWriterInvokesFlushErrorHandler(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
cfg := DefaultConfig()
|
||||
cfg.Name = "test-flush-err"
|
||||
cfg.MaxBatchSize = 1
|
||||
cfg.FlushInterval = time.Hour
|
||||
|
||||
flushErr := errors.New("flush failed")
|
||||
var (
|
||||
mu sync.Mutex
|
||||
errCount int
|
||||
batchSize int
|
||||
mu sync.Mutex
|
||||
errCount int
|
||||
gotItems []int
|
||||
)
|
||||
|
||||
writer, err := New[int](cfg, func(context.Context, []int) error {
|
||||
return flushErr
|
||||
}, WithFlushErrorHandler[int](func(_ context.Context, size int, err error) {
|
||||
}, WithFlushErrorHandler[int](func(_ context.Context, items []int, err error) {
|
||||
mu.Lock()
|
||||
defer mu.Unlock()
|
||||
errCount++
|
||||
batchSize = size
|
||||
gotItems = append([]int(nil), items...)
|
||||
if !errors.Is(err, flushErr) {
|
||||
t.Errorf("flush error = %v, want %v", err, flushErr)
|
||||
}
|
||||
@@ -272,13 +435,61 @@ func TestWriterInvokesFlushErrorHandler(t *testing.T) {
|
||||
|
||||
mu.Lock()
|
||||
gotCount := errCount
|
||||
gotSize := batchSize
|
||||
items := gotItems
|
||||
mu.Unlock()
|
||||
|
||||
if gotCount != 1 {
|
||||
t.Fatalf("flush error handler count = %d, want 1", gotCount)
|
||||
}
|
||||
if gotSize != 1 {
|
||||
t.Fatalf("flush error handler batch size = %d, want 1", gotSize)
|
||||
if diff := cmp.Diff([]int{7}, items); diff != "" {
|
||||
t.Fatalf("flush error handler items mismatch (-want +got):\n%s", diff)
|
||||
}
|
||||
}
|
||||
|
||||
stats := writer.Stats()
|
||||
if stats.FlushErrors != 1 {
|
||||
t.Fatalf("Stats().FlushErrors = %d, want 1", stats.FlushErrors)
|
||||
}
|
||||
if stats.Name != "test-flush-err" {
|
||||
t.Fatalf("Stats().Name = %q, want test-flush-err", stats.Name)
|
||||
}
|
||||
}
|
||||
|
||||
func TestWriterStatsTracksDrops(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
cfg := DefaultConfig()
|
||||
cfg.Name = "test-drops"
|
||||
cfg.QueueSize = 1
|
||||
cfg.MaxBatchSize = 10
|
||||
cfg.FlushInterval = time.Hour
|
||||
|
||||
writer, err := New[int](cfg, func(context.Context, []int) error { return nil })
|
||||
if err != nil {
|
||||
t.Fatalf("New() error = %v", err)
|
||||
}
|
||||
|
||||
writer.Start(context.Background())
|
||||
t.Cleanup(func() {
|
||||
stopCtx, cancel := context.WithTimeout(context.Background(), time.Second)
|
||||
defer cancel()
|
||||
_ = writer.Stop(stopCtx)
|
||||
})
|
||||
|
||||
if !writer.TryEnqueue(1) {
|
||||
t.Fatal("TryEnqueue(1) = false, want true")
|
||||
}
|
||||
if writer.TryEnqueue(2) {
|
||||
t.Fatal("TryEnqueue(2) = true, want false")
|
||||
}
|
||||
|
||||
stats := writer.Stats()
|
||||
if stats.Drops != 1 {
|
||||
t.Fatalf("Stats().Drops = %d, want 1", stats.Drops)
|
||||
}
|
||||
if stats.Cap != 1 {
|
||||
t.Fatalf("Stats().Cap = %d, want 1", stats.Cap)
|
||||
}
|
||||
if !stats.Running {
|
||||
t.Fatal("Stats().Running = false, want true")
|
||||
}
|
||||
}
|
||||
|
||||
+21
-74
@@ -7,32 +7,32 @@ package db
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"log"
|
||||
"net/url"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/ClickHouse/clickhouse-go/v2"
|
||||
"github.com/ClickHouse/clickhouse-go/v2/lib/driver"
|
||||
"github.com/Rain-kl/Wavelet/internal/config"
|
||||
"go.opentelemetry.io/otel/attribute"
|
||||
clickhouseDriver "gorm.io/driver/clickhouse"
|
||||
"gorm.io/gorm"
|
||||
"gorm.io/plugin/opentelemetry/tracing"
|
||||
)
|
||||
|
||||
const (
|
||||
clickhouseMaxExecTime = 60 // ClickHouse 最大执行时间(秒)
|
||||
clickhouseReadTimeoutFactor = 2 // ReadTimeout 为 DialTimeout 的倍数
|
||||
|
||||
// async_insert 仅挂在运行时 ChConn(写路径)上,不进入 migrator OpenDB:
|
||||
// 迁移/DDL 需要同步可见结果,且不应走异步 insert 缓冲。
|
||||
//
|
||||
// 为何启用:batchwriter 仍可能在短间隔内写出相对小的块;服务端 async_insert
|
||||
// 把多次 INSERT 合并成更大 part,减轻 3c6g 上 background merge 的 CPU 压力。
|
||||
// wait_for_async_insert=1:调用方在 flush 返回前等待落盘,避免进程崩溃丢批。
|
||||
// max_data_size / busy_timeout:约 10MB 或 ~2s 触发刷出,在延迟与 part 数之间折中。
|
||||
clickhouseAsyncInsertMaxDataSize = 10_000_000
|
||||
clickhouseAsyncInsertBusyTimeoutMs = 2000
|
||||
)
|
||||
|
||||
var (
|
||||
// ChConn ClickHouse 原生连接实例,用于批量写入
|
||||
// ChConn ClickHouse 原生连接实例,用于批量写入与查询
|
||||
ChConn driver.Conn
|
||||
|
||||
chDB *gorm.DB
|
||||
)
|
||||
|
||||
func init() {
|
||||
@@ -57,39 +57,11 @@ func init() {
|
||||
log.Fatalf("[ClickHouse] ping failed: %v\n", err)
|
||||
}
|
||||
|
||||
chDB, err = gorm.Open(clickhouseDriver.New(clickhouseDriver.Config{
|
||||
DSN: buildClickHouseDSN(),
|
||||
}), &gorm.Config{
|
||||
SkipDefaultTransaction: true,
|
||||
})
|
||||
if err != nil {
|
||||
log.Fatalf("[ClickHouse] init gorm connection failed: %v\n", err)
|
||||
}
|
||||
|
||||
if err = chDB.Use(
|
||||
tracing.NewPlugin(
|
||||
tracing.WithoutMetrics(),
|
||||
tracing.WithAttributes(
|
||||
attribute.String("db.instance", cfg.Database),
|
||||
attribute.String("db.system", "ClickHouse"),
|
||||
),
|
||||
),
|
||||
); err != nil {
|
||||
log.Fatalf("[ClickHouse] init trace failed: %v\n", err)
|
||||
}
|
||||
|
||||
sqlDB, err := chDB.DB()
|
||||
if err != nil {
|
||||
log.Fatalf("[ClickHouse] load sql db failed: %v\n", err)
|
||||
}
|
||||
|
||||
sqlDB.SetMaxIdleConns(cfg.MaxIdleConn)
|
||||
sqlDB.SetMaxOpenConns(cfg.MaxOpenConn)
|
||||
sqlDB.SetConnMaxLifetime(time.Duration(cfg.ConnMaxLifetime) * time.Second)
|
||||
|
||||
log.Println("[ClickHouse] connection established successfully")
|
||||
}
|
||||
|
||||
// buildClickHouseOptions builds the runtime native client options (queries + batch inserts).
|
||||
// Migrator uses a separate clickhouse.OpenDB path without async_insert settings.
|
||||
func buildClickHouseOptions() *clickhouse.Options {
|
||||
cfg := config.Config.ClickHouse
|
||||
|
||||
@@ -101,7 +73,11 @@ func buildClickHouseOptions() *clickhouse.Options {
|
||||
Password: cfg.Password,
|
||||
},
|
||||
Settings: clickhouse.Settings{
|
||||
"max_execution_time": clickhouseMaxExecTime,
|
||||
"max_execution_time": clickhouseMaxExecTime,
|
||||
"async_insert": 1,
|
||||
"wait_for_async_insert": 1,
|
||||
"async_insert_max_data_size": clickhouseAsyncInsertMaxDataSize,
|
||||
"async_insert_busy_timeout_ms": clickhouseAsyncInsertBusyTimeoutMs,
|
||||
},
|
||||
Compression: &clickhouse.Compression{
|
||||
Method: clickhouse.CompressionLZ4,
|
||||
@@ -115,38 +91,9 @@ func buildClickHouseOptions() *clickhouse.Options {
|
||||
}
|
||||
}
|
||||
|
||||
func buildClickHouseDSN() string {
|
||||
cfg := config.Config.ClickHouse
|
||||
|
||||
chURL := &url.URL{
|
||||
Scheme: "clickhouse",
|
||||
Host: strings.Join(cfg.Hosts, ","),
|
||||
Path: "/" + cfg.Database,
|
||||
}
|
||||
if cfg.Username != "" || cfg.Password != "" {
|
||||
chURL.User = url.UserPassword(cfg.Username, cfg.Password)
|
||||
}
|
||||
|
||||
query := chURL.Query()
|
||||
query.Set("dial_timeout", fmt.Sprintf("%ds", cfg.DialTimeout))
|
||||
query.Set("read_timeout", fmt.Sprintf("%ds", cfg.DialTimeout*clickhouseReadTimeoutFactor))
|
||||
query.Set("max_execution_time", strconv.Itoa(clickhouseMaxExecTime))
|
||||
chURL.RawQuery = query.Encode()
|
||||
|
||||
return chURL.String()
|
||||
}
|
||||
|
||||
// ChDB returns a context-aware GORM ClickHouse instance.
|
||||
func ChDB(ctx context.Context) *gorm.DB {
|
||||
if chDB == nil {
|
||||
return nil
|
||||
}
|
||||
return chDB.WithContext(ctx)
|
||||
}
|
||||
|
||||
// SetChDBForTest sets the package-level ClickHouse GORM instance for testing.
|
||||
func SetChDBForTest(d *gorm.DB) {
|
||||
chDB = d
|
||||
// ChConnReady reports whether the native ClickHouse connection is initialized.
|
||||
func ChConnReady() bool {
|
||||
return ChConn != nil
|
||||
}
|
||||
|
||||
// SetChConnForTest sets the package-level native ClickHouse connection for testing.
|
||||
|
||||
@@ -0,0 +1,19 @@
|
||||
-- +goose Up
|
||||
-- Add TTL policies to analytics tables so ClickHouse can expire rows automatically.
|
||||
ALTER TABLE w_user_access_logs MODIFY TTL created_at + INTERVAL 180 DAY;
|
||||
|
||||
-- DateTime64 columns must be cast for TTL (ClickHouse requires DateTime/Date in TTL expr).
|
||||
ALTER TABLE of_node_access_logs MODIFY TTL toDateTime(logged_at) + INTERVAL 90 DAY;
|
||||
|
||||
ALTER TABLE of_node_metric_snapshots MODIFY TTL toDateTime(captured_at) + INTERVAL 30 DAY;
|
||||
|
||||
ALTER TABLE of_node_request_reports MODIFY TTL toDateTime(window_ended_at) + INTERVAL 30 DAY;
|
||||
|
||||
ALTER TABLE of_node_obs_openresty MODIFY TTL toDateTime(captured_at) + INTERVAL 30 DAY;
|
||||
|
||||
ALTER TABLE of_node_obs_frps MODIFY TTL toDateTime(captured_at) + INTERVAL 30 DAY;
|
||||
|
||||
ALTER TABLE of_node_obs_frpc MODIFY TTL toDateTime(captured_at) + INTERVAL 30 DAY;
|
||||
|
||||
-- +goose Down
|
||||
-- TTL changes cannot be safely reversed without recreating tables.
|
||||
@@ -0,0 +1,28 @@
|
||||
-- +goose Up
|
||||
CREATE TABLE IF NOT EXISTS of_node_traffic_hourly
|
||||
(
|
||||
node_id String,
|
||||
hour DateTime,
|
||||
request_count UInt64,
|
||||
error_count UInt64,
|
||||
unique_visitor_count UInt64
|
||||
)
|
||||
ENGINE = SummingMergeTree()
|
||||
PARTITION BY toYYYYMM(hour)
|
||||
ORDER BY (node_id, hour);
|
||||
|
||||
CREATE MATERIALIZED VIEW IF NOT EXISTS of_node_traffic_hourly_mv
|
||||
TO of_node_traffic_hourly
|
||||
AS
|
||||
SELECT
|
||||
node_id,
|
||||
toStartOfHour(window_ended_at) AS hour,
|
||||
sum(request_count) AS request_count,
|
||||
sum(error_count) AS error_count,
|
||||
sum(unique_visitor_count) AS unique_visitor_count
|
||||
FROM of_node_request_reports
|
||||
GROUP BY node_id, hour;
|
||||
|
||||
-- +goose Down
|
||||
DROP VIEW IF EXISTS of_node_traffic_hourly_mv;
|
||||
DROP TABLE IF EXISTS of_node_traffic_hourly;
|
||||
+80
@@ -0,0 +1,80 @@
|
||||
-- +goose Up
|
||||
-- Hourly capacity rollups (avg CPU/memory + counter min/max for in-hour delta approximation).
|
||||
-- Network/disk counters are cumulative; max-min within an hour approximates that hour's delta
|
||||
-- (cross-hour continuity is intentionally approximate for dashboard trends).
|
||||
CREATE TABLE IF NOT EXISTS of_node_metric_capacity_hourly
|
||||
(
|
||||
node_id String,
|
||||
hour DateTime,
|
||||
cpu_usage_sum SimpleAggregateFunction(sum, Float64),
|
||||
cpu_usage_count SimpleAggregateFunction(sum, UInt64),
|
||||
memory_usage_sum SimpleAggregateFunction(sum, Float64),
|
||||
memory_usage_count SimpleAggregateFunction(sum, UInt64),
|
||||
network_rx_min SimpleAggregateFunction(min, Int64),
|
||||
network_rx_max SimpleAggregateFunction(max, Int64),
|
||||
network_tx_min SimpleAggregateFunction(min, Int64),
|
||||
network_tx_max SimpleAggregateFunction(max, Int64),
|
||||
disk_read_min SimpleAggregateFunction(min, Int64),
|
||||
disk_read_max SimpleAggregateFunction(max, Int64),
|
||||
disk_write_min SimpleAggregateFunction(min, Int64),
|
||||
disk_write_max SimpleAggregateFunction(max, Int64)
|
||||
)
|
||||
ENGINE = AggregatingMergeTree()
|
||||
PARTITION BY toYYYYMM(hour)
|
||||
ORDER BY (node_id, hour)
|
||||
TTL hour + INTERVAL 30 DAY;
|
||||
|
||||
CREATE MATERIALIZED VIEW IF NOT EXISTS of_node_metric_capacity_hourly_mv
|
||||
TO of_node_metric_capacity_hourly
|
||||
AS
|
||||
SELECT
|
||||
node_id,
|
||||
toStartOfHour(captured_at) AS hour,
|
||||
sum(cpu_usage_percent) AS cpu_usage_sum,
|
||||
toUInt64(count()) AS cpu_usage_count,
|
||||
sum(if(memory_total_bytes > 0, (memory_used_bytes * 100.0) / memory_total_bytes, 0)) AS memory_usage_sum,
|
||||
toUInt64(countIf(memory_total_bytes > 0)) AS memory_usage_count,
|
||||
min(network_rx_bytes) AS network_rx_min,
|
||||
max(network_rx_bytes) AS network_rx_max,
|
||||
min(network_tx_bytes) AS network_tx_min,
|
||||
max(network_tx_bytes) AS network_tx_max,
|
||||
min(disk_read_bytes) AS disk_read_min,
|
||||
max(disk_read_bytes) AS disk_read_max,
|
||||
min(disk_write_bytes) AS disk_write_min,
|
||||
max(disk_write_bytes) AS disk_write_max
|
||||
FROM of_node_metric_snapshots
|
||||
GROUP BY node_id, hour;
|
||||
|
||||
-- Hourly OpenResty counter rollups (min/max per node-hour for delta approximation).
|
||||
CREATE TABLE IF NOT EXISTS of_node_openresty_hourly
|
||||
(
|
||||
node_id String,
|
||||
hour DateTime,
|
||||
openresty_rx_min SimpleAggregateFunction(min, Int64),
|
||||
openresty_rx_max SimpleAggregateFunction(max, Int64),
|
||||
openresty_tx_min SimpleAggregateFunction(min, Int64),
|
||||
openresty_tx_max SimpleAggregateFunction(max, Int64)
|
||||
)
|
||||
ENGINE = AggregatingMergeTree()
|
||||
PARTITION BY toYYYYMM(hour)
|
||||
ORDER BY (node_id, hour)
|
||||
TTL hour + INTERVAL 30 DAY;
|
||||
|
||||
CREATE MATERIALIZED VIEW IF NOT EXISTS of_node_openresty_hourly_mv
|
||||
TO of_node_openresty_hourly
|
||||
AS
|
||||
SELECT
|
||||
node_id,
|
||||
toStartOfHour(captured_at) AS hour,
|
||||
min(openresty_rx_bytes) AS openresty_rx_min,
|
||||
max(openresty_rx_bytes) AS openresty_rx_max,
|
||||
min(openresty_tx_bytes) AS openresty_tx_min,
|
||||
max(openresty_tx_bytes) AS openresty_tx_max
|
||||
FROM of_node_obs_openresty
|
||||
GROUP BY node_id, hour;
|
||||
|
||||
-- +goose Down
|
||||
DROP VIEW IF EXISTS of_node_openresty_hourly_mv;
|
||||
DROP TABLE IF EXISTS of_node_openresty_hourly;
|
||||
DROP VIEW IF EXISTS of_node_metric_capacity_hourly_mv;
|
||||
DROP TABLE IF EXISTS of_node_metric_capacity_hourly;
|
||||
@@ -0,0 +1,42 @@
|
||||
-- +goose Up
|
||||
-- Hourly traffic rollups: 30d TTL + UV aggregation semantics.
|
||||
--
|
||||
-- unique_visitor_count on of_node_request_reports is per short report window
|
||||
-- (agent local distinct count for that window only). Summing those values in the
|
||||
-- MV (and again via SummingMergeTree part merges) invents a "true UV" number that
|
||||
-- double-counts visitors across windows. Prefer max() as a peak-window estimate;
|
||||
-- still NOT distinct visitors across the hour — UI/API must not overclaim.
|
||||
|
||||
ALTER TABLE of_node_traffic_hourly
|
||||
MODIFY TTL toDateTime(hour) + INTERVAL 30 DAY;
|
||||
|
||||
DROP VIEW IF EXISTS of_node_traffic_hourly_mv;
|
||||
|
||||
CREATE MATERIALIZED VIEW of_node_traffic_hourly_mv
|
||||
TO of_node_traffic_hourly
|
||||
AS
|
||||
SELECT
|
||||
node_id,
|
||||
toStartOfHour(window_ended_at) AS hour,
|
||||
sum(request_count) AS request_count,
|
||||
sum(error_count) AS error_count,
|
||||
-- Peak per-window UV estimate for the hour; not true cross-window distinct UV.
|
||||
max(unique_visitor_count) AS unique_visitor_count
|
||||
FROM of_node_request_reports
|
||||
GROUP BY node_id, hour;
|
||||
|
||||
-- +goose Down
|
||||
-- TTL reverse is not safe without table rewrite; restore prior MV definition only.
|
||||
DROP VIEW IF EXISTS of_node_traffic_hourly_mv;
|
||||
|
||||
CREATE MATERIALIZED VIEW of_node_traffic_hourly_mv
|
||||
TO of_node_traffic_hourly
|
||||
AS
|
||||
SELECT
|
||||
node_id,
|
||||
toStartOfHour(window_ended_at) AS hour,
|
||||
sum(request_count) AS request_count,
|
||||
sum(error_count) AS error_count,
|
||||
sum(unique_visitor_count) AS unique_visitor_count
|
||||
FROM of_node_request_reports
|
||||
GROUP BY node_id, hour;
|
||||
+66
@@ -0,0 +1,66 @@
|
||||
-- +goose Up
|
||||
-- One-time historical backfill for hours not yet present in rollup tables.
|
||||
-- MV only ingests rows after creation; without this, 24h charts rely on raw merge forever.
|
||||
-- ANTI JOIN avoids double-counting hours already filled by the live MV.
|
||||
|
||||
INSERT INTO of_node_metric_capacity_hourly
|
||||
SELECT
|
||||
s.node_id,
|
||||
toStartOfHour(s.captured_at) AS hour,
|
||||
sum(s.cpu_usage_percent) AS cpu_usage_sum,
|
||||
toUInt64(count()) AS cpu_usage_count,
|
||||
sum(if(s.memory_total_bytes > 0, (s.memory_used_bytes * 100.0) / s.memory_total_bytes, 0)) AS memory_usage_sum,
|
||||
toUInt64(countIf(s.memory_total_bytes > 0)) AS memory_usage_count,
|
||||
min(s.network_rx_bytes) AS network_rx_min,
|
||||
max(s.network_rx_bytes) AS network_rx_max,
|
||||
min(s.network_tx_bytes) AS network_tx_min,
|
||||
max(s.network_tx_bytes) AS network_tx_max,
|
||||
min(s.disk_read_bytes) AS disk_read_min,
|
||||
max(s.disk_read_bytes) AS disk_read_max,
|
||||
min(s.disk_write_bytes) AS disk_write_min,
|
||||
max(s.disk_write_bytes) AS disk_write_max
|
||||
FROM of_node_metric_snapshots AS s
|
||||
ANTI JOIN
|
||||
(
|
||||
SELECT
|
||||
node_id,
|
||||
hour
|
||||
FROM of_node_metric_capacity_hourly
|
||||
GROUP BY
|
||||
node_id,
|
||||
hour
|
||||
) AS existing
|
||||
ON s.node_id = existing.node_id AND toStartOfHour(s.captured_at) = existing.hour
|
||||
WHERE s.captured_at >= now() - INTERVAL 30 DAY
|
||||
GROUP BY
|
||||
s.node_id,
|
||||
hour;
|
||||
|
||||
INSERT INTO of_node_openresty_hourly
|
||||
SELECT
|
||||
s.node_id,
|
||||
toStartOfHour(s.captured_at) AS hour,
|
||||
min(s.openresty_rx_bytes) AS openresty_rx_min,
|
||||
max(s.openresty_rx_bytes) AS openresty_rx_max,
|
||||
min(s.openresty_tx_bytes) AS openresty_tx_min,
|
||||
max(s.openresty_tx_bytes) AS openresty_tx_max
|
||||
FROM of_node_obs_openresty AS s
|
||||
ANTI JOIN
|
||||
(
|
||||
SELECT
|
||||
node_id,
|
||||
hour
|
||||
FROM of_node_openresty_hourly
|
||||
GROUP BY
|
||||
node_id,
|
||||
hour
|
||||
) AS existing
|
||||
ON s.node_id = existing.node_id AND toStartOfHour(s.captured_at) = existing.hour
|
||||
WHERE s.captured_at >= now() - INTERVAL 30 DAY
|
||||
GROUP BY
|
||||
s.node_id,
|
||||
hour;
|
||||
|
||||
-- +goose Down
|
||||
-- Backfill is additive; down does not remove historical rollup rows (TTL still applies).
|
||||
SELECT 1;
|
||||
@@ -137,7 +137,7 @@ CREATE INDEX IF NOT EXISTS idx_templates_created_at ON templates (created_at);
|
||||
CREATE INDEX IF NOT EXISTS idx_templates_updated_at ON templates (updated_at);
|
||||
|
||||
INSERT INTO system_configs (key, value, type, visibility, description, created_at, updated_at) VALUES
|
||||
('cap_login_enabled', 'true', 'system', 1, '是否启用登录人机验证(true/false)', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP),
|
||||
('cap_login_enabled', 'false', 'system', 1, '是否启用登录人机验证(true/false)', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP),
|
||||
('cap_auto_solve', 'true', 'system', 1, '打开页面后是否自动开始计算,关闭则需用户手动点击触发', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP),
|
||||
('cap_challenge_count', '1', 'system', 0, '客户端需求解的 PoW 难题总数,默认 1,推荐 1~5', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP),
|
||||
('cap_challenge_size', '32', 'system', 0, '人机验证盐值长度', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP),
|
||||
|
||||
@@ -7,10 +7,6 @@ UPDATE w_system_configs
|
||||
SET value = 'false', updated_at = CURRENT_TIMESTAMP
|
||||
WHERE key = 'password_register_enabled' AND value = 'true';
|
||||
|
||||
UPDATE w_system_configs
|
||||
SET value = 'true', updated_at = CURRENT_TIMESTAMP
|
||||
WHERE key = 'cap_login_enabled' AND value = 'false';
|
||||
|
||||
-- +goose Down
|
||||
UPDATE w_system_configs
|
||||
SET value = 'true', updated_at = CURRENT_TIMESTAMP
|
||||
@@ -18,8 +14,4 @@ WHERE key = 'registration_enabled' AND value = 'false';
|
||||
|
||||
UPDATE w_system_configs
|
||||
SET value = 'true', updated_at = CURRENT_TIMESTAMP
|
||||
WHERE key = 'password_register_enabled' AND value = 'false';
|
||||
|
||||
UPDATE w_system_configs
|
||||
SET value = 'false', updated_at = CURRENT_TIMESTAMP
|
||||
WHERE key = 'cap_login_enabled' AND value = 'true';
|
||||
WHERE key = 'password_register_enabled' AND value = 'false';
|
||||
@@ -137,7 +137,7 @@ CREATE INDEX IF NOT EXISTS idx_templates_created_at ON templates (created_at);
|
||||
CREATE INDEX IF NOT EXISTS idx_templates_updated_at ON templates (updated_at);
|
||||
|
||||
INSERT INTO system_configs (key, value, type, visibility, description, created_at, updated_at) VALUES
|
||||
('cap_login_enabled', 'true', 'system', 1, '是否启用登录人机验证(true/false)', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP),
|
||||
('cap_login_enabled', 'false', 'system', 1, '是否启用登录人机验证(true/false)', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP),
|
||||
('cap_auto_solve', 'true', 'system', 1, '打开页面后是否自动开始计算,关闭则需用户手动点击触发', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP),
|
||||
('cap_challenge_count', '1', 'system', 0, '客户端需求解的 PoW 难题总数,默认 1,推荐 1~5', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP),
|
||||
('cap_challenge_size', '32', 'system', 0, '人机验证盐值长度', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP),
|
||||
|
||||
@@ -7,10 +7,6 @@ UPDATE w_system_configs
|
||||
SET value = 'false', updated_at = CURRENT_TIMESTAMP
|
||||
WHERE key = 'password_register_enabled' AND value = 'true';
|
||||
|
||||
UPDATE w_system_configs
|
||||
SET value = 'true', updated_at = CURRENT_TIMESTAMP
|
||||
WHERE key = 'cap_login_enabled' AND value = 'false';
|
||||
|
||||
-- +goose Down
|
||||
UPDATE w_system_configs
|
||||
SET value = 'true', updated_at = CURRENT_TIMESTAMP
|
||||
@@ -18,8 +14,4 @@ WHERE key = 'registration_enabled' AND value = 'false';
|
||||
|
||||
UPDATE w_system_configs
|
||||
SET value = 'true', updated_at = CURRENT_TIMESTAMP
|
||||
WHERE key = 'password_register_enabled' AND value = 'false';
|
||||
|
||||
UPDATE w_system_configs
|
||||
SET value = 'false', updated_at = CURRENT_TIMESTAMP
|
||||
WHERE key = 'cap_login_enabled' AND value = 'true';
|
||||
WHERE key = 'password_register_enabled' AND value = 'false';
|
||||
@@ -24,6 +24,8 @@ type openFlareAccessLogBucketAggregateRow struct {
|
||||
SuccessCount int64 `gorm:"column:success_count"`
|
||||
ClientErrorCount int64 `gorm:"column:client_error_count"`
|
||||
ServerErrorCount int64 `gorm:"column:server_error_count"`
|
||||
UniqueIPCount int64 `gorm:"column:unique_ip_count"`
|
||||
UniqueHostCount int64 `gorm:"column:unique_host_count"`
|
||||
}
|
||||
|
||||
type openFlareAccessLogBucketDimensionRow struct {
|
||||
@@ -52,9 +54,45 @@ type openFlareAccessLogIPTrendRow struct {
|
||||
RequestCount int64 `gorm:"column:request_count"`
|
||||
}
|
||||
|
||||
// ListOpenFlareAccessLogsForWAFIPGroup lists access logs in a time window for automatic IP group rules.
|
||||
func ListOpenFlareAccessLogsForWAFIPGroup(ctx context.Context, query OpenFlareAccessLogQuery) ([]*OpenFlareAccessLog, error) {
|
||||
return ListOpenFlareAccessLogs(ctx, query)
|
||||
type openFlareAccessLogWAFIPAggregateRow struct {
|
||||
RemoteAddr string
|
||||
RequestCount int64
|
||||
Status404Count int64
|
||||
ClientErrorCount int64
|
||||
ServerErrorCount int64
|
||||
IPHostCount int64
|
||||
LastSeenEpoch int64
|
||||
StatusCounts map[int]int64
|
||||
}
|
||||
|
||||
// ListOpenFlareAccessLogWAFIPAggregates returns per-IP aggregates for WAF automatic rules.
|
||||
func ListOpenFlareAccessLogWAFIPAggregates(ctx context.Context, query OpenFlareAccessLogQuery) ([]*OpenFlareAccessLogWAFIPAggregate, error) {
|
||||
rows, err := currentAccessLogStore().WAFIPAggregates(ctx, query)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
result := make([]*OpenFlareAccessLogWAFIPAggregate, 0, len(rows))
|
||||
for _, row := range rows {
|
||||
remoteAddr := strings.TrimSpace(row.RemoteAddr)
|
||||
if remoteAddr == "" {
|
||||
continue
|
||||
}
|
||||
statusCounts := make(map[int]int, len(row.StatusCounts))
|
||||
for code, count := range row.StatusCounts {
|
||||
statusCounts[code] = int(count)
|
||||
}
|
||||
result = append(result, &OpenFlareAccessLogWAFIPAggregate{
|
||||
RemoteAddr: remoteAddr,
|
||||
RequestCount: int(row.RequestCount),
|
||||
Status404Count: int(row.Status404Count),
|
||||
ClientErrorCount: int(row.ClientErrorCount),
|
||||
ServerErrorCount: int(row.ServerErrorCount),
|
||||
IPHostCount: int(row.IPHostCount),
|
||||
LastSeenEpoch: row.LastSeenEpoch,
|
||||
StatusCounts: statusCounts,
|
||||
})
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// InsertOpenFlareAccessLogsBatch inserts access log rows into ClickHouse.
|
||||
@@ -79,24 +117,17 @@ func ListOpenFlareAccessLogRegionCounts(ctx context.Context, nodeID string, sinc
|
||||
|
||||
// ListOpenFlareAccessLogBuckets lists folded access log buckets.
|
||||
func ListOpenFlareAccessLogBuckets(ctx context.Context, query OpenFlareAccessLogBucketQuery) ([]*OpenFlareAccessLogBucketRow, error) {
|
||||
rows, err := buildOpenFlareAccessLogBucketRows(ctx, query)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
start, end := openFlareAccessLogPaginateBounds(len(rows), query.Page, query.PageSize)
|
||||
if start >= len(rows) {
|
||||
return []*OpenFlareAccessLogBucketRow{}, nil
|
||||
}
|
||||
return rows[start:end], nil
|
||||
return buildOpenFlareAccessLogBucketRows(ctx, query)
|
||||
}
|
||||
|
||||
// CountOpenFlareAccessLogBuckets counts folded access log buckets.
|
||||
func CountOpenFlareAccessLogBuckets(ctx context.Context, query OpenFlareAccessLogBucketQuery) (int64, error) {
|
||||
rows, err := buildOpenFlareAccessLogBucketRows(ctx, query)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
filter := openFlareAccessLogQueryFromBucket(query)
|
||||
bucketSeconds := int64(query.FoldMinutes * secondsPerMinute)
|
||||
if bucketSeconds <= 0 {
|
||||
bucketSeconds = 180
|
||||
}
|
||||
return int64(len(rows)), nil
|
||||
return currentAccessLogStore().CountBuckets(ctx, filter, bucketSeconds)
|
||||
}
|
||||
|
||||
// ListOpenFlareAccessLogBucketIPs lists folded IP rows for a bucket window.
|
||||
@@ -123,24 +154,13 @@ func CountOpenFlareAccessLogBucketIPs(ctx context.Context, query OpenFlareAccess
|
||||
|
||||
// ListOpenFlareAccessLogIPSummaries lists IP summaries.
|
||||
func ListOpenFlareAccessLogIPSummaries(ctx context.Context, query OpenFlareAccessLogIPSummaryQuery, recentSince time.Time) ([]*OpenFlareAccessLogIPSummaryRow, error) {
|
||||
rows, err := buildOpenFlareAccessLogIPSummaryRows(ctx, query, recentSince)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
start, end := openFlareAccessLogPaginateBounds(len(rows), query.Page, query.PageSize)
|
||||
if start >= len(rows) {
|
||||
return []*OpenFlareAccessLogIPSummaryRow{}, nil
|
||||
}
|
||||
return rows[start:end], nil
|
||||
return buildOpenFlareAccessLogIPSummaryRows(ctx, query, recentSince)
|
||||
}
|
||||
|
||||
// CountOpenFlareAccessLogIPSummaries counts IP summaries.
|
||||
func CountOpenFlareAccessLogIPSummaries(ctx context.Context, query OpenFlareAccessLogIPSummaryQuery) (int64, error) {
|
||||
rows, err := buildOpenFlareAccessLogIPSummaryRows(ctx, query, time.Time{})
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return int64(len(rows)), nil
|
||||
filter := openFlareAccessLogQueryFromIPSummary(query)
|
||||
return currentAccessLogStore().CountIPSummaries(ctx, filter)
|
||||
}
|
||||
|
||||
// ListOpenFlareAccessLogIPTrend lists IP trend points.
|
||||
@@ -195,75 +215,22 @@ func buildOpenFlareAccessLogBucketRows(ctx context.Context, query OpenFlareAcces
|
||||
bucketSeconds = 180
|
||||
}
|
||||
|
||||
type bucketAccumulator struct {
|
||||
requestCount int64
|
||||
uniqueIPs map[string]struct{}
|
||||
uniqueHosts map[string]struct{}
|
||||
successCount int64
|
||||
clientErrorCount int64
|
||||
serverErrorCount int64
|
||||
}
|
||||
accumulators := make(map[int64]*bucketAccumulator)
|
||||
|
||||
partials, err := currentAccessLogStore().BucketAggregates(ctx, filter, bucketSeconds)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
rows := make([]*OpenFlareAccessLogBucketRow, 0, len(partials))
|
||||
for _, partial := range partials {
|
||||
accumulator := accumulators[partial.BucketEpoch]
|
||||
if accumulator == nil {
|
||||
accumulator = &bucketAccumulator{
|
||||
uniqueIPs: make(map[string]struct{}),
|
||||
uniqueHosts: make(map[string]struct{}),
|
||||
}
|
||||
accumulators[partial.BucketEpoch] = accumulator
|
||||
}
|
||||
accumulator.requestCount += partial.RequestCount
|
||||
accumulator.successCount += partial.SuccessCount
|
||||
accumulator.clientErrorCount += partial.ClientErrorCount
|
||||
accumulator.serverErrorCount += partial.ServerErrorCount
|
||||
}
|
||||
|
||||
for _, column := range []string{columnRemoteAddr, columnHost} {
|
||||
dimensions, err := currentAccessLogStore().BucketDimensions(ctx, filter, column, bucketSeconds)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
for _, item := range dimensions {
|
||||
accumulator := accumulators[item.BucketEpoch]
|
||||
if accumulator == nil {
|
||||
accumulator = &bucketAccumulator{
|
||||
uniqueIPs: make(map[string]struct{}),
|
||||
uniqueHosts: make(map[string]struct{}),
|
||||
}
|
||||
accumulators[item.BucketEpoch] = accumulator
|
||||
}
|
||||
trimmed := strings.TrimSpace(item.Value)
|
||||
if trimmed == "" {
|
||||
continue
|
||||
}
|
||||
switch column {
|
||||
case columnRemoteAddr:
|
||||
accumulator.uniqueIPs[trimmed] = struct{}{}
|
||||
case columnHost:
|
||||
accumulator.uniqueHosts[trimmed] = struct{}{}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
rows := make([]*OpenFlareAccessLogBucketRow, 0, len(accumulators))
|
||||
for bucketEpoch, accumulator := range accumulators {
|
||||
rows = append(rows, &OpenFlareAccessLogBucketRow{
|
||||
BucketEpoch: bucketEpoch,
|
||||
RequestCount: accumulator.requestCount,
|
||||
UniqueIPCount: int64(len(accumulator.uniqueIPs)),
|
||||
UniqueHostCount: int64(len(accumulator.uniqueHosts)),
|
||||
SuccessCount: accumulator.successCount,
|
||||
ClientErrorCount: accumulator.clientErrorCount,
|
||||
ServerErrorCount: accumulator.serverErrorCount,
|
||||
BucketEpoch: partial.BucketEpoch,
|
||||
RequestCount: partial.RequestCount,
|
||||
UniqueIPCount: partial.UniqueIPCount,
|
||||
UniqueHostCount: partial.UniqueHostCount,
|
||||
SuccessCount: partial.SuccessCount,
|
||||
ClientErrorCount: partial.ClientErrorCount,
|
||||
ServerErrorCount: partial.ServerErrorCount,
|
||||
})
|
||||
}
|
||||
sortOpenFlareAccessLogBucketRows(rows, query.SortBy, query.SortOrder)
|
||||
return rows, nil
|
||||
}
|
||||
|
||||
@@ -293,12 +260,7 @@ func buildOpenFlareAccessLogBucketIPRows(ctx context.Context, query OpenFlareAcc
|
||||
}
|
||||
|
||||
func buildOpenFlareAccessLogIPSummaryRows(ctx context.Context, query OpenFlareAccessLogIPSummaryQuery, recentSince time.Time) ([]*OpenFlareAccessLogIPSummaryRow, error) {
|
||||
filter := OpenFlareAccessLogQuery{
|
||||
NodeID: query.NodeID,
|
||||
RemoteAddr: query.RemoteAddr,
|
||||
Host: query.Host,
|
||||
Since: query.Since,
|
||||
}
|
||||
filter := openFlareAccessLogQueryFromIPSummary(query)
|
||||
partials, err := currentAccessLogStore().IPSummaries(ctx, filter, recentSince)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
@@ -316,7 +278,6 @@ func buildOpenFlareAccessLogIPSummaryRows(ctx context.Context, query OpenFlareAc
|
||||
LastSeenEpoch: partial.LastSeenEpoch,
|
||||
})
|
||||
}
|
||||
sortOpenFlareAccessLogIPSummaryRows(rows, query.SortBy, query.SortOrder)
|
||||
return rows, nil
|
||||
}
|
||||
|
||||
@@ -350,6 +311,23 @@ func openFlareAccessLogQueryFromBucket(query OpenFlareAccessLogBucketQuery) Open
|
||||
Host: query.Host,
|
||||
Path: query.Path,
|
||||
Since: query.Since,
|
||||
Page: query.Page,
|
||||
PageSize: query.PageSize,
|
||||
SortBy: query.SortBy,
|
||||
SortOrder: query.SortOrder,
|
||||
}
|
||||
}
|
||||
|
||||
func openFlareAccessLogQueryFromIPSummary(query OpenFlareAccessLogIPSummaryQuery) OpenFlareAccessLogQuery {
|
||||
return OpenFlareAccessLogQuery{
|
||||
NodeID: query.NodeID,
|
||||
RemoteAddr: query.RemoteAddr,
|
||||
Host: query.Host,
|
||||
Since: query.Since,
|
||||
Page: query.Page,
|
||||
PageSize: query.PageSize,
|
||||
SortBy: query.SortBy,
|
||||
SortOrder: query.SortOrder,
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -8,20 +8,46 @@ import (
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/Rain-kl/Wavelet/internal/apps/openflare/chwriter"
|
||||
analyticsmodel "github.com/Rain-kl/Wavelet/internal/model/analytics"
|
||||
analyticsrepo "github.com/Rain-kl/Wavelet/internal/repository/analytics"
|
||||
)
|
||||
|
||||
// AccessLogInsertHooks queues node access logs for async ClickHouse write.
|
||||
// Wired from openflare/chwriter.Init so model never imports the apps layer.
|
||||
type AccessLogInsertHooks struct {
|
||||
QueueNodeAccessLogs func(logs []analyticsmodel.NodeAccessLog)
|
||||
}
|
||||
|
||||
var (
|
||||
accessLogInsertHooksMu sync.RWMutex
|
||||
accessLogInsertHooks AccessLogInsertHooks
|
||||
)
|
||||
|
||||
// SetAccessLogInsertHooks registers async queue callbacks for access log inserts.
|
||||
func SetAccessLogInsertHooks(hooks AccessLogInsertHooks) {
|
||||
accessLogInsertHooksMu.Lock()
|
||||
accessLogInsertHooks = hooks
|
||||
accessLogInsertHooksMu.Unlock()
|
||||
}
|
||||
|
||||
func currentAccessLogInsertHooks() AccessLogInsertHooks {
|
||||
accessLogInsertHooksMu.RLock()
|
||||
defer accessLogInsertHooksMu.RUnlock()
|
||||
return accessLogInsertHooks
|
||||
}
|
||||
|
||||
type accessLogStore interface {
|
||||
InsertBatch(ctx context.Context, records []*OpenFlareAccessLog) error
|
||||
List(ctx context.Context, query OpenFlareAccessLogQuery) ([]*OpenFlareAccessLog, error)
|
||||
Count(ctx context.Context, query OpenFlareAccessLogQuery) (int64, int64, error)
|
||||
RegionCounts(ctx context.Context, nodeID string, since time.Time, limit int) ([]*OpenFlareAccessLogRegionCount, error)
|
||||
BucketAggregates(ctx context.Context, filter OpenFlareAccessLogQuery, bucketSeconds int64) ([]openFlareAccessLogBucketAggregateRow, error)
|
||||
CountBuckets(ctx context.Context, filter OpenFlareAccessLogQuery, bucketSeconds int64) (int64, error)
|
||||
BucketDimensions(ctx context.Context, filter OpenFlareAccessLogQuery, column string, bucketSeconds int64) ([]openFlareAccessLogBucketDimensionRow, error)
|
||||
IPAggregates(ctx context.Context, filter OpenFlareAccessLogQuery, exactRemoteAddr bool) ([]openFlareAccessLogIPAggregateRow, error)
|
||||
WAFIPAggregates(ctx context.Context, filter OpenFlareAccessLogQuery) ([]openFlareAccessLogWAFIPAggregateRow, error)
|
||||
IPSummaries(ctx context.Context, filter OpenFlareAccessLogQuery, recentSince time.Time) ([]openFlareAccessLogIPSummaryRow, error)
|
||||
CountIPSummaries(ctx context.Context, filter OpenFlareAccessLogQuery) (int64, error)
|
||||
IPTrend(ctx context.Context, filter OpenFlareAccessLogQuery, bucketSeconds int64) ([]openFlareAccessLogIPTrendRow, error)
|
||||
DeleteAll(ctx context.Context) (int64, error)
|
||||
DeleteBefore(ctx context.Context, cutoff time.Time) (int64, error)
|
||||
@@ -72,7 +98,9 @@ func (clickhouseAccessLogStore) InsertBatch(_ context.Context, records []*OpenFl
|
||||
}
|
||||
logs = append(logs, toAnalyticsNodeAccessLog(record))
|
||||
}
|
||||
chwriter.QueueNodeAccessLogs(logs)
|
||||
if hook := currentAccessLogInsertHooks().QueueNodeAccessLogs; hook != nil {
|
||||
hook(logs)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -116,11 +144,17 @@ func (clickhouseAccessLogStore) BucketAggregates(ctx context.Context, filter Ope
|
||||
SuccessCount: row.SuccessCount,
|
||||
ClientErrorCount: row.ClientErrorCount,
|
||||
ServerErrorCount: row.ServerErrorCount,
|
||||
UniqueIPCount: row.UniqueIPCount,
|
||||
UniqueHostCount: row.UniqueHostCount,
|
||||
}
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
func (clickhouseAccessLogStore) CountBuckets(ctx context.Context, filter OpenFlareAccessLogQuery, bucketSeconds int64) (int64, error) {
|
||||
return analyticsrepo.CountBucketAggregatesNodeAccessLogs(ctx, toNodeAccessLogFilter(filter), bucketSeconds)
|
||||
}
|
||||
|
||||
func (clickhouseAccessLogStore) BucketDimensions(ctx context.Context, filter OpenFlareAccessLogQuery, column string, bucketSeconds int64) ([]openFlareAccessLogBucketDimensionRow, error) {
|
||||
rows, err := analyticsrepo.BucketDimensionsNodeAccessLogs(ctx, toNodeAccessLogFilter(filter), column, bucketSeconds)
|
||||
if err != nil {
|
||||
@@ -172,6 +206,31 @@ func (clickhouseAccessLogStore) IPSummaries(ctx context.Context, filter OpenFlar
|
||||
return result, nil
|
||||
}
|
||||
|
||||
func (clickhouseAccessLogStore) CountIPSummaries(ctx context.Context, filter OpenFlareAccessLogQuery) (int64, error) {
|
||||
return analyticsrepo.CountIPSummaryNodeAccessLogs(ctx, toNodeAccessLogFilter(filter))
|
||||
}
|
||||
|
||||
func (clickhouseAccessLogStore) WAFIPAggregates(ctx context.Context, filter OpenFlareAccessLogQuery) ([]openFlareAccessLogWAFIPAggregateRow, error) {
|
||||
rows, err := analyticsrepo.IPAggregatesForWAFNodeAccessLogs(ctx, toNodeAccessLogFilter(filter))
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
result := make([]openFlareAccessLogWAFIPAggregateRow, len(rows))
|
||||
for index, row := range rows {
|
||||
result[index] = openFlareAccessLogWAFIPAggregateRow{
|
||||
RemoteAddr: row.RemoteAddr,
|
||||
RequestCount: row.RequestCount,
|
||||
Status404Count: row.Status404Count,
|
||||
ClientErrorCount: row.ClientErrorCount,
|
||||
ServerErrorCount: row.ServerErrorCount,
|
||||
IPHostCount: row.IPHostCount,
|
||||
LastSeenEpoch: row.LastSeenEpoch,
|
||||
StatusCounts: row.StatusCounts,
|
||||
}
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
func (clickhouseAccessLogStore) IPTrend(ctx context.Context, filter OpenFlareAccessLogQuery, bucketSeconds int64) ([]openFlareAccessLogIPTrendRow, error) {
|
||||
rows, err := analyticsrepo.IPTrendNodeAccessLogs(ctx, toNodeAccessLogFilter(filter), bucketSeconds)
|
||||
if err != nil {
|
||||
|
||||
@@ -5,6 +5,9 @@ package model
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net"
|
||||
"net/http"
|
||||
"net/netip"
|
||||
"sort"
|
||||
"strings"
|
||||
"sync"
|
||||
@@ -99,12 +102,21 @@ func (s *memoryAccessLogStore) BucketAggregates(_ context.Context, filter OpenFl
|
||||
s.mu.RLock()
|
||||
defer s.mu.RUnlock()
|
||||
rows := s.filterRecords(filter)
|
||||
aggregates := make(map[int64]*openFlareAccessLogBucketAggregateRow)
|
||||
type bucketAccumulator struct {
|
||||
openFlareAccessLogBucketAggregateRow
|
||||
uniqueIPs map[string]struct{}
|
||||
uniqueHosts map[string]struct{}
|
||||
}
|
||||
aggregates := make(map[int64]*bucketAccumulator)
|
||||
for _, row := range rows {
|
||||
bucketEpoch := memoryAccessLogBucketEpoch(row.LoggedAt, bucketSeconds)
|
||||
item := aggregates[bucketEpoch]
|
||||
if item == nil {
|
||||
item = &openFlareAccessLogBucketAggregateRow{BucketEpoch: bucketEpoch}
|
||||
item = &bucketAccumulator{
|
||||
openFlareAccessLogBucketAggregateRow: openFlareAccessLogBucketAggregateRow{BucketEpoch: bucketEpoch},
|
||||
uniqueIPs: make(map[string]struct{}),
|
||||
uniqueHosts: make(map[string]struct{}),
|
||||
}
|
||||
aggregates[bucketEpoch] = item
|
||||
}
|
||||
item.RequestCount++
|
||||
@@ -116,14 +128,61 @@ func (s *memoryAccessLogStore) BucketAggregates(_ context.Context, filter OpenFl
|
||||
default:
|
||||
item.ServerErrorCount++
|
||||
}
|
||||
if remoteAddr := strings.TrimSpace(row.RemoteAddr); remoteAddr != "" {
|
||||
item.uniqueIPs[remoteAddr] = struct{}{}
|
||||
}
|
||||
if host := strings.TrimSpace(row.Host); host != "" {
|
||||
item.uniqueHosts[host] = struct{}{}
|
||||
}
|
||||
}
|
||||
result := make([]openFlareAccessLogBucketAggregateRow, 0, len(aggregates))
|
||||
for _, item := range aggregates {
|
||||
result = append(result, *item)
|
||||
item.UniqueIPCount = int64(len(item.uniqueIPs))
|
||||
item.UniqueHostCount = int64(len(item.uniqueHosts))
|
||||
result = append(result, item.openFlareAccessLogBucketAggregateRow)
|
||||
}
|
||||
bucketRows := make([]*OpenFlareAccessLogBucketRow, len(result))
|
||||
for index := range result {
|
||||
bucketRows[index] = &OpenFlareAccessLogBucketRow{
|
||||
BucketEpoch: result[index].BucketEpoch,
|
||||
RequestCount: result[index].RequestCount,
|
||||
UniqueIPCount: result[index].UniqueIPCount,
|
||||
UniqueHostCount: result[index].UniqueHostCount,
|
||||
SuccessCount: result[index].SuccessCount,
|
||||
ClientErrorCount: result[index].ClientErrorCount,
|
||||
ServerErrorCount: result[index].ServerErrorCount,
|
||||
}
|
||||
}
|
||||
sortOpenFlareAccessLogBucketRows(bucketRows, filter.SortBy, filter.SortOrder)
|
||||
for index := range result {
|
||||
result[index] = openFlareAccessLogBucketAggregateRow{
|
||||
BucketEpoch: bucketRows[index].BucketEpoch,
|
||||
RequestCount: bucketRows[index].RequestCount,
|
||||
UniqueIPCount: bucketRows[index].UniqueIPCount,
|
||||
UniqueHostCount: bucketRows[index].UniqueHostCount,
|
||||
SuccessCount: bucketRows[index].SuccessCount,
|
||||
ClientErrorCount: bucketRows[index].ClientErrorCount,
|
||||
ServerErrorCount: bucketRows[index].ServerErrorCount,
|
||||
}
|
||||
}
|
||||
if filter.PageSize > 0 {
|
||||
start, end := openFlareAccessLogPaginateBounds(len(result), filter.Page, filter.PageSize)
|
||||
return result[start:end], nil
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
func (s *memoryAccessLogStore) CountBuckets(_ context.Context, filter OpenFlareAccessLogQuery, bucketSeconds int64) (int64, error) {
|
||||
s.mu.RLock()
|
||||
defer s.mu.RUnlock()
|
||||
rows := s.filterRecords(filter)
|
||||
seen := make(map[int64]struct{})
|
||||
for _, row := range rows {
|
||||
seen[memoryAccessLogBucketEpoch(row.LoggedAt, bucketSeconds)] = struct{}{}
|
||||
}
|
||||
return int64(len(seen)), nil
|
||||
}
|
||||
|
||||
func (s *memoryAccessLogStore) BucketDimensions(_ context.Context, filter OpenFlareAccessLogQuery, column string, bucketSeconds int64) ([]openFlareAccessLogBucketDimensionRow, error) {
|
||||
s.mu.RLock()
|
||||
defer s.mu.RUnlock()
|
||||
@@ -222,9 +281,91 @@ func (s *memoryAccessLogStore) IPSummaries(_ context.Context, filter OpenFlareAc
|
||||
item.LastSeenEpoch = epoch
|
||||
}
|
||||
}
|
||||
result := make([]openFlareAccessLogIPSummaryRow, 0, len(aggregates))
|
||||
summaryRows := make([]*OpenFlareAccessLogIPSummaryRow, 0, len(aggregates))
|
||||
for _, item := range aggregates {
|
||||
result = append(result, *item)
|
||||
summaryRows = append(summaryRows, &OpenFlareAccessLogIPSummaryRow{
|
||||
RemoteAddr: item.RemoteAddr,
|
||||
TotalRequests: item.TotalRequests,
|
||||
RecentRequests: item.RecentRequests,
|
||||
LastSeenEpoch: item.LastSeenEpoch,
|
||||
})
|
||||
}
|
||||
sortOpenFlareAccessLogIPSummaryRows(summaryRows, filter.SortBy, filter.SortOrder)
|
||||
if filter.PageSize > 0 {
|
||||
start, end := openFlareAccessLogPaginateBounds(len(summaryRows), filter.Page, filter.PageSize)
|
||||
summaryRows = summaryRows[start:end]
|
||||
}
|
||||
result := make([]openFlareAccessLogIPSummaryRow, len(summaryRows))
|
||||
for index, item := range summaryRows {
|
||||
result[index] = openFlareAccessLogIPSummaryRow{
|
||||
RemoteAddr: item.RemoteAddr,
|
||||
TotalRequests: item.TotalRequests,
|
||||
RecentRequests: item.RecentRequests,
|
||||
LastSeenEpoch: item.LastSeenEpoch,
|
||||
}
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
func (s *memoryAccessLogStore) CountIPSummaries(_ context.Context, filter OpenFlareAccessLogQuery) (int64, error) {
|
||||
s.mu.RLock()
|
||||
defer s.mu.RUnlock()
|
||||
rows := s.filterRecords(filter)
|
||||
seen := make(map[string]struct{})
|
||||
for _, row := range rows {
|
||||
remoteAddr := strings.TrimSpace(row.RemoteAddr)
|
||||
if remoteAddr == "" {
|
||||
continue
|
||||
}
|
||||
seen[remoteAddr] = struct{}{}
|
||||
}
|
||||
return int64(len(seen)), nil
|
||||
}
|
||||
|
||||
func (s *memoryAccessLogStore) WAFIPAggregates(_ context.Context, filter OpenFlareAccessLogQuery) ([]openFlareAccessLogWAFIPAggregateRow, error) {
|
||||
s.mu.RLock()
|
||||
defer s.mu.RUnlock()
|
||||
rows := s.filterRecords(filter)
|
||||
aggregates := make(map[string]*openFlareAccessLogWAFIPAggregateRow)
|
||||
order := make([]string, 0)
|
||||
for _, row := range rows {
|
||||
remoteAddr := strings.TrimSpace(row.RemoteAddr)
|
||||
if remoteAddr == "" {
|
||||
continue
|
||||
}
|
||||
item := aggregates[remoteAddr]
|
||||
if item == nil {
|
||||
item = &openFlareAccessLogWAFIPAggregateRow{
|
||||
RemoteAddr: remoteAddr,
|
||||
StatusCounts: make(map[int]int64),
|
||||
}
|
||||
aggregates[remoteAddr] = item
|
||||
order = append(order, remoteAddr)
|
||||
}
|
||||
item.RequestCount++
|
||||
item.StatusCounts[row.StatusCode]++
|
||||
if row.StatusCode == http.StatusNotFound {
|
||||
item.Status404Count++
|
||||
}
|
||||
if row.StatusCode >= 400 && row.StatusCode < 500 {
|
||||
item.ClientErrorCount++
|
||||
}
|
||||
if row.StatusCode >= http.StatusInternalServerError {
|
||||
item.ServerErrorCount++
|
||||
}
|
||||
if memoryAccessLogHostIsIPLiteral(row.Host) {
|
||||
item.IPHostCount++
|
||||
}
|
||||
epoch := row.LoggedAt.UTC().Unix()
|
||||
if epoch > item.LastSeenEpoch {
|
||||
item.LastSeenEpoch = epoch
|
||||
}
|
||||
}
|
||||
result := make([]openFlareAccessLogWAFIPAggregateRow, 0, len(order))
|
||||
for _, remoteAddr := range order {
|
||||
if item := aggregates[remoteAddr]; item != nil {
|
||||
result = append(result, *item)
|
||||
}
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
@@ -324,6 +465,19 @@ func memoryAccessLogMatches(row *OpenFlareAccessLog, query OpenFlareAccessLogQue
|
||||
return true
|
||||
}
|
||||
|
||||
func memoryAccessLogHostIsIPLiteral(value string) bool {
|
||||
host := strings.TrimSpace(value)
|
||||
if host == "" {
|
||||
return false
|
||||
}
|
||||
if parsedHost, _, err := net.SplitHostPort(host); err == nil {
|
||||
host = parsedHost
|
||||
}
|
||||
host = strings.Trim(host, "[]")
|
||||
_, err := netip.ParseAddr(host)
|
||||
return err == nil
|
||||
}
|
||||
|
||||
func memoryAccessLogBucketEpoch(loggedAt time.Time, bucketSeconds int64) int64 {
|
||||
if bucketSeconds <= 0 {
|
||||
bucketSeconds = 180
|
||||
|
||||
@@ -0,0 +1,75 @@
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
package model
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
analyticsmodel "github.com/Rain-kl/Wavelet/internal/model/analytics"
|
||||
)
|
||||
|
||||
// Hook setters are process-global; keep these tests serial.
|
||||
|
||||
func TestObservabilityInsertHooksAreInvoked(t *testing.T) {
|
||||
var gotSnapshot analyticsmodel.NodeMetricSnapshot
|
||||
SetObservabilityInsertHooks(ObservabilityInsertHooks{
|
||||
QueueMetricSnapshot: func(s analyticsmodel.NodeMetricSnapshot) {
|
||||
gotSnapshot = s
|
||||
},
|
||||
})
|
||||
t.Cleanup(func() {
|
||||
SetObservabilityInsertHooks(ObservabilityInsertHooks{})
|
||||
})
|
||||
|
||||
record := &OpenFlareMetricSnapshot{
|
||||
NodeID: "node-1",
|
||||
CapturedAt: time.Unix(100, 0).UTC(),
|
||||
}
|
||||
if err := (clickhouseObservabilityStore{}).InsertMetricSnapshot(context.Background(), record); err != nil {
|
||||
t.Fatalf("InsertMetricSnapshot error = %v", err)
|
||||
}
|
||||
if gotSnapshot.NodeID != "node-1" {
|
||||
t.Fatalf("hook node id = %q, want node-1", gotSnapshot.NodeID)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAccessLogInsertHooksAreInvoked(t *testing.T) {
|
||||
var got []analyticsmodel.NodeAccessLog
|
||||
SetAccessLogInsertHooks(AccessLogInsertHooks{
|
||||
QueueNodeAccessLogs: func(logs []analyticsmodel.NodeAccessLog) {
|
||||
got = append([]analyticsmodel.NodeAccessLog(nil), logs...)
|
||||
},
|
||||
})
|
||||
t.Cleanup(func() {
|
||||
SetAccessLogInsertHooks(AccessLogInsertHooks{})
|
||||
})
|
||||
|
||||
records := []*OpenFlareAccessLog{
|
||||
{NodeID: "n1", Path: "/a"},
|
||||
{NodeID: "n1", Path: "/b"},
|
||||
}
|
||||
if err := (clickhouseAccessLogStore{}).InsertBatch(context.Background(), records); err != nil {
|
||||
t.Fatalf("InsertBatch error = %v", err)
|
||||
}
|
||||
if len(got) != 2 {
|
||||
t.Fatalf("hook logs = %d, want 2", len(got))
|
||||
}
|
||||
if got[0].Path != "/a" || got[1].Path != "/b" {
|
||||
t.Fatalf("hook paths = %q/%q, want /a /b", got[0].Path, got[1].Path)
|
||||
}
|
||||
}
|
||||
|
||||
func TestInsertHooksNoopWhenUnset(t *testing.T) {
|
||||
SetObservabilityInsertHooks(ObservabilityInsertHooks{})
|
||||
SetAccessLogInsertHooks(AccessLogInsertHooks{})
|
||||
|
||||
if err := (clickhouseObservabilityStore{}).InsertMetricSnapshot(context.Background(), &OpenFlareMetricSnapshot{NodeID: "x"}); err != nil {
|
||||
t.Fatalf("InsertMetricSnapshot with nil hook error = %v", err)
|
||||
}
|
||||
if err := (clickhouseAccessLogStore{}).InsertBatch(context.Background(), []*OpenFlareAccessLog{{NodeID: "x"}}); err != nil {
|
||||
t.Fatalf("InsertBatch with nil hook error = %v", err)
|
||||
}
|
||||
}
|
||||
@@ -10,6 +10,7 @@ import (
|
||||
"time"
|
||||
|
||||
"github.com/Rain-kl/Wavelet/internal/db"
|
||||
analyticsrepo "github.com/Rain-kl/Wavelet/internal/repository/analytics"
|
||||
"gorm.io/gorm"
|
||||
)
|
||||
|
||||
@@ -277,6 +278,18 @@ type OpenFlareAccessLogIPTrendRow struct {
|
||||
RequestCount int64 `json:"request_count"`
|
||||
}
|
||||
|
||||
// OpenFlareAccessLogWAFIPAggregate is a per-IP aggregate row for WAF automatic rules.
|
||||
type OpenFlareAccessLogWAFIPAggregate struct {
|
||||
RemoteAddr string
|
||||
RequestCount int
|
||||
Status404Count int
|
||||
ClientErrorCount int
|
||||
ServerErrorCount int
|
||||
IPHostCount int
|
||||
LastSeenEpoch int64
|
||||
StatusCounts map[int]int
|
||||
}
|
||||
|
||||
func isMissingTableError(err error) bool {
|
||||
if err == nil {
|
||||
return false
|
||||
@@ -320,11 +333,179 @@ func ListOpenFlareMetricSnapshotsSince(ctx context.Context, nodeID string, since
|
||||
return currentObservabilityStore().ListMetricSnapshots(ctx, nodeID, since, limit)
|
||||
}
|
||||
|
||||
// ListOpenFlareLatestMetricSnapshotsSince returns the latest metric snapshot per node.
|
||||
// Prefer ClickHouse LIMIT 1 BY; on CH unavailability fall back to store list + reduce.
|
||||
func ListOpenFlareLatestMetricSnapshotsSince(ctx context.Context, nodeID string, since time.Time) ([]*OpenFlareMetricSnapshot, error) {
|
||||
rows, err := analyticsrepo.ListLatestNodeMetricSnapshots(ctx, analyticsrepo.NodeObservabilityFilter{
|
||||
NodeID: nodeID,
|
||||
Since: since,
|
||||
})
|
||||
if err == nil {
|
||||
return fromAnalyticsNodeMetricSnapshots(rows), nil
|
||||
}
|
||||
// Fallback for unit tests (memory store) and environments without ClickHouse.
|
||||
all, listErr := ListOpenFlareMetricSnapshotsSince(ctx, nodeID, since, 0)
|
||||
if listErr != nil {
|
||||
return nil, err
|
||||
}
|
||||
return openFlareLatestMetricSnapshots(all), nil
|
||||
}
|
||||
|
||||
// ListOpenFlareRequestReportsSince returns request reports since the given time.
|
||||
func ListOpenFlareRequestReportsSince(ctx context.Context, nodeID string, since time.Time, limit int) ([]*OpenFlareRequestReport, error) {
|
||||
return currentObservabilityStore().ListRequestReports(ctx, nodeID, since, limit)
|
||||
}
|
||||
|
||||
// ListOpenFlareLatestRequestReportsSince returns the latest request report per node.
|
||||
// Prefer ClickHouse LIMIT 1 BY; on CH unavailability fall back to store list + reduce.
|
||||
func ListOpenFlareLatestRequestReportsSince(ctx context.Context, nodeID string, since time.Time) ([]*OpenFlareRequestReport, error) {
|
||||
rows, err := analyticsrepo.ListLatestNodeRequestReports(ctx, analyticsrepo.NodeObservabilityFilter{
|
||||
NodeID: nodeID,
|
||||
Since: since,
|
||||
})
|
||||
if err == nil {
|
||||
return fromAnalyticsNodeRequestReports(rows), nil
|
||||
}
|
||||
all, listErr := ListOpenFlareRequestReportsSince(ctx, nodeID, since, 0)
|
||||
if listErr != nil {
|
||||
return nil, err
|
||||
}
|
||||
return openFlareLatestRequestReports(all), nil
|
||||
}
|
||||
|
||||
func openFlareLatestMetricSnapshots(snapshots []*OpenFlareMetricSnapshot) []*OpenFlareMetricSnapshot {
|
||||
latestByNode := make(map[string]*OpenFlareMetricSnapshot, len(snapshots))
|
||||
for _, snapshot := range snapshots {
|
||||
if snapshot == nil || snapshot.NodeID == "" {
|
||||
continue
|
||||
}
|
||||
if existing, ok := latestByNode[snapshot.NodeID]; ok && !snapshot.CapturedAt.After(existing.CapturedAt) {
|
||||
continue
|
||||
}
|
||||
latestByNode[snapshot.NodeID] = snapshot
|
||||
}
|
||||
result := make([]*OpenFlareMetricSnapshot, 0, len(latestByNode))
|
||||
for _, snapshot := range latestByNode {
|
||||
result = append(result, snapshot)
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
func openFlareLatestRequestReports(reports []*OpenFlareRequestReport) []*OpenFlareRequestReport {
|
||||
latestByNode := make(map[string]*OpenFlareRequestReport, len(reports))
|
||||
for _, report := range reports {
|
||||
if report == nil || report.NodeID == "" {
|
||||
continue
|
||||
}
|
||||
if existing, ok := latestByNode[report.NodeID]; ok && !report.WindowEndedAt.After(existing.WindowEndedAt) {
|
||||
continue
|
||||
}
|
||||
latestByNode[report.NodeID] = report
|
||||
}
|
||||
result := make([]*OpenFlareRequestReport, 0, len(latestByNode))
|
||||
for _, report := range latestByNode {
|
||||
result = append(result, report)
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
// OpenFlareTrafficHourly is an hourly traffic rollup row.
|
||||
type OpenFlareTrafficHourly struct {
|
||||
NodeID string `json:"node_id"`
|
||||
Hour time.Time `json:"hour"`
|
||||
RequestCount int64 `json:"request_count"`
|
||||
ErrorCount int64 `json:"error_count"`
|
||||
UniqueVisitorCount int64 `json:"unique_visitor_count"`
|
||||
}
|
||||
|
||||
// ListOpenFlareTrafficHourlySince returns hourly traffic rollup rows since the given time.
|
||||
func ListOpenFlareTrafficHourlySince(ctx context.Context, nodeID string, since time.Time) ([]*OpenFlareTrafficHourly, error) {
|
||||
rows, err := analyticsrepo.ListNodeTrafficHourly(ctx, analyticsrepo.NodeObservabilityFilter{
|
||||
NodeID: nodeID,
|
||||
Since: since,
|
||||
})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
result := make([]*OpenFlareTrafficHourly, len(rows))
|
||||
for index, row := range rows {
|
||||
result[index] = &OpenFlareTrafficHourly{
|
||||
NodeID: row.NodeID,
|
||||
Hour: row.Hour,
|
||||
RequestCount: row.RequestCount,
|
||||
ErrorCount: row.ErrorCount,
|
||||
UniqueVisitorCount: row.UniqueVisitorCount,
|
||||
}
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// OpenFlareMetricHourly is an hourly metric snapshot aggregation row.
|
||||
type OpenFlareMetricHourly struct {
|
||||
Hour time.Time `json:"hour"`
|
||||
AverageCPUUsagePercent float64 `json:"average_cpu_usage_percent"`
|
||||
AverageMemoryUsagePercent float64 `json:"average_memory_usage_percent"`
|
||||
NetworkRxBytes int64 `json:"network_rx_bytes"`
|
||||
NetworkTxBytes int64 `json:"network_tx_bytes"`
|
||||
DiskReadBytes int64 `json:"disk_read_bytes"`
|
||||
DiskWriteBytes int64 `json:"disk_write_bytes"`
|
||||
ReportedNodes int `json:"reported_nodes"`
|
||||
}
|
||||
|
||||
// OpenFlareOpenrestyHourly is an hourly OpenResty observation aggregation row.
|
||||
type OpenFlareOpenrestyHourly struct {
|
||||
Hour time.Time `json:"hour"`
|
||||
OpenrestyRxBytes int64 `json:"openresty_rx_bytes"`
|
||||
OpenrestyTxBytes int64 `json:"openresty_tx_bytes"`
|
||||
ReportedNodes int `json:"reported_nodes"`
|
||||
}
|
||||
|
||||
// ListOpenFlareMetricHourlySince returns hourly metric aggregates since the given time.
|
||||
func ListOpenFlareMetricHourlySince(ctx context.Context, nodeID string, since time.Time) ([]*OpenFlareMetricHourly, error) {
|
||||
rows, err := analyticsrepo.ListNodeMetricHourly(ctx, analyticsrepo.NodeObservabilityFilter{
|
||||
NodeID: nodeID,
|
||||
Since: since,
|
||||
})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
result := make([]*OpenFlareMetricHourly, len(rows))
|
||||
for index, row := range rows {
|
||||
result[index] = &OpenFlareMetricHourly{
|
||||
Hour: row.Hour,
|
||||
AverageCPUUsagePercent: row.AverageCPUUsagePercent,
|
||||
AverageMemoryUsagePercent: row.AverageMemoryUsagePercent,
|
||||
NetworkRxBytes: row.NetworkRxBytes,
|
||||
NetworkTxBytes: row.NetworkTxBytes,
|
||||
DiskReadBytes: row.DiskReadBytes,
|
||||
DiskWriteBytes: row.DiskWriteBytes,
|
||||
ReportedNodes: row.ReportedNodes,
|
||||
}
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// ListOpenFlareOpenrestyHourlySince returns hourly OpenResty aggregates since the given time.
|
||||
func ListOpenFlareOpenrestyHourlySince(ctx context.Context, nodeID string, since time.Time) ([]*OpenFlareOpenrestyHourly, error) {
|
||||
rows, err := analyticsrepo.ListNodeOpenrestyHourly(ctx, analyticsrepo.NodeObservabilityFilter{
|
||||
NodeID: nodeID,
|
||||
Since: since,
|
||||
})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
result := make([]*OpenFlareOpenrestyHourly, len(rows))
|
||||
for index, row := range rows {
|
||||
result[index] = &OpenFlareOpenrestyHourly{
|
||||
Hour: row.Hour,
|
||||
OpenrestyRxBytes: row.OpenrestyRxBytes,
|
||||
OpenrestyTxBytes: row.OpenrestyTxBytes,
|
||||
ReportedNodes: row.ReportedNodes,
|
||||
}
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// ListOpenFlareActiveHealthEvents returns active health events across all nodes.
|
||||
func ListOpenFlareActiveHealthEvents(ctx context.Context) ([]*OpenFlareHealthEvent, error) {
|
||||
conn := db.DB(ctx)
|
||||
@@ -384,6 +565,36 @@ func DeleteAllOpenFlareRequestReports(ctx context.Context) (int64, error) {
|
||||
return currentObservabilityStore().DeleteAllRequestReports(ctx)
|
||||
}
|
||||
|
||||
// DeleteOpenFlareNodeObservationOpenrestyBefore deletes OpenResty observations captured before cutoff.
|
||||
func DeleteOpenFlareNodeObservationOpenrestyBefore(ctx context.Context, cutoff time.Time) (int64, error) {
|
||||
return currentObservabilityStore().DeleteNodeObservationOpenrestyBefore(ctx, cutoff)
|
||||
}
|
||||
|
||||
// DeleteAllOpenFlareNodeObservationOpenresty deletes all OpenResty observations.
|
||||
func DeleteAllOpenFlareNodeObservationOpenresty(ctx context.Context) (int64, error) {
|
||||
return currentObservabilityStore().DeleteAllNodeObservationOpenresty(ctx)
|
||||
}
|
||||
|
||||
// DeleteOpenFlareNodeObservationFrpsBefore deletes FRPS observations captured before cutoff.
|
||||
func DeleteOpenFlareNodeObservationFrpsBefore(ctx context.Context, cutoff time.Time) (int64, error) {
|
||||
return currentObservabilityStore().DeleteNodeObservationFrpsBefore(ctx, cutoff)
|
||||
}
|
||||
|
||||
// DeleteAllOpenFlareNodeObservationFrps deletes all FRPS observations.
|
||||
func DeleteAllOpenFlareNodeObservationFrps(ctx context.Context) (int64, error) {
|
||||
return currentObservabilityStore().DeleteAllNodeObservationFrps(ctx)
|
||||
}
|
||||
|
||||
// DeleteOpenFlareNodeObservationFrpcBefore deletes FRPC observations captured before cutoff.
|
||||
func DeleteOpenFlareNodeObservationFrpcBefore(ctx context.Context, cutoff time.Time) (int64, error) {
|
||||
return currentObservabilityStore().DeleteNodeObservationFrpcBefore(ctx, cutoff)
|
||||
}
|
||||
|
||||
// DeleteAllOpenFlareNodeObservationFrpc deletes all FRPC observations.
|
||||
func DeleteAllOpenFlareNodeObservationFrpc(ctx context.Context) (int64, error) {
|
||||
return currentObservabilityStore().DeleteAllNodeObservationFrpc(ctx)
|
||||
}
|
||||
|
||||
// DeleteOpenFlareHealthEventsByNodeID deletes all health events for a node.
|
||||
func DeleteOpenFlareHealthEventsByNodeID(ctx context.Context, nodeID string) (int64, error) {
|
||||
conn := db.DB(ctx)
|
||||
|
||||
@@ -9,11 +9,38 @@ import (
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/Rain-kl/Wavelet/internal/apps/openflare/chwriter"
|
||||
analyticsmodel "github.com/Rain-kl/Wavelet/internal/model/analytics"
|
||||
analyticsrepo "github.com/Rain-kl/Wavelet/internal/repository/analytics"
|
||||
)
|
||||
|
||||
// ObservabilityInsertHooks queues observability rows for async ClickHouse write.
|
||||
// Wired from openflare/chwriter.Init so model never imports the apps layer.
|
||||
type ObservabilityInsertHooks struct {
|
||||
QueueMetricSnapshot func(analyticsmodel.NodeMetricSnapshot)
|
||||
QueueRequestReport func(analyticsmodel.NodeRequestReport)
|
||||
QueueOpenrestyObservation func(analyticsmodel.NodeObsOpenresty)
|
||||
QueueFrpsObservation func(analyticsmodel.NodeObsFrps)
|
||||
QueueFrpcObservation func(analyticsmodel.NodeObsFrpc)
|
||||
}
|
||||
|
||||
var (
|
||||
observabilityInsertHooksMu sync.RWMutex
|
||||
observabilityInsertHooks ObservabilityInsertHooks
|
||||
)
|
||||
|
||||
// SetObservabilityInsertHooks registers async queue callbacks for observability inserts.
|
||||
func SetObservabilityInsertHooks(hooks ObservabilityInsertHooks) {
|
||||
observabilityInsertHooksMu.Lock()
|
||||
observabilityInsertHooks = hooks
|
||||
observabilityInsertHooksMu.Unlock()
|
||||
}
|
||||
|
||||
func currentObservabilityInsertHooks() ObservabilityInsertHooks {
|
||||
observabilityInsertHooksMu.RLock()
|
||||
defer observabilityInsertHooksMu.RUnlock()
|
||||
return observabilityInsertHooks
|
||||
}
|
||||
|
||||
type observabilityStore interface {
|
||||
InsertMetricSnapshot(ctx context.Context, record *OpenFlareMetricSnapshot) error
|
||||
ListMetricSnapshots(ctx context.Context, nodeID string, since time.Time, limit int) ([]*OpenFlareMetricSnapshot, error)
|
||||
@@ -79,7 +106,9 @@ func (clickhouseObservabilityStore) InsertMetricSnapshot(_ context.Context, reco
|
||||
if record == nil {
|
||||
return nil
|
||||
}
|
||||
chwriter.QueueMetricSnapshot(toAnalyticsNodeMetricSnapshot(record))
|
||||
if hook := currentObservabilityInsertHooks().QueueMetricSnapshot; hook != nil {
|
||||
hook(toAnalyticsNodeMetricSnapshot(record))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -103,7 +132,9 @@ func (clickhouseObservabilityStore) InsertRequestReport(_ context.Context, recor
|
||||
if record == nil {
|
||||
return nil
|
||||
}
|
||||
chwriter.QueueRequestReport(toAnalyticsNodeRequestReport(record))
|
||||
if hook := currentObservabilityInsertHooks().QueueRequestReport; hook != nil {
|
||||
hook(toAnalyticsNodeRequestReport(record))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -127,7 +158,9 @@ func (clickhouseObservabilityStore) InsertNodeObservationOpenresty(_ context.Con
|
||||
if record == nil {
|
||||
return nil
|
||||
}
|
||||
chwriter.QueueOpenrestyObservation(toAnalyticsNodeObsOpenresty(record))
|
||||
if hook := currentObservabilityInsertHooks().QueueOpenrestyObservation; hook != nil {
|
||||
hook(toAnalyticsNodeObsOpenresty(record))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -151,7 +184,9 @@ func (clickhouseObservabilityStore) InsertNodeObservationFrps(_ context.Context,
|
||||
if record == nil {
|
||||
return nil
|
||||
}
|
||||
chwriter.QueueFrpsObservation(toAnalyticsNodeObsFrps(record))
|
||||
if hook := currentObservabilityInsertHooks().QueueFrpsObservation; hook != nil {
|
||||
hook(toAnalyticsNodeObsFrps(record))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -175,7 +210,9 @@ func (clickhouseObservabilityStore) InsertNodeObservationFrpc(_ context.Context,
|
||||
if record == nil {
|
||||
return nil
|
||||
}
|
||||
chwriter.QueueFrpcObservation(toAnalyticsNodeObsFrpc(record))
|
||||
if hook := currentObservabilityInsertHooks().QueueFrpcObservation; hook != nil {
|
||||
hook(toAnalyticsNodeObsFrpc(record))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
|
||||
@@ -7,41 +7,51 @@ package analytics
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"time"
|
||||
|
||||
"github.com/Rain-kl/Wavelet/internal/db"
|
||||
analyticsmodel "github.com/Rain-kl/Wavelet/internal/model/analytics"
|
||||
"gorm.io/gorm"
|
||||
)
|
||||
|
||||
func userAccessLogConn() error {
|
||||
if db.ChConn == nil {
|
||||
return fmt.Errorf("clickhouse native connection is not initialized")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// CountAccessLogs returns the number of access logs matching filter.
|
||||
func CountAccessLogs(ctx context.Context, filter AccessLogFilter) (uint64, error) {
|
||||
ch := db.ChDB(ctx)
|
||||
if ch == nil {
|
||||
return 0, fmt.Errorf("clickhouse gorm connection is not initialized")
|
||||
clause, args, ok := buildUserAccessLogFilterClause(filter)
|
||||
if !ok {
|
||||
return 0, nil
|
||||
}
|
||||
|
||||
var count int64
|
||||
query := applyFilter(ch.Model(&analyticsmodel.UserAccessLog{}), filter)
|
||||
if err := query.Count(&count).Error; err != nil {
|
||||
if err := userAccessLogConn(); err != nil {
|
||||
return 0, err
|
||||
}
|
||||
tableName := analyticsmodel.UserAccessLog{}.TableName()
|
||||
sql := fmt.Sprintf("SELECT count() FROM %s WHERE %s", tableName, clause)
|
||||
var count uint64
|
||||
if err := db.ChConn.QueryRow(ctx, sql, args...).Scan(&count); err != nil {
|
||||
return 0, fmt.Errorf("count access logs: %w", err)
|
||||
}
|
||||
return safeUint64Count(count), nil
|
||||
return count, nil
|
||||
}
|
||||
|
||||
// ListAccessLogs returns paginated access logs and the total match count.
|
||||
func ListAccessLogs(ctx context.Context, filter AccessLogFilter, page, pageSize int) ([]analyticsmodel.UserAccessLog, uint64, error) {
|
||||
ch := db.ChDB(ctx)
|
||||
if ch == nil {
|
||||
return nil, 0, fmt.Errorf("clickhouse gorm connection is not initialized")
|
||||
}
|
||||
|
||||
if filter.UserIDs != nil && len(filter.UserIDs) == 0 {
|
||||
clause, args, ok := buildUserAccessLogFilterClause(filter)
|
||||
if !ok {
|
||||
return []analyticsmodel.UserAccessLog{}, 0, nil
|
||||
}
|
||||
if err := userAccessLogConn(); err != nil {
|
||||
return nil, 0, err
|
||||
}
|
||||
|
||||
var total int64
|
||||
baseQuery := applyFilter(ch.Model(&analyticsmodel.UserAccessLog{}), filter)
|
||||
if err := baseQuery.Count(&total).Error; err != nil {
|
||||
tableName := analyticsmodel.UserAccessLog{}.TableName()
|
||||
countSQL := fmt.Sprintf("SELECT count() FROM %s WHERE %s", tableName, clause)
|
||||
var total uint64
|
||||
if err := db.ChConn.QueryRow(ctx, countSQL, args...).Scan(&total); err != nil {
|
||||
return nil, 0, fmt.Errorf("count access logs: %w", err)
|
||||
}
|
||||
if total == 0 {
|
||||
@@ -56,34 +66,41 @@ func ListAccessLogs(ctx context.Context, filter AccessLogFilter, page, pageSize
|
||||
}
|
||||
offset := (page - 1) * pageSize
|
||||
|
||||
var logs []analyticsmodel.UserAccessLog
|
||||
err := applyFilter(ch.Model(&analyticsmodel.UserAccessLog{}), filter).
|
||||
Order("created_at DESC, id DESC").
|
||||
Limit(pageSize).
|
||||
Offset(offset).
|
||||
Find(&logs).Error
|
||||
listSQL := fmt.Sprintf(`
|
||||
SELECT id, user_id, path, method, ip, user_agent, headers, status, latency, created_at
|
||||
FROM %s
|
||||
WHERE %s
|
||||
ORDER BY created_at DESC, id DESC
|
||||
LIMIT ? OFFSET ?`, tableName, clause)
|
||||
listArgs := append(append([]any{}, args...), pageSize, offset)
|
||||
rows, err := db.ChConn.Query(ctx, listSQL, listArgs...)
|
||||
if err != nil {
|
||||
return nil, 0, fmt.Errorf("list access logs: %w", err)
|
||||
}
|
||||
defer func() { _ = rows.Close() }()
|
||||
|
||||
return logs, safeUint64Count(total), nil
|
||||
}
|
||||
|
||||
func applyFilter(query *gorm.DB, filter AccessLogFilter) *gorm.DB {
|
||||
if filter.UserIDs != nil {
|
||||
if len(filter.UserIDs) == 0 {
|
||||
return query.Where("1 = 0")
|
||||
logs := make([]analyticsmodel.UserAccessLog, 0, pageSize)
|
||||
for rows.Next() {
|
||||
var (
|
||||
item analyticsmodel.UserAccessLog
|
||||
createdAt time.Time
|
||||
)
|
||||
if err := rows.Scan(
|
||||
&item.ID,
|
||||
&item.UserID,
|
||||
&item.Path,
|
||||
&item.Method,
|
||||
&item.IP,
|
||||
&item.UserAgent,
|
||||
&item.Headers,
|
||||
&item.Status,
|
||||
&item.Latency,
|
||||
&createdAt,
|
||||
); err != nil {
|
||||
return nil, 0, fmt.Errorf("scan access log row: %w", err)
|
||||
}
|
||||
query = query.Where("user_id IN ?", filter.UserIDs)
|
||||
item.CreatedAt = createdAt
|
||||
logs = append(logs, item)
|
||||
}
|
||||
if filter.Path != "" {
|
||||
query = query.Where("path LIKE ?", "%"+filter.Path+"%")
|
||||
}
|
||||
if filter.StartTime != nil {
|
||||
query = query.Where("created_at >= ?", *filter.StartTime)
|
||||
}
|
||||
if filter.EndTime != nil {
|
||||
query = query.Where("created_at <= ?", *filter.EndTime)
|
||||
}
|
||||
return query
|
||||
return logs, total, nil
|
||||
}
|
||||
@@ -3,7 +3,13 @@
|
||||
|
||||
package analytics
|
||||
|
||||
import "time"
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
const userAccessLogFilterClauseCapacity = 4
|
||||
|
||||
// AccessLogFilter scopes ClickHouse user access log queries.
|
||||
type AccessLogFilter struct {
|
||||
@@ -14,4 +20,37 @@ type AccessLogFilter struct {
|
||||
StartTime *time.Time
|
||||
// EndTime filters created_at <= EndTime when non-nil.
|
||||
EndTime *time.Time
|
||||
}
|
||||
|
||||
func buildUserAccessLogFilterClause(filter AccessLogFilter) (string, []any, bool) {
|
||||
if filter.UserIDs != nil && len(filter.UserIDs) == 0 {
|
||||
return "", nil, false
|
||||
}
|
||||
|
||||
parts := make([]string, 0, userAccessLogFilterClauseCapacity)
|
||||
args := make([]any, 0, userAccessLogFilterClauseCapacity)
|
||||
if filter.UserIDs != nil {
|
||||
placeholders := make([]string, len(filter.UserIDs))
|
||||
for index, userID := range filter.UserIDs {
|
||||
placeholders[index] = "?"
|
||||
args = append(args, userID)
|
||||
}
|
||||
parts = append(parts, fmt.Sprintf("user_id IN (%s)", strings.Join(placeholders, ", ")))
|
||||
}
|
||||
if trimmed := strings.TrimSpace(filter.Path); trimmed != "" {
|
||||
parts = append(parts, "path LIKE ?")
|
||||
args = append(args, "%"+trimmed+"%")
|
||||
}
|
||||
if filter.StartTime != nil {
|
||||
parts = append(parts, "created_at >= ?")
|
||||
args = append(args, *filter.StartTime)
|
||||
}
|
||||
if filter.EndTime != nil {
|
||||
parts = append(parts, "created_at <= ?")
|
||||
args = append(args, *filter.EndTime)
|
||||
}
|
||||
if len(parts) == 0 {
|
||||
return "1", args, true
|
||||
}
|
||||
return strings.Join(parts, " AND "), args, true
|
||||
}
|
||||
@@ -38,15 +38,12 @@ func GetDailyTrend(ctx context.Context, days int) ([]DailyTrend, error) {
|
||||
if days < 1 {
|
||||
days = 7
|
||||
}
|
||||
|
||||
ch := db.ChDB(ctx)
|
||||
if ch == nil {
|
||||
return nil, fmt.Errorf("clickhouse gorm connection is not initialized")
|
||||
if err := userAccessLogConn(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
startTime := time.Now().AddDate(0, 0, -(days - 1)).Truncate(hoursInDay * time.Hour)
|
||||
tableName := analyticsmodel.UserAccessLog{}.TableName()
|
||||
|
||||
query := fmt.Sprintf(`
|
||||
SELECT toDate(created_at) AS date, count() AS count
|
||||
FROM %s
|
||||
@@ -55,24 +52,26 @@ func GetDailyTrend(ctx context.Context, days int) ([]DailyTrend, error) {
|
||||
ORDER BY date ASC
|
||||
`, tableName)
|
||||
|
||||
type trendRow struct {
|
||||
Date time.Time
|
||||
Count uint64
|
||||
}
|
||||
|
||||
var rows []trendRow
|
||||
if err := ch.Raw(query, startTime).Scan(&rows).Error; err != nil {
|
||||
rows, err := db.ChConn.Query(ctx, query, startTime)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("get daily trend: %w", err)
|
||||
}
|
||||
defer func() { _ = rows.Close() }()
|
||||
|
||||
trendMap := make(map[string]uint64, days)
|
||||
for i := 0; i < days; i++ {
|
||||
dateStr := time.Now().AddDate(0, 0, -i).Format("2006-01-02")
|
||||
trendMap[dateStr] = 0
|
||||
}
|
||||
for _, row := range rows {
|
||||
dateStr := row.Date.Format("2006-01-02")
|
||||
trendMap[dateStr] = row.Count
|
||||
for rows.Next() {
|
||||
var (
|
||||
date time.Time
|
||||
count uint64
|
||||
)
|
||||
if err := rows.Scan(&date, &count); err != nil {
|
||||
return nil, fmt.Errorf("scan daily trend row: %w", err)
|
||||
}
|
||||
trendMap[date.Format("2006-01-02")] = count
|
||||
}
|
||||
|
||||
result := make([]DailyTrend, 0, days)
|
||||
@@ -88,9 +87,8 @@ func GetDailyTrend(ctx context.Context, days int) ([]DailyTrend, error) {
|
||||
|
||||
// GetBrowserDistribution returns browser-grouped access counts since startTime.
|
||||
func GetBrowserDistribution(ctx context.Context, startTime time.Time) ([]BrowserShare, error) {
|
||||
ch := db.ChDB(ctx)
|
||||
if ch == nil {
|
||||
return nil, fmt.Errorf("clickhouse gorm connection is not initialized")
|
||||
if err := userAccessLogConn(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
tableName := analyticsmodel.UserAccessLog{}.TableName()
|
||||
@@ -99,22 +97,27 @@ func GetBrowserDistribution(ctx context.Context, startTime time.Time) ([]Browser
|
||||
FROM %s
|
||||
WHERE created_at >= ?
|
||||
GROUP BY user_agent
|
||||
ORDER BY count DESC
|
||||
LIMIT 100
|
||||
`, tableName)
|
||||
|
||||
type uaRow struct {
|
||||
UserAgent string
|
||||
Count uint64
|
||||
}
|
||||
|
||||
var rows []uaRow
|
||||
if err := ch.Raw(query, startTime).Scan(&rows).Error; err != nil {
|
||||
rows, err := db.ChConn.Query(ctx, query, startTime)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("get browser distribution: %w", err)
|
||||
}
|
||||
defer func() { _ = rows.Close() }()
|
||||
|
||||
browserCounts := make(map[string]uint64)
|
||||
for _, row := range rows {
|
||||
browser := ParseBrowserName(row.UserAgent)
|
||||
browserCounts[browser] += row.Count
|
||||
for rows.Next() {
|
||||
var (
|
||||
userAgent string
|
||||
count uint64
|
||||
)
|
||||
if err := rows.Scan(&userAgent, &count); err != nil {
|
||||
return nil, fmt.Errorf("scan browser distribution row: %w", err)
|
||||
}
|
||||
browser := ParseBrowserName(userAgent)
|
||||
browserCounts[browser] += count
|
||||
}
|
||||
|
||||
result := make([]BrowserShare, 0, len(browserCounts))
|
||||
@@ -135,10 +138,8 @@ func GetTopActiveUsers(ctx context.Context, startTime time.Time, limit int) ([]T
|
||||
if limit < 1 {
|
||||
limit = 10
|
||||
}
|
||||
|
||||
ch := db.ChDB(ctx)
|
||||
if ch == nil {
|
||||
return nil, fmt.Errorf("clickhouse gorm connection is not initialized")
|
||||
if err := userAccessLogConn(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
tableName := analyticsmodel.UserAccessLog{}.TableName()
|
||||
@@ -151,9 +152,19 @@ func GetTopActiveUsers(ctx context.Context, startTime time.Time, limit int) ([]T
|
||||
LIMIT ?
|
||||
`, tableName)
|
||||
|
||||
var users []TopUser
|
||||
if err := ch.Raw(query, startTime, limit).Scan(&users).Error; err != nil {
|
||||
rows, err := db.ChConn.Query(ctx, query, startTime, limit)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("get top active users: %w", err)
|
||||
}
|
||||
defer func() { _ = rows.Close() }()
|
||||
|
||||
var users []TopUser
|
||||
for rows.Next() {
|
||||
var item TopUser
|
||||
if err := rows.Scan(&item.UserID, &item.Count); err != nil {
|
||||
return nil, fmt.Errorf("scan top active user row: %w", err)
|
||||
}
|
||||
users = append(users, item)
|
||||
}
|
||||
return users, nil
|
||||
}
|
||||
@@ -12,24 +12,10 @@ import (
|
||||
"github.com/ClickHouse/clickhouse-go/v2/lib/driver"
|
||||
"github.com/Rain-kl/Wavelet/internal/db"
|
||||
analyticsmodel "github.com/Rain-kl/Wavelet/internal/model/analytics"
|
||||
"github.com/glebarez/sqlite"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
"gorm.io/gorm"
|
||||
)
|
||||
|
||||
func setupChGormDB(t *testing.T) *gorm.DB {
|
||||
t.Helper()
|
||||
|
||||
gormDB, err := gorm.Open(sqlite.Open(":memory:"), &gorm.Config{
|
||||
DisableForeignKeyConstraintWhenMigrating: true,
|
||||
})
|
||||
require.NoError(t, err)
|
||||
require.NoError(t, gormDB.AutoMigrate(&analyticsmodel.UserAccessLog{}))
|
||||
db.SetChDBForTest(gormDB)
|
||||
return gormDB
|
||||
}
|
||||
|
||||
func TestParseBrowserName(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
@@ -52,56 +38,24 @@ func TestParseBrowserName(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestCountAccessLogs_EmptyUserIDs(t *testing.T) {
|
||||
setupChGormDB(t)
|
||||
t.Cleanup(func() { db.SetChDBForTest(nil) })
|
||||
func TestBuildUserAccessLogFilterClause_EmptyUserIDs(t *testing.T) {
|
||||
_, _, ok := buildUserAccessLogFilterClause(AccessLogFilter{UserIDs: []uint64{}})
|
||||
assert.False(t, ok)
|
||||
}
|
||||
|
||||
func TestCountAccessLogs_EmptyUserIDs(t *testing.T) {
|
||||
count, err := CountAccessLogs(context.Background(), AccessLogFilter{UserIDs: []uint64{}})
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, uint64(0), count)
|
||||
}
|
||||
|
||||
func TestListAccessLogs_EmptyUserIDs(t *testing.T) {
|
||||
setupChGormDB(t)
|
||||
t.Cleanup(func() { db.SetChDBForTest(nil) })
|
||||
|
||||
logs, total, err := ListAccessLogs(context.Background(), AccessLogFilter{UserIDs: []uint64{}}, 1, 20)
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, uint64(0), total)
|
||||
assert.Empty(t, logs)
|
||||
}
|
||||
|
||||
func TestListAccessLogs_WithFilters(t *testing.T) {
|
||||
gormDB := setupChGormDB(t)
|
||||
t.Cleanup(func() { db.SetChDBForTest(nil) })
|
||||
|
||||
now := time.Now().UTC().Truncate(time.Second)
|
||||
logs := []analyticsmodel.UserAccessLog{
|
||||
{ID: 1, UserID: 10, Path: "/api/v1/users", Method: "GET", Status: 200, CreatedAt: now},
|
||||
{ID: 2, UserID: 20, Path: "/api/v1/admin/logs", Method: "GET", Status: 200, CreatedAt: now},
|
||||
{ID: 3, UserID: 10, Path: "/api/v1/other", Method: "POST", Status: 201, CreatedAt: now},
|
||||
}
|
||||
require.NoError(t, gormDB.Create(&logs).Error)
|
||||
|
||||
start := now.Add(-time.Hour)
|
||||
filter := AccessLogFilter{
|
||||
UserIDs: []uint64{10},
|
||||
Path: "users",
|
||||
StartTime: &start,
|
||||
}
|
||||
|
||||
count, err := CountAccessLogs(context.Background(), filter)
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, uint64(1), count)
|
||||
|
||||
result, total, err := ListAccessLogs(context.Background(), filter, 1, 10)
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, uint64(1), total)
|
||||
require.Len(t, result, 1)
|
||||
assert.Equal(t, uint64(1), result[0].ID)
|
||||
assert.Equal(t, "/api/v1/users", result[0].Path)
|
||||
}
|
||||
|
||||
func TestBatchInsert_Empty(t *testing.T) {
|
||||
err := BatchInsert(context.Background(), nil)
|
||||
require.NoError(t, err)
|
||||
@@ -145,6 +99,9 @@ type mockConn struct {
|
||||
batchQuery string
|
||||
prepareCalled bool
|
||||
preparedQuery string
|
||||
queries []string
|
||||
queryArgs [][]any
|
||||
queryFn func(ctx context.Context, query string, args ...any) (driver.Rows, error)
|
||||
}
|
||||
|
||||
func (m *mockConn) Contributors() []string { return nil }
|
||||
@@ -153,8 +110,13 @@ func (m *mockConn) ServerVersion() (*driver.ServerVersion, error) { return nil,
|
||||
|
||||
func (m *mockConn) Select(_ context.Context, _ any, _ string, _ ...any) error { return nil }
|
||||
|
||||
func (m *mockConn) Query(_ context.Context, _ string, _ ...any) (driver.Rows, error) {
|
||||
return nil, nil
|
||||
func (m *mockConn) Query(ctx context.Context, query string, args ...any) (driver.Rows, error) {
|
||||
m.queries = append(m.queries, query)
|
||||
m.queryArgs = append(m.queryArgs, args)
|
||||
if m.queryFn != nil {
|
||||
return m.queryFn(ctx, query, args...)
|
||||
}
|
||||
return &mockRows{}, nil
|
||||
}
|
||||
|
||||
func (m *mockConn) QueryRow(_ context.Context, _ string, _ ...any) driver.Row { return nil }
|
||||
@@ -204,4 +166,96 @@ func (m *mockBatch) Rows() int { return len(m.rows) }
|
||||
|
||||
func (m *mockBatch) Columns() []column.Interface { return nil }
|
||||
|
||||
func (m *mockBatch) Close() error { return nil }
|
||||
func (m *mockBatch) Close() error { return nil }
|
||||
|
||||
// mockRows is an empty driver.Rows implementation for query-path unit tests.
|
||||
type mockRows struct {
|
||||
index int
|
||||
data [][]any
|
||||
err error
|
||||
}
|
||||
|
||||
func (m *mockRows) Next() bool {
|
||||
if m.err != nil {
|
||||
return false
|
||||
}
|
||||
if m.index >= len(m.data) {
|
||||
return false
|
||||
}
|
||||
m.index++
|
||||
return true
|
||||
}
|
||||
|
||||
func (m *mockRows) Scan(dest ...any) error {
|
||||
if m.err != nil {
|
||||
return m.err
|
||||
}
|
||||
if m.index == 0 || m.index > len(m.data) {
|
||||
return nil
|
||||
}
|
||||
row := m.data[m.index-1]
|
||||
for i := range dest {
|
||||
if i >= len(row) {
|
||||
break
|
||||
}
|
||||
if err := assignMockScanValue(dest[i], row[i]); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (m *mockRows) ScanStruct(_ any) error { return nil }
|
||||
|
||||
func (m *mockRows) ColumnTypes() []driver.ColumnType { return nil }
|
||||
|
||||
func (m *mockRows) Totals(_ ...any) error { return nil }
|
||||
|
||||
func (m *mockRows) Columns() []string { return nil }
|
||||
|
||||
func (m *mockRows) Close() error { return nil }
|
||||
|
||||
func (m *mockRows) Err() error { return m.err }
|
||||
|
||||
func (m *mockRows) HasData() bool { return len(m.data) > 0 }
|
||||
|
||||
func assignMockScanValue(dest any, value any) error {
|
||||
switch d := dest.(type) {
|
||||
case *string:
|
||||
if v, ok := value.(string); ok {
|
||||
*d = v
|
||||
}
|
||||
case *uint64:
|
||||
switch v := value.(type) {
|
||||
case uint64:
|
||||
*d = v
|
||||
case int:
|
||||
*d = uint64(v)
|
||||
case int64:
|
||||
*d = uint64(v)
|
||||
}
|
||||
case *int64:
|
||||
switch v := value.(type) {
|
||||
case int64:
|
||||
*d = v
|
||||
case int:
|
||||
*d = int64(v)
|
||||
case uint64:
|
||||
*d = int64(v)
|
||||
}
|
||||
case *float64:
|
||||
switch v := value.(type) {
|
||||
case float64:
|
||||
*d = v
|
||||
case float32:
|
||||
*d = float64(v)
|
||||
case int:
|
||||
*d = float64(v)
|
||||
}
|
||||
case *time.Time:
|
||||
if v, ok := value.(time.Time); ok {
|
||||
*d = v
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -5,13 +5,6 @@ package analytics
|
||||
|
||||
import "math"
|
||||
|
||||
func safeUint64Count(count int64) uint64 {
|
||||
if count < 0 {
|
||||
return 0
|
||||
}
|
||||
return uint64(count)
|
||||
}
|
||||
|
||||
func safeInt64Count(count uint64) int64 {
|
||||
if count > math.MaxInt64 {
|
||||
return math.MaxInt64
|
||||
|
||||
@@ -31,24 +31,3 @@ func TestSafeInt64Count(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestSafeUint64Count(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
count int64
|
||||
want uint64
|
||||
}{
|
||||
{name: "zero", count: 0, want: 0},
|
||||
{name: "positive", count: 42, want: 42},
|
||||
{name: "negative clamps", count: -1, want: 0},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
if got := safeUint64Count(tt.count); got != tt.want {
|
||||
t.Fatalf("safeUint64Count(%d) = %d, want %d", tt.count, got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,122 @@
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
package analytics
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"time"
|
||||
|
||||
"github.com/ClickHouse/clickhouse-go/v2/lib/driver"
|
||||
)
|
||||
|
||||
// DDL TTL days for analytics tables (must match goose ClickHouse migrations).
|
||||
const (
|
||||
// TableTTLDaysNodeAccessLogs is the of_node_access_logs TTL (90 days).
|
||||
TableTTLDaysNodeAccessLogs = 90
|
||||
// TableTTLDaysNodeMetricSnapshots is the of_node_metric_snapshots TTL (30 days).
|
||||
TableTTLDaysNodeMetricSnapshots = 30
|
||||
// TableTTLDaysNodeRequestReports is the of_node_request_reports TTL (30 days).
|
||||
TableTTLDaysNodeRequestReports = 30
|
||||
// TableTTLDaysNodeObs is the of_node_obs_* TTL (30 days).
|
||||
TableTTLDaysNodeObs = 30
|
||||
// TableTTLDaysUserAccessLogs is the w_user_access_logs TTL (180 days).
|
||||
TableTTLDaysUserAccessLogs = 180
|
||||
)
|
||||
|
||||
const (
|
||||
// CleanupModeTTLMaterialize expires rows via table TTL instead of ALTER DELETE mutations.
|
||||
// This is not a hard delete: deleted_count must stay 0; use EligibleCount as an estimate.
|
||||
CleanupModeTTLMaterialize = "ttl_materialize"
|
||||
// CleanupModeTruncate removes all rows via TRUNCATE TABLE (hard delete).
|
||||
CleanupModeTruncate = "truncate"
|
||||
)
|
||||
|
||||
// CleanupOutcome describes a non-mutation ClickHouse cleanup operation.
|
||||
//
|
||||
// For CleanupModeTruncate:
|
||||
// - DeletedCount and EligibleCount are the rows removed by TRUNCATE.
|
||||
//
|
||||
// For CleanupModeTTLMaterialize:
|
||||
// - DeletedCount is always 0 (MATERIALIZE TTL is async / not a counted hard delete).
|
||||
// - EligibleCount is an estimate of rows already past the table TTL policy (not an
|
||||
// arbitrary user cutoff younger than the DDL TTL).
|
||||
// - TableTTLDays is the DDL TTL used for the estimate and materialize.
|
||||
type CleanupOutcome struct {
|
||||
EligibleCount int64
|
||||
DeletedCount int64
|
||||
Mode string
|
||||
TableTTLDays int
|
||||
}
|
||||
|
||||
func countClickHouseRows(ctx context.Context, conn driver.Conn, countSQL string, countArgs []any) (int64, error) {
|
||||
var count uint64
|
||||
if err := conn.QueryRow(ctx, countSQL, countArgs...).Scan(&count); err != nil {
|
||||
return 0, fmt.Errorf("count clickhouse rows: %w", err)
|
||||
}
|
||||
return safeInt64Count(count), nil
|
||||
}
|
||||
|
||||
func materializeTableTTL(ctx context.Context, conn driver.Conn, tableName string) error {
|
||||
sql := fmt.Sprintf("ALTER TABLE %s MATERIALIZE TTL", tableName)
|
||||
if err := conn.Exec(ctx, sql); err != nil {
|
||||
return fmt.Errorf("materialize ttl on %s: %w", tableName, err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// tableTTLCutoff returns the UTC instant at which rows become eligible under a fixed day TTL.
|
||||
func tableTTLCutoff(tableTTLDays int, now time.Time) time.Time {
|
||||
if tableTTLDays < 1 {
|
||||
tableTTLDays = 1
|
||||
}
|
||||
return now.UTC().Add(-time.Duration(tableTTLDays) * 24 * time.Hour)
|
||||
}
|
||||
|
||||
// materializeExpiredByTableTTL force-materializes table TTL and estimates rows past that policy.
|
||||
//
|
||||
// countSQL must count only rows older than the table TTL (callers pass tableTTLCutoff args).
|
||||
// Node-scoped filters may be used for the estimate only; MATERIALIZE is always table-global.
|
||||
func materializeExpiredByTableTTL(
|
||||
ctx context.Context,
|
||||
conn driver.Conn,
|
||||
tableName string,
|
||||
tableTTLDays int,
|
||||
countSQL string,
|
||||
countArgs []any,
|
||||
) (CleanupOutcome, error) {
|
||||
outcome := CleanupOutcome{
|
||||
Mode: CleanupModeTTLMaterialize,
|
||||
TableTTLDays: tableTTLDays,
|
||||
}
|
||||
count, err := countClickHouseRows(ctx, conn, countSQL, countArgs)
|
||||
if err != nil {
|
||||
return CleanupOutcome{}, err
|
||||
}
|
||||
outcome.EligibleCount = count
|
||||
// Always force materialize so ClickHouse applies the DDL TTL policy promptly.
|
||||
// EligibleCount is informational only; MATERIALIZE does not return a deleted row count.
|
||||
if err := materializeTableTTL(ctx, conn, tableName); err != nil {
|
||||
return CleanupOutcome{}, err
|
||||
}
|
||||
return outcome, nil
|
||||
}
|
||||
|
||||
func truncateClickHouseTable(ctx context.Context, conn driver.Conn, tableName string) (CleanupOutcome, error) {
|
||||
count, err := countClickHouseRows(ctx, conn, "SELECT count() FROM "+tableName, nil)
|
||||
if err != nil {
|
||||
return CleanupOutcome{}, err
|
||||
}
|
||||
if count == 0 {
|
||||
return CleanupOutcome{Mode: CleanupModeTruncate}, nil
|
||||
}
|
||||
if err := conn.Exec(ctx, "TRUNCATE TABLE "+tableName); err != nil {
|
||||
return CleanupOutcome{}, fmt.Errorf("truncate %s: %w", tableName, err)
|
||||
}
|
||||
return CleanupOutcome{
|
||||
EligibleCount: count,
|
||||
DeletedCount: count,
|
||||
Mode: CleanupModeTruncate,
|
||||
}, nil
|
||||
}
|
||||
@@ -0,0 +1,37 @@
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
package analytics
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/stretchr/testify/assert"
|
||||
)
|
||||
|
||||
func TestTableTTLCutoff(t *testing.T) {
|
||||
now := time.Date(2026, 7, 10, 12, 0, 0, 0, time.UTC)
|
||||
got := tableTTLCutoff(30, now)
|
||||
assert.Equal(t, now.Add(-30*24*time.Hour), got)
|
||||
|
||||
got = tableTTLCutoff(90, now)
|
||||
assert.Equal(t, now.Add(-90*24*time.Hour), got)
|
||||
|
||||
// Invalid TTL floors to 1 day.
|
||||
got = tableTTLCutoff(0, now)
|
||||
assert.Equal(t, now.Add(-24*time.Hour), got)
|
||||
}
|
||||
|
||||
func TestCleanupModeConstants(t *testing.T) {
|
||||
assert.Equal(t, "ttl_materialize", CleanupModeTTLMaterialize)
|
||||
assert.Equal(t, "truncate", CleanupModeTruncate)
|
||||
}
|
||||
|
||||
func TestTableTTLDaysMatchDDL(t *testing.T) {
|
||||
assert.Equal(t, 90, TableTTLDaysNodeAccessLogs)
|
||||
assert.Equal(t, 30, TableTTLDaysNodeMetricSnapshots)
|
||||
assert.Equal(t, 30, TableTTLDaysNodeRequestReports)
|
||||
assert.Equal(t, 30, TableTTLDaysNodeObs)
|
||||
assert.Equal(t, 180, TableTTLDaysUserAccessLogs)
|
||||
}
|
||||
@@ -0,0 +1,74 @@
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
package analytics
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
|
||||
"github.com/Rain-kl/Wavelet/internal/config"
|
||||
"github.com/Rain-kl/Wavelet/internal/db"
|
||||
"github.com/Rain-kl/Wavelet/internal/db/batchwriter"
|
||||
)
|
||||
|
||||
// ClickHouseOperationalStats summarizes ClickHouse merge/mutation pressure
|
||||
// and in-process batch writer queue health.
|
||||
type ClickHouseOperationalStats struct {
|
||||
Database string `json:"database"`
|
||||
ActiveParts int64 `json:"active_parts"`
|
||||
TotalRows int64 `json:"total_rows"`
|
||||
PendingMutations int64 `json:"pending_mutations"`
|
||||
AsyncInsertQueue int64 `json:"async_insert_queue"`
|
||||
AsyncInsertBytes int64 `json:"async_insert_bytes"`
|
||||
// BatchWriters reports in-process queue depth/drops/flush errors for CH writers.
|
||||
BatchWriters []batchwriter.Stats `json:"batch_writers,omitempty"`
|
||||
}
|
||||
|
||||
// GetClickHouseOperationalStats returns operational metrics for the configured database.
|
||||
func GetClickHouseOperationalStats(ctx context.Context) (*ClickHouseOperationalStats, error) {
|
||||
if db.ChConn == nil {
|
||||
return nil, fmt.Errorf("clickhouse native connection is not initialized")
|
||||
}
|
||||
database := config.Config.ClickHouse.Database
|
||||
stats := &ClickHouseOperationalStats{Database: database}
|
||||
|
||||
partsSQL := `
|
||||
SELECT
|
||||
count() AS active_parts,
|
||||
ifNull(sum(rows), 0) AS total_rows
|
||||
FROM system.parts
|
||||
WHERE active AND database = ?`
|
||||
var activeParts, totalRows uint64
|
||||
if err := db.ChConn.QueryRow(ctx, partsSQL, database).Scan(&activeParts, &totalRows); err != nil {
|
||||
return nil, fmt.Errorf("query system.parts: %w", err)
|
||||
}
|
||||
stats.ActiveParts = safeInt64Count(activeParts)
|
||||
stats.TotalRows = safeInt64Count(totalRows)
|
||||
|
||||
mutationsSQL := `
|
||||
SELECT count()
|
||||
FROM system.mutations
|
||||
WHERE is_done = 0 AND database = ?`
|
||||
if err := db.ChConn.QueryRow(ctx, mutationsSQL, database).Scan(&stats.PendingMutations); err != nil {
|
||||
return nil, fmt.Errorf("query system.mutations: %w", err)
|
||||
}
|
||||
|
||||
asyncSQL := `
|
||||
SELECT
|
||||
count() AS queue_entries,
|
||||
ifNull(sum(bytes), 0) AS queue_bytes
|
||||
FROM system.asynchronous_inserts
|
||||
WHERE database = ?`
|
||||
var queueEntries, queueBytes uint64
|
||||
if err := db.ChConn.QueryRow(ctx, asyncSQL, database).Scan(&queueEntries, &queueBytes); err != nil {
|
||||
// Older ClickHouse versions may not expose asynchronous_inserts; treat as optional.
|
||||
stats.AsyncInsertQueue = 0
|
||||
stats.AsyncInsertBytes = 0
|
||||
} else {
|
||||
stats.AsyncInsertQueue = safeInt64Count(queueEntries)
|
||||
stats.AsyncInsertBytes = safeInt64Count(queueBytes)
|
||||
}
|
||||
|
||||
return stats, nil
|
||||
}
|
||||
@@ -87,23 +87,16 @@ func CountNodeAccessLogs(ctx context.Context, filter NodeAccessLogFilter) (int64
|
||||
clause, args := buildNodeAccessLogFilterClause(filter)
|
||||
tableName := nodeAccessLogTableName()
|
||||
|
||||
var totalRecords uint64
|
||||
countSQL := fmt.Sprintf("SELECT count() FROM %s WHERE %s", tableName, clause)
|
||||
if err := conn.QueryRow(ctx, countSQL, args...).Scan(&totalRecords); err != nil {
|
||||
countSQL := fmt.Sprintf(`
|
||||
SELECT
|
||||
count() AS total_records,
|
||||
uniqExactIf(remote_addr, remote_addr != '') AS total_ips
|
||||
FROM %s
|
||||
WHERE %s`, tableName, clause)
|
||||
var totalRecords, totalIPs uint64
|
||||
if err := conn.QueryRow(ctx, countSQL, args...).Scan(&totalRecords, &totalIPs); err != nil {
|
||||
return 0, 0, fmt.Errorf("count node access logs: %w", err)
|
||||
}
|
||||
|
||||
ipSQL := fmt.Sprintf(`
|
||||
SELECT count() FROM (
|
||||
SELECT trim(remote_addr) AS trimmed_remote_addr
|
||||
FROM %s
|
||||
WHERE %s AND trim(remote_addr) != ''
|
||||
GROUP BY trimmed_remote_addr
|
||||
)`, tableName, clause)
|
||||
var totalIPs uint64
|
||||
if err := conn.QueryRow(ctx, ipSQL, args...).Scan(&totalIPs); err != nil {
|
||||
return 0, 0, fmt.Errorf("count node access log ips: %w", err)
|
||||
}
|
||||
return safeInt64Count(totalRecords), safeInt64Count(totalIPs), nil
|
||||
}
|
||||
|
||||
|
||||
@@ -9,52 +9,79 @@ import (
|
||||
"time"
|
||||
)
|
||||
|
||||
// DeleteAllNodeAccessLogs deletes all node access logs.
|
||||
// DeleteAllNodeAccessLogs hard-deletes all node access logs via TRUNCATE.
|
||||
func DeleteAllNodeAccessLogs(ctx context.Context) (int64, error) {
|
||||
tableName := nodeAccessLogTableName()
|
||||
return deleteNodeAccessLogsWithCount(ctx, "SELECT count() FROM "+tableName, nil, "ALTER TABLE "+tableName+" DELETE WHERE 1")
|
||||
}
|
||||
|
||||
// DeleteNodeAccessLogsBefore deletes logs older than cutoff.
|
||||
func DeleteNodeAccessLogsBefore(ctx context.Context, cutoff time.Time) (int64, error) {
|
||||
tableName := nodeAccessLogTableName()
|
||||
cutoff = cutoff.UTC()
|
||||
return deleteNodeAccessLogsWithCount(
|
||||
ctx,
|
||||
fmt.Sprintf("SELECT count() FROM %s WHERE logged_at < ?", tableName),
|
||||
[]any{cutoff},
|
||||
fmt.Sprintf("ALTER TABLE %s DELETE WHERE logged_at < ?", tableName),
|
||||
cutoff,
|
||||
)
|
||||
}
|
||||
|
||||
// DeleteNodeAccessLogsByNodeBefore deletes logs for a node older than cutoff.
|
||||
func DeleteNodeAccessLogsByNodeBefore(ctx context.Context, nodeID string, before time.Time) (int64, error) {
|
||||
tableName := nodeAccessLogTableName()
|
||||
before = before.UTC()
|
||||
return deleteNodeAccessLogsWithCount(
|
||||
ctx,
|
||||
fmt.Sprintf("SELECT count() FROM %s WHERE node_id = ? AND logged_at < ?", tableName),
|
||||
[]any{nodeID, before},
|
||||
fmt.Sprintf("ALTER TABLE %s DELETE WHERE node_id = ? AND logged_at < ?", tableName),
|
||||
nodeID, before,
|
||||
)
|
||||
}
|
||||
|
||||
func deleteNodeAccessLogsWithCount(ctx context.Context, countSQL string, countArgs []any, deleteSQL string, deleteArgs ...any) (int64, error) {
|
||||
conn, err := nodeAccessLogConn()
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
var count uint64
|
||||
if err := conn.QueryRow(ctx, countSQL, countArgs...).Scan(&count); err != nil {
|
||||
return 0, fmt.Errorf("count node access logs for delete: %w", err)
|
||||
outcome, err := truncateClickHouseTable(ctx, conn, nodeAccessLogTableName())
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
if count == 0 {
|
||||
return 0, nil
|
||||
return outcome.DeletedCount, nil
|
||||
}
|
||||
|
||||
// DeleteNodeAccessLogsBefore force-materializes of_node_access_logs table TTL.
|
||||
//
|
||||
// The cutoff argument is kept for call-site compatibility and is not used to select rows:
|
||||
// ClickHouse MATERIALIZE TTL only enforces the DDL policy (TableTTLDaysNodeAccessLogs).
|
||||
// Returns an estimate of rows past table TTL as the int64 (not a hard-deleted count).
|
||||
// Callers that need honest API fields should prefer MaterializeNodeAccessLogsTTL.
|
||||
func DeleteNodeAccessLogsBefore(ctx context.Context, _ time.Time) (int64, error) {
|
||||
outcome, err := MaterializeNodeAccessLogsTTL(ctx)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
if err := conn.Exec(ctx, deleteSQL, deleteArgs...); err != nil {
|
||||
return 0, fmt.Errorf("delete node access logs: %w", err)
|
||||
return outcome.EligibleCount, nil
|
||||
}
|
||||
|
||||
// MaterializeNodeAccessLogsTTL force-materializes table TTL and reports an honest outcome.
|
||||
func MaterializeNodeAccessLogsTTL(ctx context.Context) (CleanupOutcome, error) {
|
||||
conn, err := nodeAccessLogConn()
|
||||
if err != nil {
|
||||
return CleanupOutcome{}, err
|
||||
}
|
||||
return safeInt64Count(count), nil
|
||||
}
|
||||
tableName := nodeAccessLogTableName()
|
||||
ttlDays := TableTTLDaysNodeAccessLogs
|
||||
cutoff := tableTTLCutoff(ttlDays, time.Now())
|
||||
return materializeExpiredByTableTTL(
|
||||
ctx,
|
||||
conn,
|
||||
tableName,
|
||||
ttlDays,
|
||||
fmt.Sprintf("SELECT count() FROM %s WHERE logged_at < ?", tableName),
|
||||
[]any{cutoff},
|
||||
)
|
||||
}
|
||||
|
||||
// DeleteNodeAccessLogsByNodeBefore force-materializes table-global TTL.
|
||||
//
|
||||
// Node-scoped hard delete is not supported: MATERIALIZE TTL is table-global.
|
||||
// The returned count is an estimate of rows for nodeID past table TTL only.
|
||||
func DeleteNodeAccessLogsByNodeBefore(ctx context.Context, nodeID string, _ time.Time) (int64, error) {
|
||||
outcome, err := MaterializeNodeAccessLogsTTLByNode(ctx, nodeID)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.EligibleCount, nil
|
||||
}
|
||||
|
||||
// MaterializeNodeAccessLogsTTLByNode materializes table-global TTL and estimates node-scoped rows past TTL.
|
||||
func MaterializeNodeAccessLogsTTLByNode(ctx context.Context, nodeID string) (CleanupOutcome, error) {
|
||||
conn, err := nodeAccessLogConn()
|
||||
if err != nil {
|
||||
return CleanupOutcome{}, err
|
||||
}
|
||||
tableName := nodeAccessLogTableName()
|
||||
ttlDays := TableTTLDaysNodeAccessLogs
|
||||
cutoff := tableTTLCutoff(ttlDays, time.Now())
|
||||
return materializeExpiredByTableTTL(
|
||||
ctx,
|
||||
conn,
|
||||
tableName,
|
||||
ttlDays,
|
||||
fmt.Sprintf("SELECT count() FROM %s WHERE node_id = ? AND logged_at < ?", tableName),
|
||||
[]any{nodeID, cutoff},
|
||||
)
|
||||
}
|
||||
|
||||
@@ -9,7 +9,15 @@ import (
|
||||
"time"
|
||||
)
|
||||
|
||||
const nodeAccessLogFilterClauseCapacity = 6
|
||||
const (
|
||||
nodeAccessLogFilterClauseCapacity = 6
|
||||
|
||||
nodeAccessLogSortDesc = "DESC"
|
||||
nodeAccessLogSortAsc = "ASC"
|
||||
nodeAccessLogSortAscInput = "asc"
|
||||
|
||||
nodeAccessLogColumnRemoteAddr = "remote_addr"
|
||||
)
|
||||
|
||||
// NodeAccessLogFilter scopes ClickHouse node access log queries.
|
||||
type NodeAccessLogFilter struct {
|
||||
@@ -32,7 +40,7 @@ func buildNodeAccessLogFilterClause(filter NodeAccessLogFilter) (string, []any)
|
||||
parts = append(parts, "node_id = ?")
|
||||
args = append(args, trimmed)
|
||||
}
|
||||
if trimmed := strings.TrimSpace(filter.RemoteAddr); trimmed != "" {
|
||||
if trimmed := normalizeNodeAccessLogRemoteAddr(filter.RemoteAddr); trimmed != "" {
|
||||
parts = append(parts, "remote_addr LIKE ?")
|
||||
args = append(args, trimmed+"%")
|
||||
}
|
||||
@@ -66,16 +74,16 @@ func combineNodeAccessLogSQLClauses(left string, right string) string {
|
||||
}
|
||||
|
||||
func nodeAccessLogOrderClause(sortBy string, sortOrder string) string {
|
||||
direction := "DESC"
|
||||
if normalizeNodeAccessLogSortOrder(sortOrder) == "asc" {
|
||||
direction = "ASC"
|
||||
direction := nodeAccessLogSortDesc
|
||||
if normalizeNodeAccessLogSortOrder(sortOrder) == nodeAccessLogSortAscInput {
|
||||
direction = nodeAccessLogSortAsc
|
||||
}
|
||||
column := "logged_at"
|
||||
switch strings.TrimSpace(sortBy) {
|
||||
case "status_code":
|
||||
column = "status_code"
|
||||
case "remote_addr":
|
||||
column = "remote_addr"
|
||||
case nodeAccessLogColumnRemoteAddr:
|
||||
column = nodeAccessLogColumnRemoteAddr
|
||||
case "host":
|
||||
column = "host"
|
||||
case "path":
|
||||
@@ -87,6 +95,10 @@ func nodeAccessLogOrderClause(sortBy string, sortOrder string) string {
|
||||
return column + " " + direction + ", logged_at " + direction + ", id " + direction
|
||||
}
|
||||
|
||||
func normalizeNodeAccessLogRemoteAddr(value string) string {
|
||||
return strings.TrimSpace(value)
|
||||
}
|
||||
|
||||
func normalizeNodeAccessLogSortOrder(sortOrder string) string {
|
||||
if strings.EqualFold(strings.TrimSpace(sortOrder), "asc") {
|
||||
return "asc"
|
||||
@@ -102,6 +114,43 @@ func nodeAccessLogEpochExpr() string {
|
||||
return "toInt64(toUnixTimestamp(logged_at))"
|
||||
}
|
||||
|
||||
func nodeAccessLogHostIsIPLiteralExpr() string {
|
||||
return `(
|
||||
toIPv4OrNull(trim(if(position(trim(host), ':') > 0 AND NOT startsWith(trim(host), '['), splitByChar(':', trim(host))[1], replaceRegexpAll(trim(host), '\\[|\\]', '')))) IS NOT NULL
|
||||
OR toIPv6OrNull(trim(if(position(trim(host), ':') > 0 AND NOT startsWith(trim(host), '['), splitByChar(':', trim(host))[1], replaceRegexpAll(trim(host), '\\[|\\]', '')))) IS NOT NULL
|
||||
)`
|
||||
}
|
||||
|
||||
func nodeAccessLogBucketOrderClause(sortBy string, sortOrder string) string {
|
||||
direction := nodeAccessLogSortDesc
|
||||
if normalizeNodeAccessLogSortOrder(sortOrder) == nodeAccessLogSortAscInput {
|
||||
direction = nodeAccessLogSortAsc
|
||||
}
|
||||
switch strings.TrimSpace(sortBy) {
|
||||
case "request_count":
|
||||
return "request_count " + direction + ", bucket_epoch DESC"
|
||||
default:
|
||||
return "bucket_epoch " + direction
|
||||
}
|
||||
}
|
||||
|
||||
func nodeAccessLogIPSummaryOrderClause(sortBy string, sortOrder string) string {
|
||||
direction := nodeAccessLogSortDesc
|
||||
if normalizeNodeAccessLogSortOrder(sortOrder) == nodeAccessLogSortAscInput {
|
||||
direction = nodeAccessLogSortAsc
|
||||
}
|
||||
column := "total_requests"
|
||||
switch strings.TrimSpace(sortBy) {
|
||||
case "recent_requests":
|
||||
column = "recent_requests"
|
||||
case "last_seen_at":
|
||||
column = "last_seen_epoch"
|
||||
case nodeAccessLogColumnRemoteAddr:
|
||||
column = nodeAccessLogColumnRemoteAddr
|
||||
}
|
||||
return column + " " + direction + ", last_seen_epoch DESC, remote_addr ASC"
|
||||
}
|
||||
|
||||
func nodeAccessLogTableName() string {
|
||||
return "of_node_access_logs"
|
||||
}
|
||||
|
||||
@@ -17,6 +17,20 @@ type NodeAccessLogBucketAggregate struct {
|
||||
SuccessCount int64
|
||||
ClientErrorCount int64
|
||||
ServerErrorCount int64
|
||||
UniqueIPCount int64
|
||||
UniqueHostCount int64
|
||||
}
|
||||
|
||||
// NodeAccessLogWAFIPAggregate is a per-IP aggregate row for WAF automatic rules.
|
||||
type NodeAccessLogWAFIPAggregate struct {
|
||||
RemoteAddr string
|
||||
RequestCount int64
|
||||
Status404Count int64
|
||||
ClientErrorCount int64
|
||||
ServerErrorCount int64
|
||||
IPHostCount int64
|
||||
LastSeenEpoch int64
|
||||
StatusCounts map[int]int64
|
||||
}
|
||||
|
||||
// NodeAccessLogBucketDimension is a bucket dimension value.
|
||||
@@ -49,7 +63,7 @@ type NodeAccessLogIPTrend struct {
|
||||
RequestCount int64
|
||||
}
|
||||
|
||||
// BucketAggregatesNodeAccessLogs returns folded bucket aggregates.
|
||||
// BucketAggregatesNodeAccessLogs returns folded bucket aggregates with unique IP/host counts.
|
||||
func BucketAggregatesNodeAccessLogs(ctx context.Context, filter NodeAccessLogFilter, bucketSeconds int64) ([]NodeAccessLogBucketAggregate, error) {
|
||||
conn, err := nodeAccessLogConn()
|
||||
if err != nil {
|
||||
@@ -64,10 +78,20 @@ SELECT
|
||||
count() AS request_count,
|
||||
countIf(status_code < 400) AS success_count,
|
||||
countIf(status_code >= 400 AND status_code < 500) AS client_error_count,
|
||||
countIf(status_code >= 500) AS server_error_count
|
||||
countIf(status_code >= 500) AS server_error_count,
|
||||
uniqExactIf(remote_addr, remote_addr != '') AS unique_ip_count,
|
||||
uniqExactIf(host, host != '') AS unique_host_count
|
||||
FROM %s
|
||||
WHERE %s
|
||||
GROUP BY bucket_epoch`, bucketExpr, tableName, clause)
|
||||
GROUP BY bucket_epoch
|
||||
ORDER BY %s`, bucketExpr, tableName, clause, nodeAccessLogBucketOrderClause(filter.SortBy, filter.SortOrder))
|
||||
if filter.PageSize > 0 {
|
||||
if filter.Page < 0 {
|
||||
filter.Page = 0
|
||||
}
|
||||
sql += clickHouseLimitOffsetClause
|
||||
args = append(args, filter.PageSize, filter.Page*filter.PageSize)
|
||||
}
|
||||
rows, err := conn.Query(ctx, sql, args...)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("bucket aggregates node access logs: %w", err)
|
||||
@@ -77,10 +101,10 @@ GROUP BY bucket_epoch`, bucketExpr, tableName, clause)
|
||||
var result []NodeAccessLogBucketAggregate
|
||||
for rows.Next() {
|
||||
var (
|
||||
bucketEpoch int64
|
||||
requestCount, successCount, clientErrorCount, serverErrorCount uint64
|
||||
bucketEpoch int64
|
||||
requestCount, successCount, clientErrorCount, serverErrorCount, uniqueIPCount, uniqueHostCount uint64
|
||||
)
|
||||
if err := rows.Scan(&bucketEpoch, &requestCount, &successCount, &clientErrorCount, &serverErrorCount); err != nil {
|
||||
if err := rows.Scan(&bucketEpoch, &requestCount, &successCount, &clientErrorCount, &serverErrorCount, &uniqueIPCount, &uniqueHostCount); err != nil {
|
||||
return nil, fmt.Errorf("scan bucket aggregate row: %w", err)
|
||||
}
|
||||
result = append(result, NodeAccessLogBucketAggregate{
|
||||
@@ -89,11 +113,36 @@ GROUP BY bucket_epoch`, bucketExpr, tableName, clause)
|
||||
SuccessCount: safeInt64Count(successCount),
|
||||
ClientErrorCount: safeInt64Count(clientErrorCount),
|
||||
ServerErrorCount: safeInt64Count(serverErrorCount),
|
||||
UniqueIPCount: safeInt64Count(uniqueIPCount),
|
||||
UniqueHostCount: safeInt64Count(uniqueHostCount),
|
||||
})
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// CountBucketAggregatesNodeAccessLogs returns the number of folded buckets matching filter.
|
||||
func CountBucketAggregatesNodeAccessLogs(ctx context.Context, filter NodeAccessLogFilter, bucketSeconds int64) (int64, error) {
|
||||
conn, err := nodeAccessLogConn()
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
clause, args := buildNodeAccessLogFilterClause(filter)
|
||||
bucketExpr := nodeAccessLogBucketEpochExpr(bucketSeconds)
|
||||
tableName := nodeAccessLogTableName()
|
||||
sql := fmt.Sprintf(`
|
||||
SELECT count() FROM (
|
||||
SELECT 1
|
||||
FROM %s
|
||||
WHERE %s
|
||||
GROUP BY %s
|
||||
)`, tableName, clause, bucketExpr)
|
||||
var totalBuckets uint64
|
||||
if err := conn.QueryRow(ctx, sql, args...).Scan(&totalBuckets); err != nil {
|
||||
return 0, fmt.Errorf("count bucket aggregates node access logs: %w", err)
|
||||
}
|
||||
return safeInt64Count(totalBuckets), nil
|
||||
}
|
||||
|
||||
// BucketDimensionsNodeAccessLogs returns bucket dimension values.
|
||||
func BucketDimensionsNodeAccessLogs(ctx context.Context, filter NodeAccessLogFilter, column string, bucketSeconds int64) ([]NodeAccessLogBucketDimension, error) {
|
||||
conn, err := nodeAccessLogConn()
|
||||
@@ -137,26 +186,26 @@ func IPAggregatesNodeAccessLogs(ctx context.Context, filter NodeAccessLogFilter,
|
||||
queryClause := clause
|
||||
queryArgs := append([]any{}, args...)
|
||||
if exactRemoteAddr {
|
||||
trimmed := strings.TrimSpace(filter.RemoteAddr)
|
||||
trimmed := normalizeNodeAccessLogRemoteAddr(filter.RemoteAddr)
|
||||
if trimmed == "" {
|
||||
return []NodeAccessLogIPAggregate{}, nil
|
||||
}
|
||||
queryClause = combineNodeAccessLogSQLClauses(queryClause, "trim(remote_addr) = ?")
|
||||
queryClause = combineNodeAccessLogSQLClauses(queryClause, "remote_addr = ?")
|
||||
queryArgs = append(queryArgs, trimmed)
|
||||
}
|
||||
lastSeenExpr := nodeAccessLogEpochExpr()
|
||||
tableName := nodeAccessLogTableName()
|
||||
sql := fmt.Sprintf(`
|
||||
SELECT
|
||||
trim(remote_addr) AS trimmed_remote_addr,
|
||||
remote_addr,
|
||||
count() AS request_count,
|
||||
countIf(status_code < 400) AS success_count,
|
||||
countIf(status_code >= 400 AND status_code < 500) AS client_error_count,
|
||||
countIf(status_code >= 500) AS server_error_count,
|
||||
max(%s) AS last_seen_epoch
|
||||
FROM %s
|
||||
WHERE %s AND trim(remote_addr) != ''
|
||||
GROUP BY trimmed_remote_addr`, lastSeenExpr, tableName, queryClause)
|
||||
WHERE %s AND remote_addr != ''
|
||||
GROUP BY remote_addr`, lastSeenExpr, tableName, queryClause)
|
||||
rows, err := conn.Query(ctx, sql, queryArgs...)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("ip aggregates node access logs: %w", err)
|
||||
@@ -185,7 +234,7 @@ GROUP BY trimmed_remote_addr`, lastSeenExpr, tableName, queryClause)
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// IPSummariesNodeAccessLogs returns IP summary rows.
|
||||
// IPSummariesNodeAccessLogs returns paginated IP summary rows.
|
||||
func IPSummariesNodeAccessLogs(ctx context.Context, filter NodeAccessLogFilter, recentSince time.Time) ([]NodeAccessLogIPSummary, error) {
|
||||
conn, err := nodeAccessLogConn()
|
||||
if err != nil {
|
||||
@@ -203,13 +252,21 @@ func IPSummariesNodeAccessLogs(ctx context.Context, filter NodeAccessLogFilter,
|
||||
tableName := nodeAccessLogTableName()
|
||||
sql := fmt.Sprintf(`
|
||||
SELECT
|
||||
trim(remote_addr) AS trimmed_remote_addr,
|
||||
remote_addr,
|
||||
count() AS total_requests,
|
||||
sum(%s) AS recent_requests,
|
||||
max(%s) AS last_seen_epoch
|
||||
FROM %s
|
||||
WHERE %s AND trim(remote_addr) != ''
|
||||
GROUP BY trimmed_remote_addr`, recentClause, lastSeenExpr, tableName, clause)
|
||||
WHERE %s AND remote_addr != ''
|
||||
GROUP BY remote_addr
|
||||
ORDER BY %s`, recentClause, lastSeenExpr, tableName, clause, nodeAccessLogIPSummaryOrderClause(filter.SortBy, filter.SortOrder))
|
||||
if filter.PageSize > 0 {
|
||||
if filter.Page < 0 {
|
||||
filter.Page = 0
|
||||
}
|
||||
sql += clickHouseLimitOffsetClause
|
||||
queryArgs = append(queryArgs, filter.PageSize, filter.Page*filter.PageSize)
|
||||
}
|
||||
rows, err := conn.Query(ctx, sql, queryArgs...)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("ip summaries node access logs: %w", err)
|
||||
@@ -236,6 +293,140 @@ GROUP BY trimmed_remote_addr`, recentClause, lastSeenExpr, tableName, clause)
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// CountIPSummaryNodeAccessLogs returns the number of distinct IPs matching filter.
|
||||
func CountIPSummaryNodeAccessLogs(ctx context.Context, filter NodeAccessLogFilter) (int64, error) {
|
||||
conn, err := nodeAccessLogConn()
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
clause, args := buildNodeAccessLogFilterClause(filter)
|
||||
tableName := nodeAccessLogTableName()
|
||||
sql := fmt.Sprintf(`
|
||||
SELECT count() FROM (
|
||||
SELECT 1
|
||||
FROM %s
|
||||
WHERE %s AND remote_addr != ''
|
||||
GROUP BY remote_addr
|
||||
)`, tableName, clause)
|
||||
var totalIPs uint64
|
||||
if err := conn.QueryRow(ctx, sql, args...).Scan(&totalIPs); err != nil {
|
||||
return 0, fmt.Errorf("count ip summary node access logs: %w", err)
|
||||
}
|
||||
return safeInt64Count(totalIPs), nil
|
||||
}
|
||||
|
||||
// IPAggregatesForWAFNodeAccessLogs returns per-IP aggregates for WAF automatic rules.
|
||||
func IPAggregatesForWAFNodeAccessLogs(ctx context.Context, filter NodeAccessLogFilter) ([]NodeAccessLogWAFIPAggregate, error) {
|
||||
conn, err := nodeAccessLogConn()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
clause, args := buildNodeAccessLogFilterClause(filter)
|
||||
lastSeenExpr := nodeAccessLogEpochExpr()
|
||||
hostIsIPExpr := nodeAccessLogHostIsIPLiteralExpr()
|
||||
tableName := nodeAccessLogTableName()
|
||||
sql := fmt.Sprintf(`
|
||||
SELECT
|
||||
remote_addr,
|
||||
count() AS request_count,
|
||||
countIf(status_code = 404) AS status_404_count,
|
||||
countIf(status_code >= 400 AND status_code < 500) AS client_error_count,
|
||||
countIf(status_code >= 500) AS server_error_count,
|
||||
countIf(%s) AS ip_host_count,
|
||||
max(%s) AS last_seen_epoch
|
||||
FROM %s
|
||||
WHERE %s AND remote_addr != ''
|
||||
GROUP BY remote_addr`, hostIsIPExpr, lastSeenExpr, tableName, clause)
|
||||
rows, err := conn.Query(ctx, sql, args...)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("ip aggregates for waf node access logs: %w", err)
|
||||
}
|
||||
defer func() { _ = rows.Close() }()
|
||||
|
||||
aggregates := make(map[string]*NodeAccessLogWAFIPAggregate)
|
||||
order := make([]string, 0)
|
||||
for rows.Next() {
|
||||
var (
|
||||
remoteAddr string
|
||||
lastSeenEpoch int64
|
||||
requestCount, status404Count, clientErrorCount, serverErrorCount, ipHostCount uint64
|
||||
)
|
||||
if err := rows.Scan(&remoteAddr, &requestCount, &status404Count, &clientErrorCount, &serverErrorCount, &ipHostCount, &lastSeenEpoch); err != nil {
|
||||
return nil, fmt.Errorf("scan waf ip aggregate row: %w", err)
|
||||
}
|
||||
remoteAddr = strings.TrimSpace(remoteAddr)
|
||||
if remoteAddr == "" {
|
||||
continue
|
||||
}
|
||||
aggregates[remoteAddr] = &NodeAccessLogWAFIPAggregate{
|
||||
RemoteAddr: remoteAddr,
|
||||
RequestCount: safeInt64Count(requestCount),
|
||||
Status404Count: safeInt64Count(status404Count),
|
||||
ClientErrorCount: safeInt64Count(clientErrorCount),
|
||||
ServerErrorCount: safeInt64Count(serverErrorCount),
|
||||
IPHostCount: safeInt64Count(ipHostCount),
|
||||
LastSeenEpoch: lastSeenEpoch,
|
||||
StatusCounts: make(map[int]int64),
|
||||
}
|
||||
order = append(order, remoteAddr)
|
||||
}
|
||||
if err := mergeWAFIPStatusCodeCounts(ctx, filter, aggregates); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
result := make([]NodeAccessLogWAFIPAggregate, 0, len(order))
|
||||
for _, remoteAddr := range order {
|
||||
if aggregate := aggregates[remoteAddr]; aggregate != nil {
|
||||
result = append(result, *aggregate)
|
||||
}
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
func mergeWAFIPStatusCodeCounts(ctx context.Context, filter NodeAccessLogFilter, aggregates map[string]*NodeAccessLogWAFIPAggregate) error {
|
||||
if len(aggregates) == 0 {
|
||||
return nil
|
||||
}
|
||||
conn, err := nodeAccessLogConn()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
clause, args := buildNodeAccessLogFilterClause(filter)
|
||||
tableName := nodeAccessLogTableName()
|
||||
sql := fmt.Sprintf(`
|
||||
SELECT
|
||||
remote_addr,
|
||||
status_code,
|
||||
count() AS status_count
|
||||
FROM %s
|
||||
WHERE %s AND remote_addr != ''
|
||||
GROUP BY remote_addr, status_code`, tableName, clause)
|
||||
rows, err := conn.Query(ctx, sql, args...)
|
||||
if err != nil {
|
||||
return fmt.Errorf("waf ip status code counts: %w", err)
|
||||
}
|
||||
defer func() { _ = rows.Close() }()
|
||||
for rows.Next() {
|
||||
var (
|
||||
remoteAddr string
|
||||
statusCode int32
|
||||
statusCount uint64
|
||||
)
|
||||
if err := rows.Scan(&remoteAddr, &statusCode, &statusCount); err != nil {
|
||||
return fmt.Errorf("scan waf ip status code row: %w", err)
|
||||
}
|
||||
remoteAddr = strings.TrimSpace(remoteAddr)
|
||||
aggregate := aggregates[remoteAddr]
|
||||
if aggregate == nil {
|
||||
continue
|
||||
}
|
||||
if aggregate.StatusCounts == nil {
|
||||
aggregate.StatusCounts = make(map[int]int64)
|
||||
}
|
||||
aggregate.StatusCounts[int(statusCode)] = safeInt64Count(statusCount)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// IPTrendNodeAccessLogs returns IP trend bucket rows.
|
||||
func IPTrendNodeAccessLogs(ctx context.Context, filter NodeAccessLogFilter, bucketSeconds int64) ([]NodeAccessLogIPTrend, error) {
|
||||
conn, err := nodeAccessLogConn()
|
||||
|
||||
@@ -6,6 +6,7 @@ package analytics
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/Rain-kl/Wavelet/internal/db"
|
||||
@@ -41,7 +42,7 @@ func BatchInsertNodeAccessLogs(ctx context.Context, logs []analyticsmodel.NodeAc
|
||||
id,
|
||||
logItem.NodeID,
|
||||
logItem.LoggedAt.UTC(),
|
||||
logItem.RemoteAddr,
|
||||
strings.TrimSpace(logItem.RemoteAddr),
|
||||
logItem.Region,
|
||||
logItem.Host,
|
||||
logItem.Path,
|
||||
|
||||
@@ -6,6 +6,8 @@ package analytics
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"sort"
|
||||
"time"
|
||||
|
||||
"github.com/ClickHouse/clickhouse-go/v2/lib/driver"
|
||||
"github.com/Rain-kl/Wavelet/internal/db"
|
||||
@@ -44,6 +46,27 @@ ORDER BY %s`, tableName, clause, nodeObservabilityCapturedAtOrderClause())
|
||||
return scanNodeMetricSnapshotRows(rows)
|
||||
}
|
||||
|
||||
// ListLatestNodeMetricSnapshots returns the latest metric snapshot per node_id.
|
||||
// Uses ClickHouse LIMIT 1 BY so dashboard health does not depend on a global raw LIMIT.
|
||||
func ListLatestNodeMetricSnapshots(ctx context.Context, filter NodeObservabilityFilter) ([]analyticsmodel.NodeMetricSnapshot, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
clause, args := buildNodeObservabilityFilterClause(filter, "captured_at")
|
||||
sql := fmt.Sprintf(`
|
||||
SELECT id, node_id, captured_at, cpu_usage_percent, memory_used_bytes, memory_total_bytes, storage_used_bytes, storage_total_bytes, disk_read_bytes, disk_write_bytes, network_rx_bytes, network_tx_bytes, created_at
|
||||
FROM %s
|
||||
WHERE %s
|
||||
ORDER BY %s%s`, nodeMetricSnapshotTableName(), clause, nodeObservabilityCapturedAtOrderClause(), clickHouseLimit1ByNodeIDClause)
|
||||
rows, err := conn.Query(ctx, sql, args...)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("list latest node metric snapshots: %w", err)
|
||||
}
|
||||
defer func() { _ = rows.Close() }()
|
||||
return scanNodeMetricSnapshotRows(rows)
|
||||
}
|
||||
|
||||
// ListNodeRequestReports returns request reports matching filter.
|
||||
func ListNodeRequestReports(ctx context.Context, filter NodeObservabilityFilter) ([]analyticsmodel.NodeRequestReport, error) {
|
||||
conn, err := observabilityConn()
|
||||
@@ -69,6 +92,27 @@ ORDER BY %s`, tableName, clause, nodeObservabilityWindowEndedAtOrderClause())
|
||||
return scanNodeRequestReportRows(rows)
|
||||
}
|
||||
|
||||
// ListLatestNodeRequestReports returns the latest request report per node_id.
|
||||
// Uses ClickHouse LIMIT 1 BY so dashboard traffic health is not skewed by a global raw LIMIT.
|
||||
func ListLatestNodeRequestReports(ctx context.Context, filter NodeObservabilityFilter) ([]analyticsmodel.NodeRequestReport, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
clause, args := buildNodeObservabilityFilterClause(filter, "window_ended_at")
|
||||
sql := fmt.Sprintf(`
|
||||
SELECT id, node_id, window_started_at, window_ended_at, request_count, error_count, unique_visitor_count, status_codes_json, top_domains_json, source_countries_json, created_at
|
||||
FROM %s
|
||||
WHERE %s
|
||||
ORDER BY %s%s`, nodeRequestReportTableName(), clause, nodeObservabilityWindowEndedAtOrderClause(), clickHouseLimit1ByNodeIDClause)
|
||||
rows, err := conn.Query(ctx, sql, args...)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("list latest node request reports: %w", err)
|
||||
}
|
||||
defer func() { _ = rows.Close() }()
|
||||
return scanNodeRequestReportRows(rows)
|
||||
}
|
||||
|
||||
// ListNodeObsOpenresty returns OpenResty observations matching filter.
|
||||
func ListNodeObsOpenresty(ctx context.Context, filter NodeObservabilityFilter) ([]analyticsmodel.NodeObsOpenresty, error) {
|
||||
conn, err := observabilityConn()
|
||||
@@ -244,6 +288,406 @@ func scanNodeObsFrpsRows(rows driver.Rows) ([]analyticsmodel.NodeObsFrps, error)
|
||||
return result, nil
|
||||
}
|
||||
|
||||
const nodeTrafficHourlyTableName = "of_node_traffic_hourly"
|
||||
|
||||
// NodeTrafficHourly is an hourly traffic rollup row.
|
||||
//
|
||||
// UniqueVisitorCount is a peak per-window estimate from short request reports
|
||||
// (MV uses max()), not true distinct visitors across the hour. SummingMergeTree
|
||||
// may still inflate residual unmerged parts; do not present as exact UV.
|
||||
type NodeTrafficHourly struct {
|
||||
NodeID string
|
||||
Hour time.Time
|
||||
RequestCount int64
|
||||
ErrorCount int64
|
||||
UniqueVisitorCount int64
|
||||
}
|
||||
|
||||
// NodeMetricHourly is an hourly metric snapshot aggregation row.
|
||||
//
|
||||
// Disk and host network counters are cumulative. Prefer pre-aggregated min/max
|
||||
// deltas from of_node_metric_capacity_hourly; raw fallback uses consecutive
|
||||
// lagInFrame samples per node (negative deltas after counter reset are dropped).
|
||||
type NodeMetricHourly struct {
|
||||
Hour time.Time
|
||||
AverageCPUUsagePercent float64
|
||||
AverageMemoryUsagePercent float64
|
||||
NetworkRxBytes int64
|
||||
NetworkTxBytes int64
|
||||
DiskReadBytes int64
|
||||
DiskWriteBytes int64
|
||||
ReportedNodes int
|
||||
}
|
||||
|
||||
// NodeOpenrestyHourly is an hourly OpenResty observation aggregation row.
|
||||
type NodeOpenrestyHourly struct {
|
||||
Hour time.Time
|
||||
OpenrestyRxBytes int64
|
||||
OpenrestyTxBytes int64
|
||||
ReportedNodes int
|
||||
}
|
||||
|
||||
// ListNodeTrafficHourly returns hourly traffic rollup rows matching filter.
|
||||
func ListNodeTrafficHourly(ctx context.Context, filter NodeObservabilityFilter) ([]NodeTrafficHourly, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
clause, args := buildNodeObservabilityFilterClause(filter, "hour")
|
||||
sql := fmt.Sprintf(`
|
||||
SELECT
|
||||
node_id,
|
||||
hour,
|
||||
sum(request_count) AS request_count,
|
||||
sum(error_count) AS error_count,
|
||||
max(unique_visitor_count) AS unique_visitor_count
|
||||
FROM %s
|
||||
WHERE %s
|
||||
GROUP BY node_id, hour
|
||||
ORDER BY hour ASC`, nodeTrafficHourlyTableName, clause)
|
||||
rows, err := conn.Query(ctx, sql, args...)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("list node traffic hourly: %w", err)
|
||||
}
|
||||
defer func() { _ = rows.Close() }()
|
||||
|
||||
result := make([]NodeTrafficHourly, 0)
|
||||
for rows.Next() {
|
||||
var (
|
||||
item NodeTrafficHourly
|
||||
requestCount, errorCount, uniqueVisitorCount uint64
|
||||
)
|
||||
if err := rows.Scan(&item.NodeID, &item.Hour, &requestCount, &errorCount, &uniqueVisitorCount); err != nil {
|
||||
return nil, fmt.Errorf("scan node traffic hourly row: %w", err)
|
||||
}
|
||||
item.Hour = item.Hour.UTC()
|
||||
item.RequestCount = safeInt64Count(requestCount)
|
||||
item.ErrorCount = safeInt64Count(errorCount)
|
||||
item.UniqueVisitorCount = safeInt64Count(uniqueVisitorCount)
|
||||
result = append(result, item)
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// hourlyRollupMaxLead is how far after filter.Since the earliest rollup bucket may start
|
||||
// while still treating pre-aggregated tables as a complete window (skip raw query).
|
||||
const hourlyRollupMaxLead = 2 * time.Hour
|
||||
|
||||
// hourlyRollupCoversWindow reports whether rollup coverage starts near the requested window.
|
||||
// rows must be ordered by hour ascending.
|
||||
func hourlyRollupCoversWindow(earliestHour time.Time, since time.Time) bool {
|
||||
if since.IsZero() {
|
||||
return true
|
||||
}
|
||||
sinceHour := since.UTC().Truncate(time.Hour)
|
||||
earliest := earliestHour.UTC().Truncate(time.Hour)
|
||||
return !earliest.After(sinceHour.Add(hourlyRollupMaxLead))
|
||||
}
|
||||
|
||||
// ListNodeMetricHourly returns hourly metric snapshot aggregates matching filter.
|
||||
//
|
||||
// Strategy (optimal for correctness + cost):
|
||||
// 1. Load of_node_metric_capacity_hourly rollup.
|
||||
// 2. If rollup spans the window from filter.Since, return it alone (cheap path).
|
||||
// 3. Otherwise load raw lagInFrame aggregates and merge by hour: rollup wins on
|
||||
// overlap, raw fills historical gaps (MV never backfills pre-creation data).
|
||||
func ListNodeMetricHourly(ctx context.Context, filter NodeObservabilityFilter) ([]NodeMetricHourly, error) {
|
||||
rollup, rollupErr := listNodeMetricHourlyFromRollup(ctx, filter)
|
||||
if rollupErr == nil && len(rollup) > 0 && hourlyRollupCoversWindow(rollup[0].Hour, filter.Since) {
|
||||
return rollup, nil
|
||||
}
|
||||
|
||||
raw, rawErr := listNodeMetricHourlyFromRaw(ctx, filter)
|
||||
if rawErr != nil {
|
||||
if rollupErr == nil && len(rollup) > 0 {
|
||||
return rollup, nil
|
||||
}
|
||||
return nil, rawErr
|
||||
}
|
||||
if len(rollup) == 0 {
|
||||
return raw, nil
|
||||
}
|
||||
// Partial rollup (or rollupErr with empty slice): merge; raw fills historical gaps.
|
||||
return mergeNodeMetricHourlyPreferRollup(rollup, raw), nil
|
||||
}
|
||||
|
||||
// mergeNodeMetricHourlyPreferRollup unions two hour series (both ASC by Hour).
|
||||
// Rollup values replace raw for the same hour; raw supplies missing hours.
|
||||
func mergeNodeMetricHourlyPreferRollup(rollup, raw []NodeMetricHourly) []NodeMetricHourly {
|
||||
byHour := make(map[int64]NodeMetricHourly, len(raw)+len(rollup))
|
||||
order := make([]int64, 0, len(raw)+len(rollup))
|
||||
add := func(row NodeMetricHourly, overwrite bool) {
|
||||
key := row.Hour.UTC().Truncate(time.Hour).Unix()
|
||||
if _, exists := byHour[key]; !exists {
|
||||
order = append(order, key)
|
||||
byHour[key] = row
|
||||
return
|
||||
}
|
||||
if overwrite {
|
||||
byHour[key] = row
|
||||
}
|
||||
}
|
||||
for _, row := range raw {
|
||||
add(row, false)
|
||||
}
|
||||
for _, row := range rollup {
|
||||
add(row, true)
|
||||
}
|
||||
result := make([]NodeMetricHourly, 0, len(order))
|
||||
// Keep chronological order of first-seen keys; re-sort by hour for stability.
|
||||
sort.Slice(order, func(i, j int) bool { return order[i] < order[j] })
|
||||
for _, key := range order {
|
||||
result = append(result, byHour[key])
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
func listNodeMetricHourlyFromRollup(ctx context.Context, filter NodeObservabilityFilter) ([]NodeMetricHourly, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
clause, args := buildNodeObservabilityFilterClause(filter, "hour")
|
||||
sql := fmt.Sprintf(`
|
||||
SELECT
|
||||
hour,
|
||||
if(sum(cpu_usage_count) > 0, sum(cpu_usage_sum) / sum(cpu_usage_count), 0) AS average_cpu_usage_percent,
|
||||
if(sum(memory_usage_count) > 0, sum(memory_usage_sum) / sum(memory_usage_count), 0) AS average_memory_usage_percent,
|
||||
sum(greatest(network_rx_max - network_rx_min, 0)) AS network_rx_bytes,
|
||||
sum(greatest(network_tx_max - network_tx_min, 0)) AS network_tx_bytes,
|
||||
sum(greatest(disk_read_max - disk_read_min, 0)) AS disk_read_bytes,
|
||||
sum(greatest(disk_write_max - disk_write_min, 0)) AS disk_write_bytes,
|
||||
toUInt64(uniqExact(node_id)) AS reported_nodes
|
||||
FROM %s
|
||||
WHERE %s
|
||||
GROUP BY hour
|
||||
ORDER BY hour ASC`, nodeMetricCapacityHourlyTableName(), clause)
|
||||
rows, err := conn.Query(ctx, sql, args...)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("list node metric hourly from rollup: %w", err)
|
||||
}
|
||||
defer func() { _ = rows.Close() }()
|
||||
return scanNodeMetricHourlyRows(rows)
|
||||
}
|
||||
|
||||
func listNodeMetricHourlyFromRaw(ctx context.Context, filter NodeObservabilityFilter) ([]NodeMetricHourly, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
clause, args := buildNodeObservabilityFilterClause(filter, "captured_at")
|
||||
tableName := nodeMetricSnapshotTableName()
|
||||
sql := fmt.Sprintf(`
|
||||
SELECT
|
||||
hour,
|
||||
avg(cpu_usage_percent) AS average_cpu_usage_percent,
|
||||
avg(memory_usage_percent) AS average_memory_usage_percent,
|
||||
sum(if(network_rx_delta >= 0, network_rx_delta, 0)) AS network_rx_bytes,
|
||||
sum(if(network_tx_delta >= 0, network_tx_delta, 0)) AS network_tx_bytes,
|
||||
sum(if(disk_read_delta >= 0, disk_read_delta, 0)) AS disk_read_bytes,
|
||||
sum(if(disk_write_delta >= 0, disk_write_delta, 0)) AS disk_write_bytes,
|
||||
toUInt64(uniqExact(node_id)) AS reported_nodes
|
||||
FROM (
|
||||
SELECT
|
||||
node_id,
|
||||
toStartOfHour(captured_at) AS hour,
|
||||
cpu_usage_percent,
|
||||
if(memory_total_bytes > 0, (memory_used_bytes * 100.0) / memory_total_bytes, 0) AS memory_usage_percent,
|
||||
network_rx_bytes - lagInFrame(network_rx_bytes, 1, network_rx_bytes) OVER (
|
||||
PARTITION BY node_id ORDER BY captured_at, id
|
||||
ROWS BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW
|
||||
) AS network_rx_delta,
|
||||
network_tx_bytes - lagInFrame(network_tx_bytes, 1, network_tx_bytes) OVER (
|
||||
PARTITION BY node_id ORDER BY captured_at, id
|
||||
ROWS BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW
|
||||
) AS network_tx_delta,
|
||||
disk_read_bytes - lagInFrame(disk_read_bytes, 1, disk_read_bytes) OVER (
|
||||
PARTITION BY node_id ORDER BY captured_at, id
|
||||
ROWS BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW
|
||||
) AS disk_read_delta,
|
||||
disk_write_bytes - lagInFrame(disk_write_bytes, 1, disk_write_bytes) OVER (
|
||||
PARTITION BY node_id ORDER BY captured_at, id
|
||||
ROWS BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW
|
||||
) AS disk_write_delta
|
||||
FROM %s
|
||||
WHERE %s
|
||||
)
|
||||
GROUP BY hour
|
||||
ORDER BY hour ASC`, tableName, clause)
|
||||
rows, err := conn.Query(ctx, sql, args...)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("list node metric hourly: %w", err)
|
||||
}
|
||||
defer func() { _ = rows.Close() }()
|
||||
return scanNodeMetricHourlyRows(rows)
|
||||
}
|
||||
|
||||
func scanNodeMetricHourlyRows(rows driver.Rows) ([]NodeMetricHourly, error) {
|
||||
result := make([]NodeMetricHourly, 0)
|
||||
for rows.Next() {
|
||||
var (
|
||||
item NodeMetricHourly
|
||||
reportedNodes uint64
|
||||
networkRx int64
|
||||
networkTx int64
|
||||
diskRead int64
|
||||
diskWrite int64
|
||||
)
|
||||
if err := rows.Scan(
|
||||
&item.Hour,
|
||||
&item.AverageCPUUsagePercent,
|
||||
&item.AverageMemoryUsagePercent,
|
||||
&networkRx,
|
||||
&networkTx,
|
||||
&diskRead,
|
||||
&diskWrite,
|
||||
&reportedNodes,
|
||||
); err != nil {
|
||||
return nil, fmt.Errorf("scan node metric hourly row: %w", err)
|
||||
}
|
||||
item.Hour = item.Hour.UTC()
|
||||
item.NetworkRxBytes = networkRx
|
||||
item.NetworkTxBytes = networkTx
|
||||
item.DiskReadBytes = diskRead
|
||||
item.DiskWriteBytes = diskWrite
|
||||
item.ReportedNodes = int(safeInt64Count(reportedNodes))
|
||||
result = append(result, item)
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// ListNodeOpenrestyHourly returns hourly OpenResty observation aggregates matching filter.
|
||||
// Same rollup-first / per-hour merge strategy as ListNodeMetricHourly.
|
||||
func ListNodeOpenrestyHourly(ctx context.Context, filter NodeObservabilityFilter) ([]NodeOpenrestyHourly, error) {
|
||||
rollup, rollupErr := listNodeOpenrestyHourlyFromRollup(ctx, filter)
|
||||
if rollupErr == nil && len(rollup) > 0 && hourlyRollupCoversWindow(rollup[0].Hour, filter.Since) {
|
||||
return rollup, nil
|
||||
}
|
||||
|
||||
raw, rawErr := listNodeOpenrestyHourlyFromRaw(ctx, filter)
|
||||
if rawErr != nil {
|
||||
if rollupErr == nil && len(rollup) > 0 {
|
||||
return rollup, nil
|
||||
}
|
||||
return nil, rawErr
|
||||
}
|
||||
if len(rollup) == 0 {
|
||||
return raw, nil
|
||||
}
|
||||
return mergeNodeOpenrestyHourlyPreferRollup(rollup, raw), nil
|
||||
}
|
||||
|
||||
func mergeNodeOpenrestyHourlyPreferRollup(rollup, raw []NodeOpenrestyHourly) []NodeOpenrestyHourly {
|
||||
byHour := make(map[int64]NodeOpenrestyHourly, len(raw)+len(rollup))
|
||||
order := make([]int64, 0, len(raw)+len(rollup))
|
||||
add := func(row NodeOpenrestyHourly, overwrite bool) {
|
||||
key := row.Hour.UTC().Truncate(time.Hour).Unix()
|
||||
if _, exists := byHour[key]; !exists {
|
||||
order = append(order, key)
|
||||
byHour[key] = row
|
||||
return
|
||||
}
|
||||
if overwrite {
|
||||
byHour[key] = row
|
||||
}
|
||||
}
|
||||
for _, row := range raw {
|
||||
add(row, false)
|
||||
}
|
||||
for _, row := range rollup {
|
||||
add(row, true)
|
||||
}
|
||||
sort.Slice(order, func(i, j int) bool { return order[i] < order[j] })
|
||||
result := make([]NodeOpenrestyHourly, 0, len(order))
|
||||
for _, key := range order {
|
||||
result = append(result, byHour[key])
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
func listNodeOpenrestyHourlyFromRollup(ctx context.Context, filter NodeObservabilityFilter) ([]NodeOpenrestyHourly, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
clause, args := buildNodeObservabilityFilterClause(filter, "hour")
|
||||
sql := fmt.Sprintf(`
|
||||
SELECT
|
||||
hour,
|
||||
sum(greatest(openresty_rx_max - openresty_rx_min, 0)) AS openresty_rx_bytes,
|
||||
sum(greatest(openresty_tx_max - openresty_tx_min, 0)) AS openresty_tx_bytes,
|
||||
toUInt64(uniqExact(node_id)) AS reported_nodes
|
||||
FROM %s
|
||||
WHERE %s
|
||||
GROUP BY hour
|
||||
ORDER BY hour ASC`, nodeOpenrestyHourlyTableName(), clause)
|
||||
rows, err := conn.Query(ctx, sql, args...)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("list node openresty hourly from rollup: %w", err)
|
||||
}
|
||||
defer func() { _ = rows.Close() }()
|
||||
return scanNodeOpenrestyHourlyRows(rows)
|
||||
}
|
||||
|
||||
func listNodeOpenrestyHourlyFromRaw(ctx context.Context, filter NodeObservabilityFilter) ([]NodeOpenrestyHourly, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
clause, args := buildNodeObservabilityFilterClause(filter, "captured_at")
|
||||
tableName := nodeObsOpenrestyTableName()
|
||||
sql := fmt.Sprintf(`
|
||||
SELECT
|
||||
hour,
|
||||
sum(if(openresty_rx_delta >= 0, openresty_rx_delta, 0)) AS openresty_rx_bytes,
|
||||
sum(if(openresty_tx_delta >= 0, openresty_tx_delta, 0)) AS openresty_tx_bytes,
|
||||
toUInt64(uniqExact(node_id)) AS reported_nodes
|
||||
FROM (
|
||||
SELECT
|
||||
node_id,
|
||||
toStartOfHour(captured_at) AS hour,
|
||||
openresty_rx_bytes - lagInFrame(openresty_rx_bytes, 1, openresty_rx_bytes) OVER (
|
||||
PARTITION BY node_id ORDER BY captured_at, id
|
||||
ROWS BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW
|
||||
) AS openresty_rx_delta,
|
||||
openresty_tx_bytes - lagInFrame(openresty_tx_bytes, 1, openresty_tx_bytes) OVER (
|
||||
PARTITION BY node_id ORDER BY captured_at, id
|
||||
ROWS BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW
|
||||
) AS openresty_tx_delta
|
||||
FROM %s
|
||||
WHERE %s
|
||||
)
|
||||
GROUP BY hour
|
||||
ORDER BY hour ASC`, tableName, clause)
|
||||
rows, err := conn.Query(ctx, sql, args...)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("list node openresty hourly: %w", err)
|
||||
}
|
||||
defer func() { _ = rows.Close() }()
|
||||
return scanNodeOpenrestyHourlyRows(rows)
|
||||
}
|
||||
|
||||
func scanNodeOpenrestyHourlyRows(rows driver.Rows) ([]NodeOpenrestyHourly, error) {
|
||||
result := make([]NodeOpenrestyHourly, 0)
|
||||
for rows.Next() {
|
||||
var (
|
||||
item NodeOpenrestyHourly
|
||||
reportedNodes uint64
|
||||
rx int64
|
||||
tx int64
|
||||
)
|
||||
if err := rows.Scan(&item.Hour, &rx, &tx, &reportedNodes); err != nil {
|
||||
return nil, fmt.Errorf("scan node openresty hourly row: %w", err)
|
||||
}
|
||||
item.Hour = item.Hour.UTC()
|
||||
item.OpenrestyRxBytes = rx
|
||||
item.OpenrestyTxBytes = tx
|
||||
item.ReportedNodes = int(safeInt64Count(reportedNodes))
|
||||
result = append(result, item)
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
func scanNodeObsFrpcRows(rows driver.Rows) ([]analyticsmodel.NodeObsFrpc, error) {
|
||||
var result []analyticsmodel.NodeObsFrpc
|
||||
for rows.Next() {
|
||||
|
||||
@@ -9,115 +9,212 @@ import (
|
||||
"time"
|
||||
)
|
||||
|
||||
// DeleteAllNodeMetricSnapshots deletes all node metric snapshots.
|
||||
// DeleteAllNodeMetricSnapshots hard-deletes all node metric snapshots via TRUNCATE.
|
||||
func DeleteAllNodeMetricSnapshots(ctx context.Context) (int64, error) {
|
||||
tableName := nodeMetricSnapshotTableName()
|
||||
return deleteNodeObservabilityWithCount(ctx, "SELECT count() FROM "+tableName, nil, "ALTER TABLE "+tableName+" DELETE WHERE 1")
|
||||
}
|
||||
|
||||
// DeleteNodeMetricSnapshotsBefore deletes metric snapshots captured before cutoff.
|
||||
func DeleteNodeMetricSnapshotsBefore(ctx context.Context, cutoff time.Time) (int64, error) {
|
||||
tableName := nodeMetricSnapshotTableName()
|
||||
cutoff = cutoff.UTC()
|
||||
return deleteNodeObservabilityWithCount(
|
||||
ctx,
|
||||
fmt.Sprintf("SELECT count() FROM %s WHERE captured_at < ?", tableName),
|
||||
[]any{cutoff},
|
||||
fmt.Sprintf("ALTER TABLE %s DELETE WHERE captured_at < ?", tableName),
|
||||
cutoff,
|
||||
)
|
||||
}
|
||||
|
||||
// DeleteAllNodeRequestReports deletes all node request reports.
|
||||
func DeleteAllNodeRequestReports(ctx context.Context) (int64, error) {
|
||||
tableName := nodeRequestReportTableName()
|
||||
return deleteNodeObservabilityWithCount(ctx, "SELECT count() FROM "+tableName, nil, "ALTER TABLE "+tableName+" DELETE WHERE 1")
|
||||
}
|
||||
|
||||
// DeleteNodeRequestReportsBefore deletes request reports ending before cutoff.
|
||||
func DeleteNodeRequestReportsBefore(ctx context.Context, cutoff time.Time) (int64, error) {
|
||||
tableName := nodeRequestReportTableName()
|
||||
cutoff = cutoff.UTC()
|
||||
return deleteNodeObservabilityWithCount(
|
||||
ctx,
|
||||
fmt.Sprintf("SELECT count() FROM %s WHERE window_ended_at < ?", tableName),
|
||||
[]any{cutoff},
|
||||
fmt.Sprintf("ALTER TABLE %s DELETE WHERE window_ended_at < ?", tableName),
|
||||
cutoff,
|
||||
)
|
||||
}
|
||||
|
||||
// DeleteAllNodeObsOpenresty deletes all OpenResty observations.
|
||||
func DeleteAllNodeObsOpenresty(ctx context.Context) (int64, error) {
|
||||
tableName := nodeObsOpenrestyTableName()
|
||||
return deleteNodeObservabilityWithCount(ctx, "SELECT count() FROM "+tableName, nil, "ALTER TABLE "+tableName+" DELETE WHERE 1")
|
||||
}
|
||||
|
||||
// DeleteNodeObsOpenrestyBefore deletes OpenResty observations captured before cutoff.
|
||||
func DeleteNodeObsOpenrestyBefore(ctx context.Context, cutoff time.Time) (int64, error) {
|
||||
tableName := nodeObsOpenrestyTableName()
|
||||
cutoff = cutoff.UTC()
|
||||
return deleteNodeObservabilityWithCount(
|
||||
ctx,
|
||||
fmt.Sprintf("SELECT count() FROM %s WHERE captured_at < ?", tableName),
|
||||
[]any{cutoff},
|
||||
fmt.Sprintf("ALTER TABLE %s DELETE WHERE captured_at < ?", tableName),
|
||||
cutoff,
|
||||
)
|
||||
}
|
||||
|
||||
// DeleteAllNodeObsFrps deletes all FRPS observations.
|
||||
func DeleteAllNodeObsFrps(ctx context.Context) (int64, error) {
|
||||
tableName := nodeObsFrpsTableName()
|
||||
return deleteNodeObservabilityWithCount(ctx, "SELECT count() FROM "+tableName, nil, "ALTER TABLE "+tableName+" DELETE WHERE 1")
|
||||
}
|
||||
|
||||
// DeleteNodeObsFrpsBefore deletes FRPS observations captured before cutoff.
|
||||
func DeleteNodeObsFrpsBefore(ctx context.Context, cutoff time.Time) (int64, error) {
|
||||
tableName := nodeObsFrpsTableName()
|
||||
cutoff = cutoff.UTC()
|
||||
return deleteNodeObservabilityWithCount(
|
||||
ctx,
|
||||
fmt.Sprintf("SELECT count() FROM %s WHERE captured_at < ?", tableName),
|
||||
[]any{cutoff},
|
||||
fmt.Sprintf("ALTER TABLE %s DELETE WHERE captured_at < ?", tableName),
|
||||
cutoff,
|
||||
)
|
||||
}
|
||||
|
||||
// DeleteAllNodeObsFrpc deletes all FRPC observations.
|
||||
func DeleteAllNodeObsFrpc(ctx context.Context) (int64, error) {
|
||||
tableName := nodeObsFrpcTableName()
|
||||
return deleteNodeObservabilityWithCount(ctx, "SELECT count() FROM "+tableName, nil, "ALTER TABLE "+tableName+" DELETE WHERE 1")
|
||||
}
|
||||
|
||||
// DeleteNodeObsFrpcBefore deletes FRPC observations captured before cutoff.
|
||||
func DeleteNodeObsFrpcBefore(ctx context.Context, cutoff time.Time) (int64, error) {
|
||||
tableName := nodeObsFrpcTableName()
|
||||
cutoff = cutoff.UTC()
|
||||
return deleteNodeObservabilityWithCount(
|
||||
ctx,
|
||||
fmt.Sprintf("SELECT count() FROM %s WHERE captured_at < ?", tableName),
|
||||
[]any{cutoff},
|
||||
fmt.Sprintf("ALTER TABLE %s DELETE WHERE captured_at < ?", tableName),
|
||||
cutoff,
|
||||
)
|
||||
}
|
||||
|
||||
func deleteNodeObservabilityWithCount(ctx context.Context, countSQL string, countArgs []any, deleteSQL string, deleteArgs ...any) (int64, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
var count uint64
|
||||
if err := conn.QueryRow(ctx, countSQL, countArgs...).Scan(&count); err != nil {
|
||||
return 0, fmt.Errorf("count node observability rows for delete: %w", err)
|
||||
outcome, err := truncateClickHouseTable(ctx, conn, nodeMetricSnapshotTableName())
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
if count == 0 {
|
||||
return 0, nil
|
||||
}
|
||||
if err := conn.Exec(ctx, deleteSQL, deleteArgs...); err != nil {
|
||||
return 0, fmt.Errorf("delete node observability rows: %w", err)
|
||||
}
|
||||
return safeInt64Count(count), nil
|
||||
return outcome.DeletedCount, nil
|
||||
}
|
||||
|
||||
// DeleteNodeMetricSnapshotsBefore force-materializes of_node_metric_snapshots table TTL.
|
||||
// cutoff is ignored; see MaterializeNodeMetricSnapshotsTTL.
|
||||
func DeleteNodeMetricSnapshotsBefore(ctx context.Context, _ time.Time) (int64, error) {
|
||||
outcome, err := MaterializeNodeMetricSnapshotsTTL(ctx)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.EligibleCount, nil
|
||||
}
|
||||
|
||||
// MaterializeNodeMetricSnapshotsTTL force-materializes table TTL and reports an honest outcome.
|
||||
func MaterializeNodeMetricSnapshotsTTL(ctx context.Context) (CleanupOutcome, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return CleanupOutcome{}, err
|
||||
}
|
||||
tableName := nodeMetricSnapshotTableName()
|
||||
ttlDays := TableTTLDaysNodeMetricSnapshots
|
||||
cutoff := tableTTLCutoff(ttlDays, time.Now())
|
||||
return materializeExpiredByTableTTL(
|
||||
ctx,
|
||||
conn,
|
||||
tableName,
|
||||
ttlDays,
|
||||
fmt.Sprintf("SELECT count() FROM %s WHERE captured_at < ?", tableName),
|
||||
[]any{cutoff},
|
||||
)
|
||||
}
|
||||
|
||||
// DeleteAllNodeRequestReports hard-deletes all node request reports via TRUNCATE.
|
||||
func DeleteAllNodeRequestReports(ctx context.Context) (int64, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
outcome, err := truncateClickHouseTable(ctx, conn, nodeRequestReportTableName())
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.DeletedCount, nil
|
||||
}
|
||||
|
||||
// DeleteNodeRequestReportsBefore force-materializes of_node_request_reports table TTL.
|
||||
// cutoff is ignored; see MaterializeNodeRequestReportsTTL.
|
||||
func DeleteNodeRequestReportsBefore(ctx context.Context, _ time.Time) (int64, error) {
|
||||
outcome, err := MaterializeNodeRequestReportsTTL(ctx)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.EligibleCount, nil
|
||||
}
|
||||
|
||||
// MaterializeNodeRequestReportsTTL force-materializes table TTL and reports an honest outcome.
|
||||
func MaterializeNodeRequestReportsTTL(ctx context.Context) (CleanupOutcome, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return CleanupOutcome{}, err
|
||||
}
|
||||
tableName := nodeRequestReportTableName()
|
||||
ttlDays := TableTTLDaysNodeRequestReports
|
||||
cutoff := tableTTLCutoff(ttlDays, time.Now())
|
||||
return materializeExpiredByTableTTL(
|
||||
ctx,
|
||||
conn,
|
||||
tableName,
|
||||
ttlDays,
|
||||
fmt.Sprintf("SELECT count() FROM %s WHERE window_ended_at < ?", tableName),
|
||||
[]any{cutoff},
|
||||
)
|
||||
}
|
||||
|
||||
// DeleteAllNodeObsOpenresty hard-deletes all OpenResty observations via TRUNCATE.
|
||||
func DeleteAllNodeObsOpenresty(ctx context.Context) (int64, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
outcome, err := truncateClickHouseTable(ctx, conn, nodeObsOpenrestyTableName())
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.DeletedCount, nil
|
||||
}
|
||||
|
||||
// DeleteNodeObsOpenrestyBefore force-materializes of_node_obs_openresty table TTL.
|
||||
// cutoff is ignored; see MaterializeNodeObsOpenrestyTTL.
|
||||
func DeleteNodeObsOpenrestyBefore(ctx context.Context, _ time.Time) (int64, error) {
|
||||
outcome, err := MaterializeNodeObsOpenrestyTTL(ctx)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.EligibleCount, nil
|
||||
}
|
||||
|
||||
// MaterializeNodeObsOpenrestyTTL force-materializes table TTL and reports an honest outcome.
|
||||
func MaterializeNodeObsOpenrestyTTL(ctx context.Context) (CleanupOutcome, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return CleanupOutcome{}, err
|
||||
}
|
||||
tableName := nodeObsOpenrestyTableName()
|
||||
ttlDays := TableTTLDaysNodeObs
|
||||
cutoff := tableTTLCutoff(ttlDays, time.Now())
|
||||
return materializeExpiredByTableTTL(
|
||||
ctx,
|
||||
conn,
|
||||
tableName,
|
||||
ttlDays,
|
||||
fmt.Sprintf("SELECT count() FROM %s WHERE captured_at < ?", tableName),
|
||||
[]any{cutoff},
|
||||
)
|
||||
}
|
||||
|
||||
// DeleteAllNodeObsFrps hard-deletes all FRPS observations via TRUNCATE.
|
||||
func DeleteAllNodeObsFrps(ctx context.Context) (int64, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
outcome, err := truncateClickHouseTable(ctx, conn, nodeObsFrpsTableName())
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.DeletedCount, nil
|
||||
}
|
||||
|
||||
// DeleteNodeObsFrpsBefore force-materializes of_node_obs_frps table TTL.
|
||||
// cutoff is ignored; see MaterializeNodeObsFrpsTTL.
|
||||
func DeleteNodeObsFrpsBefore(ctx context.Context, _ time.Time) (int64, error) {
|
||||
outcome, err := MaterializeNodeObsFrpsTTL(ctx)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.EligibleCount, nil
|
||||
}
|
||||
|
||||
// MaterializeNodeObsFrpsTTL force-materializes table TTL and reports an honest outcome.
|
||||
func MaterializeNodeObsFrpsTTL(ctx context.Context) (CleanupOutcome, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return CleanupOutcome{}, err
|
||||
}
|
||||
tableName := nodeObsFrpsTableName()
|
||||
ttlDays := TableTTLDaysNodeObs
|
||||
cutoff := tableTTLCutoff(ttlDays, time.Now())
|
||||
return materializeExpiredByTableTTL(
|
||||
ctx,
|
||||
conn,
|
||||
tableName,
|
||||
ttlDays,
|
||||
fmt.Sprintf("SELECT count() FROM %s WHERE captured_at < ?", tableName),
|
||||
[]any{cutoff},
|
||||
)
|
||||
}
|
||||
|
||||
// DeleteAllNodeObsFrpc hard-deletes all FRPC observations via TRUNCATE.
|
||||
func DeleteAllNodeObsFrpc(ctx context.Context) (int64, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
outcome, err := truncateClickHouseTable(ctx, conn, nodeObsFrpcTableName())
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.DeletedCount, nil
|
||||
}
|
||||
|
||||
// DeleteNodeObsFrpcBefore force-materializes of_node_obs_frpc table TTL.
|
||||
// cutoff is ignored; see MaterializeNodeObsFrpcTTL.
|
||||
func DeleteNodeObsFrpcBefore(ctx context.Context, _ time.Time) (int64, error) {
|
||||
outcome, err := MaterializeNodeObsFrpcTTL(ctx)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.EligibleCount, nil
|
||||
}
|
||||
|
||||
// MaterializeNodeObsFrpcTTL force-materializes table TTL and reports an honest outcome.
|
||||
func MaterializeNodeObsFrpcTTL(ctx context.Context) (CleanupOutcome, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return CleanupOutcome{}, err
|
||||
}
|
||||
tableName := nodeObsFrpcTableName()
|
||||
ttlDays := TableTTLDaysNodeObs
|
||||
cutoff := tableTTLCutoff(ttlDays, time.Now())
|
||||
return materializeExpiredByTableTTL(
|
||||
ctx,
|
||||
conn,
|
||||
tableName,
|
||||
ttlDays,
|
||||
fmt.Sprintf("SELECT count() FROM %s WHERE captured_at < ?", tableName),
|
||||
[]any{cutoff},
|
||||
)
|
||||
}
|
||||
|
||||
@@ -61,3 +61,14 @@ func nodeObsFrpsTableName() string {
|
||||
func nodeObsFrpcTableName() string {
|
||||
return "of_node_obs_frpc"
|
||||
}
|
||||
|
||||
func nodeMetricCapacityHourlyTableName() string {
|
||||
return "of_node_metric_capacity_hourly"
|
||||
}
|
||||
|
||||
func nodeOpenrestyHourlyTableName() string {
|
||||
return "of_node_openresty_hourly"
|
||||
}
|
||||
|
||||
// clickHouseLimit1ByNodeIDClause selects the first row per node_id after ORDER BY.
|
||||
const clickHouseLimit1ByNodeIDClause = " LIMIT 1 BY node_id"
|
||||
|
||||
@@ -0,0 +1,225 @@
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
package analytics
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/ClickHouse/clickhouse-go/v2/lib/driver"
|
||||
"github.com/Rain-kl/Wavelet/internal/db"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func TestListLatestNodeMetricSnapshots_UsesLimit1ByNodeID(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
mock := &mockConn{}
|
||||
db.SetChConnForTest(mock)
|
||||
t.Cleanup(func() { db.SetChConnForTest(nil) })
|
||||
|
||||
since := time.Date(2026, 7, 10, 0, 0, 0, 0, time.UTC)
|
||||
_, err := ListLatestNodeMetricSnapshots(ctx, NodeObservabilityFilter{Since: since})
|
||||
require.NoError(t, err)
|
||||
require.Len(t, mock.queries, 1)
|
||||
assert.Contains(t, mock.queries[0], "LIMIT 1 BY node_id")
|
||||
assert.Contains(t, mock.queries[0], nodeMetricSnapshotTableName())
|
||||
assert.Contains(t, mock.queries[0], "captured_at DESC")
|
||||
assert.NotContains(t, mock.queries[0], "LIMIT ?")
|
||||
require.Len(t, mock.queryArgs, 1)
|
||||
require.Len(t, mock.queryArgs[0], 1)
|
||||
assert.Equal(t, since, mock.queryArgs[0][0])
|
||||
}
|
||||
|
||||
func TestListLatestNodeRequestReports_UsesLimit1ByNodeID(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
mock := &mockConn{}
|
||||
db.SetChConnForTest(mock)
|
||||
t.Cleanup(func() { db.SetChConnForTest(nil) })
|
||||
|
||||
_, err := ListLatestNodeRequestReports(ctx, NodeObservabilityFilter{NodeID: "node-a"})
|
||||
require.NoError(t, err)
|
||||
require.Len(t, mock.queries, 1)
|
||||
assert.Contains(t, mock.queries[0], "LIMIT 1 BY node_id")
|
||||
assert.Contains(t, mock.queries[0], nodeRequestReportTableName())
|
||||
assert.Contains(t, mock.queries[0], "window_ended_at DESC")
|
||||
require.Len(t, mock.queryArgs, 1)
|
||||
require.Len(t, mock.queryArgs[0], 1)
|
||||
assert.Equal(t, "node-a", mock.queryArgs[0][0])
|
||||
}
|
||||
|
||||
func TestListNodeMetricHourly_PrefersRollup(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
hour := time.Date(2026, 7, 10, 12, 0, 0, 0, time.UTC)
|
||||
since := hour.Add(-1 * time.Hour)
|
||||
mock := &mockConn{
|
||||
queryFn: func(_ context.Context, query string, _ ...any) (driver.Rows, error) {
|
||||
if strings.Contains(query, nodeMetricCapacityHourlyTableName()) {
|
||||
return &mockRows{data: [][]any{{
|
||||
hour, 42.5, 60.0, int64(100), int64(200), int64(10), int64(20), uint64(2),
|
||||
}}}, nil
|
||||
}
|
||||
return nil, errors.New("raw path should not be used when rollup covers the window")
|
||||
},
|
||||
}
|
||||
db.SetChConnForTest(mock)
|
||||
t.Cleanup(func() { db.SetChConnForTest(nil) })
|
||||
|
||||
rows, err := ListNodeMetricHourly(ctx, NodeObservabilityFilter{Since: since})
|
||||
require.NoError(t, err)
|
||||
require.Len(t, rows, 1)
|
||||
assert.Equal(t, 42.5, rows[0].AverageCPUUsagePercent)
|
||||
assert.Equal(t, 60.0, rows[0].AverageMemoryUsagePercent)
|
||||
assert.Equal(t, int64(100), rows[0].NetworkRxBytes)
|
||||
assert.Equal(t, 2, rows[0].ReportedNodes)
|
||||
require.Len(t, mock.queries, 1)
|
||||
assert.Contains(t, mock.queries[0], nodeMetricCapacityHourlyTableName())
|
||||
}
|
||||
|
||||
func TestListNodeMetricHourly_MergesRawGapsWithPartialRollup(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
// 24h window starts far before the only rollup bucket (last hour).
|
||||
since := time.Date(2026, 7, 9, 12, 0, 0, 0, time.UTC)
|
||||
rollupHour := time.Date(2026, 7, 10, 12, 0, 0, 0, time.UTC)
|
||||
rawHour := time.Date(2026, 7, 9, 15, 0, 0, 0, time.UTC)
|
||||
mock := &mockConn{
|
||||
queryFn: func(_ context.Context, query string, _ ...any) (driver.Rows, error) {
|
||||
if strings.Contains(query, nodeMetricCapacityHourlyTableName()) {
|
||||
return &mockRows{data: [][]any{{
|
||||
rollupHour, 99.0, 99.0, int64(1), int64(1), int64(1), int64(1), uint64(1),
|
||||
}}}, nil
|
||||
}
|
||||
if strings.Contains(query, nodeMetricSnapshotTableName()) {
|
||||
return &mockRows{data: [][]any{
|
||||
{rawHour, 12.0, 34.0, int64(5), int64(6), int64(7), int64(8), uint64(1)},
|
||||
{rollupHour, 50.0, 50.0, int64(9), int64(9), int64(9), int64(9), uint64(1)},
|
||||
}}, nil
|
||||
}
|
||||
return &mockRows{}, nil
|
||||
},
|
||||
}
|
||||
db.SetChConnForTest(mock)
|
||||
t.Cleanup(func() { db.SetChConnForTest(nil) })
|
||||
|
||||
rows, err := ListNodeMetricHourly(ctx, NodeObservabilityFilter{Since: since})
|
||||
require.NoError(t, err)
|
||||
require.Len(t, rows, 2)
|
||||
assert.Equal(t, rawHour, rows[0].Hour)
|
||||
assert.Equal(t, 12.0, rows[0].AverageCPUUsagePercent)
|
||||
// Overlapping hour prefers rollup (99) over raw (50).
|
||||
assert.Equal(t, rollupHour, rows[1].Hour)
|
||||
assert.Equal(t, 99.0, rows[1].AverageCPUUsagePercent)
|
||||
require.GreaterOrEqual(t, len(mock.queries), 2)
|
||||
assert.Contains(t, mock.queries[1], "lagInFrame")
|
||||
}
|
||||
|
||||
func TestMergeNodeMetricHourlyPreferRollup(t *testing.T) {
|
||||
h1 := time.Date(2026, 7, 10, 10, 0, 0, 0, time.UTC)
|
||||
h2 := time.Date(2026, 7, 10, 11, 0, 0, 0, time.UTC)
|
||||
merged := mergeNodeMetricHourlyPreferRollup(
|
||||
[]NodeMetricHourly{{Hour: h2, AverageCPUUsagePercent: 80}},
|
||||
[]NodeMetricHourly{
|
||||
{Hour: h1, AverageCPUUsagePercent: 10},
|
||||
{Hour: h2, AverageCPUUsagePercent: 20},
|
||||
},
|
||||
)
|
||||
require.Len(t, merged, 2)
|
||||
assert.Equal(t, h1, merged[0].Hour)
|
||||
assert.Equal(t, 10.0, merged[0].AverageCPUUsagePercent)
|
||||
assert.Equal(t, h2, merged[1].Hour)
|
||||
assert.Equal(t, 80.0, merged[1].AverageCPUUsagePercent)
|
||||
}
|
||||
|
||||
func TestHourlyRollupCoversWindow(t *testing.T) {
|
||||
since := time.Date(2026, 7, 10, 0, 0, 0, 0, time.UTC)
|
||||
assert.True(t, hourlyRollupCoversWindow(since, since))
|
||||
assert.True(t, hourlyRollupCoversWindow(since.Add(2*time.Hour), since))
|
||||
assert.False(t, hourlyRollupCoversWindow(since.Add(3*time.Hour), since))
|
||||
assert.True(t, hourlyRollupCoversWindow(time.Date(2026, 7, 11, 0, 0, 0, 0, time.UTC), time.Time{}))
|
||||
}
|
||||
|
||||
func TestListNodeMetricHourly_FallsBackToRawOnRollupError(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
hour := time.Date(2026, 7, 10, 13, 0, 0, 0, time.UTC)
|
||||
mock := &mockConn{
|
||||
queryFn: func(_ context.Context, query string, _ ...any) (driver.Rows, error) {
|
||||
if strings.Contains(query, nodeMetricCapacityHourlyTableName()) {
|
||||
return nil, errors.New("rollup missing")
|
||||
}
|
||||
if strings.Contains(query, nodeMetricSnapshotTableName()) {
|
||||
return &mockRows{data: [][]any{{
|
||||
hour, 10.0, 20.0, int64(1), int64(2), int64(3), int64(4), uint64(1),
|
||||
}}}, nil
|
||||
}
|
||||
return &mockRows{}, nil
|
||||
},
|
||||
}
|
||||
db.SetChConnForTest(mock)
|
||||
t.Cleanup(func() { db.SetChConnForTest(nil) })
|
||||
|
||||
rows, err := ListNodeMetricHourly(ctx, NodeObservabilityFilter{})
|
||||
require.NoError(t, err)
|
||||
require.Len(t, rows, 1)
|
||||
assert.Equal(t, 10.0, rows[0].AverageCPUUsagePercent)
|
||||
assert.Equal(t, int64(3), rows[0].DiskReadBytes)
|
||||
require.GreaterOrEqual(t, len(mock.queries), 2)
|
||||
assert.Contains(t, mock.queries[0], nodeMetricCapacityHourlyTableName())
|
||||
assert.Contains(t, mock.queries[1], "lagInFrame")
|
||||
}
|
||||
|
||||
func TestListNodeOpenrestyHourly_PrefersRollup(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
hour := time.Date(2026, 7, 10, 14, 0, 0, 0, time.UTC)
|
||||
since := hour.Add(-1 * time.Hour)
|
||||
mock := &mockConn{
|
||||
queryFn: func(_ context.Context, query string, _ ...any) (driver.Rows, error) {
|
||||
if strings.Contains(query, nodeOpenrestyHourlyTableName()) {
|
||||
return &mockRows{data: [][]any{{
|
||||
hour, int64(50), int64(70), uint64(3),
|
||||
}}}, nil
|
||||
}
|
||||
return nil, errors.New("raw path should not be used when rollup covers the window")
|
||||
},
|
||||
}
|
||||
db.SetChConnForTest(mock)
|
||||
t.Cleanup(func() { db.SetChConnForTest(nil) })
|
||||
|
||||
rows, err := ListNodeOpenrestyHourly(ctx, NodeObservabilityFilter{Since: since})
|
||||
require.NoError(t, err)
|
||||
require.Len(t, rows, 1)
|
||||
assert.Equal(t, int64(50), rows[0].OpenrestyRxBytes)
|
||||
assert.Equal(t, int64(70), rows[0].OpenrestyTxBytes)
|
||||
assert.Equal(t, 3, rows[0].ReportedNodes)
|
||||
}
|
||||
|
||||
func TestListNodeOpenrestyHourly_FallsBackToRawOnEmptyRollup(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
hour := time.Date(2026, 7, 10, 15, 0, 0, 0, time.UTC)
|
||||
mock := &mockConn{
|
||||
queryFn: func(_ context.Context, query string, _ ...any) (driver.Rows, error) {
|
||||
if strings.Contains(query, nodeOpenrestyHourlyTableName()) {
|
||||
return &mockRows{}, nil
|
||||
}
|
||||
if strings.Contains(query, nodeObsOpenrestyTableName()) {
|
||||
return &mockRows{data: [][]any{{
|
||||
hour, int64(9), int64(8), uint64(1),
|
||||
}}}, nil
|
||||
}
|
||||
return &mockRows{}, nil
|
||||
},
|
||||
}
|
||||
db.SetChConnForTest(mock)
|
||||
t.Cleanup(func() { db.SetChConnForTest(nil) })
|
||||
|
||||
rows, err := ListNodeOpenrestyHourly(ctx, NodeObservabilityFilter{})
|
||||
require.NoError(t, err)
|
||||
require.Len(t, rows, 1)
|
||||
assert.Equal(t, int64(9), rows[0].OpenrestyRxBytes)
|
||||
require.GreaterOrEqual(t, len(mock.queries), 2)
|
||||
assert.Contains(t, mock.queries[1], "lagInFrame")
|
||||
}
|
||||
|
||||
@@ -51,6 +51,7 @@ func RegisterAdminRoutes(apiV1Router *gin.RouterGroup) {
|
||||
func registerAdminDiagnosticRoutes(adminRouter *gin.RouterGroup) {
|
||||
// System status
|
||||
adminRouter.GET("/status", admin_status.GetSystemStatus)
|
||||
adminRouter.GET("/status/clickhouse", admin_status.GetClickHouseStatus)
|
||||
|
||||
// Database basic info & backup export
|
||||
adminRouter.GET("/db-info", admin_status.GetDatabaseInfo)
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user