mirror of
https://github.com/Rain-kl/OpenFlare.git
synced 2026-09-28 21:56:36 +08:00
Compare commits
13 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 13c5073bf8 | |||
| da1dd92404 | |||
| 4b11279662 | |||
| bbadcca294 | |||
| 44ce6497a1 | |||
| 4b83f91b31 | |||
| 9d2fac5d4c | |||
| b4b93ff4ed | |||
| 160e63558f | |||
| 9b3555c569 | |||
| b928928958 | |||
| b312460ddf | |||
| 50f7257d93 |
+18
-4
@@ -1,7 +1,8 @@
|
||||
# ──────────────────────────────────────────────────────────────────────────────
|
||||
# openflare — 环境变量配置模板
|
||||
# 复制此文件为 .env 并填入实际值: cp .env.example .env
|
||||
# 环境变量优先级高于 config.yaml / config.docker.yaml
|
||||
# 环境变量优先级高于 config.yaml
|
||||
# docker compose 会读取本文件(env_file: .env)并替换 compose 中的 ${VAR}
|
||||
# ──────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
# ─── 时区 ─────────────────────────────────────────────────────────────────────
|
||||
@@ -22,11 +23,12 @@ APP_SESSION_HTTP_ONLY=true
|
||||
# HTTPS 部署时设为 true,HTTP 环境必须为 false
|
||||
APP_SESSION_SECURE=true
|
||||
|
||||
# ─── 数据库 ────────────────────────────────────────────────────────────────────
|
||||
# ─── 数据库(PostgreSQL)──────────────────────────────────────────────────────
|
||||
# 设置 DB_HOST 后自动启用 PostgreSQL,也可通过 DB_ENABLED 显式控制
|
||||
# DB_ENABLED=false 时使用 SQLite 作为后备数据库
|
||||
DB_ENABLED=true
|
||||
# SQLITE_PATH=./data/openflare.db
|
||||
# compose 内应用连服务名;本机直连 Docker 映射端口时用 127.0.0.1
|
||||
DB_HOST=postgres
|
||||
DB_PORT=5432
|
||||
DB_USERNAME=openflare
|
||||
@@ -38,7 +40,7 @@ DB_TIMEZONE=Asia/Shanghai
|
||||
# DB_MAX_IDLE_CONN=16
|
||||
# DB_MAX_OPEN_CONN=128
|
||||
|
||||
# ─── Redis ─────────────────────────────────────────────────────────────────────
|
||||
# ─── Redis / Valkey ────────────────────────────────────────────────────────────
|
||||
# 设置 REDIS_ADDR 后自动启用,也可通过 REDIS_ENABLED 显式控制
|
||||
REDIS_ENABLED=true
|
||||
REDIS_ADDR=redis:6379
|
||||
@@ -47,13 +49,20 @@ REDIS_ADDR=redis:6379
|
||||
# REDIS_DB=0
|
||||
REDIS_KEY_PREFIX=openflare:
|
||||
# REDIS_POOL_SIZE=100
|
||||
# compose 宿主机映射端口(仅 docker-compose 使用)
|
||||
# REDIS_PORT=6379
|
||||
|
||||
# ─── ClickHouse(必需)────────────────────────────────────────────────────
|
||||
# ─── ClickHouse(必需)────────────────────────────────────────────────────────
|
||||
# CLICKHOUSE_HOST 设置后会自动启用;测试环境可显式 CLICKHOUSE_ENABLED=true 做 live 联调
|
||||
CLICKHOUSE_ENABLED=true
|
||||
# compose 内:clickhouse:9000;本机连映射端口:127.0.0.1:9000
|
||||
CLICKHOUSE_HOST=clickhouse:9000
|
||||
CLICKHOUSE_USERNAME=default
|
||||
# 须与 compose clickhouse 服务密码一致(首次初始化后改密码需清 data/clickhouse_data)
|
||||
CLICKHOUSE_PASSWORD=replace-with-clickhouse-password
|
||||
CLICKHOUSE_NAME=openflare
|
||||
|
||||
|
||||
# ─── 日志 ──────────────────────────────────────────────────────────────────────
|
||||
LOG_LEVEL=info
|
||||
LOG_FORMAT=console
|
||||
@@ -67,6 +76,11 @@ OTEL_EXPORTER_OTLP_INSECURE=true
|
||||
OTEL_SAMPLING_RATE=0.0
|
||||
# 全局 Tracer 命名空间,默认为 github.com/Rain-kl/OpenFlare
|
||||
# OTEL_TRACER_NAME=github.com/Rain-kl/OpenFlare
|
||||
# compose 可选端口覆盖
|
||||
# JAEGER_VERSION=2.19.0
|
||||
# JAEGER_UI_PORT=16686
|
||||
# JAEGER_OTLP_GRPC_PORT=4317
|
||||
# JAEGER_OTLP_HTTP_PORT=4318
|
||||
|
||||
# ─── Worker ────────────────────────────────────────────────────────────────────
|
||||
# WORKER_CONCURRENCY=20
|
||||
|
||||
@@ -48,6 +48,20 @@ OpenFlare 是开源 CDN 编排与边缘安全平台。它支持反向代理、
|
||||
* **SSO 单点登录**:支持 GitHub OAuth 与标准 OIDC 协议,无缝接入企业身份提供商实现统一登录。
|
||||
* **统一观测**:聚合节点请求指标、实时访问日志明细、宿主机与 Nginx 资源快照、健康事件以及网络波动补传缓冲。
|
||||
|
||||
## 界面预览
|
||||
|
||||
### 仪表盘总览
|
||||
|
||||

|
||||
|
||||
### 节点详情
|
||||
|
||||

|
||||
|
||||
### 配置新增
|
||||
|
||||

|
||||
|
||||
## 快速开始
|
||||
|
||||
### 1. 启动 Server
|
||||
@@ -58,6 +72,11 @@ OpenFlare 是开源 CDN 编排与边缘安全平台。它支持反向代理、
|
||||
# 下载环境变量模板并创建 .env 文件
|
||||
curl -o .env.example https://raw.githubusercontent.com/Rain-kl/OpenFlare/refs/heads/main/.env.example
|
||||
cp .env.example .env
|
||||
|
||||
# ClickHouse 服务端:curl performance.xml 到 ./config/clickhouse,整目录挂载到 config.d(不要放 listen 配置)
|
||||
mkdir -p ./config/clickhouse
|
||||
curl -fsSL -o ./config/clickhouse/performance.xml \
|
||||
https://raw.githubusercontent.com/Rain-kl/OpenFlare/refs/heads/main/config/clickhouse/performance.xml
|
||||
```
|
||||
|
||||
```yaml
|
||||
@@ -117,15 +136,20 @@ services:
|
||||
CLICKHOUSE_PASSWORD: ${CLICKHOUSE_PASSWORD:-replace-with-clickhouse-password}
|
||||
CLICKHOUSE_DEFAULT_ACCESS_MANAGEMENT: 1
|
||||
TZ: ${TZ:-Asia/Shanghai}
|
||||
ulimits:
|
||||
nofile:
|
||||
soft: 262144
|
||||
hard: 262144
|
||||
volumes:
|
||||
- openflare_clickhouse_data:/var/lib/clickhouse
|
||||
- ./config/clickhouse:/etc/clickhouse-server/config.d:ro
|
||||
healthcheck:
|
||||
test: ["CMD", "clickhouse-client", "--user", "${CLICKHOUSE_USERNAME:-default}", "--password", "${CLICKHOUSE_PASSWORD:-replace-with-clickhouse-password}", "--query", "SELECT 1"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
start_period: 15s
|
||||
|
||||
|
||||
volumes:
|
||||
openflare_uploads:
|
||||
openflare_postgres_data:
|
||||
@@ -162,20 +186,6 @@ docker run -d --name openflare-agent --restart unless-stopped \
|
||||
ghcr.io/rain-kl/openflare-agent:latest
|
||||
```
|
||||
|
||||
## 界面预览
|
||||
|
||||
### 仪表盘总览
|
||||
|
||||

|
||||
|
||||
### 节点详情
|
||||
|
||||

|
||||
|
||||
### 配置新增
|
||||
|
||||

|
||||
|
||||
## 开源协议
|
||||
|
||||
本项目采用 [Apache License 2.0](./LICENSE) 开源。
|
||||
|
||||
+8
-5
@@ -99,15 +99,18 @@ otel:
|
||||
|
||||
|
||||
# ─── ClickHouse (required) ──────────────────────────────────────────────────────
|
||||
# Analytics / observability OLAP store. Telemetry writes are best-effort (async batch).
|
||||
clickhouse:
|
||||
enabled: true
|
||||
hosts:
|
||||
- "127.0.0.1:9000"
|
||||
- "127.0.0.1:9000" # compose 内应用可用 clickhouse:9000(经 CLICKHOUSE_HOST)
|
||||
username: "default"
|
||||
password: "123456"
|
||||
password: "replace-with-clickhouse-password" # 与 .env / compose CLICKHOUSE_PASSWORD 一致
|
||||
database: "openflare"
|
||||
max_idle_conn: 20
|
||||
max_open_conn: 50
|
||||
max_idle_conn: 8 # keep warm sockets low to save client + server RAM
|
||||
max_open_conn: 16 # cap concurrent native sessions on modest CH boxes
|
||||
conn_max_lifetime: 3600
|
||||
dial_timeout: 5
|
||||
block_buffer_size: 100
|
||||
block_buffer_size: 32 # rows buffered per block; 32 is enough for our batch sizes
|
||||
# Runtime client also enables async_insert (wait_for_async_insert=1, busy_timeout≈2s)
|
||||
# in internal/db/clickhouse.go — not configured via YAML.
|
||||
|
||||
@@ -0,0 +1,25 @@
|
||||
<?xml version="1.0"?>
|
||||
<!--
|
||||
Tuned for small control-plane hosts (e.g. 3c6g).
|
||||
|
||||
background_pool_size * background_merges_mutations_concurrency_ratio must stay
|
||||
greater than merge_tree number_of_free_entries_in_pool_to_execute_mutation
|
||||
(ClickHouse 25.x refuses to start otherwise). Keep the merge free-entry
|
||||
thresholds low so a small pool remains valid.
|
||||
-->
|
||||
<clickhouse>
|
||||
<max_concurrent_queries>20</max_concurrent_queries>
|
||||
<background_pool_size>4</background_pool_size>
|
||||
<background_merges_mutations_concurrency_ratio>2</background_merges_mutations_concurrency_ratio>
|
||||
<background_schedule_pool_size>4</background_schedule_pool_size>
|
||||
<background_common_pool_size>2</background_common_pool_size>
|
||||
<background_fetches_pool_size>2</background_fetches_pool_size>
|
||||
<background_move_pool_size>1</background_move_pool_size>
|
||||
<mark_cache_size>268435456</mark_cache_size>
|
||||
<uncompressed_cache_size>0</uncompressed_cache_size>
|
||||
<merge_tree>
|
||||
<number_of_free_entries_in_pool_to_execute_mutation>2</number_of_free_entries_in_pool_to_execute_mutation>
|
||||
<number_of_free_entries_in_pool_to_lower_max_size_of_merge>2</number_of_free_entries_in_pool_to_lower_max_size_of_merge>
|
||||
<number_of_free_entries_in_pool_to_execute_optimize_entire_partition>2</number_of_free_entries_in_pool_to_execute_optimize_entire_partition>
|
||||
</merge_tree>
|
||||
</clickhouse>
|
||||
+3
-3
@@ -84,11 +84,11 @@ services:
|
||||
soft: 262144
|
||||
hard: 262144
|
||||
ports:
|
||||
- "${CLICKHOUSE_HTTP_PORT:-8123}:8123"
|
||||
- "${CLICKHOUSE_NATIVE_PORT:-9000}:9000"
|
||||
- "8123:8123"
|
||||
- "9000:9000"
|
||||
volumes:
|
||||
- ./data/clickhouse_data:/var/lib/clickhouse
|
||||
- ./docker/clickhouse/config.d:/etc/clickhouse-server/config.d
|
||||
- ./config/clickhouse:/etc/clickhouse-server/config.d:ro
|
||||
healthcheck:
|
||||
test: ["CMD", "clickhouse-client", "--user", "${CLICKHOUSE_USERNAME:-default}", "--password", "${CLICKHOUSE_PASSWORD:-replace-with-clickhouse-password}", "--query", "SELECT 1"]
|
||||
interval: 10s
|
||||
|
||||
@@ -1,6 +0,0 @@
|
||||
<?xml version="1.0"?>
|
||||
<clickhouse>
|
||||
<max_concurrent_queries>50</max_concurrent_queries>
|
||||
<background_pool_size>8</background_pool_size>
|
||||
<background_merges_mutations_concurrency_ratio>2</background_merges_mutations_concurrency_ratio>
|
||||
</clickhouse>
|
||||
@@ -11,6 +11,9 @@ sidebar: false
|
||||
## 重大变更
|
||||
|
||||
> [!IMPORTANT]
|
||||
>
|
||||
> 3.1.2 版本更新了 CLickHouse 部署配置。
|
||||
>
|
||||
> 3.0.0 版本为 Wavelet 平台迁移与架构重构版本,涉及数据库表结构、环境变量以及前后端底层架构的重大变更。请务必在升级前备份数据库,并且更新到 V2.3.4。
|
||||
> 目前已知的兼容性问题:
|
||||
> - Pages 无法迁移, 升级前请先手动下载并备份 Pages 静态站点的 ZIP 包,升级后重新创建。
|
||||
@@ -18,6 +21,29 @@ sidebar: false
|
||||
|
||||
## [unreleased]
|
||||
|
||||
## [v3.1.2] - 2026-07-10
|
||||
|
||||
### 修复
|
||||
|
||||
- 修复节点/仪表盘 24 小时容量、网络、磁盘 IO 趋势在 ClickHouse 限流查询下几乎为空的问题:改为基于小时级聚合与计数器 delta 统计,避免仅依赖最近有限条原始快照导致历史时段全空。
|
||||
- 降低静置时 ClickHouse CPU:可观测/访问日志 batchwriter 启用 `MinBatchSize` 与 `MaxFlushWait`,减少心跳小 part 写入;Docker `performance.xml` 收紧小规格后台 merge 池。
|
||||
- ClickHouse 清理语义:按保留天数仅 `MATERIALIZE` 表 DDL TTL,`deleted_count` 不再伪报删除;短于表 TTL 的保留请求被拒绝。
|
||||
- 可观测 dedup 仅在入队成功后保留,flush 失败释放键并短重试;审计 writer 增加 `MaxFlushWait`;`/admin/status/clickhouse` 暴露 batch writer 队列深度/丢弃/flush 错误。
|
||||
- model 层通过 hooks 写入 CH,去除对 `chwriter` 的直接依赖。
|
||||
- Dashboard 每节点最新指标改为 `LIMIT 1 BY node_id`;新增 metric/openresty 小时预聚合表;读路径按小时 merge(rollup 窗口完整时仅走预聚合,不足时用 raw 补洞),并提供历史 backfill 迁移。
|
||||
- 小规格默认连接池下调;`async_insert_busy_timeout` 调至 2s;`of_node_traffic_hourly` 增加 30 天 TTL,UV 改为峰值窗口估计并修正前端文案。
|
||||
- Docker ClickHouse:`performance.xml` 下调 merge free-entry 阈值以兼容小 `background_pool`(避免 25.x 启动 Code 36)。
|
||||
|
||||
### 文档
|
||||
|
||||
- 同步 `.env.example` 与 `config.example.yaml`;ClickHouse 服务端配置改为 curl `performance.xml` 到 `./config/clickhouse` 后整目录挂载至 `config.d`(不要放入 listen 配置)。
|
||||
|
||||
## [v3.1.1] - 2026-07-06
|
||||
|
||||
### 修改
|
||||
|
||||
- 将 `cap_login_enabled` 默认值由 `true` 变更为 `false`,默认关闭登录界面 PoW 人机验证。
|
||||
|
||||
## [v3.1.0] - 2026-07-04
|
||||
|
||||
### 修改
|
||||
|
||||
@@ -8,6 +8,30 @@ OpenFlare Server 是 Gin + GORM 单体控制面,负责管理端 UI、管理 AP
|
||||
> **关于外部依赖**:
|
||||
> OpenFlare 系统内建了对后台异步任务(Asynq 框架)及海量节点日志分析与度量指标(观测面板)的支持。因此,**无论采用何种部署模式,系统都必须依赖 Redis(或 Valkey)与 ClickHouse 的运行**。各个部署方案的主要差异在于主关系型数据库的选择(SQLite vs PostgreSQL)以及是否启用链路追踪服务(Jaeger)。
|
||||
|
||||
> [!TIP]
|
||||
> **ClickHouse 服务端性能配置(推荐挂载)**
|
||||
> 控制面常见为小规格主机(如 3c6g)。仓库提供的 `performance.xml` 会收紧后台 merge/mutation 线程池,避免默认配置在小机器上静置 CPU 偏高或 ClickHouse 25.x 启动校验失败。
|
||||
> 将本地目录 `./config/clickhouse` 挂载到容器 `/etc/clickhouse-server/config.d`。
|
||||
> **目录内只放 `performance.xml`,不要放入任何 listen 相关配置**(监听地址沿用官方镜像默认即可)。
|
||||
|
||||
部署前将配置拉到本地:
|
||||
|
||||
```bash
|
||||
mkdir -p ./config/clickhouse
|
||||
curl -fsSL -o ./config/clickhouse/performance.xml \
|
||||
https://raw.githubusercontent.com/Rain-kl/OpenFlare/refs/heads/main/config/clickhouse/performance.xml
|
||||
```
|
||||
|
||||
在 ClickHouse 服务的 `volumes` 中增加(与数据卷并列):
|
||||
|
||||
```yaml
|
||||
volumes:
|
||||
- ./data/clickhouse_data:/var/lib/clickhouse # 或 named volume
|
||||
- ./config/clickhouse:/etc/clickhouse-server/config.d:ro
|
||||
```
|
||||
|
||||
修改 `performance.xml` 后需 `docker compose restart clickhouse` 才生效。
|
||||
|
||||
---
|
||||
|
||||
## 方式一:Docker 部署 (推荐)
|
||||
@@ -71,10 +95,15 @@ services:
|
||||
CLICKHOUSE_PASSWORD: 123456
|
||||
CLICKHOUSE_DEFAULT_ACCESS_MANAGEMENT: 1
|
||||
TZ: Asia/Shanghai
|
||||
ulimits:
|
||||
nofile:
|
||||
soft: 262144
|
||||
hard: 262144
|
||||
volumes:
|
||||
- ./data/clickhouse_data:/var/lib/clickhouse
|
||||
- ./config/clickhouse:/etc/clickhouse-server/config.d:ro
|
||||
healthcheck:
|
||||
test: ["CMD", "clickhouse-client", "--query", "SELECT 1"]
|
||||
test: ["CMD", "clickhouse-client", "--user", "default", "--password", "123456", "--query", "SELECT 1"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
@@ -84,6 +113,9 @@ services:
|
||||
运行启动命令:
|
||||
|
||||
```bash
|
||||
mkdir -p ./config/clickhouse
|
||||
curl -fsSL -o ./config/clickhouse/performance.xml \
|
||||
https://raw.githubusercontent.com/Rain-kl/OpenFlare/refs/heads/main/config/clickhouse/performance.xml
|
||||
docker compose up -d
|
||||
```
|
||||
|
||||
@@ -154,8 +186,13 @@ services:
|
||||
CLICKHOUSE_PASSWORD: ${CLICKHOUSE_PASSWORD:-replace-with-clickhouse-password}
|
||||
CLICKHOUSE_DEFAULT_ACCESS_MANAGEMENT: 1
|
||||
TZ: ${TZ:-Asia/Shanghai}
|
||||
ulimits:
|
||||
nofile:
|
||||
soft: 262144
|
||||
hard: 262144
|
||||
volumes:
|
||||
- openflare_clickhouse_data:/var/lib/clickhouse
|
||||
- ./config/clickhouse:/etc/clickhouse-server/config.d:ro
|
||||
healthcheck:
|
||||
test: ["CMD", "clickhouse-client", "--user", "${CLICKHOUSE_USERNAME:-default}", "--password", "${CLICKHOUSE_PASSWORD:-replace-with-clickhouse-password}", "--query", "SELECT 1"]
|
||||
interval: 10s
|
||||
@@ -173,6 +210,9 @@ volumes:
|
||||
创建对应的 `.env` 文件来配置系统环境变量(可复制并修改根目录下的 `.env.example`):
|
||||
|
||||
```bash
|
||||
mkdir -p ./config/clickhouse
|
||||
curl -fsSL -o ./config/clickhouse/performance.xml \
|
||||
https://raw.githubusercontent.com/Rain-kl/OpenFlare/refs/heads/main/config/clickhouse/performance.xml
|
||||
curl -o .env.example https://raw.githubusercontent.com/Rain-kl/OpenFlare/refs/heads/main/.env.example
|
||||
cp .env.example .env
|
||||
# 编辑 .env 文件,填入对应的数据库、Redis、ClickHouse 连接地址、密码与 APP_SESSION_SECRET
|
||||
@@ -264,25 +304,33 @@ services:
|
||||
CLICKHOUSE_PASSWORD: ${CLICKHOUSE_PASSWORD:-replace-with-clickhouse-password}
|
||||
CLICKHOUSE_DEFAULT_ACCESS_MANAGEMENT: 1
|
||||
TZ: ${TZ:-Asia/Shanghai}
|
||||
ulimits:
|
||||
nofile:
|
||||
soft: 262144
|
||||
hard: 262144
|
||||
volumes:
|
||||
- openflare_clickhouse_data:/var/lib/clickhouse
|
||||
|
||||
volumes:
|
||||
openflare_uploads:
|
||||
openflare_postgres_data:
|
||||
openflare_redis_data:
|
||||
openflare_clickhouse_data:
|
||||
- ./config/clickhouse:/etc/clickhouse-server/config.d:ro
|
||||
healthcheck:
|
||||
test: ["CMD", "clickhouse-client", "--user", "${CLICKHOUSE_USERNAME:-default}", "--password", "${CLICKHOUSE_PASSWORD:-replace-with-clickhouse-password}", "--query", "SELECT 1"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
start_period: 15s
|
||||
|
||||
volumes:
|
||||
openflare_uploads:
|
||||
openflare_postgres_data:
|
||||
openflare_redis_data:
|
||||
openflare_clickhouse_data:
|
||||
```
|
||||
|
||||
启动并验证:
|
||||
|
||||
```bash
|
||||
mkdir -p ./config/clickhouse
|
||||
curl -fsSL -o ./config/clickhouse/performance.xml \
|
||||
https://raw.githubusercontent.com/Rain-kl/OpenFlare/refs/heads/main/config/clickhouse/performance.xml
|
||||
curl -o .env.example https://raw.githubusercontent.com/Rain-kl/OpenFlare/refs/heads/main/.env.example
|
||||
cp .env.example .env
|
||||
# 编辑 .env 文件并确保设置好 APP_SESSION_SECRET 密码
|
||||
@@ -304,7 +352,7 @@ docker compose up -d
|
||||
| Go | `1.25+` |
|
||||
| Node.js | `18+` |
|
||||
| pnpm | 推荐通过 `corepack enable` 使用项目声明的 pnpm |
|
||||
| 外部服务 | 必须在本地或远端运行 Redis (Valkey) 和 ClickHouse 实例 |
|
||||
| 外部服务 | 必须在本地或远端运行 Redis (Valkey) 和 ClickHouse 实例;ClickHouse 建议挂载仓库提供的 `performance.xml`(见上文「ClickHouse 服务端性能配置」) |
|
||||
|
||||
### 1. 构建管理端前端
|
||||
|
||||
|
||||
@@ -30,6 +30,14 @@ Agent 统一通过 OpenResty 二进制控制运行时。本地部署需要节点
|
||||
|
||||
为了保证异步任务队列(Asynq 框架)及可观测流量看板功能完整运行,快速开始推荐采用 **PostgreSQL + Redis + ClickHouse** 经典单机版编排。
|
||||
|
||||
先拉取 ClickHouse 服务端性能配置到 `./config/clickhouse`(目录内**仅**放 `performance.xml`,不要放 listen 配置):
|
||||
|
||||
```bash
|
||||
mkdir -p ./config/clickhouse
|
||||
curl -fsSL -o ./config/clickhouse/performance.xml \
|
||||
https://raw.githubusercontent.com/Rain-kl/OpenFlare/refs/heads/main/config/clickhouse/performance.xml
|
||||
```
|
||||
|
||||
在空目录中创建 `docker-compose.yaml`:
|
||||
|
||||
```yaml
|
||||
@@ -54,9 +62,9 @@ services:
|
||||
DB_PASSWORD: "${DB_PASSWORD:-replace-with-strong-password}"
|
||||
DB_NAME: "${DB_NAME:-openflare}"
|
||||
REDIS_ENABLED: "true"
|
||||
REDIS_ADDRS: "redis:6379"
|
||||
REDIS_ADDR: "redis:6379"
|
||||
CLICKHOUSE_ENABLED: "true"
|
||||
CLICKHOUSE_HOSTS: "clickhouse:9000"
|
||||
CLICKHOUSE_HOST: "clickhouse:9000"
|
||||
depends_on:
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
@@ -101,10 +109,15 @@ services:
|
||||
CLICKHOUSE_PASSWORD: ${CLICKHOUSE_PASSWORD:-replace-with-clickhouse-password}
|
||||
CLICKHOUSE_DEFAULT_ACCESS_MANAGEMENT: 1
|
||||
TZ: Asia/Shanghai
|
||||
ulimits:
|
||||
nofile:
|
||||
soft: 262144
|
||||
hard: 262144
|
||||
volumes:
|
||||
- openflare_clickhouse_data:/var/lib/clickhouse
|
||||
- ./config/clickhouse:/etc/clickhouse-server/config.d:ro
|
||||
healthcheck:
|
||||
test: ["CMD", "clickhouse-client", "--user", "${CLICKHOUSE_USERNAME:-default}", "--password", "${CLICKHOUSE_PASSWORD:-replace-with-clickhouse-password}", "--query", "SELECT 1"]
|
||||
test: ["CMD", "clickhouse-client", "--user", "default", "--password", "${CLICKHOUSE_PASSWORD:-replace-with-clickhouse-password}", "--query", "SELECT 1"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
|
||||
@@ -0,0 +1,22 @@
|
||||
# ClickHouse P0–P3 修复计划
|
||||
|
||||
> 状态: 已完成(已合并主工作区,`make code-check` 通过)
|
||||
> 策略: 4 个互不干扰 worktree 并行,最后由主代理合并
|
||||
|
||||
## 任务拆分
|
||||
|
||||
| ID | Worktree 主题 | 范围 | 禁止改动 |
|
||||
|----|---------------|------|----------|
|
||||
| WT1 | P0 清理语义 C1 | cleanup maintenance / delete / tasks | chwriter、dashboard、DDL 新 MV |
|
||||
| WT2 | 写路径 C2+H1+H2+H3 | chwriter、batchwriter、risk_control、model store 分层、status 指标 | goose 迁移、dashboard 读逻辑 |
|
||||
| WT3 | 读路径 H4+H5 | 最新快照查询、metric/openresty 小时 MV + 读路径 | chwriter、cleanup |
|
||||
| WT4 | P3 打磨 | 连接池/async_insert、traffic hourly TTL、UV 语义 | model store 分层、cleanup |
|
||||
|
||||
## 合并顺序
|
||||
|
||||
1. WT1 → 2. WT2 → 3. WT3 → 4. WT4
|
||||
(迁移文件时间戳已错开,changelog 由主代理统一写)
|
||||
|
||||
## 验收
|
||||
|
||||
各 worktree: 相关 `go test` + 可运行部分;合并后 `make code-check`。
|
||||
@@ -99,13 +99,21 @@ Server 的所有核心基础配置定义在 `config.yaml` 中,且均支持环
|
||||
| `redis.pool_size` | `REDIS_POOL_SIZE` | Redis 连接池大小 | `100` |
|
||||
|
||||
### 4. ClickHouse 配置 (`clickhouse:`)
|
||||
|
||||
> **说明**:下列为 OpenFlare **客户端**连接参数。ClickHouse **服务端**小规格调优:将 `performance.xml` curl 到 `./config/clickhouse/`,compose 挂载 `./config/clickhouse:/etc/clickhouse-server/config.d:ro`(目录内不要放 listen 配置),详见 [启动 Server](../deployment/server.md)。
|
||||
|
||||
| 配置文件 YAML 路径 | 对应覆盖环境变量 | 作用说明 | 默认值 |
|
||||
| --- | --- | --- | --- |
|
||||
| `clickhouse.enabled` | `CLICKHOUSE_ENABLED` | 是否启用 ClickHouse。**系统节点指标与访问日志在此进行海量写入** | `true` |
|
||||
| `clickhouse.hosts` | `CLICKHOUSE_HOST` | ClickHouse 集群连接地址数组(环境变量仅设置单地址) | `["127.0.0.1:9000"]` |
|
||||
| `clickhouse.username` | `CLICKHOUSE_USERNAME` | ClickHouse 账号用户名 | `default` |
|
||||
| `clickhouse.password` | `CLICKHOUSE_PASSWORD` | ClickHouse 密码 | `123456` |
|
||||
| `clickhouse.password` | `CLICKHOUSE_PASSWORD` | ClickHouse 密码 | `replace-with-clickhouse-password` |
|
||||
| `clickhouse.database` | `CLICKHOUSE_NAME` | ClickHouse 存储的数据库名称 | `openflare` |
|
||||
| `clickhouse.max_idle_conn` | - | 客户端空闲连接数(小规格默认偏低) | `8` |
|
||||
| `clickhouse.max_open_conn` | - | 客户端最大打开连接数 | `16` |
|
||||
| `clickhouse.conn_max_lifetime` | - | 连接最大存活时间(秒) | `3600` |
|
||||
| `clickhouse.dial_timeout` | - | 建连超时(秒) | `5` |
|
||||
| `clickhouse.block_buffer_size` | - | 原生协议 block 缓冲行数 | `32` |
|
||||
|
||||
### 5. 系统日志配置 (`log:`)
|
||||
| 配置文件 YAML 路径 | 对应覆盖环境变量 | 作用说明 | 默认值 |
|
||||
@@ -161,7 +169,7 @@ Server 的所有核心基础配置定义在 `config.yaml` 中,且均支持环
|
||||
### 2. 人机安全校验 (PoW Captcha)
|
||||
| 配置键 (Key) | 数据类型 | 作用说明 | 默认值 |
|
||||
| --- | --- | --- | --- |
|
||||
| `cap_login_enabled` | `bool` | 是否在登录界面强制要求进行本地 PoW 算力防爆破人机验证 | `true` |
|
||||
| `cap_login_enabled` | `bool` | 是否在登录界面强制要求进行本地 PoW 算力防爆破人机验证 | `false` |
|
||||
| `cap_auto_solve` | `bool` | 打开页面后是否由浏览器自动开始后台背景计算算力(无需用户手动点击)| `true` |
|
||||
| `cap_challenge_count` | `int` | 人机验证所需的计算难题数。数量越大,计算要求时间越长(推荐 1~5) | `1` |
|
||||
| `cap_challenge_difficulty`| `int`| 每次计算所需的 PoW 哈希前缀匹配难度。推荐数值在 3-5 之间 | `4` |
|
||||
|
||||
@@ -34,7 +34,7 @@ export function DashboardStatCards({
|
||||
{formatCompactNumber(traffic.request_count)}
|
||||
</div>
|
||||
<p className="text-[10px] text-muted-foreground">
|
||||
独立访客 {formatCompactNumber(traffic.unique_visitors)} · 错误{' '}
|
||||
窗口UV(估) {formatCompactNumber(traffic.unique_visitors)} · 错误{' '}
|
||||
{formatCompactNumber(traffic.error_count)} · 估算 QPS{' '}
|
||||
{traffic.estimated_qps.toFixed(2)}
|
||||
</p>
|
||||
|
||||
@@ -661,7 +661,7 @@ export function NodeObservability({
|
||||
</p>
|
||||
<p className="mt-2 text-sm text-muted-foreground">
|
||||
{trafficSummary
|
||||
? `近 60 秒 · UV ${formatMetricCount(trafficSummary.unique_visitor_count)}`
|
||||
? `近 60 秒 · 窗口UV ${formatMetricCount(trafficSummary.unique_visitor_count)}`
|
||||
: '暂无窗口流量摘要'}
|
||||
</p>
|
||||
</div>
|
||||
|
||||
@@ -73,7 +73,7 @@ export function LoginForm({ onOTPStateChange }: { onOTPStateChange?: (show: bool
|
||||
queryFn: () => AuthService.getAuthSources(),
|
||||
})
|
||||
|
||||
const capEnabled = configBool(publicConfigQuery.data?.cap_login_enabled, true)
|
||||
const capEnabled = configBool(publicConfigQuery.data?.cap_login_enabled, false)
|
||||
const capAutoSolve = configBool(publicConfigQuery.data?.cap_auto_solve, true)
|
||||
|
||||
const loginMutation = useMutation({
|
||||
|
||||
@@ -68,7 +68,7 @@ export function RegisterForm() {
|
||||
|
||||
const emailRegisterEnabled = configBool(publicConfigQuery.data?.email_register_verification_enabled, false)
|
||||
|
||||
const capEnabled = configBool(publicConfigQuery.data?.cap_login_enabled, true)
|
||||
const capEnabled = configBool(publicConfigQuery.data?.cap_login_enabled, false)
|
||||
const capAutoSolve = configBool(publicConfigQuery.data?.cap_auto_solve, true)
|
||||
|
||||
const [capScope, setCapScope] = useState<'send_email_code' | 'register'>('send_email_code')
|
||||
|
||||
@@ -245,7 +245,7 @@ export const apiSections: PolicySection[] = [
|
||||
"registration_enabled": "false",
|
||||
"password_login_enabled": "true",
|
||||
"password_register_enabled": "false",
|
||||
"cap_login_enabled": "true",
|
||||
"cap_login_enabled": "false",
|
||||
"oidc_login_enabled": "true"
|
||||
}
|
||||
}`}
|
||||
|
||||
@@ -6,16 +6,19 @@ package status
|
||||
import (
|
||||
"net/http"
|
||||
|
||||
"github.com/Rain-kl/Wavelet/internal/apps/openflare/chwriter"
|
||||
"github.com/Rain-kl/Wavelet/internal/apps/risk_control"
|
||||
"github.com/Rain-kl/Wavelet/internal/common/response"
|
||||
"github.com/Rain-kl/Wavelet/internal/config"
|
||||
"github.com/Rain-kl/Wavelet/internal/db"
|
||||
"github.com/Rain-kl/Wavelet/internal/db/batchwriter"
|
||||
analyticsrepo "github.com/Rain-kl/Wavelet/internal/repository/analytics"
|
||||
"github.com/gin-gonic/gin"
|
||||
)
|
||||
|
||||
// GetClickHouseStatus returns ClickHouse operational metrics for administrators.
|
||||
// @Summary 获取 ClickHouse 运行指标
|
||||
// @Description 返回 ClickHouse parts、mutation、async_insert 队列等运维指标,需要管理员权限
|
||||
// @Description 返回 ClickHouse parts、mutation、async_insert 队列及进程内 batch writer 指标,需要管理员权限
|
||||
// @Tags admin
|
||||
// @Produce json
|
||||
// @Security SessionCookie
|
||||
@@ -36,5 +39,15 @@ func GetClickHouseStatus(c *gin.Context) {
|
||||
response.AbortInternal(c, "获取 ClickHouse 运行指标失败")
|
||||
return
|
||||
}
|
||||
stats.BatchWriters = collectBatchWriterStats()
|
||||
c.JSON(http.StatusOK, response.OK(stats))
|
||||
}
|
||||
}
|
||||
|
||||
func collectBatchWriterStats() []batchwriter.Stats {
|
||||
out := chwriter.WriterStats()
|
||||
if out == nil {
|
||||
out = make([]batchwriter.Stats, 0, 1)
|
||||
}
|
||||
out = append(out, risk_control.LogWriterStats())
|
||||
return out
|
||||
}
|
||||
|
||||
@@ -56,13 +56,13 @@ func TestProtectionEnabledReflectsLoginSwitch(t *testing.T) {
|
||||
|
||||
ResetRuntimeSettingsForTest()
|
||||
|
||||
if !ProtectionEnabled(ctx) {
|
||||
t.Fatal("ProtectionEnabled() = false, want true from seed defaults")
|
||||
if ProtectionEnabled(ctx) {
|
||||
t.Fatal("ProtectionEnabled() = true, want false from seed defaults")
|
||||
}
|
||||
|
||||
if err := db.DB(ctx).Model(&model.SystemConfig{}).
|
||||
Where("key = ?", model.ConfigKeyCapLoginEnabled).
|
||||
Update("value", "false").Error; err != nil {
|
||||
Update("value", "true").Error; err != nil {
|
||||
t.Fatalf("Update(cap_login_enabled) error = %v", err)
|
||||
}
|
||||
if err := repository.InvalidateSystemConfigCache(ctx, model.ConfigKeyCapLoginEnabled); err != nil {
|
||||
@@ -70,8 +70,8 @@ func TestProtectionEnabledReflectsLoginSwitch(t *testing.T) {
|
||||
}
|
||||
InvalidateRuntimeSettings()
|
||||
|
||||
if ProtectionEnabled(ctx) {
|
||||
t.Fatal("ProtectionEnabled() = true, want false after config update")
|
||||
if !ProtectionEnabled(ctx) {
|
||||
t.Fatal("ProtectionEnabled() = false, want true after config update")
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -25,7 +25,7 @@ func newDedupSet() *dedupSet {
|
||||
|
||||
// markIfNew records key when it has not been seen within dedupTTL.
|
||||
func (s *dedupSet) markIfNew(key string) bool {
|
||||
if key == "" {
|
||||
if s == nil || key == "" {
|
||||
return false
|
||||
}
|
||||
|
||||
@@ -33,19 +33,33 @@ func (s *dedupSet) markIfNew(key string) bool {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
|
||||
// Periodically clean up all expired keys (e.g., every 30 seconds)
|
||||
if now.Sub(s.lastCleanup) >= 30*time.Second {
|
||||
for existing, expiresAt := range s.keys {
|
||||
if now.After(expiresAt) {
|
||||
delete(s.keys, existing)
|
||||
}
|
||||
}
|
||||
s.lastCleanup = now
|
||||
}
|
||||
s.cleanupExpiredLocked(now)
|
||||
|
||||
if expiresAt, exists := s.keys[key]; exists && now.Before(expiresAt) {
|
||||
return false
|
||||
}
|
||||
s.keys[key] = now.Add(dedupTTL)
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
// unmark removes a key so a later enqueue or flush retry may accept it again.
|
||||
func (s *dedupSet) unmark(key string) {
|
||||
if s == nil || key == "" {
|
||||
return
|
||||
}
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
delete(s.keys, key)
|
||||
}
|
||||
|
||||
func (s *dedupSet) cleanupExpiredLocked(now time.Time) {
|
||||
if now.Sub(s.lastCleanup) < 30*time.Second {
|
||||
return
|
||||
}
|
||||
for existing, expiresAt := range s.keys {
|
||||
if now.After(expiresAt) {
|
||||
delete(s.keys, existing)
|
||||
}
|
||||
}
|
||||
s.lastCleanup = now
|
||||
}
|
||||
|
||||
@@ -3,7 +3,16 @@
|
||||
|
||||
package chwriter
|
||||
|
||||
import "testing"
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/Rain-kl/Wavelet/internal/db/batchwriter"
|
||||
analyticsmodel "github.com/Rain-kl/Wavelet/internal/model/analytics"
|
||||
)
|
||||
|
||||
func TestDedupSetMarkIfNew(t *testing.T) {
|
||||
t.Parallel()
|
||||
@@ -21,4 +30,157 @@ func TestDedupSetMarkIfNew(t *testing.T) {
|
||||
if set.markIfNew("") {
|
||||
t.Fatal("markIfNew() = true, want false on empty key")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestDedupSetUnmarkAllowsRetry(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
set := newDedupSet()
|
||||
if !set.markIfNew("k") {
|
||||
t.Fatal("markIfNew() = false, want true")
|
||||
}
|
||||
set.unmark("k")
|
||||
if !set.markIfNew("k") {
|
||||
t.Fatal("markIfNew() after unmark = false, want true")
|
||||
}
|
||||
}
|
||||
|
||||
func TestQueueWithDedupDoesNotMarkWhenEnqueueFails(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
cfg := batchwriter.DefaultConfig()
|
||||
cfg.QueueSize = 1
|
||||
cfg.MaxBatchSize = 10
|
||||
cfg.FlushInterval = time.Hour
|
||||
|
||||
// Block the worker so the queue stays full after one enqueue.
|
||||
block := make(chan struct{})
|
||||
writer, err := batchwriter.New[int](cfg, func(context.Context, []int) error {
|
||||
<-block
|
||||
return nil
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("New() error = %v", err)
|
||||
}
|
||||
writer.Start(context.Background())
|
||||
t.Cleanup(func() {
|
||||
close(block)
|
||||
stopCtx, cancel := context.WithTimeout(context.Background(), time.Second)
|
||||
defer cancel()
|
||||
_ = writer.Stop(stopCtx)
|
||||
})
|
||||
|
||||
// Fill the channel buffer (and the worker's current receive slot may empty one).
|
||||
// Keep enqueueing until full so subsequent queueWithDedup fails.
|
||||
for i := 0; i < cfg.QueueSize+2; i++ {
|
||||
_ = writer.TryEnqueue(i)
|
||||
if writer.IsFull() {
|
||||
break
|
||||
}
|
||||
}
|
||||
if !writer.IsFull() {
|
||||
t.Fatal("writer not full after filling; cannot test enqueue failure path")
|
||||
}
|
||||
|
||||
dedup := newDedupSet()
|
||||
queueWithDedup(writer, dedup, "dedup-key", 99)
|
||||
// Key must not remain marked after failed enqueue.
|
||||
if !dedup.markIfNew("dedup-key") {
|
||||
t.Fatal("dedup key still marked after failed enqueue; want unmark")
|
||||
}
|
||||
}
|
||||
|
||||
func TestQueueWithDedupMarksOnlyOnSuccess(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
cfg := batchwriter.DefaultConfig()
|
||||
cfg.MaxBatchSize = 100
|
||||
cfg.FlushInterval = time.Hour
|
||||
|
||||
writer, err := batchwriter.New[int](cfg, func(context.Context, []int) error { return nil })
|
||||
if err != nil {
|
||||
t.Fatalf("New() error = %v", err)
|
||||
}
|
||||
writer.Start(context.Background())
|
||||
t.Cleanup(func() {
|
||||
stopCtx, cancel := context.WithTimeout(context.Background(), time.Second)
|
||||
defer cancel()
|
||||
_ = writer.Stop(stopCtx)
|
||||
})
|
||||
|
||||
dedup := newDedupSet()
|
||||
queueWithDedup(writer, dedup, "ok-key", 1)
|
||||
if dedup.markIfNew("ok-key") {
|
||||
t.Fatal("markIfNew() = true after successful enqueue, want false (key marked)")
|
||||
}
|
||||
}
|
||||
|
||||
func TestFlushErrorHandlerUnmarksKeys(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dedup := newDedupSet()
|
||||
flushErr := errors.New("ch down")
|
||||
|
||||
var (
|
||||
mu sync.Mutex
|
||||
errCount int
|
||||
)
|
||||
|
||||
cfg := batchwriter.Config{
|
||||
Name: "test_obs",
|
||||
QueueSize: 10,
|
||||
MaxBatchSize: 1,
|
||||
FlushInterval: time.Hour,
|
||||
}
|
||||
keyFn := func(s analyticsmodel.NodeMetricSnapshot) string {
|
||||
return metricSnapshotKey(s)
|
||||
}
|
||||
writer, err := batchwriter.New(
|
||||
cfg,
|
||||
func(context.Context, []analyticsmodel.NodeMetricSnapshot) error { return flushErr },
|
||||
batchwriter.WithFlushErrorHandler[analyticsmodel.NodeMetricSnapshot](func(_ context.Context, items []analyticsmodel.NodeMetricSnapshot, err error) {
|
||||
mu.Lock()
|
||||
errCount++
|
||||
mu.Unlock()
|
||||
for _, item := range items {
|
||||
dedup.unmark(keyFn(item))
|
||||
}
|
||||
}),
|
||||
)
|
||||
if err != nil {
|
||||
t.Fatalf("New() error = %v", err)
|
||||
}
|
||||
writer.Start(context.Background())
|
||||
t.Cleanup(func() {
|
||||
stopCtx, cancel := context.WithTimeout(context.Background(), time.Second)
|
||||
defer cancel()
|
||||
_ = writer.Stop(stopCtx)
|
||||
})
|
||||
|
||||
item := analyticsmodel.NodeMetricSnapshot{
|
||||
NodeID: "n1",
|
||||
CapturedAt: time.Unix(1, 0).UTC(),
|
||||
}
|
||||
key := keyFn(item)
|
||||
if !dedup.markIfNew(key) {
|
||||
t.Fatal("markIfNew failed")
|
||||
}
|
||||
if !writer.TryEnqueue(item) {
|
||||
t.Fatal("TryEnqueue failed")
|
||||
}
|
||||
|
||||
deadline := time.Now().Add(time.Second)
|
||||
for {
|
||||
mu.Lock()
|
||||
ready := errCount >= 1
|
||||
mu.Unlock()
|
||||
if ready || time.Now().After(deadline) {
|
||||
break
|
||||
}
|
||||
time.Sleep(5 * time.Millisecond)
|
||||
}
|
||||
|
||||
if !dedup.markIfNew(key) {
|
||||
t.Fatal("key still marked after flush error unmark; want available for retry")
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,86 @@
|
||||
//go:build live_ch
|
||||
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
package chwriter_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/Rain-kl/Wavelet/internal/apps/openflare/chwriter"
|
||||
"github.com/Rain-kl/Wavelet/internal/db"
|
||||
"github.com/Rain-kl/Wavelet/internal/model"
|
||||
)
|
||||
|
||||
// Run with Docker ClickHouse + config.yaml:
|
||||
//
|
||||
// go test -tags live_ch ./internal/apps/openflare/chwriter -run TestLiveAppWritePath -count=1 -timeout 2m
|
||||
func TestLiveAppWritePath(t *testing.T) {
|
||||
if !db.ChConnReady() {
|
||||
t.Skip("ClickHouse connection not ready")
|
||||
}
|
||||
ctx := context.Background()
|
||||
chwriter.Init(ctx)
|
||||
|
||||
now := time.Now().UTC()
|
||||
nodeID := "e2e-app-write-" + now.Format("150405")
|
||||
if err := model.InsertOpenFlareMetricSnapshot(ctx, &model.OpenFlareMetricSnapshot{
|
||||
NodeID: nodeID,
|
||||
CapturedAt: now,
|
||||
CPUUsagePercent: 33.3,
|
||||
MemoryUsedBytes: 111,
|
||||
MemoryTotalBytes: 1000,
|
||||
StorageUsedBytes: 222,
|
||||
StorageTotalBytes: 2000,
|
||||
DiskReadBytes: 10,
|
||||
DiskWriteBytes: 20,
|
||||
NetworkRxBytes: 30,
|
||||
NetworkTxBytes: 40,
|
||||
}); err != nil {
|
||||
t.Fatalf("InsertOpenFlareMetricSnapshot: %v", err)
|
||||
}
|
||||
|
||||
deadline := time.Now().Add(45 * time.Second)
|
||||
var found bool
|
||||
for time.Now().Before(deadline) {
|
||||
rows, err := model.ListOpenFlareMetricSnapshotsSince(ctx, nodeID, now.Add(-time.Minute), 10)
|
||||
if err != nil {
|
||||
t.Fatalf("ListOpenFlareMetricSnapshotsSince: %v", err)
|
||||
}
|
||||
if len(rows) > 0 {
|
||||
found = true
|
||||
t.Logf("found snapshot id=%d cpu=%.1f after flush", rows[0].ID, rows[0].CPUUsagePercent)
|
||||
break
|
||||
}
|
||||
time.Sleep(2 * time.Second)
|
||||
}
|
||||
if !found {
|
||||
t.Fatal("metric snapshot not visible in ClickHouse after flush wait")
|
||||
}
|
||||
|
||||
latest, err := model.ListOpenFlareLatestMetricSnapshotsSince(ctx, "", now.Add(-time.Hour))
|
||||
if err != nil {
|
||||
t.Fatalf("ListOpenFlareLatestMetricSnapshotsSince: %v", err)
|
||||
}
|
||||
var latestOK bool
|
||||
for _, row := range latest {
|
||||
if row != nil && row.NodeID == nodeID {
|
||||
latestOK = true
|
||||
break
|
||||
}
|
||||
}
|
||||
if !latestOK {
|
||||
t.Fatalf("latest-per-node query missing node %s (rows=%d)", nodeID, len(latest))
|
||||
}
|
||||
|
||||
stats := chwriter.WriterStats()
|
||||
if len(stats) == 0 {
|
||||
t.Fatal("WriterStats empty after Init")
|
||||
}
|
||||
for _, s := range stats {
|
||||
t.Logf("writer %s running=%v depth=%d drops=%d flush_err=%d", s.Name, s.Running, s.Depth, s.Drops, s.FlushErrors)
|
||||
}
|
||||
}
|
||||
@@ -14,19 +14,30 @@ import (
|
||||
"github.com/Rain-kl/Wavelet/internal/config"
|
||||
"github.com/Rain-kl/Wavelet/internal/db/batchwriter"
|
||||
"github.com/Rain-kl/Wavelet/internal/lifecycle"
|
||||
"github.com/Rain-kl/Wavelet/internal/model"
|
||||
analyticsmodel "github.com/Rain-kl/Wavelet/internal/model/analytics"
|
||||
analyticsrepo "github.com/Rain-kl/Wavelet/internal/repository/analytics"
|
||||
"github.com/Rain-kl/Wavelet/pkg/logger"
|
||||
)
|
||||
|
||||
const (
|
||||
// Observability traffic is sparse (heartbeat ~10s/node). Prefer larger batches to
|
||||
// cut ClickHouse parts/merges; MaxFlushWait bounds visibility lag for single-node labs.
|
||||
observabilityQueueSize = 5_000
|
||||
observabilityMaxBatchSize = 500
|
||||
observabilityFlushEvery = 5 * time.Second
|
||||
observabilityMinBatchSize = 20
|
||||
observabilityFlushEvery = 10 * time.Second
|
||||
observabilityMaxFlushWait = 30 * time.Second
|
||||
|
||||
nodeAccessLogQueueSize = 10_000
|
||||
nodeAccessLogMaxBatchSize = 1_000
|
||||
nodeAccessLogFlushEvery = time.Second
|
||||
nodeAccessLogMinBatchSize = 50
|
||||
nodeAccessLogFlushEvery = 2 * time.Second
|
||||
nodeAccessLogMaxFlushWait = 5 * time.Second
|
||||
|
||||
// flushAttempts is total tries (1 initial + short retries) before giving up a batch.
|
||||
flushAttempts = 2
|
||||
flushRetryBackoff = 50 * time.Millisecond
|
||||
)
|
||||
|
||||
var (
|
||||
@@ -59,11 +70,36 @@ func Init(ctx context.Context) {
|
||||
frpsDedup = newDedupSet()
|
||||
frpcDedup = newDedupSet()
|
||||
|
||||
metricSnapshotWriter = mustNewObservabilityWriter("metric_snapshots", analyticsrepo.BatchInsertNodeMetricSnapshots)
|
||||
requestReportWriter = mustNewObservabilityWriter("request_reports", analyticsrepo.BatchInsertNodeRequestReports)
|
||||
openrestyWriter = mustNewObservabilityWriter("openresty_obs", analyticsrepo.BatchInsertNodeObsOpenresty)
|
||||
frpsWriter = mustNewObservabilityWriter("frps_obs", analyticsrepo.BatchInsertNodeObsFrps)
|
||||
frpcWriter = mustNewObservabilityWriter("frpc_obs", analyticsrepo.BatchInsertNodeObsFrpc)
|
||||
metricSnapshotWriter = mustNewObservabilityWriter(
|
||||
"metric_snapshots",
|
||||
withFlushRetries(analyticsrepo.BatchInsertNodeMetricSnapshots),
|
||||
metricSnapshotDedup,
|
||||
metricSnapshotKey,
|
||||
)
|
||||
requestReportWriter = mustNewObservabilityWriter(
|
||||
"request_reports",
|
||||
withFlushRetries(analyticsrepo.BatchInsertNodeRequestReports),
|
||||
requestReportDedup,
|
||||
requestReportKey,
|
||||
)
|
||||
openrestyWriter = mustNewObservabilityWriter(
|
||||
"openresty_obs",
|
||||
withFlushRetries(analyticsrepo.BatchInsertNodeObsOpenresty),
|
||||
openrestyDedup,
|
||||
openrestyKey,
|
||||
)
|
||||
frpsWriter = mustNewObservabilityWriter(
|
||||
"frps_obs",
|
||||
withFlushRetries(analyticsrepo.BatchInsertNodeObsFrps),
|
||||
frpsDedup,
|
||||
frpsKey,
|
||||
)
|
||||
frpcWriter = mustNewObservabilityWriter(
|
||||
"frpc_obs",
|
||||
withFlushRetries(analyticsrepo.BatchInsertNodeObsFrpc),
|
||||
frpcDedup,
|
||||
frpcKey,
|
||||
)
|
||||
nodeAccessLogWriter = mustNewNodeAccessLogWriter()
|
||||
|
||||
metricSnapshotWriter.Start(ctx)
|
||||
@@ -73,6 +109,7 @@ func Init(ctx context.Context) {
|
||||
frpcWriter.Start(ctx)
|
||||
nodeAccessLogWriter.Start(ctx)
|
||||
|
||||
wireModelInsertHooks()
|
||||
lifecycle.OnShutdown("openflare_chwriter", Stop)
|
||||
})
|
||||
}
|
||||
@@ -102,69 +139,49 @@ func Stop(ctx context.Context) error {
|
||||
return firstErr
|
||||
}
|
||||
|
||||
// WriterStats returns queue depth and failure counters for all OpenFlare writers.
|
||||
func WriterStats() []batchwriter.Stats {
|
||||
writers := []statsProvider{
|
||||
metricSnapshotWriter,
|
||||
requestReportWriter,
|
||||
openrestyWriter,
|
||||
frpsWriter,
|
||||
frpcWriter,
|
||||
nodeAccessLogWriter,
|
||||
}
|
||||
out := make([]batchwriter.Stats, 0, len(writers))
|
||||
for _, w := range writers {
|
||||
if w == nil {
|
||||
continue
|
||||
}
|
||||
out = append(out, w.Stats())
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// QueueMetricSnapshot enqueues a metric snapshot for asynchronous flush.
|
||||
func QueueMetricSnapshot(snapshot analyticsmodel.NodeMetricSnapshot) {
|
||||
if metricSnapshotWriter == nil {
|
||||
return
|
||||
}
|
||||
key := fmt.Sprintf("%s|%d", snapshot.NodeID, snapshot.CapturedAt.UTC().UnixNano())
|
||||
if !metricSnapshotDedup.markIfNew(key) {
|
||||
return
|
||||
}
|
||||
metricSnapshotWriter.TryEnqueue(snapshot)
|
||||
queueWithDedup(metricSnapshotWriter, metricSnapshotDedup, metricSnapshotKey(snapshot), snapshot)
|
||||
}
|
||||
|
||||
// QueueRequestReport enqueues a request report for asynchronous flush.
|
||||
func QueueRequestReport(report analyticsmodel.NodeRequestReport) {
|
||||
if requestReportWriter == nil {
|
||||
return
|
||||
}
|
||||
key := fmt.Sprintf(
|
||||
"%s|%d|%d",
|
||||
report.NodeID,
|
||||
report.WindowStartedAt.UTC().UnixNano(),
|
||||
report.WindowEndedAt.UTC().UnixNano(),
|
||||
)
|
||||
if !requestReportDedup.markIfNew(key) {
|
||||
return
|
||||
}
|
||||
requestReportWriter.TryEnqueue(report)
|
||||
queueWithDedup(requestReportWriter, requestReportDedup, requestReportKey(report), report)
|
||||
}
|
||||
|
||||
// QueueOpenrestyObservation enqueues an OpenResty observation for asynchronous flush.
|
||||
func QueueOpenrestyObservation(observation analyticsmodel.NodeObsOpenresty) {
|
||||
if openrestyWriter == nil {
|
||||
return
|
||||
}
|
||||
key := fmt.Sprintf("%s|%d", observation.NodeID, observation.CapturedAt.UTC().UnixNano())
|
||||
if !openrestyDedup.markIfNew(key) {
|
||||
return
|
||||
}
|
||||
openrestyWriter.TryEnqueue(observation)
|
||||
queueWithDedup(openrestyWriter, openrestyDedup, openrestyKey(observation), observation)
|
||||
}
|
||||
|
||||
// QueueFrpsObservation enqueues an FRPS observation for asynchronous flush.
|
||||
func QueueFrpsObservation(observation analyticsmodel.NodeObsFrps) {
|
||||
if frpsWriter == nil {
|
||||
return
|
||||
}
|
||||
key := fmt.Sprintf("%s|%d", observation.NodeID, observation.CapturedAt.UTC().UnixNano())
|
||||
if !frpsDedup.markIfNew(key) {
|
||||
return
|
||||
}
|
||||
frpsWriter.TryEnqueue(observation)
|
||||
queueWithDedup(frpsWriter, frpsDedup, frpsKey(observation), observation)
|
||||
}
|
||||
|
||||
// QueueFrpcObservation enqueues an FRPC observation for asynchronous flush.
|
||||
func QueueFrpcObservation(observation analyticsmodel.NodeObsFrpc) {
|
||||
if frpcWriter == nil {
|
||||
return
|
||||
}
|
||||
key := fmt.Sprintf("%s|%d", observation.NodeID, observation.CapturedAt.UTC().UnixNano())
|
||||
if !frpcDedup.markIfNew(key) {
|
||||
return
|
||||
}
|
||||
frpcWriter.TryEnqueue(observation)
|
||||
queueWithDedup(frpcWriter, frpcDedup, frpcKey(observation), observation)
|
||||
}
|
||||
|
||||
// QueueNodeAccessLogs enqueues node access logs for asynchronous flush.
|
||||
@@ -177,19 +194,46 @@ func QueueNodeAccessLogs(logs []analyticsmodel.NodeAccessLog) {
|
||||
}
|
||||
}
|
||||
|
||||
func mustNewObservabilityWriter[T any](name string, flush batchwriter.FlushFunc[T]) *batchwriter.Writer[T] {
|
||||
func queueWithDedup[T any](writer *batchwriter.Writer[T], dedup *dedupSet, key string, item T) {
|
||||
if writer == nil {
|
||||
return
|
||||
}
|
||||
// Mark first so concurrent duplicates still collapse; release on enqueue failure
|
||||
// so a full queue does not permanently suppress the item.
|
||||
if !dedup.markIfNew(key) {
|
||||
return
|
||||
}
|
||||
if !writer.TryEnqueue(item) {
|
||||
dedup.unmark(key)
|
||||
}
|
||||
}
|
||||
|
||||
func mustNewObservabilityWriter[T any](
|
||||
name string,
|
||||
flush batchwriter.FlushFunc[T],
|
||||
dedup *dedupSet,
|
||||
keyFn func(T) string,
|
||||
) *batchwriter.Writer[T] {
|
||||
cfg := batchwriter.Config{
|
||||
Name: name,
|
||||
QueueSize: observabilityQueueSize,
|
||||
MaxBatchSize: observabilityMaxBatchSize,
|
||||
MinBatchSize: observabilityMinBatchSize,
|
||||
FlushInterval: observabilityFlushEvery,
|
||||
MaxFlushWait: observabilityMaxFlushWait,
|
||||
}
|
||||
writer, err := batchwriter.New(
|
||||
cfg,
|
||||
flush,
|
||||
withObservabilityDropHandler[T](name),
|
||||
batchwriter.WithFlushErrorHandler[T](func(ctx context.Context, batchSize int, err error) {
|
||||
logger.ErrorF(ctx, "[OpenFlare] flush %s failed (batch=%d): %v", name, batchSize, err)
|
||||
batchwriter.WithFlushErrorHandler[T](func(ctx context.Context, items []T, err error) {
|
||||
logger.ErrorF(ctx, "[OpenFlare] flush %s failed (batch=%d): %v", name, len(items), err)
|
||||
if dedup == nil || keyFn == nil {
|
||||
return
|
||||
}
|
||||
for _, item := range items {
|
||||
dedup.unmark(keyFn(item))
|
||||
}
|
||||
}),
|
||||
)
|
||||
if err != nil {
|
||||
@@ -203,14 +247,18 @@ func mustNewNodeAccessLogWriter() *batchwriter.Writer[analyticsmodel.NodeAccessL
|
||||
Name: "node_access_logs",
|
||||
QueueSize: nodeAccessLogQueueSize,
|
||||
MaxBatchSize: nodeAccessLogMaxBatchSize,
|
||||
MinBatchSize: nodeAccessLogMinBatchSize,
|
||||
FlushInterval: nodeAccessLogFlushEvery,
|
||||
MaxFlushWait: nodeAccessLogMaxFlushWait,
|
||||
}
|
||||
writer, err := batchwriter.New[analyticsmodel.NodeAccessLog](cfg, analyticsrepo.BatchInsertNodeAccessLogs,
|
||||
writer, err := batchwriter.New[analyticsmodel.NodeAccessLog](
|
||||
cfg,
|
||||
withFlushRetries(analyticsrepo.BatchInsertNodeAccessLogs),
|
||||
batchwriter.WithDropHandler[analyticsmodel.NodeAccessLog](func(item analyticsmodel.NodeAccessLog) {
|
||||
logger.WarnF(context.Background(), "[OpenFlare] node access log queue full, dropping log for node %s path %s", item.NodeID, item.Path)
|
||||
}),
|
||||
batchwriter.WithFlushErrorHandler[analyticsmodel.NodeAccessLog](func(ctx context.Context, batchSize int, err error) {
|
||||
logger.ErrorF(ctx, "[OpenFlare] flush node access logs failed (batch=%d): %v", batchSize, err)
|
||||
batchwriter.WithFlushErrorHandler[analyticsmodel.NodeAccessLog](func(ctx context.Context, items []analyticsmodel.NodeAccessLog, err error) {
|
||||
logger.ErrorF(ctx, "[OpenFlare] flush node access logs failed (batch=%d): %v", len(items), err)
|
||||
}),
|
||||
)
|
||||
if err != nil {
|
||||
@@ -225,10 +273,74 @@ func withObservabilityDropHandler[T any](name string) batchwriter.Option[T] {
|
||||
})
|
||||
}
|
||||
|
||||
// withFlushRetries wraps a flush function with a short retry to ride out brief CH blips.
|
||||
func withFlushRetries[T any](flush batchwriter.FlushFunc[T]) batchwriter.FlushFunc[T] {
|
||||
return func(ctx context.Context, items []T) error {
|
||||
var err error
|
||||
for attempt := 1; attempt <= flushAttempts; attempt++ {
|
||||
err = flush(ctx, items)
|
||||
if err == nil {
|
||||
return nil
|
||||
}
|
||||
if attempt == flushAttempts {
|
||||
break
|
||||
}
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return ctx.Err()
|
||||
case <-time.After(flushRetryBackoff * time.Duration(attempt)):
|
||||
}
|
||||
}
|
||||
return err
|
||||
}
|
||||
}
|
||||
|
||||
func wireModelInsertHooks() {
|
||||
model.SetObservabilityInsertHooks(model.ObservabilityInsertHooks{
|
||||
QueueMetricSnapshot: QueueMetricSnapshot,
|
||||
QueueRequestReport: QueueRequestReport,
|
||||
QueueOpenrestyObservation: QueueOpenrestyObservation,
|
||||
QueueFrpsObservation: QueueFrpsObservation,
|
||||
QueueFrpcObservation: QueueFrpcObservation,
|
||||
})
|
||||
model.SetAccessLogInsertHooks(model.AccessLogInsertHooks{
|
||||
QueueNodeAccessLogs: QueueNodeAccessLogs,
|
||||
})
|
||||
}
|
||||
|
||||
func metricSnapshotKey(snapshot analyticsmodel.NodeMetricSnapshot) string {
|
||||
return fmt.Sprintf("%s|%d", snapshot.NodeID, snapshot.CapturedAt.UTC().UnixNano())
|
||||
}
|
||||
|
||||
func requestReportKey(report analyticsmodel.NodeRequestReport) string {
|
||||
return fmt.Sprintf(
|
||||
"%s|%d|%d",
|
||||
report.NodeID,
|
||||
report.WindowStartedAt.UTC().UnixNano(),
|
||||
report.WindowEndedAt.UTC().UnixNano(),
|
||||
)
|
||||
}
|
||||
|
||||
func openrestyKey(observation analyticsmodel.NodeObsOpenresty) string {
|
||||
return fmt.Sprintf("%s|%d", observation.NodeID, observation.CapturedAt.UTC().UnixNano())
|
||||
}
|
||||
|
||||
func frpsKey(observation analyticsmodel.NodeObsFrps) string {
|
||||
return fmt.Sprintf("%s|%d", observation.NodeID, observation.CapturedAt.UTC().UnixNano())
|
||||
}
|
||||
|
||||
func frpcKey(observation analyticsmodel.NodeObsFrpc) string {
|
||||
return fmt.Sprintf("%s|%d", observation.NodeID, observation.CapturedAt.UTC().UnixNano())
|
||||
}
|
||||
|
||||
type batchStopper interface {
|
||||
Stop(ctx context.Context) error
|
||||
}
|
||||
|
||||
type statsProvider interface {
|
||||
Stats() batchwriter.Stats
|
||||
}
|
||||
|
||||
func running() bool {
|
||||
return metricSnapshotWriter != nil && metricSnapshotWriter.Running()
|
||||
}
|
||||
}
|
||||
|
||||
@@ -117,6 +117,16 @@ func buildOverviewView(ctx context.Context) (*OverviewView, error) {
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
// Latest-per-node health: dedicated LIMIT 1 BY queries (not a global raw LIMIT).
|
||||
latestSnapshotRows, err := model.ListOpenFlareLatestMetricSnapshotsSince(ctx, "", since)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
latestTrafficRows, err := model.ListOpenFlareLatestRequestReportsSince(ctx, "", since)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
// Bounded raw windows remain for distributions and trend fallbacks; trends prefer hourly rollups.
|
||||
snapshots, err := model.ListOpenFlareMetricSnapshotsSince(ctx, "", since, dashboardOverviewSnapshotLimit)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
@@ -137,27 +147,17 @@ func buildOverviewView(ctx context.Context) (*OverviewView, error) {
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
trafficTrend := observability.BuildTrafficTrendPoints(now, reports)
|
||||
if trafficHourly, hourlyErr := model.ListOpenFlareTrafficHourlySince(ctx, "", since); hourlyErr == nil && len(trafficHourly) > 0 {
|
||||
trafficTrend = observability.BuildTrafficTrendPointsFromHourly(now, trafficHourly)
|
||||
}
|
||||
|
||||
view := &OverviewView{
|
||||
GeneratedAt: now,
|
||||
Nodes: make([]NodeHealth, 0, len(nodes)),
|
||||
Distributions: observability.BuildTrafficDistributions(reports, accessLogRegions, dashboardDistributionLimit),
|
||||
Trends: observability.NodeTrends{
|
||||
Traffic24h: trafficTrend,
|
||||
Capacity24h: observability.BuildCapacityTrendPoints(now, snapshots),
|
||||
Network24h: observability.BuildNetworkTrendPoints(now, snapshots, openrestySnapshots),
|
||||
DiskIO24h: observability.BuildDiskIOTrendPoints(now, snapshots),
|
||||
},
|
||||
Trends: observability.BuildNodeTrends(ctx, now, "", snapshots, openrestySnapshots, reports),
|
||||
}
|
||||
|
||||
var cpuNodeCount int
|
||||
var memoryNodeCount int
|
||||
latestSnapshots := observability.LatestMetricSnapshotsByNode(snapshots)
|
||||
latestTrafficReports := observability.LatestTrafficReportsByNode(reports)
|
||||
latestSnapshots := observability.LatestMetricSnapshotsByNode(latestSnapshotRows)
|
||||
latestTrafficReports := observability.LatestTrafficReportsByNode(latestTrafficRows)
|
||||
activeEventsByNode := observability.ActiveHealthEventsByNode(activeEvents)
|
||||
|
||||
for _, node := range nodes {
|
||||
|
||||
@@ -58,6 +58,32 @@ func TestGetOverviewStructure(t *testing.T) {
|
||||
OpenrestyStatus: "unknown",
|
||||
}).Error)
|
||||
|
||||
// Seed older + newer snapshots per node; health must use latest-per-node, not a global raw limit.
|
||||
require.NoError(t, model.InsertOpenFlareMetricSnapshot(ctx, &model.OpenFlareMetricSnapshot{
|
||||
NodeID: "node-dashboard-1",
|
||||
CapturedAt: now.Add(-2 * time.Hour),
|
||||
CPUUsagePercent: 10,
|
||||
MemoryUsedBytes: 1,
|
||||
MemoryTotalBytes: 10,
|
||||
}))
|
||||
require.NoError(t, model.InsertOpenFlareMetricSnapshot(ctx, &model.OpenFlareMetricSnapshot{
|
||||
NodeID: "node-dashboard-1",
|
||||
CapturedAt: now.Add(-time.Minute),
|
||||
CPUUsagePercent: 55,
|
||||
MemoryUsedBytes: 5,
|
||||
MemoryTotalBytes: 10,
|
||||
StorageUsedBytes: 2,
|
||||
StorageTotalBytes: 10,
|
||||
}))
|
||||
require.NoError(t, model.InsertOpenFlareRequestReport(ctx, &model.OpenFlareRequestReport{
|
||||
NodeID: "node-dashboard-1",
|
||||
WindowStartedAt: now.Add(-2 * time.Minute),
|
||||
WindowEndedAt: now.Add(-time.Minute),
|
||||
RequestCount: 12,
|
||||
ErrorCount: 1,
|
||||
UniqueVisitorCount: 4,
|
||||
}))
|
||||
|
||||
overview, err := GetOverview(ctx)
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, overview)
|
||||
@@ -69,14 +95,14 @@ func TestGetOverviewStructure(t *testing.T) {
|
||||
assert.Equal(t, 0, overview.Summary.OfflineNodes)
|
||||
assert.Equal(t, 0, overview.Summary.UnhealthyNodes)
|
||||
|
||||
assert.Equal(t, int64(0), overview.Traffic.RequestCount)
|
||||
assert.Equal(t, int64(0), overview.Traffic.UniqueVisitors)
|
||||
assert.Equal(t, int64(0), overview.Traffic.ErrorCount)
|
||||
assert.Equal(t, float64(0), overview.Traffic.EstimatedQPS)
|
||||
assert.Equal(t, 0, overview.Traffic.ReportedNodes)
|
||||
assert.Equal(t, int64(12), overview.Traffic.RequestCount)
|
||||
assert.Equal(t, int64(4), overview.Traffic.UniqueVisitors)
|
||||
assert.Equal(t, int64(1), overview.Traffic.ErrorCount)
|
||||
assert.InDelta(t, 0.2, overview.Traffic.EstimatedQPS, 0.0001)
|
||||
assert.Equal(t, 1, overview.Traffic.ReportedNodes)
|
||||
|
||||
assert.Equal(t, float64(0), overview.Capacity.AverageCPUUsagePercent)
|
||||
assert.Equal(t, float64(0), overview.Capacity.AverageMemoryUsagePercent)
|
||||
assert.Equal(t, 55.0, overview.Capacity.AverageCPUUsagePercent)
|
||||
assert.Equal(t, 50.0, overview.Capacity.AverageMemoryUsagePercent)
|
||||
assert.Equal(t, 0, overview.Capacity.HighCPUNodes)
|
||||
assert.Equal(t, 0, overview.Capacity.HighMemoryNodes)
|
||||
assert.Equal(t, 0, overview.Capacity.HighStorageNodes)
|
||||
@@ -120,10 +146,20 @@ func TestGetOverviewStructure(t *testing.T) {
|
||||
assert.Equal(t, "Edge 1", onlineNode[2])
|
||||
assert.Equal(t, "online", onlineNode[6])
|
||||
assert.Equal(t, "healthy", onlineNode[7])
|
||||
// Latest-per-node health fields (indexes match compressDashboardNodes).
|
||||
assert.Equal(t, 55.0, onlineNode[11]) // cpu_usage_percent from latest snapshot
|
||||
assert.Equal(t, 50.0, onlineNode[12]) // memory_usage_percent
|
||||
assert.Equal(t, int64(12), onlineNode[14])
|
||||
assert.Equal(t, int64(1), onlineNode[15])
|
||||
assert.Equal(t, int64(4), onlineNode[16])
|
||||
|
||||
pendingNode := nodeByID["node-dashboard-2"]
|
||||
require.NotNil(t, pendingNode)
|
||||
assert.Equal(t, "Edge 2", pendingNode[2])
|
||||
assert.Equal(t, "pending", pendingNode[6])
|
||||
assert.Equal(t, "unknown", pendingNode[7])
|
||||
|
||||
assert.Equal(t, 55.0, overview.Capacity.AverageCPUUsagePercent)
|
||||
assert.Equal(t, 1, overview.Traffic.ReportedNodes)
|
||||
assert.Equal(t, int64(4), overview.Traffic.UniqueVisitors)
|
||||
}
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
package observability
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"sort"
|
||||
"strings"
|
||||
@@ -13,6 +14,7 @@ import (
|
||||
)
|
||||
|
||||
const observabilityTrendBuckets = 24
|
||||
const unknownTrendNodeKey = "__unknown__"
|
||||
|
||||
const (
|
||||
healthEventStatusActive = "active"
|
||||
@@ -133,6 +135,12 @@ type diskCounterState struct {
|
||||
seen bool
|
||||
}
|
||||
|
||||
type networkCounterState struct {
|
||||
rx int64
|
||||
tx int64
|
||||
seen bool
|
||||
}
|
||||
|
||||
func buildTrafficWindowSummary(report *model.OpenFlareRequestReport) *TrafficWindowSummary {
|
||||
if report == nil {
|
||||
return nil
|
||||
@@ -257,6 +265,44 @@ func buildHealthSummary(
|
||||
return summary
|
||||
}
|
||||
|
||||
// BuildNodeTrends builds 24h trend series, preferring ClickHouse hourly aggregates
|
||||
// over limited raw snapshot windows so capacity/network/disk charts stay complete.
|
||||
func BuildNodeTrends(
|
||||
ctx context.Context,
|
||||
now time.Time,
|
||||
nodeID string,
|
||||
snapshots []*model.OpenFlareMetricSnapshot,
|
||||
openrestyObs []*model.OpenFlareNodeObservationOpenresty,
|
||||
reports []*model.OpenFlareRequestReport,
|
||||
) NodeTrends {
|
||||
trendSince := now.Add(-24 * time.Hour)
|
||||
trafficTrend := BuildTrafficTrendPoints(now, reports)
|
||||
if trafficHourly, err := model.ListOpenFlareTrafficHourlySince(ctx, nodeID, trendSince); err == nil && len(trafficHourly) > 0 {
|
||||
trafficTrend = BuildTrafficTrendPointsFromHourly(now, trafficHourly)
|
||||
}
|
||||
|
||||
capacityTrend := BuildCapacityTrendPoints(now, snapshots)
|
||||
networkTrend := BuildNetworkTrendPoints(now, snapshots, openrestyObs)
|
||||
diskIOTrend := BuildDiskIOTrendPoints(now, snapshots)
|
||||
|
||||
metricHourly, metricErr := model.ListOpenFlareMetricHourlySince(ctx, nodeID, trendSince)
|
||||
if metricErr == nil && len(metricHourly) > 0 {
|
||||
capacityTrend = BuildCapacityTrendPointsFromHourly(now, metricHourly)
|
||||
diskIOTrend = BuildDiskIOTrendPointsFromHourly(now, metricHourly)
|
||||
}
|
||||
openrestyHourly, openrestyErr := model.ListOpenFlareOpenrestyHourlySince(ctx, nodeID, trendSince)
|
||||
if metricErr == nil && openrestyErr == nil && (len(metricHourly) > 0 || len(openrestyHourly) > 0) {
|
||||
networkTrend = BuildNetworkTrendPointsFromHourly(now, metricHourly, openrestyHourly)
|
||||
}
|
||||
|
||||
return NodeTrends{
|
||||
Traffic24h: trafficTrend,
|
||||
Capacity24h: capacityTrend,
|
||||
Network24h: networkTrend,
|
||||
DiskIO24h: diskIOTrend,
|
||||
}
|
||||
}
|
||||
|
||||
// BuildTrafficTrendPointsFromHourly builds 24h traffic trend buckets from hourly rollups.
|
||||
func BuildTrafficTrendPointsFromHourly(now time.Time, hourly []*model.OpenFlareTrafficHourly) []TrafficTrendPoint {
|
||||
start := trendWindowStart(now)
|
||||
@@ -336,7 +382,30 @@ func BuildCapacityTrendPoints(now time.Time, snapshots []*model.OpenFlareMetricS
|
||||
return points
|
||||
}
|
||||
|
||||
// BuildCapacityTrendPointsFromHourly builds 24h capacity trend buckets from hourly aggregates.
|
||||
func BuildCapacityTrendPointsFromHourly(now time.Time, hourly []*model.OpenFlareMetricHourly) []CapacityTrendPoint {
|
||||
start := trendWindowStart(now)
|
||||
points := make([]CapacityTrendPoint, observabilityTrendBuckets)
|
||||
for index := range points {
|
||||
points[index].BucketStartedAt = start.Add(time.Duration(index) * time.Hour)
|
||||
}
|
||||
for _, row := range hourly {
|
||||
if row == nil {
|
||||
continue
|
||||
}
|
||||
index, ok := trendBucketIndex(row.Hour, start)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
points[index].AverageCPUUsagePercent = row.AverageCPUUsagePercent
|
||||
points[index].AverageMemoryUsagePercent = row.AverageMemoryUsagePercent
|
||||
points[index].ReportedNodes = row.ReportedNodes
|
||||
}
|
||||
return points
|
||||
}
|
||||
|
||||
// BuildNetworkTrendPoints builds 24h network trend buckets.
|
||||
// Host and OpenResty counters are cumulative; values are consecutive deltas.
|
||||
func BuildNetworkTrendPoints(
|
||||
now time.Time,
|
||||
snapshots []*model.OpenFlareMetricSnapshot,
|
||||
@@ -349,24 +418,70 @@ func BuildNetworkTrendPoints(
|
||||
points[index].BucketStartedAt = start.Add(time.Duration(index) * time.Hour)
|
||||
accumulators[index].nodes = make(map[string]struct{})
|
||||
}
|
||||
sort.Slice(snapshots, func(i int, j int) bool {
|
||||
if snapshots[i].CapturedAt.Equal(snapshots[j].CapturedAt) {
|
||||
return snapshots[i].NodeID < snapshots[j].NodeID
|
||||
}
|
||||
return snapshots[i].CapturedAt.Before(snapshots[j].CapturedAt)
|
||||
})
|
||||
previousHostByNode := make(map[string]networkCounterState, len(snapshots))
|
||||
for _, snapshot := range snapshots {
|
||||
if snapshot == nil {
|
||||
continue
|
||||
}
|
||||
nodeKey := snapshot.NodeID
|
||||
if nodeKey == "" {
|
||||
nodeKey = unknownTrendNodeKey
|
||||
}
|
||||
previous := previousHostByNode[nodeKey]
|
||||
previousHostByNode[nodeKey] = networkCounterState{
|
||||
rx: snapshot.NetworkRxBytes,
|
||||
tx: snapshot.NetworkTxBytes,
|
||||
seen: true,
|
||||
}
|
||||
if !previous.seen {
|
||||
continue
|
||||
}
|
||||
index, ok := trendBucketIndex(snapshot.CapturedAt, start)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
points[index].NetworkRxBytes += snapshot.NetworkRxBytes
|
||||
points[index].NetworkTxBytes += snapshot.NetworkTxBytes
|
||||
points[index].NetworkRxBytes += nonNegativeDelta(snapshot.NetworkRxBytes, previous.rx)
|
||||
points[index].NetworkTxBytes += nonNegativeDelta(snapshot.NetworkTxBytes, previous.tx)
|
||||
if snapshot.NodeID != "" {
|
||||
accumulators[index].nodes[snapshot.NodeID] = struct{}{}
|
||||
}
|
||||
}
|
||||
sort.Slice(openrestyObs, func(i int, j int) bool {
|
||||
if openrestyObs[i].CapturedAt.Equal(openrestyObs[j].CapturedAt) {
|
||||
return openrestyObs[i].NodeID < openrestyObs[j].NodeID
|
||||
}
|
||||
return openrestyObs[i].CapturedAt.Before(openrestyObs[j].CapturedAt)
|
||||
})
|
||||
previousOpenrestyByNode := make(map[string]networkCounterState, len(openrestyObs))
|
||||
for _, obs := range openrestyObs {
|
||||
if obs == nil {
|
||||
continue
|
||||
}
|
||||
nodeKey := obs.NodeID
|
||||
if nodeKey == "" {
|
||||
nodeKey = unknownTrendNodeKey
|
||||
}
|
||||
previous := previousOpenrestyByNode[nodeKey]
|
||||
previousOpenrestyByNode[nodeKey] = networkCounterState{
|
||||
rx: obs.OpenrestyRxBytes,
|
||||
tx: obs.OpenrestyTxBytes,
|
||||
seen: true,
|
||||
}
|
||||
if !previous.seen {
|
||||
continue
|
||||
}
|
||||
index, ok := trendBucketIndex(obs.CapturedAt, start)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
points[index].OpenrestyRxBytes += obs.OpenrestyRxBytes
|
||||
points[index].OpenrestyTxBytes += obs.OpenrestyTxBytes
|
||||
points[index].OpenrestyRxBytes += nonNegativeDelta(obs.OpenrestyRxBytes, previous.rx)
|
||||
points[index].OpenrestyTxBytes += nonNegativeDelta(obs.OpenrestyTxBytes, previous.tx)
|
||||
if obs.NodeID != "" {
|
||||
accumulators[index].nodes[obs.NodeID] = struct{}{}
|
||||
}
|
||||
@@ -377,6 +492,48 @@ func BuildNetworkTrendPoints(
|
||||
return points
|
||||
}
|
||||
|
||||
// BuildNetworkTrendPointsFromHourly builds 24h network trend buckets from hourly aggregates.
|
||||
func BuildNetworkTrendPointsFromHourly(
|
||||
now time.Time,
|
||||
metricHourly []*model.OpenFlareMetricHourly,
|
||||
openrestyHourly []*model.OpenFlareOpenrestyHourly,
|
||||
) []NetworkTrendPoint {
|
||||
start := trendWindowStart(now)
|
||||
points := make([]NetworkTrendPoint, observabilityTrendBuckets)
|
||||
for index := range points {
|
||||
points[index].BucketStartedAt = start.Add(time.Duration(index) * time.Hour)
|
||||
}
|
||||
for _, row := range metricHourly {
|
||||
if row == nil {
|
||||
continue
|
||||
}
|
||||
index, ok := trendBucketIndex(row.Hour, start)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
points[index].NetworkRxBytes += row.NetworkRxBytes
|
||||
points[index].NetworkTxBytes += row.NetworkTxBytes
|
||||
if row.ReportedNodes > points[index].ReportedNodes {
|
||||
points[index].ReportedNodes = row.ReportedNodes
|
||||
}
|
||||
}
|
||||
for _, row := range openrestyHourly {
|
||||
if row == nil {
|
||||
continue
|
||||
}
|
||||
index, ok := trendBucketIndex(row.Hour, start)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
points[index].OpenrestyRxBytes += row.OpenrestyRxBytes
|
||||
points[index].OpenrestyTxBytes += row.OpenrestyTxBytes
|
||||
if row.ReportedNodes > points[index].ReportedNodes {
|
||||
points[index].ReportedNodes = row.ReportedNodes
|
||||
}
|
||||
}
|
||||
return points
|
||||
}
|
||||
|
||||
// BuildDiskIOTrendPoints builds 24h disk IO trend buckets.
|
||||
func BuildDiskIOTrendPoints(now time.Time, snapshots []*model.OpenFlareMetricSnapshot) []DiskIOTrendPoint {
|
||||
start := trendWindowStart(now)
|
||||
@@ -396,7 +553,7 @@ func BuildDiskIOTrendPoints(now time.Time, snapshots []*model.OpenFlareMetricSna
|
||||
for _, snapshot := range snapshots {
|
||||
nodeKey := snapshot.NodeID
|
||||
if nodeKey == "" {
|
||||
nodeKey = "__unknown__"
|
||||
nodeKey = unknownTrendNodeKey
|
||||
}
|
||||
previous := previousByNode[nodeKey]
|
||||
previousByNode[nodeKey] = diskCounterState{
|
||||
@@ -411,16 +568,8 @@ func BuildDiskIOTrendPoints(now time.Time, snapshots []*model.OpenFlareMetricSna
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
readDelta := snapshot.DiskReadBytes - previous.read
|
||||
writeDelta := snapshot.DiskWriteBytes - previous.write
|
||||
if readDelta < 0 {
|
||||
readDelta = 0
|
||||
}
|
||||
if writeDelta < 0 {
|
||||
writeDelta = 0
|
||||
}
|
||||
points[index].DiskReadBytes += readDelta
|
||||
points[index].DiskWriteBytes += writeDelta
|
||||
points[index].DiskReadBytes += nonNegativeDelta(snapshot.DiskReadBytes, previous.read)
|
||||
points[index].DiskWriteBytes += nonNegativeDelta(snapshot.DiskWriteBytes, previous.write)
|
||||
if snapshot.NodeID != "" {
|
||||
accumulators[index].nodes[snapshot.NodeID] = struct{}{}
|
||||
}
|
||||
@@ -431,6 +580,36 @@ func BuildDiskIOTrendPoints(now time.Time, snapshots []*model.OpenFlareMetricSna
|
||||
return points
|
||||
}
|
||||
|
||||
// BuildDiskIOTrendPointsFromHourly builds 24h disk IO trend buckets from hourly aggregates.
|
||||
func BuildDiskIOTrendPointsFromHourly(now time.Time, hourly []*model.OpenFlareMetricHourly) []DiskIOTrendPoint {
|
||||
start := trendWindowStart(now)
|
||||
points := make([]DiskIOTrendPoint, observabilityTrendBuckets)
|
||||
for index := range points {
|
||||
points[index].BucketStartedAt = start.Add(time.Duration(index) * time.Hour)
|
||||
}
|
||||
for _, row := range hourly {
|
||||
if row == nil {
|
||||
continue
|
||||
}
|
||||
index, ok := trendBucketIndex(row.Hour, start)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
points[index].DiskReadBytes += row.DiskReadBytes
|
||||
points[index].DiskWriteBytes += row.DiskWriteBytes
|
||||
points[index].ReportedNodes = row.ReportedNodes
|
||||
}
|
||||
return points
|
||||
}
|
||||
|
||||
func nonNegativeDelta(current int64, previous int64) int64 {
|
||||
delta := current - previous
|
||||
if delta < 0 {
|
||||
return 0
|
||||
}
|
||||
return delta
|
||||
}
|
||||
|
||||
func latestMetricSnapshot(snapshots []*model.OpenFlareMetricSnapshot) *model.OpenFlareMetricSnapshot {
|
||||
var latest *model.OpenFlareMetricSnapshot
|
||||
for _, snapshot := range snapshots {
|
||||
|
||||
@@ -132,3 +132,84 @@ func TestBuildTrafficWindowSummaryNilWithoutReport(t *testing.T) {
|
||||
t.Fatalf("buildTrafficWindowSummary(nil) = %#v, want nil", summary)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildCapacityTrendPointsFromHourlyFillsBuckets(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
now := time.Date(2026, 7, 10, 9, 30, 0, 0, time.UTC)
|
||||
hourly := []*model.OpenFlareMetricHourly{
|
||||
{
|
||||
Hour: now.Add(-3 * time.Hour).Truncate(time.Hour),
|
||||
AverageCPUUsagePercent: 42.5,
|
||||
AverageMemoryUsagePercent: 61.2,
|
||||
ReportedNodes: 1,
|
||||
},
|
||||
{
|
||||
Hour: now.Truncate(time.Hour),
|
||||
AverageCPUUsagePercent: 12.0,
|
||||
AverageMemoryUsagePercent: 50.0,
|
||||
ReportedNodes: 2,
|
||||
},
|
||||
}
|
||||
|
||||
points := BuildCapacityTrendPointsFromHourly(now, hourly)
|
||||
if len(points) != observabilityTrendBuckets {
|
||||
t.Fatalf("len = %d, want %d", len(points), observabilityTrendBuckets)
|
||||
}
|
||||
if points[len(points)-4].AverageCPUUsagePercent != 42.5 {
|
||||
t.Fatalf("hour-3 cpu = %v, want 42.5", points[len(points)-4].AverageCPUUsagePercent)
|
||||
}
|
||||
if points[len(points)-1].ReportedNodes != 2 {
|
||||
t.Fatalf("current hour reported_nodes = %d, want 2", points[len(points)-1].ReportedNodes)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildNetworkTrendPointsUsesCounterDeltas(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
now := time.Date(2026, 7, 10, 9, 30, 0, 0, time.UTC)
|
||||
base := now.Truncate(time.Hour)
|
||||
snapshots := []*model.OpenFlareMetricSnapshot{
|
||||
{NodeID: "n1", CapturedAt: base.Add(10 * time.Minute), NetworkRxBytes: 1000, NetworkTxBytes: 2000},
|
||||
{NodeID: "n1", CapturedAt: base.Add(20 * time.Minute), NetworkRxBytes: 1500, NetworkTxBytes: 2600},
|
||||
}
|
||||
openrestyObs := []*model.OpenFlareNodeObservationOpenresty{
|
||||
{NodeID: "n1", CapturedAt: base.Add(10 * time.Minute), OpenrestyRxBytes: 100, OpenrestyTxBytes: 200},
|
||||
{NodeID: "n1", CapturedAt: base.Add(20 * time.Minute), OpenrestyRxBytes: 180, OpenrestyTxBytes: 250},
|
||||
}
|
||||
|
||||
points := BuildNetworkTrendPoints(now, snapshots, openrestyObs)
|
||||
current := points[len(points)-1]
|
||||
if current.NetworkRxBytes != 500 {
|
||||
t.Fatalf("network_rx_bytes = %d, want 500", current.NetworkRxBytes)
|
||||
}
|
||||
if current.NetworkTxBytes != 600 {
|
||||
t.Fatalf("network_tx_bytes = %d, want 600", current.NetworkTxBytes)
|
||||
}
|
||||
if current.OpenrestyRxBytes != 80 {
|
||||
t.Fatalf("openresty_rx_bytes = %d, want 80", current.OpenrestyRxBytes)
|
||||
}
|
||||
if current.OpenrestyTxBytes != 50 {
|
||||
t.Fatalf("openresty_tx_bytes = %d, want 50", current.OpenrestyTxBytes)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildDiskIOTrendPointsFromHourlyFillsBuckets(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
now := time.Date(2026, 7, 10, 9, 30, 0, 0, time.UTC)
|
||||
hourly := []*model.OpenFlareMetricHourly{
|
||||
{
|
||||
Hour: now.Add(-1 * time.Hour).Truncate(time.Hour),
|
||||
DiskReadBytes: 1024,
|
||||
DiskWriteBytes: 2048,
|
||||
ReportedNodes: 1,
|
||||
},
|
||||
}
|
||||
|
||||
points := BuildDiskIOTrendPointsFromHourly(now, hourly)
|
||||
prev := points[len(points)-2]
|
||||
if prev.DiskReadBytes != 1024 || prev.DiskWriteBytes != 2048 {
|
||||
t.Fatalf("previous hour disk io = %#v, want read=1024 write=2048", prev)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -134,11 +134,6 @@ func GetNodeObservability(ctx context.Context, id uint, query NodeQuery) (*NodeV
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
trafficTrend := BuildTrafficTrendPoints(now, reports)
|
||||
if trafficHourly, hourlyErr := model.ListOpenFlareTrafficHourlySince(ctx, node.NodeID, now.Add(-24*time.Hour)); hourlyErr == nil && len(trafficHourly) > 0 {
|
||||
trafficTrend = BuildTrafficTrendPointsFromHourly(now, trafficHourly)
|
||||
}
|
||||
|
||||
view := &NodeView{
|
||||
NodeID: node.NodeID,
|
||||
Profile: profile,
|
||||
@@ -150,12 +145,7 @@ func GetNodeObservability(ctx context.Context, id uint, query NodeQuery) (*NodeV
|
||||
Distributions: BuildTrafficDistributions(reports, accessLogRegions, defaultTrafficDistributionLimit),
|
||||
Health: buildHealthSummary(latestMetricSnapshot(snapshots), latestTrafficReport(reports), events),
|
||||
},
|
||||
Trends: NodeTrends{
|
||||
Traffic24h: trafficTrend,
|
||||
Capacity24h: BuildCapacityTrendPoints(now, snapshots),
|
||||
Network24h: BuildNetworkTrendPoints(now, snapshots, openrestyObs),
|
||||
DiskIO24h: BuildDiskIOTrendPoints(now, snapshots),
|
||||
},
|
||||
Trends: BuildNodeTrends(ctx, now, node.NodeID, snapshots, openrestyObs, reports),
|
||||
}
|
||||
if node.NodeType == "tunnel_relay" {
|
||||
frpsObs, frpsErr := model.ListOpenFlareNodeObservationFrps(ctx, node.NodeID, time.Time{}, 1)
|
||||
|
||||
@@ -59,6 +59,9 @@ type databaseCleanupResult struct {
|
||||
Target string `json:"target"`
|
||||
TargetLabel string `json:"target_label"`
|
||||
DeletedCount int64 `json:"deleted_count"`
|
||||
EligibleCount int64 `json:"eligible_count,omitempty"`
|
||||
CleanupMode string `json:"cleanup_mode,omitempty"`
|
||||
TableTTLDays int `json:"table_ttl_days,omitempty"`
|
||||
DeleteAll bool `json:"delete_all"`
|
||||
RetentionDays *int `json:"retention_days,omitempty"`
|
||||
}
|
||||
@@ -198,6 +201,9 @@ func cleanupDatabaseObservability(ctx context.Context, input databaseCleanupInpu
|
||||
Target: result.Target,
|
||||
TargetLabel: result.TargetLabel,
|
||||
DeletedCount: result.DeletedCount,
|
||||
EligibleCount: result.EligibleCount,
|
||||
CleanupMode: result.CleanupMode,
|
||||
TableTTLDays: result.TableTTLDays,
|
||||
DeleteAll: result.DeleteAll,
|
||||
RetentionDays: result.RetentionDays,
|
||||
}, nil
|
||||
|
||||
@@ -146,21 +146,26 @@ func TestCleanupDatabaseObservabilityDeletesRows(t *testing.T) {
|
||||
},
|
||||
}))
|
||||
|
||||
retention := 7
|
||||
result, err := cleanupDatabaseObservability(ctx, databaseCleanupInput{
|
||||
// Retention shorter than table TTL (90d for access logs) must be rejected.
|
||||
shortRetention := 7
|
||||
_, err := cleanupDatabaseObservability(ctx, databaseCleanupInput{
|
||||
Target: "node_access_logs",
|
||||
RetentionDays: &retention,
|
||||
RetentionDays: &shortRetention,
|
||||
})
|
||||
require.Error(t, err)
|
||||
|
||||
// Full truncate still hard-deletes all rows.
|
||||
result, err := cleanupDatabaseObservability(ctx, databaseCleanupInput{
|
||||
Target: "node_access_logs",
|
||||
})
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, "node_access_logs", result.Target)
|
||||
assert.Equal(t, "访问日志", result.TargetLabel)
|
||||
assert.Equal(t, int64(1), result.DeletedCount)
|
||||
assert.False(t, result.DeleteAll)
|
||||
require.NotNil(t, result.RetentionDays)
|
||||
assert.Equal(t, 7, *result.RetentionDays)
|
||||
assert.Equal(t, int64(2), result.DeletedCount)
|
||||
assert.True(t, result.DeleteAll)
|
||||
assert.Equal(t, "truncate", result.CleanupMode)
|
||||
|
||||
rows, err := model.ListOpenFlareAccessLogs(ctx, model.OpenFlareAccessLogQuery{Page: 0, PageSize: 10})
|
||||
require.NoError(t, err)
|
||||
require.Len(t, rows, 1)
|
||||
assert.Equal(t, "/recent", rows[0].Path)
|
||||
assert.Empty(t, rows)
|
||||
}
|
||||
|
||||
@@ -34,9 +34,19 @@ var databaseCleanupTargets = map[string]string{
|
||||
DatabaseCleanupTargetAccessLogs: "访问日志",
|
||||
DatabaseCleanupTargetMetricSnapshots: "性能快照",
|
||||
DatabaseCleanupTargetRequestReports: "请求聚合",
|
||||
DatabaseCleanupTargetObsOpenresty: "OpenResty 观测",
|
||||
DatabaseCleanupTargetObsFrps: "FRPS 观测",
|
||||
DatabaseCleanupTargetObsFrpc: "FRPC 观测",
|
||||
DatabaseCleanupTargetObsOpenresty: "OpenResty 观测",
|
||||
DatabaseCleanupTargetObsFrps: "FRPS 观测",
|
||||
DatabaseCleanupTargetObsFrpc: "FRPC 观测",
|
||||
}
|
||||
|
||||
// databaseCleanupTableTTLDays maps API targets to ClickHouse DDL TTL days.
|
||||
var databaseCleanupTableTTLDays = map[string]int{
|
||||
DatabaseCleanupTargetAccessLogs: analyticsrepo.TableTTLDaysNodeAccessLogs,
|
||||
DatabaseCleanupTargetMetricSnapshots: analyticsrepo.TableTTLDaysNodeMetricSnapshots,
|
||||
DatabaseCleanupTargetRequestReports: analyticsrepo.TableTTLDaysNodeRequestReports,
|
||||
DatabaseCleanupTargetObsOpenresty: analyticsrepo.TableTTLDaysNodeObs,
|
||||
DatabaseCleanupTargetObsFrps: analyticsrepo.TableTTLDaysNodeObs,
|
||||
DatabaseCleanupTargetObsFrpc: analyticsrepo.TableTTLDaysNodeObs,
|
||||
}
|
||||
|
||||
// DatabaseCleanupInput describes a manual observability cleanup request.
|
||||
@@ -46,14 +56,21 @@ type DatabaseCleanupInput struct {
|
||||
}
|
||||
|
||||
// DatabaseCleanupResult summarizes a manual observability cleanup run.
|
||||
//
|
||||
// Semantics:
|
||||
// - delete_all / cleanup_mode=truncate: DeletedCount is hard-deleted rows (TRUNCATE).
|
||||
// - retention path / cleanup_mode=ttl_materialize: DeletedCount is always 0;
|
||||
// EligibleCount estimates rows past the table DDL TTL (not an arbitrary younger cutoff).
|
||||
type DatabaseCleanupResult struct {
|
||||
Target string `json:"target"`
|
||||
TargetLabel string `json:"target_label"`
|
||||
DeletedCount int64 `json:"deleted_count"`
|
||||
CleanupMode string `json:"cleanup_mode,omitempty"`
|
||||
DeleteAll bool `json:"delete_all"`
|
||||
RetentionDays *int `json:"retention_days,omitempty"`
|
||||
Cutoff *time.Time `json:"cutoff,omitempty"`
|
||||
Target string `json:"target"`
|
||||
TargetLabel string `json:"target_label"`
|
||||
DeletedCount int64 `json:"deleted_count"`
|
||||
EligibleCount int64 `json:"eligible_count,omitempty"`
|
||||
CleanupMode string `json:"cleanup_mode,omitempty"`
|
||||
TableTTLDays int `json:"table_ttl_days,omitempty"`
|
||||
DeleteAll bool `json:"delete_all"`
|
||||
RetentionDays *int `json:"retention_days,omitempty"`
|
||||
Cutoff *time.Time `json:"cutoff,omitempty"`
|
||||
}
|
||||
|
||||
// DatabaseAutoCleanupSummary summarizes a scheduled auto-cleanup run.
|
||||
@@ -63,7 +80,17 @@ type DatabaseAutoCleanupSummary struct {
|
||||
Results []DatabaseCleanupResult `json:"results"`
|
||||
}
|
||||
|
||||
// TableTTLDaysForCleanupTarget returns the DDL TTL days for a cleanup target.
|
||||
func TableTTLDaysForCleanupTarget(target string) (int, bool) {
|
||||
days, ok := databaseCleanupTableTTLDays[strings.TrimSpace(target)]
|
||||
return days, ok
|
||||
}
|
||||
|
||||
// CleanupDatabaseObservability deletes observability rows for the given target.
|
||||
//
|
||||
// When RetentionDays is nil, rows are hard-deleted via TRUNCATE.
|
||||
// When RetentionDays is set, ClickHouse only force-materializes the table TTL policy:
|
||||
// retention_days shorter than the table TTL is rejected (do not fake success).
|
||||
func CleanupDatabaseObservability(ctx context.Context, input DatabaseCleanupInput) (*DatabaseCleanupResult, error) {
|
||||
target := strings.TrimSpace(input.Target)
|
||||
targetLabel, ok := databaseCleanupTargets[target]
|
||||
@@ -74,10 +101,12 @@ func CleanupDatabaseObservability(ctx context.Context, input DatabaseCleanupInpu
|
||||
return nil, errors.New("retention_days 必须为大于 0 的整数")
|
||||
}
|
||||
|
||||
tableTTLDays := databaseCleanupTableTTLDays[target]
|
||||
result := &DatabaseCleanupResult{
|
||||
Target: target,
|
||||
TargetLabel: targetLabel,
|
||||
DeleteAll: input.RetentionDays == nil,
|
||||
Target: target,
|
||||
TargetLabel: targetLabel,
|
||||
DeleteAll: input.RetentionDays == nil,
|
||||
TableTTLDays: tableTTLDays,
|
||||
}
|
||||
|
||||
if input.RetentionDays == nil {
|
||||
@@ -86,24 +115,37 @@ func CleanupDatabaseObservability(ctx context.Context, input DatabaseCleanupInpu
|
||||
return nil, err
|
||||
}
|
||||
result.DeletedCount = deleted
|
||||
result.EligibleCount = deleted
|
||||
result.CleanupMode = mode
|
||||
return result, nil
|
||||
}
|
||||
|
||||
retentionDays := *input.RetentionDays
|
||||
cutoff := time.Now().UTC().Add(-time.Duration(retentionDays) * 24 * time.Hour)
|
||||
deleted, mode, err := deleteObservabilityRowsBefore(ctx, target, cutoff)
|
||||
if retentionDays < tableTTLDays {
|
||||
return nil, fmt.Errorf(
|
||||
"retention_days 不能小于表 TTL(%d 天);ClickHouse 仅支持按表 TTL 物化过期,更短保留请使用清空全部或调整 DDL",
|
||||
tableTTLDays,
|
||||
)
|
||||
}
|
||||
|
||||
// MATERIALIZE TTL only enforces DDL policy; cutoff reported is the table TTL boundary.
|
||||
tableCutoff := time.Now().UTC().Add(-time.Duration(tableTTLDays) * 24 * time.Hour)
|
||||
eligible, mode, err := materializeObservabilityTableTTL(ctx, target)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
result.DeletedCount = deleted
|
||||
result.DeletedCount = 0
|
||||
result.EligibleCount = eligible
|
||||
result.CleanupMode = mode
|
||||
result.RetentionDays = &retentionDays
|
||||
result.Cutoff = &cutoff
|
||||
result.Cutoff = &tableCutoff
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// RunDatabaseAutoCleanupOnce runs retention-based cleanup for all observability targets.
|
||||
//
|
||||
// Configured retention shorter than a target's table TTL is clamped up to the table TTL
|
||||
// so the scheduled job can force-materialize each table policy without failing.
|
||||
func RunDatabaseAutoCleanupOnce(ctx context.Context, now time.Time) (*DatabaseAutoCleanupSummary, error) {
|
||||
enabled, err := repository.GetBoolByKey(ctx, model.ConfigKeyDatabaseAutoCleanupEnabled)
|
||||
if err != nil {
|
||||
@@ -128,9 +170,13 @@ func RunDatabaseAutoCleanupOnce(ctx context.Context, now time.Time) (*DatabaseAu
|
||||
DatabaseCleanupTargetObsFrps,
|
||||
DatabaseCleanupTargetObsFrpc,
|
||||
} {
|
||||
effectiveDays := retentionDays
|
||||
if ttl, ok := databaseCleanupTableTTLDays[target]; ok && effectiveDays < ttl {
|
||||
effectiveDays = ttl
|
||||
}
|
||||
result, err := CleanupDatabaseObservability(ctx, DatabaseCleanupInput{
|
||||
Target: target,
|
||||
RetentionDays: &retentionDays,
|
||||
RetentionDays: &effectiveDays,
|
||||
})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
@@ -172,29 +218,37 @@ func deleteAllObservabilityRows(ctx context.Context, target string) (int64, stri
|
||||
return deleted, analyticsrepo.CleanupModeTruncate, nil
|
||||
}
|
||||
|
||||
func deleteObservabilityRowsBefore(ctx context.Context, target string, cutoff time.Time) (int64, string, error) {
|
||||
// materializeObservabilityTableTTL triggers table-TTL materialize (or memory-store delete-before
|
||||
// with the table TTL cutoff for tests) and returns the eligible/estimate row count.
|
||||
func materializeObservabilityTableTTL(ctx context.Context, target string) (int64, string, error) {
|
||||
ttlDays, ok := databaseCleanupTableTTLDays[target]
|
||||
if !ok {
|
||||
return 0, "", errors.New("unsupported cleanup target")
|
||||
}
|
||||
cutoff := time.Now().UTC().Add(-time.Duration(ttlDays) * 24 * time.Hour)
|
||||
|
||||
var (
|
||||
deleted int64
|
||||
err error
|
||||
eligible int64
|
||||
err error
|
||||
)
|
||||
switch target {
|
||||
case DatabaseCleanupTargetAccessLogs:
|
||||
deleted, err = model.DeleteOpenFlareAccessLogsBefore(ctx, cutoff)
|
||||
eligible, err = model.DeleteOpenFlareAccessLogsBefore(ctx, cutoff)
|
||||
case DatabaseCleanupTargetMetricSnapshots:
|
||||
deleted, err = model.DeleteOpenFlareMetricSnapshotsBefore(ctx, cutoff)
|
||||
eligible, err = model.DeleteOpenFlareMetricSnapshotsBefore(ctx, cutoff)
|
||||
case DatabaseCleanupTargetRequestReports:
|
||||
deleted, err = model.DeleteOpenFlareRequestReportsBefore(ctx, cutoff)
|
||||
eligible, err = model.DeleteOpenFlareRequestReportsBefore(ctx, cutoff)
|
||||
case DatabaseCleanupTargetObsOpenresty:
|
||||
deleted, err = model.DeleteOpenFlareNodeObservationOpenrestyBefore(ctx, cutoff)
|
||||
eligible, err = model.DeleteOpenFlareNodeObservationOpenrestyBefore(ctx, cutoff)
|
||||
case DatabaseCleanupTargetObsFrps:
|
||||
deleted, err = model.DeleteOpenFlareNodeObservationFrpsBefore(ctx, cutoff)
|
||||
eligible, err = model.DeleteOpenFlareNodeObservationFrpsBefore(ctx, cutoff)
|
||||
case DatabaseCleanupTargetObsFrpc:
|
||||
deleted, err = model.DeleteOpenFlareNodeObservationFrpcBefore(ctx, cutoff)
|
||||
eligible, err = model.DeleteOpenFlareNodeObservationFrpcBefore(ctx, cutoff)
|
||||
default:
|
||||
return 0, "", errors.New("unsupported cleanup target")
|
||||
}
|
||||
if err != nil {
|
||||
return 0, "", err
|
||||
}
|
||||
return deleted, analyticsrepo.CleanupModeTTLMaterialize, nil
|
||||
return eligible, analyticsrepo.CleanupModeTTLMaterialize, nil
|
||||
}
|
||||
|
||||
@@ -11,6 +11,7 @@ import (
|
||||
"github.com/Rain-kl/Wavelet/internal/db"
|
||||
"github.com/Rain-kl/Wavelet/internal/model"
|
||||
"github.com/Rain-kl/Wavelet/internal/repository"
|
||||
analyticsrepo "github.com/Rain-kl/Wavelet/internal/repository/analytics"
|
||||
"github.com/glebarez/sqlite"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
@@ -36,13 +37,41 @@ func setupDatabaseCleanupTestDB(t *testing.T) context.Context {
|
||||
return context.Background()
|
||||
}
|
||||
|
||||
func TestCleanupDatabaseObservabilityDeletesTargetedRows(t *testing.T) {
|
||||
func TestCleanupDatabaseObservabilityRejectsRetentionShorterThanTableTTL(t *testing.T) {
|
||||
ctx := setupDatabaseCleanupTestDB(t)
|
||||
|
||||
retentionDays := 7 // metric snapshots DDL TTL is 30 days
|
||||
result, err := CleanupDatabaseObservability(ctx, DatabaseCleanupInput{
|
||||
Target: DatabaseCleanupTargetMetricSnapshots,
|
||||
RetentionDays: &retentionDays,
|
||||
})
|
||||
require.Error(t, err)
|
||||
assert.Nil(t, result)
|
||||
assert.Contains(t, err.Error(), "不能小于表 TTL")
|
||||
assert.Contains(t, err.Error(), "30")
|
||||
}
|
||||
|
||||
func TestCleanupDatabaseObservabilityRejectsAccessLogRetentionShorterThanTableTTL(t *testing.T) {
|
||||
ctx := setupDatabaseCleanupTestDB(t)
|
||||
|
||||
retentionDays := 30 // access logs DDL TTL is 90 days
|
||||
result, err := CleanupDatabaseObservability(ctx, DatabaseCleanupInput{
|
||||
Target: DatabaseCleanupTargetAccessLogs,
|
||||
RetentionDays: &retentionDays,
|
||||
})
|
||||
require.Error(t, err)
|
||||
assert.Nil(t, result)
|
||||
assert.Contains(t, err.Error(), "90")
|
||||
}
|
||||
|
||||
func TestCleanupDatabaseObservabilityMaterializeDoesNotClaimHardDelete(t *testing.T) {
|
||||
ctx := setupDatabaseCleanupTestDB(t)
|
||||
now := time.Now().UTC()
|
||||
|
||||
// One row past metric table TTL (30d), one still inside the window.
|
||||
require.NoError(t, model.InsertOpenFlareMetricSnapshot(ctx, &model.OpenFlareMetricSnapshot{
|
||||
NodeID: "node-a",
|
||||
CapturedAt: now.Add(-10 * 24 * time.Hour),
|
||||
CapturedAt: now.Add(-40 * 24 * time.Hour),
|
||||
CPUUsagePercent: 10,
|
||||
}))
|
||||
require.NoError(t, model.InsertOpenFlareMetricSnapshot(ctx, &model.OpenFlareMetricSnapshot{
|
||||
@@ -51,15 +80,22 @@ func TestCleanupDatabaseObservabilityDeletesTargetedRows(t *testing.T) {
|
||||
CPUUsagePercent: 20,
|
||||
}))
|
||||
|
||||
retentionDays := 7
|
||||
retentionDays := analyticsrepo.TableTTLDaysNodeMetricSnapshots
|
||||
result, err := CleanupDatabaseObservability(ctx, DatabaseCleanupInput{
|
||||
Target: DatabaseCleanupTargetMetricSnapshots,
|
||||
RetentionDays: &retentionDays,
|
||||
})
|
||||
require.NoError(t, err)
|
||||
assert.False(t, result.DeleteAll)
|
||||
assert.Equal(t, int64(1), result.DeletedCount)
|
||||
assert.Equal(t, analyticsrepo.CleanupModeTTLMaterialize, result.CleanupMode)
|
||||
assert.Equal(t, analyticsrepo.TableTTLDaysNodeMetricSnapshots, result.TableTTLDays)
|
||||
// MATERIALIZE is not a counted hard delete.
|
||||
assert.Equal(t, int64(0), result.DeletedCount)
|
||||
assert.Equal(t, int64(1), result.EligibleCount)
|
||||
require.NotNil(t, result.Cutoff)
|
||||
assert.True(t, result.Cutoff.Before(now.Add(-29*24*time.Hour)))
|
||||
|
||||
// Memory store applies the table-TTL cutoff for tests; only the recent row remains.
|
||||
rows, err := model.ListOpenFlareMetricSnapshotsSince(ctx, "", time.Time{}, 0)
|
||||
require.NoError(t, err)
|
||||
require.Len(t, rows, 1)
|
||||
@@ -94,20 +130,23 @@ func TestCleanupDatabaseObservabilityDeletesAllRowsWhenRetentionMissing(t *testi
|
||||
})
|
||||
require.NoError(t, err)
|
||||
assert.True(t, result.DeleteAll)
|
||||
assert.Equal(t, analyticsrepo.CleanupModeTruncate, result.CleanupMode)
|
||||
assert.Equal(t, int64(2), result.DeletedCount)
|
||||
assert.Equal(t, int64(2), result.EligibleCount)
|
||||
|
||||
rows, err := model.ListOpenFlareAccessLogs(ctx, model.OpenFlareAccessLogQuery{Page: 0, PageSize: 10})
|
||||
require.NoError(t, err)
|
||||
assert.Empty(t, rows)
|
||||
}
|
||||
|
||||
func TestRunDatabaseAutoCleanupOnceDeletesAllObservabilityTargets(t *testing.T) {
|
||||
func TestRunDatabaseAutoCleanupOnceClampsRetentionToTableTTL(t *testing.T) {
|
||||
ctx := setupDatabaseCleanupTestDB(t)
|
||||
now := time.Now().UTC()
|
||||
|
||||
// Access logs TTL=90d, metrics TTL=30d. Config retention=1 must clamp, not reject.
|
||||
require.NoError(t, model.InsertOpenFlareAccessLogsBatch(ctx, []*model.OpenFlareAccessLog{{
|
||||
NodeID: "node-a",
|
||||
LoggedAt: now.Add(-48 * time.Hour),
|
||||
LoggedAt: now.Add(-100 * 24 * time.Hour),
|
||||
RemoteAddr: "203.0.113.10",
|
||||
Host: "example.com",
|
||||
Path: "/access",
|
||||
@@ -115,13 +154,13 @@ func TestRunDatabaseAutoCleanupOnceDeletesAllObservabilityTargets(t *testing.T)
|
||||
}}))
|
||||
require.NoError(t, model.InsertOpenFlareMetricSnapshot(ctx, &model.OpenFlareMetricSnapshot{
|
||||
NodeID: "node-a",
|
||||
CapturedAt: now.Add(-48 * time.Hour),
|
||||
CapturedAt: now.Add(-40 * 24 * time.Hour),
|
||||
CPUUsagePercent: 10,
|
||||
}))
|
||||
require.NoError(t, model.InsertOpenFlareRequestReport(ctx, &model.OpenFlareRequestReport{
|
||||
NodeID: "node-a",
|
||||
WindowStartedAt: now.Add(-49 * time.Hour),
|
||||
WindowEndedAt: now.Add(-48 * time.Hour),
|
||||
WindowStartedAt: now.Add(-41 * 24 * time.Hour),
|
||||
WindowEndedAt: now.Add(-40 * 24 * time.Hour),
|
||||
RequestCount: 15,
|
||||
}))
|
||||
|
||||
@@ -132,6 +171,15 @@ func TestRunDatabaseAutoCleanupOnceDeletesAllObservabilityTargets(t *testing.T)
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, summary)
|
||||
require.Len(t, summary.Results, 6)
|
||||
assert.Equal(t, 1, summary.RetentionDays)
|
||||
|
||||
for _, result := range summary.Results {
|
||||
assert.Equal(t, analyticsrepo.CleanupModeTTLMaterialize, result.CleanupMode)
|
||||
assert.Equal(t, int64(0), result.DeletedCount, "target %s must not claim hard delete", result.Target)
|
||||
assert.GreaterOrEqual(t, result.TableTTLDays, 30)
|
||||
require.NotNil(t, result.RetentionDays)
|
||||
assert.GreaterOrEqual(t, *result.RetentionDays, result.TableTTLDays)
|
||||
}
|
||||
|
||||
accessLogs, err := model.ListOpenFlareAccessLogs(ctx, model.OpenFlareAccessLogQuery{Page: 0, PageSize: 10})
|
||||
require.NoError(t, err)
|
||||
@@ -145,3 +193,16 @@ func TestRunDatabaseAutoCleanupOnceDeletesAllObservabilityTargets(t *testing.T)
|
||||
require.NoError(t, err)
|
||||
assert.Empty(t, requestReports)
|
||||
}
|
||||
|
||||
func TestTableTTLDaysForCleanupTarget(t *testing.T) {
|
||||
days, ok := TableTTLDaysForCleanupTarget(DatabaseCleanupTargetAccessLogs)
|
||||
require.True(t, ok)
|
||||
assert.Equal(t, 90, days)
|
||||
|
||||
days, ok = TableTTLDaysForCleanupTarget(DatabaseCleanupTargetMetricSnapshots)
|
||||
require.True(t, ok)
|
||||
assert.Equal(t, 30, days)
|
||||
|
||||
_, ok = TableTTLDaysForCleanupTarget("unknown")
|
||||
assert.False(t, ok)
|
||||
}
|
||||
|
||||
@@ -6,6 +6,7 @@ package risk_control
|
||||
import (
|
||||
"context"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/Rain-kl/Wavelet/internal/config"
|
||||
"github.com/Rain-kl/Wavelet/internal/db/batchwriter"
|
||||
@@ -15,6 +16,11 @@ import (
|
||||
"github.com/Rain-kl/Wavelet/pkg/logger"
|
||||
)
|
||||
|
||||
const (
|
||||
// Bound visibility lag for sparse access-log traffic when MinBatchSize is not met.
|
||||
accessLogMaxFlushWait = 3 * time.Second
|
||||
)
|
||||
|
||||
var (
|
||||
logWriterMu sync.RWMutex
|
||||
logWriter *batchwriter.Writer[*analytics.UserAccessLog]
|
||||
@@ -33,6 +39,8 @@ func InitLogWriter(ctx context.Context) {
|
||||
}
|
||||
|
||||
cfg := batchwriter.DefaultConfig()
|
||||
cfg.Name = "user_access_logs"
|
||||
cfg.MaxFlushWait = accessLogMaxFlushWait
|
||||
writer, err := batchwriter.New[*analytics.UserAccessLog](cfg, func(ctx context.Context, items []*analytics.UserAccessLog) error {
|
||||
rows := make([]analytics.UserAccessLog, 0, len(items))
|
||||
for _, item := range items {
|
||||
@@ -50,8 +58,8 @@ func InitLogWriter(ctx context.Context) {
|
||||
}
|
||||
logger.WarnF(context.Background(), "[RiskControl] Log queue full, dropping log item for path: %s", path)
|
||||
}),
|
||||
batchwriter.WithFlushErrorHandler[*analytics.UserAccessLog](func(ctx context.Context, batchSize int, err error) {
|
||||
logger.ErrorF(ctx, "[RiskControl] Send ClickHouse batch failed (batch=%d): %v", batchSize, err)
|
||||
batchwriter.WithFlushErrorHandler[*analytics.UserAccessLog](func(ctx context.Context, items []*analytics.UserAccessLog, err error) {
|
||||
logger.ErrorF(ctx, "[RiskControl] Send ClickHouse batch failed (batch=%d): %v", len(items), err)
|
||||
}),
|
||||
)
|
||||
if err != nil {
|
||||
@@ -82,6 +90,16 @@ func IsBufferFull() bool {
|
||||
return writer.IsFull()
|
||||
}
|
||||
|
||||
// LogWriterStats returns queue depth and failure counters for the access-log writer.
|
||||
// When the writer is not initialized, it returns a zero-value Stats with the expected name.
|
||||
func LogWriterStats() batchwriter.Stats {
|
||||
writer := currentLogWriter()
|
||||
if writer == nil {
|
||||
return batchwriter.Stats{Name: "user_access_logs"}
|
||||
}
|
||||
return writer.Stats()
|
||||
}
|
||||
|
||||
// QueueAccessLog enqueues an access log without blocking.
|
||||
func QueueAccessLog(logItem *analytics.UserAccessLog) {
|
||||
writer := currentLogWriter()
|
||||
@@ -108,4 +126,4 @@ func currentLogWriter() *batchwriter.Writer[*analytics.UserAccessLog] {
|
||||
logWriterMu.RLock()
|
||||
defer logWriterMu.RUnlock()
|
||||
return logWriter
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,30 @@
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
package risk_control
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func TestAccessLogMaxFlushWaitInRange(t *testing.T) {
|
||||
t.Parallel()
|
||||
if accessLogMaxFlushWait < 2*time.Second || accessLogMaxFlushWait > 5*time.Second {
|
||||
t.Fatalf("accessLogMaxFlushWait = %v, want in [2s, 5s]", accessLogMaxFlushWait)
|
||||
}
|
||||
}
|
||||
|
||||
func TestLogWriterStatsWhenNil(t *testing.T) {
|
||||
t.Parallel()
|
||||
reset := SetLogWriterForTest(nil)
|
||||
t.Cleanup(reset)
|
||||
|
||||
stats := LogWriterStats()
|
||||
if stats.Name != "user_access_logs" {
|
||||
t.Fatalf("LogWriterStats().Name = %q, want user_access_logs", stats.Name)
|
||||
}
|
||||
if stats.Running {
|
||||
t.Fatal("LogWriterStats().Running = true for nil writer, want false")
|
||||
}
|
||||
}
|
||||
@@ -118,9 +118,17 @@ func applyDefaults(c *configModel) {
|
||||
}
|
||||
|
||||
func applyClickHouseDefaults(c *configModel) {
|
||||
// Tests disable ClickHouse by default to avoid accidental connections.
|
||||
// Opt in with CLICKHOUSE_ENABLED=true for live integration tests (e.g. -tags live_ch).
|
||||
if isTest() {
|
||||
c.ClickHouse.Enabled = false
|
||||
return
|
||||
if v, ok := os.LookupEnv("CLICKHOUSE_ENABLED"); !ok {
|
||||
c.ClickHouse.Enabled = false
|
||||
return
|
||||
} else if b, err := strconv.ParseBool(v); err != nil || !b {
|
||||
c.ClickHouse.Enabled = false
|
||||
return
|
||||
}
|
||||
// Keep Enabled=true from env and continue applying host/pool defaults.
|
||||
}
|
||||
if !c.ClickHouse.Enabled {
|
||||
c.ClickHouse.Enabled = true
|
||||
@@ -134,11 +142,14 @@ func applyClickHouseDefaults(c *configModel) {
|
||||
if c.ClickHouse.Username == "" {
|
||||
c.ClickHouse.Username = "default"
|
||||
}
|
||||
// Pool / buffer defaults target small control-plane hosts (e.g. 3c6g):
|
||||
// oversized open/idle pools waste RAM and amplify concurrent CH pressure;
|
||||
// large block buffers add client memory without helping our small batch inserts.
|
||||
if c.ClickHouse.MaxIdleConn <= 0 {
|
||||
c.ClickHouse.MaxIdleConn = 20
|
||||
c.ClickHouse.MaxIdleConn = 8
|
||||
}
|
||||
if c.ClickHouse.MaxOpenConn <= 0 {
|
||||
c.ClickHouse.MaxOpenConn = 50
|
||||
c.ClickHouse.MaxOpenConn = 16
|
||||
}
|
||||
if c.ClickHouse.ConnMaxLifetime <= 0 {
|
||||
c.ClickHouse.ConnMaxLifetime = 3600
|
||||
@@ -147,7 +158,7 @@ func applyClickHouseDefaults(c *configModel) {
|
||||
c.ClickHouse.DialTimeout = 5
|
||||
}
|
||||
if c.ClickHouse.BlockBufferSize == 0 {
|
||||
c.ClickHouse.BlockBufferSize = 100
|
||||
c.ClickHouse.BlockBufferSize = 32
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -71,18 +71,19 @@ type databaseReplicaConfig struct {
|
||||
Password string `mapstructure:"password"`
|
||||
}
|
||||
|
||||
// clickhouse 配置
|
||||
// clickHouseConfig ClickHouse 原生客户端配置。
|
||||
// 连接池 / block_buffer 默认值按小型控制面主机(如 3c6g)收敛,见 applyClickHouseDefaults。
|
||||
type clickHouseConfig struct {
|
||||
Enabled bool `mapstructure:"enabled"`
|
||||
Hosts []string `mapstructure:"hosts"`
|
||||
Username string `mapstructure:"username"`
|
||||
Password string `mapstructure:"password"`
|
||||
Database string `mapstructure:"database"`
|
||||
MaxIdleConn int `mapstructure:"max_idle_conn"`
|
||||
MaxOpenConn int `mapstructure:"max_open_conn"`
|
||||
ConnMaxLifetime int `mapstructure:"conn_max_lifetime"`
|
||||
DialTimeout int `mapstructure:"dial_timeout"`
|
||||
BlockBufferSize uint8 `mapstructure:"block_buffer_size"`
|
||||
MaxIdleConn int `mapstructure:"max_idle_conn"` // 默认 8
|
||||
MaxOpenConn int `mapstructure:"max_open_conn"` // 默认 16
|
||||
ConnMaxLifetime int `mapstructure:"conn_max_lifetime"` // 秒
|
||||
DialTimeout int `mapstructure:"dial_timeout"` // 秒
|
||||
BlockBufferSize uint8 `mapstructure:"block_buffer_size"` // 默认 32
|
||||
}
|
||||
|
||||
// redisConfig Redis配置
|
||||
|
||||
@@ -28,10 +28,15 @@ type Config struct {
|
||||
|
||||
// MinBatchSize is the minimum in-memory batch size for time-based flushes.
|
||||
// Zero disables the threshold and preserves legacy interval flush behavior.
|
||||
// When set, interval flushes below this size are skipped unless MaxFlushWait elapses.
|
||||
MinBatchSize int
|
||||
|
||||
// FlushInterval triggers a time-based flush even when the batch is smaller.
|
||||
// FlushInterval is how often the worker checks whether a time-based flush should run.
|
||||
FlushInterval time.Duration
|
||||
|
||||
// MaxFlushWait forces a flush of any non-empty batch once the oldest item has waited
|
||||
// this long, even if MinBatchSize has not been reached. Zero disables the force path.
|
||||
MaxFlushWait time.Duration
|
||||
}
|
||||
|
||||
// DefaultConfig returns production-friendly defaults aligned with audit log batching.
|
||||
@@ -57,5 +62,8 @@ func (c Config) validate() error {
|
||||
if c.FlushInterval <= 0 {
|
||||
return fmt.Errorf("batchwriter: flush interval must be positive")
|
||||
}
|
||||
if c.MaxFlushWait < 0 {
|
||||
return fmt.Errorf("batchwriter: max flush wait must be non-negative")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -9,22 +9,34 @@ package batchwriter
|
||||
import (
|
||||
"context"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
)
|
||||
|
||||
// FlushFunc persists a batch of queued items. It is invoked from the worker goroutine.
|
||||
type FlushFunc[T any] func(ctx context.Context, items []T) error
|
||||
|
||||
// FlushErrorHandler is called when FlushFunc returns an error. The batch is discarded
|
||||
// after the handler returns; the worker continues processing.
|
||||
type FlushErrorHandler func(ctx context.Context, batchSize int, err error)
|
||||
// FlushErrorHandler is called when FlushFunc returns an error after optional retries.
|
||||
// The batch is discarded after the handler returns; the worker continues processing.
|
||||
// Handlers receive the failed items so callers can release dedup keys or re-queue.
|
||||
type FlushErrorHandler[T any] func(ctx context.Context, items []T, err error)
|
||||
|
||||
// Stats is a point-in-time snapshot of Writer queue and failure counters.
|
||||
type Stats struct {
|
||||
Name string `json:"name"`
|
||||
Depth int `json:"depth"`
|
||||
Cap int `json:"cap"`
|
||||
Drops int64 `json:"drops"`
|
||||
FlushErrors int64 `json:"flush_errors"`
|
||||
Running bool `json:"running"`
|
||||
}
|
||||
|
||||
// Writer buffers items and flushes them by size or interval.
|
||||
type Writer[T any] struct {
|
||||
cfg Config
|
||||
flush FlushFunc[T]
|
||||
|
||||
onFlushError FlushErrorHandler
|
||||
onFlushError FlushErrorHandler[T]
|
||||
onDrop func(T)
|
||||
|
||||
startOnce sync.Once
|
||||
@@ -34,13 +46,16 @@ type Writer[T any] struct {
|
||||
ch chan T
|
||||
workerCtx context.Context
|
||||
done chan struct{}
|
||||
|
||||
drops atomic.Int64
|
||||
flushErrors atomic.Int64
|
||||
}
|
||||
|
||||
// Option configures optional Writer callbacks.
|
||||
type Option[T any] func(*Writer[T])
|
||||
|
||||
// WithFlushErrorHandler registers a callback for flush failures.
|
||||
func WithFlushErrorHandler[T any](handler FlushErrorHandler) Option[T] {
|
||||
func WithFlushErrorHandler[T any](handler FlushErrorHandler[T]) Option[T] {
|
||||
return func(w *Writer[T]) {
|
||||
w.onFlushError = handler
|
||||
}
|
||||
@@ -168,22 +183,37 @@ func (w *Writer[T]) Cap() int {
|
||||
return w.cfg.QueueSize
|
||||
}
|
||||
|
||||
// Stats returns a point-in-time snapshot of queue depth and failure counters.
|
||||
func (w *Writer[T]) Stats() Stats {
|
||||
return Stats{
|
||||
Name: w.cfg.Name,
|
||||
Depth: w.Len(),
|
||||
Cap: w.Cap(),
|
||||
Drops: w.drops.Load(),
|
||||
FlushErrors: w.flushErrors.Load(),
|
||||
Running: w.Running(),
|
||||
}
|
||||
}
|
||||
|
||||
func (w *Writer[T]) run() {
|
||||
ticker := time.NewTicker(w.cfg.FlushInterval)
|
||||
defer ticker.Stop()
|
||||
|
||||
batch := make([]T, 0, w.cfg.MaxBatchSize)
|
||||
var batchStartedAt time.Time
|
||||
flush := func() {
|
||||
if len(batch) == 0 {
|
||||
return
|
||||
}
|
||||
items := append([]T(nil), batch...)
|
||||
if err := w.flush(w.workerCtx, items); err != nil {
|
||||
w.flushErrors.Add(1)
|
||||
if w.onFlushError != nil {
|
||||
w.onFlushError(w.workerCtx, len(items), err)
|
||||
w.onFlushError(w.workerCtx, items, err)
|
||||
}
|
||||
}
|
||||
batch = batch[:0]
|
||||
batchStartedAt = time.Time{}
|
||||
}
|
||||
|
||||
defer func() {
|
||||
@@ -197,21 +227,38 @@ func (w *Writer[T]) run() {
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
if len(batch) == 0 {
|
||||
batchStartedAt = time.Now()
|
||||
}
|
||||
batch = append(batch, item)
|
||||
if len(batch) >= w.cfg.MaxBatchSize {
|
||||
flush()
|
||||
}
|
||||
case <-ticker.C:
|
||||
if len(batch) > 0 && (w.cfg.MinBatchSize == 0 || len(batch) >= w.cfg.MinBatchSize) {
|
||||
if w.shouldFlushOnInterval(len(batch), batchStartedAt, time.Now()) {
|
||||
flush()
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (w *Writer[T]) shouldFlushOnInterval(batchLen int, batchStartedAt time.Time, now time.Time) bool {
|
||||
if batchLen == 0 {
|
||||
return false
|
||||
}
|
||||
if w.cfg.MinBatchSize == 0 || batchLen >= w.cfg.MinBatchSize {
|
||||
return true
|
||||
}
|
||||
if w.cfg.MaxFlushWait <= 0 || batchStartedAt.IsZero() {
|
||||
return false
|
||||
}
|
||||
return !now.Before(batchStartedAt.Add(w.cfg.MaxFlushWait))
|
||||
}
|
||||
|
||||
func (w *Writer[T]) notifyDrop(item T) {
|
||||
w.drops.Add(1)
|
||||
if w.onDrop == nil {
|
||||
return
|
||||
}
|
||||
w.onDrop(item)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -36,7 +36,7 @@ func TestWriterFlushesOnMaxBatchSize(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
var (
|
||||
mu sync.Mutex
|
||||
mu sync.Mutex
|
||||
batches [][]int
|
||||
)
|
||||
cfg := DefaultConfig()
|
||||
@@ -326,27 +326,83 @@ func TestWriterFlushesOnIntervalWhenMinBatchSizeReached(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestWriterForcesFlushAfterMaxFlushWait(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
var (
|
||||
mu sync.Mutex
|
||||
batch []int
|
||||
)
|
||||
cfg := DefaultConfig()
|
||||
cfg.MaxBatchSize = 100
|
||||
cfg.MinBatchSize = 50
|
||||
cfg.FlushInterval = 20 * time.Millisecond
|
||||
cfg.MaxFlushWait = 80 * time.Millisecond
|
||||
|
||||
writer, err := New[int](cfg, func(_ context.Context, items []int) error {
|
||||
mu.Lock()
|
||||
defer mu.Unlock()
|
||||
batch = append([]int(nil), items...)
|
||||
return nil
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("New() error = %v", err)
|
||||
}
|
||||
|
||||
writer.Start(context.Background())
|
||||
t.Cleanup(func() {
|
||||
stopCtx, cancel := context.WithTimeout(context.Background(), time.Second)
|
||||
defer cancel()
|
||||
if err := writer.Stop(stopCtx); err != nil {
|
||||
t.Fatalf("Stop() error = %v", err)
|
||||
}
|
||||
})
|
||||
|
||||
if !writer.TryEnqueue(1) {
|
||||
t.Fatal("TryEnqueue(1) = false, want true")
|
||||
}
|
||||
|
||||
deadline := time.Now().Add(time.Second)
|
||||
for {
|
||||
mu.Lock()
|
||||
ready := len(batch) == 1
|
||||
mu.Unlock()
|
||||
if ready || time.Now().After(deadline) {
|
||||
break
|
||||
}
|
||||
time.Sleep(5 * time.Millisecond)
|
||||
}
|
||||
|
||||
mu.Lock()
|
||||
got := batch
|
||||
mu.Unlock()
|
||||
if diff := cmp.Diff([]int{1}, got); diff != "" {
|
||||
t.Fatalf("max flush wait mismatch (-want +got):\n%s", diff)
|
||||
}
|
||||
}
|
||||
|
||||
func TestWriterInvokesFlushErrorHandler(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
cfg := DefaultConfig()
|
||||
cfg.Name = "test-flush-err"
|
||||
cfg.MaxBatchSize = 1
|
||||
cfg.FlushInterval = time.Hour
|
||||
|
||||
flushErr := errors.New("flush failed")
|
||||
var (
|
||||
mu sync.Mutex
|
||||
errCount int
|
||||
batchSize int
|
||||
mu sync.Mutex
|
||||
errCount int
|
||||
gotItems []int
|
||||
)
|
||||
|
||||
writer, err := New[int](cfg, func(context.Context, []int) error {
|
||||
return flushErr
|
||||
}, WithFlushErrorHandler[int](func(_ context.Context, size int, err error) {
|
||||
}, WithFlushErrorHandler[int](func(_ context.Context, items []int, err error) {
|
||||
mu.Lock()
|
||||
defer mu.Unlock()
|
||||
errCount++
|
||||
batchSize = size
|
||||
gotItems = append([]int(nil), items...)
|
||||
if !errors.Is(err, flushErr) {
|
||||
t.Errorf("flush error = %v, want %v", err, flushErr)
|
||||
}
|
||||
@@ -379,13 +435,61 @@ func TestWriterInvokesFlushErrorHandler(t *testing.T) {
|
||||
|
||||
mu.Lock()
|
||||
gotCount := errCount
|
||||
gotSize := batchSize
|
||||
items := gotItems
|
||||
mu.Unlock()
|
||||
|
||||
if gotCount != 1 {
|
||||
t.Fatalf("flush error handler count = %d, want 1", gotCount)
|
||||
}
|
||||
if gotSize != 1 {
|
||||
t.Fatalf("flush error handler batch size = %d, want 1", gotSize)
|
||||
if diff := cmp.Diff([]int{7}, items); diff != "" {
|
||||
t.Fatalf("flush error handler items mismatch (-want +got):\n%s", diff)
|
||||
}
|
||||
}
|
||||
|
||||
stats := writer.Stats()
|
||||
if stats.FlushErrors != 1 {
|
||||
t.Fatalf("Stats().FlushErrors = %d, want 1", stats.FlushErrors)
|
||||
}
|
||||
if stats.Name != "test-flush-err" {
|
||||
t.Fatalf("Stats().Name = %q, want test-flush-err", stats.Name)
|
||||
}
|
||||
}
|
||||
|
||||
func TestWriterStatsTracksDrops(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
cfg := DefaultConfig()
|
||||
cfg.Name = "test-drops"
|
||||
cfg.QueueSize = 1
|
||||
cfg.MaxBatchSize = 10
|
||||
cfg.FlushInterval = time.Hour
|
||||
|
||||
writer, err := New[int](cfg, func(context.Context, []int) error { return nil })
|
||||
if err != nil {
|
||||
t.Fatalf("New() error = %v", err)
|
||||
}
|
||||
|
||||
writer.Start(context.Background())
|
||||
t.Cleanup(func() {
|
||||
stopCtx, cancel := context.WithTimeout(context.Background(), time.Second)
|
||||
defer cancel()
|
||||
_ = writer.Stop(stopCtx)
|
||||
})
|
||||
|
||||
if !writer.TryEnqueue(1) {
|
||||
t.Fatal("TryEnqueue(1) = false, want true")
|
||||
}
|
||||
if writer.TryEnqueue(2) {
|
||||
t.Fatal("TryEnqueue(2) = true, want false")
|
||||
}
|
||||
|
||||
stats := writer.Stats()
|
||||
if stats.Drops != 1 {
|
||||
t.Fatalf("Stats().Drops = %d, want 1", stats.Drops)
|
||||
}
|
||||
if stats.Cap != 1 {
|
||||
t.Fatalf("Stats().Cap = %d, want 1", stats.Cap)
|
||||
}
|
||||
if !stats.Running {
|
||||
t.Fatal("Stats().Running = false, want true")
|
||||
}
|
||||
}
|
||||
|
||||
@@ -16,10 +16,18 @@ import (
|
||||
)
|
||||
|
||||
const (
|
||||
clickhouseMaxExecTime = 60 // ClickHouse 最大执行时间(秒)
|
||||
clickhouseReadTimeoutFactor = 2 // ReadTimeout 为 DialTimeout 的倍数
|
||||
clickhouseMaxExecTime = 60 // ClickHouse 最大执行时间(秒)
|
||||
clickhouseReadTimeoutFactor = 2 // ReadTimeout 为 DialTimeout 的倍数
|
||||
|
||||
// async_insert 仅挂在运行时 ChConn(写路径)上,不进入 migrator OpenDB:
|
||||
// 迁移/DDL 需要同步可见结果,且不应走异步 insert 缓冲。
|
||||
//
|
||||
// 为何启用:batchwriter 仍可能在短间隔内写出相对小的块;服务端 async_insert
|
||||
// 把多次 INSERT 合并成更大 part,减轻 3c6g 上 background merge 的 CPU 压力。
|
||||
// wait_for_async_insert=1:调用方在 flush 返回前等待落盘,避免进程崩溃丢批。
|
||||
// max_data_size / busy_timeout:约 10MB 或 ~2s 触发刷出,在延迟与 part 数之间折中。
|
||||
clickhouseAsyncInsertMaxDataSize = 10_000_000
|
||||
clickhouseAsyncInsertBusyTimeoutMs = 1000
|
||||
clickhouseAsyncInsertBusyTimeoutMs = 2000
|
||||
)
|
||||
|
||||
var (
|
||||
@@ -52,6 +60,8 @@ func init() {
|
||||
log.Println("[ClickHouse] connection established successfully")
|
||||
}
|
||||
|
||||
// buildClickHouseOptions builds the runtime native client options (queries + batch inserts).
|
||||
// Migrator uses a separate clickhouse.OpenDB path without async_insert settings.
|
||||
func buildClickHouseOptions() *clickhouse.Options {
|
||||
cfg := config.Config.ClickHouse
|
||||
|
||||
|
||||
+80
@@ -0,0 +1,80 @@
|
||||
-- +goose Up
|
||||
-- Hourly capacity rollups (avg CPU/memory + counter min/max for in-hour delta approximation).
|
||||
-- Network/disk counters are cumulative; max-min within an hour approximates that hour's delta
|
||||
-- (cross-hour continuity is intentionally approximate for dashboard trends).
|
||||
CREATE TABLE IF NOT EXISTS of_node_metric_capacity_hourly
|
||||
(
|
||||
node_id String,
|
||||
hour DateTime,
|
||||
cpu_usage_sum SimpleAggregateFunction(sum, Float64),
|
||||
cpu_usage_count SimpleAggregateFunction(sum, UInt64),
|
||||
memory_usage_sum SimpleAggregateFunction(sum, Float64),
|
||||
memory_usage_count SimpleAggregateFunction(sum, UInt64),
|
||||
network_rx_min SimpleAggregateFunction(min, Int64),
|
||||
network_rx_max SimpleAggregateFunction(max, Int64),
|
||||
network_tx_min SimpleAggregateFunction(min, Int64),
|
||||
network_tx_max SimpleAggregateFunction(max, Int64),
|
||||
disk_read_min SimpleAggregateFunction(min, Int64),
|
||||
disk_read_max SimpleAggregateFunction(max, Int64),
|
||||
disk_write_min SimpleAggregateFunction(min, Int64),
|
||||
disk_write_max SimpleAggregateFunction(max, Int64)
|
||||
)
|
||||
ENGINE = AggregatingMergeTree()
|
||||
PARTITION BY toYYYYMM(hour)
|
||||
ORDER BY (node_id, hour)
|
||||
TTL hour + INTERVAL 30 DAY;
|
||||
|
||||
CREATE MATERIALIZED VIEW IF NOT EXISTS of_node_metric_capacity_hourly_mv
|
||||
TO of_node_metric_capacity_hourly
|
||||
AS
|
||||
SELECT
|
||||
node_id,
|
||||
toStartOfHour(captured_at) AS hour,
|
||||
sum(cpu_usage_percent) AS cpu_usage_sum,
|
||||
toUInt64(count()) AS cpu_usage_count,
|
||||
sum(if(memory_total_bytes > 0, (memory_used_bytes * 100.0) / memory_total_bytes, 0)) AS memory_usage_sum,
|
||||
toUInt64(countIf(memory_total_bytes > 0)) AS memory_usage_count,
|
||||
min(network_rx_bytes) AS network_rx_min,
|
||||
max(network_rx_bytes) AS network_rx_max,
|
||||
min(network_tx_bytes) AS network_tx_min,
|
||||
max(network_tx_bytes) AS network_tx_max,
|
||||
min(disk_read_bytes) AS disk_read_min,
|
||||
max(disk_read_bytes) AS disk_read_max,
|
||||
min(disk_write_bytes) AS disk_write_min,
|
||||
max(disk_write_bytes) AS disk_write_max
|
||||
FROM of_node_metric_snapshots
|
||||
GROUP BY node_id, hour;
|
||||
|
||||
-- Hourly OpenResty counter rollups (min/max per node-hour for delta approximation).
|
||||
CREATE TABLE IF NOT EXISTS of_node_openresty_hourly
|
||||
(
|
||||
node_id String,
|
||||
hour DateTime,
|
||||
openresty_rx_min SimpleAggregateFunction(min, Int64),
|
||||
openresty_rx_max SimpleAggregateFunction(max, Int64),
|
||||
openresty_tx_min SimpleAggregateFunction(min, Int64),
|
||||
openresty_tx_max SimpleAggregateFunction(max, Int64)
|
||||
)
|
||||
ENGINE = AggregatingMergeTree()
|
||||
PARTITION BY toYYYYMM(hour)
|
||||
ORDER BY (node_id, hour)
|
||||
TTL hour + INTERVAL 30 DAY;
|
||||
|
||||
CREATE MATERIALIZED VIEW IF NOT EXISTS of_node_openresty_hourly_mv
|
||||
TO of_node_openresty_hourly
|
||||
AS
|
||||
SELECT
|
||||
node_id,
|
||||
toStartOfHour(captured_at) AS hour,
|
||||
min(openresty_rx_bytes) AS openresty_rx_min,
|
||||
max(openresty_rx_bytes) AS openresty_rx_max,
|
||||
min(openresty_tx_bytes) AS openresty_tx_min,
|
||||
max(openresty_tx_bytes) AS openresty_tx_max
|
||||
FROM of_node_obs_openresty
|
||||
GROUP BY node_id, hour;
|
||||
|
||||
-- +goose Down
|
||||
DROP VIEW IF EXISTS of_node_openresty_hourly_mv;
|
||||
DROP TABLE IF EXISTS of_node_openresty_hourly;
|
||||
DROP VIEW IF EXISTS of_node_metric_capacity_hourly_mv;
|
||||
DROP TABLE IF EXISTS of_node_metric_capacity_hourly;
|
||||
@@ -0,0 +1,42 @@
|
||||
-- +goose Up
|
||||
-- Hourly traffic rollups: 30d TTL + UV aggregation semantics.
|
||||
--
|
||||
-- unique_visitor_count on of_node_request_reports is per short report window
|
||||
-- (agent local distinct count for that window only). Summing those values in the
|
||||
-- MV (and again via SummingMergeTree part merges) invents a "true UV" number that
|
||||
-- double-counts visitors across windows. Prefer max() as a peak-window estimate;
|
||||
-- still NOT distinct visitors across the hour — UI/API must not overclaim.
|
||||
|
||||
ALTER TABLE of_node_traffic_hourly
|
||||
MODIFY TTL toDateTime(hour) + INTERVAL 30 DAY;
|
||||
|
||||
DROP VIEW IF EXISTS of_node_traffic_hourly_mv;
|
||||
|
||||
CREATE MATERIALIZED VIEW of_node_traffic_hourly_mv
|
||||
TO of_node_traffic_hourly
|
||||
AS
|
||||
SELECT
|
||||
node_id,
|
||||
toStartOfHour(window_ended_at) AS hour,
|
||||
sum(request_count) AS request_count,
|
||||
sum(error_count) AS error_count,
|
||||
-- Peak per-window UV estimate for the hour; not true cross-window distinct UV.
|
||||
max(unique_visitor_count) AS unique_visitor_count
|
||||
FROM of_node_request_reports
|
||||
GROUP BY node_id, hour;
|
||||
|
||||
-- +goose Down
|
||||
-- TTL reverse is not safe without table rewrite; restore prior MV definition only.
|
||||
DROP VIEW IF EXISTS of_node_traffic_hourly_mv;
|
||||
|
||||
CREATE MATERIALIZED VIEW of_node_traffic_hourly_mv
|
||||
TO of_node_traffic_hourly
|
||||
AS
|
||||
SELECT
|
||||
node_id,
|
||||
toStartOfHour(window_ended_at) AS hour,
|
||||
sum(request_count) AS request_count,
|
||||
sum(error_count) AS error_count,
|
||||
sum(unique_visitor_count) AS unique_visitor_count
|
||||
FROM of_node_request_reports
|
||||
GROUP BY node_id, hour;
|
||||
+66
@@ -0,0 +1,66 @@
|
||||
-- +goose Up
|
||||
-- One-time historical backfill for hours not yet present in rollup tables.
|
||||
-- MV only ingests rows after creation; without this, 24h charts rely on raw merge forever.
|
||||
-- ANTI JOIN avoids double-counting hours already filled by the live MV.
|
||||
|
||||
INSERT INTO of_node_metric_capacity_hourly
|
||||
SELECT
|
||||
s.node_id,
|
||||
toStartOfHour(s.captured_at) AS hour,
|
||||
sum(s.cpu_usage_percent) AS cpu_usage_sum,
|
||||
toUInt64(count()) AS cpu_usage_count,
|
||||
sum(if(s.memory_total_bytes > 0, (s.memory_used_bytes * 100.0) / s.memory_total_bytes, 0)) AS memory_usage_sum,
|
||||
toUInt64(countIf(s.memory_total_bytes > 0)) AS memory_usage_count,
|
||||
min(s.network_rx_bytes) AS network_rx_min,
|
||||
max(s.network_rx_bytes) AS network_rx_max,
|
||||
min(s.network_tx_bytes) AS network_tx_min,
|
||||
max(s.network_tx_bytes) AS network_tx_max,
|
||||
min(s.disk_read_bytes) AS disk_read_min,
|
||||
max(s.disk_read_bytes) AS disk_read_max,
|
||||
min(s.disk_write_bytes) AS disk_write_min,
|
||||
max(s.disk_write_bytes) AS disk_write_max
|
||||
FROM of_node_metric_snapshots AS s
|
||||
ANTI JOIN
|
||||
(
|
||||
SELECT
|
||||
node_id,
|
||||
hour
|
||||
FROM of_node_metric_capacity_hourly
|
||||
GROUP BY
|
||||
node_id,
|
||||
hour
|
||||
) AS existing
|
||||
ON s.node_id = existing.node_id AND toStartOfHour(s.captured_at) = existing.hour
|
||||
WHERE s.captured_at >= now() - INTERVAL 30 DAY
|
||||
GROUP BY
|
||||
s.node_id,
|
||||
hour;
|
||||
|
||||
INSERT INTO of_node_openresty_hourly
|
||||
SELECT
|
||||
s.node_id,
|
||||
toStartOfHour(s.captured_at) AS hour,
|
||||
min(s.openresty_rx_bytes) AS openresty_rx_min,
|
||||
max(s.openresty_rx_bytes) AS openresty_rx_max,
|
||||
min(s.openresty_tx_bytes) AS openresty_tx_min,
|
||||
max(s.openresty_tx_bytes) AS openresty_tx_max
|
||||
FROM of_node_obs_openresty AS s
|
||||
ANTI JOIN
|
||||
(
|
||||
SELECT
|
||||
node_id,
|
||||
hour
|
||||
FROM of_node_openresty_hourly
|
||||
GROUP BY
|
||||
node_id,
|
||||
hour
|
||||
) AS existing
|
||||
ON s.node_id = existing.node_id AND toStartOfHour(s.captured_at) = existing.hour
|
||||
WHERE s.captured_at >= now() - INTERVAL 30 DAY
|
||||
GROUP BY
|
||||
s.node_id,
|
||||
hour;
|
||||
|
||||
-- +goose Down
|
||||
-- Backfill is additive; down does not remove historical rollup rows (TTL still applies).
|
||||
SELECT 1;
|
||||
@@ -137,7 +137,7 @@ CREATE INDEX IF NOT EXISTS idx_templates_created_at ON templates (created_at);
|
||||
CREATE INDEX IF NOT EXISTS idx_templates_updated_at ON templates (updated_at);
|
||||
|
||||
INSERT INTO system_configs (key, value, type, visibility, description, created_at, updated_at) VALUES
|
||||
('cap_login_enabled', 'true', 'system', 1, '是否启用登录人机验证(true/false)', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP),
|
||||
('cap_login_enabled', 'false', 'system', 1, '是否启用登录人机验证(true/false)', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP),
|
||||
('cap_auto_solve', 'true', 'system', 1, '打开页面后是否自动开始计算,关闭则需用户手动点击触发', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP),
|
||||
('cap_challenge_count', '1', 'system', 0, '客户端需求解的 PoW 难题总数,默认 1,推荐 1~5', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP),
|
||||
('cap_challenge_size', '32', 'system', 0, '人机验证盐值长度', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP),
|
||||
|
||||
@@ -7,10 +7,6 @@ UPDATE w_system_configs
|
||||
SET value = 'false', updated_at = CURRENT_TIMESTAMP
|
||||
WHERE key = 'password_register_enabled' AND value = 'true';
|
||||
|
||||
UPDATE w_system_configs
|
||||
SET value = 'true', updated_at = CURRENT_TIMESTAMP
|
||||
WHERE key = 'cap_login_enabled' AND value = 'false';
|
||||
|
||||
-- +goose Down
|
||||
UPDATE w_system_configs
|
||||
SET value = 'true', updated_at = CURRENT_TIMESTAMP
|
||||
@@ -18,8 +14,4 @@ WHERE key = 'registration_enabled' AND value = 'false';
|
||||
|
||||
UPDATE w_system_configs
|
||||
SET value = 'true', updated_at = CURRENT_TIMESTAMP
|
||||
WHERE key = 'password_register_enabled' AND value = 'false';
|
||||
|
||||
UPDATE w_system_configs
|
||||
SET value = 'false', updated_at = CURRENT_TIMESTAMP
|
||||
WHERE key = 'cap_login_enabled' AND value = 'true';
|
||||
WHERE key = 'password_register_enabled' AND value = 'false';
|
||||
@@ -137,7 +137,7 @@ CREATE INDEX IF NOT EXISTS idx_templates_created_at ON templates (created_at);
|
||||
CREATE INDEX IF NOT EXISTS idx_templates_updated_at ON templates (updated_at);
|
||||
|
||||
INSERT INTO system_configs (key, value, type, visibility, description, created_at, updated_at) VALUES
|
||||
('cap_login_enabled', 'true', 'system', 1, '是否启用登录人机验证(true/false)', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP),
|
||||
('cap_login_enabled', 'false', 'system', 1, '是否启用登录人机验证(true/false)', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP),
|
||||
('cap_auto_solve', 'true', 'system', 1, '打开页面后是否自动开始计算,关闭则需用户手动点击触发', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP),
|
||||
('cap_challenge_count', '1', 'system', 0, '客户端需求解的 PoW 难题总数,默认 1,推荐 1~5', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP),
|
||||
('cap_challenge_size', '32', 'system', 0, '人机验证盐值长度', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP),
|
||||
|
||||
@@ -7,10 +7,6 @@ UPDATE w_system_configs
|
||||
SET value = 'false', updated_at = CURRENT_TIMESTAMP
|
||||
WHERE key = 'password_register_enabled' AND value = 'true';
|
||||
|
||||
UPDATE w_system_configs
|
||||
SET value = 'true', updated_at = CURRENT_TIMESTAMP
|
||||
WHERE key = 'cap_login_enabled' AND value = 'false';
|
||||
|
||||
-- +goose Down
|
||||
UPDATE w_system_configs
|
||||
SET value = 'true', updated_at = CURRENT_TIMESTAMP
|
||||
@@ -18,8 +14,4 @@ WHERE key = 'registration_enabled' AND value = 'false';
|
||||
|
||||
UPDATE w_system_configs
|
||||
SET value = 'true', updated_at = CURRENT_TIMESTAMP
|
||||
WHERE key = 'password_register_enabled' AND value = 'false';
|
||||
|
||||
UPDATE w_system_configs
|
||||
SET value = 'false', updated_at = CURRENT_TIMESTAMP
|
||||
WHERE key = 'cap_login_enabled' AND value = 'true';
|
||||
WHERE key = 'password_register_enabled' AND value = 'false';
|
||||
@@ -8,11 +8,34 @@ import (
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/Rain-kl/Wavelet/internal/apps/openflare/chwriter"
|
||||
analyticsmodel "github.com/Rain-kl/Wavelet/internal/model/analytics"
|
||||
analyticsrepo "github.com/Rain-kl/Wavelet/internal/repository/analytics"
|
||||
)
|
||||
|
||||
// AccessLogInsertHooks queues node access logs for async ClickHouse write.
|
||||
// Wired from openflare/chwriter.Init so model never imports the apps layer.
|
||||
type AccessLogInsertHooks struct {
|
||||
QueueNodeAccessLogs func(logs []analyticsmodel.NodeAccessLog)
|
||||
}
|
||||
|
||||
var (
|
||||
accessLogInsertHooksMu sync.RWMutex
|
||||
accessLogInsertHooks AccessLogInsertHooks
|
||||
)
|
||||
|
||||
// SetAccessLogInsertHooks registers async queue callbacks for access log inserts.
|
||||
func SetAccessLogInsertHooks(hooks AccessLogInsertHooks) {
|
||||
accessLogInsertHooksMu.Lock()
|
||||
accessLogInsertHooks = hooks
|
||||
accessLogInsertHooksMu.Unlock()
|
||||
}
|
||||
|
||||
func currentAccessLogInsertHooks() AccessLogInsertHooks {
|
||||
accessLogInsertHooksMu.RLock()
|
||||
defer accessLogInsertHooksMu.RUnlock()
|
||||
return accessLogInsertHooks
|
||||
}
|
||||
|
||||
type accessLogStore interface {
|
||||
InsertBatch(ctx context.Context, records []*OpenFlareAccessLog) error
|
||||
List(ctx context.Context, query OpenFlareAccessLogQuery) ([]*OpenFlareAccessLog, error)
|
||||
@@ -75,7 +98,9 @@ func (clickhouseAccessLogStore) InsertBatch(_ context.Context, records []*OpenFl
|
||||
}
|
||||
logs = append(logs, toAnalyticsNodeAccessLog(record))
|
||||
}
|
||||
chwriter.QueueNodeAccessLogs(logs)
|
||||
if hook := currentAccessLogInsertHooks().QueueNodeAccessLogs; hook != nil {
|
||||
hook(logs)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,75 @@
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
package model
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
analyticsmodel "github.com/Rain-kl/Wavelet/internal/model/analytics"
|
||||
)
|
||||
|
||||
// Hook setters are process-global; keep these tests serial.
|
||||
|
||||
func TestObservabilityInsertHooksAreInvoked(t *testing.T) {
|
||||
var gotSnapshot analyticsmodel.NodeMetricSnapshot
|
||||
SetObservabilityInsertHooks(ObservabilityInsertHooks{
|
||||
QueueMetricSnapshot: func(s analyticsmodel.NodeMetricSnapshot) {
|
||||
gotSnapshot = s
|
||||
},
|
||||
})
|
||||
t.Cleanup(func() {
|
||||
SetObservabilityInsertHooks(ObservabilityInsertHooks{})
|
||||
})
|
||||
|
||||
record := &OpenFlareMetricSnapshot{
|
||||
NodeID: "node-1",
|
||||
CapturedAt: time.Unix(100, 0).UTC(),
|
||||
}
|
||||
if err := (clickhouseObservabilityStore{}).InsertMetricSnapshot(context.Background(), record); err != nil {
|
||||
t.Fatalf("InsertMetricSnapshot error = %v", err)
|
||||
}
|
||||
if gotSnapshot.NodeID != "node-1" {
|
||||
t.Fatalf("hook node id = %q, want node-1", gotSnapshot.NodeID)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAccessLogInsertHooksAreInvoked(t *testing.T) {
|
||||
var got []analyticsmodel.NodeAccessLog
|
||||
SetAccessLogInsertHooks(AccessLogInsertHooks{
|
||||
QueueNodeAccessLogs: func(logs []analyticsmodel.NodeAccessLog) {
|
||||
got = append([]analyticsmodel.NodeAccessLog(nil), logs...)
|
||||
},
|
||||
})
|
||||
t.Cleanup(func() {
|
||||
SetAccessLogInsertHooks(AccessLogInsertHooks{})
|
||||
})
|
||||
|
||||
records := []*OpenFlareAccessLog{
|
||||
{NodeID: "n1", Path: "/a"},
|
||||
{NodeID: "n1", Path: "/b"},
|
||||
}
|
||||
if err := (clickhouseAccessLogStore{}).InsertBatch(context.Background(), records); err != nil {
|
||||
t.Fatalf("InsertBatch error = %v", err)
|
||||
}
|
||||
if len(got) != 2 {
|
||||
t.Fatalf("hook logs = %d, want 2", len(got))
|
||||
}
|
||||
if got[0].Path != "/a" || got[1].Path != "/b" {
|
||||
t.Fatalf("hook paths = %q/%q, want /a /b", got[0].Path, got[1].Path)
|
||||
}
|
||||
}
|
||||
|
||||
func TestInsertHooksNoopWhenUnset(t *testing.T) {
|
||||
SetObservabilityInsertHooks(ObservabilityInsertHooks{})
|
||||
SetAccessLogInsertHooks(AccessLogInsertHooks{})
|
||||
|
||||
if err := (clickhouseObservabilityStore{}).InsertMetricSnapshot(context.Background(), &OpenFlareMetricSnapshot{NodeID: "x"}); err != nil {
|
||||
t.Fatalf("InsertMetricSnapshot with nil hook error = %v", err)
|
||||
}
|
||||
if err := (clickhouseAccessLogStore{}).InsertBatch(context.Background(), []*OpenFlareAccessLog{{NodeID: "x"}}); err != nil {
|
||||
t.Fatalf("InsertBatch with nil hook error = %v", err)
|
||||
}
|
||||
}
|
||||
@@ -333,11 +333,82 @@ func ListOpenFlareMetricSnapshotsSince(ctx context.Context, nodeID string, since
|
||||
return currentObservabilityStore().ListMetricSnapshots(ctx, nodeID, since, limit)
|
||||
}
|
||||
|
||||
// ListOpenFlareLatestMetricSnapshotsSince returns the latest metric snapshot per node.
|
||||
// Prefer ClickHouse LIMIT 1 BY; on CH unavailability fall back to store list + reduce.
|
||||
func ListOpenFlareLatestMetricSnapshotsSince(ctx context.Context, nodeID string, since time.Time) ([]*OpenFlareMetricSnapshot, error) {
|
||||
rows, err := analyticsrepo.ListLatestNodeMetricSnapshots(ctx, analyticsrepo.NodeObservabilityFilter{
|
||||
NodeID: nodeID,
|
||||
Since: since,
|
||||
})
|
||||
if err == nil {
|
||||
return fromAnalyticsNodeMetricSnapshots(rows), nil
|
||||
}
|
||||
// Fallback for unit tests (memory store) and environments without ClickHouse.
|
||||
all, listErr := ListOpenFlareMetricSnapshotsSince(ctx, nodeID, since, 0)
|
||||
if listErr != nil {
|
||||
return nil, err
|
||||
}
|
||||
return openFlareLatestMetricSnapshots(all), nil
|
||||
}
|
||||
|
||||
// ListOpenFlareRequestReportsSince returns request reports since the given time.
|
||||
func ListOpenFlareRequestReportsSince(ctx context.Context, nodeID string, since time.Time, limit int) ([]*OpenFlareRequestReport, error) {
|
||||
return currentObservabilityStore().ListRequestReports(ctx, nodeID, since, limit)
|
||||
}
|
||||
|
||||
// ListOpenFlareLatestRequestReportsSince returns the latest request report per node.
|
||||
// Prefer ClickHouse LIMIT 1 BY; on CH unavailability fall back to store list + reduce.
|
||||
func ListOpenFlareLatestRequestReportsSince(ctx context.Context, nodeID string, since time.Time) ([]*OpenFlareRequestReport, error) {
|
||||
rows, err := analyticsrepo.ListLatestNodeRequestReports(ctx, analyticsrepo.NodeObservabilityFilter{
|
||||
NodeID: nodeID,
|
||||
Since: since,
|
||||
})
|
||||
if err == nil {
|
||||
return fromAnalyticsNodeRequestReports(rows), nil
|
||||
}
|
||||
all, listErr := ListOpenFlareRequestReportsSince(ctx, nodeID, since, 0)
|
||||
if listErr != nil {
|
||||
return nil, err
|
||||
}
|
||||
return openFlareLatestRequestReports(all), nil
|
||||
}
|
||||
|
||||
func openFlareLatestMetricSnapshots(snapshots []*OpenFlareMetricSnapshot) []*OpenFlareMetricSnapshot {
|
||||
latestByNode := make(map[string]*OpenFlareMetricSnapshot, len(snapshots))
|
||||
for _, snapshot := range snapshots {
|
||||
if snapshot == nil || snapshot.NodeID == "" {
|
||||
continue
|
||||
}
|
||||
if existing, ok := latestByNode[snapshot.NodeID]; ok && !snapshot.CapturedAt.After(existing.CapturedAt) {
|
||||
continue
|
||||
}
|
||||
latestByNode[snapshot.NodeID] = snapshot
|
||||
}
|
||||
result := make([]*OpenFlareMetricSnapshot, 0, len(latestByNode))
|
||||
for _, snapshot := range latestByNode {
|
||||
result = append(result, snapshot)
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
func openFlareLatestRequestReports(reports []*OpenFlareRequestReport) []*OpenFlareRequestReport {
|
||||
latestByNode := make(map[string]*OpenFlareRequestReport, len(reports))
|
||||
for _, report := range reports {
|
||||
if report == nil || report.NodeID == "" {
|
||||
continue
|
||||
}
|
||||
if existing, ok := latestByNode[report.NodeID]; ok && !report.WindowEndedAt.After(existing.WindowEndedAt) {
|
||||
continue
|
||||
}
|
||||
latestByNode[report.NodeID] = report
|
||||
}
|
||||
result := make([]*OpenFlareRequestReport, 0, len(latestByNode))
|
||||
for _, report := range latestByNode {
|
||||
result = append(result, report)
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
// OpenFlareTrafficHourly is an hourly traffic rollup row.
|
||||
type OpenFlareTrafficHourly struct {
|
||||
NodeID string `json:"node_id"`
|
||||
@@ -369,6 +440,72 @@ func ListOpenFlareTrafficHourlySince(ctx context.Context, nodeID string, since t
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// OpenFlareMetricHourly is an hourly metric snapshot aggregation row.
|
||||
type OpenFlareMetricHourly struct {
|
||||
Hour time.Time `json:"hour"`
|
||||
AverageCPUUsagePercent float64 `json:"average_cpu_usage_percent"`
|
||||
AverageMemoryUsagePercent float64 `json:"average_memory_usage_percent"`
|
||||
NetworkRxBytes int64 `json:"network_rx_bytes"`
|
||||
NetworkTxBytes int64 `json:"network_tx_bytes"`
|
||||
DiskReadBytes int64 `json:"disk_read_bytes"`
|
||||
DiskWriteBytes int64 `json:"disk_write_bytes"`
|
||||
ReportedNodes int `json:"reported_nodes"`
|
||||
}
|
||||
|
||||
// OpenFlareOpenrestyHourly is an hourly OpenResty observation aggregation row.
|
||||
type OpenFlareOpenrestyHourly struct {
|
||||
Hour time.Time `json:"hour"`
|
||||
OpenrestyRxBytes int64 `json:"openresty_rx_bytes"`
|
||||
OpenrestyTxBytes int64 `json:"openresty_tx_bytes"`
|
||||
ReportedNodes int `json:"reported_nodes"`
|
||||
}
|
||||
|
||||
// ListOpenFlareMetricHourlySince returns hourly metric aggregates since the given time.
|
||||
func ListOpenFlareMetricHourlySince(ctx context.Context, nodeID string, since time.Time) ([]*OpenFlareMetricHourly, error) {
|
||||
rows, err := analyticsrepo.ListNodeMetricHourly(ctx, analyticsrepo.NodeObservabilityFilter{
|
||||
NodeID: nodeID,
|
||||
Since: since,
|
||||
})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
result := make([]*OpenFlareMetricHourly, len(rows))
|
||||
for index, row := range rows {
|
||||
result[index] = &OpenFlareMetricHourly{
|
||||
Hour: row.Hour,
|
||||
AverageCPUUsagePercent: row.AverageCPUUsagePercent,
|
||||
AverageMemoryUsagePercent: row.AverageMemoryUsagePercent,
|
||||
NetworkRxBytes: row.NetworkRxBytes,
|
||||
NetworkTxBytes: row.NetworkTxBytes,
|
||||
DiskReadBytes: row.DiskReadBytes,
|
||||
DiskWriteBytes: row.DiskWriteBytes,
|
||||
ReportedNodes: row.ReportedNodes,
|
||||
}
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// ListOpenFlareOpenrestyHourlySince returns hourly OpenResty aggregates since the given time.
|
||||
func ListOpenFlareOpenrestyHourlySince(ctx context.Context, nodeID string, since time.Time) ([]*OpenFlareOpenrestyHourly, error) {
|
||||
rows, err := analyticsrepo.ListNodeOpenrestyHourly(ctx, analyticsrepo.NodeObservabilityFilter{
|
||||
NodeID: nodeID,
|
||||
Since: since,
|
||||
})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
result := make([]*OpenFlareOpenrestyHourly, len(rows))
|
||||
for index, row := range rows {
|
||||
result[index] = &OpenFlareOpenrestyHourly{
|
||||
Hour: row.Hour,
|
||||
OpenrestyRxBytes: row.OpenrestyRxBytes,
|
||||
OpenrestyTxBytes: row.OpenrestyTxBytes,
|
||||
ReportedNodes: row.ReportedNodes,
|
||||
}
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// ListOpenFlareActiveHealthEvents returns active health events across all nodes.
|
||||
func ListOpenFlareActiveHealthEvents(ctx context.Context) ([]*OpenFlareHealthEvent, error) {
|
||||
conn := db.DB(ctx)
|
||||
|
||||
@@ -9,11 +9,38 @@ import (
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/Rain-kl/Wavelet/internal/apps/openflare/chwriter"
|
||||
analyticsmodel "github.com/Rain-kl/Wavelet/internal/model/analytics"
|
||||
analyticsrepo "github.com/Rain-kl/Wavelet/internal/repository/analytics"
|
||||
)
|
||||
|
||||
// ObservabilityInsertHooks queues observability rows for async ClickHouse write.
|
||||
// Wired from openflare/chwriter.Init so model never imports the apps layer.
|
||||
type ObservabilityInsertHooks struct {
|
||||
QueueMetricSnapshot func(analyticsmodel.NodeMetricSnapshot)
|
||||
QueueRequestReport func(analyticsmodel.NodeRequestReport)
|
||||
QueueOpenrestyObservation func(analyticsmodel.NodeObsOpenresty)
|
||||
QueueFrpsObservation func(analyticsmodel.NodeObsFrps)
|
||||
QueueFrpcObservation func(analyticsmodel.NodeObsFrpc)
|
||||
}
|
||||
|
||||
var (
|
||||
observabilityInsertHooksMu sync.RWMutex
|
||||
observabilityInsertHooks ObservabilityInsertHooks
|
||||
)
|
||||
|
||||
// SetObservabilityInsertHooks registers async queue callbacks for observability inserts.
|
||||
func SetObservabilityInsertHooks(hooks ObservabilityInsertHooks) {
|
||||
observabilityInsertHooksMu.Lock()
|
||||
observabilityInsertHooks = hooks
|
||||
observabilityInsertHooksMu.Unlock()
|
||||
}
|
||||
|
||||
func currentObservabilityInsertHooks() ObservabilityInsertHooks {
|
||||
observabilityInsertHooksMu.RLock()
|
||||
defer observabilityInsertHooksMu.RUnlock()
|
||||
return observabilityInsertHooks
|
||||
}
|
||||
|
||||
type observabilityStore interface {
|
||||
InsertMetricSnapshot(ctx context.Context, record *OpenFlareMetricSnapshot) error
|
||||
ListMetricSnapshots(ctx context.Context, nodeID string, since time.Time, limit int) ([]*OpenFlareMetricSnapshot, error)
|
||||
@@ -79,7 +106,9 @@ func (clickhouseObservabilityStore) InsertMetricSnapshot(_ context.Context, reco
|
||||
if record == nil {
|
||||
return nil
|
||||
}
|
||||
chwriter.QueueMetricSnapshot(toAnalyticsNodeMetricSnapshot(record))
|
||||
if hook := currentObservabilityInsertHooks().QueueMetricSnapshot; hook != nil {
|
||||
hook(toAnalyticsNodeMetricSnapshot(record))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -103,7 +132,9 @@ func (clickhouseObservabilityStore) InsertRequestReport(_ context.Context, recor
|
||||
if record == nil {
|
||||
return nil
|
||||
}
|
||||
chwriter.QueueRequestReport(toAnalyticsNodeRequestReport(record))
|
||||
if hook := currentObservabilityInsertHooks().QueueRequestReport; hook != nil {
|
||||
hook(toAnalyticsNodeRequestReport(record))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -127,7 +158,9 @@ func (clickhouseObservabilityStore) InsertNodeObservationOpenresty(_ context.Con
|
||||
if record == nil {
|
||||
return nil
|
||||
}
|
||||
chwriter.QueueOpenrestyObservation(toAnalyticsNodeObsOpenresty(record))
|
||||
if hook := currentObservabilityInsertHooks().QueueOpenrestyObservation; hook != nil {
|
||||
hook(toAnalyticsNodeObsOpenresty(record))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -151,7 +184,9 @@ func (clickhouseObservabilityStore) InsertNodeObservationFrps(_ context.Context,
|
||||
if record == nil {
|
||||
return nil
|
||||
}
|
||||
chwriter.QueueFrpsObservation(toAnalyticsNodeObsFrps(record))
|
||||
if hook := currentObservabilityInsertHooks().QueueFrpsObservation; hook != nil {
|
||||
hook(toAnalyticsNodeObsFrps(record))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -175,7 +210,9 @@ func (clickhouseObservabilityStore) InsertNodeObservationFrpc(_ context.Context,
|
||||
if record == nil {
|
||||
return nil
|
||||
}
|
||||
chwriter.QueueFrpcObservation(toAnalyticsNodeObsFrpc(record))
|
||||
if hook := currentObservabilityInsertHooks().QueueFrpcObservation; hook != nil {
|
||||
hook(toAnalyticsNodeObsFrpc(record))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
|
||||
@@ -99,6 +99,9 @@ type mockConn struct {
|
||||
batchQuery string
|
||||
prepareCalled bool
|
||||
preparedQuery string
|
||||
queries []string
|
||||
queryArgs [][]any
|
||||
queryFn func(ctx context.Context, query string, args ...any) (driver.Rows, error)
|
||||
}
|
||||
|
||||
func (m *mockConn) Contributors() []string { return nil }
|
||||
@@ -107,8 +110,13 @@ func (m *mockConn) ServerVersion() (*driver.ServerVersion, error) { return nil,
|
||||
|
||||
func (m *mockConn) Select(_ context.Context, _ any, _ string, _ ...any) error { return nil }
|
||||
|
||||
func (m *mockConn) Query(_ context.Context, _ string, _ ...any) (driver.Rows, error) {
|
||||
return nil, nil
|
||||
func (m *mockConn) Query(ctx context.Context, query string, args ...any) (driver.Rows, error) {
|
||||
m.queries = append(m.queries, query)
|
||||
m.queryArgs = append(m.queryArgs, args)
|
||||
if m.queryFn != nil {
|
||||
return m.queryFn(ctx, query, args...)
|
||||
}
|
||||
return &mockRows{}, nil
|
||||
}
|
||||
|
||||
func (m *mockConn) QueryRow(_ context.Context, _ string, _ ...any) driver.Row { return nil }
|
||||
@@ -158,4 +166,96 @@ func (m *mockBatch) Rows() int { return len(m.rows) }
|
||||
|
||||
func (m *mockBatch) Columns() []column.Interface { return nil }
|
||||
|
||||
func (m *mockBatch) Close() error { return nil }
|
||||
func (m *mockBatch) Close() error { return nil }
|
||||
|
||||
// mockRows is an empty driver.Rows implementation for query-path unit tests.
|
||||
type mockRows struct {
|
||||
index int
|
||||
data [][]any
|
||||
err error
|
||||
}
|
||||
|
||||
func (m *mockRows) Next() bool {
|
||||
if m.err != nil {
|
||||
return false
|
||||
}
|
||||
if m.index >= len(m.data) {
|
||||
return false
|
||||
}
|
||||
m.index++
|
||||
return true
|
||||
}
|
||||
|
||||
func (m *mockRows) Scan(dest ...any) error {
|
||||
if m.err != nil {
|
||||
return m.err
|
||||
}
|
||||
if m.index == 0 || m.index > len(m.data) {
|
||||
return nil
|
||||
}
|
||||
row := m.data[m.index-1]
|
||||
for i := range dest {
|
||||
if i >= len(row) {
|
||||
break
|
||||
}
|
||||
if err := assignMockScanValue(dest[i], row[i]); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (m *mockRows) ScanStruct(_ any) error { return nil }
|
||||
|
||||
func (m *mockRows) ColumnTypes() []driver.ColumnType { return nil }
|
||||
|
||||
func (m *mockRows) Totals(_ ...any) error { return nil }
|
||||
|
||||
func (m *mockRows) Columns() []string { return nil }
|
||||
|
||||
func (m *mockRows) Close() error { return nil }
|
||||
|
||||
func (m *mockRows) Err() error { return m.err }
|
||||
|
||||
func (m *mockRows) HasData() bool { return len(m.data) > 0 }
|
||||
|
||||
func assignMockScanValue(dest any, value any) error {
|
||||
switch d := dest.(type) {
|
||||
case *string:
|
||||
if v, ok := value.(string); ok {
|
||||
*d = v
|
||||
}
|
||||
case *uint64:
|
||||
switch v := value.(type) {
|
||||
case uint64:
|
||||
*d = v
|
||||
case int:
|
||||
*d = uint64(v)
|
||||
case int64:
|
||||
*d = uint64(v)
|
||||
}
|
||||
case *int64:
|
||||
switch v := value.(type) {
|
||||
case int64:
|
||||
*d = v
|
||||
case int:
|
||||
*d = int64(v)
|
||||
case uint64:
|
||||
*d = int64(v)
|
||||
}
|
||||
case *float64:
|
||||
switch v := value.(type) {
|
||||
case float64:
|
||||
*d = v
|
||||
case float32:
|
||||
*d = float64(v)
|
||||
case int:
|
||||
*d = float64(v)
|
||||
}
|
||||
case *time.Time:
|
||||
if v, ok := value.(time.Time); ok {
|
||||
*d = v
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -6,21 +6,48 @@ package analytics
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"time"
|
||||
|
||||
"github.com/ClickHouse/clickhouse-go/v2/lib/driver"
|
||||
)
|
||||
|
||||
// DDL TTL days for analytics tables (must match goose ClickHouse migrations).
|
||||
const (
|
||||
// TableTTLDaysNodeAccessLogs is the of_node_access_logs TTL (90 days).
|
||||
TableTTLDaysNodeAccessLogs = 90
|
||||
// TableTTLDaysNodeMetricSnapshots is the of_node_metric_snapshots TTL (30 days).
|
||||
TableTTLDaysNodeMetricSnapshots = 30
|
||||
// TableTTLDaysNodeRequestReports is the of_node_request_reports TTL (30 days).
|
||||
TableTTLDaysNodeRequestReports = 30
|
||||
// TableTTLDaysNodeObs is the of_node_obs_* TTL (30 days).
|
||||
TableTTLDaysNodeObs = 30
|
||||
// TableTTLDaysUserAccessLogs is the w_user_access_logs TTL (180 days).
|
||||
TableTTLDaysUserAccessLogs = 180
|
||||
)
|
||||
|
||||
const (
|
||||
// CleanupModeTTLMaterialize expires rows via table TTL instead of ALTER DELETE mutations.
|
||||
// This is not a hard delete: deleted_count must stay 0; use EligibleCount as an estimate.
|
||||
CleanupModeTTLMaterialize = "ttl_materialize"
|
||||
// CleanupModeTruncate removes all rows via TRUNCATE TABLE.
|
||||
// CleanupModeTruncate removes all rows via TRUNCATE TABLE (hard delete).
|
||||
CleanupModeTruncate = "truncate"
|
||||
)
|
||||
|
||||
// CleanupOutcome describes a non-mutation ClickHouse cleanup operation.
|
||||
//
|
||||
// For CleanupModeTruncate:
|
||||
// - DeletedCount and EligibleCount are the rows removed by TRUNCATE.
|
||||
//
|
||||
// For CleanupModeTTLMaterialize:
|
||||
// - DeletedCount is always 0 (MATERIALIZE TTL is async / not a counted hard delete).
|
||||
// - EligibleCount is an estimate of rows already past the table TTL policy (not an
|
||||
// arbitrary user cutoff younger than the DDL TTL).
|
||||
// - TableTTLDays is the DDL TTL used for the estimate and materialize.
|
||||
type CleanupOutcome struct {
|
||||
EligibleCount int64
|
||||
DeletedCount int64
|
||||
Mode string
|
||||
TableTTLDays int
|
||||
}
|
||||
|
||||
func countClickHouseRows(ctx context.Context, conn driver.Conn, countSQL string, countArgs []any) (int64, error) {
|
||||
@@ -39,21 +66,41 @@ func materializeTableTTL(ctx context.Context, conn driver.Conn, tableName string
|
||||
return nil
|
||||
}
|
||||
|
||||
func expireRowsViaTTL(ctx context.Context, conn driver.Conn, tableName string, countSQL string, countArgs []any) (CleanupOutcome, error) {
|
||||
// tableTTLCutoff returns the UTC instant at which rows become eligible under a fixed day TTL.
|
||||
func tableTTLCutoff(tableTTLDays int, now time.Time) time.Time {
|
||||
if tableTTLDays < 1 {
|
||||
tableTTLDays = 1
|
||||
}
|
||||
return now.UTC().Add(-time.Duration(tableTTLDays) * 24 * time.Hour)
|
||||
}
|
||||
|
||||
// materializeExpiredByTableTTL force-materializes table TTL and estimates rows past that policy.
|
||||
//
|
||||
// countSQL must count only rows older than the table TTL (callers pass tableTTLCutoff args).
|
||||
// Node-scoped filters may be used for the estimate only; MATERIALIZE is always table-global.
|
||||
func materializeExpiredByTableTTL(
|
||||
ctx context.Context,
|
||||
conn driver.Conn,
|
||||
tableName string,
|
||||
tableTTLDays int,
|
||||
countSQL string,
|
||||
countArgs []any,
|
||||
) (CleanupOutcome, error) {
|
||||
outcome := CleanupOutcome{
|
||||
Mode: CleanupModeTTLMaterialize,
|
||||
TableTTLDays: tableTTLDays,
|
||||
}
|
||||
count, err := countClickHouseRows(ctx, conn, countSQL, countArgs)
|
||||
if err != nil {
|
||||
return CleanupOutcome{}, err
|
||||
}
|
||||
if count == 0 {
|
||||
return CleanupOutcome{Mode: CleanupModeTTLMaterialize}, nil
|
||||
}
|
||||
outcome.EligibleCount = count
|
||||
// Always force materialize so ClickHouse applies the DDL TTL policy promptly.
|
||||
// EligibleCount is informational only; MATERIALIZE does not return a deleted row count.
|
||||
if err := materializeTableTTL(ctx, conn, tableName); err != nil {
|
||||
return CleanupOutcome{}, err
|
||||
}
|
||||
return CleanupOutcome{
|
||||
EligibleCount: count,
|
||||
Mode: CleanupModeTTLMaterialize,
|
||||
}, nil
|
||||
return outcome, nil
|
||||
}
|
||||
|
||||
func truncateClickHouseTable(ctx context.Context, conn driver.Conn, tableName string) (CleanupOutcome, error) {
|
||||
@@ -69,6 +116,7 @@ func truncateClickHouseTable(ctx context.Context, conn driver.Conn, tableName st
|
||||
}
|
||||
return CleanupOutcome{
|
||||
EligibleCount: count,
|
||||
DeletedCount: count,
|
||||
Mode: CleanupModeTruncate,
|
||||
}, nil
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,37 @@
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
package analytics
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/stretchr/testify/assert"
|
||||
)
|
||||
|
||||
func TestTableTTLCutoff(t *testing.T) {
|
||||
now := time.Date(2026, 7, 10, 12, 0, 0, 0, time.UTC)
|
||||
got := tableTTLCutoff(30, now)
|
||||
assert.Equal(t, now.Add(-30*24*time.Hour), got)
|
||||
|
||||
got = tableTTLCutoff(90, now)
|
||||
assert.Equal(t, now.Add(-90*24*time.Hour), got)
|
||||
|
||||
// Invalid TTL floors to 1 day.
|
||||
got = tableTTLCutoff(0, now)
|
||||
assert.Equal(t, now.Add(-24*time.Hour), got)
|
||||
}
|
||||
|
||||
func TestCleanupModeConstants(t *testing.T) {
|
||||
assert.Equal(t, "ttl_materialize", CleanupModeTTLMaterialize)
|
||||
assert.Equal(t, "truncate", CleanupModeTruncate)
|
||||
}
|
||||
|
||||
func TestTableTTLDaysMatchDDL(t *testing.T) {
|
||||
assert.Equal(t, 90, TableTTLDaysNodeAccessLogs)
|
||||
assert.Equal(t, 30, TableTTLDaysNodeMetricSnapshots)
|
||||
assert.Equal(t, 30, TableTTLDaysNodeRequestReports)
|
||||
assert.Equal(t, 30, TableTTLDaysNodeObs)
|
||||
assert.Equal(t, 180, TableTTLDaysUserAccessLogs)
|
||||
}
|
||||
@@ -9,16 +9,20 @@ import (
|
||||
|
||||
"github.com/Rain-kl/Wavelet/internal/config"
|
||||
"github.com/Rain-kl/Wavelet/internal/db"
|
||||
"github.com/Rain-kl/Wavelet/internal/db/batchwriter"
|
||||
)
|
||||
|
||||
// ClickHouseOperationalStats summarizes ClickHouse merge/mutation pressure.
|
||||
// ClickHouseOperationalStats summarizes ClickHouse merge/mutation pressure
|
||||
// and in-process batch writer queue health.
|
||||
type ClickHouseOperationalStats struct {
|
||||
Database string `json:"database"`
|
||||
ActiveParts int64 `json:"active_parts"`
|
||||
TotalRows int64 `json:"total_rows"`
|
||||
PendingMutations int64 `json:"pending_mutations"`
|
||||
AsyncInsertQueue int64 `json:"async_insert_queue"`
|
||||
AsyncInsertBytes int64 `json:"async_insert_bytes"`
|
||||
Database string `json:"database"`
|
||||
ActiveParts int64 `json:"active_parts"`
|
||||
TotalRows int64 `json:"total_rows"`
|
||||
PendingMutations int64 `json:"pending_mutations"`
|
||||
AsyncInsertQueue int64 `json:"async_insert_queue"`
|
||||
AsyncInsertBytes int64 `json:"async_insert_bytes"`
|
||||
// BatchWriters reports in-process queue depth/drops/flush errors for CH writers.
|
||||
BatchWriters []batchwriter.Stats `json:"batch_writers,omitempty"`
|
||||
}
|
||||
|
||||
// GetClickHouseOperationalStats returns operational metrics for the configured database.
|
||||
@@ -67,4 +71,4 @@ WHERE database = ?`
|
||||
}
|
||||
|
||||
return stats, nil
|
||||
}
|
||||
}
|
||||
|
||||
@@ -9,7 +9,7 @@ import (
|
||||
"time"
|
||||
)
|
||||
|
||||
// DeleteAllNodeAccessLogs deletes all node access logs.
|
||||
// DeleteAllNodeAccessLogs hard-deletes all node access logs via TRUNCATE.
|
||||
func DeleteAllNodeAccessLogs(ctx context.Context) (int64, error) {
|
||||
conn, err := nodeAccessLogConn()
|
||||
if err != nil {
|
||||
@@ -19,47 +19,69 @@ func DeleteAllNodeAccessLogs(ctx context.Context) (int64, error) {
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.EligibleCount, nil
|
||||
return outcome.DeletedCount, nil
|
||||
}
|
||||
|
||||
// DeleteNodeAccessLogsBefore expires logs older than cutoff via table TTL.
|
||||
func DeleteNodeAccessLogsBefore(ctx context.Context, cutoff time.Time) (int64, error) {
|
||||
conn, err := nodeAccessLogConn()
|
||||
// DeleteNodeAccessLogsBefore force-materializes of_node_access_logs table TTL.
|
||||
//
|
||||
// The cutoff argument is kept for call-site compatibility and is not used to select rows:
|
||||
// ClickHouse MATERIALIZE TTL only enforces the DDL policy (TableTTLDaysNodeAccessLogs).
|
||||
// Returns an estimate of rows past table TTL as the int64 (not a hard-deleted count).
|
||||
// Callers that need honest API fields should prefer MaterializeNodeAccessLogsTTL.
|
||||
func DeleteNodeAccessLogsBefore(ctx context.Context, _ time.Time) (int64, error) {
|
||||
outcome, err := MaterializeNodeAccessLogsTTL(ctx)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.EligibleCount, nil
|
||||
}
|
||||
|
||||
// MaterializeNodeAccessLogsTTL force-materializes table TTL and reports an honest outcome.
|
||||
func MaterializeNodeAccessLogsTTL(ctx context.Context) (CleanupOutcome, error) {
|
||||
conn, err := nodeAccessLogConn()
|
||||
if err != nil {
|
||||
return CleanupOutcome{}, err
|
||||
}
|
||||
tableName := nodeAccessLogTableName()
|
||||
cutoff = cutoff.UTC()
|
||||
outcome, err := expireRowsViaTTL(
|
||||
ttlDays := TableTTLDaysNodeAccessLogs
|
||||
cutoff := tableTTLCutoff(ttlDays, time.Now())
|
||||
return materializeExpiredByTableTTL(
|
||||
ctx,
|
||||
conn,
|
||||
tableName,
|
||||
ttlDays,
|
||||
fmt.Sprintf("SELECT count() FROM %s WHERE logged_at < ?", tableName),
|
||||
[]any{cutoff},
|
||||
)
|
||||
}
|
||||
|
||||
// DeleteNodeAccessLogsByNodeBefore force-materializes table-global TTL.
|
||||
//
|
||||
// Node-scoped hard delete is not supported: MATERIALIZE TTL is table-global.
|
||||
// The returned count is an estimate of rows for nodeID past table TTL only.
|
||||
func DeleteNodeAccessLogsByNodeBefore(ctx context.Context, nodeID string, _ time.Time) (int64, error) {
|
||||
outcome, err := MaterializeNodeAccessLogsTTLByNode(ctx, nodeID)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.EligibleCount, nil
|
||||
}
|
||||
|
||||
// DeleteNodeAccessLogsByNodeBefore expires logs for a node older than cutoff via table TTL.
|
||||
func DeleteNodeAccessLogsByNodeBefore(ctx context.Context, nodeID string, before time.Time) (int64, error) {
|
||||
// MaterializeNodeAccessLogsTTLByNode materializes table-global TTL and estimates node-scoped rows past TTL.
|
||||
func MaterializeNodeAccessLogsTTLByNode(ctx context.Context, nodeID string) (CleanupOutcome, error) {
|
||||
conn, err := nodeAccessLogConn()
|
||||
if err != nil {
|
||||
return 0, err
|
||||
return CleanupOutcome{}, err
|
||||
}
|
||||
tableName := nodeAccessLogTableName()
|
||||
before = before.UTC()
|
||||
outcome, err := expireRowsViaTTL(
|
||||
ttlDays := TableTTLDaysNodeAccessLogs
|
||||
cutoff := tableTTLCutoff(ttlDays, time.Now())
|
||||
return materializeExpiredByTableTTL(
|
||||
ctx,
|
||||
conn,
|
||||
tableName,
|
||||
ttlDays,
|
||||
fmt.Sprintf("SELECT count() FROM %s WHERE node_id = ? AND logged_at < ?", tableName),
|
||||
[]any{nodeID, before},
|
||||
[]any{nodeID, cutoff},
|
||||
)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.EligibleCount, nil
|
||||
}
|
||||
}
|
||||
|
||||
@@ -6,6 +6,7 @@ package analytics
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"sort"
|
||||
"time"
|
||||
|
||||
"github.com/ClickHouse/clickhouse-go/v2/lib/driver"
|
||||
@@ -45,6 +46,27 @@ ORDER BY %s`, tableName, clause, nodeObservabilityCapturedAtOrderClause())
|
||||
return scanNodeMetricSnapshotRows(rows)
|
||||
}
|
||||
|
||||
// ListLatestNodeMetricSnapshots returns the latest metric snapshot per node_id.
|
||||
// Uses ClickHouse LIMIT 1 BY so dashboard health does not depend on a global raw LIMIT.
|
||||
func ListLatestNodeMetricSnapshots(ctx context.Context, filter NodeObservabilityFilter) ([]analyticsmodel.NodeMetricSnapshot, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
clause, args := buildNodeObservabilityFilterClause(filter, "captured_at")
|
||||
sql := fmt.Sprintf(`
|
||||
SELECT id, node_id, captured_at, cpu_usage_percent, memory_used_bytes, memory_total_bytes, storage_used_bytes, storage_total_bytes, disk_read_bytes, disk_write_bytes, network_rx_bytes, network_tx_bytes, created_at
|
||||
FROM %s
|
||||
WHERE %s
|
||||
ORDER BY %s%s`, nodeMetricSnapshotTableName(), clause, nodeObservabilityCapturedAtOrderClause(), clickHouseLimit1ByNodeIDClause)
|
||||
rows, err := conn.Query(ctx, sql, args...)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("list latest node metric snapshots: %w", err)
|
||||
}
|
||||
defer func() { _ = rows.Close() }()
|
||||
return scanNodeMetricSnapshotRows(rows)
|
||||
}
|
||||
|
||||
// ListNodeRequestReports returns request reports matching filter.
|
||||
func ListNodeRequestReports(ctx context.Context, filter NodeObservabilityFilter) ([]analyticsmodel.NodeRequestReport, error) {
|
||||
conn, err := observabilityConn()
|
||||
@@ -70,6 +92,27 @@ ORDER BY %s`, tableName, clause, nodeObservabilityWindowEndedAtOrderClause())
|
||||
return scanNodeRequestReportRows(rows)
|
||||
}
|
||||
|
||||
// ListLatestNodeRequestReports returns the latest request report per node_id.
|
||||
// Uses ClickHouse LIMIT 1 BY so dashboard traffic health is not skewed by a global raw LIMIT.
|
||||
func ListLatestNodeRequestReports(ctx context.Context, filter NodeObservabilityFilter) ([]analyticsmodel.NodeRequestReport, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
clause, args := buildNodeObservabilityFilterClause(filter, "window_ended_at")
|
||||
sql := fmt.Sprintf(`
|
||||
SELECT id, node_id, window_started_at, window_ended_at, request_count, error_count, unique_visitor_count, status_codes_json, top_domains_json, source_countries_json, created_at
|
||||
FROM %s
|
||||
WHERE %s
|
||||
ORDER BY %s%s`, nodeRequestReportTableName(), clause, nodeObservabilityWindowEndedAtOrderClause(), clickHouseLimit1ByNodeIDClause)
|
||||
rows, err := conn.Query(ctx, sql, args...)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("list latest node request reports: %w", err)
|
||||
}
|
||||
defer func() { _ = rows.Close() }()
|
||||
return scanNodeRequestReportRows(rows)
|
||||
}
|
||||
|
||||
// ListNodeObsOpenresty returns OpenResty observations matching filter.
|
||||
func ListNodeObsOpenresty(ctx context.Context, filter NodeObservabilityFilter) ([]analyticsmodel.NodeObsOpenresty, error) {
|
||||
conn, err := observabilityConn()
|
||||
@@ -248,6 +291,10 @@ func scanNodeObsFrpsRows(rows driver.Rows) ([]analyticsmodel.NodeObsFrps, error)
|
||||
const nodeTrafficHourlyTableName = "of_node_traffic_hourly"
|
||||
|
||||
// NodeTrafficHourly is an hourly traffic rollup row.
|
||||
//
|
||||
// UniqueVisitorCount is a peak per-window estimate from short request reports
|
||||
// (MV uses max()), not true distinct visitors across the hour. SummingMergeTree
|
||||
// may still inflate residual unmerged parts; do not present as exact UV.
|
||||
type NodeTrafficHourly struct {
|
||||
NodeID string
|
||||
Hour time.Time
|
||||
@@ -256,6 +303,30 @@ type NodeTrafficHourly struct {
|
||||
UniqueVisitorCount int64
|
||||
}
|
||||
|
||||
// NodeMetricHourly is an hourly metric snapshot aggregation row.
|
||||
//
|
||||
// Disk and host network counters are cumulative. Prefer pre-aggregated min/max
|
||||
// deltas from of_node_metric_capacity_hourly; raw fallback uses consecutive
|
||||
// lagInFrame samples per node (negative deltas after counter reset are dropped).
|
||||
type NodeMetricHourly struct {
|
||||
Hour time.Time
|
||||
AverageCPUUsagePercent float64
|
||||
AverageMemoryUsagePercent float64
|
||||
NetworkRxBytes int64
|
||||
NetworkTxBytes int64
|
||||
DiskReadBytes int64
|
||||
DiskWriteBytes int64
|
||||
ReportedNodes int
|
||||
}
|
||||
|
||||
// NodeOpenrestyHourly is an hourly OpenResty observation aggregation row.
|
||||
type NodeOpenrestyHourly struct {
|
||||
Hour time.Time
|
||||
OpenrestyRxBytes int64
|
||||
OpenrestyTxBytes int64
|
||||
ReportedNodes int
|
||||
}
|
||||
|
||||
// ListNodeTrafficHourly returns hourly traffic rollup rows matching filter.
|
||||
func ListNodeTrafficHourly(ctx context.Context, filter NodeObservabilityFilter) ([]NodeTrafficHourly, error) {
|
||||
conn, err := observabilityConn()
|
||||
@@ -269,7 +340,7 @@ SELECT
|
||||
hour,
|
||||
sum(request_count) AS request_count,
|
||||
sum(error_count) AS error_count,
|
||||
sum(unique_visitor_count) AS unique_visitor_count
|
||||
max(unique_visitor_count) AS unique_visitor_count
|
||||
FROM %s
|
||||
WHERE %s
|
||||
GROUP BY node_id, hour
|
||||
@@ -298,6 +369,325 @@ ORDER BY hour ASC`, nodeTrafficHourlyTableName, clause)
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// hourlyRollupMaxLead is how far after filter.Since the earliest rollup bucket may start
|
||||
// while still treating pre-aggregated tables as a complete window (skip raw query).
|
||||
const hourlyRollupMaxLead = 2 * time.Hour
|
||||
|
||||
// hourlyRollupCoversWindow reports whether rollup coverage starts near the requested window.
|
||||
// rows must be ordered by hour ascending.
|
||||
func hourlyRollupCoversWindow(earliestHour time.Time, since time.Time) bool {
|
||||
if since.IsZero() {
|
||||
return true
|
||||
}
|
||||
sinceHour := since.UTC().Truncate(time.Hour)
|
||||
earliest := earliestHour.UTC().Truncate(time.Hour)
|
||||
return !earliest.After(sinceHour.Add(hourlyRollupMaxLead))
|
||||
}
|
||||
|
||||
// ListNodeMetricHourly returns hourly metric snapshot aggregates matching filter.
|
||||
//
|
||||
// Strategy (optimal for correctness + cost):
|
||||
// 1. Load of_node_metric_capacity_hourly rollup.
|
||||
// 2. If rollup spans the window from filter.Since, return it alone (cheap path).
|
||||
// 3. Otherwise load raw lagInFrame aggregates and merge by hour: rollup wins on
|
||||
// overlap, raw fills historical gaps (MV never backfills pre-creation data).
|
||||
func ListNodeMetricHourly(ctx context.Context, filter NodeObservabilityFilter) ([]NodeMetricHourly, error) {
|
||||
rollup, rollupErr := listNodeMetricHourlyFromRollup(ctx, filter)
|
||||
if rollupErr == nil && len(rollup) > 0 && hourlyRollupCoversWindow(rollup[0].Hour, filter.Since) {
|
||||
return rollup, nil
|
||||
}
|
||||
|
||||
raw, rawErr := listNodeMetricHourlyFromRaw(ctx, filter)
|
||||
if rawErr != nil {
|
||||
if rollupErr == nil && len(rollup) > 0 {
|
||||
return rollup, nil
|
||||
}
|
||||
return nil, rawErr
|
||||
}
|
||||
if len(rollup) == 0 {
|
||||
return raw, nil
|
||||
}
|
||||
// Partial rollup (or rollupErr with empty slice): merge; raw fills historical gaps.
|
||||
return mergeNodeMetricHourlyPreferRollup(rollup, raw), nil
|
||||
}
|
||||
|
||||
// mergeNodeMetricHourlyPreferRollup unions two hour series (both ASC by Hour).
|
||||
// Rollup values replace raw for the same hour; raw supplies missing hours.
|
||||
func mergeNodeMetricHourlyPreferRollup(rollup, raw []NodeMetricHourly) []NodeMetricHourly {
|
||||
byHour := make(map[int64]NodeMetricHourly, len(raw)+len(rollup))
|
||||
order := make([]int64, 0, len(raw)+len(rollup))
|
||||
add := func(row NodeMetricHourly, overwrite bool) {
|
||||
key := row.Hour.UTC().Truncate(time.Hour).Unix()
|
||||
if _, exists := byHour[key]; !exists {
|
||||
order = append(order, key)
|
||||
byHour[key] = row
|
||||
return
|
||||
}
|
||||
if overwrite {
|
||||
byHour[key] = row
|
||||
}
|
||||
}
|
||||
for _, row := range raw {
|
||||
add(row, false)
|
||||
}
|
||||
for _, row := range rollup {
|
||||
add(row, true)
|
||||
}
|
||||
result := make([]NodeMetricHourly, 0, len(order))
|
||||
// Keep chronological order of first-seen keys; re-sort by hour for stability.
|
||||
sort.Slice(order, func(i, j int) bool { return order[i] < order[j] })
|
||||
for _, key := range order {
|
||||
result = append(result, byHour[key])
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
func listNodeMetricHourlyFromRollup(ctx context.Context, filter NodeObservabilityFilter) ([]NodeMetricHourly, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
clause, args := buildNodeObservabilityFilterClause(filter, "hour")
|
||||
sql := fmt.Sprintf(`
|
||||
SELECT
|
||||
hour,
|
||||
if(sum(cpu_usage_count) > 0, sum(cpu_usage_sum) / sum(cpu_usage_count), 0) AS average_cpu_usage_percent,
|
||||
if(sum(memory_usage_count) > 0, sum(memory_usage_sum) / sum(memory_usage_count), 0) AS average_memory_usage_percent,
|
||||
sum(greatest(network_rx_max - network_rx_min, 0)) AS network_rx_bytes,
|
||||
sum(greatest(network_tx_max - network_tx_min, 0)) AS network_tx_bytes,
|
||||
sum(greatest(disk_read_max - disk_read_min, 0)) AS disk_read_bytes,
|
||||
sum(greatest(disk_write_max - disk_write_min, 0)) AS disk_write_bytes,
|
||||
toUInt64(uniqExact(node_id)) AS reported_nodes
|
||||
FROM %s
|
||||
WHERE %s
|
||||
GROUP BY hour
|
||||
ORDER BY hour ASC`, nodeMetricCapacityHourlyTableName(), clause)
|
||||
rows, err := conn.Query(ctx, sql, args...)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("list node metric hourly from rollup: %w", err)
|
||||
}
|
||||
defer func() { _ = rows.Close() }()
|
||||
return scanNodeMetricHourlyRows(rows)
|
||||
}
|
||||
|
||||
func listNodeMetricHourlyFromRaw(ctx context.Context, filter NodeObservabilityFilter) ([]NodeMetricHourly, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
clause, args := buildNodeObservabilityFilterClause(filter, "captured_at")
|
||||
tableName := nodeMetricSnapshotTableName()
|
||||
sql := fmt.Sprintf(`
|
||||
SELECT
|
||||
hour,
|
||||
avg(cpu_usage_percent) AS average_cpu_usage_percent,
|
||||
avg(memory_usage_percent) AS average_memory_usage_percent,
|
||||
sum(if(network_rx_delta >= 0, network_rx_delta, 0)) AS network_rx_bytes,
|
||||
sum(if(network_tx_delta >= 0, network_tx_delta, 0)) AS network_tx_bytes,
|
||||
sum(if(disk_read_delta >= 0, disk_read_delta, 0)) AS disk_read_bytes,
|
||||
sum(if(disk_write_delta >= 0, disk_write_delta, 0)) AS disk_write_bytes,
|
||||
toUInt64(uniqExact(node_id)) AS reported_nodes
|
||||
FROM (
|
||||
SELECT
|
||||
node_id,
|
||||
toStartOfHour(captured_at) AS hour,
|
||||
cpu_usage_percent,
|
||||
if(memory_total_bytes > 0, (memory_used_bytes * 100.0) / memory_total_bytes, 0) AS memory_usage_percent,
|
||||
network_rx_bytes - lagInFrame(network_rx_bytes, 1, network_rx_bytes) OVER (
|
||||
PARTITION BY node_id ORDER BY captured_at, id
|
||||
ROWS BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW
|
||||
) AS network_rx_delta,
|
||||
network_tx_bytes - lagInFrame(network_tx_bytes, 1, network_tx_bytes) OVER (
|
||||
PARTITION BY node_id ORDER BY captured_at, id
|
||||
ROWS BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW
|
||||
) AS network_tx_delta,
|
||||
disk_read_bytes - lagInFrame(disk_read_bytes, 1, disk_read_bytes) OVER (
|
||||
PARTITION BY node_id ORDER BY captured_at, id
|
||||
ROWS BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW
|
||||
) AS disk_read_delta,
|
||||
disk_write_bytes - lagInFrame(disk_write_bytes, 1, disk_write_bytes) OVER (
|
||||
PARTITION BY node_id ORDER BY captured_at, id
|
||||
ROWS BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW
|
||||
) AS disk_write_delta
|
||||
FROM %s
|
||||
WHERE %s
|
||||
)
|
||||
GROUP BY hour
|
||||
ORDER BY hour ASC`, tableName, clause)
|
||||
rows, err := conn.Query(ctx, sql, args...)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("list node metric hourly: %w", err)
|
||||
}
|
||||
defer func() { _ = rows.Close() }()
|
||||
return scanNodeMetricHourlyRows(rows)
|
||||
}
|
||||
|
||||
func scanNodeMetricHourlyRows(rows driver.Rows) ([]NodeMetricHourly, error) {
|
||||
result := make([]NodeMetricHourly, 0)
|
||||
for rows.Next() {
|
||||
var (
|
||||
item NodeMetricHourly
|
||||
reportedNodes uint64
|
||||
networkRx int64
|
||||
networkTx int64
|
||||
diskRead int64
|
||||
diskWrite int64
|
||||
)
|
||||
if err := rows.Scan(
|
||||
&item.Hour,
|
||||
&item.AverageCPUUsagePercent,
|
||||
&item.AverageMemoryUsagePercent,
|
||||
&networkRx,
|
||||
&networkTx,
|
||||
&diskRead,
|
||||
&diskWrite,
|
||||
&reportedNodes,
|
||||
); err != nil {
|
||||
return nil, fmt.Errorf("scan node metric hourly row: %w", err)
|
||||
}
|
||||
item.Hour = item.Hour.UTC()
|
||||
item.NetworkRxBytes = networkRx
|
||||
item.NetworkTxBytes = networkTx
|
||||
item.DiskReadBytes = diskRead
|
||||
item.DiskWriteBytes = diskWrite
|
||||
item.ReportedNodes = int(safeInt64Count(reportedNodes))
|
||||
result = append(result, item)
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// ListNodeOpenrestyHourly returns hourly OpenResty observation aggregates matching filter.
|
||||
// Same rollup-first / per-hour merge strategy as ListNodeMetricHourly.
|
||||
func ListNodeOpenrestyHourly(ctx context.Context, filter NodeObservabilityFilter) ([]NodeOpenrestyHourly, error) {
|
||||
rollup, rollupErr := listNodeOpenrestyHourlyFromRollup(ctx, filter)
|
||||
if rollupErr == nil && len(rollup) > 0 && hourlyRollupCoversWindow(rollup[0].Hour, filter.Since) {
|
||||
return rollup, nil
|
||||
}
|
||||
|
||||
raw, rawErr := listNodeOpenrestyHourlyFromRaw(ctx, filter)
|
||||
if rawErr != nil {
|
||||
if rollupErr == nil && len(rollup) > 0 {
|
||||
return rollup, nil
|
||||
}
|
||||
return nil, rawErr
|
||||
}
|
||||
if len(rollup) == 0 {
|
||||
return raw, nil
|
||||
}
|
||||
return mergeNodeOpenrestyHourlyPreferRollup(rollup, raw), nil
|
||||
}
|
||||
|
||||
func mergeNodeOpenrestyHourlyPreferRollup(rollup, raw []NodeOpenrestyHourly) []NodeOpenrestyHourly {
|
||||
byHour := make(map[int64]NodeOpenrestyHourly, len(raw)+len(rollup))
|
||||
order := make([]int64, 0, len(raw)+len(rollup))
|
||||
add := func(row NodeOpenrestyHourly, overwrite bool) {
|
||||
key := row.Hour.UTC().Truncate(time.Hour).Unix()
|
||||
if _, exists := byHour[key]; !exists {
|
||||
order = append(order, key)
|
||||
byHour[key] = row
|
||||
return
|
||||
}
|
||||
if overwrite {
|
||||
byHour[key] = row
|
||||
}
|
||||
}
|
||||
for _, row := range raw {
|
||||
add(row, false)
|
||||
}
|
||||
for _, row := range rollup {
|
||||
add(row, true)
|
||||
}
|
||||
sort.Slice(order, func(i, j int) bool { return order[i] < order[j] })
|
||||
result := make([]NodeOpenrestyHourly, 0, len(order))
|
||||
for _, key := range order {
|
||||
result = append(result, byHour[key])
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
func listNodeOpenrestyHourlyFromRollup(ctx context.Context, filter NodeObservabilityFilter) ([]NodeOpenrestyHourly, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
clause, args := buildNodeObservabilityFilterClause(filter, "hour")
|
||||
sql := fmt.Sprintf(`
|
||||
SELECT
|
||||
hour,
|
||||
sum(greatest(openresty_rx_max - openresty_rx_min, 0)) AS openresty_rx_bytes,
|
||||
sum(greatest(openresty_tx_max - openresty_tx_min, 0)) AS openresty_tx_bytes,
|
||||
toUInt64(uniqExact(node_id)) AS reported_nodes
|
||||
FROM %s
|
||||
WHERE %s
|
||||
GROUP BY hour
|
||||
ORDER BY hour ASC`, nodeOpenrestyHourlyTableName(), clause)
|
||||
rows, err := conn.Query(ctx, sql, args...)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("list node openresty hourly from rollup: %w", err)
|
||||
}
|
||||
defer func() { _ = rows.Close() }()
|
||||
return scanNodeOpenrestyHourlyRows(rows)
|
||||
}
|
||||
|
||||
func listNodeOpenrestyHourlyFromRaw(ctx context.Context, filter NodeObservabilityFilter) ([]NodeOpenrestyHourly, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
clause, args := buildNodeObservabilityFilterClause(filter, "captured_at")
|
||||
tableName := nodeObsOpenrestyTableName()
|
||||
sql := fmt.Sprintf(`
|
||||
SELECT
|
||||
hour,
|
||||
sum(if(openresty_rx_delta >= 0, openresty_rx_delta, 0)) AS openresty_rx_bytes,
|
||||
sum(if(openresty_tx_delta >= 0, openresty_tx_delta, 0)) AS openresty_tx_bytes,
|
||||
toUInt64(uniqExact(node_id)) AS reported_nodes
|
||||
FROM (
|
||||
SELECT
|
||||
node_id,
|
||||
toStartOfHour(captured_at) AS hour,
|
||||
openresty_rx_bytes - lagInFrame(openresty_rx_bytes, 1, openresty_rx_bytes) OVER (
|
||||
PARTITION BY node_id ORDER BY captured_at, id
|
||||
ROWS BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW
|
||||
) AS openresty_rx_delta,
|
||||
openresty_tx_bytes - lagInFrame(openresty_tx_bytes, 1, openresty_tx_bytes) OVER (
|
||||
PARTITION BY node_id ORDER BY captured_at, id
|
||||
ROWS BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW
|
||||
) AS openresty_tx_delta
|
||||
FROM %s
|
||||
WHERE %s
|
||||
)
|
||||
GROUP BY hour
|
||||
ORDER BY hour ASC`, tableName, clause)
|
||||
rows, err := conn.Query(ctx, sql, args...)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("list node openresty hourly: %w", err)
|
||||
}
|
||||
defer func() { _ = rows.Close() }()
|
||||
return scanNodeOpenrestyHourlyRows(rows)
|
||||
}
|
||||
|
||||
func scanNodeOpenrestyHourlyRows(rows driver.Rows) ([]NodeOpenrestyHourly, error) {
|
||||
result := make([]NodeOpenrestyHourly, 0)
|
||||
for rows.Next() {
|
||||
var (
|
||||
item NodeOpenrestyHourly
|
||||
reportedNodes uint64
|
||||
rx int64
|
||||
tx int64
|
||||
)
|
||||
if err := rows.Scan(&item.Hour, &rx, &tx, &reportedNodes); err != nil {
|
||||
return nil, fmt.Errorf("scan node openresty hourly row: %w", err)
|
||||
}
|
||||
item.Hour = item.Hour.UTC()
|
||||
item.OpenrestyRxBytes = rx
|
||||
item.OpenrestyTxBytes = tx
|
||||
item.ReportedNodes = int(safeInt64Count(reportedNodes))
|
||||
result = append(result, item)
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
func scanNodeObsFrpcRows(rows driver.Rows) ([]analyticsmodel.NodeObsFrpc, error) {
|
||||
var result []analyticsmodel.NodeObsFrpc
|
||||
for rows.Next() {
|
||||
|
||||
@@ -9,7 +9,7 @@ import (
|
||||
"time"
|
||||
)
|
||||
|
||||
// DeleteAllNodeMetricSnapshots deletes all node metric snapshots.
|
||||
// DeleteAllNodeMetricSnapshots hard-deletes all node metric snapshots via TRUNCATE.
|
||||
func DeleteAllNodeMetricSnapshots(ctx context.Context) (int64, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
@@ -19,31 +19,39 @@ func DeleteAllNodeMetricSnapshots(ctx context.Context) (int64, error) {
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.EligibleCount, nil
|
||||
return outcome.DeletedCount, nil
|
||||
}
|
||||
|
||||
// DeleteNodeMetricSnapshotsBefore expires metric snapshots captured before cutoff via table TTL.
|
||||
func DeleteNodeMetricSnapshotsBefore(ctx context.Context, cutoff time.Time) (int64, error) {
|
||||
conn, err := observabilityConn()
|
||||
// DeleteNodeMetricSnapshotsBefore force-materializes of_node_metric_snapshots table TTL.
|
||||
// cutoff is ignored; see MaterializeNodeMetricSnapshotsTTL.
|
||||
func DeleteNodeMetricSnapshotsBefore(ctx context.Context, _ time.Time) (int64, error) {
|
||||
outcome, err := MaterializeNodeMetricSnapshotsTTL(ctx)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.EligibleCount, nil
|
||||
}
|
||||
|
||||
// MaterializeNodeMetricSnapshotsTTL force-materializes table TTL and reports an honest outcome.
|
||||
func MaterializeNodeMetricSnapshotsTTL(ctx context.Context) (CleanupOutcome, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return CleanupOutcome{}, err
|
||||
}
|
||||
tableName := nodeMetricSnapshotTableName()
|
||||
cutoff = cutoff.UTC()
|
||||
outcome, err := expireRowsViaTTL(
|
||||
ttlDays := TableTTLDaysNodeMetricSnapshots
|
||||
cutoff := tableTTLCutoff(ttlDays, time.Now())
|
||||
return materializeExpiredByTableTTL(
|
||||
ctx,
|
||||
conn,
|
||||
tableName,
|
||||
ttlDays,
|
||||
fmt.Sprintf("SELECT count() FROM %s WHERE captured_at < ?", tableName),
|
||||
[]any{cutoff},
|
||||
)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.EligibleCount, nil
|
||||
}
|
||||
|
||||
// DeleteAllNodeRequestReports deletes all node request reports.
|
||||
// DeleteAllNodeRequestReports hard-deletes all node request reports via TRUNCATE.
|
||||
func DeleteAllNodeRequestReports(ctx context.Context) (int64, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
@@ -53,31 +61,39 @@ func DeleteAllNodeRequestReports(ctx context.Context) (int64, error) {
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.EligibleCount, nil
|
||||
return outcome.DeletedCount, nil
|
||||
}
|
||||
|
||||
// DeleteNodeRequestReportsBefore expires request reports ending before cutoff via table TTL.
|
||||
func DeleteNodeRequestReportsBefore(ctx context.Context, cutoff time.Time) (int64, error) {
|
||||
conn, err := observabilityConn()
|
||||
// DeleteNodeRequestReportsBefore force-materializes of_node_request_reports table TTL.
|
||||
// cutoff is ignored; see MaterializeNodeRequestReportsTTL.
|
||||
func DeleteNodeRequestReportsBefore(ctx context.Context, _ time.Time) (int64, error) {
|
||||
outcome, err := MaterializeNodeRequestReportsTTL(ctx)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.EligibleCount, nil
|
||||
}
|
||||
|
||||
// MaterializeNodeRequestReportsTTL force-materializes table TTL and reports an honest outcome.
|
||||
func MaterializeNodeRequestReportsTTL(ctx context.Context) (CleanupOutcome, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return CleanupOutcome{}, err
|
||||
}
|
||||
tableName := nodeRequestReportTableName()
|
||||
cutoff = cutoff.UTC()
|
||||
outcome, err := expireRowsViaTTL(
|
||||
ttlDays := TableTTLDaysNodeRequestReports
|
||||
cutoff := tableTTLCutoff(ttlDays, time.Now())
|
||||
return materializeExpiredByTableTTL(
|
||||
ctx,
|
||||
conn,
|
||||
tableName,
|
||||
ttlDays,
|
||||
fmt.Sprintf("SELECT count() FROM %s WHERE window_ended_at < ?", tableName),
|
||||
[]any{cutoff},
|
||||
)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.EligibleCount, nil
|
||||
}
|
||||
|
||||
// DeleteAllNodeObsOpenresty deletes all OpenResty observations.
|
||||
// DeleteAllNodeObsOpenresty hard-deletes all OpenResty observations via TRUNCATE.
|
||||
func DeleteAllNodeObsOpenresty(ctx context.Context) (int64, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
@@ -87,31 +103,39 @@ func DeleteAllNodeObsOpenresty(ctx context.Context) (int64, error) {
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.EligibleCount, nil
|
||||
return outcome.DeletedCount, nil
|
||||
}
|
||||
|
||||
// DeleteNodeObsOpenrestyBefore expires OpenResty observations captured before cutoff via table TTL.
|
||||
func DeleteNodeObsOpenrestyBefore(ctx context.Context, cutoff time.Time) (int64, error) {
|
||||
conn, err := observabilityConn()
|
||||
// DeleteNodeObsOpenrestyBefore force-materializes of_node_obs_openresty table TTL.
|
||||
// cutoff is ignored; see MaterializeNodeObsOpenrestyTTL.
|
||||
func DeleteNodeObsOpenrestyBefore(ctx context.Context, _ time.Time) (int64, error) {
|
||||
outcome, err := MaterializeNodeObsOpenrestyTTL(ctx)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.EligibleCount, nil
|
||||
}
|
||||
|
||||
// MaterializeNodeObsOpenrestyTTL force-materializes table TTL and reports an honest outcome.
|
||||
func MaterializeNodeObsOpenrestyTTL(ctx context.Context) (CleanupOutcome, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return CleanupOutcome{}, err
|
||||
}
|
||||
tableName := nodeObsOpenrestyTableName()
|
||||
cutoff = cutoff.UTC()
|
||||
outcome, err := expireRowsViaTTL(
|
||||
ttlDays := TableTTLDaysNodeObs
|
||||
cutoff := tableTTLCutoff(ttlDays, time.Now())
|
||||
return materializeExpiredByTableTTL(
|
||||
ctx,
|
||||
conn,
|
||||
tableName,
|
||||
ttlDays,
|
||||
fmt.Sprintf("SELECT count() FROM %s WHERE captured_at < ?", tableName),
|
||||
[]any{cutoff},
|
||||
)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.EligibleCount, nil
|
||||
}
|
||||
|
||||
// DeleteAllNodeObsFrps deletes all FRPS observations.
|
||||
// DeleteAllNodeObsFrps hard-deletes all FRPS observations via TRUNCATE.
|
||||
func DeleteAllNodeObsFrps(ctx context.Context) (int64, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
@@ -121,31 +145,39 @@ func DeleteAllNodeObsFrps(ctx context.Context) (int64, error) {
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.EligibleCount, nil
|
||||
return outcome.DeletedCount, nil
|
||||
}
|
||||
|
||||
// DeleteNodeObsFrpsBefore expires FRPS observations captured before cutoff via table TTL.
|
||||
func DeleteNodeObsFrpsBefore(ctx context.Context, cutoff time.Time) (int64, error) {
|
||||
conn, err := observabilityConn()
|
||||
// DeleteNodeObsFrpsBefore force-materializes of_node_obs_frps table TTL.
|
||||
// cutoff is ignored; see MaterializeNodeObsFrpsTTL.
|
||||
func DeleteNodeObsFrpsBefore(ctx context.Context, _ time.Time) (int64, error) {
|
||||
outcome, err := MaterializeNodeObsFrpsTTL(ctx)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.EligibleCount, nil
|
||||
}
|
||||
|
||||
// MaterializeNodeObsFrpsTTL force-materializes table TTL and reports an honest outcome.
|
||||
func MaterializeNodeObsFrpsTTL(ctx context.Context) (CleanupOutcome, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return CleanupOutcome{}, err
|
||||
}
|
||||
tableName := nodeObsFrpsTableName()
|
||||
cutoff = cutoff.UTC()
|
||||
outcome, err := expireRowsViaTTL(
|
||||
ttlDays := TableTTLDaysNodeObs
|
||||
cutoff := tableTTLCutoff(ttlDays, time.Now())
|
||||
return materializeExpiredByTableTTL(
|
||||
ctx,
|
||||
conn,
|
||||
tableName,
|
||||
ttlDays,
|
||||
fmt.Sprintf("SELECT count() FROM %s WHERE captured_at < ?", tableName),
|
||||
[]any{cutoff},
|
||||
)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.EligibleCount, nil
|
||||
}
|
||||
|
||||
// DeleteAllNodeObsFrpc deletes all FRPC observations.
|
||||
// DeleteAllNodeObsFrpc hard-deletes all FRPC observations via TRUNCATE.
|
||||
func DeleteAllNodeObsFrpc(ctx context.Context) (int64, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
@@ -155,26 +187,34 @@ func DeleteAllNodeObsFrpc(ctx context.Context) (int64, error) {
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.DeletedCount, nil
|
||||
}
|
||||
|
||||
// DeleteNodeObsFrpcBefore force-materializes of_node_obs_frpc table TTL.
|
||||
// cutoff is ignored; see MaterializeNodeObsFrpcTTL.
|
||||
func DeleteNodeObsFrpcBefore(ctx context.Context, _ time.Time) (int64, error) {
|
||||
outcome, err := MaterializeNodeObsFrpcTTL(ctx)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.EligibleCount, nil
|
||||
}
|
||||
|
||||
// DeleteNodeObsFrpcBefore expires FRPC observations captured before cutoff via table TTL.
|
||||
func DeleteNodeObsFrpcBefore(ctx context.Context, cutoff time.Time) (int64, error) {
|
||||
// MaterializeNodeObsFrpcTTL force-materializes table TTL and reports an honest outcome.
|
||||
func MaterializeNodeObsFrpcTTL(ctx context.Context) (CleanupOutcome, error) {
|
||||
conn, err := observabilityConn()
|
||||
if err != nil {
|
||||
return 0, err
|
||||
return CleanupOutcome{}, err
|
||||
}
|
||||
tableName := nodeObsFrpcTableName()
|
||||
cutoff = cutoff.UTC()
|
||||
outcome, err := expireRowsViaTTL(
|
||||
ttlDays := TableTTLDaysNodeObs
|
||||
cutoff := tableTTLCutoff(ttlDays, time.Now())
|
||||
return materializeExpiredByTableTTL(
|
||||
ctx,
|
||||
conn,
|
||||
tableName,
|
||||
ttlDays,
|
||||
fmt.Sprintf("SELECT count() FROM %s WHERE captured_at < ?", tableName),
|
||||
[]any{cutoff},
|
||||
)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return outcome.EligibleCount, nil
|
||||
}
|
||||
}
|
||||
|
||||
@@ -61,3 +61,14 @@ func nodeObsFrpsTableName() string {
|
||||
func nodeObsFrpcTableName() string {
|
||||
return "of_node_obs_frpc"
|
||||
}
|
||||
|
||||
func nodeMetricCapacityHourlyTableName() string {
|
||||
return "of_node_metric_capacity_hourly"
|
||||
}
|
||||
|
||||
func nodeOpenrestyHourlyTableName() string {
|
||||
return "of_node_openresty_hourly"
|
||||
}
|
||||
|
||||
// clickHouseLimit1ByNodeIDClause selects the first row per node_id after ORDER BY.
|
||||
const clickHouseLimit1ByNodeIDClause = " LIMIT 1 BY node_id"
|
||||
|
||||
@@ -0,0 +1,225 @@
|
||||
// Copyright 2026 Arctel.net
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
package analytics
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/ClickHouse/clickhouse-go/v2/lib/driver"
|
||||
"github.com/Rain-kl/Wavelet/internal/db"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func TestListLatestNodeMetricSnapshots_UsesLimit1ByNodeID(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
mock := &mockConn{}
|
||||
db.SetChConnForTest(mock)
|
||||
t.Cleanup(func() { db.SetChConnForTest(nil) })
|
||||
|
||||
since := time.Date(2026, 7, 10, 0, 0, 0, 0, time.UTC)
|
||||
_, err := ListLatestNodeMetricSnapshots(ctx, NodeObservabilityFilter{Since: since})
|
||||
require.NoError(t, err)
|
||||
require.Len(t, mock.queries, 1)
|
||||
assert.Contains(t, mock.queries[0], "LIMIT 1 BY node_id")
|
||||
assert.Contains(t, mock.queries[0], nodeMetricSnapshotTableName())
|
||||
assert.Contains(t, mock.queries[0], "captured_at DESC")
|
||||
assert.NotContains(t, mock.queries[0], "LIMIT ?")
|
||||
require.Len(t, mock.queryArgs, 1)
|
||||
require.Len(t, mock.queryArgs[0], 1)
|
||||
assert.Equal(t, since, mock.queryArgs[0][0])
|
||||
}
|
||||
|
||||
func TestListLatestNodeRequestReports_UsesLimit1ByNodeID(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
mock := &mockConn{}
|
||||
db.SetChConnForTest(mock)
|
||||
t.Cleanup(func() { db.SetChConnForTest(nil) })
|
||||
|
||||
_, err := ListLatestNodeRequestReports(ctx, NodeObservabilityFilter{NodeID: "node-a"})
|
||||
require.NoError(t, err)
|
||||
require.Len(t, mock.queries, 1)
|
||||
assert.Contains(t, mock.queries[0], "LIMIT 1 BY node_id")
|
||||
assert.Contains(t, mock.queries[0], nodeRequestReportTableName())
|
||||
assert.Contains(t, mock.queries[0], "window_ended_at DESC")
|
||||
require.Len(t, mock.queryArgs, 1)
|
||||
require.Len(t, mock.queryArgs[0], 1)
|
||||
assert.Equal(t, "node-a", mock.queryArgs[0][0])
|
||||
}
|
||||
|
||||
func TestListNodeMetricHourly_PrefersRollup(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
hour := time.Date(2026, 7, 10, 12, 0, 0, 0, time.UTC)
|
||||
since := hour.Add(-1 * time.Hour)
|
||||
mock := &mockConn{
|
||||
queryFn: func(_ context.Context, query string, _ ...any) (driver.Rows, error) {
|
||||
if strings.Contains(query, nodeMetricCapacityHourlyTableName()) {
|
||||
return &mockRows{data: [][]any{{
|
||||
hour, 42.5, 60.0, int64(100), int64(200), int64(10), int64(20), uint64(2),
|
||||
}}}, nil
|
||||
}
|
||||
return nil, errors.New("raw path should not be used when rollup covers the window")
|
||||
},
|
||||
}
|
||||
db.SetChConnForTest(mock)
|
||||
t.Cleanup(func() { db.SetChConnForTest(nil) })
|
||||
|
||||
rows, err := ListNodeMetricHourly(ctx, NodeObservabilityFilter{Since: since})
|
||||
require.NoError(t, err)
|
||||
require.Len(t, rows, 1)
|
||||
assert.Equal(t, 42.5, rows[0].AverageCPUUsagePercent)
|
||||
assert.Equal(t, 60.0, rows[0].AverageMemoryUsagePercent)
|
||||
assert.Equal(t, int64(100), rows[0].NetworkRxBytes)
|
||||
assert.Equal(t, 2, rows[0].ReportedNodes)
|
||||
require.Len(t, mock.queries, 1)
|
||||
assert.Contains(t, mock.queries[0], nodeMetricCapacityHourlyTableName())
|
||||
}
|
||||
|
||||
func TestListNodeMetricHourly_MergesRawGapsWithPartialRollup(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
// 24h window starts far before the only rollup bucket (last hour).
|
||||
since := time.Date(2026, 7, 9, 12, 0, 0, 0, time.UTC)
|
||||
rollupHour := time.Date(2026, 7, 10, 12, 0, 0, 0, time.UTC)
|
||||
rawHour := time.Date(2026, 7, 9, 15, 0, 0, 0, time.UTC)
|
||||
mock := &mockConn{
|
||||
queryFn: func(_ context.Context, query string, _ ...any) (driver.Rows, error) {
|
||||
if strings.Contains(query, nodeMetricCapacityHourlyTableName()) {
|
||||
return &mockRows{data: [][]any{{
|
||||
rollupHour, 99.0, 99.0, int64(1), int64(1), int64(1), int64(1), uint64(1),
|
||||
}}}, nil
|
||||
}
|
||||
if strings.Contains(query, nodeMetricSnapshotTableName()) {
|
||||
return &mockRows{data: [][]any{
|
||||
{rawHour, 12.0, 34.0, int64(5), int64(6), int64(7), int64(8), uint64(1)},
|
||||
{rollupHour, 50.0, 50.0, int64(9), int64(9), int64(9), int64(9), uint64(1)},
|
||||
}}, nil
|
||||
}
|
||||
return &mockRows{}, nil
|
||||
},
|
||||
}
|
||||
db.SetChConnForTest(mock)
|
||||
t.Cleanup(func() { db.SetChConnForTest(nil) })
|
||||
|
||||
rows, err := ListNodeMetricHourly(ctx, NodeObservabilityFilter{Since: since})
|
||||
require.NoError(t, err)
|
||||
require.Len(t, rows, 2)
|
||||
assert.Equal(t, rawHour, rows[0].Hour)
|
||||
assert.Equal(t, 12.0, rows[0].AverageCPUUsagePercent)
|
||||
// Overlapping hour prefers rollup (99) over raw (50).
|
||||
assert.Equal(t, rollupHour, rows[1].Hour)
|
||||
assert.Equal(t, 99.0, rows[1].AverageCPUUsagePercent)
|
||||
require.GreaterOrEqual(t, len(mock.queries), 2)
|
||||
assert.Contains(t, mock.queries[1], "lagInFrame")
|
||||
}
|
||||
|
||||
func TestMergeNodeMetricHourlyPreferRollup(t *testing.T) {
|
||||
h1 := time.Date(2026, 7, 10, 10, 0, 0, 0, time.UTC)
|
||||
h2 := time.Date(2026, 7, 10, 11, 0, 0, 0, time.UTC)
|
||||
merged := mergeNodeMetricHourlyPreferRollup(
|
||||
[]NodeMetricHourly{{Hour: h2, AverageCPUUsagePercent: 80}},
|
||||
[]NodeMetricHourly{
|
||||
{Hour: h1, AverageCPUUsagePercent: 10},
|
||||
{Hour: h2, AverageCPUUsagePercent: 20},
|
||||
},
|
||||
)
|
||||
require.Len(t, merged, 2)
|
||||
assert.Equal(t, h1, merged[0].Hour)
|
||||
assert.Equal(t, 10.0, merged[0].AverageCPUUsagePercent)
|
||||
assert.Equal(t, h2, merged[1].Hour)
|
||||
assert.Equal(t, 80.0, merged[1].AverageCPUUsagePercent)
|
||||
}
|
||||
|
||||
func TestHourlyRollupCoversWindow(t *testing.T) {
|
||||
since := time.Date(2026, 7, 10, 0, 0, 0, 0, time.UTC)
|
||||
assert.True(t, hourlyRollupCoversWindow(since, since))
|
||||
assert.True(t, hourlyRollupCoversWindow(since.Add(2*time.Hour), since))
|
||||
assert.False(t, hourlyRollupCoversWindow(since.Add(3*time.Hour), since))
|
||||
assert.True(t, hourlyRollupCoversWindow(time.Date(2026, 7, 11, 0, 0, 0, 0, time.UTC), time.Time{}))
|
||||
}
|
||||
|
||||
func TestListNodeMetricHourly_FallsBackToRawOnRollupError(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
hour := time.Date(2026, 7, 10, 13, 0, 0, 0, time.UTC)
|
||||
mock := &mockConn{
|
||||
queryFn: func(_ context.Context, query string, _ ...any) (driver.Rows, error) {
|
||||
if strings.Contains(query, nodeMetricCapacityHourlyTableName()) {
|
||||
return nil, errors.New("rollup missing")
|
||||
}
|
||||
if strings.Contains(query, nodeMetricSnapshotTableName()) {
|
||||
return &mockRows{data: [][]any{{
|
||||
hour, 10.0, 20.0, int64(1), int64(2), int64(3), int64(4), uint64(1),
|
||||
}}}, nil
|
||||
}
|
||||
return &mockRows{}, nil
|
||||
},
|
||||
}
|
||||
db.SetChConnForTest(mock)
|
||||
t.Cleanup(func() { db.SetChConnForTest(nil) })
|
||||
|
||||
rows, err := ListNodeMetricHourly(ctx, NodeObservabilityFilter{})
|
||||
require.NoError(t, err)
|
||||
require.Len(t, rows, 1)
|
||||
assert.Equal(t, 10.0, rows[0].AverageCPUUsagePercent)
|
||||
assert.Equal(t, int64(3), rows[0].DiskReadBytes)
|
||||
require.GreaterOrEqual(t, len(mock.queries), 2)
|
||||
assert.Contains(t, mock.queries[0], nodeMetricCapacityHourlyTableName())
|
||||
assert.Contains(t, mock.queries[1], "lagInFrame")
|
||||
}
|
||||
|
||||
func TestListNodeOpenrestyHourly_PrefersRollup(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
hour := time.Date(2026, 7, 10, 14, 0, 0, 0, time.UTC)
|
||||
since := hour.Add(-1 * time.Hour)
|
||||
mock := &mockConn{
|
||||
queryFn: func(_ context.Context, query string, _ ...any) (driver.Rows, error) {
|
||||
if strings.Contains(query, nodeOpenrestyHourlyTableName()) {
|
||||
return &mockRows{data: [][]any{{
|
||||
hour, int64(50), int64(70), uint64(3),
|
||||
}}}, nil
|
||||
}
|
||||
return nil, errors.New("raw path should not be used when rollup covers the window")
|
||||
},
|
||||
}
|
||||
db.SetChConnForTest(mock)
|
||||
t.Cleanup(func() { db.SetChConnForTest(nil) })
|
||||
|
||||
rows, err := ListNodeOpenrestyHourly(ctx, NodeObservabilityFilter{Since: since})
|
||||
require.NoError(t, err)
|
||||
require.Len(t, rows, 1)
|
||||
assert.Equal(t, int64(50), rows[0].OpenrestyRxBytes)
|
||||
assert.Equal(t, int64(70), rows[0].OpenrestyTxBytes)
|
||||
assert.Equal(t, 3, rows[0].ReportedNodes)
|
||||
}
|
||||
|
||||
func TestListNodeOpenrestyHourly_FallsBackToRawOnEmptyRollup(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
hour := time.Date(2026, 7, 10, 15, 0, 0, 0, time.UTC)
|
||||
mock := &mockConn{
|
||||
queryFn: func(_ context.Context, query string, _ ...any) (driver.Rows, error) {
|
||||
if strings.Contains(query, nodeOpenrestyHourlyTableName()) {
|
||||
return &mockRows{}, nil
|
||||
}
|
||||
if strings.Contains(query, nodeObsOpenrestyTableName()) {
|
||||
return &mockRows{data: [][]any{{
|
||||
hour, int64(9), int64(8), uint64(1),
|
||||
}}}, nil
|
||||
}
|
||||
return &mockRows{}, nil
|
||||
},
|
||||
}
|
||||
db.SetChConnForTest(mock)
|
||||
t.Cleanup(func() { db.SetChConnForTest(nil) })
|
||||
|
||||
rows, err := ListNodeOpenrestyHourly(ctx, NodeObservabilityFilter{})
|
||||
require.NoError(t, err)
|
||||
require.Len(t, rows, 1)
|
||||
assert.Equal(t, int64(9), rows[0].OpenrestyRxBytes)
|
||||
require.GreaterOrEqual(t, len(mock.queries), 2)
|
||||
assert.Contains(t, mock.queries[1], "lagInFrame")
|
||||
}
|
||||
|
||||
@@ -142,7 +142,7 @@ func getSeedConfigsPart1() []model.SystemConfig {
|
||||
},
|
||||
{
|
||||
Key: model.ConfigKeyCapLoginEnabled,
|
||||
Value: configValueTrue,
|
||||
Value: configValueFalse,
|
||||
Type: configTypeSystem,
|
||||
Description: "是否启用登录人机验证(true/false)",
|
||||
},
|
||||
|
||||
@@ -0,0 +1,92 @@
|
||||
// Package main is a manual smoke tool for ClickHouse app write path.
|
||||
// Usage (from repo root, with config.yaml and Docker CH up):
|
||||
//
|
||||
// go run ./scripts/live_ch_smoke
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"os"
|
||||
"time"
|
||||
|
||||
"github.com/Rain-kl/Wavelet/internal/apps/openflare/chwriter"
|
||||
"github.com/Rain-kl/Wavelet/internal/db"
|
||||
"github.com/Rain-kl/Wavelet/internal/model"
|
||||
)
|
||||
|
||||
const (
|
||||
flushWaitTimeout = 45 * time.Second
|
||||
listLimit = 5
|
||||
pollInterval = 2 * time.Second
|
||||
)
|
||||
|
||||
func main() {
|
||||
if err := run(); err != nil {
|
||||
fmt.Fprintln(os.Stderr, err)
|
||||
os.Exit(1)
|
||||
}
|
||||
}
|
||||
|
||||
func run() error {
|
||||
if !db.ChConnReady() {
|
||||
return fmt.Errorf("ChConn not ready — check config.yaml clickhouse.enabled")
|
||||
}
|
||||
ctx := context.Background()
|
||||
chwriter.Init(ctx)
|
||||
defer func() { _ = chwriter.Stop(ctx) }()
|
||||
|
||||
now := time.Now().UTC()
|
||||
nodeID := "e2e-app-" + now.Format("150405")
|
||||
if err := model.InsertOpenFlareMetricSnapshot(ctx, &model.OpenFlareMetricSnapshot{
|
||||
NodeID: nodeID, CapturedAt: now, CPUUsagePercent: 41.2,
|
||||
MemoryUsedBytes: 123, MemoryTotalBytes: 1000,
|
||||
StorageUsedBytes: 456, StorageTotalBytes: 2000,
|
||||
DiskReadBytes: 11, DiskWriteBytes: 22, NetworkRxBytes: 33, NetworkTxBytes: 44,
|
||||
}); err != nil {
|
||||
return fmt.Errorf("insert: %w", err)
|
||||
}
|
||||
fmt.Println("queued", nodeID)
|
||||
|
||||
if err := waitForSnapshot(ctx, nodeID, now); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := assertLatestIncludes(ctx, nodeID, now); err != nil {
|
||||
return err
|
||||
}
|
||||
for _, s := range chwriter.WriterStats() {
|
||||
fmt.Printf("writer %s running=%v depth=%d drops=%d flush_err=%d\n",
|
||||
s.Name, s.Running, s.Depth, s.Drops, s.FlushErrors)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func waitForSnapshot(ctx context.Context, nodeID string, now time.Time) error {
|
||||
deadline := time.Now().Add(flushWaitTimeout)
|
||||
for time.Now().Before(deadline) {
|
||||
rows, err := model.ListOpenFlareMetricSnapshotsSince(ctx, nodeID, now.Add(-time.Minute), listLimit)
|
||||
if err != nil {
|
||||
return fmt.Errorf("list: %w", err)
|
||||
}
|
||||
if len(rows) > 0 {
|
||||
fmt.Printf("OK flushed id=%d cpu=%.1f\n", rows[0].ID, rows[0].CPUUsagePercent)
|
||||
return nil
|
||||
}
|
||||
time.Sleep(pollInterval)
|
||||
}
|
||||
return fmt.Errorf("not flushed within timeout")
|
||||
}
|
||||
|
||||
func assertLatestIncludes(ctx context.Context, nodeID string, now time.Time) error {
|
||||
latest, err := model.ListOpenFlareLatestMetricSnapshotsSince(ctx, "", now.Add(-time.Hour))
|
||||
if err != nil {
|
||||
return fmt.Errorf("latest: %w", err)
|
||||
}
|
||||
for _, r := range latest {
|
||||
if r != nil && r.NodeID == nodeID {
|
||||
fmt.Println("OK latest-per-node includes node")
|
||||
return nil
|
||||
}
|
||||
}
|
||||
return fmt.Errorf("latest-per-node missing node %s", nodeID)
|
||||
}
|
||||
Reference in New Issue
Block a user