diff --git a/docs/config.ts b/docs/config.ts index cab01868..be7880ed 100644 --- a/docs/config.ts +++ b/docs/config.ts @@ -132,6 +132,9 @@ function sidebarDesign(): DefaultTheme.SidebarItem[] { { text: 'WAF 设计', link: 'waf-design' }, { text: 'WAF 可编排规则设计', link: 'waf-orchestration-design' }, { text: 'Pages 静态托管设计', link: 'pages-design' }, + { text: '边缘可观测与业务流量统计', link: 'observability-design' }, + { text: '观测数据传输模型', link: 'observability-transport-model' }, + { text: '观测上报协议与表结构', link: 'observability-data-model' }, { text: 'Uptime Kuma 监控同步设计', link: 'kuma-design' }, { text: '登录验证码设计', link: 'login-captcha' } ] diff --git a/docs/deployment/agent.md b/docs/deployment/agent.md index 370465e1..fff193e7 100644 --- a/docs/deployment/agent.md +++ b/docs/deployment/agent.md @@ -90,8 +90,8 @@ curl -fsSL https://raw.githubusercontent.com/Rain-kl/OpenFlare/main/scripts/inst "data_dir": "./data", "openresty_path": "openresty", "openresty_observability_port": 18081, - "observability_replay_minutes": 15, - "heartbeat_interval": 10000, + "observability_replay_minutes": 60, + "heartbeat_interval": 3000, "request_timeout": 10000 } ``` @@ -110,7 +110,7 @@ curl -fsSL https://raw.githubusercontent.com/Rain-kl/OpenFlare/main/scripts/inst "cert_dir": "/var/lib/openflare-agent/etc/nginx/certs", "lua_dir": "/var/lib/openflare-agent/etc/nginx/lua", "runtime_config_dir": "/var/lib/openflare-agent/etc/openflare", - "heartbeat_interval": 10000, + "heartbeat_interval": 3000, "request_timeout": 10000 } ``` diff --git a/docs/deployment/deployment.md b/docs/deployment/deployment.md index 5b3b5f9b..fcec892f 100644 --- a/docs/deployment/deployment.md +++ b/docs/deployment/deployment.md @@ -194,7 +194,7 @@ export LOG_LEVEL='info' "agent_token": "replace-with-node-auth-token", "data_dir": "./data", "openresty_path": "openresty", - "heartbeat_interval": 10000, + "heartbeat_interval": 3000, "request_timeout": 10000 } ``` diff --git a/docs/design/agent-design.md b/docs/design/agent-design.md index ddf250f8..4bb2f665 100644 --- a/docs/design/agent-design.md +++ b/docs/design/agent-design.md @@ -27,7 +27,7 @@ Agent 主要由以下核心子模块组成,共同配合完成其完整的生 | **OpenResty 管控** | `nginx/` | 执行 Nginx 配置校验 (`openresty -t`)、重写、平滑重载 (`reload`) 及进程自启动。 | | **本地状态库** | `state/` | 持久化记录本地应用版本、错误日志及未成功上报的可观测性指标缓冲。 | | **自更新服务** | `updater/` | 监听 Server 自更新指令,安全拉取新版本二进制并完成原地热升级。 | -| **可观测性** | `observability/` | 采集系统宿主机 CPU/内存/磁盘及 Nginx 性能指标,处理访问日志并上报。 | +| **可观测性** | `observability/` | 采集宿主机资源读数、OpenResty 健康/连接,并 tail 访问日志明细上报;**不做** UV/TopN/吞吐等业务预聚合。详见 [边缘可观测与业务流量统计](./observability-design.md)。 | | **GeoIP 维护** | `geoipdata/` `geoipupdate/` | 维护并定期更新本地 GeoIP 数据库,为 WAF 地域过滤提供支撑。 | --- @@ -173,3 +173,4 @@ graph TD 1. **零特权指令通道**:Server 绝对禁止向 Agent 传递任何任意 shell 命令或远程执行脚本(如 exec/eval 等)。所有系统控制原语(如启动、停止、重载、更新)必须硬编码在 Agent 二进制内部。 2. **严格的 Token 过滤与前缀验证**:Agent 侧向 Server 请求资源时,接口端点固定以 `/api/v1/agent/` 为前缀,并强制携带 `X-Agent-Token` 进行签名或令牌核验。 3. **节点自治原则**:Agent 须具备完备的离线工作能力。在与 Server 失去连接期间,本地 OpenResty 必须依靠本地已落地的配置保持反向代理服务的绝对正常运行。 +4. **观测只上报事实**:访问日志以明细形式上送;主机指标上报计数器/瞬时读数。禁止在 Agent 内计算业务 UV、Top 域名、24h 已提供数据等结论性指标(由 Server 聚合)。详见 [边缘可观测与业务流量统计](./observability-design.md)。 diff --git a/docs/design/architecture.md b/docs/design/architecture.md index a04bfceb..477a21f5 100644 --- a/docs/design/architecture.md +++ b/docs/design/architecture.md @@ -65,8 +65,8 @@ OpenResty (Agent, TLS/WAF) | 组件 | 职责 | 详细设计参考 | | --------------- | ---------------------------------------------------------------------- | ------------ | -| **Server** | 管理端 UI/API、控制面状态持久化、配置编译渲染、发布版本控制、Pages 部署包存储、Uptime Kuma 监控同步与登录验证码防护 | [Agent 与发布模型](./agent-design.md) / [Uptime Kuma 监控同步设计](./kuma-design.md) / [登录验证码设计](./login-captcha.md) | -| **Agent** | 周期心跳与 WS 同步、静态资源包拉取与解压、OpenResty 配置写入/校验/重载与自愈 | [Agent 与发布模型](./agent-design.md) | +| **Server** | 管理端 UI/API、控制面状态持久化、配置编译渲染、发布版本控制、Pages 部署包存储、访问日志入库与业务流量聚合、Uptime Kuma 监控同步与登录验证码防护 | [Agent 与发布模型](./agent-design.md) / [边缘可观测与业务流量统计](./observability-design.md) / [Uptime Kuma 监控同步设计](./kuma-design.md) / [登录验证码设计](./login-captcha.md) | +| **Agent** | 周期心跳与 WS 同步、静态资源包拉取与解压、OpenResty 配置写入/校验/重载与自愈;观测仅上报访问明细与主机/健康读数,不做业务预聚合 | [Agent 与发布模型](./agent-design.md) / [边缘可观测与业务流量统计](./observability-design.md) | | **OpenResty** | 接收真实流量,执行 WAF 过滤、PoW 防护、Basic Auth 认证与静态/反代服务 | [WAF 设计](./waf-design.md) / [Pages 设计](./pages-design.md) | | **Relay** | 部署于边缘节点,管理 `frps` 守护进程生命周期,接受心跳派发的穿透中继配置 | [内网穿透设计](./tunnel-design.md) | | **OpenFlared** | 部署于内网,管理 `frpc` 进程组,向多个 Relay 建立反向隧道,上报连接状态 | [内网穿透设计](./tunnel-design.md) | @@ -135,6 +135,24 @@ OpenResty (Agent, TLS/WAF) * IP 组成员独立热更新:协调 Worker 每 5 秒检查一次 checksum,仅在变化时加载完整快照,各 Worker 的请求路径始终读取本地内存对象。 * *IP 组来源与同步机制详见:[WAF 设计文档](./waf-design.md);图模型、执行语义与发布约束详见:[WAF 可编排规则设计](./waf-orchestration-design.md)。* +### 4. 边缘可观测与业务流量统计流 +```text +OpenResty access.log(业务事实) + | + | Agent tail 增量明细(不 sum/count/uniq) + v +Server 入库 ClickHouse + | + +---> 全局聚合 --> 看板「已提供数据 / 请求 / UV」 + +---> host∈Zone --> Zone「已提供数据」等(同一套语义) + +---> node_id 过滤 --> 节点业务量 + +主机 /proc 网卡与 CPU 等 --> Agent 读数快照 --> 宿主机资源趋势(与业务交付分开展示) +OpenResty 健康与连接数 --> 边缘健康(瞬时,不作 24h 业务总量) +``` +* **原则**:Agent 只上报事实,Server 解释事实;业务流量唯一真相为访问日志。`openresty_tx` 与「已提供数据」不得双轨并存。 +* *传输模型、示例与采集频率详见:[观测数据传输模型](./observability-transport-model.md);字段收敛与迁移详见:[边缘可观测与业务流量统计](./observability-design.md)* + --- ## 核心对象 @@ -159,6 +177,8 @@ OpenResty (Agent, TLS/WAF) | Zone 域名与路由策略分离 | Zone 提供根域入口与域名边界;路由仍可复用同一套站点级策略并按域名绑定证书 | | 内网穿透基于 frp 整合 | 复用成熟隧道协议,避免自研隧道引起稳定性风险;其 Vhost 机制天然适配反代路由 | | 运行时配置与控制库解耦 | WAF 规则发布时编译并随 OpenResty reload 加载;动态 IP 组通过 checksum 驱动的内存快照独立刷新 | +| 业务流量以访问日志为唯一真相 | Agent 禁止业务预聚合;看板与 Zone 共用 Server 侧聚合,避免 openresty_tx 与 bytes_sent 双轨 | +| 业务交付 / 边缘健康 / 主机资源分层 | 已提供数据≠宿主机网卡出站≠OpenResty 连接数,UI 与 API 分名分区 | --- @@ -174,4 +194,5 @@ OpenResty (Agent, TLS/WAF) * WAF 相关开发:阅读 [WAF 设计](./waf-design.md) 与 [WAF 可编排规则设计](./waf-orchestration-design.md)。 * Pages 托管开发:阅读 [Pages 静态托管设计](./pages-design.md)。 * 监控同步开发:阅读 [Uptime Kuma 监控同步设计](./kuma-design.md)。 + * 看板/访问日志/节点指标开发:阅读 [观测数据传输模型](./observability-transport-model.md) 与 [边缘可观测与业务流量统计](./observability-design.md)。 5. **[仓库结构](./index.md#仓库结构)**:明确各个物理目录分层职责,避免堆砌和重复开发。 diff --git a/docs/design/index.md b/docs/design/index.md index 9b429870..7d4d4bc6 100644 --- a/docs/design/index.md +++ b/docs/design/index.md @@ -28,7 +28,7 @@ OpenFlare 适合需要统一管理多台 OpenResty 代理节点的团队,具 | **内网穿透** | 通过中继节点(Relay)与内网客户端(OpenFlared),反向穿透暴露内网 Web 服务 | [内网穿透设计](./tunnel-design.md) / [穿透使用指南](../guide/tunnel-usage.md) | | **Pages 静态托管** | 直接上传前端压缩包(zip / tar.gz / tar.xz / 7z 等),由边缘节点拉取并由 OpenResty 本地服务,支持 API 反代与 SPA Fallback | [Pages 静态托管设计](./pages-design.md) | | **TLS 证书自动续期** | 将证书显式绑定到 Zone 域名,并通过 ACME 协议向 Let's Encrypt 申请/续期证书 | [Zone 与域名资源设计](./zone-design.md) | -| **多节点监控与观测** | 收集节点资源快照、健康事件,聚合请求指标与访问日志明细 | [系统架构](./architecture.md) | +| **多节点监控与观测** | 访问日志为业务流量唯一真相;Agent 只上报明细与主机读数,Server 统一聚合;与 Zone/看板对账 | [观测数据传输模型](./observability-transport-model.md) / [边缘可观测与业务流量统计](./observability-design.md) / [上报协议与表结构](./observability-data-model.md) / [系统架构](./architecture.md) | --- diff --git a/docs/design/observability-data-model.md b/docs/design/observability-data-model.md new file mode 100644 index 00000000..d97dc0b9 --- /dev/null +++ b/docs/design/observability-data-model.md @@ -0,0 +1,723 @@ +# Agent 上报协议与观测落库数据模型 + +你会学到:重构后 Agent 心跳/WS 上报的 **数据结构**、Server **如何解析与写入**、ClickHouse / 关系库 **目标表结构**,以及与旧字段/旧表的兼容关系。 + +本设计是 [边缘可观测与业务流量统计重构](./observability-design.md) 的 **协议与存储专章**,实现时以本文字段与 DDL 为准。 + +**先读传输全景与示例:** [观测数据传输模型](./observability-transport-model.md)。 + +--- + +## 1. 设计目标 + +| 目标 | 说明 | +| --- | --- | +| Agent 只报事实 | 明细 + 主机读数 + 边缘健康瞬时态;无业务预聚合 | +| 一张业务明细表 | 访问日志是 L1 唯一写入路径 | +| 聚合在库内/控制面 | 小时汇总由 ClickHouse MV 或查询生成,Agent 不写汇总表 | +| 字段不重叠 | `bytes_sent` = 已提供数据;网卡 `network_*` = 宿主机;不再有业务 `openresty_tx` | +| 可演进 | 新字段可选;旧 Agent 缺字段时 Server 填默认值 | + +--- + +## 2. 分层与写入总览 + +```text + Agent NodePayload (v2) + │ + ┌───────────────┼───────────────┐ + ▼ ▼ ▼ + access_logs host_metrics edge_health + (L1 明细) (L3 读数) (L2 瞬时) + │ │ │ + ▼ ▼ ▼ + of_node_access_logs of_node_metric_ of_node_edge_health + │ snapshots │ + │ │ │ + ▼ ▼ │ + of_access_log_hourly of_node_metric_ │ + (MV, Server 侧) capacity_hourly (MV) │ + │ │ │ + └─────── 管理端聚合 API ───────────┘ + +关系库 (PostgreSQL/SQLite):节点最新状态、Profile、健康事件(非明细湖) +``` + +| 层 | 含义 | Agent 上报块 | ClickHouse 事实表 | +| --- | --- | --- | --- | +| L1 | 业务交付 | `access_logs` | `of_node_access_logs` | +| L2 | 边缘健康 | `edge_health` | `of_node_edge_health` | +| L3 | 宿主机资源 | `host_metrics` | `of_node_metric_snapshots` | + +--- + +## 3. Agent 上报数据结构(协议 v2) + +### 3.1 顶层 `NodePayload` + +传输:HTTP 心跳 body 与 WebSocket `status` 消息共用同一结构。 + +```json +{ + "schema_version": 2, + "node_id": "n_xxx", + "name": "edge-1", + "ip": "1.2.3.4", + "version": "3.3.0", + "ext_version": "", + "current_version": "cfg-checksum-or-version", + "last_error": "", + "profile": { }, + "host_metrics": { }, + "edge_health": { }, + "access_logs": [ ], + "buffered": [ ], + "health_events": [ ], + "waf_ip_group_checksums": { "1": "md5..." } +} +``` + +| 字段 | 类型 | 必填 | 说明 | +| --- | --- | --- | --- | +| `schema_version` | int | 建议 | `2` = 本设计;缺省或 `0/1` 按旧协议兼容解析 | +| `node_id` | string | ✅ | 节点 ID | +| `name` | string | ✅ | 显示名 | +| `ip` | string | ✅ | 上报 IP | +| `version` / `ext_version` | string | ✅ | Agent 版本 | +| `current_version` | string | | 本地激活配置版本摘要 | +| `last_error` | string | | 最近同步/运行错误,可空 | +| `profile` | object | | 主机概况,变化时上报(可节流) | +| `host_metrics` | object | 建议每拍 | L3 资源快照 | +| `edge_health` | object | 建议每拍 | L2 OpenResty 健康 | +| `access_logs` | array | | 本拍增量访问明细 | +| `buffered` | array | | 离线补传的事实批次(见 §3.6) | +| `health_events` | array | | 边缘健康事件 | +| `waf_ip_group_checksums` | map | | 差分同步用,非观测湖 | + +**协议 v2 删除(不再作为权威,兼容期可忽略):** + +| 旧字段 | 处置 | +| --- | --- | +| `traffic_report` | 忽略,不落业务表 | +| `openresty_observation.openresty_rx_bytes` / `tx` | 忽略 | +| `openresty_status` / `openresty_message`(顶层) | 迁入 `edge_health`;兼容期从旧字段回填 | +| `snapshot` | 重命名为 `host_metrics`;兼容期别名读取 | +| `buffered_observability` | 重命名为 `buffered`;结构见 §3.6 | + +### 3.2 `profile` — 主机概况(低频) + +对应关系库 `of_node_system_profiles`(或现有等价表),**不进 ClickHouse 明细湖**。 + +```json +{ + "hostname": "edge-1", + "os_name": "linux", + "os_version": "...", + "kernel_version": "...", + "architecture": "amd64", + "cpu_model": "...", + "cpu_cores": 8, + "total_memory_bytes": 16106127360, + "total_disk_bytes": 107374182400, + "uptime_seconds": 864000, + "reported_at_unix": 1720000000 +} +``` + +| 字段 | 语义 | +| --- | --- | +| 硬件/OS 描述字段 | 事实读数 | +| `reported_at_unix` | Agent 采集时刻(UTC 秒) | + +### 3.3 `host_metrics` — 宿主机资源(L3) + +**全部为读数,不做 24h 业务总量。** +网卡/磁盘字节为 **内核累计计数器原值**(单调递增,重启可归零);CPU 为瞬时百分比;内存/磁盘占用为当前用量。 + +```json +{ + "captured_at_unix": 1720000000, + "cpu_usage_percent": 12.5, + "memory_used_bytes": 4294967296, + "memory_total_bytes": 16106127360, + "storage_used_bytes": 50000000000, + "storage_total_bytes": 107374182400, + "disk_read_bytes": 9000000000, + "disk_write_bytes": 12000000000, + "network_rx_bytes": 500000000000, + "network_tx_bytes": 800000000000 +} +``` + +| 字段 | 类型 | 语义 | Server 如何用 | +| --- | --- | --- | --- | +| `captured_at_unix` | int64 | 采样时刻 | `captured_at` | +| `cpu_usage_percent` | float | 瞬时 CPU% | 直接存;趋势取平均 | +| `memory_*` / `storage_*` | int64 | 当前用量/总量 | 直接存;算占用率 | +| `disk_read_bytes` / `disk_write_bytes` | int64 | **累计** IO 字节 | 存原值;查询时相邻差分 | +| `network_rx_bytes` / `network_tx_bytes` | int64 | **累计** 网卡字节 | 存原值;查询时相邻差分 →「宿主机网卡入/出站」 | + +> Agent **禁止** 在上报前对网卡/磁盘做「本周期增量」替换累计值(否则 Server 差分会错)。 + +### 3.4 `edge_health` — OpenResty 边缘健康(L2) + +**仅瞬时态,不包含业务吞吐。** + +```json +{ + "captured_at_unix": 1720000000, + "status": "healthy", + "message": "", + "connections": 42 +} +``` + +| 字段 | 类型 | 语义 | +| --- | --- | --- | +| `status` | string | `healthy` / `unhealthy` / `unknown` | +| `message` | string | 状态说明 | +| `connections` | int64 | stub_status Active connections | + +`status`/`message` 同步更新关系库节点最新状态;`connections` 写入 CH `of_node_edge_health` 供节点详情曲线(可选)。 + +### 3.5 `access_logs[]` — 访问明细(L1,业务唯一事实) + +Agent:tail access.log → 解析 JSON 行 → 原样字段上报(可截断 path)。 + +```json +{ + "logged_at_unix": 1720000001, + "remote_addr": "203.0.113.10", + "host": "www.example.com", + "path": "/api/v1/ping", + "status_code": 200, + "bytes_sent": 1024, + "request_length": 128, + "request_time_ms": 15 +} +``` + +| 字段 | 类型 | 必填 | 来源(OpenResty) | 业务含义 | +| --- | --- | --- | --- | --- | +| `logged_at_unix` | int64 | ✅ | `$time_iso8601` 解析 | 请求完成时间 | +| `remote_addr` | string | ✅ | `$remote_addr` | 客户端 IP → UV | +| `host` | string | ✅ | `$host` | 域名 → Zone 归属 | +| `path` | string | ✅ | `$request_uri`,Agent 可截断 | 路径 | +| `status_code` | int | ✅ | `$status` | 状态码 | +| `bytes_sent` | int64 | ✅ | **`$body_bytes_sent`** | **已提供数据**(响应体) | +| `request_length` | int64 | 建议 | `$request_length` | **接收数据**;旧 Agent 缺省 0 | +| `request_time_ms` | int64 | 可选 | `$request_time * 1000` | 耗时;缺省 0 | + +**明确不由 Agent 上报(由 Server 写入):** + +* `region` / 国家:入库时 GeoIP 解析 +* `id` / `created_at`:Server 生成 +* `node_id`:取自 payload / 鉴权上下文 + +**单次心跳条数建议:** + +* 软上限例如 2000 条/拍;超出进入 `buffered` 下一批,**禁止** 在 Agent 压成 TrafficReport。 + +### 3.6 `buffered[]` — 离线补传(只装事实) + +```json +{ + "captured_at_unix": 1719999900, + "host_metrics": { }, + "edge_health": { }, + "access_logs": [ ] +} +``` + +| 字段 | 说明 | +| --- | --- | +| `captured_at_unix` | 该批次采集/缓冲时刻,用于 ack 与去重窗口 | +| `host_metrics` / `edge_health` / `access_logs` | 与主 payload 同结构;可省略空块 | + +**禁止** 在 buffered 中携带 `traffic_report` 或 rx/tx 吞吐。 + +### 3.7 `health_events[]` + +```json +{ + "event_type": "openresty_unhealthy", + "severity": "critical", + "message": "...", + "triggered_at_unix": 1720000000, + "metadata": { } +} +``` + +写入关系库健康事件表(现有模型即可),不进访问日志湖。 + +### 3.8 Go 协议草图(目标) + +```go +// pkg/protocol/agent.go(目标形态,实现时替换旧类型) + +type NodePayload struct { + SchemaVersion int `json:"schema_version,omitempty"` + NodeID string `json:"node_id"` + Name string `json:"name"` + IP string `json:"ip"` + Version string `json:"version"` + ExtVersion string `json:"ext_version"` + CurrentVersion string `json:"current_version"` + LastError string `json:"last_error"` + Profile *NodeSystemProfile `json:"profile,omitempty"` + HostMetrics *NodeHostMetrics `json:"host_metrics,omitempty"` + EdgeHealth *NodeEdgeHealth `json:"edge_health,omitempty"` + AccessLogs []NodeAccessLog `json:"access_logs,omitempty"` + Buffered []BufferedFacts `json:"buffered,omitempty"` + HealthEvents []NodeHealthEvent `json:"health_events"` + WAFIPGroupChecksums map[string]string `json:"waf_ip_group_checksums,omitempty"` + + // Deprecated: schema_version < 2 兼容 + Snapshot *NodeHostMetrics `json:"snapshot,omitempty"` + OpenrestyStatus string `json:"openresty_status,omitempty"` + OpenrestyMessage string `json:"openresty_message,omitempty"` + OpenrestyObservation json.RawMessage `json:"openresty_observation,omitempty"` // 仅解析 connections + TrafficReport json.RawMessage `json:"traffic_report,omitempty"` // 忽略 + BufferedObservability []BufferedFacts `json:"buffered_observability,omitempty"` +} + +type NodeHostMetrics struct { + CapturedAtUnix int64 `json:"captured_at_unix"` + CPUUsagePercent float64 `json:"cpu_usage_percent"` + MemoryUsedBytes int64 `json:"memory_used_bytes"` + MemoryTotalBytes int64 `json:"memory_total_bytes"` + StorageUsedBytes int64 `json:"storage_used_bytes"` + StorageTotalBytes int64 `json:"storage_total_bytes"` + DiskReadBytes int64 `json:"disk_read_bytes"` + DiskWriteBytes int64 `json:"disk_write_bytes"` + NetworkRxBytes int64 `json:"network_rx_bytes"` + NetworkTxBytes int64 `json:"network_tx_bytes"` +} + +type NodeEdgeHealth struct { + CapturedAtUnix int64 `json:"captured_at_unix"` + Status string `json:"status"` + Message string `json:"message"` + Connections int64 `json:"connections"` +} + +type NodeAccessLog struct { + LoggedAtUnix int64 `json:"logged_at_unix"` + RemoteAddr string `json:"remote_addr"` + Host string `json:"host"` + Path string `json:"path"` + StatusCode int `json:"status_code"` + BytesSent int64 `json:"bytes_sent"` // body_bytes_sent,已提供数据 + RequestLength int64 `json:"request_length"` // 接收数据 + RequestTimeMs int64 `json:"request_time_ms"` // 可选 +} + +type BufferedFacts struct { + CapturedAtUnix int64 `json:"captured_at_unix"` + HostMetrics *NodeHostMetrics `json:"host_metrics,omitempty"` + EdgeHealth *NodeEdgeHealth `json:"edge_health,omitempty"` + AccessLogs []NodeAccessLog `json:"access_logs,omitempty"` +} +``` + +--- + +## 4. Server 解析与落库流程 + +### 4.1 入口 + +* HTTP:`POST /api/v1/agent/...` 心跳(现有路径) +* WebSocket:`type=status` payload = `NodePayload` +* 鉴权:`X-Agent-Token` → 绑定 `node_id`(payload.node_id 必须与 token 节点一致) + +### 4.2 处理流水线(单次 payload) + +```text +1. 反序列化 NodePayload +2. 归一化(normalize) + - schema_version < 2: + host_metrics ← snapshot + edge_health.status ← openresty_status + edge_health.connections ← openresty_observation.connections(若有) + traffic_report → drop + openresty_observation.rx/tx → drop + buffered ← buffered_observability + - path 再截断、status 范围钳制、负数字节 → 0 +3. 关系库事务(节点最新态) + - 更新 node 在线时间、IP、版本、edge_health.status/message + - upsert profile(若有) + - insert health_events(若有) +4. ClickHouse 异步 batch(失败记日志,不阻断心跳响应的配置下发) + a. access_logs + buffered[].access_logs + → 补 region(GeoIP) + → 分配 snowflake id + → BatchInsert of_node_access_logs + b. host_metrics + buffered[].host_metrics + → of_node_metric_snapshots + c. edge_health + buffered[].edge_health + → of_node_edge_health(仅 connections + status 快照可选) +5. 返回心跳响应(settings / active_config / waf 差分) +6. 若使用 buffer ack:按 buffered.captured_at_unix 列表确认 +``` + +### 4.3 归一化规则(硬约束) + +| 规则 | 行为 | +| --- | --- | +| `logged_at` 超前 now+5m | 钳制为 now 或丢弃该条(实现选定一种并单测) | +| `logged_at` 早于 now−TTL | 仍可写入,依赖表 TTL 清理 | +| 空 `host` | 允许,聚合进「未归属」 | +| `bytes_sent` / `request_length` < 0 | 置 0 | +| 单批 access_logs > N | 截断并打点监控(或只入 buffer 队列),不改为预聚合 | +| 重复补传 | CH 允许少量重复行;查询用 sum 近似(不强制精确去重) | + +### 4.4 字段映射表(上报 → 表) + +| 上报路径 | 目标存储 | 列 | +| --- | --- | --- | +| `access_logs[]` | CH `of_node_access_logs` | 见 §5.1 | +| `host_metrics` | CH `of_node_metric_snapshots` | 见 §5.2 | +| `edge_health` | CH `of_node_edge_health` + PG node 最新状态 | 见 §5.3 / §5.6 | +| `profile` | PG `of_node_system_profiles` | 现有列 | +| `health_events` | PG 健康事件表 | 现有模型 | +| `waf_ip_group_checksums` | 不落观测表 | 同步逻辑 | +| `traffic_report`(旧) | **不写** | — | +| `openresty_rx/tx`(旧) | **不写** | — | + +### 4.5 查询侧(不落新「业务出站」列) + +| 产品指标 | SQL 语义(示意) | +| --- | --- | +| 已提供数据 | `sum(bytes_sent)` | +| 接收数据 | `sum(request_length)` | +| 请求数 | `count()` | +| UV | `uniqExact(remote_addr)` | +| 5xx | `countIf(status_code >= 500)` | +| 按域名/状态码/地区 | `GROUP BY host / status_code / region` | +| 宿主机网卡出站 | 对 `network_tx_bytes` 按 node 时间序非负差分后 sum | +| OpenResty 连接 | `of_node_edge_health.connections` 最新或平均 | + +--- + +## 5. 表结构(目标 DDL) + +> 引擎与 TTL 与现网一致倾向:访问日志 90 天,指标 30 天。 +> `id` 使用控制面 Snowflake/唯一 UInt64。 + +### 5.1 L1 事实表:`of_node_access_logs` + +```sql +CREATE TABLE IF NOT EXISTS of_node_access_logs +( + id UInt64, + node_id String, + logged_at DateTime64(3, 'UTC'), + remote_addr String, + region String, -- Server GeoIP 写入,Agent 不传 + host String, + path String, + status_code Int32, + bytes_sent UInt64, -- 已提供数据(body) + request_length UInt64 DEFAULT 0, -- 接收数据;新增 + request_time_ms UInt32 DEFAULT 0, -- 可选;新增 + created_at DateTime64(3, 'UTC') +) +ENGINE = MergeTree() +PARTITION BY toYYYYMM(logged_at) +ORDER BY (node_id, logged_at, host, status_code, remote_addr) +TTL toDateTime(logged_at) + INTERVAL 90 DAY +SETTINGS index_granularity = 8192; +``` + +| 列 | 类型 | 来源 | +| --- | --- | --- | +| `id` | UInt64 | Server | +| `node_id` | String | 鉴权/payload | +| `logged_at` | DateTime64(3) | `logged_at_unix` | +| `remote_addr` | String | 上报 | +| `region` | String | Server GeoIP | +| `host` | String | 上报 | +| `path` | String | 上报 | +| `status_code` | Int32 | 上报 | +| `bytes_sent` | UInt64 | 上报 → **已提供数据** | +| `request_length` | UInt64 | 上报 → **接收数据** | +| `request_time_ms` | UInt32 | 上报可选 | +| `created_at` | DateTime64(3) | Server now | + +**迁移:** 现表已有 `bytes_sent`;新增: + +```sql +ALTER TABLE of_node_access_logs + ADD COLUMN IF NOT EXISTS request_length UInt64 DEFAULT 0, + ADD COLUMN IF NOT EXISTS request_time_ms UInt32 DEFAULT 0; +``` + +### 5.2 L1 小时汇总(Server 侧 MV) + +**禁止 Agent 写入。** 供看板/节点 24h 快速查询请求数、错误数、字节量。 + +**已实现选型:`SummingMergeTree` + 不含 UV 列。** + +```sql +CREATE TABLE IF NOT EXISTS of_access_log_hourly +( + node_id String, + hour DateTime('UTC'), + host String, + request_count UInt64, + error_count UInt64, + bytes_sent UInt64, + request_length UInt64 +) +ENGINE = SummingMergeTree() +PARTITION BY toYYYYMM(hour) +ORDER BY (node_id, hour, host) +TTL hour + INTERVAL 90 DAY; + +CREATE MATERIALIZED VIEW IF NOT EXISTS of_access_log_hourly_mv +TO of_access_log_hourly +AS +SELECT + node_id, + toStartOfHour(logged_at) AS hour, + host, + toUInt64(count()) AS request_count, + toUInt64(countIf(status_code >= 500)) AS error_count, + sum(bytes_sent) AS bytes_sent, + sum(request_length) AS request_length +FROM of_node_access_logs +GROUP BY node_id, hour, host; +``` + +历史小时(MV 创建前已入库的明细)需一次性回填,见迁移 `202607180003_backfill_access_log_hourly.sql`(ANTI JOIN 防重)。 + +#### UV 策略(必须遵守) + +| 场景 | 数据源 | 算法 | 说明 | +| --- | --- | --- | --- | +| **窗口总 UV**(看板汇总、节点卡片、Zone 汇总) | `of_node_access_logs` 明细 | `uniqExact(remote_addr)`(`TrafficSummary` / 节点聚合) | **唯一权威**;不可用小时 UV 相加 | +| **24h 趋势折线请求/错误/字节** | `of_access_log_hourly` 优先,缺数据回落明细桶 | `sum(request_count)` 等 | 小时路径 **不填** `unique_visitor_count`(恒为 0) | +| **24h 趋势折线分时 UV** | 仅明细桶路径 | 桶内 `uniqExact` | 走 hourly 时 UI 应展示空/0 或隐藏 UV 序列,**禁止**对小时行做 `sum(UV)` | + +**为何 hourly 不存 UV:** + +1. `SummingMergeTree` 只能安全合并可加和计数;`uniqExact` 跨 part 合并需要 `AggregatingMergeTree` + state,实现与查询更重。 +2. 即便存每小时 UV,对多小时窗口 **相加会严重高估**(同一 IP 跨小时重复计)。 +3. 产品「24h 独立访客」只认整窗 `uniqExact`;趋势图主序列是请求量/错误/字节,分时 UV 非主指标。 + +可选未来:若需要分时 UV 曲线,再单独加 `AggregatingMergeTree` 状态表或查询时对明细做 `uniqExact` 按小时 group(成本更高,不阻塞当前看板)。 +### 5.3 L3 事实表:`of_node_metric_snapshots`(保留,语义明确) + +```sql +CREATE TABLE IF NOT EXISTS of_node_metric_snapshots +( + id UInt64, + node_id String, + captured_at DateTime64(3, 'UTC'), + cpu_usage_percent Float64, + memory_used_bytes Int64, + memory_total_bytes Int64, + storage_used_bytes Int64, + storage_total_bytes Int64, + disk_read_bytes Int64, -- 累计原值 + disk_write_bytes Int64, + network_rx_bytes Int64, -- 累计原值 → 宿主机网卡入站 + network_tx_bytes Int64, -- 累计原值 → 宿主机网卡出站 + created_at DateTime64(3, 'UTC') +) +ENGINE = MergeTree() +PARTITION BY toYYYYMM(captured_at) +ORDER BY (node_id, captured_at, id) +TTL toDateTime(captured_at) + INTERVAL 30 DAY +SETTINGS index_granularity = 8192; +``` + +列与现网一致;**文档与 API 必须标注 network_* 为宿主机网卡累计值**。 + +### 5.4 L3 小时汇总:`of_node_metric_capacity_hourly`(保留) + +现有 min/max 用于累计计数器小时增量近似 + CPU/内存平均。逻辑不变: + +* `network_tx_max - network_tx_min` ≈ 该小时宿主机出站 +* **不得** 用于「已提供数据」 + +### 5.5 L2 事实表:`of_node_edge_health`(新建,替换吞吐型 openresty 表) + +```sql +CREATE TABLE IF NOT EXISTS of_node_edge_health +( + id UInt64, + node_id String, + captured_at DateTime64(3, 'UTC'), + status LowCardinality(String), -- healthy / unhealthy / unknown + connections Int64, + created_at DateTime64(3, 'UTC') +) +ENGINE = MergeTree() +PARTITION BY toYYYYMM(captured_at) +ORDER BY (node_id, captured_at, id) +TTL toDateTime(captured_at) + INTERVAL 30 DAY +SETTINGS index_granularity = 8192; +``` + +| 列 | 说明 | +| --- | --- | +| `status` | 瞬时健康 | +| `connections` | 当前连接数 | + +**无** `openresty_rx_bytes` / `openresty_tx_bytes`。 + +### 5.6 关系库(节点最新态,非分析湖) + +与观测湖分离,保持「最新一份」: + +| 表(逻辑名) | 用途 | 关键列 | +| --- | --- | --- | +| `of_nodes`(或现节点表) | 在线、版本、IP | `last_seen_at`, `openresty_status`, `openresty_message`, `agent_version` | +| `of_node_system_profiles` | profile upsert | hostname, cpu_cores, total_memory_bytes, ... | +| 健康事件表 | `health_events` | event_type, severity, message, triggered_at | + +> 具体物理表名以仓库现有 GORM 模型为准;本设计不强制改名,只强制 **不再把业务吞吐写进节点表**。 + +### 5.7 废弃表(停止写入 → TTL 后删除) + +| 表 | 原因 | 替代 | +| --- | --- | --- | +| `of_node_request_reports` | Agent 预聚合 | `of_node_access_logs` + hourly | +| `of_node_traffic_hourly` + MV | 依赖 request_reports | `of_access_log_hourly` | +| `of_node_obs_openresty` | 含业务 rx/tx | `of_node_edge_health` | +| `of_node_openresty_hourly` + MV | 业务吞吐差分 | `of_access_log_hourly` 的 bytes_* | + +Relay 专用 `of_node_obs_frps` / `of_node_obs_frpc` **保留**(非本 Agent 主路径,但同属 CH 观测)。 + +--- + +## 6. 表与协议对照总表 + +| 产品概念 | 协议字段 | 表.列 | 聚合 | +| --- | --- | --- | --- | +| 已提供数据 | `access_logs[].bytes_sent` | `of_node_access_logs.bytes_sent` | `sum` | +| 接收数据 | `access_logs[].request_length` | `...request_length` | `sum` | +| 请求数 | 行数 | — | `count` | +| UV(窗口总) | `remote_addr` | 同左明细 | `uniqExact`(**禁止** sum 小时 UV) | +| Top 域名 | `host` | 同左 | `group by` | +| 状态码分布 | `status_code` | 同左 | `group by` | +| 来源地区 | — | `region`(Server) | `group by` | +| 宿主机网卡出站 | `host_metrics.network_tx_bytes` | `of_node_metric_snapshots.network_tx_bytes` | 时间序差分 | +| 宿主机网卡入站 | `network_rx_bytes` | 同左 | 差分 | +| 磁盘读/写 | `disk_*_bytes` | 同左 | 差分 | +| CPU/内存 | 瞬时字段 | 同左 | avg | +| OpenResty 连接 | `edge_health.connections` | `of_node_edge_health.connections` | 最新/avg | +| OpenResty 健康 | `edge_health.status` | 节点表 + 可选 CH | 最新 | + +**不再存在的映射:** + +| 旧概念 | 旧字段 | 处置 | +| --- | --- | --- | +| OpenResty 出站 | `openresty_tx_bytes` | 删除;用已提供数据 | +| OpenResty 入站 | `openresty_rx_bytes` | 删除;用接收数据 | +| 窗口请求报告 | `traffic_report` | 删除 | + +--- + +## 7. OpenResty 日志格式(与明细对齐) + +目标 `log_format`(与现网一致,保证 `bytes_sent` 键 = body): + +```nginx +log_format openflare_json escape=json + '{"ts":"$time_iso8601","host":"$host","path":"$request_uri",' + '"remote_addr":"$remote_addr","status":$status,' + '"request_time":$request_time,' + '"bytes_sent":$body_bytes_sent,"request_length":$request_length}'; +``` + +Agent 解析: + +* `ts` → `logged_at_unix` +* `bytes_sent` → 协议 `bytes_sent`(已提供) +* `request_length` → 协议 `request_length` +* `request_time` → 可选 `request_time_ms = round(sec * 1000)` + +--- + +## 8. 兼容策略(协议 v1 → v2) + +| 客户端 | Server 行为 | +| --- | --- | +| 新 Agent `schema_version=2` | 按本文写入 L1/L2/L3 | +| 旧 Agent 带 `snapshot` + `access_logs` | 映射为 host_metrics;access_logs 无 request_length 则 0 | +| 旧 Agent 带 `traffic_report` | **丢弃** | +| 旧 Agent 带 `openresty_observation` | 只取 `connections` + 顶层 status;rx/tx 丢弃 | +| 读路径 | 业务 API **只读** access_logs(及 hourly);不再读 request_reports / openresty 吞吐 | + +--- + +## 9. 示例:一次心跳的落库结果 + +**Agent 上报(节选):** + +```json +{ + "schema_version": 2, + "node_id": "n1", + "host_metrics": { + "captured_at_unix": 1720000000, + "cpu_usage_percent": 10, + "memory_used_bytes": 1, + "memory_total_bytes": 2, + "storage_used_bytes": 3, + "storage_total_bytes": 4, + "disk_read_bytes": 100, + "disk_write_bytes": 200, + "network_rx_bytes": 1000, + "network_tx_bytes": 2000 + }, + "edge_health": { + "captured_at_unix": 1720000000, + "status": "healthy", + "message": "", + "connections": 5 + }, + "access_logs": [ + { + "logged_at_unix": 1720000001, + "remote_addr": "1.1.1.1", + "host": "a.example.com", + "path": "/", + "status_code": 200, + "bytes_sent": 500, + "request_length": 80 + } + ] +} +``` + +**写入:** + +1. `of_node_metric_snapshots` 1 行(network_tx=2000 累计) +2. `of_node_edge_health` 1 行(connections=5) +3. `of_node_access_logs` 1 行(bytes_sent=500, request_length=80, region=Server 填充) +4. MV 异步计入 `of_access_log_hourly` + +**查询 24h 已提供数据:** `sum(bytes_sent)` → 至少 500(加历史) +**查询宿主机出站:** 对 snapshots 差分,与 500 **无强制相等关系**。 + +--- + +## 10. 实现检查清单 + +- [x] `pkg/protocol`:v2 类型与 deprecated 兼容字段 +- [x] Agent:只组 `host_metrics` / `edge_health` / `access_logs` / `buffered` +- [x] Server normalize + 停写 request_reports / openresty rx/tx +- [x] CH migration:`request_length`、`request_time_ms`、`of_node_edge_health`、`of_access_log_hourly`、hourly 回填 +- [x] 看板/Zone API 统一读 access log 聚合 +- [x] 文档与前端文案:已提供数据 ≠ 宿主机网卡出站;UV 策略(整窗 uniqExact / 小时路径 UV=0) + +--- + +## 11. 修订记录 + +| 日期 | 说明 | +| --- | --- | +| 2026-07-17 | 初稿:协议 v2、Server 落库流水线、CH/关系库目标表结构与废弃表清单 | diff --git a/docs/design/observability-design.md b/docs/design/observability-design.md new file mode 100644 index 00000000..db2f60a4 --- /dev/null +++ b/docs/design/observability-design.md @@ -0,0 +1,578 @@ +# 边缘可观测与业务流量统计重构设计 + +你会学到:当前观测链路为何出现「看板 OpenResty 出站」与「Zone 已提供数据」不一致、字段与聚合为何冗余,以及目标架构如何让 **Agent 只上报事实、Server 只解释事实**,业务流量以访问日志为唯一真相源。 + +--- + +## 1. 目标 + +### 1.1 要解决的问题 + +1. **双真相源**:业务吞吐同时来自访问日志聚合与 OpenResty 观测差分,数值长期对不上。 +2. **Agent 越权计算**:边缘预聚合 `TrafficReport`、吞吐累计,控制面再聚合一遍,语义难演进、难对账。 +3. **字段语义重叠**:「OpenResty 出站」与「已提供数据」对用户是同一业务问题,系统却用两套字段、两条管道。 +4. **瞬时与累计混用**:60 秒窗口计数被当成进程累计做 24h 差分,造成严重偏低。 +5. **UI 诱导错误对比**:看板与 Zone 页使用相近「流量/数据」文案,却未声明范围与口径差异。 + +### 1.2 重构目标 + +| 目标 | 说明 | +| --- | --- | +| **单一业务真相** | 请求数、已提供数据、UV、状态码分布、Top 域名等 **只** 从访问日志(及其 Server 侧派生汇总)得出 | +| **Agent 只上报事实** | 明细日志 + 机器读数 + 健康瞬时态;**禁止** 业务 UV/TopN/24h 总量等预聚合 | +| **字段收敛** | 一个业务概念对应一个权威字段;机器网卡与业务交付严格分名 | +| **可对账** | 全局「已提供数据」≈ 各 Zone「已提供数据」之和(差仅为未绑定/未知 Host) | +| **可演进** | 改时间窗、TopN、归属规则只改 Server,不升 Agent | + +### 1.3 非目标(本设计不覆盖) + +* 建成通用日志平台、全量日志长期归档或检索产品。 +* 替换 ClickHouse / 取消分析库依赖。 +* 改造 Relay / OpenFlared 的主机指标采集(可对齐原则,但不在本轮协议主路径)。 +* 实时流式告警引擎、APM 链路追踪(OpenTelemetry 服务端已有,与本业务流量模型正交)。 + +--- + +## 2. 范围与约束 + +### 2.1 产品约束(继承) + +* 单租户、全局单激活配置;观测不引入多租户计费隔离。 +* ClickHouse 为访问日志与时序观测的强制分析存储。 +* Agent 无入向控制、Pull 模型;离线期间本地 OpenResty 继续服务,观测可本地缓冲后补传。 + +### 2.2 工程约束 + +* Agent 保持轻量:解析日志行、读 `/proc`、健康检查;不做业务分析。 +* 控制面 API 错误仍走统一信封与 `response.Abort*`。 +* 访问日志字段变更须同时更新 OpenResty `log_format` 与 Agent 解析器,并保证向后兼容至少一个小版本。 + +--- + +## 3. 设计原则 + +### 原则 P1:Agent 上报事实,Server 解释事实 + +```text +Agent = 采集 + 可靠投递(原始/近原始) +Server = 入库 + 聚合 + 归属 + 趋势 + 对账 +``` + +**允许的边缘处理(采集)** + +* 将 JSON access.log 行解析为结构化字段 +* path 长度上限、丢弃非法行、跳过观测端口自身请求 +* 读取网卡/CPU/内存等计数器 **原值** +* 批量、压缩、离线缓冲与重试 + +**禁止的边缘处理(业务计算)** + +* UV / Top 域名 / 状态码直方图 / 窗口 request_count 作为权威指标 +* 为看板单独维护「业务入出站累计」 +* Zone / 域名归属统计、国家分布(国家可在 Server 入库时解析) + +### 原则 P2:业务流量唯一真相 = 访问日志 + +| 业务问题 | 唯一答案 | +| --- | --- | +| 提供了多少数据 | `sum(bytes_sent)` | +| 多少请求 | `count()` | +| 多少独立访客 | `uniqExact(remote_addr)`(或产品约定哈希) | +| 状态码 / Top 域名 | 对日志 `group by` | + +### 原则 P3:三层指标互不混用 + +| 层 | 名称 | 用途 | 典型字段 | +| --- | --- | --- | --- | +| L1 业务交付 | Business Traffic | 用户与 Zone 对账、看板业务趋势 | access log | +| L2 边缘健康 | Edge Health | OpenResty 是否活着、当前连接 | status、connections | +| L3 宿主机资源 | Host Capacity | 容量规划、机器是否打满 | CPU、内存、磁盘、**网卡** | + +禁止将 L3 网卡或 L2 瞬时计数命名为「已提供数据」;禁止将 L1 与 L3 画在同一摘要卡片上却不标注语义。 + +### 原则 P4:一个业务概念一个字段 + +* **已提供数据** ≡ 响应体交付量 ≡ 历史文案中的「OpenResty 出站(业务含义)」→ **只保留 `bytes_sent` 聚合** +* **接收数据**(可选)≡ 请求侧体量 → 日志 `request_length` 聚合 +* **宿主机出站** ≡ `network_tx` 差分,文案必须含「宿主机/网卡」 + +--- + +## 4. 现状问题(基线) + +### 4.1 当前数据流(冗余) + +```text +一次 HTTP 请求 + │ + ├─ access.log 一行 + │ → Agent tail → AccessLogs[] + │ → CH of_node_access_logs + │ → Zone「已提供数据」✅ + │ + ├─ Lua shared dict 窗口/累计计数 + │ → /openflare/observability + │ → TrafficReport + OpenrestyObservation(rx/tx) + │ → CH request_reports / obs_openresty + │ → 看板「OpenResty 入/出站」❌ 易与 Zone 不一致 + │ + ├─ access.log 二次汇总(观测 endpoint 失败时回退) + │ → 又一份 TrafficReport / 吞吐 + │ + └─ 宿主机 network_rx/tx + → Snapshot → 网络趋势中的「主机」曲线 +``` + +### 4.2 字段重叠 + +| 用户感知 | 系统字段 A | 系统字段 B | 问题 | +| --- | --- | --- | --- | +| 出站 / 已提供 | `openresty_tx_bytes` | `bytes_sent` | 业务语义重复 | +| 入站 | `openresty_rx_bytes` | `request_length`(日志) | 业务语义重复 | +| 请求数 | `TrafficReport.request_count` | `count(access_logs)` | 聚合重复且窗口易重计 | +| 出站(机器) | `network_tx_bytes` | (无业务对应) | 应单独命名,勿与业务对账 | + +### 4.3 典型故障模式 + +1. 窗口计数被当累计差分 → 24h 业务吞吐严重偏低。 +2. 小时 rollup `max−min` 对重置型计数失效。 +3. Zone 用日志、看板用观测 → 用户认为系统算错。 +4. 改口径需同步改 Lua、Agent 状态累计、Server 差分、前端文案。 + +--- + +## 5. 目标架构 + +### 5.1 目标数据流 + +```mermaid +flowchart TB + subgraph edge [边缘节点] + OR[OpenResty] + LOG[access.log] + PROC[主机 /proc 与磁盘] + STUB[stub_status 连接数] + AG[Agent] + OR -->|log_format 写行| LOG + LOG -->|仅 tail 增量明细| AG + PROC -->|读数快照| AG + STUB -->|瞬时连接| AG + OR -->|健康探测| AG + end + + subgraph server [控制面 Server] + HB[心跳 / WS 接收] + CH[(ClickHouse)] + AGG[聚合查询层] + API[管理端 API] + HB --> CH + CH --> AGG + AGG --> API + end + + subgraph ui [管理端] + DASH[看板:全局业务趋势] + ZONE[Zone:按域名过滤] + NODE[节点:主机资源 + 健康] + end + + AG -->|AccessLogs + HostSnapshot + Health| HB + API --> DASH + API --> ZONE + API --> NODE +``` + +### 5.2 职责矩阵 + +| 能力 | Agent | Server | 前端 | +| --- | --- | --- | --- | +| 写 access.log | OpenResty | — | — | +| 读并上报明细 | ✅ | 入库 | — | +| sum/count/uniq/TopN | ❌ | ✅ | 展示 | +| Zone 域名过滤 | ❌ | ✅ | 选择 Zone | +| 主机 CPU/内存/网卡 | 读原值上报 | 差分/平均 | 节点/看板资源区 | +| OpenResty 连接数 | 读瞬时上报 | 最近值 | 节点健康 | +| 业务 24h 入出站 | ❌ | 日志聚合 | 统一称「已提供/接收数据」 | + +--- + +## 6. 指标与字段模型 + +### 6.1 权威字段表(目标) + +#### L1 业务交付(来自访问日志) + +| 概念 | 存储字段 | 聚合 | 展示名 | +| --- | --- | --- | --- | +| 请求时间 | `logged_at` | 时间窗过滤 | — | +| 节点 | `node_id` | group | — | +| 客户端 IP | `remote_addr` | `uniq` → UV | 唯一访问者 | +| Host | `host` | group / Zone 映射 | 域名 | +| 路径 | `path` | 可选 | — | +| 状态码 | `status_code` | group | 状态码分布 | +| **已提供数据** | **`bytes_sent`** | **`sum`** | **已提供数据** | +| **接收数据** | **`request_length`** | **`sum`** | **接收数据**(可选展示) | +| 地区 | `region`(Server 解析写入) | group | 来源地区 | + +> 说明:OpenResty `log_format` 中 JSON 键名可继续叫 `bytes_sent`,值必须来自 **`$body_bytes_sent`**(与现网一致),表示响应体交付量,即「已提供数据」。 + +#### L2 边缘健康(瞬时,不做 24h 业务总量) + +| 概念 | 字段 | 说明 | +| --- | --- | --- | +| OpenResty 健康 | `openresty_status` / message | 已有 | +| 当前连接 | `openresty_connections` | stub_status | +| (可选)近窗 QPS 粗估 | 仅节点详情「此刻」,**不得**作为 24h 总量权威 | 若实现须标明「瞬时」 | + +#### L3 宿主机资源 + +| 概念 | 字段 | 展示名 | +| --- | --- | --- | +| CPU / 内存 / 磁盘占用 | 现有 snapshot | 保持 | +| 网卡累计字节 | `network_rx_bytes` / `network_tx_bytes` | **宿主机网卡入/出站** | +| 磁盘 IO 累计 | `disk_read_bytes` / `disk_write_bytes` | 磁盘读/写 | + +### 6.2 废弃或降级字段 + +| 现字段 | 处置 | 原因 | +| --- | --- | --- | +| `openresty_tx_bytes` | **废弃业务用途**;迁移期可读但 UI 不再展示为业务出站 | 与 `bytes_sent` 重复 | +| `openresty_rx_bytes` | **废弃业务用途**;由 `request_length` 聚合替代 | 与日志重复 | +| `TrafficReport` 全量 | **废弃权威地位**;迁移期可停写或仅兼容旧 Agent | 边缘预聚合 | +| `TrafficReport.top_domains` / `status_codes` / `unique_visitor_count` | 改由 Server 查日志 | 同上 | +| Agent state 内业务 lifetime 累计 | 删除 | 违背 P1 | +| Lua shared dict 业务吞吐/窗口请求计数 | 删除或仅保留本地诊断 | 非投递主路径 | + +### 6.3 命名对照(前端文案强制) + +| 禁止混用文案 | 正确文案 | 数据来源 | +| --- | --- | --- | +| OpenResty 出站(指业务量) | **已提供数据** | `sum(bytes_sent)` | +| OpenResty 入站(指业务量) | **接收数据** | `sum(request_length)` | +| 网络出站(未说明) | **宿主机网卡出站** | `network_tx` 差分 | +| 已提供数据 vs 出站 两套卡片 | **只保留一套业务卡片** | 日志 | + +--- + +## 7. Agent 设计 + +### 7.1 心跳载荷(目标协议) + +保留并强化: + +```text +NodePayload + identity / version / openresty_status + profile # 主机概况(低频) + snapshot # L3 资源读数(含网卡累计原值) + openresty_connections # L2 瞬时(可挂在精简 observation 或 snapshot 扩展) + access_logs[] # L1 明细(主路径) + health_events[] + buffered_observability[] # 缓冲的是上述事实,不是报表 + waf_ip_group_checksums +``` + +移除或标记 deprecated(兼容窗口内 Server 忽略写入分析权威路径): + +```text +traffic_report # deprecated +openresty_observation.rx/tx # deprecated(connections 迁出后可删结构) +``` + +### 7.2 Access log 上报要求 + +每条明细至少包含: + +| 字段 | 必填 | 备注 | +| --- | --- | --- | +| `logged_at_unix` | ✅ | 请求完成时间 | +| `remote_addr` | ✅ | UV | +| `host` | ✅ | Zone 映射 | +| `path` | ✅ | 可截断 | +| `status_code` | ✅ | | +| `bytes_sent` | ✅ | body 字节,已提供数据 | +| `request_length` | ✅(协议补齐) | 接收数据;旧 Agent 可缺省为 0 | + +Agent 职责: + +1. 按 offset tail `access.log`(截断/轮转时重置 offset,**只上报文件中仍存在的新行**)。 +2. 结构化解析后批量放入心跳 / WS。 +3. 离线写入本地 buffer,连通后按窗口补传。 +4. **不对明细做 sum/count/uniq。** + +### 7.3 主机 Snapshot + +* 继续上报网卡/磁盘 **累计计数器原值**(非业务预聚合)。 +* Server 侧对累计值做相邻采样非负差分 → 宿主机趋势。 +* 这与「已提供数据」无关,UI 必须分区展示。 + +### 7.4 OpenResty 本地观测 + +**收敛后建议:** + +* 保留:健康检查、`stub_status` 当前连接。 +* 删除主路径依赖:`log.lua` 中对 request/status/domain/rx/tx 的 shared dict 业务计数,以及 `/openflare/observability` 作为 TrafficReport 来源。 +* 若短期内保留 endpoint 供调试,不得再写入 Server 权威分析表。 + +### 7.5 与 Agent 设计文档的关系 + +本设计强化 [Agent 与发布模型](./agent-design.md) 中的「纯粹数据落地」: + +* 配置与证书:落地与上报应用状态。 +* 观测:只搬运事实,不搬运业务结论。 + +--- + +## 8. Server 设计 + +### 8.1 入库 + +| 输入 | 表 | 说明 | +| --- | --- | --- | +| `access_logs[]` | `of_node_access_logs` | 权威业务明细;补齐 `request_length` 列(若尚无) | +| `snapshot` | `of_node_metric_snapshots` | L3;网卡/磁盘累计 | +| 连接数 / 健康 | 现有节点状态或精简 obs 表 | L2 | +| `traffic_report` / openresty rx/tx | **停止作为权威写入** 或兼容期双写但不读 | 迁移后删除写入 | + +GeoIP:继续在 Server 入库路径解析 `remote_addr` → `region`,不在 Agent 做。 + +### 8.2 聚合层(统一) + +所有业务趋势与 Zone 统计共用同一查询语义: + +```text +过滤:logged_at ∈ [since, until] +可选:node_id / host IN (...) +指标: + request_count = count() + unique_visitors = uniqExact(remote_addr) + bytes_provided = sum(bytes_sent) -- 已提供数据 + bytes_received = sum(request_length) -- 接收数据 + 按 hour/bucket 折叠 series + 按 status_code / host / region 分布 +``` + +实现位置: + +* Zone:`GET .../zones/:id/stats`(已有,对齐字段命名) +* 看板:overview 的 traffic / 业务网络趋势 **改为调用同一聚合**(全局、无 host 过滤或 Top 过滤) +* 节点详情:业务量 = 该 `node_id` 过滤的同一聚合;主机网卡仍走 metric 差分 + +### 8.3 派生汇总(可选性能路径) + +当明细查询在 24h 全量节点上过重时,允许 **Server 侧** 物化视图: + +```text +of_access_log_hourly + (hour, node_id, host, request_count, bytes_sent, bytes_received, ...) +``` + +约束: + +* 仅由 CH 从 `of_node_access_logs` 派生,**禁止** Agent 直接写该表。 +* Zone / 看板优先读 rollup,缺口回退明细(与现有 metric hourly 策略类似)。 + +### 8.4 停用的分析路径 + +| 路径 | 迁移后 | +| --- | --- | +| `BuildNetworkTrendPoints` 对 openresty_rx/tx 差分 | 删除或仅保留 network_* 主机曲线 | +| `of_node_obs_openresty` 吞吐字段 | 停止写入;TTL 过期后删表或缩列 | +| `of_node_request_reports` + traffic hourly | 业务趋势不再依赖;可整表废弃 | +| Dashboard compact 中 openresty_tx 序列 | 改为 bytes_provided 序列 | + +--- + +## 9. API 与前端 + +### 9.1 语义统一的响应字段 + +建议在业务统计 API 中统一使用: + +```json +{ + "request_count": 0, + "unique_visitors": 0, + "bytes_provided": 0, + "bytes_received": 0, + "series": [ + { + "bucket_started_at": "...", + "request_count": 0, + "unique_visitors": 0, + "bytes_provided": 0, + "bytes_received": 0 + } + ] +} +``` + +兼容:旧字段 `bytes_sent` 可在一个版本内作为 `bytes_provided` 的别名返回,文档标注 deprecated。 + +### 9.2 看板 + +* **业务区**:请求趋势、已提供数据、接收数据(可选)、状态码、Top 域名、来源地区 —— 全部 L1。 +* **资源区**:CPU/内存、**宿主机网卡**、磁盘 IO —— 全部 L3。 +* **禁止**:在业务区展示「OpenResty 入/出站」作为与 Zone 对账的指标。 + +「24 小时网络与磁盘趋势」建议拆分或改标题: + +* 「24 小时业务流量」→ `bytes_provided` / `bytes_received` / 请求 +* 「24 小时宿主机网络与磁盘」→ `network_*` / `disk_*` + +### 9.3 Zone `/websites/:id` + +* 保持「已提供的数据总计」等卡片。 +* 数据与看板业务区 **同一聚合函数**,仅 `hosts = zone 域名列表`。 +* 文档与 UI 可注明:全局看板含全部 Host;本页仅本 Zone。 + +### 9.4 节点详情 + +* 业务吞吐:该节点 `sum(bytes_sent)` 等。 +* OpenResty:健康 + 当前连接。 +* 网卡:明确「宿主机」。 + +--- + +## 10. OpenResty 与日志格式 + +### 10.1 保持 + +现有 JSON `log_format` 核心字段: + +```text +ts, host, path, remote_addr, status, request_time, +bytes_sent (= $body_bytes_sent), request_length +``` + +### 10.2 变更 + +* 不再依赖 log phase 写入业务 shared dict 计数作为控制面输入。 +* 观测端口请求继续不写业务统计(或 access_log off)。 + +### 10.3 Agent 解析 + +* 协议 `NodeAccessLog` 增加 `request_length`。 +* 旧日志行缺字段时按 0,不阻断整批。 + +--- + +## 11. 兼容与迁移 + +### 11.1 阶段划分 + +| 阶段 | 内容 | 结果 | +| --- | --- | --- | +| **M1 读路径切换** | 看板/节点业务趋势改为 access log 聚合;UI 文案改为已提供/接收数据 | 对账立刻成立;旧字段可仍写入 | +| **M2 协议补齐** | AccessLog 上报 `request_length`;Server 入库 | 接收数据可用 | +| **M3 停写预聚合** | Server 忽略/停写 TrafficReport 与 openresty rx/tx 权威路径 | 减负 | +| **M4 Agent 瘦身** | 移除边缘 TrafficReport 构建、Lua 业务计数、lifetime 累计 state | 符合 P1 | +| **M5 清理** | 删除废弃 CH 表/列、API 字段、前端类型 | 无冗余 | + +### 11.2 兼容策略 + +* 旧 Agent 仍发 `TrafficReport`:Server **不用于** 看板业务趋势。 +* 旧 Agent 无 `request_length`:`bytes_received` 为 0 或不展示。 +* 明细缺失时段:业务图为空或仅部分;**不得**回退到 openresty_tx 冒充已提供数据(避免再次双真相)。 + +### 11.3 数据回填 + +* 历史「已提供数据」以 access log 为准,无需从 openresty 观测回填。 +* 历史看板 openresty 曲线可保留只读至 TTL,或直接隐藏。 + +--- + +## 12. 存储与容量 + +* 业务趋势依赖明细或 hourly rollup,需关注 `of_node_access_logs` TTL 与采样。 +* 若明细量过大:优先 **Server 侧 rollup**,而不是恢复 Agent 预聚合。 +* 可对 path 高基数场景限制明细 path 长度(已有),聚合默认不按完整 path 做全局 Top。 + +--- + +## 13. 验证标准 + +### 13.1 对账 + +在仅有单一 Zone 产生流量的环境: + +```text +看板「已提供数据」(24h) ≈ Zone「已提供的数据总计」(24h) +误差仅来自时间窗对齐(整点截断)与未计入 Host +``` + +多 Zone 时: + +```text +sum(各 Zone 已提供) + sum(未归属 Host) = 全局已提供 +``` + +### 13.2 回归 + +* Agent 单测:只解析与 offset,不出现业务 sum 断言为「上报契约」。 +* Server:Zone stats 与 dashboard business traffic 共用聚合测例。 +* 前端:文案快照/测试中不再出现业务含义的「OpenResty 出站」与「已提供数据」双卡片。 + +### 13.3 性能 + +* 24h 看板聚合 P95 可接受(必要时 hourly MV)。 +* 心跳 payload 体积:明细批量有上限;超限拆缓冲,不在 Agent 做摘要替代。 + +--- + +## 14. 风险与权衡 + +| 风险 | 缓解 | +| --- | --- | +| 明细量大导致 CH 与心跳变重 | 批量、压缩、采样策略评估;Server rollup;限制单次条数 | +| 短暂丢失日志导致业务量偏低 | 本地 buffer 与轮转处理;监控 access log 采集滞后 | +| 用户仍对比「网卡出站」与「已提供」 | UI 分区与文案强制「宿主机」前缀 | +| 旧 Agent 长期在线 | 兼容期忽略预聚合;文档要求升级 Agent 以获得接收数据 | + +**为何不保留 Agent 预聚合作为优化?** + +* 省带宽的代价是再次分裂真相、口径漂移、本次问题重演。 +* 优化应落在 Server 派生表与查询,而不是边缘业务计算。 + +--- + +## 15. 关键决策摘要 + +| 决策 | 选择 | 否决方案 | +| --- | --- | --- | +| 业务流量真相 | 访问日志 | OpenResty dict / TrafficReport | +| Agent 角色 | 只上报事实 | 边缘 UV/TopN/吞吐累计 | +| 「出站」与「已提供」 | 合并为已提供数据 | 双字段双管道长期并存 | +| 网卡流量 | 独立 L3,单独文案 | 与业务出站并列对账 | +| 性能 | CH rollup | Agent 预聚合 | +| 迁移 | 先切读路径再瘦身 Agent | 先删明细依赖预聚合 | + +--- + +## 16. 文档与代码映射(落地时) + +| 区域 | 主要路径 | +| --- | --- | +| 协议 | `pkg/protocol/agent.go` | +| Agent 采集 | `internal/apps/agent/observability/`、`heartbeat/` | +| OpenResty 日志与 Lua | `pkg/render/openresty/`、`internal/apps/agent/nginx/observability_assets.go` | +| Server 入库 | `internal/apps/openflare/agent/observability.go` | +| 日志聚合 | `internal/repository/analytics/node_access_log*.go`、`internal/apps/openflare/zone/stats.go` | +| 看板 | `internal/apps/openflare/dashboard/`、`internal/apps/openflare/observability/analytics.go` | +| 前端 | `frontend/app/(main)/page.tsx`、`components/dashboard/*`、`websites/.../zone-overview.tsx` | + +实现计划见:`docs/plan/20260717-observability-redesign.md`。 + +**推荐阅读顺序:** + +1. **[观测数据传输模型](./observability-transport-model.md)**(最新:传什么、从哪采、频率、示例 JSON) +2. [Agent 上报协议与观测落库数据模型](./observability-data-model.md)(协议字段与 DDL) + +--- + +## 17. 修订记录 + +| 日期 | 说明 | +| --- | --- | +| 2026-07-17 | 初稿:针对双真相、Agent 预聚合、字段冗余给出目标架构与迁移阶段 | +| 2026-07-17 | 增补协议/表结构专章链接 `observability-data-model.md` | diff --git a/docs/design/observability-transport-model.md b/docs/design/observability-transport-model.md new file mode 100644 index 00000000..a0615c52 --- /dev/null +++ b/docs/design/observability-transport-model.md @@ -0,0 +1,488 @@ +# 边缘观测数据传输模型(现行目标版) + +> **本文是「Agent ↔ Server 观测数据怎么传」的最新权威说明。** +> 读完应能回答:传什么、从哪采、多久采一次、Server 怎么存、产品指标从哪查。 +> 协议字段与 DDL 细节另见 [观测上报协议与表结构](./observability-data-model.md);问题背景见 [边缘可观测与业务流量统计](./observability-design.md)。 + +--- + +## 0. 先记住三层(不要混) + +| 层 | 回答的问题 | 唯一数据来源 | 产品例子 | +| --- | --- | --- | --- | +| **L1 业务交付** | 提供了多少数据?多少请求? | **access.log 明细** | 已提供数据、请求数、UV、状态码、Top 域名 | +| **L2 边缘健康** | OpenResty 活着吗?现在多少连接? | **本机 `/openflare/observability`** | 节点健康、当前连接 | +| **L3 宿主机资源** | CPU/内存/磁盘/网卡怎样? | **操作系统读数** | 容量趋势、宿主机网卡 | + +**三层互不对账。** +「已提供数据」≠「当前连接」≠「宿主机网卡出站」。 + +--- + +## 1. 总览:谁采集、谁上报、谁聚合 + +```text +┌─────────────────────────────────────────────────────────────┐ +│ 边缘节点 │ +│ │ +│ 访客请求 ──► OpenResty │ +│ │ │ +│ ├─ access.log(每请求一行) ←── L1 采集点 │ +│ │ │ +│ └─ 连接状态(进程内维护) │ +│ │ │ +│ ▼ │ +│ GET /openflare/observability ←── L2 读快照 │ +│ (不扫日志、不重算业务量) │ +│ │ +│ 操作系统 /proc 等 ────────────────────── L3 读快照 │ +│ │ +│ ┌────────── Agent ──────────┐ │ +│ │ 默认每 3s 组一包 NodePayload │ │ +│ │ · tail access.log 增量 │ │ +│ │ · GET 本机 observability │ │ +│ │ · 读 host_metrics │ │ +│ └────────────┬──────────────┘ │ +└─────────────────────────────│──────────────────────────────────┘ + │ HTTP 心跳 或 WebSocket status + ▼ +┌─────────────────────────────────────────────────────────────┐ +│ Server(控制面) │ +│ · 明细 → ClickHouse of_node_access_logs │ +│ · 健康 → 节点最新态 + of_node_edge_health │ +│ · 主机 → of_node_metric_snapshots │ +│ · 业务趋势 / Zone 统计 = 只对 access_logs 做 sum/count/uniq │ +└─────────────────────────────────────────────────────────────┘ +``` + +| 角色 | 做什么 | 不做什么 | +| --- | --- | --- | +| OpenResty | 写 access.log;维护连接数 | 不向控制面直接上报 | +| Agent | **采集事实并上报** | **不算** UV/TopN/24h 已提供数据 | +| Server | 入库 + **聚合解释** | 不信任边缘业务预汇总 | + +--- + +## 2. 采集频率(默认) + +| 动作 | 默认频率 | 配置 | +| --- | --- | --- | +| Agent → Server 上报 | **每 3 秒** 一次完整 payload | `heartbeat_interval` / 控制面 `agent_heartbeat_interval`(毫秒,默认 `3000`) | +| 组包时 tail access.log | **随上报**(两次上报之间的新行) | 同上 | +| 组包时 GET `/openflare/observability` | **随上报**(读**当前**连接快照) | 同上 | +| 组包时读主机指标 | **随上报** | 同上 | +| OpenResty 写 access.log | **每个请求结束时** 1 行 | 与心跳无关 | +| 连接数在进程内更新 | **连接变化时**(内核维护) | 与心跳无关 | +| 离线补传窗口 | 默认保留约 **60 分钟** | `observability_replay_minutes` | +| 节点离线判定 | 约 **60 秒** 无成功心跳 | `node_offline_threshold`(默认 `60000` 毫秒) | + +**说明:** + +- Agent **没有**单独的「采样时钟」;**采样点 = 上报点**(默认 3s)。 +- access.log 是「请求级连续写入」;Agent 只是周期性 **搬运增量行**。 +- `/openflare/observability` **不是**「被调用才开始统计业务」;对连接而言是 **读 Nginx 已有瞬时值**。 + +传输通道: + +- **HTTP 心跳**:按间隔 POST 整包。 +- **WebSocket**:连通后按同一间隔发 `status` 消息(内容同构);此时不再走 HTTP 心跳双发。 + +--- + +## 3. Agent → Server 数据包(NodePayload v2) + +### 3.1 结构骨架 + +```json +{ + "schema_version": 2, + "node_id": "n_01hxyz", + "name": "edge-shanghai-1", + "ip": "203.0.113.10", + "version": "3.4.0", + "ext_version": "", + "current_version": "20260718-abc", + "last_error": "", + "profile": { }, + "host_metrics": { }, + "edge_health": { }, + "access_logs": [ ], + "buffered": [ ], + "health_events": [ ], + "waf_ip_group_checksums": { } +} +``` + +| 字段 | 层 | 含义 | +| --- | --- | --- | +| 身份/版本/last_error | 控制 | 节点是谁、跑什么版本 | +| `profile` | 低频概况 | 主机名、核数等(变化才报) | +| `access_logs` | **L1** | 访问明细增量 | +| `edge_health` | **L2** | OpenResty 健康 + 当前连接 | +| `host_metrics` | **L3** | CPU/内存/磁盘/网卡读数 | +| `buffered` | 补传 | 离线期间攒的事实批次 | +| `health_events` | 事件 | 如 openresty_unhealthy | +| `waf_ip_group_checksums` | 同步 | 非观测湖 | + +**目标态不再作为权威业务数据(旧字段,兼容期可忽略):** + +- `traffic_report`(窗内请求/UV/Top 域名等预聚合) +- `openresty_observation.openresty_rx_bytes` / `openresty_tx_bytes` +- 顶层与「已提供数据」平行的业务「出站」字段 + +--- + +## 4. L1 业务:access_logs + +### 4.1 采集从哪里来 + +| 步骤 | 位置 | 说明 | +| --- | --- | --- | +| 1 | OpenResty `log_format openflare_json` | 每请求写一行 JSON 到 `access_log_path` | +| 2 | Agent 按文件 offset **tail 增量** | 两次心跳之间的新行 | +| 3 | 解析后放入 `access_logs[]` | 可截断过长 path;**不做 sum/count** | + +日志格式(OpenResty 变量): + +```text +ts ← $time_iso8601 +host ← $host +path ← $request_uri +remote_addr ← $remote_addr +status ← $status +request_time ← $request_time +bytes_sent ← $body_bytes_sent 【已提供数据 = 响应体字节】 +request_length← $request_length 【接收数据】 +``` + +观测端口请求 **不写** 业务 access.log(独立 server `access_log off`)。 + +### 4.2 上报示例 + +```json +"access_logs": [ + { + "logged_at_unix": 1721289601, + "remote_addr": "198.51.100.20", + "host": "www.example.com", + "path": "/api/v1/ping", + "status_code": 200, + "bytes_sent": 1024, + "request_length": 128, + "request_time_ms": 15 + }, + { + "logged_at_unix": 1721289602, + "remote_addr": "198.51.100.21", + "host": "www.example.com", + "path": "/index.html", + "status_code": 200, + "bytes_sent": 8192, + "request_length": 300, + "request_time_ms": 8 + } +] +``` + +| 字段 | 解释 | +| --- | --- | +| `bytes_sent` | **已提供数据**(单请求);全局/Zone 合计 = Server `sum` | +| `request_length` | **接收数据**(单请求) | +| `logged_at_unix` | 请求完成时间(业务时间轴) | +| `host` | 用于 Zone 域名过滤 | +| 无 `region` | **Server 入库时** GeoIP 写入 | + +### 4.3 Server 如何用(产品指标) + +| 产品指标 | 算法(仅 L1) | +| --- | --- | +| 已提供数据 | `sum(bytes_sent)` | +| 接收数据 | `sum(request_length)` | +| 请求数 | `count()` | +| UV | `uniqExact(remote_addr)` | +| 状态码分布 | `group by status_code` | +| Top 域名 | `group by host` | +| Zone 页 | 同上 + `host IN (该 Zone 域名)` | +| 看板业务区 | 同上,全局或 Top 过滤 | + +落库表:`of_node_access_logs`(可选 Server 侧 `of_access_log_hourly` 加速,**Agent 不写**)。 + +### 4.4 频率再强调 + +```text +请求发生 ──立即──► 写 access.log +Agent 每 3s ──搬运──► 这 3s 内新行(可能 0 行,也可能很多行) +Server ──立即/批量──► CH +``` + +业务量正确性 **不依赖** 3s 对齐;3s 只影响「明细到达控制面的延迟」和单包条数。 + +--- + +## 5. L2 健康:edge_health 与 `/openflare/observability` + +### 5.1 本机监测口(合并后目标) + +**只保留一个接口:** + +```http +GET http://127.0.0.1:{openresty_observability_port}/openflare/observability +``` + +默认端口:**18081**(`openresty_observability_port`)。 + +**职责:** 回答「OpenResty 此刻怎样」,**不**回答业务已提供多少数据。 + +#### 返回示例(目标 JSON) + +```json +{ + "ok": true, + "captured_at_unix": 1721289600, + "connections": { + "active": 42, + "reading": 0, + "writing": 1, + "waiting": 41 + } +} +``` + +| 字段 | 是否瞬时 | 从哪来 | 说明 | +| --- | --- | --- | --- | +| `ok` | 当次探测 | 能返回 200 即 true | 探活 | +| `captured_at_unix` | 采样时刻 | `ngx.time()` | 与上报对齐 | +| `connections.active` | **瞬时** | Nginx 连接状态(原 stub_status Active) | 当前活跃连接 | +| `reading` / `writing` / `waiting` | **瞬时** | 同上细分 | 可选但建议带 | + +**不返回(已从目标模型删除):** + +| 旧字段 | 原因 | +| --- | --- | +| `request_count` / `error_count` / UV / status_codes / top_domains | 业务窗汇总,改由 access log | +| `openresty_rx_bytes` / `openresty_tx_bytes` | 与已提供/接收数据重复且易错 | +| `source_countries` | 从未实现;国家走 Server GeoIP | +| `server.accepts/handled/requests` | 进程累计 counter,易与业务请求混淆;主路径不收录 | + +**`/openflare/stub_status`:** 合并进上述 JSON 后 **删除**(过渡期可双挂,Agent 只打合并口)。 + +### 5.2 采集机制(读快照,不是「调用才开始统计业务」) + +```text +Nginx 在连接建立/释放时维护 Active connections 等 + │ +Agent GET /openflare/observability + │ +只读取「当前值」拼 JSON 返回 +``` + +- **不是** GET 一次才去扫 access.log。 +- **不是** 60 秒业务均值。 +- 是 **瞬时 gauge 快照**。 + +### 5.3 上报示例(装进 NodePayload) + +```json +"edge_health": { + "captured_at_unix": 1721289600, + "status": "healthy", + "message": "", + "connections": 42 +} +``` + +| 字段 | 来源 | +| --- | --- | +| `status` / `message` | Agent 对 OpenResty 健康探测结果(配置校验/进程等,可与观测口 `ok` 配合) | +| `connections` | 观测口 `connections.active` | + +落库:关系库节点最新状态 + 可选 CH `of_node_edge_health` 做连接曲线。 + +--- + +## 6. L3 主机:host_metrics + +### 6.1 采集从哪里来 + +Agent 读本机(如 `/proc`、磁盘统计等),**每次组包时读一次**。 + +| 字段 | 语义 | 说明 | +| --- | --- | --- | +| `cpu_usage_percent` | 瞬时 | 当前 CPU% | +| `memory_*` / `storage_*` | 瞬时用量/总量 | 占用率在 Server 或展示层算 | +| `disk_read_bytes` / `disk_write_bytes` | **累计 counter** | 内核累计 IO | +| `network_rx_bytes` / `network_tx_bytes` | **累计 counter** | **宿主机网卡**,不是已提供数据 | + +### 6.2 上报示例 + +```json +"host_metrics": { + "captured_at_unix": 1721289600, + "cpu_usage_percent": 12.5, + "memory_used_bytes": 4294967296, + "memory_total_bytes": 16106127360, + "storage_used_bytes": 50000000000, + "storage_total_bytes": 107374182400, + "disk_read_bytes": 9000000000, + "disk_write_bytes": 12000000000, + "network_rx_bytes": 500000000000, + "network_tx_bytes": 800000000000 +} +``` + +### 6.3 Server 如何处理累计字段 + +```text +存原值时间序列 +展示「这段时间网卡出站」时: + delta = 本次 - 上次 + 若 delta < 0 → 视为重启/计数器归零,本段增量记 0,从新基线继续 + 若 delta >= 0 → 记入该时段增量 +``` + +- Agent **上报原值**,不在边缘算 24h 总量。 +- **禁止** 对累计原值做 `sum` 当业务量。 +- 文案必须是 **「宿主机网卡」**,禁止叫「已提供数据 / OpenResty 出站」。 + +落库:`of_node_metric_snapshots`(可选 capacity hourly MV)。 + +--- + +## 7. 一次完整上报示例(拼起来) + +```json +{ + "schema_version": 2, + "node_id": "n_01hxyz", + "name": "edge-shanghai-1", + "ip": "203.0.113.10", + "version": "3.4.0", + "ext_version": "", + "current_version": "20260718-abc", + "last_error": "", + "host_metrics": { + "captured_at_unix": 1721289600, + "cpu_usage_percent": 12.5, + "memory_used_bytes": 4294967296, + "memory_total_bytes": 16106127360, + "storage_used_bytes": 50000000000, + "storage_total_bytes": 107374182400, + "disk_read_bytes": 9000000000, + "disk_write_bytes": 12000000000, + "network_rx_bytes": 500000000000, + "network_tx_bytes": 800000000000 + }, + "edge_health": { + "captured_at_unix": 1721289600, + "status": "healthy", + "message": "", + "connections": 42 + }, + "access_logs": [ + { + "logged_at_unix": 1721289595, + "remote_addr": "198.51.100.20", + "host": "www.example.com", + "path": "/", + "status_code": 200, + "bytes_sent": 4096, + "request_length": 200, + "request_time_ms": 12 + } + ], + "buffered": [], + "health_events": [], + "waf_ip_group_checksums": { + "1": "d41d8cd98f00b204e9800998ecf8427e" + } +} +``` + +**Server 落库示意:** + +| payload 块 | 写入 | +| --- | --- | +| `access_logs[0]` | CH 一行,`bytes_sent=4096`,`region` 由 GeoIP 填 | +| `edge_health` | 节点 `openresty_status=healthy`,connections=42 | +| `host_metrics` | CH metric 一行累计/瞬时字段 | + +**产品查询示意(24h):** + +- 已提供数据 = 该节点(或全局)日志 `sum(bytes_sent)` +- 当前连接 = 最新 `edge_health.connections` +- 宿主机网卡出站 = metric 上 `network_tx` 非负差分之和 + +三者数字 **不必相等**。 + +--- + +## 8. 离线补传 `buffered` + +Agent 上报失败时,把 **同一类事实** 按窗口缓存在本地(默认约 60 分钟),恢复后塞进 `buffered[]`: + +```json +"buffered": [ + { + "captured_at_unix": 1721289500, + "host_metrics": { }, + "edge_health": { }, + "access_logs": [ ] + } +] +``` + +- 只装事实,不装旧 TrafficReport。 +- Server 处理逻辑与主字段相同。 + +--- + +## 9. 端到端时序(默认 3s) + +```text +t=0.0s 访客请求完成 → 写 access.log 一行;连接数可能变化 +t=0.1s 又一请求 → 又一行 log +… +t=3s Agent 心跳: + · 读走 2 行 access_logs + · GET observability → connections=42 + · 读 host_metrics + · 发给 Server +t=3s+ Server 入库;看板/Zone 查询时聚合日志 +t=6s 下一轮… +``` + +--- + +## 10. 旧模型对照(帮助消歧) + +| 旧做法 | 新模型 | +| --- | --- | +| Lua dict 60s 窗 request_count + Agent 10s 拉 + Server sum | **删除**;请求数 = 日志 count | +| openresty_tx 当「出站」 | **删除**;已提供数据 = `sum(bytes_sent)` | +| 两个口 observability + stub_status | **合并为一个** observability,只返回连接/探活 | +| TrafficReport 预聚合 | **不作为权威**;兼容期可忽略 | +| 业务与网卡混称「流量」 | **分文案、分 API、分表** | + +--- + +## 11. 配置与实现索引 + +| 项 | 位置/键 | +| --- | --- | +| 心跳间隔 | Agent `heartbeat_interval`;控制面 `agent_heartbeat_interval`(默认 3000ms) | +| 离线阈值 | 控制面 `node_offline_threshold`(默认 60000ms) | +| 观测端口 | `openresty_observability_port`(默认 18081) | +| access.log 路径 | `access_log_path` | +| 补传分钟数 | `observability_replay_minutes`(默认 60) | +| 协议类型 | `pkg/protocol/agent.go`(落地时按 v2 演进) | +| 表结构 DDL | [observability-data-model.md](./observability-data-model.md) | + +--- + +## 12. 修订记录 + +| 日期 | 说明 | +| --- | --- | +| 2026-07-18 | 初稿:作为「最新传输模型」单页说明——三层、频率、示例 JSON、采集来源、与旧模型对照 | +| 2026-07-18 | 默认上报间隔 3s;离线阈值 60s;补传窗口 60 分钟 | +| 2026-07-18 | M5:edge_health 表、access_log_hourly、废弃 request_reports/obs_openresty 吞吐表 | diff --git a/docs/plan/20260717-observability-redesign.md b/docs/plan/20260717-observability-redesign.md new file mode 100644 index 00000000..8104ea54 --- /dev/null +++ b/docs/plan/20260717-observability-redesign.md @@ -0,0 +1,99 @@ +# 边缘可观测与业务流量统计重构 — 实现计划 + +说明:本计划对应设计文档 [observability-design.md](../design/observability-design.md)。重大架构重构,按阶段交付,避免一次大爆炸。 + +--- + +## 0. 落地进度(2026-07-18) + +* [x] M1 看板业务趋势改读 access log;网络图文案改为已提供/接收 + 宿主机网卡 +* [x] M2 协议 v2 字段(host_metrics/edge_health/request_length);CH 列 `request_length`/`request_time_ms` +* [x] M3 Agent:观测口仅健康连接;payload 不再发 TrafficReport;access_logs 带 request_length +* [x] M4 Server:停写 TrafficReport;openresty 仅存 connections;明细入库带 request_length +* [x] 分布图 status/top domains + 节点行请求/UV 改 access log;24h UV 用 uniqExact;API bytes_provided/received +* [x] M5:`of_node_edge_health`、`of_access_log_hourly`(+MV);删除 request_reports/traffic_hourly/openresty_hourly/obs_openresty;写入/查询改道 +* [x] 收尾:清 openresty hourly / request_report 死路径;edge_health 写全 status;cleanup 命名 `node_edge_health`;hourly 回填 SQL + UV 策略文档 +* [x] 协议/API 去兼容层(Agent 销毁重建):删除 TrafficReport / openresty_observation / snapshot 别名 / request_reports API 字段 / openresty_rx|tx +* [x] 前端 UV 文案:24h/查询窗口独立访客;趋势图不绘分时 UV +* [ ] 真实环境 ClickHouse 迁移 + `202607180003` 回填(本机 Docker 未起时需运维执行) + +## 1. 目标与背景 (Goal & Context) + +* **需求背景**:看板「OpenResty 入/出站」与 Zone「已提供数据」不一致;Agent 预聚合与访问日志双轨;`openresty_tx` 与 `bytes_sent` 业务语义重复。 +* **开发范围 (Scope)**: + * **必做**:业务趋势统一为访问日志聚合;UI 字段与文案收敛;协议补齐 `request_length`;停用预聚合作为权威源;Agent 瘦身。 + * **后续**:废弃 CH 表清理、hourly rollup 性能优化、Relay 指标对齐。 +* **Out of Scope**:通用日志平台、替换 ClickHouse、APM。 + +--- + +## 2. 设计与决策 (Design & Decisions) + +* **核心对象**:以 `of_node_access_logs` 为 L1 权威;主机 snapshot 为 L3;OpenResty 仅健康/连接为 L2。 +* **传输模型(示例与频率)**:见 [observability-transport-model.md](../design/observability-transport-model.md)。 +* **协议与表结构**:见 [observability-data-model.md](../design/observability-data-model.md)(NodePayload v2、落库流水线、DDL、废弃表)。 +* **API**:看板与 Zone 共用聚合语义;`bytes_provided` / `bytes_received`(兼容 `bytes_sent` 别名)。 +* **数据流**:见 [observability-design.md](../design/observability-design.md) §5。 +* **权衡**:性能用 Server 侧 rollup,不恢复 Agent 预聚合。 + +--- + +## 3. 阶段与修改清单 (Proposed Changes) + +### 阶段 M1 — 读路径切换(优先对账) + +* #### [MODIFY] `internal/apps/openflare/dashboard/*`、`observability/analytics.go` + * 业务 24h 趋势改为 access log 聚合(全局)。 + * 网络趋势中业务曲线与主机网卡分离。 +* #### [MODIFY] 前端 dashboard 组件与文案 + * 「OpenResty 出站/入站」→「已提供数据/接收数据」或拆卡片。 +* #### [MODIFY] Zone stats 字段对齐(如需别名) +* **验收**:单 Zone 流量时看板已提供 ≈ Zone 已提供。 + +### 阶段 M2 — 协议与入库补齐 + +* #### [MODIFY] `pkg/protocol/agent.go` — `NodeAccessLog.request_length` +* #### [MODIFY] Agent 解析与 CH 写入列 +* #### [MODIFY] goose ClickHouse migration(如缺列) + +### 阶段 M3 — 停写预聚合权威路径 + +* #### [MODIFY] Server persist:TrafficReport / openresty rx/tx 不再驱动看板 +* 可选:直接停写以减 CH 压力 + +### 阶段 M4 — Agent 瘦身 + +* #### [MODIFY] 移除 TrafficReport 构建主路径、Lua 业务 dict 计数、state 内业务累计 +* #### [MODIFY] 心跳仅明细 + snapshot + 连接/健康 + +### 阶段 M5 — 清理 + +* 删除废弃 API 字段、前端类型、CH 表/MV、相关测试夹具 +* 更新 agent-design / changelog(代码变更时) + +--- + +## 4. 验证计划 (Verification Plan) + +### 自动化 + +* `go test`:zone stats、dashboard 聚合、agent access log 解析 +* 前端:zone / dashboard 文案与字段测试 + +### 手动 + +* 制造已知大小响应,对比 Zone 与看板 24h 已提供数据 +* 确认宿主机网卡曲线与业务已提供数据分区展示、数值可不一致且文案不诱导对账 + +### 质量门禁 + +* `make swagger`(若 API 变更) +* `make code-check` +* `make prettier` + +--- + +## 5. 依赖与风险 + +* 明细量大时 M1 需同步评估 hourly rollup(仍 Server 侧)。 +* 旧 Agent 无 `request_length` 时接收数据为空,需 UI 降级。 diff --git a/docs/plan/20260718-observability-ch-migrate-runbook.md b/docs/plan/20260718-observability-ch-migrate-runbook.md new file mode 100644 index 00000000..61e63929 --- /dev/null +++ b/docs/plan/20260718-observability-ch-migrate-runbook.md @@ -0,0 +1,71 @@ +# ClickHouse 观测表迁移与小时汇总回填(运维手册) + +适用:M5 观测存储(`of_node_edge_health`、`of_access_log_hourly`、删旧表)及历史小时回填。 + +## 前提 + +* 控制面 `config.yaml` / 环境变量中 ClickHouse 已启用,账号可写 `openflare` 库。 +* 备份策略已就绪(可选:对 `of_node_access_logs` 做快照)。 +* **Agent 升级策略为销毁重建**;勿混跑旧 Agent(旧协议字段已从 Server 删除)。 + +## 1. 自动迁移(推荐) + +进程启动时 `migrator.MigrateClickHouse()` 会按 goose 顺序执行: + +| 版本 | 作用 | +| --- | --- | +| `202607180001` | access log 增加 `request_length` / `request_time_ms` | +| `202607180002` | 建 `of_node_edge_health`、`of_access_log_hourly`(+MV);删 request_reports / openresty 吞吐表 | +| `202607180003` | 从明细 ANTI JOIN 回填近 90 天 `of_access_log_hourly` | + +启动 API / all 模式一次即可: + +```bash +# 示例:本地 +./bin/openflare api +# 或 +make run # 以项目实际入口为准 +``` + +查看 goose 版本表(ClickHouse)确认三版本均已应用。 + +## 2. 仅回填(迁移已执行、MV 创建前缺历史) + +若只需重跑回填 SQL: + +```bash +clickhouse-client --host 127.0.0.1 --port 9000 \ + --user default --password "$CLICKHOUSE_PASSWORD" \ + --database openflare \ + --multiquery < internal/db/migrator/goose/clickhouse/202607180003_backfill_access_log_hourly.sql +``` + +(goose 文件含 `+goose Up` 注释,若 client 报错可去掉注释行后执行 INSERT 主体。) + +回填可重复:`ANTI JOIN` 跳过已有 `(node_id, hour, host)`。 + +## 3. 验收 + +```sql +-- 新表存在 +SHOW TABLES FROM openflare LIKE 'of_node_edge_health'; +SHOW TABLES FROM openflare LIKE 'of_access_log_hourly'; + +-- 旧表应不存在 +SHOW TABLES FROM openflare LIKE 'of_node_request_reports'; +SHOW TABLES FROM openflare LIKE 'of_node_obs_openresty'; + +-- 小时汇总有数据(有历史访问时) +SELECT count() FROM of_access_log_hourly; +SELECT min(hour), max(hour), sum(request_count) FROM of_access_log_hourly; +``` + +看板 24h 请求趋势应优先走 hourly;UV 卡片为整窗独立访客,**不等于**小时 UV 之和。 + +## 4. 本机执行记录 + +| 日期 | 环境 | 结果 | +| --- | --- | --- | +| 2026-07-18 | 开发机 | Docker daemon 未启动,未能 live 迁移;SQL 与 goose 文件已入库 | + +运维在目标环境按 §1–§3 执行后更新本表。 diff --git a/docs/plan/index.md b/docs/plan/index.md index 05bfb1f3..5c31cf48 100644 --- a/docs/plan/index.md +++ b/docs/plan/index.md @@ -15,6 +15,7 @@ * [Zone 与域名资源重构](./20260712-zone-domain-refactor.md):以 Zone 和正规化 Zone 域名替代托管域名及反代路由中的域名/证书冗余字段。 * [WAF 可编排规则](./20260713-waf-orchestration.md):使用 React Flow 编辑 DAG 规则,发布时编译并由 OpenResty 纯内存执行。 +* [边缘可观测与业务流量统计重构](./20260717-observability-redesign.md):访问日志为业务唯一真相;Agent 只上报明细与主机读数;收敛「出站/已提供」双字段。 ## 使用建议 diff --git a/docs/reference/configuration.md b/docs/reference/configuration.md index b9a906ef..ee8d3d2f 100644 --- a/docs/reference/configuration.md +++ b/docs/reference/configuration.md @@ -192,9 +192,9 @@ Server 的所有核心基础配置定义在 `config.yaml` 中,且均支持环 | 配置键 (Key) | 数据类型 | 作用说明 | 默认值 | | --- | --- | --- | --- | | `agent_discovery_token` | `string` | 新节点首次一键接入并自动注册的全局通用验证发现 Token | 无(系统初始化生成) | -| `agent_heartbeat_interval`| `int` | 控制并向所有接入 Agent 周期下发的标准心跳检测间隔(毫秒) | `10000` (10s) | +| `agent_heartbeat_interval`| `int` | 控制并向所有接入 Agent 周期下发的标准心跳检测间隔(毫秒) | `3000` (3s) | | `agent_websocket_upgrade_enabled` | `bool` | 是否授权 Agent 在 HTTP 心跳握手成功后升级建立持久 WebSocket 实时连接 | `true` | -| `node_offline_threshold` | `int` | 在管理后台中判定节点失去心跳并标注为离线状态的无响应阈值(毫秒) | `120000` (120s) | +| `node_offline_threshold` | `int` | 在管理后台中判定节点失去心跳并标注为离线状态的无响应阈值(毫秒) | `60000` (60s) | | `agent_update_repo` | `string` | Agent 节点更新下载自身二进制的 Release 仓库源 | `Rain-kl/OpenFlare` | | `geoip_provider` | `string` | GeoIP 提供商,支持 `maxmind` 等,用于 WAF 防护时地域分析 | `ipinfo` | | `database_auto_cleanup_enabled` | `bool` | 是否在每天凌晨 3:00 自动清理过期观测历史日志(降低数据库空间) | `true` | @@ -323,9 +323,9 @@ Server 的所有核心基础配置定义在 `config.yaml` 中,且均支持环 | `mmdb_download_url` | WAF GeoIP mmdb 周期更新地址 | 否 | GeoLite2 Country 更新地址;首次缺失时从程序内嵌数据库初始化 | | `city_mmdb_download_url` | WAF City MMDB 周期更新地址 | 否 | GeoLite2 City 更新地址;首次缺失时从程序内嵌数据库初始化 | | `observability_buffer_path` | 观测补报缓冲文件路径 | 否 | `data_dir/var/lib/openflare/observability-buffer.json` | -| `observability_replay_minutes` | 自动补传最近观测窗口分钟数 | 否 | `15` | +| `observability_replay_minutes` | 自动补传最近观测窗口分钟数 | 否 | `60` | | `state_path` | Agent 本地状态文件路径 | 否 | `data_dir/var/lib/openflare/agent-state.json` | -| `heartbeat_interval` | 心跳间隔 | 否 | `10000` 毫秒 | +| `heartbeat_interval` | 心跳间隔 | 否 | `3000` 毫秒 | | `request_timeout` | HTTP 请求超时 | 否 | `10000` 毫秒 | ---