diff --git a/README.en.md b/README.en.md deleted file mode 100644 index 3c8170c7..00000000 --- a/README.en.md +++ /dev/null @@ -1,180 +0,0 @@ -
- -# OpenFlare - -**[📖 中文](./README.md) | [English](./README.en.md)** - -OpenFlare is an open-source CDN orchestration and edge security platform. It supports reverse proxy, centralized configuration synchronization, in-network tunneling (Tunnels), dynamic WAF protection, and CC defense challenges. - -
- -

- - license - - - release - - - ghcr - -

- -> [!WARNING] -> After the first login with the `admin` user, you must change the default password `12345678`. -> -> The BETA version is a temporary product in the development and testing stage and may have unknown issues. It should not be used in production environments. - -## Documentation - -**https://open-flare.pages.dev** - -Common entry points: - -* [Quick Start](https://open-flare.pages.dev/guide/quick-start) -* [Deployment Guide](https://open-flare.pages.dev/deployment/deployment) -* [Configuration Reference](https://open-flare.pages.dev/reference/configuration) -* [System Design](https://open-flare.pages.dev/design/) - -## Core Capabilities - -* **Reverse Proxy Configuration Management**: Uses website rules as the aggregation boundary, supports multi-domain binding and multi-upstream load balancing, and centrally manages reverse proxy configurations for all OpenResty nodes. -* **Secure In-Network Tunneling (Tunnels)**: Open-source version of Cloudflare Tunnels. No public IP or exposed inbound ports are required. Securely reverse-proxy internal web services to the public internet through Relay relay nodes and OpenFlared clients. -* **Edge WAF Security Protection**: Provides global and custom rule groups, supports manual/auto/subscription-type IP groups, MaxMind GeoIP national-level geographic access control, IP group member Checksum differential synchronization (no Nginx reload required), and custom blocking responses. -* **CC Defense and Human-Computer Challenge (PoW)**: Built-in high-performance client-side cryptography Proof of Work challenge (similar to Turnstile). Secures high-speed interception and blocking of zombie networks and crawlers at the gateway edge. -* **Pages Static Hosting**: Supports uploading or synchronizing pre-built artifacts from restricted Remote URLs or public GitHub Release assets. GitHub latest can be checked periodically and optionally auto-published. All sources are unified to generate immutable deployments, pulled by the edge Agent and served locally by OpenResty, supporting rollbacks, SPA Fallback, and API reverse proxy. -* **TLS Certificate Automation**: Supports dynamic certificate uploads, automatic multi-domain certificate matching and binding, and automatic issuance and renewal of certificates from Let's Encrypt via the ACME protocol. -* **Uptime Kuma Monitoring Synchronization**: Integrated with Uptime Kuma to automatically perform differential synchronization of monitoring site lists, real-time awareness of node availability and service status. -* **SSO Single Sign-On**: Supports GitHub OAuth and standard OIDC protocol for seamless integration with enterprise identity providers to achieve unified login. -* **Unified Observability**: Aggregates node request metrics, real-time access log details, host and Nginx resource snapshots, health events, and network fluctuation replenishment buffers. - -## Interface Preview - -### Dashboard Overview - -![OpenFlare dashboard overview](./docs/assets/readme/dashboard-overview.png) - -### Access Logs - -![OpenFlare version release](./docs/assets/readme/domain_overview.png) - -### WAF Protection - -![OpenFlare version release](./docs/assets/readme/waf.png) - -## Quick Start - -### Hardware Configuration Recommendations - -| Component | Minimum Hardware Requirements | Recommended Hardware Requirements | Notes | -|------------------------|-----------------------------------|-----------------------------------|-------| -| **Server Control Plane** | 1 CPU core / 2 GB RAM / 20 GB disk | 2 CPU cores / 4 GB RAM / 50 GB+ disk | Disk usage should be expanded reasonably based on access log retention duration and concurrent traffic | -| **Agent Data Plane** | 1 CPU core / 512 MB RAM / 2 GB disk | 2 CPU cores / 2 GB RAM / 10 GB+ disk | Expanded based on OpenResty concurrent proxy connections and WAF interception processing | -| **Relay Relay Node** | 1 CPU core / 1 GB RAM / 5 GB disk | 2 CPU cores / 2 GB RAM / 20 GB disk | frps transmission relay throughput is mainly limited by bandwidth and CPU throughput | -| **OpenFlared Client** | 1 CPU core / 256 MB RAM / 1 GB disk | 1 CPU core / 512 MB RAM / 5 GB disk | Runs independently on the internal network with extremely low resource consumption; only network throughput needs to be guaranteed | - -### 1. Start the Server - -Use `docker-compose`: - -```bash -# Download environment variable template and create .env file -curl -o .env.example https://raw.githubusercontent.com/Rain-kl/OpenFlare/refs/heads/main/.env.example -cp .env.example .env -``` - -```yaml -services: - openflare: - image: ghcr.io/rain-kl/openflare:latest - restart: unless-stopped - env_file: .env - environment: - TZ: ${TZ:-Asia/Shanghai} - ports: - - "3000:3000" - volumes: - - openflare_uploads:/app/uploads - depends_on: - postgres: - condition: service_healthy - redis: - condition: service_healthy - - postgres: - image: postgres:17-alpine - restart: unless-stopped - environment: - POSTGRES_DB: ${DB_NAME:-openflare} - POSTGRES_USER: ${DB_USERNAME:-openflare} - POSTGRES_PASSWORD: ${DB_PASSWORD:-replace-with-strong-password} - volumes: - - openflare_postgres_data:/var/lib/postgresql/data - healthcheck: - test: ["CMD-SHELL", "pg_isready -U ${DB_USERNAME:-openflare} -d ${DB_NAME:-openflare}"] - interval: 10s - timeout: 5s - retries: 5 - - redis: - image: valkey/valkey:8.0-alpine - restart: unless-stopped - command: ["valkey-server", "--appendonly", "yes"] - volumes: - - openflare_redis_data:/data - healthcheck: - test: ["CMD", "valkey-cli", "ping"] - interval: 10s - timeout: 5s - retries: 5 - start_period: 5s - -volumes: - openflare_uploads: - openflare_postgres_data: - openflare_redis_data: -``` - -See the [deployment documentation](https://open-flare.pages.dev/deployment/deployment) for details. - -Access address: `http://localhost:3000` - -Default account: - -* Username: `admin` -* Password: `12345678` - -### 2. Install Agent - -Before installing the Agent, first install OpenResty on the node or use the built-in OpenResty Agent Docker image. - -You can copy the installation command from the control panel's **Nodes Management -> Details -> Node Information -> Node ID and Deployment**, or use the script below: - -#### Docker Deployment - -Docker deployment can directly run the Agent image: - -```bash -docker pull ghcr.io/rain-kl/openflare-agent:latest -docker rm -f openflare-agent 2>/dev/null || true -docker run -d --name openflare-agent --restart unless-stopped \ - -p 80:80 -p 443:443/tcp -p 443:443/udp \ - -v openflare-agent-pages:/data/var/lib/openflare/pages \ - -e OPENFLARE_SERVER_URL=http://your-server:3000 \ - -e OPENFLARE_AGENT_TOKEN=YOUR_AGENT_TOKEN \ - ghcr.io/rain-kl/openflare-agent:latest -``` - -## Open Source License - -This project is licensed under the [Apache License 2.0](./LICENSE). - -## Star History - - - - - - Star History Chart - - diff --git a/README.md b/README.md index c92e6a48..1647adcf 100644 --- a/README.md +++ b/README.md @@ -2,9 +2,9 @@ # OpenFlare -**[📖 中文](./README.md) | [English](./README.en.md)** +**[English](./README.md) | [简体中文](./README.zh-CN.md)** -OpenFlare 是开源 CDN 编排与边缘安全平台。它支持反向代理、集中式配置同步、内网穿透(Tunnels)、动态 WAF 防护以及防 CC 挑战。 +OpenFlare is an open-source CDN orchestration and edge security platform. It supports reverse proxy, centralized configuration synchronization, in-network tunneling (Tunnels), dynamic WAF protection, and CC defense challenges. @@ -21,64 +21,64 @@ OpenFlare 是开源 CDN 编排与边缘安全平台。它支持反向代理、

> [!WARNING] -> 使用 `admin` 用户初次登录系统后,务必修改默认密码 `12345678`。 +> After the first login with the `admin` user, you must change the default password `12345678`. > -> BETA 版本为开发测试阶段的临时产物,可能存在未知问题,请勿在生产环境使用。 +> The BETA version is a temporary product in the development and testing stage and may have unknown issues. It should not be used in production environments. -## 文档 +## Documentation -**https://open-flare.pages.dev** +**https://openflare.fyrn.link** -常用入口: +Common entry points: -* [快速开始](https://open-flare.pages.dev/guide/quick-start) -* [部署说明](https://open-flare.pages.dev/deployment/deployment) -* [配置项参考](https://open-flare.pages.dev/reference/configuration) -* [系统设计](https://open-flare.pages.dev/design/) +* [Quick Start](https://openflare.fyrn.link/guide/quick-start) +* [Deployment Guide](https://openflare.fyrn.link/deployment/deployment) +* [Configuration Reference](https://openflare.fyrn.link/reference/configuration) +* [System Design](https://openflare.fyrn.link/design/) -## 核心能力 +## Core Capabilities -* **反代配置管理**:以网站规则为聚合边界,支持多域名绑定与多上游负载均衡,统一管理所有 OpenResty 节点的反代配置。 -* **安全内网穿透(Tunnels)**:开源版的 Cloudflare Tunnels。无须公网 IP 或暴露入向端口,通过 Relay 中继节点与 OpenFlared 客户端安全反向穿透内网 Web 服务至公网。 -* **边缘 WAF 安全防护**:提供全局与自定义规则组,支持手动/自动/订阅型 IP 组、MaxMind GeoIP 国家级地域准入、IP 组成员 Checksum 差分同步(无需 Nginx 重载)以及自定义拦截响应。 -* **防 CC 与人机挑战(PoW)**:内置高性能客户端密码学 Proof of Work 挑战(类似 Turnstile),在网关边缘秒级拦截并阻断僵尸网络与爬虫。 -* **Pages 静态托管**:支持上传或从受限 Remote URL、公开 GitHub Release asset 同步预构建产物;GitHub latest 可定时检查并可选自动发布。所有来源统一生成不可变部署,由边缘 Agent 拉取并通过 OpenResty 本地提供服务,支持回滚、SPA Fallback 与 API 反向代理。 -* **TLS 证书自动化**:支持证书动态上传、多域名证书自动匹配绑定,以及通过 ACME 协议向 Let's Encrypt 自动申请与续期证书。 -* **Uptime Kuma 监控同步**:与 Uptime Kuma 集成,自动差分同步监控站点列表,实时感知节点存活与服务可用状态。 -* **SSO 单点登录**:支持 GitHub OAuth 与标准 OIDC 协议,无缝接入企业身份提供商实现统一登录。 -* **统一观测**:聚合节点请求指标、实时访问日志明细、宿主机与 Nginx 资源快照、健康事件以及网络波动补传缓冲。 +* **Reverse Proxy Configuration Management**: Uses website rules as the aggregation boundary, supports multi-domain binding and multi-upstream load balancing, and centrally manages reverse proxy configurations for all OpenResty nodes. +* **Secure In-Network Tunneling (Tunnels)**: Open-source version of Cloudflare Tunnels. No public IP or exposed inbound ports are required. Securely reverse-proxy internal web services to the public internet through Relay relay nodes and OpenFlared clients. +* **Edge WAF Security Protection**: Provides global and custom rule groups, supports manual/auto/subscription-type IP groups, MaxMind GeoIP national-level geographic access control, IP group member Checksum differential synchronization (no Nginx reload required), and custom blocking responses. +* **CC Defense and Human-Computer Challenge (PoW)**: Built-in high-performance client-side cryptography Proof of Work challenge (similar to Turnstile). Secures high-speed interception and blocking of zombie networks and crawlers at the gateway edge. +* **Pages Static Hosting**: Supports uploading or synchronizing pre-built artifacts from restricted Remote URLs or public GitHub Release assets. GitHub latest can be checked periodically and optionally auto-published. All sources are unified to generate immutable deployments, pulled by the edge Agent and served locally by OpenResty, supporting rollbacks, SPA Fallback, and API reverse proxy. +* **TLS Certificate Automation**: Supports dynamic certificate uploads, automatic multi-domain certificate matching and binding, and automatic issuance and renewal of certificates from Let's Encrypt via the ACME protocol. +* **Uptime Kuma Monitoring Synchronization**: Integrated with Uptime Kuma to automatically perform differential synchronization of monitoring site lists, real-time awareness of node availability and service status. +* **SSO Single Sign-On**: Supports GitHub OAuth and standard OIDC protocol for seamless integration with enterprise identity providers to achieve unified login. +* **Unified Observability**: Aggregates node request metrics, real-time access log details, host and Nginx resource snapshots, health events, and network fluctuation replenishment buffers. -## 界面预览 +## Interface Preview -### 仪表盘总览 +### Dashboard Overview ![OpenFlare dashboard overview](./docs/assets/readme/dashboard-overview.png) -### 访问日志 +### Access Logs ![OpenFlare version release](./docs/assets/readme/domain_overview.png) -### WAF 防护 +### WAF Protection ![OpenFlare version release](./docs/assets/readme/waf.png) -## 快速开始 +## Quick Start -### 硬件配置推荐 +### Hardware Configuration Recommendations -| 组件 | 最低硬件配额 | 推荐硬件配额 | 说明 | -| --- |-------------------------------| --- | --- | -| **Server 控制面** | 1 核 CPU / 2 GB 内存 / 20 GB 磁盘 | 2 核 CPU / 4 GB 内存 / 50 GB+ 磁盘 | 磁盘用量需根据访问日志留存时长与并发流量合理扩容 | -| **Agent 数据面** | 1 核 CPU / 512 MB 内存 / 2 GB 磁盘 | 2 核 CPU / 2 GB 内存 / 10 GB+ 磁盘 | 根据 OpenResty 的并发代理连接量与 WAF 拦截处理扩容 | -| **Relay 中继节点**| 1 核 CPU / 1 GB 内存 / 5 GB 磁盘 | 2 核 CPU / 2 GB 内存 / 20 GB 磁盘 | frps 传输中继吞吐量主要受带宽与 CPU 吞吐能力限制 | -| **OpenFlared 客户端**| 1 核 CPU / 256 MB 内存 / 1 GB 磁盘 | 1 核 CPU / 512 MB 内存 / 5 GB 磁盘 | 独立运行于内网,自身资源占用极小,保障网络吞吐即可 | +| Component | Minimum Hardware Requirements | Recommended Hardware Requirements | Notes | +|------------------------|-----------------------------------|-----------------------------------|-------| +| **Server Control Plane** | 1 CPU core / 2 GB RAM / 20 GB disk | 2 CPU cores / 4 GB RAM / 50 GB+ disk | Disk usage should be expanded reasonably based on access log retention duration and concurrent traffic | +| **Agent Data Plane** | 1 CPU core / 512 MB RAM / 2 GB disk | 2 CPU cores / 2 GB RAM / 10 GB+ disk | Expanded based on OpenResty concurrent proxy connections and WAF interception processing | +| **Relay Relay Node** | 1 CPU core / 1 GB RAM / 5 GB disk | 2 CPU cores / 2 GB RAM / 20 GB disk | frps transmission relay throughput is mainly limited by bandwidth and CPU throughput | +| **OpenFlared Client** | 1 CPU core / 256 MB RAM / 1 GB disk | 1 CPU core / 512 MB RAM / 5 GB disk | Runs independently on the internal network with extremely low resource consumption; only network throughput needs to be guaranteed | -### 1. 启动 Server +### 1. Start the Server -使用 docker-compose +Use `docker-compose`: ```bash -# 下载环境变量模板并创建 .env 文件 +# Download environment variable template and create .env file curl -o .env.example https://raw.githubusercontent.com/Rain-kl/OpenFlare/refs/heads/main/.env.example cp .env.example .env ``` @@ -135,24 +135,24 @@ volumes: openflare_redis_data: ``` -详细部署说明见 [部署文档](https://open-flare.pages.dev/deployment/deployment)。 +See the [deployment documentation](https://openflare.fyrn.link/deployment/deployment) for details. -访问地址:`http://localhost:3000` +Access address: `http://localhost:3000` -默认账号: +Default account: -* 用户名:`admin` -* 密码:`12345678` +* Username: `admin` +* Password: `12345678` -### 2. 安装 Agent +### 2. Install Agent -安装 Agent 前请先在节点上安装 OpenResty,或改用内置 OpenResty 的 Agent Docker 镜像。 +Before installing the Agent, first install OpenResty on the node or use the built-in OpenResty Agent Docker image. -你可以在控制面板的节点管理->详情->节点信息->节点标识与部署复制安装命令,或直接使用下面的脚本: +You can copy the installation command from the control panel's **Nodes Management -> Details -> Node Information -> Node ID and Deployment**, or use the script below: -#### Docker 部署 +#### Docker Deployment -Docker 部署可直接运行 Agent 镜像: +Docker deployment can directly run the Agent image: ```bash docker pull ghcr.io/rain-kl/openflare-agent:latest @@ -165,9 +165,9 @@ docker run -d --name openflare-agent --restart unless-stopped \ ghcr.io/rain-kl/openflare-agent:latest ``` -## 开源协议 +## Open Source License -本项目采用 [Apache License 2.0](./LICENSE) 开源。 +This project is licensed under the [Apache License 2.0](./LICENSE). ## Star History diff --git a/README.zh-CN.md b/README.zh-CN.md new file mode 100644 index 00000000..365dca96 --- /dev/null +++ b/README.zh-CN.md @@ -0,0 +1,180 @@ +
+ +# OpenFlare + +**[English](./README.md) | [简体中文](./README.zh-CN.md)** + +OpenFlare 是开源 CDN 编排与边缘安全平台。它支持反向代理、集中式配置同步、内网穿透(Tunnels)、动态 WAF 防护以及防 CC 挑战。 + +
+ +

+ + license + + + release + + + ghcr + +

+ +> [!WARNING] +> 使用 `admin` 用户初次登录系统后,务必修改默认密码 `12345678`。 +> +> BETA 版本为开发测试阶段的临时产物,可能存在未知问题,请勿在生产环境使用。 + +## 文档 + +**https://openflare.fyrn.link** + +常用入口: + +* [快速开始](https://openflare.fyrn.link/guide/quick-start) +* [部署说明](https://openflare.fyrn.link/deployment/deployment) +* [配置项参考](https://openflare.fyrn.link/reference/configuration) +* [系统设计](https://openflare.fyrn.link/design/) + +## 核心能力 + +* **反代配置管理**:以网站规则为聚合边界,支持多域名绑定与多上游负载均衡,统一管理所有 OpenResty 节点的反代配置。 +* **安全内网穿透(Tunnels)**:开源版的 Cloudflare Tunnels。无须公网 IP 或暴露入向端口,通过 Relay 中继节点与 OpenFlared 客户端安全反向穿透内网 Web 服务至公网。 +* **边缘 WAF 安全防护**:提供全局与自定义规则组,支持手动/自动/订阅型 IP 组、MaxMind GeoIP 国家级地域准入、IP 组成员 Checksum 差分同步(无需 Nginx 重载)以及自定义拦截响应。 +* **防 CC 与人机挑战(PoW)**:内置高性能客户端密码学 Proof of Work 挑战(类似 Turnstile),在网关边缘秒级拦截并阻断僵尸网络与爬虫。 +* **Pages 静态托管**:支持上传或从受限 Remote URL、公开 GitHub Release asset 同步预构建产物;GitHub latest 可定时检查并可选自动发布。所有来源统一生成不可变部署,由边缘 Agent 拉取并通过 OpenResty 本地提供服务,支持回滚、SPA Fallback 与 API 反向代理。 +* **TLS 证书自动化**:支持证书动态上传、多域名证书自动匹配绑定,以及通过 ACME 协议向 Let's Encrypt 自动申请与续期证书。 +* **Uptime Kuma 监控同步**:与 Uptime Kuma 集成,自动差分同步监控站点列表,实时感知节点存活与服务可用状态。 +* **SSO 单点登录**:支持 GitHub OAuth 与标准 OIDC 协议,无缝接入企业身份提供商实现统一登录。 +* **统一观测**:聚合节点请求指标、实时访问日志明细、宿主机与 Nginx 资源快照、健康事件以及网络波动补传缓冲。 + +## 界面预览 + +### 仪表盘总览 + +![OpenFlare dashboard overview](./docs/assets/readme/dashboard-overview.png) + +### 访问日志 + +![OpenFlare version release](./docs/assets/readme/domain_overview.png) + +### WAF 防护 + +![OpenFlare version release](./docs/assets/readme/waf.png) + +## 快速开始 + +### 硬件配置推荐 + +| 组件 | 最低硬件配额 | 推荐硬件配额 | 说明 | +| --- |-------------------------------| --- | --- | +| **Server 控制面** | 1 核 CPU / 2 GB 内存 / 20 GB 磁盘 | 2 核 CPU / 4 GB 内存 / 50 GB+ 磁盘 | 磁盘用量需根据访问日志留存时长与并发流量合理扩容 | +| **Agent 数据面** | 1 核 CPU / 512 MB 内存 / 2 GB 磁盘 | 2 核 CPU / 2 GB 内存 / 10 GB+ 磁盘 | 根据 OpenResty 的并发代理连接量与 WAF 拦截处理扩容 | +| **Relay 中继节点**| 1 核 CPU / 1 GB 内存 / 5 GB 磁盘 | 2 核 CPU / 2 GB 内存 / 20 GB 磁盘 | frps 传输中继吞吐量主要受带宽与 CPU 吞吐能力限制 | +| **OpenFlared 客户端**| 1 核 CPU / 256 MB 内存 / 1 GB 磁盘 | 1 核 CPU / 512 MB 内存 / 5 GB 磁盘 | 独立运行于内网,自身资源占用极小,保障网络吞吐即可 | + +### 1. 启动 Server + +使用 docker-compose + +```bash +# 下载环境变量模板并创建 .env 文件 +curl -o .env.example https://raw.githubusercontent.com/Rain-kl/OpenFlare/refs/heads/main/.env.example +cp .env.example .env +``` + +```yaml +services: + openflare: + image: ghcr.io/rain-kl/openflare:latest + restart: unless-stopped + env_file: .env + environment: + TZ: ${TZ:-Asia/Shanghai} + ports: + - "3000:3000" + volumes: + - openflare_uploads:/app/uploads + depends_on: + postgres: + condition: service_healthy + redis: + condition: service_healthy + + postgres: + image: postgres:17-alpine + restart: unless-stopped + environment: + POSTGRES_DB: ${DB_NAME:-openflare} + POSTGRES_USER: ${DB_USERNAME:-openflare} + POSTGRES_PASSWORD: ${DB_PASSWORD:-replace-with-strong-password} + volumes: + - openflare_postgres_data:/var/lib/postgresql/data + healthcheck: + test: ["CMD-SHELL", "pg_isready -U ${DB_USERNAME:-openflare} -d ${DB_NAME:-openflare}"] + interval: 10s + timeout: 5s + retries: 5 + + redis: + image: valkey/valkey:8.0-alpine + restart: unless-stopped + command: ["valkey-server", "--appendonly", "yes"] + volumes: + - openflare_redis_data:/data + healthcheck: + test: ["CMD", "valkey-cli", "ping"] + interval: 10s + timeout: 5s + retries: 5 + start_period: 5s + +volumes: + openflare_uploads: + openflare_postgres_data: + openflare_redis_data: +``` + +详细部署说明见 [部署文档](https://openflare.fyrn.link/deployment/deployment)。 + +访问地址:`http://localhost:3000` + +默认账号: + +* 用户名:`admin` +* 密码:`12345678` + +### 2. 安装 Agent + +安装 Agent 前请先在节点上安装 OpenResty,或改用内置 OpenResty 的 Agent Docker 镜像。 + +你可以在控制面板的节点管理->详情->节点信息->节点标识与部署复制安装命令,或直接使用下面的脚本: + +#### Docker 部署 + +Docker 部署可直接运行 Agent 镜像: + +```bash +docker pull ghcr.io/rain-kl/openflare-agent:latest +docker rm -f openflare-agent 2>/dev/null || true +docker run -d --name openflare-agent --restart unless-stopped \ + -p 80:80 -p 443:443/tcp -p 443:443/udp \ + -v openflare-agent-pages:/data/var/lib/openflare/pages \ + -e OPENFLARE_SERVER_URL=http://your-server:3000 \ + -e OPENFLARE_AGENT_TOKEN=YOUR_AGENT_TOKEN \ + ghcr.io/rain-kl/openflare-agent:latest +``` + +## 开源协议 + +本项目采用 [Apache License 2.0](./LICENSE) 开源。 + +## Star History + + + + + + Star History Chart + + diff --git a/docs/en/changelog/index.md b/docs/en/changelog/index.md new file mode 100644 index 00000000..9ba486ec --- /dev/null +++ b/docs/en/changelog/index.md @@ -0,0 +1,5 @@ +# Changelog + +The changelog is maintained in Simplified Chinese in this repository. See the Chinese version: + +* [更新日志(简体中文)](../changelog/) diff --git a/docs/en/config.ts b/docs/en/config.ts new file mode 100644 index 00000000..49e5093e --- /dev/null +++ b/docs/en/config.ts @@ -0,0 +1,147 @@ +import { defineAdditionalConfig, type DefaultTheme } from 'vitepress' + +export default defineAdditionalConfig({ + description: + 'OpenFlare is a lightweight, self-hosted OpenResty control plane for managing reverse proxy rules, configuration publishing, node synchronization, TLS certificates, and basic observability.', + + themeConfig: { + nav: nav(), + + sidebar: { + '/en/guide/': { base: '/en/guide/', items: sidebarGuide() }, + '/en/reference/': { base: '/en/reference/', items: sidebarReference() }, + '/en/deployment/': { base: '/en/deployment/', items: sidebarDeployment() }, + '/en/design/': { base: '/en/design/', items: sidebarDesign() }, + '/en/changelog/': { base: '/en/changelog/', items: [] } + }, + + editLink: { + pattern: 'https://github.com/Rain-kl/OpenFlare/edit/main/docs/:path', + text: 'Edit this page on GitHub' + }, + + footer: { + message: 'Released under the Apache License 2.0', + copyright: 'Copyright © OpenFlare contributors' + }, + + docFooter: { + prev: 'Previous Page', + next: 'Next Page' + }, + + outline: { + label: 'On this page' + }, + + lastUpdated: { + text: 'Last updated at' + }, + + notFound: { + title: 'Page Not Found', + quote: 'This document does not have a corresponding page yet.', + linkLabel: 'Go to Home', + linkText: 'Back to OpenFlare Docs' + }, + + langMenuLabel: 'Language', + returnToTopLabel: 'Back to top', + sidebarMenuLabel: 'Menu', + darkModeSwitchLabel: 'Theme', + lightModeSwitchTitle: 'Switch to light theme', + darkModeSwitchTitle: 'Switch to dark theme', + skipToContentLabel: 'Skip to content' + } +}) + +function nav(): DefaultTheme.NavItem[] { + return [ + { text: 'Guide', link: '/en/guide/', activeMatch: '/en/guide/' }, + { text: 'Deployment', link: '/en/deployment/', activeMatch: '/en/deployment/' }, + { text: 'Reference', link: '/en/reference/', activeMatch: '/en/reference/' }, + { text: 'Design', link: '/en/design/', activeMatch: '/en/design/' }, + { text: 'Changelog', link: '/en/changelog/', activeMatch: '/en/changelog/' } + ] +} + +function sidebarGuide(): DefaultTheme.SidebarItem[] { + return [ + { + text: 'Guide', + items: [ + { text: 'Overview', link: '' }, + { text: 'Quick Start', link: 'quick-start' }, + { text: 'TLS Certificates & Auto-Renewal', link: 'certificates' }, + { text: 'Zone Domain Migration', link: 'zone-domain-migration' }, + { text: 'Create a Reverse Proxy Config', link: 'proxy-config' }, + { text: 'Pages Static Hosting Usage', link: 'pages-usage' }, + { text: 'Tunnel & Intranet Penetration', link: 'tunnel-usage' }, + { text: 'WAF Security Protection', link: 'waf-usage' }, + { text: 'WAF Auto IP Group Expressions', link: 'waf-ip-group-expr' }, + { text: 'Uptime Kuma Monitoring Sync', link: 'uptime-kuma' }, + { text: 'SSO Login Configuration', link: 'sso' }, + { text: 'Publish First Configuration', link: 'first-site' }, + { text: 'Troubleshooting', link: 'troubleshooting' }, + { text: 'Credits', link: 'credits' } + ] + } + ] +} + +function sidebarReference(): DefaultTheme.SidebarItem[] { + return [ + { + text: 'Reference', + items: [ + { text: 'Overview', link: '' }, + { text: 'Configuration Options', link: 'configuration' }, + { text: 'CLI Commands', link: 'cli' } + ] + } + ] +} + +function sidebarDeployment(): DefaultTheme.SidebarItem[] { + return [ + { + text: 'Deployment', + items: [ + { text: 'Overview', link: '' }, + { text: 'Deployment Guide', link: 'deployment' }, + { text: 'Start the Server', link: 'server' }, + { text: 'Access Agent', link: 'agent' }, + { text: 'Deploy Relay (Tunnel)', link: 'relay' }, + { text: 'Deploy OpenFlared', link: 'openflared' }, + { text: 'Upgrade & Maintenance', link: 'upgrade' } + ] + } + ] +} + +function sidebarDesign(): DefaultTheme.SidebarItem[] { + return [ + { + text: 'Design', + items: [ + { text: 'Product Boundaries', link: '' }, + { text: 'System Architecture', link: 'architecture' }, + { text: 'Zone & Domain Resource Design', link: 'zone-design' }, + { text: 'Cloudflare DNS Pointing Design', link: 'cloudflare-pointing' }, + { text: 'Agent & Publish Model', link: 'agent-design' }, + { text: 'Tunnel & Intranet Penetration', link: 'tunnel-design' }, + { text: 'WAF Design', link: 'waf-design' }, + { text: 'WAF Orchestration Rule Design', link: 'waf-orchestration-design' }, + { text: 'Pages Static Hosting Design', link: 'pages-design' }, + { text: 'Edge Cache Strategy Design', link: 'edge-cache-design' }, + { text: 'Origin Error Page Design', link: 'origin-error-page' }, + { text: 'Edge Observability & Traffic Stats', link: 'observability-design' }, + { text: 'Observability Transport Model', link: 'observability-transport-model' }, + { text: 'Observability Protocol & Tables', link: 'observability-data-model' }, + { text: 'Log Store Decoupling', link: 'logstore' }, + { text: 'Uptime Kuma Sync Design', link: 'kuma-design' }, + { text: 'Login CAPTCHA Design', link: 'login-captcha' } + ] + } + ] +} diff --git a/docs/en/deployment/agent.md b/docs/en/deployment/agent.md new file mode 100644 index 00000000..fc5c7408 --- /dev/null +++ b/docs/en/deployment/agent.md @@ -0,0 +1,176 @@ +# Access Agent + +You will learn: The responsibilities of the Agent, the difference between the two access Tokens, installation script parameters, `agent.json` settings, and how to verify that the node has successfully connected. + +The OpenFlare Agent runs on the proxy node. It does not receive arbitrary remote shell commands; instead, it pulls the configuration version published by the control plane via the Agent API, writes files for OpenResty locally, executes configuration validation, reloads, and attempts to roll back to a working configuration if it fails. + +## Connection Credentials + +| Method | Applicable Scenario | +| --- | --- | +| `discovery_token` | Automatically registers a node for the first time, which the Server exchanges for a node-specific credential | +| `agent_token` | Node has already been created/allocated in the management console, directly uses this node-specific credential | + +At least one of `agent_token` or `discovery_token` must be configured. + +### Credential Retrieval Path + +- **`discovery_token` (Auto Registration Token)**: Log into the management console, navigate to "System Settings" -> "Auto Registration", where you can generate, view, and copy the global auto-registration credential. +- **`agent_token` (Node Specific Token)**: Log into the management console, navigate to "Node Management" -> "Add Node", fill in basic node information, save, and copy the node-specific access Token in the node details. + +## One-Click Installation + +Using the `discovery_token`: + +```bash +curl -fsSL https://raw.githubusercontent.com/Rain-kl/OpenFlare/main/scripts/install-agent.sh | bash -s -- \ + --server-url http://your-server:3000 \ + --discovery-token YOUR_DISCOVERY_TOKEN +``` + +Using the node-specific `agent_token`: + +```bash +curl -fsSL https://raw.githubusercontent.com/Rain-kl/OpenFlare/main/scripts/install-agent.sh | bash -s -- \ + --server-url http://your-server:3000 \ + --agent-token YOUR_AGENT_TOKEN +``` + +The installation script downloads the latest Agent, writes to `/opt/openflare-agent` by default, generates `agent.json`, and registers `openflare-agent.service` on Linux + systemd environments. + +Supported arguments: + +| Argument | Description | Default Value | +| --- | --- | --- | +| `--server-url` | Server address (required) | | +| `--discovery-token` | One-time auto-registration Token | | +| `--agent-token` | Node-specific Token | | +| `--install-dir` | Target installation directory | `/opt/openflare-agent` | +| `--openresty-path` | Path to the OpenResty binary; automatically detects `openresty` if unspecified | | +| `--repo` | GitHub repository to download from | `Rain-kl/OpenFlare` | +| `--no-service` | Do not register systemd service | | + +## Configuration File + +Default configuration file path: + +```text +/opt/openflare-agent/agent.json +``` + +Example local configuration: + +```json +{ + "server_url": "http://127.0.0.1:3000", + "agent_token": "replace-with-node-auth-token", + "data_dir": "./data", + "openresty_path": "openresty", + "openresty_observability_port": 18081, + "observability_replay_minutes": 15, + "heartbeat_interval": 10000, + "request_timeout": 10000 +} +``` + +Example customized OpenResty paths configuration: + +```json +{ + "server_url": "http://127.0.0.1:3000", + "agent_token": "replace-with-node-auth-token", + "data_dir": "/var/lib/openflare-agent", + "openresty_path": "/usr/local/openresty/nginx/sbin/openresty", + "main_config_path": "/var/lib/openflare-agent/etc/nginx/nginx.conf", + "route_config_path": "/var/lib/openflare-agent/etc/nginx/conf.d/openflare_routes.conf", + "access_log_path": "/var/lib/openflare-agent/var/log/openflare/access.log", + "cert_dir": "/var/lib/openflare-agent/etc/nginx/certs", + "lua_dir": "/var/lib/openflare-agent/etc/nginx/lua", + "runtime_config_dir": "/var/lib/openflare-agent/etc/openflare", + "heartbeat_interval": 10000, + "request_timeout": 10000 +} +``` + +If `openresty_path` is not configured, the Agent calls `openresty` by default. For the full fields, see [Configurations Reference](../reference/configuration.md#agent-configurations-fields). + +## Running in Docker + +For Docker deployments, run the Agent image containing built-in OpenResty directly: + +```bash +docker pull ghcr.io/rain-kl/openflare-agent:latest +docker rm -f openflare-agent 2>/dev/null || true +docker run -d --name openflare-agent --restart unless-stopped \ + -p 80:80 -p 443:443 \ + -e OPENFLARE_SERVER_URL=http://your-server:3000 \ + -e OPENFLARE_AGENT_TOKEN=YOUR_AGENT_TOKEN \ + ghcr.io/rain-kl/openflare-agent:latest +``` + +## Start & Validate + +In a systemd environment: + +```bash +systemctl start openflare-agent +systemctl status openflare-agent +journalctl -u openflare-agent -f +``` + +Manual execution: + +```bash +/opt/openflare-agent/openflare-agent -config /opt/openflare-agent/agent.json +``` + +Running from source: + +```bash +cd openflare-agent +export LOG_LEVEL='info' +go run ./cmd/agent -config /path/to/agent.json +``` + +Running compiled binary: + +```bash +cd openflare-agent +go build -o openflare-agent ./cmd/agent +export LOG_LEVEL='info' +./openflare-agent -config /path/to/agent.json +``` + +Confirm in the management console: + +| Position | Expected Result | +| --- | --- | +| Node List | Node status is online | +| Node Details | Heartbeat, current version, and basic resource metrics display correctly | +| Apply Logs | Application result displays after publishing | + +## Uninstall + +To completely uninstall the Agent and wipe local data: + +```bash +curl -fsSL https://raw.githubusercontent.com/Rain-kl/OpenFlare/main/scripts/uninstall-agent.sh | bash +``` + +Supported arguments: + +| Argument | Description | Default Value | +| --- | --- | --- | +| `--install-dir` | Installation directory | `/opt/openflare-agent` | +| `--service-name` | systemd service name | `openflare-agent` | + +The uninstallation script only removes the Agent service, processes, and installation directory; it does not uninstall OpenResty from the host. + +## Common Questions + +| Symptom | Actions | +| --- | --- | +| `agent_token and discovery_token cannot both be empty` | Check if at least one Token is configured in `agent.json` | +| Node stays offline | Run `curl -I http://your-server:3000` on the Agent node to verify that the Server is reachable | +| OpenResty is not running | Review `journalctl -u openflare-agent`, checking that `openresty_path` is executable and ports 80/443 are not bound | +| Repeated application failures after publishing | The Agent blocks repeated sync attempts of the same failing `version + checksum`; fix the configuration and republish, or activate an older version to roll back | diff --git a/docs/en/deployment/deployment.md b/docs/en/deployment/deployment.md new file mode 100644 index 00000000..f17e9834 --- /dev/null +++ b/docs/en/deployment/deployment.md @@ -0,0 +1,287 @@ +# Deployment Guide + +You will learn: The recommended deployment strategies for OpenFlare, the system requirements for Server and Agent, how to run from source, integration steps, upgrades, and uninstallation entrypoints. + +In production environments, we highly recommend using PostgreSQL as the Server database and explicitly configuring `SESSION_SECRET` for the Server. The recommended Agent deployment method is Docker (which runs the Agent image containing built-in OpenResty); host systemd service installation via script and manual local run are also supported. + +## Deployment Topology + +### Standard Reverse Proxy Traffic Path + +```text +Browser + | + v +OpenFlare Server :3000 + | + | Agent API / heartbeat / config pull + v +OpenFlare Agent + | + v +OpenResty binary + | + v +Origin service +``` + +### Intranet Penetration Traffic Path + +```text +Browser + | + v +OpenResty (Agent, WAF/HTTPS Termination) <-- TunnelRelay Node + | + | proxy_pass (127.0.0.1:{vhost_port}) + v +OpenFlareRelay (frps process) <-- TunnelRelay Node + | + | frp tunnel protocol + v +OpenFlared (frpc client) <-- Intranet Server + | + v +Internal Service (192.168.x.x) +``` + +## Prerequisites + +Server: + +| Item | Requirement | +| --- | --- | +| Go | `1.25+`, required only when running from source | +| Node.js | `18+`, required only when building the admin frontend from source | +| Database | Writable SQLite parent directory, or a reachable PostgreSQL instance | +| Port | Listens on port `3000` by default | + +Agent: + +| Item | Requirement | +| --- | --- | +| System | The installation script supports Linux and macOS; the systemd service is created only on Linux + systemd environments | +| Architecture | `amd64` or `arm64` | +| OpenResty | Required to have the `openresty` executable when deploying locally, or specify its path via `--openresty-path` | +| Docker | Required only when deploying the Agent via Docker image | +| Network | The Agent node must be able to reach the Server address | +| GeoIP | WAF regional rules rely on the Agent's local MaxMind mmdb; the Agent initializes a built-in library on startup and updates it periodically | + +### Hardware Allocation Recommendations + +| Component | Minimum Allocation | Recommended Allocation | Note | +| --- | --- | --- | --- | +| **Server Control Plane** | 1 Core CPU / 1 GB RAM / 10 GB Disk | 2 Cores CPU / 4 GB RAM / 50 GB+ Disk | Expand disk allocation according to log retention windows and concurrency. | +| **Agent Data Plane** | 1 Core CPU / 512 MB RAM / 2 GB Disk | 2 Cores CPU / 2 GB RAM / 10 GB+ Disk | Expand according to concurrent reverse proxy connections and WAF workloads. | +| **Relay Node** | 1 Core CPU / 1 GB RAM / 5 GB Disk | 2 Cores CPU / 2 GB RAM / 20 GB Disk | frps throughput is primarily bounded by CPU processing capacity and bandwidth. | +| **OpenFlared Client** | 1 Core CPU / 256 MB RAM / 1 GB Disk | 1 Core CPU / 512 MB RAM / 5 GB Disk | Runs inside the intranet; utilizes minimal CPU/RAM, optimize for network throughput. | + +## Docker Compose Deployment for Server + +Create a `docker-compose.yml` file: + +```yaml +services: + postgres: + image: postgres:17-alpine + restart: unless-stopped + environment: + POSTGRES_DB: openflare + POSTGRES_USER: openflare + POSTGRES_PASSWORD: replace-with-strong-password + volumes: + - postgres-data:/var/lib/postgresql/data + healthcheck: + test: ["CMD-SHELL", "pg_isready -U openflare -d openflare"] + interval: 10s + timeout: 5s + retries: 5 + + openflare: + image: ghcr.io/rain-kl/openflare:latest + container_name: openflare + restart: unless-stopped + depends_on: + postgres: + condition: service_healthy + ports: + - "3000:3000" + environment: + SESSION_SECRET: replace-with-a-long-random-string + DSN: postgres://openflare:replace-with-strong-password@postgres:5432/openflare?sslmode=disable + GIN_MODE: release + LOG_LEVEL: info + volumes: + - openflare-data:/data + +volumes: + postgres-data: + openflare-data: +``` + +Start the Server: + +```bash +docker compose up -d +docker compose ps +docker compose logs -f openflare +``` + +Access `http://localhost:3000` for the first time, using the default credentials `root` / `123456`. Please change the default password immediately after logging in. + +## Start Server from Source + +First, build the admin frontend: + +```bash +cd openflare-server/web +corepack enable +pnpm install +pnpm build +``` + +Then, launch the Server: + +```bash +cd openflare-server +export SESSION_SECRET='replace-with-a-long-random-string' +export SQLITE_PATH='./openflare.db' +export LOG_LEVEL='info' +# Optional: Prefer PostgreSQL by setting DSN +# export DSN='postgres://openflare:secret@127.0.0.1:5432/openflare?sslmode=disable' +go run . +``` + +By default, the Server listens on port `3000`. You can also specify it explicitly: + +```bash +go run . --port 3000 --log-dir ./logs +``` + +## Running Agent in Docker (Recommended) + +Docker is the recommended deployment method for the Agent. Running the Agent image directly launches the Agent controller alongside the built-in OpenResty binary. If `node_ip` is left blank, the Agent automatically resolves its outbound public IP via third-party APIs, avoiding registering the Docker bridge address as the node IP. + +Mounting the configuration file: + +```bash +docker pull ghcr.io/rain-kl/openflare-agent:latest +docker rm -f openflare-agent 2>/dev/null || true +docker run -d --name openflare-agent --restart unless-stopped \ + -p 80:80 -p 443:443 \ + -v openflare-agent-data:/data \ + -v ./agent.json:/etc/openflare/agent.json:ro \ + ghcr.io/rain-kl/openflare-agent:latest +``` + +Using environment variables: + +```bash +docker pull ghcr.io/rain-kl/openflare-agent:latest +docker rm -f openflare-agent 2>/dev/null || true +docker run -d --name openflare-agent --restart unless-stopped \ + -p 80:80 -p 443:443 \ + -e OPENFLARE_SERVER_URL=http://your-server:3000 \ + -e OPENFLARE_AGENT_TOKEN=YOUR_AGENT_TOKEN \ + ghcr.io/rain-kl/openflare-agent:latest +``` + +## Agent Connection via Installation Script + +Apart from Docker, you can deploy the Agent directly on a Linux/macOS host using the installation script. + +Auto-register using `discovery_token`: + +```bash +curl -fsSL https://raw.githubusercontent.com/Rain-kl/OpenFlare/main/scripts/install-agent.sh | bash -s -- \ + --server-url http://your-server:3000 \ + --discovery-token YOUR_DISCOVERY_TOKEN +``` + +Connect using node-specific `agent_token`: + +```bash +curl -fsSL https://raw.githubusercontent.com/Rain-kl/OpenFlare/main/scripts/install-agent.sh | bash -s -- \ + --server-url http://your-server:3000 \ + --agent-token YOUR_AGENT_TOKEN +``` + +Installation script arguments: + +| Argument | Description | Default Value | +| --- | --- | --- | +| `--server-url` | Server address (required) | | +| `--discovery-token` | Auto-registration Token; mutually exclusive with `--agent-token` | | +| `--agent-token` | Node-specific Token; mutually exclusive with `--discovery-token` | | +| `--install-dir` | Target installation directory | `/opt/openflare-agent` | +| `--openresty-path` | Path to the OpenResty binary; automatically detects `openresty` if unspecified | | +| `--repo` | GitHub repository to download from | `Rain-kl/OpenFlare` | +| `--no-service` | Do not register systemd service | | + +Confirm service status: + +```bash +systemctl status openflare-agent +journalctl -u openflare-agent -f +``` + +## Running the Agent Manually + +Running from source: + +```bash +cd openflare-agent +export LOG_LEVEL='info' +go run ./cmd/agent -config /path/to/agent.json +``` + +Running compiled binary: + +```bash +cd openflare-agent +go build -o openflare-agent ./cmd/agent +export LOG_LEVEL='info' +./openflare-agent -config /path/to/agent.json +``` + +Minimal `agent.json` example: + +```json +{ + "server_url": "http://127.0.0.1:3000", + "agent_token": "replace-with-node-auth-token", + "data_dir": "./data", + "openresty_path": "openresty", + "heartbeat_interval": 10000, + "request_timeout": 10000 +} +``` + +If `openresty_path` is left blank, the Agent calls `openresty` by default. + +By default, the Agent attempts to upgrade the HTTP heartbeat connection to WebSocket once successfully registered. Once upgraded, configuration activations on the Server notify the Agent instantly; if WebSocket disconnects or fails to establish, the Agent gracefully falls back to HTTP polling. + +WAF geographical filtering depends on the local `GeoLite2-Country.mmdb`. The Agent automatically writes the built-in database to `data_dir/etc/openflare/GeoLite2-Country.mmdb` on startup and checks for periodic updates. Muted warnings are logged if updates fail, having no impact on Nginx configuration sync or reloads. + +## Upgrades & Uninstallation + +Server: + +* Root users can check and trigger Server upgrades in the top header of the management console. +* To deploy preview releases, manually check the GitHub Releases page. +* You can also trigger upgrades by uploading the compiled Server binary in the console. + +Agent: + +* By default, the Agent automatically upgrades following stable releases. +* Agent self-updates require the GitHub Release to contain the compiled binary and a matching `.sha256` checksum file; updates are blocked if the downloaded binary fails the SHA-256 validation. +* You can re-execute the installation script to redeploy or force-update the Agent. +* Upgrading to preview releases requires a manual trigger. + +Uninstalling the Agent: + +```bash +curl -fsSL https://raw.githubusercontent.com/Rain-kl/OpenFlare/main/scripts/uninstall-agent.sh | bash +``` + +The uninstallation script stops the Agent process, removes the systemd service unit, and wipes the installation directory, without uninstalling OpenResty from the host. diff --git a/docs/en/deployment/index.md b/docs/en/deployment/index.md new file mode 100644 index 00000000..570f8f3f --- /dev/null +++ b/docs/en/deployment/index.md @@ -0,0 +1,24 @@ +# Deployment & Upgrade + +This section provides detailed deployment guides, configuration instructions, and upgrade maintenance procedures for the OpenFlare Server, Agent, Relay, and the OpenFlared client. + +## Content Navigation + +### Quick Start +* **[Quick Start](../guide/quick-start.md)**: Start the Server and your first Agent in under 5 minutes using Docker Compose (recommended for new users). + +### Server Deployment +* **[Launch Server](./server.md)**: Learn how to build the frontend from source, start the Server, and choose between SQLite or PostgreSQL. + +### Agent Deployment +* **[Deploy Agent](./agent.md)**: Explore Agent connection methods, Docker deployment, host script installation, config files, and troubleshooting. + +### Tunnel Intranet Penetration Deployment +* **[Deploy Relay](./relay.md)**: View config descriptions, Docker deployment, and host runtime guides for TunnelRelay nodes. +* **[Deploy OpenFlared](./openflared.md)**: Access config descriptions, Docker runtime, and auto-sync mechanisms for the intranet client. + +### Upgrade & Maintenance +* **[Upgrade & Maintenance](./upgrade.md)**: Discover upgrading procedures for Server/Agent, data retention rules, and validation commands. + +### Reference Manuals +* **[Deployment Guide](./deployment.md)**: Browse deployment topologies, prerequisites, Docker Compose samples, and multiple deployment strategies. diff --git a/docs/en/deployment/openflared.md b/docs/en/deployment/openflared.md new file mode 100644 index 00000000..a8ad41d5 --- /dev/null +++ b/docs/en/deployment/openflared.md @@ -0,0 +1,119 @@ +# Deploy OpenFlared Client + +You will learn: The responsibilities of the OpenFlared client, configuration parameters and environment variables, how to run the client via Docker, and how to deploy the client on an intranet server using the compiled host binary. + +**OpenFlared** is a tunnel client deployed in the user's intranet environment (LANs, private VPCs, or other environments that cannot be directly accessed from the public internet). Its core responsibility is to establish communication with the control plane (OpenFlare Server) via the `X-Tunnel-Token` header, automatically spawning and managing one or more **frpc (Fast Reverse Proxy Client)** subprocesses locally to securely and stably tunnel HTTP traffic back to public relay nodes. + +--- + +## Prerequisites + +1. **Retrieve Tunnel Token**: Create a new tunnel instance on the "Intranet Penetration" or "Tunnel Management" page in the OpenFlare management console; the system will automatically generate a unique `tunnel_id` and a `tunnel_token` (e.g., `tun-<32hex>`). +2. **Outbound Network Permissions**: The intranet server does not require any inbound public IPs or port mappings, but it must be able to reach the **OpenFlare Server URL** and the corresponding **TunnelRelay node control port (default 7000)** over the outbound network. +3. **Software Dependencies** (Host deployment only): + - You must have an executable `frpc` binary locally (recommended version `v0.61.0+` or the latest stable `v0.69.0`), or specify its path explicitly in the configuration. + +--- + +## Configuration & Environment Variables + +`openflared` reads `flared.json` in the working directory by default on startup. Overriding options via environment variables is fully supported. + +### Configuration Fields Details + +| JSON Field | Environment Variable | Description | Default Value | +| --- | --- | --- | --- | +| `server_url` | `OPENFLARE_SERVER_URL` | OpenFlare Server API base URL | **None (Required)** | +| `tunnel_token` | `OPENFLARE_TUNNEL_TOKEN` | Tunnel client dedicated access Token | **None (Required)** | +| `frpc_path` | `OPENFLARE_FRPC_PATH` | Path to the `frpc` executable binary | `"frpc"` | +| `data_dir` | `OPENFLARE_DATA_DIR` | Directory to store local data and generated `frpc_{relayNodeID}.toml` configs | `"./data"` | +| `state_path` | - | Path to store local state JSON file (saving the last applied version) | `"{data_dir}/flared-state.json"` | +| `heartbeat_interval`| - | Heartbeat reporting interval (ms or Go Duration string) | `10000` (10s) | +| `sync_interval` | - | Tunnel config polling interval (ms or Go Duration string) | `30000` (30s) | +| `request_timeout` | - | HTTP request timeout duration | `10000` (10s) | + +--- + +## Docker Deployment (Recommended) + +Docker is the simplest and safest way to run the client inside the intranet. The official `openflared` image embeds the client controller and `frpc v0.69.0` out of the box, requiring no environment setup. + +```bash +docker pull ghcr.io/rain-kl/openflared:latest +docker rm -f openflared 2>/dev/null || true + +docker run -d --name openflared --restart unless-stopped \ + -e OPENFLARE_SERVER_URL=http://your-server:3000 \ + -e OPENFLARE_TUNNEL_TOKEN=YOUR_TUNNEL_TOKEN \ + -v openflared-data:/app/data \ + ghcr.io/rain-kl/openflared:latest +``` + +--- + +## Manual Host Deployment + +If you need to run the client directly on a Linux/macOS/Windows host inside the intranet: + +### 1. Compile the Binary + +```bash +cd openflared +go build -o flared ./cmd/flared +``` + +### 2. Prepare `flared.json` + +Create a `flared.json` configuration file in the same directory as the executable: + +```json +{ + "server_url": "http://your-server-ip:3000", + "tunnel_token": "your-tunnel-auth-token", + "frpc_path": "/usr/local/bin/frpc", + "data_dir": "./data", + "heartbeat_interval": "10s", + "sync_interval": "30s" +} +``` + +### 3. Start the Service + +```bash +export LOG_LEVEL='info' +./flared -config ./flared.json +``` + +--- + +## Start & Validate + +### 1. Auto-Sync Workflow + +Once started successfully, OpenFlared operates the following workflow: +- **Heartbeat & Config Fetching**: Periodically polls `/api/flared/heartbeat` and `/api/flared/config` endpoints to validate the Token and evaluate configuration versions. +- **File Rendering**: When a new configuration version (or checksum mismatch) is detected, it pulls the complete tunnel routing rules. If multiple Relays are bound, it renders `frpc_{relayNodeID}.toml` configurations in `data_dir` for each Relay. +- **Hot Reload or Restart**: Spawns the corresponding `frpc` subprocesses, or executes `frpc reload` / restart actions when configurations change, ensuring traffic mappings are kept up to date. +- **Process Auto-Recovery**: If a local `frpc` tunnel process exits unexpectedly, the master program automatically restarts it after a 5-second backoff penalty. + +### 2. View Logs & Connection Status + +```bash +# Docker container logs +docker logs -f openflared +``` + +If running correctly, the logs will show output similar to: +```text +flared config loaded ... +detected frpc version v0.69.0 +flared process started +applying new tunnel config {"version": "...", "checksum": "..."} +frpc process missing, starting {"relay_id": "..."} +``` + +### 3. Verify in the Management Console + +Open the **"Intranet Penetration"** page in the management console: +- Check the online status of the corresponding tunnel; it should display green as **"Online"**. +- You can inspect which relay nodes the tunnel is connected to, and view the detailed routing configurations of the intranet services. diff --git a/docs/en/deployment/relay.md b/docs/en/deployment/relay.md new file mode 100644 index 00000000..ee0b8eb9 --- /dev/null +++ b/docs/en/deployment/relay.md @@ -0,0 +1,126 @@ +# Deploy Relay (Tunnel Relay) + +You will learn: The responsibilities of a TunnelRelay node, `openflare-relay` configuration parameters and environment variables, how to run the Relay via Docker, and how to build and deploy the Relay from source manually. + +In the OpenFlare intranet penetration architecture, the **TunnelRelay node** plays a key role. Unlike standard Edge Nodes, in addition to running the traditional Agent (managing OpenResty for HTTPS/WAF processing), it co-locates the **Relay (frps tunnel manager)** service, responsible for listening to intranet client (OpenFlared) tunnel connections and relaying traffic. + +--- + +## Prerequisites + +Before deploying a TunnelRelay node, ensure: + +1. **Registered as a TunnelRelay node**: Add a node of type `tunnel_relay` in the OpenFlare management console under "Node Management", and retrieve its node-specific `agent_token` or use the global `discovery_token`. +2. **Network Ports**: + - Ensure `bindPort` (the port frpc clients connect to, default `7000`) is accessible from the public/intranet client networks. + - Ensure `vhostHTTPPort` (the HTTP Vhost port, default `8080`) is free and not bound by other processes, as the Agent routes traffic to frps on this port. +3. **Software Dependencies** (Host deployment only): + - You must have an executable `frps` binary locally (recommended version `v0.61.0+` or the latest stable `v0.69.0`), or specify its path explicitly in the configuration. + +--- + +## Configuration & Environment Variables + +`openflare-relay` reads `relay.json` in the working directory by default on startup. Overriding options via environment variables is fully supported. + +### Configuration Fields Details + +| JSON Field | Environment Variable | Description | Default Value | +| --- | --- | --- | --- | +| `server_url` | `OPENFLARE_SERVER_URL` | OpenFlare Server API base URL | **None (Required)** | +| `agent_token` | `OPENFLARE_AGENT_TOKEN` | Node-specific Token | Mutually exclusive with below | +| `discovery_token` | `OPENFLARE_DISCOVERY_TOKEN` | One-time auto-registration Token | Mutually exclusive with above | +| `node_name` | `OPENFLARE_NODE_NAME` | Custom name for the node | Hostname by default | +| `node_ip` | `OPENFLARE_NODE_IP` | Outbound/listening IP of the node | Automatically detects real outbound IP | +| `frps_path` | `OPENFLARE_FRPS_PATH` | Path to the `frps` executable binary | `"frps"` | +| `data_dir` | `OPENFLARE_DATA_DIR` | Directory to store local data and generated `frps.toml` | `"./data"` | +| `state_path` | - | Path to store local state JSON file | `"{data_dir}/relay-state.json"` | +| `heartbeat_interval`| - | Heartbeat interval (integer ms or Go Duration string) | `10000` (10s) | +| `request_timeout` | - | HTTP request timeout duration | `10000` (10s) | + +--- + +## Docker Deployment (Recommended) + +Docker is the most convenient way to deploy a TunnelRelay node. The official Docker image embeds the `openflare-relay` controller and `frps v0.69.0` out of the box. + +```bash +docker pull ghcr.io/rain-kl/openflare-relay:latest +docker rm -f openflare-relay 2>/dev/null || true + +docker run -d --name openflare-relay --restart unless-stopped \ + -p 7000:7000 \ + -e OPENFLARE_SERVER_URL=http://your-server:3000 \ + -e OPENFLARE_AGENT_TOKEN=YOUR_AGENT_TOKEN \ + -v openflare-relay-data:/var/lib/openflare-relay \ + ghcr.io/rain-kl/openflare-relay:latest +``` + +> [!TIP] +> The `-p 7000:7000` option maps the port `frpc` clients connect to. If a custom `relay_bind_port` is configured in the management console, change this port mapping on the host accordingly. + +--- + +## Manual Host Deployment + +If you prefer to run the Relay directly on a physical host or VM: + +### 1. Compile the Binary + +```bash +cd openflare-relay +go build -o openflare-relay ./cmd/relay +``` + +### 2. Prepare `relay.json` + +Create a `relay.json` configuration file in the same directory as the executable: + +```json +{ + "server_url": "http://127.0.0.1:3000", + "agent_token": "your-relay-node-agent-token", + "frps_path": "/usr/local/bin/frps", + "data_dir": "./data", + "heartbeat_interval": "10s", + "request_timeout": "10s" +} +``` + +### 3. Start the Service + +```bash +export LOG_LEVEL='info' +./openflare-relay -config ./relay.json +``` + +--- + +## Start & Validate + +### 1. View Process Logs + +```bash +# Docker container logs +docker logs -f openflare-relay +``` + +If managed via systemd on Linux, execute: +```bash +journalctl -u openflare-relay -f +``` + +### 2. Verify Runtime Status + +Upon starting successfully, the Relay operates as follows: +- Sends HTTP heartbeats to register and go online with the control plane. +- Retrieves the active frps baseline settings (including `bindPort`, `vhostHTTPPort`, and the auto-generated `auth_token`). +- Automatically renders the `data/frps.toml` configuration locally. +- Spawns the subprocess `frps -c data/frps.toml`. +- If the `frps` process crashes, the Relay automatically restarts it after 2 seconds. + +### 3. Verify in the Management Console + +Log into the management console and navigate to **"Node Management"** to verify: +- The TunnelRelay node status is marked as **"Online"**. +- The Node Type is correctly displayed as **Relay Node** and the frps status displays as **Healthy**. diff --git a/docs/en/deployment/server.md b/docs/en/deployment/server.md new file mode 100644 index 00000000..9f862980 --- /dev/null +++ b/docs/en/deployment/server.md @@ -0,0 +1,170 @@ +# Launch Server + +You will learn: How to build the admin frontend from source, start the OpenFlare Server, choose between SQLite or PostgreSQL, and access Swagger. + +OpenFlare Server is a Gin + GORM monolithic control plane, responsible for managing the Admin UI, Admin API, Agent API, configuration rendering, version publishing, data storage, and aggregated queries. + +## Prerequisites + +| Item | Requirement | +| --- | --- | +| Go | `1.25+` | +| Node.js | `18+` | +| pnpm | Recommended enabling via `corepack enable` | +| Database | SQLite parent directory must be writable, or a reachable PostgreSQL instance | + +In production environments, we highly recommend explicitly configuring `SESSION_SECRET` and prioritizing PostgreSQL. + +## Build the Admin Frontend + +The Go Server hosts static assets located in `openflare-server/web/build`. Before starting the Server from source, build the frontend: + +```bash +cd openflare-server/web +corepack enable +pnpm install +pnpm build +``` + +Common frontend quality checks: + +```bash +pnpm lint +pnpm typecheck +pnpm test +``` + +## Start with SQLite + +```bash +cd openflare-server +export SESSION_SECRET='replace-with-a-long-random-string' +export SQLITE_PATH='./openflare.db' +export LOG_LEVEL='info' +go run . +``` + +By default, the Server listens on port `3000`. Access it at: + +```text +http://localhost:3000 +``` + +## Start with PostgreSQL + +```bash +cd openflare-server +export SESSION_SECRET='replace-with-a-long-random-string' +export DSN='postgres://openflare:secret@127.0.0.1:5432/openflare?sslmode=disable' +export LOG_LEVEL='info' +go run . +``` + +If `DSN` is set, it takes precedence over SQLite. When both `DSN` and the legacy `SQL_DSN` exist, `DSN` is prioritized. + +If the target PostgreSQL database is empty and a local SQLite database exists at `SQLITE_PATH`, the Server automatically migrates the SQLite data into PostgreSQL during startup, outputting the migration progress in the logs. + +## Start with Docker + +Deploying with Docker avoids the hassle of setting up local Go and Node.js environments. OpenFlare provides official Dockerfiles and Compose configurations to support independent container startups and multi-service orchestrations. + +### 1. Quick Start via Docker Run (SQLite Example) + +Ensure that a local directory for persisting databases and logs has been created. Run the following command to start the Server: + +```bash +# Create local mount directory +mkdir -p ./openflare-data + +# Start the container +docker run -d \ + --name openflare-server \ + -p 3000:3000 \ + -v $(pwd)/openflare-data:/data \ + -e SESSION_SECRET='replace-with-a-long-random-string' \ + -e SQLITE_PATH='/data/openflare.db' \ + -e GIN_MODE='release' \ + -e LOG_LEVEL='info' \ + ghcr.io/rain-kl/openflare:latest +``` + +Startup parameters: +* **`-p 3000:3000`**: Maps port `3000` on the host to port `3000` inside the container. +* **`-v $(pwd)/openflare-data:/data`**: Mounts the local directory to `/data` in the container, ensuring that the SQLite database `openflare.db` is not lost when restarting or rebuilding the container. +* **`SESSION_SECRET`**: The session signing hash key (required). + +--- + +### 2. One-click Startup via Docker Compose (Integrated PostgreSQL) + +We recommend using Docker Compose in production environments to orchestrate an independent PostgreSQL database and establish high-availability relationships. + +Create a `docker-compose.yml` file: + +```yaml +services: + postgres: + image: postgres:17-alpine + restart: unless-stopped + environment: + POSTGRES_DB: openflare + POSTGRES_USER: openflare + POSTGRES_PASSWORD: replace-with-strong-password + volumes: + - ./postgres-data:/var/lib/postgresql/data + healthcheck: + test: ["CMD-SHELL", "pg_isready -U openflare -d openflare"] + interval: 10s + timeout: 5s + retries: 5 + + openflare: + image: ghcr.io/rain-kl/openflare:latest + restart: unless-stopped + depends_on: + postgres: + condition: service_healthy + ports: + - "3000:3000" + environment: + SESSION_SECRET: replace-with-random-string + SQLITE_PATH: /data/openflare.db + DSN: postgres://openflare:replace-with-strong-password@postgres:5432/openflare?sslmode=disable + GIN_MODE: release + LOG_LEVEL: info + volumes: + - ./openflare-data:/data +``` + +Start the services: + +```bash +docker compose up -d +``` + +Compose configuration options: +* **`depends_on` and `healthcheck`**: Uses PostgreSQL's health check (`pg_isready`) to ensure that the database is fully initialized and ready before launching the OpenFlare Server, preventing panics from failed database connection attempts on first launch. +* **Separated Data Volume Mounts**: PostgreSQL data is mounted under `./postgres-data`, and OpenFlare data and backups are mounted under `./openflare-data`, making backups and maintenance simple. + +## CLI Arguments + +```bash +go run . --port 3000 --log-dir ./logs +``` + +| Argument | Description | Default Value | +| --- | --- | --- | +| `--port` | The port the Server listens to | `3000` | +| `--log-dir` | The directory to write logs to | Empty, outputs to stdout | +| `--version` | Outputs version and exits | `false` | +| `--help` | Outputs help and exits | `false` | + +## First Login + +Default credentials: + +| Username | Password | +| --- | --- | +| `root` | `123456` | + +Please change the default password immediately after your first login. diff --git a/docs/en/deployment/upgrade.md b/docs/en/deployment/upgrade.md new file mode 100644 index 00000000..67abd0ed --- /dev/null +++ b/docs/en/deployment/upgrade.md @@ -0,0 +1,52 @@ +# Upgrade & Maintenance + +You will learn: How to upgrade the Server and the Agent, how to clean up observability data, and which validation commands to execute before and after maintenance. + +Before upgrading, verify the currently active version, the most recent Agent application results, and your database backup strategy. In production environments, never trigger upgrades while a configuration is being published, during large-scale Agent reconnections, or while database migrations are in progress. + +## Server Upgrade + +Root users can check and trigger stable Server upgrades in the top header of the management console. You can also trigger upgrades by uploading the compiled Server binary in the console. + +To deploy preview releases, manually check the GitHub Releases page. We highly recommend prioritizing stable releases in production environments. + +Verify after upgrading: + +```bash +docker compose ps +docker compose logs -n 100 openflare +``` + +If deployed from source, restart the Server and verify that no database migration or startup errors appear in the logs. + +## Agent Upgrade + +Node Agents automatically update following stable releases by default. Upgrading to preview releases requires a manual trigger. + +You can re-execute the installation script to redeploy or force-update the Agent: + +```bash +curl -fsSL https://raw.githubusercontent.com/Rain-kl/OpenFlare/main/scripts/install-agent.sh | bash -s -- \ + --server-url http://your-server:3000 \ + --agent-token YOUR_AGENT_TOKEN +``` + +Note: Re-executing the current installation script wipes the entire installation directory, including the existing `agent.json`, local states, cached databases, and downloaded binaries. Ensure you have the node Token handy before executing the script. + +Verify after upgrading: + +```bash +systemctl status openflare-agent +journalctl -u openflare-agent -n 100 --no-pager +``` + +## Data Maintenance + +The management console's Settings page maintains options for automatic cleanup of observability data: + +| Parameter | Description | +| --- | --- | +| `DatabaseAutoCleanupEnabled` | Toggles daily automatic cleanup | +| `DatabaseAutoCleanupRetentionDays` | Data retention duration in days, minimum 1 day | + +When enabled, the Server cleans up access logs, metrics snapshots, and request reports at 3:00 AM daily. diff --git a/docs/en/design/agent-design.md b/docs/en/design/agent-design.md new file mode 100644 index 00000000..c804a0f2 --- /dev/null +++ b/docs/en/design/agent-design.md @@ -0,0 +1,181 @@ +# Agent Design Document + +You will learn: Agent design principles, core functional modules, interaction links with the Server, and how configuration applications are secured and made reliable through immutable version models and the three-stage disaster recovery rollback mechanism. + +--- + +## Requirements Analysis + +In distributed reverse proxy and edge security gateway scenarios, the Agent plays a central role in connecting the control plane (Server) and the data plane (OpenResty). Since the Agent runs on the user's actual node server, its design must adhere to the following core security and high-availability requirements: + +1. **Active Pull (Pull Model) instead of Push**: The Server does not hold the SSH keys of the nodes, nor does it actively initiate inbound connections to the nodes. All control directives and configuration updates are actively pulled by the Agent via heartbeats or long-lived connections (WebSockets). This eliminates inbound firewall security risks on the node side and prevents control channels from being hijacked. +2. **Minimal Intrusiveness**: The Agent runs as an independent Go binary process. It only interacts with the local OpenResty process through file-based configuration rewriting and signal notifications, without interfering with other system services on the node. +3. **Robust Disaster Recovery & Self-Healing**: Since network jitter, disk exhaustion, or erroneous configurations can easily lead to configuration sync failures, the Agent must possess zero-dependency local rollback and self-healing capabilities, strictly preventing a single configuration error from causing a complete node outage. +4. **Pure Data and State Landing**: The Agent is only responsible for executing file generation and control intentions rendered by the Server. It does not carry complex control plane duties like business logic validation or multi-tenant authorization, ensuring the node side remains highly efficient and lightweight. + +--- + +## Core Capabilities + +The Agent is composed of the following core sub-modules, cooperating to manage its complete lifecycle: + +| Module Name | Directory | Responsibilities | +| :--- | :--- | :--- | +| **Config Sync** | `sync/` | Pulls full configuration packages, writes files, triggers reloads, and records and reports sync statuses. | +| **Heartbeat** | `heartbeat/` | Periodically reports node health and resource metrics to the Server and retrieves the latest active version summary. | +| **WebSocket** | `wsclient/` | Maintains a persistent connection with the Server, providing sub-second real-time configuration pushes and commands. | +| **OpenResty Control** | `nginx/` | Executes Nginx config validation (`openresty -t`), rewrites, graceful reloads (`reload`), and process auto-start. | +| **Local State Store** | `state/` | Persistently records local applied versions, error logs, and buffers unsent observability metrics. | +| **Self-Updater** | `updater/` | Listens to Server self-update commands, securely pulls new binary versions, and completes in-place upgrades. | +| **Observability** | `observability/` | Collects host CPU/memory/disk and Nginx performance metrics, processes access logs, and uploads them. | +| **GeoIP Maintenance** | `geoipdata/` `geoipupdate/` | Maintains and updates the local GeoIP database periodically to support WAF country-level filtering. | + +--- + +## Interaction Flows with Server + +The Agent communicates with the control plane through **Token-based Auto-Registration** and a **Dual-channel Heartbeat/WebSocket** system during its lifecycle. + +### 1. Auto-Registration Flow + +If the Agent starts with an empty `access_token` in its local `agent.json`, but has a `discovery_token` configured, it triggers the auto-registration flow: +1. The Agent sends a registration request to `/api/agent/register`, carrying a local hardware fingerprint, IP, and hostname. +2. After validating the `discovery_token`, the Server generates a unique `NodeID` and a dedicated `AccessToken` (i.e., `agent_token`) in the database and returns them. +3. The Agent writes the dedicated Token to its local configuration file, clears the one-time `discovery_token`, and uses the `AccessToken` for all subsequent authenticated communications. + +### 2. Dual-Channel Heartbeat & Sync Mechanism + +* **HTTP Polling (Fallback and Detection)**: The Agent sends POST heartbeat packets at configured `heartbeat_interval` intervals by default. It reports health metrics while retrieving the currently active configuration version summary (Version & Checksum). +* **WebSocket Channel (Real-time Communication)**: Upon a successful HTTP heartbeat, the Agent automatically attempts to upgrade the connection to WebSocket (`/api/agent/ws`). + * Once the WS connection is established, heartbeats and metrics reporting shift entirely to the WS pipeline, reducing network overhead. + * When the Server publishes or activates a new version, it broadcasts a notification to the Agent via WS. The Agent triggers the synchronization flow **immediately** upon receiving the change event, achieving sub-second configuration deployment. + * If the WS connection drops due to network issues, the Agent automatically falls back to HTTP polling and uses an exponential backoff algorithm to attempt rebuilding the WS channel. + +### 3. Interaction Sequence Diagram + +```mermaid +sequenceDiagram + autonumber + participant Agent as OpenFlare Agent + participant OR as Local OpenResty + participant Server as OpenFlare Server + + Note over Agent: First Startup (No AccessToken) + Agent->>Server: 1. Auto-registration request (carrying discovery_token) + Server-->>Agent: 2. Issue NodeID & dedicated AccessToken (agent_token) + Note over Agent: Store Token in local configuration file + + rect rgb(240, 248, 255) + Note over Agent, Server: HTTP Fallback & WebSocket Upgrade + Agent->>Server: 3. Send HTTP Heartbeat (report system metrics & health) + Server-->>Agent: 4. Return ActiveConfig summary & AgentSettings + Agent->>Server: 5. Initiate WebSocket upgrade request (/api/agent/ws) + Server-->>Agent: 6. Upgrade successful (persistent bi-directional channel) + end + + rect rgb(245, 245, 245) + Note over Agent, Server: Real-time Configuration Publication + Note over Server: Administrator clicks publish config in UI + Server->>Agent: 7. Broadcast active config summary via WS (WSMessageTypeActiveConfig) + Agent->>Server: 8. Request full configuration details (carrying target Version/Checksum) + Server-->>Agent: 9. Return complete configuration snapshot (Nginx configs, certs, WAF rules, etc.) + Note over Agent: Backup old files, write new config to local temp path + Agent->>OR: 10. Execute config syntax validation (openresty -t) + OR-->>Agent: 11. Return validation result (OK) + Agent->>OR: 12. Send graceful reload signal (openresty -s reload) + Agent->>Server: 13. Report application success status (Apply Log & ActiveVersion) + end +``` + +--- + +## Control of OpenResty + +The Agent implements end-to-end closed-loop control of the data plane OpenResty, including configuration rendering, syntax validation, graceful reloading, and exception state capturing: + +### 1. Configuration Layout on Disk + +Upon successful sync, the Agent writes configuration files to `/etc/nginx/openflare-lua/` (or the configured `LuaDir`) according to a strict physical structure: +* `nginx.conf`: Main configuration file (replaces absolute path placeholders, configures performance parameters, shared dictionaries, and global server blocks). +* `routes.conf`: Route configuration file (generated by the Agent, containing all website server blocks, certificate paths, cache settings, and rate limit directives). +* `certs/`: Certificate storage directory (files named as `{cert_id}.crt` and `{cert_id}.key`). +* `waf/` and `pow/`: Dedicated Lua runtime scripts required for WAF and CC mitigation. +* `waf_config.json` and `waf_ip_groups.json`: Structured rules and IP databases required by the WAF filtering engine. + +### 2. Refined Reload Operations + +1. **Backup Current Config**: Before writing new files, the Agent copies the existing configuration files to a `.backup` directory, keeping a complete rollback snapshot. +2. **Write and Replace Placeholders**: Writes the pulled templates, automatically replacing absolute path placeholders (e.g., `__OPENFLARE_LUA_DIR__`) with actual local execution paths. +3. **Syntax Validation**: Calls `openresty -t -c ` to run a strict syntax test. +4. **Graceful Reload**: If validation passes, the Agent moves the files to the official paths and executes `openresty -s reload`. If OpenResty is not running, it launches the process. +5. **Exception Capture**: If validation or reload fails, the Agent intercepts the standard error output (stderr) and extracts the first 2000 characters of the detailed error log. + +--- + +## Publishing & Config Application Model + +OpenFlare discards the fragile mechanism of dynamically patching node configurations, instead using an **immutable configuration version publishing model**. + +```text +Edit rules -> Preview / View diff -> Publish -> Generate full configuration version -> Activate version -> Agent pulls -> Local application -> Report result +``` + +### 1. Core Design Principles + +* **Complete Publication**: Every publication compiles all enabled proxy routes, certificates, and global/custom WAF rules at once, generating a complete version package with a unique `checksum`. +* **Version Format**: Uses the `YYYYMMDD-NNN` incremental format, ensuring version histories are intuitive and strictly monotonic. +* **Global Single Active Version**: The system supports only one globally `active` configuration version at any given time. Rollbacks do not require reverse patching; they simply transition an older healthy version to the `active` state, and the Agent pulls and applies it. + +### 2. Three-Stage Disaster Recovery & Rollback Mechanism + +If the Agent fails to apply a configuration (or reload fails), it automatically triggers the following three-stage self-healing pipeline: + +```mermaid +graph TD + A[Config Application Failed] --> B[Stage 1: Attempt Local Backup Recovery] + B -- Backup Exists --> C[Write Local Backup Files] + C --> D[Run openresty -t Validation] + D -- Validation OK --> E[Reload Old Configuration] + D -- Validation Failed --> F[Proceed to Stage 2] + B -- No Backup --> F[Stage 2: Write Built-in Safe Fallback Config] + F --> G[Write fallback nginx.conf: Listen on Port 80 Only] + G --> H[Enable stub_status health checks] + G --> I[Return 503 for all other routes & block errors] + G --> J[Attempt to launch OpenResty to maintain basic survival] + J --> K[Proceed to Stage 3] + E --> L[Report Apply Warning] + K --> M[Block Local Repeated Application of Failed Version] + M --> N[Report Apply Error with detailed logs] +``` + +1. **Stage 1: Local Backup Rollback** + * The Agent attempts to restore the main configuration, routes, and certificates from the `.backup` directory. + * It runs `openresty -t` validation on the restored backup. If successful, it reloads and reports a `Warning` to the Server (Warning: failed to apply new version, automatically rolled back to the previous healthy version). +2. **Stage 2: Built-in Safe Fallback Runtime** + * If no local backup exists (e.g., first deployment failed) or if the rollback validation fails, the Agent activates the ultimate self-healing mechanism: writing a **built-in safe fallback configuration**. + * **Fallback Configuration Specification**: + * Listens only on port `80`, containing no real user reverse proxy routes. + * The `/openflare/stub_status` endpoint returns a healthy response, while all other requests uniformly return a `503 Service Unavailable` status code with the fixed response body `OpenFlare: No Valid Configuration`. + * It attempts to launch OpenResty with this minimal configuration. This keeps the Nginx process alive, preserving underlying health probes and metric endpoints, preventing containers/pods from being repeatedly killed and restarted by orchestration systems, while keeping sensitive routes secure. +3. **Stage 3: Local Configuration Blocking** + * The Agent records the failing configuration's `version + checksum` in its local state store blacklist. + * Until the control plane activates a new configuration (resulting in a changed `checksum`), the Agent's heartbeat blocks repeated synchronization pulls of this erroneous version, preventing nodes from entering an infinite loop of "heartbeat -> pull failing config -> crash rollback". + +### 3. WAF IP Group Asynchronous Runtime Synchronization + +To prevent highly volatile IP blacklists from triggering frequent full config publications and Nginx reloads (which still incur minor CPU and connection overhead), WAF IP groups are synchronized via an **asynchronous differential sync design**: + +* **Static Publication Snapshot**: The `waf_config.json` generated upon publication only contains the group ID reference mapping (i.e., `ip_whitelist_group_ids` / `ip_blacklist_group_ids`) and does not contain the actual list of IP addresses. +* **Heartbeat Differential Check**: The Agent uploads its locally cached IP groups MD5 checksum map in its heartbeat. +* **Differential Delivery**: The Server compares checksums and only delivers missing or modified IP groups, which are written directly to `waf_ip_groups.json` on the node without reload. +* **WebSocket Real-time Push**: When an administrator updates an IP group, or a threat intelligence subscription successfully pulls, or a security rule triggers a temporary block, the Server immediately broadcasts the IP group update package via WebSocket. The Agent receives and applies it instantly **without Nginx reloads**. + +--- + +## Design Constraints + +To protect the security boundary of the data and control plane, Agent development must strictly comply with the following engineering constraints: + +1. **Zero-Privilege Command Execution**: The Server is strictly prohibited from sending any arbitrary shell commands or scripts to the Agent (such as exec/eval). All system control operations (such as start, stop, reload, update) must be hardcoded inside the Agent binary. +2. **Strict Token Filtering and Prefix Validation**: Agent requests to the Server must be prefixed with `/api/agent/` and must carry the `X-Agent-Token` header for signature or token verification. +3. **Node Autonomy**: The Agent must support complete offline capabilities. During disconnected periods, the local OpenResty must rely on local configuration copies to keep reverse proxy services running normally. diff --git a/docs/en/design/architecture.md b/docs/en/design/architecture.md new file mode 100644 index 00000000..7dc48291 --- /dev/null +++ b/docs/en/design/architecture.md @@ -0,0 +1,223 @@ +# System Architecture + +You will learn: The overall architecture of OpenFlare, the boundaries of responsibilities for Server, Agent, OpenResty, and Admin Frontend, and the request flow of a configuration publication from the admin dashboard to activation on a node. + +OpenFlare consists of the Server, the Agent, the node-local OpenResty, and the Admin Frontend. The Server is the control plane, the Agent is the only controlled entry point on the node side, and OpenResty serves as the actual data plane. In intranet penetration scenarios, the Relay (frps manager) and OpenFlared (frpc manager) extend the data plane traffic path. + +### Standard Reverse Proxy Traffic Path + +```text +Browser + | + | Management UI / API + v +OpenFlare Server (Gin + GORM + SQLite/PostgreSQL) + | + | Agent API / heartbeat / config pull + v +OpenFlare Agent + | + | write config / openresty -t / reload / rollback + v +OpenResty binary + | + | reverse proxy + v +Origin +``` + +### Intranet Penetration Traffic Path + +```text +Browser + | + | HTTPS request + v +OpenResty (Agent, TLS/WAF) <-- TunnelRelay Node + | + | proxy_pass http://localhost:vhost_port (Host header preserved) + v +OpenFlareRelay (frps) <-- TunnelRelay Node, co-located with Agent + | + | frp tunnel protocol (HTTP Vhost routing by Host header) + v +OpenFlared (frpc) <-- Intranet Server + | + | HTTP/HTTPS forward + v +Internal Service (192.168.x.x) +``` + +## Component Responsibilities + +| Component | Responsibility | +| --- | --- | +| Server | Admin UI, Admin API, Agent/Relay/Client API, configuration rendering, version publishing, data storage, and aggregated queries. | +| Agent | Registration, heartbeats, synchronization, file writing, validation, reload, rollback on failure, self-updating, and light metrics collection. | +| OpenResty | Receives real traffic, executing WAF, PoW, authentication, and reverse proxying according to the configuration rendered by OpenFlare. | +| OpenFlareRelay | Manages the lifecycle of the frps process, providing tunnel relay services and receiving frps configurations via heartbeat. | +| OpenFlared | Manages frpc processes (can be multiple), connecting to the Relay and forwarding traffic to intranet services. | +| Frontend | Manages pages for website configs, WAF, origins, certificates, nodes, tunnels, versions, users, settings, and observability. | + +## Server + +`openflare-server` is the single-control-plane monolith: + +* Gin provides the HTTP services. +* GORM accesses SQLite or PostgreSQL. +* The existing login system provides Admin Session management. +* Authentication sources support GitHub OAuth and standard OIDC logins with external account binding. +* The Go Server hosts the `openflare-server/web` static build assets. + +The Server does not directly SSH to nodes, nor does it modify node files online. It only stores control plane state, generates complete configuration versions, and lets nodes actively pull them via the Agent API. + +## Agent + +`openflare-agent` is a Go monolithic application: + +* Runs as a single binary on the node side. +* Reads or generates local node information on startup. +* Performs periodic heartbeat check-ins to report status and retrieve active version summaries. +* Upon discovering a new version, it pulls the configuration, backs up old files, writes new files, validates them, and reloads. +* Automatically rolls back to restore operations if the application fails. +* Maintains the local WAF GeoIP mmdb, writing the built-in library on startup and updating it periodically based on configuration. + +The Agent executes validation, reload, startup, and restart uniformly via the path specified in `openresty_path`; if unconfigured, it defaults to calling `openresty`. During Docker deployments, the Agent image packages OpenResty and follows the same execution control logic. + +The node IP is maintained by default through Agent registration and heartbeat reporting; if the administrator locks the node IP, the Server only updates running status, versions, and observability fields, and no longer accepts reports from the Agent to override the locked IP. + +## Frontend + +`openflare-server/web` is the official Next.js-based frontend: + +* Next.js 15 App Router. +* React 19. +* TypeScript. +* Tailwind CSS. +* TanStack Query for server-side state. + +The frontend uses static export mode (`output: 'export'`), which is then hosted by the Go Server using `embed.FS`. All API requests must go through `lib/api/` and process the `success/message/data` response structure. + +The Server integrates the following security features: +* CORS middleware: Cross-Origin Resource Sharing protection. +* Rate limiting: Global and key API endpoint throttling. +* Session management: Cookie/Redis-based session storage. + +## Data & Request Flow + +### Management Request Flow + +```text +Browser -> Frontend -> /api/* -> controller -> service -> model -> database +``` + +Admin mutation APIs use `POST`, while read-only APIs use `GET`. Both success and failure responses return a clear `message`. + +### Agent Sync Flow + +```text +Agent HTTP heartbeat -> Server returns active version summary +Agent detects new version -> Pulls complete configuration details +Agent writes main configuration / route configurations / certificates / Lua resources / WAF runtimes +Agent runs OpenResty validation (openresty -t) and reload +Agent reports application result +``` + +### Relay Sync Flow + +The Relay (OpenFlareRelay process) runs on the TunnelRelay node and shares the same `agent_token` with the Agent: + +```text +Relay HTTP heartbeat -> Server returns frps base configuration (bindPort, vhostHTTPPort, auth_token) +Relay generates frps.toml and starts or updates the frps process +Relay periodically reports frps health status and connection statistics +Relay attempts WebSocket upgrade for real-time configuration pushes +``` + +frps configurations are relatively static (ports, auth token), dispatched via heartbeats, and **not included in the versioned publishing flow**. The Relay must monitor the frps process and auto-recover it on failures. Authentication: `X-Agent-Token` + API path prefix `/api/relay/*`, distinguished by Server via `node_type = tunnel_relay`. + +### OpenFlared Sync Flow + +OpenFlared (client) runs inside the intranet server, using independent `tunnel_token` authentication: + +```text +Client HTTP heartbeat -> Server returns tunnel configuration version summary (version, checksum) +Client detects new version -> Pulls complete tunnel route configuration (relay list + frpc proxy definitions) +Client generates independent frpc.toml configuration files for each Relay +Client starts a new frpc process for new Relays, or hot-reloads (frpc reload) existing ones +Client reports application results (success/failure details) +``` + +OpenFlared communicates with the Server via `/api/flared/*` using the `X-Tunnel-Token` header. Tunnel route configurations are versioned along with the publishing flow, ensuring all configuration changes are consistently published to both Agents and Clients via a single version number. + +**WebSocket Upgrade Flow** (Optional, controlled via `AgentWebsocketUpgradeEnabled`): + +When WebSocket upgrade is enabled: +1. The Agent retrieves run configurations and settings via HTTP heartbeat. +2. The Agent attempts to upgrade the connection to `GET /api/agent/ws` (WebSocket). +3. Once the WS connection is established, periodic state reporting and real-time commands are carried over the WebSocket pipeline, minimizing latency. +4. When the Server publishes or activates a version, it immediately broadcasts the active version summary to connected Agents, triggering the sync flow instantly. +5. If the WebSocket disconnects or fails to establish, the Agent automatically falls back to HTTP heartbeats, ensuring high availability. + +Through the `OpenRestyWebsocketEnabled` option, WebSocket reverse proxy support can be enabled or disabled at the OpenResty layer. + +### Reverse Proxy Flow + +```text +Client -> OpenResty server block -> WAF Lua -> named upstream -> Origin +``` + +Website configurations are the boundaries of reverse proxy aggregation. A single website configuration can bind multiple domains, sharing site-level rate limiting, reverse proxy, and cache settings. + +WAF executes in the OpenResty `access_by_lua_file` phase. Rules originate from the `waf_config.json` carried in the currently active version; global rule groups take effect by default, and websites can overlay custom rule groups. `waf_config.json` only stores rule group references and IP group IDs; IP group members are synchronized independently by the Agent into `waf_ip_groups.json`, and the OpenResty Lua engine merges and evaluates them by reference ID. + +WAF IP groups are managed by the Server. Manual IP groups store IP/CIDR lists directly; auto IP groups are evaluated by Server cron jobs reading request logs and applying Expr boolean rules; subscription IP groups are fetched by Server cron jobs from remote text or JSON sources. The Agent reports local IP group checksums in heartbeats, and the Server only returns mismatched IP groups. When an IP group is updated on the Server, a broadcast is sent via WebSocket to push changes, and the OpenResty Lua reads the local JSON file directly without querying the DB, request logs, or remote subscription sources. + +## Core Objects + +Current valid entities include: + +* `proxy_routes` +* `origins` +* `config_versions` +* `nodes` +* `tunnels` +* `auth_sources` +* `external_accounts` +* `node_system_profiles` +* `apply_logs` +* `tls_certificates` +* `managed_domains` +* `node_request_reports` +* `node_access_logs` +* `node_metric_snapshots` +* `traffic_analytics_rollups` +* `node_health_events` +* `waf_rule_groups` +* `waf_ip_groups` +* `waf_rule_group_bindings` +* `acme_accounts` +* `dns_accounts` +* `geoip_update_configs` + +## Key Design Decisions + +| Decision | Rationale | +| --- | --- | +| Full Config Versioning instead of Patches | Provides stable, verifiable boundaries for previewing, activating, history, and rollbacks. | +| Pull Model (Agent-driven) | Server does not need SSH keys or inbound command ports, preventing control channel hijacking. Supports HTTP and WebSocket. | +| Global Single Active Version | Reduces MVP complexity, ensuring all nodes are uniform by default. Supports previews, version history, and one-click rollback. | +| Website Multi-Domain Aggregation | Enables sharing site-level policies across domains while supporting per-domain certificate binding. | +| Server-side Observability Aggregation | Prevents UI-side temporary statistical calculations from producing inconsistent data metrics. | +| Intranet Penetration based on frp | Reuses a mature tunnel protocol rather than custom implementations to minimize stability risks. frps Vhost routing aligns naturally with HTTP. | +| Independent Binary for Relay/Client | Separation of concerns: Relay manages frps, Client manages frpc, allowing independent updates and deployments. | +| Tunnel decoupled from Node system | Tunnel clients run internally, using completely different registration and authentication flows compared to edge nodes. | + +## Recommended Reading for Contributors + +Before modifying architectural code, please read: + +1. [Product Boundaries](./index.md) +2. [Agent & Publish Model](./agent-design.md) +3. [Development Constraints](../../guideline/Constraints.md) +4. [Repository Structure](./repository.md) diff --git a/docs/en/design/cloudflare-pointing.md b/docs/en/design/cloudflare-pointing.md new file mode 100644 index 00000000..b2b615ab --- /dev/null +++ b/docs/en/design/cloudflare-pointing.md @@ -0,0 +1,223 @@ +# Cloudflare DNS Pointing Design + +## Goals + +Point **ZoneDomains (explicit FQDNs)** in OpenFlare to edge node IPs quickly via the Cloudflare API, replacing manual A-record edits in the CF console. Users organize domains into **pointing groups**: each group configures a primary node, a backup node, and a default orange-cloud (proxied) policy; members can override orange-cloud individually. The system treats the database tables as the desired state and idempotently syncs remote DNS. + +This module is an **optional integration capability**; it does not turn Zones into an authoritative DNS control plane. Zones still only handle root-domain boundaries, domains, certificates, and reverse-proxy associations; A record create/update/delete is driven by this module through Cloudflare. + +## Scope and Phasing + +### Phase 1 (this design's scope) + +* Sidebar **Cloudflare** entry with a Token-ready gate +* Connection config: import from an existing DNS account **or** standalone entry within the module (mixed sources), stored encrypted +* Pointing group CRUD: primary node, backup node (reserved), group default orange-cloud +* Member management: add/remove at the granularity of `zone_domain_id`; member-level orange-cloud +* Sync: write each member as a **single A record** on Cloudflare → current active node IPv4 +* Triggers: manual sync, member add, node/orange-cloud change, node IP change enqueue +* Async task batch sync; member sync status with readable errors + +### Phase 2 + +* Agent heartbeat offline detection of primary node failure → `active_node` switches to backup → whole-group auto sync +* Optional auto failback, failure notification push + +### Explicitly Out of Scope (later or permanent) + +* Multiple parallel Cloudflare accounts (one global connection config) +* AAAA / multi-A load balancing / CNAME to node hostnames +* Managing MX/TXT/Page Rules and other non-module A records +* Non-Cloudflare DNS providers +* Merging DNS record management into the Zone core model + +## Relationship with Existing Capabilities + +| Existing Capability | Relationship | +| --- | --- | +| `of_zones` / `of_zone_domains` | Provide the pointable FQDN list; this module only references `zone_domain_id` | +| `of_nodes.ip` | Source of A record `content`; recommended to restrict to edge nodes with valid IPv4 | +| `of_dns_accounts` + `sealSensitive` | ACME DNS-01 already supports Cloudflare Token; this module can **import** the same account or store a Token standalone | +| lego Cloudflare provider | **Only** TXT/DNS-01; this module builds its own CF HTTP client for Zone/DNS Record APIs | + +## Core Model + +```mermaid +erDiagram + CF_CONNECTIONS ||--o| DNS_ACCOUNTS : optional_import + CF_POINTING_GROUPS ||--o{ CF_POINTING_MEMBERS : contains + ZONE_DOMAINS ||--o| CF_POINTING_MEMBERS : pointed_as + NODES ||--o{ CF_POINTING_GROUPS : primary + NODES ||--o{ CF_POINTING_GROUPS : backup + NODES ||--o{ CF_POINTING_GROUPS : active + + CF_CONNECTIONS { + uint id PK + string source + uint dns_account_id + string authorization + string status + time verified_at + } + CF_POINTING_GROUPS { + uint id PK + string name + uint primary_node_id + uint backup_node_id + uint active_node_id + bool default_proxied + bool enabled + } + CF_POINTING_MEMBERS { + uint id PK + uint group_id + uint zone_domain_id UK + bool proxied + string cf_zone_id + string cf_record_id + string desired_ip + string sync_status + string last_error + time synced_at + } +``` + +### `of_cf_connections` (one valid connection globally) + +| Field | Description | +| --- | --- | +| `source` | `dns_account` \| `standalone` | +| `dns_account_id` | references `of_dns_accounts` (type=cloudflare) when `source=dns_account` | +| `authorization` | encrypted storage when `source=standalone`, payload shape `{"api_token":"..."}`, consistent with DNS accounts; the API **never returns it** | +| `status` / `verified_at` | connectivity check result and time | + +**Token resolution:** `dns_account` → decrypt the associated account; `standalone` → decrypt this row. Associated account deleted or validation failed → module not ready, sync forbidden. + +**Recommended permissions:** Cloudflare API Token with `Zone:Read`, `DNS:Edit`. + +### `of_cf_pointing_groups` + +| Field | Description | +| --- | --- | +| `name` | display name | +| `primary_node_id` | primary node | +| `backup_node_id` | backup (nullable; phase 1 stores only) | +| `active_node_id` | currently effective node; equals primary in phase 1; rewritten by phase 2 failover | +| `default_proxied` | group default orange-cloud; **only affects newly added members** | +| `enabled` | whether to participate in sync | + +Constraints: primary and backup must not be the same node; the node chosen as the active target must have a valid IPv4. + +### `of_cf_pointing_members` + +| Field | Description | +| --- | --- | +| `group_id` | owning group | +| `zone_domain_id` | globally unique: one domain belongs to at most one group | +| `proxied` | member orange-cloud (the only runtime basis) | +| `cf_zone_id` / `cf_record_id` | Cloudflare cache for idempotent updates | +| `desired_ip` / `sync_status` / `last_error` / `synced_at` | desired and sync state | + +`sync_status`: `pending` \| `syncing` \| `ok` \| `error`. + +No physical foreign keys; `zone_domain_id` unique index; query indexes on `group_id` etc. + +## Orange-Cloud Priority + +1. **Member `proxied`**: the only basis written to CF during sync. +2. **Group `default_proxied`**: copied to `proxied` when a member is **added**. +3. Later changes to the group default **do not rewrite** existing members. + +## Sync Semantics + +### Desired State + +The OpenFlare DB tables are the Source of Truth. Each member expects: + +| Item | Value | +| --- | --- | +| type | `A` | +| name | the ZoneDomain's FQDN | +| content | the group `active_node`'s IPv4 | +| proxied | member `proxied` | +| ttl | forced Auto by CF when orange-cloud is on; unified default (e.g. 300) when off | + +Phase 1 does not write AAAA. Node IP not a valid IPv4 → that member is `error`. + +### Triggers + +| Trigger | Behavior | +| --- | --- | +| Manual sync (all / group / member) | reconcile | +| Member added | initialize `proxied`, then enqueue sync | +| Member removed / group deleted | delete the remote A managed by this module by default (configurable keep) | +| Primary node / active / member `proxied` changed | re-sync the corresponding scope | +| Node IP changed (heartbeat or manual) | enqueue members whose `active_node_id` points to that node | +| Token not ready | refuse sync | + +Phase 1 does not do scheduled full reconciliation. + +### Reconcile (single member, idempotent) + +1. Resolve the CF Zone by the FQDN's registrable root domain, cache `cf_zone_id`. +2. With a `cf_record_id`, prefer Update; if stale, list by `name+type=A`. +3. **0 records** → Create; **exactly 1** → take over and Update; **multiple** → fail and tell the user to clean up in CF. +4. Write back `cf_record_id`, `desired_ip`, `sync_status`, `synced_at` / `last_error`. +5. On rate limiting, retry with bounded backoff. + +**Ownership:** only manage records cached by this module or taken over as "the only same-name A"; do not clear the Zone or touch other record types. After a user edits in the CF console, the next sync overwrites with the OpenFlare desired state. + +### Execution Carrier + +* Single record: can sync on the request path. +* Whole group / per-node batch: Asynq tasks (`cloudflare:sync_member` / `sync_group` / `sync_by_node`), registered in `bootstrap`. +* Per-member mutex to prevent concurrent double-writes. +* Node IP change path delivers tasks **best-effort**, not blocking the heartbeat. + +## API (Admin Panel) + +Prefix: `/api/v1/d/cloudflare`, Session admin auth. Package: `internal/apps/openflare/cloudflare/`; routes: `internal/router/v1/openflare/register_cloudflare.go`. + +| Resource | Method & Path | +| --- | --- | +| Connection | `GET/PUT /connection`, `POST /connection/verify`, `POST /connection/clear` | +| Overview | `GET /overview` | +| Groups | `GET/POST /groups`, `GET /groups/:id`, `POST /groups/:id/update|delete|sync` | +| Members | `GET/POST /groups/:id/members`, `POST .../members/:memberId/update|remove|sync` | +| Available domains | `GET /domains/available` | + +* Success `response.OK`; failure `response.Abort*`; the Token is **never** returned in JSON. +* Handlers separated from `logics.go`; the CF client is abstracted behind an interface for replaceability. + +## Frontend + +* Navigation: `frontend/lib/navigation/openflare-nav.ts` adds **Cloudflare** → `/cloudflare` (near Website Management / DNS Accounts). +* Routes: + * `/cloudflare`: overview; guide to configure when not ready + * `/cloudflare/settings`: mixed Token config and test connection + * `/cloudflare/groups`, `/cloudflare/groups/[id]`: list and detail (members, orange-cloud, sync) +* Services: independent service under `frontend/lib/services/openflare/`, extending `BaseService`. +* Pages follow the existing title-bar and component-split conventions; destructive actions need double confirmation. +* Copy that must be visible: sync overwrites module-managed A records; multiple same-name A records need manual cleanup; removal deletes remote records by default; phase 1 has no automatic failover. + +## Errors and Security + +* User-visible copy is module-internal constants; internal errors log via `pkg/logger`. +* Typical: token not configured, invalid token, node without IP, no CF Zone, multiple same-name A records, rate limiting. +* Token is only decrypted server-side for use; responses and logs must never contain plaintext tokens. + +## Data Migration + +* goose both dialects (PG/SQLite) create the three tables; defaults match Go zero values. + +## Key Decision Summary + +| Decision | Conclusion | +| --- | --- | +| Module shape | standalone Cloudflare pointing module, not embedded Zone fields | +| Token | mixed: imported from DNS account or encrypted standalone | +| Domain granularity | ZoneDomain (FQDN) | +| Record shape | single A → active node IPv4 | +| Failover | phase 2; heartbeat offline; phase 1 only stores backup/active | +| Orange-cloud | member-level effective; group default only initializes | +| SoT | DB tables as desired state drive CF | diff --git a/docs/en/design/edge-cache-design.md b/docs/en/design/edge-cache-design.md new file mode 100644 index 00000000..7cdde9c0 --- /dev/null +++ b/docs/en/design/edge-cache-design.md @@ -0,0 +1,264 @@ +# Edge Cache Strategy Design + +You will learn: how OpenFlare's edge `proxy_cache` aligns with the Cloudflare default loop between "should cache" and "should not cache": request eligibility (extension/policy) × response shareability (origin `Cache-Control` / `Expires` / `Set-Cookie`), and the differences from the previous over-strict request bypass. + +This design is the productized chapter on "basic caching" in [System Architecture](./architecture.md); cache results in access logs are in [Observability Data Model §3.5.1](./observability-data-model.md). + +--- + +## 1. Goals and Non-Goals + +### 1.1 Goals + +* **Close to CF default out of the box**: after enabling cache on a route, **only static extensions are cached by default** — HTML is not cached by default; **request session cookies / Authorization / client Cache-Control no longer cause a blanket BYPASS**. +* **Cacheable content hits**: a logged-in user visiting `/_app/**/*.js` and other static assets can show `MISS` → `HIT`. +* **Non-cacheable stays blocked**: policy not eligible (equivalent to CF `DYNAMIC`); origin `private` / `no-store`; responses with **`Set-Cookie` not stored** (aligned with CF OCC default); `all` is an advanced option with documented warnings. +* **Default Edge TTL when no origin freshness**: aligned with CF's per-status default TTL (see §3.5). +* **Consistent observability**: keep relying on `$upstream_cache_status` → three-state `cache_status` detail. +* **Backward compatible**: legacy route `cache_policy=url` maps to `all`; policy enum and migration rules stay in [§5](#5-兼容与迁移). + +### 1.2 Non-Goals (later iterations) + +* Cache Rules expression engine +* Forced Edge TTL ignoring origin `Cache-Control` (CF Cache Rules "Ignore cache-control") +* Purge (by URL/prefix/site-wide) +* Browser TTL rewriting, client `CF-Cache-Status` response header +* Full RFC conditions: `Authorization` cached only when the response has `public`/`s-maxage`/`must-revalidate` (needs Lua; this iteration deletes the request-side bypass entirely, relying on policy + origin headers) +* HEAD → GET conversion then cache +* Hit-rate dashboard + +--- + +## 2. Cloudflare Decision Loop (Alignment Baseline) + +CF default is a **two-stage decision**, **not** "request has Cookie → don't cache". + +### 2.1 Stage A — Eligible at Request Time + +| Condition | CF Result | +| --- | --- | +| Non-GET | not cached by default | +| Extension not in default cacheable table, no Rules forcing eligible | **`DYNAMIC`** (no cache lookup) | +| Extension in default table, or Rules eligible | continue to Stage B | +| **Request Cookie** | **no effect by default** | +| Cache Rules Bypass | `DYNAMIC` | + +CF's default cacheable extensions are keyed by **extension** rather than MIME; **HTML / JSON are not cached by default**. + +### 2.2 Stage B — Response Storeable (OCC on, Free/Pro/Biz default) + +| Condition | Result | +| --- | --- | +| `Cache-Control: no-store` / `private` | not stored | +| `public` + `max-age>0`, or future `Expires` | cacheable | +| No Cache-Control / Expires | still cacheable with per-status **default Edge TTL** (e.g. 200 → 120m) | +| Response **`Set-Cookie`** (default cache level + OCC) | **not stored**, status tends toward **BYPASS** | +| Request `Authorization` | cacheable only when the response also has `public` / `s-maxage` / `must-revalidate` (full condition simplified with Nginx this iteration, see §3.4) | + +### 2.3 Status Semantics (vs. Observability) + +| CF | Meaning | OpenFlare `cache_status` | +| --- | --- | --- | +| HIT / STALE / UPDATING / REVALIDATED | hit class | same-name or equivalent | +| MISS / EXPIRED | fetch from origin | same-name | +| BYPASS | eligible at request time, response not cacheable | `BYPASS` → UI "not cached" | +| DYNAMIC | not eligible at request time | policy skip mostly `BYPASS` or empty → UI "not cached" | + +--- + +## 3. Product Semantics + +### 3.1 Two-Level Switch (unchanged) + +* **Global** `openresty_cache_enabled`: generates `proxy_cache_path` etc.; when off, route-level cache directives are inert. +* **Route** `cache_enabled`: whether to enable `proxy_cache` in that site's `location`. + +Cache logic only runs when both are on. + +### 3.2 Policy Enum + +| `cache_policy` | Meaning | New Default | Legacy Compatibility | +| --- | --- | --- | --- | +| **`static`** | only eligible when URI matches **standard static extensions** | **yes** | — | +| **`all`** | after method bypass, no path/extension restriction (advanced; risk similar to CF Cache Everything) | no | legacy `url` → `all` | +| **`suffix`** | custom extension list (`cache_rules`) | no | kept | +| **`path_prefix`** | custom path prefix | no | kept | +| **`path_exact`** | custom exact path | no | kept | + +Render layer: historical `url` is treated as `all`; API/UI only expose the enum above. + +### 3.3 Standard Static Extensions (built-in) + +Aligned with CF default "no HTML/JSON caching"; keeps modern frontend-friendly enhancements: + +```text +css js mjs map +ico cur gif jpg jpeg png webp avif svg svgz +ttf otf woff woff2 eot +mp3 mp4 webm ogg flac +wasm pdf +zip 7z gz tar +``` + +* **Excludes** `html` / `htm` / **`json`** (aligned with CF not caching JSON by default). +* **Includes** `map` / `mjs` / `wasm` (deliberate enhancement for sourcemap / ES module / WASM hits). +* Matching: `$uri` extension, case-insensitive: + `if ($uri !~* \.(?:css|js|…)$) { set $openflare_skip_cache 1; }` + +### 3.4 Request-Side Bypass (after CF alignment) + +Only kept: + +1. `$request_method != GET` (HEAD included, consistent with current network; no CF HEAD→GET) + +**Removed** (previously over-strict, causing low hit rates): + +* Session-cookie regex +* `$http_authorization != ""` +* request `$http_cache_control` matching `no-cache|no-store|private` + +**How security still holds:** + +| Threat | Gate | +| --- | --- | +| Accidentally caching HTML/API | default `static` extensions (no html/json) | +| Personalized content | origin `private` / `no-store` (respected by Nginx) | +| Response writes session | **`Set-Cookie` → not stored** (§3.6) | +| `all` too broad | UI/doc warning: needs correct origin Cache-Control | +| API with Bearer | rely on policy (don't use `all` for APIs) + origin headers; full Auth conditional caching is later | + +### 3.5 Default Edge TTL (no origin freshness) + +Aligned with CF's per-status default TTL without `Cache-Control`/`Expires`, emitted in cache-enabled locations: + +| Status | TTL | +| --- | --- | +| 200, 206, 301 | 120m | +| 302, 303 | 20m | +| 404, 410 | 3m | + +```nginx +proxy_cache_valid 200 206 301 120m; +proxy_cache_valid 302 303 20m; +proxy_cache_valid 404 410 3m; +``` + +* When the origin provides valid `Cache-Control` / `Expires`, the origin freshness wins (no `proxy_ignore_headers`). +* **No** forced Edge TTL override ignoring origin headers. + +### 3.6 Response Side: Set-Cookie Not Stored + +Aligned with CF OCC default: an eligible request whose origin returns **`Set-Cookie`** is **not written** into `proxy_cache` (read path may still have MISS/BYPASS semantics). + +```nginx +proxy_no_cache $openflare_skip_cache $upstream_http_set_cookie; +``` + +(`proxy_no_cache` multi-arg: any non-empty and non-`"0"` arg means no write.) + +`proxy_cache_bypass` still only binds `$openflare_skip_cache` (request-side skip); the response side only affects **writes**, consistent with CF "eligible but response not cacheable". + +### 3.7 Relationship with Origin Headers + +* **Eligibility**: policy + method bypass. +* **Store / duration**: origin `Cache-Control` / `Expires` + default `proxy_cache_valid` + Set-Cookie gate + global `inactive`. + +--- + +## 4. Rendering and Data Flow + +```text +Global cache_enabled? + │ no → no proxy_cache_* generated + ▼ yes +Route cache_enabled? + │ no → location without proxy_cache + ▼ yes +set $openflare_skip_cache 0 + → non-GET → set 1 + → policy if (static/all/suffix/…) → may set 1 +proxy_cache openflare_cache +proxy_cache_methods GET +proxy_cache_bypass $openflare_skip_cache +proxy_no_cache $openflare_skip_cache $upstream_http_set_cookie +proxy_cache_valid … + → +access.log cache_status=$upstream_cache_status +``` + +### 4.1 Policy → Nginx Conditions + +| Policy | Extra Condition | +| --- | --- | +| `static` | `$uri` not matching built-in extension table → skip | +| `all` | no extra path condition | +| `suffix` | not matching `cache_rules` extensions → skip | +| `path_prefix` / `path_exact` | same as current implementation | + +### 4.2 Code Areas Involved + +| Area | Path | +| --- | --- | +| Rendering | `pkg/render/openresty/render.go` (bypass, Set-Cookie, `proxy_cache_valid`, extension constants) | +| Validation | `internal/apps/openflare/proxy_route/helpers.go` | +| Model/defaults | creating a route defaults `cache_policy=static`; `url`→`all` on read/write | +| Snapshot | `config_version` snapshot normalization | +| UI | `proxy-routes/detail/components/cache-section.tsx` | + +--- + +## 5. Compatibility and Migration + +| Data | Handling | +| --- | --- | +| `cache_policy=''` or `url` in DB (and cache enabled) | read / snapshot / render → **`all`** | +| API write with enabled and empty policy | normalized to **`all`**; UI new-create with cache on **explicitly submits** `static` | +| New routes | default **`static`** when cache enabled | +| Bypass behavior change | **breaking vs. old implementation**: cookie/auth traffic goes from "not cached" to cacheable HIT; requires **republishing node configs** | +| Default extensions | **remove `json`** from the table; sites relying on caching `*.json` can use custom `suffix` or `all` | + +**Release note:** document this alignment with the CF default model; hit rate expected to rise; `all` and wrong origin headers need ops self-check. + +--- + +## 6. UI Copy Points (Cache Tab) + +* After enabling cache, default: **standard static assets** (summary extensions, **excluding HTML/JSON**; including map/mjs etc.). +* Options: standard static / all cacheable GET (advanced) / custom suffix / path prefix / exact path. +* CF-aligned notes: + * login cookies are **not** separately skipped from caching; + * origin `private` / `no-store` / response **`Set-Cookie`** are not written to the edge cache; + * default Edge TTL used when no origin cache headers. +* **Advanced `all`**: warn "similar to Cache Everything; personalized pages must declare private/no-store from the origin". +* Global Performance cache master switch must be on. + +--- + +## 7. Decision Matrix (Avoid Missed Judgments) + +| Scenario | CF | OpenFlare (this design) | +| --- | --- | --- | +| GET static + session Cookie + origin public max-age | HIT | HIT | +| GET HTML + static policy | DYNAMIC | policy skip → not cached | +| GET + all + origin private | not stored | not stored | +| GET static + response Set-Cookie | BYPASS (OCC) | not stored | +| GET + Authorization + static public | conditional cache | cacheable (simplified; rely on origin not marking sensitive APIs public) | +| GET + no-CC 200 static | default 120m | `proxy_cache_valid` 120m | +| DevTools Disable cache (request no-cache) | edge may still HIT by default | edge may still HIT by default | +| POST | not cached | non-GET skip | + +--- + +## 8. Decision Record + +| Decision | Choice | Reason | +| --- | --- | --- | +| Request Cookie bypass | **removed** | aligned with CF; restore static hit rate for logged-in users | +| Request Authorization / Cache-Control bypass | **removed** | aligned with CF request-eligibility model; response gate as backstop | +| Set-Cookie | **bind to proxy_no_cache** | aligned with CF OCC "response Set-Cookie not stored" | +| Default Edge TTL | **per-status proxy_cache_valid** | aligned with CF default TTL when headerless, avoiding "never stored" | +| Remove json from default table | **yes** | aligned with CF not caching JSON by default | +| Keep map/mjs/wasm | **yes** | useful hits for modern frontend, deliberate enhancement | +| Default cacheable scope | cache-on defaults to `static` | benchmarked to CF, reduces HTML/API mis-caching | +| Legacy `url` | maps to `all` | doesn't narrow existing behavior | +| Full Auth conditions / Purge / Rules | later | close the default loop first, then extend | diff --git a/docs/en/design/index.md b/docs/en/design/index.md new file mode 100644 index 00000000..35e842b0 --- /dev/null +++ b/docs/en/design/index.md @@ -0,0 +1,204 @@ +# Product Boundaries + +You will learn: What OpenFlare is, what problems it solves, who the target audience is, what current stable features are available, and which design boundaries cannot be bypassed during implementation. + +OpenFlare is a self-hosted OpenResty control plane designed for single-team or single-organization internal operations. It solves the problems of decentralized management of reverse proxy configurations, node synchronization, certificate hosting, configuration publication and rollback, and basic observability. + +## Project Positioning + +OpenFlare is suitable for teams that need to centrally manage multiple OpenResty proxy nodes: + +* Wanting to maintain reverse proxy website configurations using a management dashboard. +* Wanting every configuration change to have a complete version history, preview, activation, and rollback support. +* Wanting nodes to actively synchronize configurations, rather than the control plane SSHing into nodes to execute commands. +* Wanting to manage TLS certificates, domain assets, node statuses, and basic access analytics in a single system. + +OpenFlare is currently not positioned as a general-purpose logging platform, service mesh, Kubernetes Ingress Controller, or multi-tenant cloud platform. + +## Current Capabilities + +| Capability | Description | +| --- | --- | +| Reverse Proxy Rules | Uses website configuration as the aggregation boundary, supporting multiple domains and origin settings. | +| Website-level Config | One rule corresponds to one website, which can bind one or more domains and share site-level configurations. | +| Origin Management | Maintains a lightweight origin directory and allows websites to save renderable origin snapshots. | +| Config Versioning | Supports previews, publishing, activation, immutable history, and rollbacks. | +| Agent Sync | Supports registration, heartbeats, synchronization, application result reporting, and self-updating. | +| OpenResty Hosting | Manages main config templates, performance parameters, cache parameters, and Lua resources. | +| HTTPS/TLS | Hosts certificate and domain assets, binding certificates on a per-domain basis. | +| WAF | Maintains IP/CIDR block blacklists/whitelists, IP groups, and country-level geographic access controls at both global and site-specific levels. | +| Basic Observability | Aggregates node requests, resource snapshots, health events, and access analytics. | +| Node Management | Manages node status, token systems, and deployment/update lifecycles. | +| Admin UI | Next.js-based official management dashboard. | +| Auth Source Login | Supports configuring GitHub OAuth and standard OIDC login portals, allowing third-party accounts to bind to existing local users. | +| Intranet Penetration | Securely exposes intranet HTTP services to the public internet using TunnelRelay nodes and the OpenFlared client, reusing the Agent's HTTPS/WAF capabilities. | + +Default Working Model: + +* All nodes consume the same globally activated configuration version. +* The Server stores configurations and state, and does not directly SSH to manage nodes. +* The Agent is the only controlled entry point on the node side. +* TunnelRelay nodes run both the Agent (OpenResty) and the Relay (frps manager) to provide intranet penetration relays. +* The OpenFlared client runs inside the intranet, managing the frpc process to connect to the Relay and forward traffic to intranet services. + +## Typical Use Cases + +| Scenario | Description | +| --- | --- | +| Unified Entrance | Exposes multiple internal HTTP services via a unified domain and TLS certificate. | +| Multi-Node Sync | Multiple OpenResty nodes consume the same active configuration version. | +| Change Review | View previews or diffs before publishing, keeping an immutable history post-publish. | +| Rapid Rollback | Re-activate an older version, letting the Agent pull and apply it. | +| Certificate Hosting | Bind TLS certificates to different domains under the same website. | +| Observability | Check node health status, aggregated requests, traffic analytics, and health events. | +| Intranet Penetration | Exposes intranet HTTP services that are not directly reachable from the public internet using Tunnels, benefiting from HTTPS, WAF, and all other protections. | + +## Website Configuration Constraints + +`proxy_routes` is the aggregate object for "website configurations". One record corresponds to one website, which can bind one or more domains and share a set of site-level configurations. + +Constraints: + +* `proxy_routes.site_name` is the unique business identifier of the website. +* `proxy_routes.domains` must contain at least one domain, and `domains[0]` is treated as the primary domain. +* Any domain can globally belong to only one `proxy_routes`. +* Site-level rate limits, reverse proxies, and caching configurations are shared by the site, with no per-domain differences allowed within the same website. +* HTTPS allows binding certificates on a per-domain basis within the same site. + +## Origin & Upstream Constraints + +`origins` serve the reuse of the origin directory, storing only the origin address, display name, and remarks, without carrying protocols, ports, paths, weights, or health check policies. `proxy_routes` can optionally associate with an `origins` record, but the rule internally still saves a complete upstream snapshot for rendering. + +Upstream Constraints: + +* `proxy_routes` must contain at least one upstream address (for direct type `direct`), or be associated with a Tunnel (for intranet penetration type `tunnel`). +* Multi-upstream load balancing is uniformly rendered into a named `upstream` with keepalive enabled. +* A single upstream is allowed to carry a base path or query, which is appended in `proxy_pass`. Multi-upstream is strictly limited to pure `scheme://host[:port]` structures, and all upstreams in the same rule must use the same protocol. +* `proxy_routes.origin_host` is an optional field used to override the `Host` header during back-to-source requests. +* All direct upstream addresses must be valid `http://` or `https://` URLs. +* Intranet penetration upstreams must associate with a valid `tunnel_id` and specify the intranet target address and protocol. + +## Intranet Penetration Constraints + +OpenFlare implements intranet penetration through TunnelRelay nodes and the OpenFlared client, built on top of frp (Fast Reverse Proxy). + +### Node & Component Model + +**Node Types**: + +* `nodes.node_type` distinguishes the node type: `edge_node` (edge node, default) and `tunnel_relay` (tunnel relay). +* TunnelRelay nodes run both the Agent (OpenResty) and the Relay (frps manager) concurrently, sharing the same `agent_token`. + - The Agent is responsible for HTTPS termination, WAF protection, caching, and rate limiting. + - The Relay manages the frps process, providing tunnel relay services for intranet clients. +* TunnelRelay nodes introduce new fields: `node_type`, `relay_bind_port` (frpc connection port, default 7000), `relay_vhost_http_port` (HTTP Vhost port, default 8080), `relay_auth_token` (automatically generated), `relay_status`, etc. + +**Tunnel Client**: + +* The `tunnels` table independently stores intranet penetration client registration info and is decoupled from the `nodes` system. +* Each Tunnel has a unique `tunnel_id` (format `tun-<32hex>`) and `tunnel_token` (client authentication credential). +* The OpenFlared client runs inside the intranet, is not exposed to the public internet, uses `tunnel_token` for authentication, and communicates with the Server via `/api/flared/*` endpoints. +* An OpenFlared client can connect to multiple Relays simultaneously for high availability. + +### Upstream Type Expansion + +The upstream configuration of `proxy_routes` is divided into two types, distinguished by the `upstream_type` field: + +* **Direct Upstream (`direct`, default)**: Forwards traffic directly to the origin address, behaving exactly like the existing mechanism. +* **Intranet Penetration Upstream (`tunnel`)**: Forwards traffic to the intranet service via a TunnelRelay node. + - Must specify `tunnel_id` (associated with the `tunnels` table). + - Must specify `tunnel_target_addr` (intranet target address, e.g., `192.168.1.100:8080`) and `tunnel_target_protocol` (`http` or `https`). + - During publication, the Server automatically replaces the upstream address with `http://127.0.0.1:{relay_vhost_http_port}`. + +### Traffic Paths & Protocols + +**Complete Data Plane Traffic Path**: + +``` +Browser → OpenResty (Agent, TLS/WAF) [TunnelRelay Node] + ↓ + frps (Relay, HTTP Vhost Routing) [TunnelRelay Node, 127.0.0.1:{vhost_port}] + ↓ + frp Tunnel Protocol (Host Header Routing) + ↓ + frpc (Client, Multi-process) [Intranet Server] + ↓ + Intranet Service (192.168.x.x:port) +``` + +**Key Features**: + +* frps uses the HTTP Vhost single-port reuse mechanism; all HTTP tunnels share one `vhost_port`, automatically routed to the corresponding frpc based on the Host header. +* The Agent preserves the original `Host` header, which frps uses to match the virtual host. +* Each tunnel corresponds to a single `proxy_routes` and can bind multiple domains. +* The OpenFlared client manages an independent frpc process for each connected Relay, transmitting multiple HTTP proxy definitions via a single frp tunnel. + +### Configuration Sync Model + +The publication process generates two types of configuration version data simultaneously, linked by a single `config_version` version number: + +* **Agent-side Config**: OpenResty main configuration + route configurations + WAF rules. If a tunnel upstream is included, it is automatically rendered as a `http://127.0.0.1:{vhost_port}` upstream. +* **Tunnel-side Config**: Relay list + frpc proxy definitions. Versioned alongside the publishing process; changes are hot-reloaded using `frpc reload` first. +* **Relay Config**: Dispatched via heartbeat responses, relatively static, and not included in the versioned publishing flow. + +### Tunnel Design Constraints + +* Only HTTP protocol tunnel traffic is supported (keeping TCP/UDP tunnels extensible); separate TCP/UDP port allocation is not supported for now. +* The DNS for domains using Tunnel upstreams should resolve to the designated TunnelRelay node. +* frp binaries (v0.61+) are packaged and provided by the system deployment script or container images. + +## HTTPS Constraints + +`proxy_routes.domain_cert_ids` is used to record the domain-certificate bindings parallel to `domains`; a value of `0` means the domain does not have HTTPS enabled and stays HTTP-only. + +During rendering: + +* Domains with certificates are grouped by certificate and output as independent `443 ssl` `server` blocks. +* Domains without certificates bound must not be automatically routed to HTTPS. +* All domains in `proxy_routes.domains` must be kept in the same site configuration to avoid being split across version snapshots. + +## WAF Constraints + +WAF centers around rule groups. The system provides a single global rule group (applied to all sites by default), on top of which websites can overlay multiple custom rule groups. + +Core Capabilities: + +* Supports individual IP / CIDR block whitelists and blacklists. +* Supports IP group references (including manual, automatic Expr calculated, and URL subscribed IP groups). +* Supports GeoIP-based country/region level admission filtering. +* Supports custom interception responses for rule groups (custom status codes and interception HTML pages, default is `418`). + +IP Group & Judgment Constraints: + +* **Runtime Decoupling**: The WAF runtime only reads local JSON files and does not access the Server database; configuration versions only store referenced IP group IDs. IP group members are synchronized via MD5 checksum differences and WebSocket push notifications, achieving hot activation without reloading Nginx. +* **Built-in Expr Rules**: + * High-frequency 404 scanning block: `request_count > 100 && status_404_ratio >= 0.8` + * Malicious IP direct probe: `ip_host_count > 50 && ip_host_ratio > 0.5` +* **Decision Priority**: The whitelist has absolute priority. If it does not match the whitelist, the blacklist funnel is triggered (global rule group first, custom groups matched in ascending ID order). +* GeoIP resolution depends on the local MaxMind database; if GeoIP is anomalous, region rules are automatically ignored and must not disrupt the availability of IP rules and the main reverse proxy chain. + +## Authentication Source Constraints + +`auth_sources` uniformly supports `github` and `oidc` login configurations. `external_accounts` stores bindings between third-party accounts and local users. Logic for first-time third-party login: + +* If already bound, directly authorize login; if there is an active local session, automatically bind. +* If unbound and registration is enabled, automatically create a local account; if registration is closed, require the user to provide an existing local username and password to establish the association. + +## Version & Observability Constraints + +* `config_versions` must save the complete snapshot, rendering result, and `checksum`. +* Globally, only one version can be active at a time. +* Rollback is achieved by re-activating an older version. +* `nodes` only carry control plane state and low-frequency summaries; they do not carry high-frequency observability facts. +* Metrics, trends, and access analytics prioritize server-side aggregation rather than client-side temporary statistics. +* Access detail logs are only retained within a controlled time window, not evolving into a general logging platform. + +## Documentation Maintenance Principles + +* Update this document when the product range or system boundaries change. +* Update [System Architecture](./architecture.md) when the system structure or module responsibilities change. +* Update [Agent & Publish Model](./agent-design.md) when the publishing, synchronization, rollback, or Agent model changes. +* Update [Development Constraints](../../guideline/Constraints.md) when developer constraints, code specifications, or API conventions change. +* Update README and [Deployment Instructions](../../deployment/deployment.md) when deployment methods change. +* Update [Configurations Reference](../reference/configuration.md) when configuration items change. +* Completed phases should no longer be backfilled as "version plans". +* Before starting a new phase, complement the design first, then proceed to implementation. diff --git a/docs/en/design/kuma-design.md b/docs/en/design/kuma-design.md new file mode 100644 index 00000000..36a4592b --- /dev/null +++ b/docs/en/design/kuma-design.md @@ -0,0 +1,109 @@ +# Uptime Kuma Sync Design + +You will learn: the design background of the OpenFlare × Uptime Kuma monitoring integration, the control-flow design based on the Socket.IO protocol, the anti-pollution model centered on tag isolation, and the differential incremental sync state machine. + +--- + +## Requirements Analysis + +In a multi-node gateway architecture, monitoring system state and reverse proxy route state are usually disconnected: +1. **High entry overhead**: every time the gateway control plane adds or decommissions a site, the admin must re-configure the corresponding probe address and alert policy in the monitoring system (e.g. Uptime Kuma). +2. **Data inconsistency**: when a proxy route domain changes or switches to HTTPS, monitoring parameters are easily left un-updated, causing false positives or missed alerts. +3. **Environment pollution risk**: a full "delete-recreate" sync in monitoring would wipe historical statistics and SLA curves, and would also affect other monitor tasks the user configured manually on the instance that are unrelated to the gateway. + +To address these, OpenFlare introduces a **Uptime Kuma auto-monitoring sync mechanism** based on the client/server model, achieving strongly consistent, low-overhead, zero-pollution synchronization between gateway site route definitions and the availability monitoring system. + +--- + +## Core Architecture + +The Uptime Kuma sync subsystem runs entirely in the **Server control plane** background scheduler. + +```text + [ OpenFlare Control Plane / DB ] [ Uptime Kuma Instance ] + │ │ + 1. Scheduled Cron trigger (Job) │ + │ │ + 2. Read proxy routes & options config │ + │ │ + 3. Connect to Socket.IO <──── 4. Socket.IO handshake & login ────┤ + │ │ + ├────── 5. Validate / create "OpenFlare" tag ──►│ + ├────── 6. Compare site attrs vs Kuma monitor list ─►│ + │ │ + └────── 7. Execute differential ops (add / edit / delete) ─►│ +``` + +The sync subsystem does not pass through the data-plane Agent nodes; the Server talks directly to Uptime Kuma's exposed Socket.IO endpoint. This reduces edge node network overhead and keeps auth credentials (Kuma username/password) safely inside the control plane. + +--- + +## Tag Isolation and Anti-Pollution Design + +To run safely in a shared Uptime Kuma instance without disturbing manually created monitors, a **dedicated tag isolation mechanism** is used: + +1. **`OpenFlare`-specific tag**: + * On first connect, the sync routine calls `getTags` to fetch all tags in the instance. + * It checks whether a tag named `OpenFlare` exists (default color indigo `#4f46e5`). If not, it creates it automatically via the `addTag` API. +2. **Filtered scope**: + * After fetching Uptime Kuma's monitor list (`monitorList`), the sync task only keeps monitors **tagged with `OpenFlare`**. + * All modification comparisons (`editMonitor`) and offline cleanups (`deleteMonitor`) operate **only within this filtered subset**. Any monitor not bound with the `OpenFlare` tag is "invisible" to the sync routine — perfect anti-pollution isolation. + +--- + +## Differential Sync State Machine + +On each run, the sync routine computes a diff between OpenFlare's local config and Uptime Kuma's data, then executes different Socket.IO events based on the comparison: + +```mermaid +stateDiagram-v2 + [*] --> 检查站点状态与监控范围 + + state "检查监控范围" as Scope { + [*] --> 校验站点是否启用并且在 Scope 内 + 校验站点是否启用并且在 Scope 内 --> 在Scope内 : 是 + 校验站点是否启用并且在 Scope 内 --> 不在Scope内 : 否 + } + + 不在Scope内 --> 检查Kuma中是否存在同名且带标签的监控 + 检查Kuma中是否存在同名且带标签的监控 --> 执行清理 : 存在 + 检查Kuma中是否存在同名且带标签的监控 --> 忽略 : 不存在 + + 在Scope内 --> 检查Kuma中是否存在同名监控 + + state "比对属性" as Compare { + [*] --> 检查是否存在 + 检查是否存在 --> 新建监控项 : 否 + 检查是否存在 --> 比对元数据 : 是 + 比对元数据 --> 属性一致 : 匹配 + 比对元数据 --> 属性不一致 : 不匹配 + } + + 新建监控项 --> 发送add指令并绑定Tag + 属性不一致 --> 发送editMonitor指令 + 属性一致 --> 忽略 + + 执行清理 --> 发送deleteMonitor指令 + 忽略 --> [*] +``` + +### 1. Monitor URL Normalization +A site route in OpenFlare can configure multiple domains; the sync routine automatically extracts the primary domain and assembles a standard `http://` or `https://` prefix based on whether HTTPS is enabled. + +### 2. Compared Attribute Set +If a same-named, tagged monitor already exists, the sync routine compares the following 5 key fields against the current gateway global option. Any mismatch triggers an update: +* **URL**: `Url` +* **Probe interval**: `Interval` (default 60s) +* **Max retries**: `MaxRetries` +* **Retry interval**: `RetryInterval` (default 60s) +* **Request timeout**: `Timeout` (default 48s) + +--- + +## Scheduler and High-Concurrency Protection + +1. **Cron-based single-thread execution**: + * The Server periodically (every 1 minute) probes via a background Cron Job whether the configured sync interval (`UptimeKumaSyncInterval`) is reached. + * The task uses mutex locking internally. If a previous sync request is still running due to network latency, the next schedule is skipped automatically, preventing concurrent Socket.IO connections from DDOS-ing the Uptime Kuma instance. +2. **WebSocket state listening**: + * The sync routine uses Socket.IO's event listener; after the connection is established, it only proceeds to the differential algorithm once the full `monitorList` event list push is received, avoiding monitor deletion caused by incomplete data loading. diff --git a/docs/en/design/login-captcha.md b/docs/en/design/login-captcha.md new file mode 100644 index 00000000..daa35a61 --- /dev/null +++ b/docs/en/design/login-captcha.md @@ -0,0 +1,127 @@ +# Login CAPTCHA Integration (Cap) + +This document describes the design of introducing **Cap** — an open-source CAPTCHA solution based on Proof-of-Work (PoW) and invisible browser fingerprint features — into the OpenFlare control plane, to protect the login API against brute-force attacks and credential-stuffing by crawlers. + +--- + +## 1. Business Background and Product Scope + +### Background and Pain Points +The OpenFlare login endpoint `/api/v1/user/login` lacks user-dimension protection; attackers can use proxy pools to perform credential stuffing and brute-force attacks on high-privilege accounts (such as `root`). At the same time, standard visual CAPTCHAs are unfriendly to login-page UX and accessibility. + +### Product Scope and Technology Choice +* **Technology choice**: Cap (a Proof-of-Work-driven, invisible, image-free CAPTCHA solution). + - **Core principle**: the client (Widget/page) obtains a proof-of-work (PoW) challenge from the server, computes the solution in the browser background, and sends the answer back. The server verifies the answer to complete human-machine verification. + - **Advantages**: invisible, image-free, no dependency on external third-party API nodes (private), tiny package size. +* **Integration scope**: the control-plane Server login API (`/api/v1/user/login`) and the frontend login page. +* **Config granularity**: admins can toggle the CAPTCHA on/off anytime via the console Option table (`cap_login_enabled`). + +--- + +## 2. System Architecture and Interaction Sequence + +### 2.1 Module Responsibilities +1. **Frontend**: + * Introduces the `cap-widget` (React 19 custom element) on the login page. + * On form submit, accompanies the submission with the `cap-token` solved by the Widget. +2. **Server (control-plane backend)**: + * Exposes `POST /api/cap/challenge` to distribute the PoW challenge and a signed JWT token to the client. + * Exposes `POST /api/cap/redeem` to verify the submitted PoW solution and issue a login credential (Redeem Token) with an expiry time. + * Stores the Redeem Token and its expiry in the in-memory/Redis cache. + * In `POST /api/v1/user/login`, when CAPTCHA protection is enabled, first validates and consumes (single-use) the corresponding `cap-token`. + +### 2.2 Verification Flow Sequence Diagram +```mermaid +sequenceDiagram + autonumber + actor User as User + participant Browser as Browser (Frontend Web) + participant Server as OpenFlare Server (Backend) + participant Cache as Memory/Redis Cache + + User->>Browser: Open login page + Browser->>Server: POST /api/cap/challenge (get challenge) + Server->>Browser: Return {challenge, token, expires} (JWT format) + Note over Browser: Widget computes the PoW challenge in background (WASM/Worker) + Browser->>Server: POST /api/cap/redeem (submit solutions + token) + alt PoW solution valid + Server->>Cache: Store Redeem Token (tokenKey:expires) + Server->>Browser: Return {success: true, token} (i.e. cap-token) + else validation failed + Server->>Browser: Return {success: false, reason} + end + User->>Browser: Enter account/password, click login + Browser->>Server: POST /api/v1/user/login (with X-Cap-Token in HTTP header) + alt CapLoginEnabled = true + Server->>Server: Middleware (CapAuth) validates and consumes X-Cap-Token + alt token valid, not expired, not consumed + Server->>Server: c.Next() -> normal login logic (Bcrypt password check) + Server->>Browser: Return login success (Session Cookie) + else token invalid or already consumed + Server->>Browser: Intercept and return CAPTCHA error (401 Unauthorized) + end + else CapLoginEnabled = false + Server->>Server: c.Next() -> normal login logic + end +``` + +--- + +## 3. Core APIs and Data Model + +### 3.1 API Definitions + +#### 1. Get Challenge (POST /api/cap/challenge) +* **Method**: `POST` +* **Auth**: public +* **Response payload** (unified API envelope, `data` is the business payload): + ```json + { + "error_msg": "", + "data": { + "challenge": { + "c": 1, + "s": 32, + "d": 4 + }, + "token": "eyJhbGciOiJIUzI1NiIsInR5cCI6IkpXVCJ9...", + "expires": 1717660800000 + } + } + ``` + +#### 2. Redeem Challenge (POST /api/cap/redeem) +* **Method**: `POST` +* **Request payload**: + ```json + { + "token": "challenge_jwt_token_here", + "solutions": [12345, 67890, 54321] + } + ``` +* **Response payload (success)**: + ```json + { + "success": true, + "token": "random_id:ver_token", + "expires": 1717661000000 + } + ``` + +#### 3. Login API (POST /api/v1/user/login) +* **Request payload unchanged**: + ```json + { + "username": "root", + "password": "your_password" + } + ``` +* **CAPTCHA carrier**: placed in the HTTP Request Header `X-Cap-Token`. + +--- + +## 4. Replay Attack Protection and Security Trade-offs +1. **JWT temporary state binding**: the challenge is signed into the JWT payload at generation time, including an expiry limit (10 minutes). +2. **Replay interception (nonce consumption)**: when the client calls `/redeem` to submit the solution, the backend marks the JWT signature as used in the cache. Re-submitting the same solution package returns `already_redeemed`. +3. **Redeem single-use (one-time invalidation)**: when the client logs in and submits the `cap-token`, the backend immediately deletes the key from the cache after validating it, preventing attackers from extracting historical valid `cap-token`s for login replay. +4. **Seamless verification**: by tuning parameters like `c` (challenge count) and `d` (difficulty), you balance solve time against anti-crawler strength; users solve silently in the background without interrupting the login flow. diff --git a/docs/en/design/logstore.md b/docs/en/design/logstore.md new file mode 100644 index 00000000..156f3cf1 --- /dev/null +++ b/docs/en/design/logstore.md @@ -0,0 +1,86 @@ +# Log Store Decoupling + +You will learn: which tables are log-purpose, why they must not be pinned to ClickHouse, and which code path a new log table must follow. + +Observability fields and the reporting protocol are still governed by [Observability Protocol & Tables](./observability-data-model.md); this document only defines **where data is stored and how to switch databases**. + +--- + +## 1. Goals + +* **ClickHouse optional**: when not enabled, PostgreSQL (or SQLite when the primary DB is off) fully takes over writes, queries, aggregation, and cleanup. +* **Upper layers don't touch the underlying DB**: apps only face `internal/repository/logstore` (or the `repository` facade). `repository/analytics` and `db.ChConn` / `db.ChDB` are only used by logstore's ClickHouse implementation. +* **Switchable**: 「Switch Log Database」in Task Management copies data between PostgreSQL/SQLite and ClickHouse and flips the primary; writes are frozen during migration, the switch only happens on success, and source data is not deleted. + +--- + +## 2. What Counts as a Log Table + +A table enters logstore only if it meets all of: + +* Append-only writes, almost no row updates +* Query or aggregate by time, deletable by retention days +* Must still support writes and queries when ClickHouse is off +* Does not participate in transactional consistency for websites / nodes / certificates, etc. + +**Don't** make these log tables: Zones, nodes, config versions, task executions, upload metadata. These go through the business primary DB `repository`. + +Current log domains: + +| Domain | Interface | Tables | +| --- | --- | --- | +| Node access logs | `AccessLogStore` | `of_node_access_logs` | +| Observability time series | `ObservabilityStore` | `of_node_metric_snapshots` / `of_node_edge_health` / `of_node_obs_frps` / `of_node_obs_frpc` | +| User access audit | `UserAccessLogStore` | `w_user_access_logs` | + +Hourly materialized views on ClickHouse (e.g. `of_access_log_hourly`) only serve CH query acceleration. PostgreSQL / SQLite **do not** build isomorphic aggregation tables; queries aggregate in real time from raw logs. + +--- + +## 3. Layering + +| Layer | Path | Responsibility | +| --- | --- | --- | +| Abstraction | `internal/repository/logstore` | Interfaces + `Active` / `BuildForMigration`; selects implementation by `log_database` | +| CH implementation | `logstore/clickhouse_store.go` | Delegates to `repository/analytics` (native batch + existing aggregation SQL) | +| Primary DB implementation | `logstore/postgres_store.go` | PostgreSQL (high-frequency tables partitioned monthly) and SQLite (plain tables) share GORM | +| Model | `internal/model/analytics` | Entities and batch SQL, no IO | +| Enqueue | `chwriter` / `risk_control` + `batchwriter` | `FlushFunc` calls `logstore.Active`; node logs / observability enqueue via hooks | +| Constraint | `logstore/imports_test.go` | apps are forbidden from importing `repository/analytics` | + +`log_database` has only two legal states: **follow the business primary DB** (`postgres` or `sqlite`) or **`clickhouse`**. "Primary PostgreSQL + log SQLite" does not exist. `log_database` / `log_db_migration` are protected and cannot be changed from the admin panel. + +At startup: `log_database=clickhouse` but ClickHouse not enabled → startup is refused; you must re-enable ClickHouse, switch back to the primary DB, and only then turn it off. + +--- + +## 4. Switch Protocol + +Task type `of_log_db_switch` (admin name 「Switch Log Database」), parameter `target`. + +1. Validate the target is legal and not the current DB. +2. Write `log_db_migration=migrating`, drain in-flight batchwriter (`Drain`, not `Stop` writer). Writes return a clear error afterward (HTTP 503), not queued backlog. +3. Clear the target log tables, then copy by id in pages; call `EnsurePartitions` on the PostgreSQL target before copying. +4. Only on full success write `log_database=target` and clear the migration marker; on failure clear the marker and writes continue on the source DB. +5. Source data is not deleted; re-clear the target before retry to guarantee idempotency. + +Don't invent another switch protocol, and don't connect `analyticsrepo` directly inside tasks. + +--- + +## 5. Adding a New Log Table + +Column names must be identical across the three goose migrations (ClickHouse / PostgreSQL / SQLite). Key points: + +* High-frequency tables: CH uses `MergeTree` + `toYYYYMM`; PG uses `PARTITION BY RANGE(time column)` with the partition key in the primary key; SQLite uses a plain table + indexes. +* IDs use snowflake `uint64`, preserved as-is during migration. +* Writes go through a dedicated `batchwriter`; flush calls `logstore.Active`, not `analyticsrepo.BatchInsert`. +* The switch task's `copy*` must cover the new table; cleanup uses existing `log_retention_days_*` or `metric_retention_days`, don't use the wrong TTL. + +Runtime config: [Configuration Reference · Log Storage](../reference/configuration.md#8-日志存储log-database). + +--- + +## 6. Related Docs + +* Observability fields and reporting protocol: [Observability Protocol & Tables](./observability-data-model.md) diff --git a/docs/en/design/observability-data-model.md b/docs/en/design/observability-data-model.md new file mode 100644 index 00000000..673055d4 --- /dev/null +++ b/docs/en/design/observability-data-model.md @@ -0,0 +1,769 @@ +# Agent Reporting Protocol and Observability Data Model + +You will learn: the **data structures** of the refactored Agent heartbeat/WS reports, how the Server **parses and writes** them, and the **target table structures** in ClickHouse / relational DBs. +**No protocol compatibility layer**: Agents upgrade via destroy-and-recreate or binary replacement; old fields are not parsed, old buffers are discarded wholesale. + +This design is the **protocol & storage chapter** of [Edge Observability & Business Traffic Stats Refactor](./observability-design.md); implement against the fields and DDL in this document. + +**First read the transport overview and examples:** [Observability Transport Model](./observability-transport-model.md). + +--- + +## 1. Design Goals + +| Goal | Description | +| --- | --- | +| Agent reports only facts | details + host readings + edge health instant state; no business pre-aggregation | +| One business detail table | access logs are the only L1 write path | +| Aggregation in DB/control plane | hourly summaries come from ClickHouse MV or queries; the Agent never writes summary tables | +| No field overlap | `bytes_sent` = data provided; NIC `network_*` = host; no business `openresty_tx` anymore | +| Evolvable | new fields optional; missing numeric values default to 0; removed legacy protocol fields are not parsed | + +--- + +## 2. Layering and Write Overview + +```text + Agent NodePayload (v2) + │ + ┌───────────────┼───────────────┐ + ▼ ▼ ▼ + access_logs host_metrics edge_health + (L1 details) (L3 readings) (L2 instant) + │ │ │ + ▼ ▼ ▼ + of_node_access_logs of_node_metric_ of_node_edge_health + │ snapshots │ + │ │ │ + ▼ ▼ │ + of_access_log_hourly of_node_metric_ │ + (MV, Server side) capacity_hourly (MV) │ + │ │ │ + └─────── admin aggregation API ────┘ + +Relational DB (PostgreSQL/SQLite): node latest state, Profile, health events (not a detail lake) +``` + +| Layer | Meaning | Agent Report Block | ClickHouse Fact Table | +| --- | --- | --- | --- | +| L1 | business delivery | `access_logs` | `of_node_access_logs` | +| L2 | edge health | `edge_health` | `of_node_edge_health` | +| L3 | host capacity | `host_metrics` | `of_node_metric_snapshots` | + +--- + +## 3. Agent Report Data Structures (protocol v2) + +### 3.1 Top-Level `NodePayload` + +Transport: HTTP heartbeat body and WebSocket `status` messages share the same structure. + +```json +{ + "schema_version": 2, + "node_id": "n_xxx", + "name": "edge-1", + "ip": "1.2.3.4", + "version": "3.3.0", + "ext_version": "", + "current_version": "cfg-checksum-or-version", + "last_error": "", + "profile": { }, + "host_metrics": { }, + "edge_health": { }, + "access_logs": [ ], + "buffered": [ ], + "health_events": [ ], + "waf_ip_group_checksums": { "1": "md5..." } +} +``` + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `schema_version` | int | suggested | fixed to `2` (this design) | +| `node_id` | string | ✅ | node ID | +| `name` | string | ✅ | display name | +| `ip` | string | ✅ | reporting IP | +| `version` / `ext_version` | string | ✅ | Agent version | +| `current_version` | string | | locally active config version summary | +| `last_error` | string | | latest sync/runtime error, nullable | +| `openresty_status` | string | ✅ (when OpenResty present) | **latest health-state authoritative field** → written to PG node table | +| `openresty_message` | string | | **latest health-description authoritative field** → written to PG node table (**not into CH**) | +| `profile` | object | | host overview, report on change (may throttle) | +| `host_metrics` | object | suggested each beat | L3 resource snapshot | +| `edge_health` | object | suggested each beat | L2 connection time series + status aligned with top level | +| `access_logs` | array | | this beat's incremental access details | +| `buffered` | array | | offline backfill fact batches (see §3.6) | +| `health_events` | array | | edge health events | +| `waf_ip_group_checksums` | map | | for differential sync, not an observability lake | + +**Removed, Server no longer parses (no compatibility layer):** + +| Old Field | Disposition | +| --- | --- | +| `traffic_report` | not in the protocol; not stored | +| `openresty_observation` | not present; connections/status go through `edge_health` | +| `snapshot` | not present; only `host_metrics` | +| `buffered_observability` | not present; only `buffered` | + +### 3.2 `profile` — Host Overview (low frequency) + +Maps to relational `of_node_system_profiles` (or an existing equivalent), **not into the ClickHouse detail lake**. + +```json +{ + "hostname": "edge-1", + "os_name": "linux", + "os_version": "...", + "kernel_version": "...", + "architecture": "amd64", + "cpu_model": "...", + "cpu_cores": 8, + "total_memory_bytes": 16106127360, + "total_disk_bytes": 107374182400, + "uptime_seconds": 864000, + "reported_at_unix": 1720000000 +} +``` + +| Field | Semantics | +| --- | --- | +| hardware/OS description fields | factual readings | +| `reported_at_unix` | Agent collection time (UTC seconds) | + +### 3.3 `host_metrics` — Host Capacity (L3) + +**All readings, no 24h business totals.** +NIC/disk bytes are **kernel cumulative counter raw values** (monotonically increasing, may reset on restart); CPU is an instant percentage; memory/disk usage is current usage. + +```json +{ + "captured_at_unix": 1720000000, + "cpu_usage_percent": 12.5, + "memory_used_bytes": 4294967296, + "memory_total_bytes": 16106127360, + "storage_used_bytes": 50000000000, + "storage_total_bytes": 107374182400, + "disk_read_bytes": 9000000000, + "disk_write_bytes": 12000000000, + "network_rx_bytes": 500000000000, + "network_tx_bytes": 800000000000 +} +``` + +| Field | Type | Semantics | How Server Uses It | +| --- | --- | --- | --- | +| `captured_at_unix` | int64 | sampling time | `captured_at` | +| `cpu_usage_percent` | float | instant CPU% | store directly; average for trends | +| `memory_*` / `storage_*` | int64 | current used/total | store directly; compute usage rate | +| `disk_read_bytes` / `disk_write_bytes` | int64 | **cumulative** IO bytes | store raw; adjacent deltas at query time | +| `network_rx_bytes` / `network_tx_bytes` | int64 | **cumulative** NIC bytes | store raw; adjacent deltas at query time → "host NIC in/outbound" | + +> The Agent is **forbidden** from replacing cumulative values with "this period's delta" before reporting (otherwise Server deltas would be wrong). + +### 3.4 `edge_health` — OpenResty Edge Health (L2) + +**Instant state only, no business throughput.** + +```json +{ + "captured_at_unix": 1720000000, + "status": "healthy", + "message": "", + "connections": 42 +} +``` + +| Field | Type | Semantics | +| --- | --- | --- | +| `status` | string | `healthy` / `unhealthy` / `unknown` (must match top-level `openresty_status`) | +| `message` | string | status description (may be reported; **only backfills PG latest state, not into CH**) | +| `connections` | int64 | stub_status Active connections | + +#### Health-State Authoritative Sources (converged) + +| Data | Authoritative Storage | Description | +| --- | --- | --- | +| **Current** OpenResty health + description | **PG node table** `openresty_status` / `openresty_message` | UI badges, lists, alerts use this | +| **Time series** health status + connections | **CH** `of_node_edge_health` (`status`, `connections`) | connection curves / health history; **no message column** | +| Agent report | top-level status/message + `edge_health` | Server normalizes both statuses aligned; message **only written to PG** | + +So: "is it unhealthy now" → read PG; "connections over the past 24h" → read CH. + +### 3.5 `access_logs[]` — Access Details (L1, single business truth) + +Agent: tail access.log → parse JSON lines → report fields as-is (path may be truncated). + +```json +{ + "logged_at_unix": 1720000001, + "remote_addr": "203.0.113.10", + "host": "www.example.com", + "path": "/api/v1/ping", + "status_code": 200, + "bytes_sent": 1024, + "request_length": 128, + "request_time_ms": 15, + "user_agent": "Mozilla/5.0 ...", + "cache_status": "HIT" +} +``` + +| Field | Type | Required | Source (OpenResty) | Business Meaning | +| --- | --- | --- | --- | --- | +| `logged_at_unix` | int64 | ✅ | parse `$time_iso8601` | request completion time | +| `remote_addr` | string | ✅ | `$remote_addr` | client IP → UV | +| `host` | string | ✅ | `$host` | domain → Zone ownership | +| `path` | string | ✅ | `$request_uri`, Agent may truncate | path | +| `status_code` | int | ✅ | `$status` | status code | +| `bytes_sent` | int64 | ✅ | **`$body_bytes_sent`** | **data provided** (response body) | +| `request_length` | int64 | suggested | `$request_length` | **data received** | +| `request_time_ms` | int64 | optional | `$request_time * 1000` | latency; default 0 | +| `user_agent` | string | suggested | `$http_user_agent` | UA; may truncate on store | +| `cache_status` | string | suggested | **`$upstream_cache_status`** | edge cache result (see §3.5.1) | + +**Explicitly not reported by Agent (written by Server):** + +* `region` / country: GeoIP resolved at insert time +* `id` / `created_at`: Server-generated +* `node_id`: from payload / auth context + +**Explicitly not reported:** + +* `upstream_addr` / origin address / `origin_fetched`: no origin-endpoint tracking; "did it fetch from origin" is only derived from `cache_status` at the control plane (§3.5.1) + +### 3.5.1 `cache_status` — Cache Hit and Origin Fetch (detail-first) + +**Goal (phase 1):** access log details/list can show "cache hit / origin fetch / no cache used". +**Caliber:** only store the raw OpenResty `$upstream_cache_status`; **no upstream address reported**. + +#### Raw Values (stored) + +| Value | Meaning (OpenResty) | +| --- | --- | +| `HIT` | cache hit | +| `MISS` | miss, fetched from upstream | +| `BYPASS` | cache skipped (e.g. method/cookie/policy caused `$openflare_skip_cache`) | +| `EXPIRED` | expired then origin fetch | +| `STALE` | served stale | +| `UPDATING` | background updating, may return old cache | +| `REVALIDATED` | revalidated, still used cache | +| `-` or empty | didn't pass through `proxy_cache` (e.g. Pages local static, non-proxy location) | + +#### UI Three-State Derivation (not stored) + +Control-plane display uses derived enum `cache_outcome`, **not written to CH**: + +| Three-State | Condition (`cache_status`) | Suggested List Label | +| --- | --- | --- | +| **Cache hit** | `HIT` / `STALE` / `REVALIDATED` / `UPDATING` | hit | +| **Origin fetch** | `MISS` / `EXPIRED` | origin | +| **No cache used** | `BYPASS` / `-` / `""` | not cached | + +Details can show both the three-state and the raw `cache_status`. + +#### Boundaries + +* Pages static / locations without `proxy_cache`: mostly empty or `-` → **no cache used**, must not be labeled "hit". +* Detail pages show cache state; hit-rate dashboards and hourly dimensions can extend from the same column. + +**Per-heartbeat count suggestion:** + +* Soft cap e.g. 2000 lines/beat; overflow goes into `buffered` next batch, **forbidden** to compress into a TrafficReport in the Agent. + +### 3.6 `buffered[]` — Offline Backfill (facts only) + +```json +{ + "captured_at_unix": 1719999900, + "host_metrics": { }, + "edge_health": { }, + "access_logs": [ ] +} +``` + +| Field | Description | +| --- | --- | +| `captured_at_unix` | batch collection/buffer time, used for ack and dedup window | +| `host_metrics` / `edge_health` / `access_logs` | same structures as the main payload; empty blocks may be omitted | + +**Forbidden** to carry `traffic_report` or rx/tx throughput in buffered. + +### 3.7 `health_events[]` + +```json +{ + "event_type": "openresty_unhealthy", + "severity": "critical", + "message": "...", + "triggered_at_unix": 1720000000, + "metadata": { } +} +``` + +Written to the relational health-event table (existing model suffices), not into the access log lake. + +### 3.8 Go Protocol Structures + +```go +// pkg/protocol/agent.go (current implementation) + +type NodePayload struct { + SchemaVersion int `json:"schema_version,omitempty"` + NodeID string `json:"node_id"` + Name string `json:"name"` + IP string `json:"ip"` + Version string `json:"version"` + ExtVersion string `json:"ext_version"` + CurrentVersion string `json:"current_version"` + LastError string `json:"last_error"` + OpenrestyStatus string `json:"openresty_status"` // PG latest-state authority + OpenrestyMessage string `json:"openresty_message"` // PG latest-state authority; not into CH + Profile *NodeSystemProfile `json:"profile,omitempty"` + HostMetrics *NodeHostMetrics `json:"host_metrics,omitempty"` + EdgeHealth *NodeEdgeHealth `json:"edge_health,omitempty"` + AccessLogs []NodeAccessLog `json:"access_logs,omitempty"` + Buffered []BufferedFacts `json:"buffered,omitempty"` + HealthEvents []NodeHealthEvent `json:"health_events"` + WAFIPGroupChecksums map[string]string `json:"waf_ip_group_checksums,omitempty"` +} + +type NodeHostMetrics struct { + CapturedAtUnix int64 `json:"captured_at_unix"` + CPUUsagePercent float64 `json:"cpu_usage_percent"` + MemoryUsedBytes int64 `json:"memory_used_bytes"` + MemoryTotalBytes int64 `json:"memory_total_bytes"` + StorageUsedBytes int64 `json:"storage_used_bytes"` + StorageTotalBytes int64 `json:"storage_total_bytes"` + DiskReadBytes int64 `json:"disk_read_bytes"` + DiskWriteBytes int64 `json:"disk_write_bytes"` + NetworkRxBytes int64 `json:"network_rx_bytes"` + NetworkTxBytes int64 `json:"network_tx_bytes"` +} + +type NodeEdgeHealth struct { + CapturedAtUnix int64 `json:"captured_at_unix"` + Status string `json:"status"` + Message string `json:"message"` + Connections int64 `json:"connections"` +} + +type NodeAccessLog struct { + LoggedAtUnix int64 `json:"logged_at_unix"` + RemoteAddr string `json:"remote_addr"` + Host string `json:"host"` + Path string `json:"path"` + UserAgent string `json:"user_agent,omitempty"` + CacheStatus string `json:"cache_status,omitempty"` // $upstream_cache_status + StatusCode int `json:"status_code"` + BytesSent int64 `json:"bytes_sent"` // body_bytes_sent, data provided + RequestLength int64 `json:"request_length"` // data received + RequestTimeMs int64 `json:"request_time_ms"` // optional +} + +type BufferedFacts struct { + CapturedAtUnix int64 `json:"captured_at_unix"` + HostMetrics *NodeHostMetrics `json:"host_metrics,omitempty"` + EdgeHealth *NodeEdgeHealth `json:"edge_health,omitempty"` + AccessLogs []NodeAccessLog `json:"access_logs,omitempty"` +} +``` + +--- + +## 4. Server Parsing and Storage Flow + +### 4.1 Entry Points + +* HTTP: `POST /api/v1/agent/...` heartbeat (existing path) +* WebSocket: `type=status` payload = `NodePayload` +* Auth: `X-Agent-Token` → binds `node_id` (payload.node_id must match the token node) + +### 4.2 Processing Pipeline (single payload) + +```text +1. Deserialize NodePayload +2. Normalize + - schema_version < 2: + host_metrics ← snapshot + edge_health.status ← openresty_status + edge_health.connections ← openresty_observation.connections (if present) + traffic_report → drop + openresty_observation.rx/tx → drop + buffered ← buffered_observability + - path truncated again, status clamped, negative bytes → 0 +3. Relational transaction (node latest state) + - update node online time, IP, version, edge_health.status/message + - upsert profile (if present) + - insert health_events (if present) +4. ClickHouse async batch (log on failure; doesn't block heartbeat response config delivery) + a. access_logs + buffered[].access_logs + → fill region (GeoIP) + → assign snowflake id + → BatchInsert of_node_access_logs + b. host_metrics + buffered[].host_metrics + → of_node_metric_snapshots + c. edge_health + buffered[].edge_health + → of_node_edge_health (connections + status snapshot only, optional) +5. Return heartbeat response (settings / active_config / waf diff) +6. If using buffer ack: confirm by the list of buffered.captured_at_unix +``` + +### 4.3 Normalization Rules (hard constraints) + +| Rule | Behavior | +| --- | --- | +| `logged_at` ahead of now+5m | clamp to now or drop the line (pick one in implementation and unit-test it) | +| `logged_at` older than now−TTL | still writable; relies on table TTL cleanup | +| empty `host` | allowed, aggregated into "unassigned" | +| `bytes_sent` / `request_length` < 0 | set to 0 | +| single batch access_logs > N | truncate and alert-metric it (or only queue into buffer), never switch to pre-aggregation | +| duplicate backfill | CH tolerates a few duplicate rows; queries approximate with sum (no forced exact dedup) | + +### 4.4 Field Mapping Table (report → table) + +| Report Path | Target Storage | Columns | +| --- | --- | --- | +| `access_logs[]` | CH `of_node_access_logs` | see §5.1 | +| `host_metrics` | CH `of_node_metric_snapshots` | see §5.2 | +| `edge_health` | CH `of_node_edge_health` + PG node latest state | see §5.3 / §5.6 | +| `profile` | PG `of_node_system_profiles` | existing columns | +| `health_events` | PG health-event table | existing model | +| `waf_ip_group_checksums` | not into observability tables | sync logic | +| `traffic_report` (legacy) | **not written** | — | +| `openresty_rx/tx` (legacy) | **not written** | — | + +### 4.5 Query Side (no new "business outbound" column) + +| Product Metric | SQL Semantics (sketch) | +| --- | --- | +| Data provided | `sum(bytes_sent)` | +| Data received | `sum(request_length)` | +| Request count | `count()` | +| UV | `uniqExact(remote_addr)` | +| 5xx | `countIf(status_code >= 500)` | +| By domain/status/region | `GROUP BY host / status_code / region` | +| Host NIC outbound | non-negative delta over `network_tx_bytes` per node in time order, then sum | +| OpenResty connections | `of_node_edge_health.connections` latest or average | + +--- + +## 5. Table Structures (DDL) + +> Engine and TTL tend to match production: access logs 90 days, metrics 30 days. +> `id` uses control-plane Snowflake/unique UInt64. + +### 5.1 L1 Fact Table: `of_node_access_logs` + +```sql +CREATE TABLE IF NOT EXISTS of_node_access_logs +( + id UInt64, + node_id String, + logged_at DateTime64(3, 'UTC'), + remote_addr String, + region String, -- Server GeoIP writes, Agent doesn't send + host String, + path String, + user_agent String DEFAULT '', -- $http_user_agent + cache_status String DEFAULT '', -- $upstream_cache_status + status_code Int32, + bytes_sent UInt64, -- data provided (body) + request_length UInt64 DEFAULT 0, -- data received + request_time_ms UInt32 DEFAULT 0, -- optional + created_at DateTime64(3, 'UTC') +) +ENGINE = MergeTree() +PARTITION BY toYYYYMM(logged_at) +ORDER BY (node_id, logged_at, host, status_code, remote_addr) +TTL toDateTime(logged_at) + INTERVAL 90 DAY +SETTINGS index_granularity = 8192; +``` + +| Column | Type | Source | +| --- | --- | --- | +| `id` | UInt64 | Server | +| `node_id` | String | auth/payload | +| `logged_at` | DateTime64(3) | `logged_at_unix` | +| `remote_addr` | String | report | +| `region` | String | Server GeoIP | +| `host` | String | report | +| `path` | String | report | +| `user_agent` | String | report (nullable) | +| `cache_status` | String | report (nullable) → **cache status** | +| `status_code` | Int32 | report | +| `bytes_sent` | UInt64 | report → **data provided** | +| `request_length` | UInt64 | report → **data received** | +| `request_time_ms` | UInt32 | report optional | +| `created_at` | DateTime64(3) | Server now | + +**Migration:** the current table already has `bytes_sent` / `request_length` / `request_time_ms` / `user_agent`; cache status adds: + +```sql +ALTER TABLE of_node_access_logs + ADD COLUMN IF NOT EXISTS cache_status String DEFAULT ''; +``` + +### 5.2 L1 Hourly Rollup (Server-side MV) + +**Agent forbidden to write.** Serves dashboard/node 24h fast queries of request count, error count, bytes. + +**Implemented choice: `SummingMergeTree` + no UV column.** + +```sql +CREATE TABLE IF NOT EXISTS of_access_log_hourly +( + node_id String, + hour DateTime('UTC'), + host String, + request_count UInt64, + error_count UInt64, + bytes_sent UInt64, + request_length UInt64 +) +ENGINE = SummingMergeTree() +PARTITION BY toYYYYMM(hour) +ORDER BY (node_id, hour, host) +TTL hour + INTERVAL 90 DAY; + +CREATE MATERIALIZED VIEW IF NOT EXISTS of_access_log_hourly_mv +TO of_access_log_hourly +AS +SELECT + node_id, + toStartOfHour(logged_at) AS hour, + host, + toUInt64(count()) AS request_count, + toUInt64(countIf(status_code >= 500)) AS error_count, + sum(bytes_sent) AS bytes_sent, + sum(request_length) AS request_length +FROM of_node_access_logs +GROUP BY node_id, hour, host; +``` + +Historical hours (details stored before the MV existed) need a one-time backfill, see migration `202607180003_backfill_access_log_hourly.sql` (ANTI JOIN to prevent duplicates). + +#### UV Policy (must follow) + +| Scenario | Data Source | Algorithm | Notes | +| --- | --- | --- | --- | +| **Window total UV** (dashboard totals, node cards, Zone totals) | `of_node_access_logs` details | `uniqExact(remote_addr)` (`TrafficSummary` / node aggregation) | **single authority**; never sum hourly UV | +| **24h trend line request/error/bytes** | `of_access_log_hourly` preferred, fall back to detail buckets | `sum(request_count)` etc. | hourly path **doesn't fill** `unique_visitor_count` (always 0) | +| **24h trend per-hour UV** | detail bucket path only | in-bucket `uniqExact` | when using hourly, UI should show empty/0 or hide the UV series; **forbidden** to `sum(UV)` over hourly rows | + +**Why hourly doesn't store UV:** + +1. `SummingMergeTree` can only safely merge addable counts; `uniqExact` across parts needs `AggregatingMergeTree` + state, heavier to implement and query. +2. Even if hourly UV were stored, **summing over multi-hour windows severely overestimates** (the same IP is counted once per hour). +3. Product "24h unique visitors" only recognizes whole-window `uniqExact`; the trend chart's main series are requests/errors/bytes — per-hour UV is not a primary metric. + +### 5.3 L3 Fact Table: `of_node_metric_snapshots` (kept, semantics clarified) + +```sql +CREATE TABLE IF NOT EXISTS of_node_metric_snapshots +( + id UInt64, + node_id String, + captured_at DateTime64(3, 'UTC'), + cpu_usage_percent Float64, + memory_used_bytes Int64, + memory_total_bytes Int64, + storage_used_bytes Int64, + storage_total_bytes Int64, + disk_read_bytes Int64, -- cumulative raw + disk_write_bytes Int64, + network_rx_bytes Int64, -- cumulative raw → host NIC inbound + network_tx_bytes Int64, -- cumulative raw → host NIC outbound + created_at DateTime64(3, 'UTC') +) +ENGINE = MergeTree() +PARTITION BY toYYYYMM(captured_at) +ORDER BY (node_id, captured_at, id) +TTL toDateTime(captured_at) + INTERVAL 30 DAY +SETTINGS index_granularity = 8192; +``` + +Columns match production; **docs and API must label `network_*` as host NIC cumulative values**. + +### 5.4 L3 Hourly Rollup: `of_node_metric_capacity_hourly` (kept) + +Existing min/max used for cumulative-counter hourly increment approximation + CPU/memory averages. Logic unchanged: + +* `network_tx_max - network_tx_min` ≈ that hour's host outbound +* **must not** be used for "data provided" + +### 5.5 L2 Fact Table: `of_node_edge_health` (new, replaces throughput-style openresty table) + +```sql +CREATE TABLE IF NOT EXISTS of_node_edge_health +( + id UInt64, + node_id String, + captured_at DateTime64(3, 'UTC'), + status LowCardinality(String), -- healthy / unhealthy / unknown + connections Int64, + created_at DateTime64(3, 'UTC') +) +ENGINE = MergeTree() +PARTITION BY toYYYYMM(captured_at) +ORDER BY (node_id, captured_at, id) +TTL toDateTime(captured_at) + INTERVAL 30 DAY +SETTINGS index_granularity = 8192; +``` + +| Column | Description | +| --- | --- | +| `status` | instant health (same source as PG current state; for time series, not the sole UI authority) | +| `connections` | current connection count | + +**No** `message` column (description only in PG latest state). +**No** `openresty_rx_bytes` / `openresty_tx_bytes`. + +### 5.6 Relational DB (node latest state, not an analytics lake) + +Separate from the observability lake, keeping "latest one": + +| Table (logical name) | Purpose | Key Columns | +| --- | --- | --- | +| `of_nodes` (or current node table) | online, version, IP | `last_seen_at`, `openresty_status`, `openresty_message`, `agent_version` | +| `of_node_system_profiles` | profile upsert | hostname, cpu_cores, total_memory_bytes, ... | +| health-event table | `health_events` | event_type, severity, message, triggered_at | + +> Actual physical table names follow the repo's existing GORM models; this design doesn't force renames, only forces **business throughput no longer written into node tables**. + +### 5.7 Deprecated Tables (stop writing → delete after TTL) + +| Table | Reason | Replacement | +| --- | --- | --- | +| `of_node_request_reports` | Agent pre-aggregation | `of_node_access_logs` + hourly | +| `of_node_traffic_hourly` + MV | depends on request_reports | `of_access_log_hourly` | +| `of_node_obs_openresty` | contains business rx/tx | `of_node_edge_health` | +| `of_node_openresty_hourly` + MV | business throughput deltas | `of_access_log_hourly` bytes_* | + +Relay-specific `of_node_obs_frps` / `of_node_obs_frpc` **kept** (not this Agent's main path, but same CH observability). + +--- + +## 6. Table-Protocol Cross-Reference + +| Product Concept | Protocol Field | Table.Column | Aggregation | +| --- | --- | --- | --- | +| Data provided | `access_logs[].bytes_sent` | `of_node_access_logs.bytes_sent` | `sum` | +| Data received | `access_logs[].request_length` | `...request_length` | `sum` | +| Request count | row count | — | `count` | +| UV (window total) | `remote_addr` | same details | `uniqExact` (**forbidden** to sum hourly UV) | +| Top domains | `host` | same | `group by` | +| Status distribution | `status_code` | same | `group by` | +| Source region | — | `region` (Server) | `group by` | +| Host NIC outbound | `host_metrics.network_tx_bytes` | `of_node_metric_snapshots.network_tx_bytes` | time-series delta | +| Host NIC inbound | `network_rx_bytes` | same | delta | +| Disk read/write | `disk_*_bytes` | same | delta | +| CPU/memory | instant fields | same | avg | +| OpenResty connections | `edge_health.connections` | `of_node_edge_health.connections` | latest/avg | +| OpenResty health | `edge_health.status` | node table + optional CH | latest | + +**Mappings that no longer exist:** + +| Old Concept | Old Field | Disposition | +| --- | --- | --- | +| OpenResty outbound | `openresty_tx_bytes` | removed; use data provided | +| OpenResty inbound | `openresty_rx_bytes` | removed; use data received | +| Window request report | `traffic_report` | removed | + +--- + +## 7. OpenResty Log Format (aligned with details) + +Target `log_format` (ensures the `bytes_sent` key = body; includes UA and cache status): + +```nginx +log_format openflare_json escape=json + '{"ts":"$time_iso8601","host":"$host","path":"$request_uri",' + '"remote_addr":"$remote_addr","status":$status,' + '"request_time":$request_time,' + '"bytes_sent":$body_bytes_sent,"request_length":$request_length,' + '"user_agent":"$http_user_agent",' + '"cache_status":"$upstream_cache_status"}'; +``` + +Agent parsing: + +* `ts` → `logged_at_unix` +* `bytes_sent` → protocol `bytes_sent` (provided) +* `request_length` → protocol `request_length` +* `request_time` → optional `request_time_ms = round(sec * 1000)` +* `user_agent` → protocol `user_agent` +* `cache_status` → protocol `cache_status` (passed through as-is, no three-state compression) + +--- + +## 8. Upgrade Strategy (no compatibility layer) + +| Item | Strategy | +| --- | --- | +| Agent upgrade | **destroy-and-recreate** preferred; **binary replacement** allowed | +| Protocol | only schema v2 fields; legacy JSON fields not parsed | +| Local observability buffer | if still in old format (containing `snapshot` / `openresty_observation` / `traffic_report`) or corrupt → **delete the file wholesale**, rebuild at runtime | +| Read path | business APIs **only read** access_logs (and hourly); current health reads PG; connection series reads CH edge_health | +| Old Agents | must upgrade; the control plane provides no v1 dual-read path | + +--- + +## 9. Example: Storage Result of One Heartbeat + +**Agent report (excerpt):** + +```json +{ + "schema_version": 2, + "node_id": "n1", + "host_metrics": { + "captured_at_unix": 1720000000, + "cpu_usage_percent": 10, + "memory_used_bytes": 1, + "memory_total_bytes": 2, + "storage_used_bytes": 3, + "storage_total_bytes": 4, + "disk_read_bytes": 100, + "disk_write_bytes": 200, + "network_rx_bytes": 1000, + "network_tx_bytes": 2000 + }, + "edge_health": { + "captured_at_unix": 1720000000, + "status": "healthy", + "message": "", + "connections": 5 + }, + "access_logs": [ + { + "logged_at_unix": 1720000001, + "remote_addr": "1.1.1.1", + "host": "a.example.com", + "path": "/", + "status_code": 200, + "bytes_sent": 500, + "request_length": 80 + } + ] +} +``` + +**Written:** + +1. PG node latest state: `openresty_status` / `openresty_message` (if reported) +2. `of_node_metric_snapshots` 1 row (network_tx=2000 cumulative) +3. `of_node_edge_health` 1 row (status + connections=5; **no message**) +4. `of_node_access_logs` 1 row (bytes_sent=500, request_length=80, region filled by Server) +5. MV asynchronously counts into `of_access_log_hourly` + +**Query 24h data provided:** `sum(bytes_sent)` → at least 500 (plus history) +**Query host outbound:** delta over snapshots; **no forced equality** with 500. + +--- + +## 10. Revision History + +| Date | Notes | +| --- | --- | +| 2026-07-17 | initial draft: protocol v2, Server storage pipeline, CH/relational target table structures and deprecated table list | diff --git a/docs/en/design/observability-design.md b/docs/en/design/observability-design.md new file mode 100644 index 00000000..4022b61f --- /dev/null +++ b/docs/en/design/observability-design.md @@ -0,0 +1,552 @@ +# Edge Observability and Business Traffic Statistics Refactor + +You will learn: the problems this refactor solves (dashboard "OpenResty outbound" vs "Zone data provided" inconsistency, field and aggregation redundancy), and how the target architecture makes **the Agent report only facts and the Server interpret facts**, with access logs as the single source of truth for business traffic. + +--- + +## 1. Goals + +### 1.1 Problems to Solve + +1. **Dual sources of truth**: business throughput comes from both access-log aggregation and OpenResty observability deltas, and the numbers never match long-term. +2. **Agent over-computes**: the edge pre-aggregates `TrafficReport`, throughput accumulation, and the control plane aggregates again — semantics are hard to evolve and reconcile. +3. **Field semantic overlap**: "OpenResty outbound" and "data provided" are the same business problem for users, but the system uses two field sets and two pipelines. +4. **Instant vs cumulative mixed**: 60-second window counts are treated as process cumulative values for 24h deltas, causing severe underestimation. +5. **UI induces wrong comparisons**: the dashboard and Zone page use similar "traffic/data" wording without declaring scope and caliber differences. + +### 1.2 Refactor Goals + +| Goal | Description | +| --- | --- | +| **Single business truth** | request count, data provided, UV, status distribution, Top domains etc. **only** derived from access logs (and Server-side rollups) | +| **Agent reports only facts** | detail logs + machine readings + health snapshots; **business UV/TopN/24h totals pre-aggregation is forbidden** | +| **Field convergence** | one business concept maps to one authoritative field; machine NIC and business delivery strictly separated by name | +| **Reconcilable** | global "data provided" ≈ sum of per-Zone "data provided" (difference only from unbound/unknown Hosts) | +| **Evolvable** | changing time windows, TopN, ownership rules only changes the Server, not the Agent | + +### 1.3 Non-Goals (outside this design) + +* Building a general log platform, full-log long-term archive, or search product. +* Replacing ClickHouse / removing the analytics DB dependency. +* Reworking Relay / OpenFlared host metric collection (principles align, but not in this round's protocol main path). +* Real-time streaming alert engine, APM tracing (the OpenTelemetry server side already exists and is orthogonal to this business traffic model). + +--- + +## 2. Scope and Constraints + +### 2.1 Product Constraints (inherited) + +* Single-tenant, single globally active config; observability introduces no multi-tenant billing isolation. +* Access logs and time-series observability use the switchable log primary DB (ClickHouse by default; switchable to PostgreSQL/SQLite), see [Log Store Decoupling](./logstore.md). +* Agent has no inbound control, Pull model; during offline periods local OpenResty keeps serving, and observability can buffer locally and backfill. + +### 2.2 Engineering Constraints + +* Agent stays lightweight: parse log lines, read `/proc`, health checks; no business analysis. +* Control-plane API errors still use the unified envelope and `response.Abort*`. +* Access log field changes must update both the OpenResty `log_format` and the Agent parser simultaneously; Agent and control plane ship at the same version, no legacy protocol parsing. + +--- + +## 3. Design Principles + +### Principle P1: Agent Reports Facts, Server Interprets Facts + +```text +Agent = collection + reliable delivery (raw / near-raw) +Server = storage + aggregation + ownership + trends + reconciliation +``` + +**Allowed edge processing (collection)** + +* Parsing JSON access.log lines into structured fields +* path length caps, dropping invalid lines, skipping observability-port's own requests +* Reading NIC/CPU/memory counters as **raw values** +* Batching, compression, offline buffering and retries + +**Forbidden edge processing (business computation)** + +* UV / Top domains / status histograms / window request_count as authoritative metrics +* Maintaining "business in/out cumulative" for the dashboard +* Zone / domain ownership stats, country distribution (country can be resolved at Server insert time) + +### Principle P2: Single Truth for Business Traffic = Access Logs + +| Business Question | Single Answer | +| --- | --- | +| How much data was provided | `sum(bytes_sent)` | +| How many requests | `count()` | +| How many unique visitors | `uniqExact(remote_addr)` (or product-defined hashing) | +| Status codes / Top domains | `group by` on logs | + +### Principle P3: Three Metric Layers Never Mixed + +| Layer | Name | Purpose | Typical Fields | +| --- | --- | --- | --- | +| L1 Business delivery | Business Traffic | user & Zone reconciliation, dashboard business trends | access log | +| L2 Edge health | Edge Health | is OpenResty alive, current connections | status, connections | +| L3 Host capacity | Host Capacity | capacity planning, is the machine saturated | CPU, memory, disk, **NIC** | + +Never name L3 NIC or L2 instant counts as "data provided"; never draw L1 and L3 on the same summary card without labeling semantics. + +### Principle P4: One Business Concept, One Field + +* **Data provided** ≡ response body delivered ≡ "OpenResty outbound (business meaning)" in legacy copy → **keep only `bytes_sent` aggregation** +* **Data received** (optional) ≡ request-side volume → log `request_length` aggregation +* **Host outbound** ≡ `network_tx` delta, copy must include "host/NIC" + +--- + +## 4. Pre-Refactor Problems (Baseline) + +### 4.1 Pre-Refactor Data Flow (redundant) + +```text +One HTTP request + │ + ├─ access.log line + │ → Agent tail → AccessLogs[] + │ → CH of_node_access_logs + │ → Zone "data provided" ✅ + │ + ├─ Lua shared dict window/cumulative counts + │ → /openflare/observability + │ → TrafficReport + OpenrestyObservation(rx/tx) + │ → CH request_reports / obs_openresty + │ → dashboard "OpenResty in/outbound" ❌ easily inconsistent with Zone + │ + ├─ second access.log aggregation (fallback when observability endpoint fails) + │ → yet another TrafficReport / throughput + │ + └─ host network_rx/tx + → Snapshot → "host" curve in network trends +``` + +### 4.2 Field Overlap + +| User Perception | System Field A | System Field B | Problem | +| --- | --- | --- | --- | +| Outbound / provided | `openresty_tx_bytes` | `bytes_sent` | duplicate business semantics | +| Inbound | `openresty_rx_bytes` | `request_length` (log) | duplicate business semantics | +| Request count | `TrafficReport.request_count` | `count(access_logs)` | duplicate aggregation, window easily double-counted | +| Outbound (machine) | `network_tx_bytes` | (no business equivalent) | should be named separately, never reconciled with business | + +### 4.3 Typical Failure Modes + +1. Window counts treated as cumulative deltas → 24h business throughput severely underestimated. +2. Hourly rollup `max−min` broken for resetting counters. +3. Zone uses logs, dashboard uses observability → users think the system is wrong. +4. Changing caliber requires syncing Lua, Agent state accumulation, Server deltas, and frontend copy. + +--- + +## 5. Target Architecture + +### 5.1 Target Data Flow + +```mermaid +flowchart TB + subgraph edge [Edge Node] + OR[OpenResty] + LOG[access.log] + PROC[host /proc and disk] + STUB[stub_status connections] + AG[Agent] + OR -->|log_format writes line| LOG + LOG -->|tail incremental details only| AG + PROC -->|reading snapshots| AG + STUB -->|instant connections| AG + OR -->|health probe| AG + end + + subgraph server [Control-Plane Server] + HB[Heartbeat / WS receive] + CH[(ClickHouse)] + AGG[Aggregation query layer] + API[Admin API] + HB --> CH + CH --> AGG + AGG --> API + end + + subgraph ui [Admin Panel] + DASH[Dashboard: global business trends] + ZONE[Zone: filter by domain] + NODE[Node: host resources + health] + end + + AG -->|AccessLogs + HostSnapshot + Health| HB + API --> DASH + API --> ZONE + API --> NODE +``` + +### 5.2 Responsibility Matrix + +| Capability | Agent | Server | Frontend | +| --- | --- | --- | --- | +| Write access.log | OpenResty | — | — | +| Read and report details | ✅ | store | — | +| sum/count/uniq/TopN | ❌ | ✅ | display | +| Zone domain filtering | ❌ | ✅ | select Zone | +| Host CPU/memory/NIC | read raw values and report | delta/average | node/dashboard resource area | +| OpenResty connections | read instant and report | latest value | node health | +| Business 24h in/outbound | ❌ | log aggregation | uniformly called "data provided/received" | + +--- + +## 6. Metrics and Field Model + +### 6.1 Authoritative Field Table (target) + +#### L1 Business Delivery (from access logs) + +| Concept | Storage Field | Aggregation | Display Name | +| --- | --- | --- | --- | +| Request time | `logged_at` | window filter | — | +| Node | `node_id` | group | — | +| Client IP | `remote_addr` | `uniq` → UV | Unique visitors | +| Host | `host` | group / Zone mapping | Domain | +| Path | `path` | optional | — | +| Status code | `status_code` | group | Status distribution | +| **Data provided** | **`bytes_sent`** | **`sum`** | **Data provided** | +| **Data received** | **`request_length`** | **`sum`** | **Data received** (optional display) | +| Region | `region` (resolved & written by Server) | group | Source region | + +> Note: the JSON key in the OpenResty `log_format` may keep the name `bytes_sent`; the value must come from **`$body_bytes_sent`** (consistent with production), representing response body delivered, i.e. "data provided". + +#### L2 Edge Health (instant; no 24h business totals) + +| Concept | Field | Description | +| --- | --- | --- | +| OpenResty health | `openresty_status` / message | existing | +| Current connections | `openresty_connections` | stub_status | +| (optional) rough recent-window QPS | node detail "right now" only, **never** authoritative 24h totals | if implemented must be labeled "instant" | + +#### L3 Host Capacity + +| Concept | Field | Display Name | +| --- | --- | --- | +| CPU / memory / disk usage | `host_metrics` | keep | +| NIC cumulative bytes | `network_rx_bytes` / `network_tx_bytes` | **Host NIC in/outbound** | +| Disk IO cumulative | `disk_read_bytes` / `disk_write_bytes` | Disk read/write | + +### 6.2 Removed Fields (no compatibility layer) + +| Original Field | Disposition | Reason | +| --- | --- | --- | +| `openresty_tx_bytes` / `openresty_rx_bytes` | **removed** | business bytes follow access logs | +| `TrafficReport` and TopN/window UV | **removed** | edge pre-aggregation | +| Agent state business lifetime accumulators | removed | violates P1 | +| Lua shared dict business throughput/window request counts | removed | not the delivery main path | + +### 6.3 Naming Reference (frontend copy enforced) + +| Forbidden Copy | Correct Copy | Data Source | +| --- | --- | --- | +| OpenResty outbound (business volume) | **Data provided** | `sum(bytes_sent)` | +| OpenResty inbound (business volume) | **Data received** | `sum(request_length)` | +| Network outbound (unspecified) | **Host NIC outbound** | `network_tx` delta | +| Two cards: data provided vs outbound | **keep only one business card** | logs | + +--- + +## 7. Agent Design + +### 7.1 Heartbeat Payload (target protocol) + +Keep and strengthen: + +```text +NodePayload + identity / version / openresty_status / openresty_message # latest state → PG + profile # host overview (low frequency) + host_metrics # L3 resource readings (incl. NIC cumulative raw values) + edge_health # L2: status + connections (CH time series; message not in CH) + access_logs[] # L1 details (main path) + health_events[] + buffered[] # buffered facts above, not reports + waf_ip_group_checksums +``` + +Removed from the protocol (no compatibility layer): + +```text +traffic_report +openresty_observation +snapshot / buffered_observability aliases +``` + +### 7.2 Access Log Reporting Requirements + +Each detail at minimum contains: + +| Field | Required | Note | +| --- | --- | --- | +| `logged_at_unix` | ✅ | request completion time | +| `remote_addr` | ✅ | UV | +| `host` | ✅ | Zone mapping | +| `path` | ✅ | may be truncated | +| `status_code` | ✅ | | +| `bytes_sent` | ✅ | body bytes, data provided | +| `request_length` | ✅ | data received | + +Agent responsibilities: + +1. Tail `access.log` by offset (reset offset on truncation/rotation, **only report new lines still present in the file**). +2. Parse into structured form, batch into heartbeat / WS. +3. Offline writes to local buffer, backfill by window once connected. +4. **No sum/count/uniq on details.** + +### 7.3 Host Snapshot + +* Keep reporting NIC/disk **cumulative counter raw values** (not business pre-aggregation). +* Server does non-negative deltas between adjacent samples → host trends. +* This is unrelated to "data provided"; the UI must display it in a separate section. + +### 7.4 OpenResty Local Observability + +Converged state: + +* Keep: health checks, `stub_status` current connections. +* The main path no longer relies on `log.lua` shared dict business counts; `/openflare/observability` only returns health and connection snapshots, not business report sources. + +### 7.5 Relationship with the Agent Design Doc + +This design strengthens "pure data landing" in [Agent & Publish Model](./agent-design.md): + +* Config and certificates: land and report applied state. +* Observability: only carry facts, not business conclusions. + +--- + +## 8. Server Design + +### 8.1 Storage + +| Input | Table | Description | +| --- | --- | --- | +| `access_logs[]` | `of_node_access_logs` | authoritative business details | +| `host_metrics` | `of_node_metric_snapshots` | L3; NIC/disk cumulative | +| `openresty_status` / `openresty_message` | **PG node table** | L2 **latest-state authority** (message only here) | +| `edge_health` | `of_node_edge_health` | L2 time series: status + connections (**no message**) | + +GeoIP: continue resolving `remote_addr` → `region` in the Server insert path, not in the Agent. + +### 8.2 Aggregation Layer (unified) + +All business trends and Zone stats share the same query semantics: + +```text +filter: logged_at ∈ [since, until] +optional: node_id / host IN (...) +metrics: + request_count = count() + unique_visitors = uniqExact(remote_addr) + bytes_provided = sum(bytes_sent) -- data provided + bytes_received = sum(request_length) -- data received + series folded by hour/bucket + distributions by status_code / host / region +``` + +Implementation locations: + +* Zone: `GET .../zones/:id/stats` (existing, align field naming) +* Dashboard: overview traffic / business network trends **switch to the same aggregation** (global, no host filter or Top filter) +* Node detail: business volume = the same aggregation filtered by that `node_id`; host NIC still uses metric deltas + +### 8.3 Derived Rollups (optional performance path) + +When detail queries over all nodes for 24h are too heavy, allow **Server-side** materialized views: + +```text +of_access_log_hourly + (hour, node_id, host, request_count, bytes_sent, bytes_received, ...) +``` + +Constraints: + +* Derived only by CH from `of_node_access_logs`; **Agent is forbidden from writing this table directly**. +* Zone / dashboard prefer reading the rollup, falling back to details (similar to the existing metric hourly policy). + +### 8.4 Decommissioned Analytics Paths + +| Path | After Migration | +| --- | --- | +| `BuildNetworkTrendPoints` delta on openresty_rx/tx | deleted, or keep only `network_*` host curves | +| `of_node_obs_openresty` throughput fields | stop writing; drop table or shrink columns after TTL expiry | +| `of_node_request_reports` + traffic hourly | business trends no longer depend on it; table can be deprecated wholesale | +| Dashboard compact openresty_tx series | change to bytes_provided series | + +--- + +## 9. API and Frontend + +### 9.1 Semantically Unified Response Fields + +Business stats APIs should uniformly use: + +```json +{ + "request_count": 0, + "unique_visitors": 0, + "bytes_provided": 0, + "bytes_received": 0, + "series": [ + { + "bucket_started_at": "...", + "request_count": 0, + "unique_visitors": 0, + "bytes_provided": 0, + "bytes_received": 0 + } + ] +} +``` + +API business byte fields use `bytes_provided` / `bytes_received` (access-log aggregation); no more openresty throughput aliases. + +### 9.2 Dashboard + +* **Business area**: request trend, data provided, data received (optional), status codes, Top domains, source regions — all L1. +* **Resource area**: CPU/memory, **host NIC**, disk IO — all L3. +* **Forbidden**: showing "OpenResty in/outbound" in the business area as a metric reconciled with Zone. + +Suggest splitting or retitling "24-hour network and disk trends": + +* "24-hour business traffic" → `bytes_provided` / `bytes_received` / requests +* "24-hour host network and disk" → `network_*` / `disk_*` + +### 9.3 Zone `/websites/:id` + +* Keep cards like "total data provided". +* Data and the dashboard business area use **the same aggregation function**, only `hosts = zone domain list`. +* Docs and UI may note: the global dashboard includes all Hosts; this page is only this Zone. + +### 9.4 Node Detail + +* Business throughput: that node's `sum(bytes_sent)` etc. +* OpenResty: health + current connections. +* NIC: clearly "host". + +--- + +## 10. OpenResty and Log Format + +### 10.1 Keep + +Existing JSON `log_format` core fields: + +```text +ts, host, path, remote_addr, status, request_time, +bytes_sent (= $body_bytes_sent), request_length +``` + +### 10.2 Changes + +* No longer rely on log-phase writes of business shared dict counts as control-plane input. +* Observability-port requests continue not writing business stats (or `access_log off`). + +### 10.3 Agent Parsing + +* Protocol `NodeAccessLog` adds `request_length`. +* Legacy log lines missing fields default to 0, not blocking the whole batch. + +--- + +## 11. Upgrade and Migration (no compatibility layer) + +### 11.1 Phase Review (shipped) + +| Phase | Content | +| --- | --- | +| **M1–M5** | read path switches to access logs; protocol v2; stop pre-aggregation; edge_health + access_log_hourly; drop old tables and API compat fields | + +### 11.2 Upgrade Strategy + +* **Agent: destroy-and-recreate preferred**; binary replacement allowed. +* On binary replacement: the local old observability buffer (including `snapshot` / `openresty_observation` / `traffic_report`) is **deleted wholesale**, rebuilt after running. +* Server **does not** parse v1 fields, **does not** dual-read request_reports / openresty throughput. +* Detail-missing periods: business charts are empty or partial; **never** impersonate data provided with NIC or removed openresty throughput. + +### 11.3 Data Backfill + +* Historical "data provided" follows access logs. +* Before `of_access_log_hourly` is created, history is backfilled with goose SQL (ANTI JOIN to prevent duplicates). + +### 11.4 Health-State Authority + +* **Current state**: PG `openresty_status` / `openresty_message`. +* **Time series**: log primary DB `of_node_edge_health` (status + connections; no message). + +### 11.5 UV + +* **Whole-window unique visitors**: `uniqExact(remote_addr)` (dashboard totals, Zone totals). +* **Bucketed UV** (Zone curves): per-bucket uniq, **not summable across buckets**; UI must note it. +* **Hourly trend path**: don't plot / fill per-hour UV (hourly table has no UV). + +--- + +## 12. Storage and Capacity + +* Business trends rely on details or hourly rollups; watch `of_node_access_logs` TTL and sampling. +* If details are too large: prefer **Server-side rollup** rather than restoring Agent pre-aggregation. +* For high-cardinality path scenarios, limit detail path length (existing); aggregation doesn't do global Top over full paths by default. + +--- + +## 13. Risks and Trade-offs + +| Risk | Mitigation | +| --- | --- | +| Large detail volume makes CH and heartbeat heavy | batching, compression, sampling policy evaluation; Server rollup; limit per-batch count | +| Brief log loss lowers business volume | local buffer and rotation handling; monitor access log collection lag | +| Users still compare "NIC outbound" with "data provided" | UI sections and copy enforce the "host" prefix | +| Old Agents stay online long-term | **no compatibility layer**; Agents must be upgraded/rebuilt | + +**Why not keep Agent pre-aggregation as an optimization?** + +* Saving bandwidth re-splits the truth, drifts calibers, and repeats this problem. +* Optimization belongs in Server derived tables and queries, not edge business computation. + +--- + +## 14. Key Decision Summary + +| Decision | Choice | Rejected Alternative | +| --- | --- | --- | +| Business traffic truth | access logs | OpenResty dict / TrafficReport | +| Agent role | report only facts | edge UV/TopN/throughput accumulation | +| "Outbound" vs "provided" | merged into data provided | dual fields and dual pipelines long-term | +| NIC traffic | independent L3, separate copy | reconciled side-by-side with business outbound | +| Performance | CH rollup | Agent pre-aggregation | +| Migration | switch read path first, then slim Agent | drop details first, rely on pre-aggregation | + +--- + +## 15. Doc and Code Mapping + +| Area | Main Paths | +| --- | --- | +| Protocol | `pkg/protocol/agent.go` | +| Agent collection | `internal/apps/agent/observability/`, `heartbeat/` | +| OpenResty logging and Lua | `pkg/render/openresty/`, `internal/apps/agent/nginx/observability_assets.go` | +| Server storage | `internal/apps/openflare/agent/observability.go` | +| Log aggregation | `internal/repository/analytics/node_access_log*.go`, `internal/apps/openflare/zone/stats.go` | +| Dashboard | `internal/apps/openflare/dashboard/`, `internal/apps/openflare/observability/analytics.go` | +| Frontend | `frontend/app/(main)/page.tsx`, `components/dashboard/*`, `websites/.../zone-overview.tsx` | + +**Recommended reading order:** + +1. **[Observability Transport Model](./observability-transport-model.md)** (latest: what to send, where collected from, frequency, sample JSON) +2. [Agent Reporting Protocol and Observability Data Model](./observability-data-model.md) (protocol fields and DDL) + +--- + +## 16. Revision History + +| Date | Notes | +| --- | --- | +| 2026-07-17 | initial draft: target architecture and migration phases for dual truth, Agent pre-aggregation, field redundancy | +| 2026-07-17 | added protocol/table-structure chapter links `observability-data-model.md` | diff --git a/docs/en/design/observability-transport-model.md b/docs/en/design/observability-transport-model.md new file mode 100644 index 00000000..c0ff9ad6 --- /dev/null +++ b/docs/en/design/observability-transport-model.md @@ -0,0 +1,502 @@ +# Edge Observability Transport Model (current target version) + +> **This document is the latest authoritative description of "how Agent ↔ Server observability data is transmitted".** +> After reading you should be able to answer: what is sent, where it is collected from, how often, how the Server stores it, and where product metrics are queried from. +> Protocol fields and DDL details: [Observability Reporting Protocol & Data Model](./observability-data-model.md); background: [Edge Observability & Business Traffic Stats](./observability-design.md). + +--- + +## 0. Remember the Three Layers First + +| Layer | Question Answered | Single Data Source | Product Examples | +| --- | --- | --- | --- | +| **L1 Business delivery** | How much data provided? How many requests? | **access.log details** | data provided, request count, UV, status codes, Top domains | +| **L2 Edge health** | Is OpenResty alive? Current connections? | **local `/openflare/observability`** | node health, current connections | +| **L3 Host capacity** | How are CPU/memory/disk/NIC? | **OS readings** | capacity trends, host NIC | + +**The three layers are never reconciled against each other.** +"Data provided" ≠ "current connections" ≠ "host NIC outbound". + +--- + +## 1. Overview: Who Collects, Who Reports, Who Aggregates + +```text +┌─────────────────────────────────────────────────────────────┐ +│ Edge Node │ +│ │ +│ Visitor request ──► OpenResty │ +│ │ │ +│ ├─ access.log (one line per request) ←── L1 collection point │ +│ │ │ +│ └─ connection state (maintained in-process) │ +│ │ │ +│ ▼ │ +│ GET /openflare/observability ←── L2 reads snapshot │ +│ (no log scanning, no business recomputation) │ +│ │ +│ OS /proc etc. ──────────────────────────── L3 reads snapshot │ +│ │ +│ ┌────────── Agent ──────────┐ │ +│ │ default: one NodePayload per 3s │ │ +│ │ · tail access.log incremental │ │ +│ │ · GET local observability │ │ +│ │ · read host_metrics │ │ +│ └────────────┬──────────────┘ │ +└─────────────────────────────│──────────────────────────────────┘ + │ HTTP heartbeat or WebSocket status + ▼ +┌─────────────────────────────────────────────────────────────┐ +│ Server (control plane) │ +│ · details → ClickHouse of_node_access_logs │ +│ · health → node latest state + of_node_edge_health │ +│ · host → of_node_metric_snapshots │ +│ · business trends / Zone stats = sum/count/uniq over access_logs only │ +└─────────────────────────────────────────────────────────────┘ +``` + +| Role | Does | Doesn't | +| --- | --- | --- | +| OpenResty | writes access.log; maintains connection counts | no direct reporting to the control plane | +| Agent | **collects facts and reports them** | **does not compute** UV/TopN/24h data provided | +| Server | stores + **aggregates/interpreters** | does not trust edge business pre-summaries | + +--- + +## 2. Collection Frequency (defaults) + +| Action | Default Frequency | Config | +| --- | --- | --- | +| Agent → Server report | **every 3 seconds** a full payload | `heartbeat_interval` / control-plane `agent_heartbeat_interval` (ms, default `3000`) | +| Tail access.log when packing | **with report** (new lines since last report) | same | +| GET `/openflare/observability` when packing | **with report** (reads **current** connection snapshot) | same | +| Read host metrics when packing | **with report** | same | +| OpenResty writes access.log | **1 line at each request end** | unrelated to heartbeat | +| Connection counts update in-process | **on connection change** (kernel-maintained) | unrelated to heartbeat | +| Offline replay window | keep ~**60 minutes** by default | `observability_replay_minutes` | +| Node offline detection | ~**60s** without a successful heartbeat | `node_offline_threshold` (default `60000` ms) | + +**Notes:** + +- The Agent has **no separate "sampling clock"**; **sampling points = report points** (default 3s). +- access.log is "per-request continuous writes"; the Agent only **moves incremental lines** periodically. +- `/openflare/observability` is **not** "business stats start being counted when called"; for connections it **reads Nginx's existing instantaneous values**. + +Transport channels: + +- **HTTP heartbeat**: POST the full payload at the interval. +- **WebSocket**: after connecting, sends `status` messages at the same interval (same content shape); HTTP heartbeat is not double-sent then. + +--- + +## 3. Agent → Server Packet (NodePayload v2) + +### 3.1 Structure Skeleton + +```json +{ + "schema_version": 2, + "node_id": "n_01hxyz", + "name": "edge-shanghai-1", + "ip": "203.0.113.10", + "version": "3.4.0", + "ext_version": "", + "current_version": "20260718-abc", + "last_error": "", + "profile": { }, + "host_metrics": { }, + "edge_health": { }, + "access_logs": [ ], + "buffered": [ ], + "health_events": [ ], + "waf_ip_group_checksums": { } +} +``` + +| Field | Layer | Meaning | +| --- | --- | --- | +| identity/version/last_error | control | who the node is, what version it runs | +| `profile` | low-frequency overview | hostname, core count, etc. (report on change) | +| `access_logs` | **L1** | access detail increments | +| `edge_health` | **L2** | OpenResty health + current connections | +| `host_metrics` | **L3** | CPU/memory/disk/NIC readings | +| `buffered` | backfill | batches of facts accumulated while offline | +| `health_events` | events | e.g. openresty_unhealthy | +| `waf_ip_group_checksums` | sync | not an observability lake | + +**Removed from the protocol (no compatibility layer; old Agents must upgrade):** + +- `traffic_report` +- `openresty_observation` (incl. rx/tx) +- `snapshot` / `buffered_observability` +- business-meaning openresty throughput fields + +--- + +## 4. L1 Business: access_logs + +### 4.1 Where Collection Comes From + +| Step | Location | Description | +| --- | --- | --- | +| 1 | OpenResty `log_format openflare_json` | writes one JSON line per request to `access_log_path` | +| 2 | Agent **tails increments** by file offset | new lines between two heartbeats | +| 3 | parse and put into `access_logs[]` | overlong paths may be truncated; **no sum/count** | + +Log format (OpenResty variables): + +```text +ts ← $time_iso8601 +host ← $host +path ← $request_uri +remote_addr ← $remote_addr +status ← $status +request_time ← $request_time +bytes_sent ← $body_bytes_sent 【data provided = response body bytes】 +request_length← $request_length 【data received】 +user_agent ← $http_user_agent +cache_status ← $upstream_cache_status 【cache status; UI can derive hit/origin/un-cached】 +``` + +Observability-port requests **don't write** business access.log (separate server with `access_log off`). + +### 4.2 Report Example + +```json +"access_logs": [ + { + "logged_at_unix": 1721289601, + "remote_addr": "198.51.100.20", + "host": "www.example.com", + "path": "/api/v1/ping", + "status_code": 200, + "bytes_sent": 1024, + "request_length": 128, + "request_time_ms": 15, + "user_agent": "curl/8.0", + "cache_status": "MISS" + }, + { + "logged_at_unix": 1721289602, + "remote_addr": "198.51.100.21", + "host": "www.example.com", + "path": "/index.html", + "status_code": 200, + "bytes_sent": 8192, + "request_length": 300, + "request_time_ms": 8, + "user_agent": "Mozilla/5.0", + "cache_status": "HIT" + } +] +``` + +| Field | Explanation | +| --- | --- | +| `bytes_sent` | **data provided** (single request); global/Zone totals = Server `sum` | +| `request_length` | **data received** (single request) | +| `logged_at_unix` | request completion time (business timeline) | +| `host` | used for Zone domain filtering | +| `cache_status` | `$upstream_cache_status` as-is; detail/list can derive three states (hit/origin/un-cached); **no upstream address reported** | +| no `region` | **written by Server at insert time** via GeoIP | + +### 4.3 How the Server Uses It (product metrics) + +| Product Metric | Algorithm (L1 only) | +| --- | --- | +| Data provided | `sum(bytes_sent)` | +| Data received | `sum(request_length)` | +| Request count | `count()` | +| UV | `uniqExact(remote_addr)` | +| Status distribution | `group by status_code` | +| Top domains | `group by host` | +| Zone page | same + `host IN (that Zone's domains)` | +| Dashboard business area | same, global or Top-filtered | + +Stored in: `of_node_access_logs` (optional Server-side `of_access_log_hourly` acceleration, **Agent never writes it**). + +### 4.4 Report Frequency + +```text +Request happens ──immediately──► write access.log +Agent every 3s ──moves──► new lines in those 3s (possibly 0, possibly many) +Server ──immediately/batched──► CH +``` + +Business volume correctness does **not** depend on 3s alignment; 3s only affects "detail arrival latency at the control plane" and per-packet line count. + +--- + +## 5. L2 Health: edge_health and `/openflare/observability` + +### 5.1 Local Monitoring Endpoint + +**Data collection endpoint:** + +```http +GET http://127.0.0.1:{openresty_observability_port}/openflare/observability +``` + +Default port: **18081** (`openresty_observability_port`). + +**Responsibility:** answers "how is OpenResty right now", **not** "how much business data was provided". + +#### Response Example + +```json +{ + "ok": true, + "captured_at_unix": 1721289600, + "connections": { + "active": 42, + "reading": 0, + "writing": 1, + "waiting": 41 + } +} +``` + +| Field | Instant? | Source | Description | +| --- | --- | --- | --- | +| `ok` | this probe | returns 200 → true | liveness | +| `captured_at_unix` | sampling time | `ngx.time()` | aligned with report | +| `connections.active` | **instant** | Nginx connection state (original stub_status Active) | current active connections | +| `reading` / `writing` / `waiting` | **instant** | same, subdivided | optional but recommended | + +**Not returned (removed):** + +| Old Field | Reason | +| --- | --- | +| `request_count` / `error_count` / UV / status_codes / top_domains | business window summaries, now from access log | +| `openresty_rx_bytes` / `openresty_tx_bytes` | duplicates data provided/received and error-prone | +| `source_countries` | never implemented; countries go through Server GeoIP | +| `server.accepts/handled/requests` | process cumulative counters, easily confused with business requests; not on the main path | + +**`/openflare/stub_status`:** kept; `/openflare/observability` internally reads that endpoint to assemble the connection-count JSON, and the Agent health check also probes it directly. + +### 5.2 Collection Mechanism (read snapshot) + +```text +Nginx maintains Active connections etc. on connect/disconnect + │ +Agent GET /openflare/observability + │ +only reads "current values" and returns JSON +``` + +- No access.log scanning, no 60-second business averages. +- Returns an **instant gauge snapshot**. + +### 5.3 Report Example (packed into NodePayload) + +```json +"edge_health": { + "captured_at_unix": 1721289600, + "status": "healthy", + "message": "", + "connections": 42 +} +``` + +| Field | Source | +| --- | --- | +| `status` / `message` | Agent health probe (config validation/process etc., may work with the observability endpoint's `ok`); must align with top-level `openresty_status` / `openresty_message` | +| `connections` | observability endpoint `connections.active` | + +**Storage split (authoritative sources):** + +| Content | Written To | +| --- | --- | +| latest `status` + `message` | **PG node table** (UI / list / alerts) | +| time-series `status` + `connections` | **CH `of_node_edge_health`** (**no message**) | + +--- + +## 6. L3 Host: host_metrics + +### 6.1 Where Collection Comes From + +The Agent reads the local machine (e.g. `/proc`, disk stats), **once per packet**. + +| Field | Semantics | Description | +| --- | --- | --- | +| `cpu_usage_percent` | instant | current CPU% | +| `memory_*` / `storage_*` | instant used/total | usage rates computed at Server or display layer | +| `disk_read_bytes` / `disk_write_bytes` | **cumulative counter** | kernel cumulative IO | +| `network_rx_bytes` / `network_tx_bytes` | **cumulative counter** | **host NIC**, not data provided | + +### 6.2 Report Example + +```json +"host_metrics": { + "captured_at_unix": 1721289600, + "cpu_usage_percent": 12.5, + "memory_used_bytes": 4294967296, + "memory_total_bytes": 16106127360, + "storage_used_bytes": 50000000000, + "storage_total_bytes": 107374182400, + "disk_read_bytes": 9000000000, + "disk_write_bytes": 12000000000, + "network_rx_bytes": 500000000000, + "network_tx_bytes": 800000000000 +} +``` + +### 6.3 How the Server Handles Cumulative Fields + +```text +store raw-value time series +when displaying "NIC outbound over this period": + delta = current - previous + if delta < 0 → treat as restart/counter reset, record this segment's increment as 0, continue from new baseline + if delta >= 0 → record into that period's increment +``` + +- The Agent **reports raw values**, never computes 24h totals at the edge. +- **Forbidden** to `sum` cumulative raw values as business volume. +- Copy must be **"host NIC"**, never "data provided / OpenResty outbound". + +Stored in: `of_node_metric_snapshots` (optional capacity hourly MV). + +--- + +## 7. One Complete Report Example + +```json +{ + "schema_version": 2, + "node_id": "n_01hxyz", + "name": "edge-shanghai-1", + "ip": "203.0.113.10", + "version": "3.4.0", + "ext_version": "", + "current_version": "20260718-abc", + "last_error": "", + "host_metrics": { + "captured_at_unix": 1721289600, + "cpu_usage_percent": 12.5, + "memory_used_bytes": 4294967296, + "memory_total_bytes": 16106127360, + "storage_used_bytes": 50000000000, + "storage_total_bytes": 107374182400, + "disk_read_bytes": 9000000000, + "disk_write_bytes": 12000000000, + "network_rx_bytes": 500000000000, + "network_tx_bytes": 800000000000 + }, + "edge_health": { + "captured_at_unix": 1721289600, + "status": "healthy", + "message": "", + "connections": 42 + }, + "access_logs": [ + { + "logged_at_unix": 1721289595, + "remote_addr": "198.51.100.20", + "host": "www.example.com", + "path": "/", + "status_code": 200, + "bytes_sent": 4096, + "request_length": 200, + "request_time_ms": 12 + } + ], + "buffered": [], + "health_events": [], + "waf_ip_group_checksums": { + "1": "d41d8cd98f00b204e9800998ecf8427e" + } +} +``` + +**Server storage sketch:** + +| payload block | written to | +| --- | --- | +| `access_logs[0]` | one CH row, `bytes_sent=4096`, `region` filled by GeoIP | +| `edge_health` | node `openresty_status=healthy`, connections=42 | +| `host_metrics` | one CH metric row with cumulative/instant fields | + +**Product query sketch (24h):** + +- data provided = `sum(bytes_sent)` over that node's (or global) logs +- current connections = latest `edge_health.connections` +- host NIC outbound = sum of non-negative `network_tx` deltas over metrics + +The three numbers **need not be equal**. + +--- + +## 8. Offline Backfill `buffered` + +When reporting fails, the Agent caches **the same kind of facts** locally by window (default ~60 minutes), then packs them into `buffered[]` after recovery: + +```json +"buffered": [ + { + "captured_at_unix": 1721289500, + "host_metrics": { }, + "edge_health": { }, + "access_logs": [ ] + } +] +``` + +- Only facts, no legacy TrafficReport. +- Server processing logic is identical to the main fields. + +--- + +## 9. End-to-End Timeline (default 3s) + +```text +t=0.0s visitor request completes → writes one access.log line; connection count may change +t=0.1s another request → another log line +… +t=3s Agent heartbeat: + · reads 2 access_logs lines + · GET observability → connections=42 + · reads host_metrics + · sends to Server +t=3s+ Server stores; dashboard/Zone queries aggregate logs +t=6s next round… +``` + +--- + +## 10. Old Model Comparison + +| Old Approach | New Model | +| --- | --- | +| Lua dict 60s window request_count + Agent 10s pull + Server sum | **removed**; request count = log count | +| openresty_tx as "outbound" | **removed**; data provided = `sum(bytes_sent)` | +| Two endpoints observability + stub_status | data collection unified through observability; stub_status kept as liveness and internal read endpoint | +| TrafficReport pre-aggregation | **removed**; no such path in protocol or API | +| Business and NIC both called "traffic" | **separate copy, separate APIs, separate tables** | +| health status/message | **PG latest-state authority**; CH only status+connection time series | + +--- + +## 11. Config and Implementation Index + +| Item | Location/Key | +| --- | --- | +| Heartbeat interval | Agent `heartbeat_interval`; control plane `agent_heartbeat_interval` (default 3000ms) | +| Offline threshold | control plane `node_offline_threshold` (default 60000ms) | +| Observability port | `openresty_observability_port` (default 18081) | +| access.log path | `access_log_path` | +| Replay minutes | `observability_replay_minutes` (default 60) | +| Protocol types | `pkg/protocol/agent.go` (evolves to v2 when landing) | +| Table DDL | [observability-data-model.md](./observability-data-model.md) | + +--- + +## 12. Revision History + +| Date | Notes | +| --- | --- | +| 2026-07-18 | initial draft: single-page "latest transport model" — three layers, frequency, sample JSON, collection sources, old-model comparison | +| 2026-07-18 | default report interval 3s; offline threshold 60s; replay window 60 minutes | +| 2026-07-18 | M5: edge_health table, access_log_hourly, deprecate request_reports/obs_openresty throughput tables | +| 2026-07-18 | no compatibility layer: removed "may ignore during compat period" wording; health message only in PG, no message in CH | diff --git a/docs/en/design/origin-error-page.md b/docs/en/design/origin-error-page.md new file mode 100644 index 00000000..8a3e06db --- /dev/null +++ b/docs/en/design/origin-error-page.md @@ -0,0 +1,204 @@ +# Origin Error Page Design + +You will learn: when an origin or the gateway returns a specified error status code, how OpenFlare replaces the pass-through response with a globally configurable page; how the config enters the immutable config version; and how the edge OpenResty keeps the real HTTP status code while displaying it in the page. + +This design is the productized complement of the reverse proxy traffic path in [System Architecture](./architecture.md); the config release model is in [Agent & Publish Model](./agent-design.md). + +--- + +## 1. Goals and Non-Goals + +### 1.1 Goals + +* **Interceptable**: for a user-configured status code set, replace the previously pass-through origin/Nginx default error response with a unified HTML. +* **Disableable**: when the global switch is off, behavior matches today (pass-through / Nginx default page). +* **Visible by default**: enabled by default, default status code tag `500-599`, default minimal OpenFlare error page. +* **Customizable**: admins can edit the full HTML online; empty HTML means the built-in default template. +* **Status passthrough**: the HTTP response `status` keeps the original error code (e.g. 502, 522); the page body shows the same value via `{{status}}`. +* **Globally unified**: a single config under sidebar「Website Management → Response Pages」shared by all reverse proxy routes. +* **Consistent with release**: the config persists via Option, enters the config version snapshot, and is distributed with release/rollback. + +### 1.2 Non-Goals + +* Per-route / per-Zone error page overrides +* Hosting error pages via file upload (online HTML only) +* Modifying WAF / PoW / rate-limit's own response pages (unless the user adds those status codes to the list) +* Pages static route error pages +* Multi-language error pages, brand asset CDN + +--- + +## 2. Product Behavior + +### 2.1 When to Replace + +| Condition | Behavior | +| --- | --- | +| Switch on and the response status falls in the expanded set | return custom/default HTML, **status unchanged** | +| Switch on with GET-only enabled, non-GET request returns a matching status | pass through the origin's raw response, no replacement | +| Switch off | no `error_page` directives generated, pass through | +| Status not in the set | no replacement | +| Pages upstream routes | this feature is not applied | +| Origin returns 2xx/3xx/4xx successfully (not configured) | no replacement | + +In all-methods mode, `proxy_intercept_errors on` is enabled on the reverse proxy `location`, so **origin-returned** matching 5xx etc. are also intercepted, not just gateway-local 502s; GET-only mode switches to Lua header/body filters that only replace GET response bodies. + +### 2.2 Status Code Tag Syntax + +Each Tags Input entry: + +| Form | Example | Meaning | +| --- | --- | --- | +| Single code | `522` | only that code | +| Closed range | `500-599` | expand including endpoints | + +* Valid range: single codes and range endpoints must be in **400–599**; `lo ≤ hi`. +* Default tag list: `["500-599"]`. +* Persist the **raw tags** (JSON array string); expand, dedupe, and sort at render time. +* If the expanded result is empty while enabled → save rejected. +* Invalid tags → save rejected with a readable error. + +### 2.3 Page Placeholders + +| Placeholder | Meaning | +| --- | --- | +| `{{status}}` | the current response status code (consistent with the HTTP status) | +| `{{host}}` | request Host | + +Both custom HTML and the default template support these placeholders; replaced at the edge at runtime. Unused placeholders may be omitted from the template. + +### 2.4 Default Page + +Built-in minimal white-background OpenFlare default page: large pass-through status code, short English description, Host, and a brand footer. Supports `{{status}}` / `{{host}}`; the frontend can load prebuilt styles from the built-in template catalog on the edit page. + +--- + +## 3. Config Model + +### 3.1 Option Keys (`w_system_configs` / OpenFlare Option API) + +| Key | Type | Default | Description | +| --- | --- | --- | --- | +| `origin_error_page_enabled` | bool string | `true` | master switch | +| `origin_error_page_status_codes` | JSON string array | `["500-599"]` | raw tags | +| `origin_error_page_html` | text | `""` | empty = built-in default; max **256 KiB** | +| `origin_error_page_get_only` | bool string | `false` | replace error pages only for GET; other methods pass through | + +Reuses APIs: + +* `GET /api/v1/d/option` +* `POST /api/v1/d/option/update-batch` + +No new resource routes. goose migration writes the seed; constants defined in the `internal/model` config key area. + +### 3.2 Validation (update-batch) + +1. `enabled`: parseable as bool. +2. `status_codes`: valid JSON array; each entry `^\d{3}$` or `^\d{3}-\d{3}$`; expanded values all in 400–599; non-empty when enabled. +3. `html`: length ≤ 256 KiB (bytes); empty allowed. +4. Parse/expand logic is a **pure function** shared by the API and `pkg/render/openresty` to avoid semantic forks. + +No XSS sanitization on HTML: it's an admin global ops config consistent with public edge display; docs warn not to embed untrusted third-party scripts. + +### 3.3 Config Version Snapshot + +`ConfigSnapshot` adds fields: + +```text +OriginErrorPageEnabled bool +OriginErrorPageStatusCodes []string // raw tags +OriginErrorPageHTML string // empty => renderer uses built-in default +OriginErrorPageGetOnly bool +``` + +Read from Option when building the snapshot; the Agent only consumes the snapshot, never reading the control-plane DB directly. + +--- + +## 4. Edge Rendering + +### 4.1 Content Generated When Enabled + +1. **SupportFile**: error page template (e.g. `error_pages/origin_error.html.tmpl`), content is the custom HTML or built-in default, keeping `{{status}}` / `{{host}}`. +2. **Each reverse proxy server** (HTTP/HTTPS proxy; excluding Pages): + +```nginx +proxy_intercept_errors on; +error_page @__openflare_origin_error; + +location @__openflare_origin_error { + default_type text/html; + charset utf-8; + content_by_lua_block { + # read template, replace {{status}} / {{host}}, output body + # ngx.status keeps the original error code + } +} +``` + +### 4.2 Runtime Replacement + +Use a **named location with `content_by_lua_block`** to read the template and replace placeholders — the status is **not** baked into a static file (status differs per request). GET-only mode uses `header_filter_by_lua_block` + `body_filter_by_lua_block` inside the reverse proxy location to replace only GET response bodies; non-GET requests pass through. + +Never rewrite the error page to HTTP 200. + +### 4.3 When Disabled + +Do not output `proxy_intercept_errors`, `error_page`, the internal location, or the corresponding SupportFile (or the file may be written but unreferenced). GET-only mode also omits the Lua filters. + +### 4.4 Interaction with Cache / Stale + +If global `proxy_cache_use_stale` returns stale cache for some error codes, **successful stale responses never enter `error_page`**. The error page is only shown when the client actually receives an error status in the configured list. Behavior depends on existing cache directives; this feature does not change stale policy. + +--- + +## 5. Frontend + +### 5.1 Entry + +* Sidebar「Website Management → Response Pages」: Error Page tab (`/responses`), edit page `/responses/error-page/edit`, preview page `/responses/error-page/preview`. + +### 5.2 Page Structure + +* Header note: takes effect after releasing via「Version Release」. +* **Switch + Tags Input** (shadcn-extension Tags Input: `@/components/ui/tags-input`): status code tags. +* **HTML editor area** +「Load default template」「Restore default (clear)」+ placeholder docs. +* **Client-side preview**: replace with sample `status=502`, `host=example.com` and preview in sandbox/iframe. +* Save: `OptionService.updateBatch`; permissions same as the performance tuning page (admin). + +### 5.3 Component Dependencies + +Tags Input and the HTML editor reuse existing shadcn/ui components, consistent with the existing UI style. + +--- + +## 6. Data Flow + +```text +Admin /responses (Error Page tab) + → Option update-batch (validate tags & HTML) + → w_system_configs + +Release config version + → snapshot writes OriginErrorPage* + → render OpenResty conf + SupportFile + → Agent pulls and reloads + +Visitor requests a proxied domain + → origin/gateway produces a matching status code + → error_page → named location + → replace placeholders, keep original status, return HTML +``` + +--- + +## 7. Decision Record + +| Decision | Choice | Reason | +| --- | --- | --- | +| Config scope | global | product requirement; simple implementation and ops | +| Storage | Option + config version | consistent with performance tuning, rollbackable | +| Status input | tags: single code and range | default whole 5xx, but can name 522 | +| Response status | keep original | correct for monitoring/SEO/client semantics | +| Runtime replacement | internal + lightweight template replacement | status differs per request | +| Customization | online HTML | flexible without a file-upload chain | diff --git a/docs/en/design/pages-design.md b/docs/en/design/pages-design.md new file mode 100644 index 00000000..12df5496 --- /dev/null +++ b/docs/en/design/pages-design.md @@ -0,0 +1,252 @@ +# Pages Static Hosting Design + +You will learn: the architecture design of OpenFlare Pages static hosting, the immutable deployment and secure extraction flow, the OpenResty static serving and API reverse proxy config rendering, and the cooperative workflow between the control plane and the Agent. + +--- + +## Requirements Analysis + +In modern web operations, besides reverse proxying dynamic applications, deploying and hosting static frontend sites (SPA apps built with React/Vue, or static generator output like Hugo/VitePress) is extremely common. Traditional approaches suffer from: +1. **Release disconnected from proxy config**: after uploading frontend build artifacts to the Nginx host, you still need to modify the Nginx vhost config manually or via other scripts — error-prone and without version control. +2. **Multi-node distribution is hard**: with multiple edge nodes managed by the control plane, syncing static files to all nodes consistently requires complex sync scripts (e.g. rsync). +3. **Rollback lacks consistency**: once a new frontend package fails or has serious defects, you must restore both the static files and the proxy rules — atomic rollback is hard. + +To solve these, OpenFlare introduces **Pages static hosting**, inspired by Cloudflare Pages. It brings "pre-built artifact import" and "website proxy rule config" into the same control plane, leveraging OpenFlare's pull-based cooperative architecture to achieve eventual convergence across multiple Agents via immutable deployments, single-node atomic switching, and periodic reconciliation, with fast rollback. + +--- + +## Core Features + +The Pages static hosting subsystem includes: +* **Pre-built artifact deployment**: upload a static resource archive directly, or save a Remote URL or public GitHub Release asset source for a project. External sources are only accessed by the Server; on successful sync they uniformly create or reuse an immutable deployment and activate it atomically. +* **Immutable deployment snapshots**: each local upload creates a new candidate deployment; persistent-source sync creates or reuses a deployment by source identity/revision and activates it. All deployments have a unique ID and whole-package SHA-256, support keeping the most recent N historical versions per system config, and can be rolled back anytime. +* **Check and auto-update**: GitHub latest can be checked periodically per project; by default it only hints at available updates; only after an admin explicitly enables it does it auto-sync and publish by the exact revision found. +* **SPA Fallback**: supports fallback routing for single-page apps; when a static file isn't found, requests redirect to the entry file. +* **Built-in API reverse proxy**: enables API proxying within Pages rules with one click, eliminating cross-origin issues by forwarding requests to a designated backend. +* **Secure package validation and extraction**: built-in path-traversal defense, symlink-hijack protection, file size/count limits, and configurable upload package size control to keep nodes physically safe. +* **Configurable limits**: admins can adjust the "deployment package size limit" and "historical deployment retention" in ops settings. + +### Deployment Sources + +Projects currently support three source views: manual, Remote URL, and GitHub Release. No source record means manual; switching or deleting a source doesn't delete historical deployments or change the current active deployment. Remote URL only allows manual "Sync and Publish"; GitHub Release supports manual check/sync for latest/tag, and only latest can opt into scheduled checking and auto-update. + +A source is mutable config; a deployment is an immutable fact. Source config and runtime cursor/state/lease are stored separately; deployments only save the security provenance snapshot at creation time. All artifacts reuse the "download or receive artifact → real-byte and entry validation → `upload.Ingest` → deployment" artifact pipeline: manual uploads stop at candidate, waiting for explicit admin activation; persistent-source sync creates-or-loads and atomically activates in the same business transaction. + +The admin project detail is organized as "current production deployment → deployment source → deployment history". + +--- + +## Pages Static Hosting Architecture + +Pages hosting is logically split into a **Control Plane** and a **Data Plane**. + +```mermaid +graph TD + %% data flow + Browser[1. Browser / Visitor] -->|HTTPS request / traffic| OpenResty[2. OpenResty / WAF] + OpenResty -->|1. static serving try_files| StaticFiles[3. Edge node local static dir current] + OpenResty -->|2. forward API proxy| BackEnd[4. Backend API service] + + %% control flow & heartbeat + Admin[Admin / CI] -->|upload or configure source| Server[OpenFlare Server control plane] + Providers[Remote / GitHub Provider] -->|restricted artifact candidate| Server + Scanner[internal scanner / action task] -->|check & auto sync| Server + Server <-->|Agent API / Heartbeat| Agent[openflare-agent process] + Server -.->|unified upload.Ingest| UploadStore[(platform upload backend)] + + Agent -->|1. discover new version| Server + Agent -->|2. download deployment package| Server + Agent -->|3. validate, extract, atomic switch| StaticFiles +``` + +* **Control Plane**: the Server receives local uploads or fetches Remote/GitHub pre-built artifacts via restricted providers; action tasks and the internal scanner handle checking, syncing, and auto-updates. All artifacts pass unified inspection and `upload.Ingest` into the platform storage backend; manual uploads create a new candidate, persistent-source sync creates-or-loads a deployment and activates it atomically. Config release only compiles the stable project anchor and static serving metadata. +* **Data Plane**: the Agent discovers Pages projects referenced by config during heartbeat/WS reconciliation, pulls the project's currently active package via a dedicated API, and performs validation and extraction. OpenResty serves static files locally; the Agent doesn't know whether the artifact came from upload, Remote, or GitHub. + +--- + +## Data Model and Metadata Design + +### 1. Core DB Entities +* **Pages project (`of_pages_projects`)**: + * Records the business name, Slug (URL-friendly), enabled state, static serving root dir (RootDir, nullable), entry filename (EntryFile, default `index.html`), SPA Fallback settings, and API reverse proxy config (APIProxyPath, APIProxyPass, APIProxyRewrite). +* **Source config (`of_pages_project_sources`)**: + * At most one mutable source config per project, distinguishing Remote URL and GitHub Release by `source_type`. `config_version` fences stale tasks; the full Remote URL lives only in the config table — never in responses, logs, task payloads, or deployment provenance. V2 does not promise DB column encryption. +* **Source runtime (`of_pages_project_source_runtime`)**: + * 1:1 with the source, storing ETag, seen/applied revision, last check/sync, next check, errors, and lease. State is fixed to `idle | checking | update_available | syncing | failed | attention`; queued/completed state is carried by `TaskExecution`. +* **Pages deployment (`of_pages_deployments`)**: + * Records immutable deployment facts: project-incremental deployment number, whole-package SHA-256, `upload_id`, file count/total bytes, creator, and nullable source identity/revision, source security snapshot, and trigger. `artifact_path` is only a legacy-compat field and is no longer the storage truth for new deployments. +* **Deployment file manifest (`of_pages_deployment_files`)**: + * Stores the full regular-file path list and actual byte counts per deployment for console display and statistics. + * No longer computes content hashes per file; integrity is guaranteed by the **whole-package** SHA-256 (`of_pages_deployments.checksum`), verified by the Agent when pulling. + * Control-plane inspection reads the archive via file handles, streaming each regular file body and checking declared vs actual size — no whole-package `ReadFile` into memory, no per-file disk write for hashing. + +### 2. Route Association and Snapshot +`proxy_routes` rules associate with a Pages project via `upstream_type = "pages"` and `pages_project_id`. A route may join the release flow only when its type is `pages` and the project has an activated deployment. +The version snapshot emitted at release includes `snapshotPagesDeployment`: +```json +{ + "project_id": 1, + "project_slug": "my-spa-app", + "deployment_id": 12, + "deployment_number": 3, + "checksum": "a7b3c2...", + "entry_file": "index.html", + "spa_fallback_enabled": true, + "spa_fallback_path": "/index.html", + "api_proxy_enabled": true, + "api_proxy_path": "/api", + "api_proxy_pass": "http://api.internal:8000", + "api_proxy_rewrite": "/api/(.*) /$1", + "local_root": "__OPENFLARE_PAGES_DIR__/projects/1/current" +} +``` + +### 3. Dual-Track Relationship with the Main Config Version (project anchor + latest pull) +* The **main config version** and **Pages deployments** are two independent version systems. +* The stable anchor of a Pages route in the main config is **`pages_project_id` (project ID)**, not a deployment ID. +* The OpenResty `root` uses the project-level path `__OPENFLARE_PAGES_DIR__/projects/{project_id}/current`; the path stays unchanged on activation switch, so swapping packages never requires republishing the main config. +* The Agent requests the "latest active package" per project (like `github/release/latest`): + * `GET /api/v1/agent/pages/projects/:project_id/latest/hash` + * `GET /api/v1/agent/pages/projects/:project_id/latest/package` + * The control plane returns the deployment ID, hash, package size, and expanded manifest metadata for the project's **currently active deployment**. The Agent uses the deployment ID and other latest metadata to detect pointer races during download, but the stable anchor of the main config and local dir remains the project ID. +* Therefore: switching the active deployment within a project **does not require publishing the main config**; the Agent polls the latest hash during periodic reconciliation, downloads on change, and switches `current`. +* The `pages_deployment` field in the snapshot still records release-time metadata (entry file, SPA/API proxy, etc.) but does not lock the Agent's package version. + +--- + +## Server (Control Plane) Responsibilities and Lifecycle + +### 1. Deployment Package Security Validation and Analysis +To protect the server from untrusted artifacts, the control plane applies the same strict validation to local uploads and all external sources: +* **Format support**: `zip`, `tar.gz` / `tgz`, `tar.xz` / `txz`, `tar.bz2` / `tbz2`, `tar`, `7z`. +* **Size limits**: archive size is controlled by system config `pages_max_package_size_mb` (default 100 MiB, range 1–2048); expanded single-file and total limits are "package size × 4" with a floor of 100 MiB. Inspection always streams regular file bodies, checking declared vs actual size and enforcing limits on actual values. +* **Count limit**: at most 1,000 static files per package. +* **Symlink blocking**: any symlink detected while walking the archive immediately errors and rejects the upload, defending against symlink-hijack attacks. +* **Path traversal defense**: every archive file path is `Clean`ed and checked for `..` or leading `/`, defending against directory-traversal writes to sensitive system paths. +* **Entry file validation**: the project's entry file (e.g. `index.html`, possibly under `project.RootDir`) must exist in the package, otherwise the upload is rejected. +* **Common root prefix stripping**: many packaging tools add a redundant top folder as a common root prefix; the control plane auto-detects and safely strips it. +* **Whole-package integrity**: SHA-256 is computed once over the archive bytes at upload/import and written to the deployment record; the Agent reconciles against the whole-package hash after pulling. No per-file content hashes. +* **Actual size recheck**: `InspectOptions.VerifySizes` is kept only for compatibility; current inspection always reads regular file bodies, checks declared values, and accumulates actual sizes, but still does not compute per-file content hashes. +* **History retention**: system config `pages_max_history_count` (default 20; 0 = unlimited) trims after successful deployment. Semantics: **each project keeps at most N deployments**; the currently active deployment is always kept, remaining slots fill from newest to oldest by deployment ID. With `history_count=1`, manual uploads temporarily keep both the active and the newest candidate; the next upload replaces the old candidate; after the candidate activates, the strict limit resumes. Exceeding non-active deployments and their file manifests are deleted; the corresponding upload record is soft-deleted idempotently via platform primitives — Pages never physically deletes blobs that may be shared by dedup. If trimming fails after a successful deployment, it only logs and doesn't roll back activation; concurrent operations may temporarily exceed N and converge on later trims. Main config version rollback does not depend on old Pages packages (see the dual-track section above). + +### 2. Deployment Package Storage Planning +The control plane stores local, Remote, and GitHub artifacts into the configured local/S3 backend via the unified upload framework (`upload.Ingest`), recording `upload_id` and the file manifest in the DB. **Large static packages never enter `config_versions` records or any config push channel**, keeping control-plane data sync lightweight. + +### 3. Source Check, Auto-Update, and Upload Compensation + +* `openflare:pages_source_action` executes admin check/sync or scanner-dispatched exact-revision sync; payloads never carry URL, Token, ETag, or lease tokens. Manual sync only accepts real user actors; auto sync only accepts the system actor with the `scheduled_auto_update` trigger. +* `openflare:pages_source_scan` is a fixed `*/5 * * * *` internal-only TaskHandler accepting only `{}`; it never appears in generic task types or the schedule management UI. Each round runs "recover expired leases → compensate orphan uploads → scan due sources". +* The scanner sorts stably by `next_check_at, source_id`, serially checking at most 20 GitHub latest sources per batch; ETag/304 still advances the check time; 403/429 record the status code and the actual backoff deadline; a single source failure doesn't block subsequent sources. +* On finding an update, the seen cursor is always saved first. Only with `auto_update_enabled=true` and a normal `update_available` state does it dispatch sync with the exact revision found this check; `attention`, Remote, and fixed tags never auto-publish. Manually activating another deployment fences in-flight tasks and disables auto. +* Orphan compensation checks at most 100 upload records per round that have been quarantined for at least 2 hours, requiring a system owner, Pages retention type, V2 marker, and no deployment references. Candidates are re-checked in the `project → source → runtime → upload` lock order and only soft-deleted via the upload framework with stat updates — never physically deleting blobs possibly shared by dedup. + +--- + +## Agent (Data Landing) Responsibilities and Self-Healing + +The Agent runs on each edge proxy node: on first applying config referencing a Pages project, and on subsequent periodic latest reconciliation, it "atomically" pulls the currently active static assets to the node. + +### 1. Pull Latest Per Project +1. The Agent parses routes with `UpstreamType == "pages"` from the active main config and collects the stable anchor **`pages_project_id`**. +2. For each project it calls `GET /api/v1/agent/pages/projects/:project_id/latest/hash` to get the control plane's currently active package hash (a latest pointer). +3. If the local `projects/{project_id}/releases/{hash}` isn't ready, it streams `.../latest/package` to a temp file, enforcing real response limits and SHA-256; after download it **requests the hash again** to avoid activation-switch races, retrying a bounded number of times on mismatch. +4. The request carries the node's `X-Agent-Token`. + +### 2. Secure Extraction, Atomic Switch, Keep Only Latest +1. Absolute package cap is 2 GiB; the downloaded content's SHA-256 must match the latest hash from the post-download re-query; the whole package never enters `[]byte`. +2. Extract into a random staging dir `projects/{project_id}/releases/.{hash}-.tmp` (zip / tar.* / 7z supported), rejecting path traversal, links, and special files. The Agent obeys both Server metadata limits and local absolute limits: at most 1,000 files, single file and total at most 8 GiB. +3. After extraction, walk the actual file tree and precisely recheck file count and total bytes against the Server metadata; mismatch → refuse to switch. +4. Write `.openflare-pages.json`, then rename to `releases/{hash}`. +5. **Atomic switch** `projects/{project_id}/current` to the new release (symlink preferred, copy on failure). +6. **Only after the new package is ready and current has switched successfully**, delete other `releases/*` (including `.tmp`) under the project — **historical deployment packages are never kept on the edge**. Each project always keeps exactly one latest content per node. +7. Multi-project reconciliation **isolates failures**: a single project failure logs and continues with others, finally aggregating errors. + +--- + +## OpenResty (Static Serving and Proxy) Config Rendering + +For Pages-hosted sites, the control plane renders the corresponding `server` block, replacing the regular proxy route's `proxy_pass`. + +### 1. Static Serving Directive Rendering +* **`root` and `index`**: + The Server points `root` at the project-level placeholder path `__OPENFLARE_PAGES_DIR__/projects/{project_id}/current` (optionally appending `RootDir`). Activation switching only changes directory contents, not the path, so swapping packages never requires republishing the main config. + ```nginx + server { + listen 80; + server_name myapp.example.com; + + root "/var/lib/openflare/pages/projects/3/current"; + index "index.html"; + ... + } + ``` + +### 2. try_files and SPA Fallback +* **SPA Fallback disabled (default)**: + only match physically existing files, otherwise strict 404: + ```nginx + location / { + try_files $uri $uri/ =404; + } + ``` +* **SPA Fallback enabled**: + if the requested file doesn't exist, redirect to the project's configured entry fallback (usually `/index.html`): + ```nginx + location / { + try_files $uri $uri/ /index.html; + } + ``` + +### 3. API Reverse Proxy and Rewrite Rendering +When a static frontend needs backend API access without cross-origin issues, enable the API proxy. The OpenResty renderer nests a dedicated API `location` branch inside the static `server` block: +```nginx +server { + listen 80; + server_name myapp.example.com; + ... + # API proxy path match + location /api { + # apply rewrite rules when configured + rewrite ^/api/(.*)$ /v1/$1 break; + rewrite ^/api$ / break; + + proxy_pass http://api.internal:8000; + proxy_http_version 1.1; + proxy_set_header Host $http_host; + proxy_set_header X-Real-IP $remote_addr; + proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for; + proxy_set_header X-Forwarded-Proto $scheme; + proxy_set_header Upgrade $http_upgrade; + proxy_set_header Connection $connection_upgrade; + } + + location / { + try_files $uri $uri/ /index.html; + } +} +``` + +--- + +## Interaction Logic and Sync Flow + +A full pre-built artifact import and activation lifecycle looks like this. Binding a project for the first time requires publishing the main config; subsequent active deployment changes converge independently via the project latest: + +```text + [Admin / scanner] [Server control plane] [Agent] [OpenResty] + | | | | + |-- manual upload ----->|-- inspect / Ingest ---->| | + | |-- create candidate | | + |-- explicit activate -->|-- switch active | | + | | | | + |-- source sync ------->|-- inspect / Ingest | | + | |-- create/load + atomic activation | | + | | | | + |-- first bind & publish|-- broadcast project anchor -->|-- write/reload route -->| + | | | | + |-- later activate/sync/rollback -->|-- active latest changed -->| | + | |<-- latest metadata reconciliation ----| | + | |--- stream package -------------------->| | + | | |-- validate, extract, recheck --| + | | |-- atomic switch current ------>| +``` diff --git a/docs/en/design/tunnel-design.md b/docs/en/design/tunnel-design.md new file mode 100644 index 00000000..84955ad7 --- /dev/null +++ b/docs/en/design/tunnel-design.md @@ -0,0 +1,130 @@ +# Intranet Penetration Tunnel Design Document + +You will learn: The architectural design of the OpenFlare intranet penetration tunnel, the internal principles of the dual-ended control components (Relay and Client), their interaction logics, and the communication flows for the data plane and control plane. + +--- + +## Requirements Analysis + +In typical web application hosting scenarios, many origin servers (Origin Servers) are deployed in local intranet environments (such as local development machines, LAN servers, or firewalled private clusters). These servers typically suffer from: +1. **No Public IP**: Cannot be directly accessed by public internet traffic. +2. **Security Compliance Restrictions**: Creating port mappings (NAT) on border routers is strictly prohibited by security policies. +3. **Dynamic IP Changes**: Traditional DDNS solutions exhibit high latency and are highly unstable. + +To allow internal origin servers to seamlessly integrate into the OpenFlare global data gateway, benefiting from premium features like WAF geographic protection and TLS certificate hosting, OpenFlare designed an end-to-end solution based on a **reverse relay penetration tunnel**. In this architecture, public edge nodes act as reverse proxy entrances and traffic relays, while the intranet side only needs to initiate secure outbound connections to achieve secure and stable reverse penetration of public traffic to internal origin servers. + +--- + +## Core Capabilities + +The intranet penetration tunnel subsystem includes the following core capabilities: + +* **Dynamic Relay Node Management**: The control plane dynamically dispatches relay services (frps), distributing service ports and authentication tokens dynamically. +* **Multi-Tunnel Reverse Proxy Mapping**: Supports mapping multiple internal web ports on a single intranet client, binding multiple domain routes to corresponding relay nodes. +* **Independent Process Lifecycle Control**: Both the relay and client are independent daemon processes written in Go, responsible for spawning, monitoring, self-healing, and hot-upgrading the underlying frp engine. +* **Token-based Independent Authentication**: The relay uses `agent_token` for authorization, whereas the intranet client uses its dedicated `tunnel_token`, enforcing isolation of permissions and routing boundaries. +* **Validation & Incremental Hot Reload**: Config files are rewritten and processes are gracefully reloaded only when tunnel bindings, certificates, or Relay topologies change, reducing runtime overhead. + +--- + +## Intranet Penetration & Tunnel Architecture + +The intranet penetration subsystem is integrated on top of the mature and high-performance `frp` tunnel protocol, divided into the **Control Plane** and the **Data Plane**. + +```mermaid +graph TD + %% Data Flow + Browser[1. Browser / Visitor] -->|HTTPS Request| Agent[2. OpenResty / Agent] + Agent -->|Local proxy_pass| RelayFrps[3. OpenFlare Relay / frps] + RelayFrps -->|Encrypted Tunnel Protocol| FlaredFrpc[4. OpenFlared / frpc] + FlaredFrpc -->|Forward Local Request| LocalOrigin[5. Intranet Origin 192.168.x.x] + + %% Control Flow & Heartbeats + Server[OpenFlare Server Control Plane] <-->|Relay API / Heartbeat| RelayManager[openflare-relay process] + Server <-->|Client API / Heartbeat| ClientManager[openflared process] + + RelayManager -.->|Control Process & Config| RelayFrps + ClientManager -.->|Control Multi-Relay Processes| FlaredFrpc + + style Browser fill:#f9f,stroke:#333,stroke-width:2px + style LocalOrigin fill:#9f9,stroke:#333,stroke-width:2px + style Server fill:#f96,stroke:#333,stroke-width:2px +``` + +* **Control Plane**: The Server maintains the database state. The `openflare-relay` process on relay nodes and the `openflared` process on intranet servers synchronize tunnel configurations via HTTP heartbeats and long-lived WebSocket connections. +* **Data Plane**: Public traffic enters the public edge Agent (OpenResty), where the TLS handshake, HTTPS termination, and WAF filtering are executed. It is then forwarded via `proxy_pass` to the co-located `openflare-relay (frps)` on the loopback address. `frps` encapsulates the HTTP requests into the encrypted TCP tunnel and sends them down to the intranet `openflared (frpc)`. Finally, `frpc` unpacks the requests and forwards them to the actual intranet origin service. + +--- + +## Relay (Server-side) Design + +`openflare-relay` is a relay manager deployed on the public edge, running on nodes of type `tunnel_relay`. + +### 1. Core Architecture & Logic +* **Process Daemon**: The Relay process embeds the `frps` binary, spawning the `frps -c frps.toml` subprocess via `exec.Command` and using goroutines to asynchronously listen to its exit status. If `frps` exits unexpectedly, it automatically restarts using an exponential backoff policy. +* **Dynamic Configuration Rendering**: Periodically synchronizes status with the control plane via HTTP heartbeats to retrieve the active `RelayConfig`, including: + * `bindPort`: The public control port that frps listens to for incoming intranet frpc connections. + * `vhostHTTPPort`: The virtual host HTTP listening port where the Agent's proxy_pass points. + * `authToken`: The security credential used during the client connection handshake. + * `webServer`: Enables the frps dashboard API, which the Relay queries to collect active tunnel counts and traffic metrics. +* **Status Reporting**: In each heartbeat cycle, the Relay reports the active connections, registered clients, individual proxy tunnel statuses, and Relay version back to the Server. + +--- + +## Openflared (Client-side) Design + +`openflared` is the client manager running inside the user's intranet server, authenticated using a dedicated `tunnel_token`. + +### 1. Core Design Mechanisms +* **Multi-Relay Support (Multiplexing)**: + To guarantee high availability and geographical proximity, the control plane may schedule the client to connect to multiple public Relays. `openflared` parses the list of Relays dispatched in the `TunnelConfig`, generating dedicated configurations (`frpc_.toml`) and allocating distinct cancelable contexts for each Relay process locally. +* **Independent Subprocess Monitoring**: + `openflared` maintains a local `processes` map to manage the lifecycles of individual `frpc` subprocesses. When the control plane adds or removes Relays, the client incrementally spawns new processes or gracefully shuts down obsolete ones without affecting other functioning tunnels. +* **Dynamic TOML Generation**: + When rendering TOML configs for each Relay, the client iterates over the Proxies list, writing each intranet service's `LocalAddr`, `LocalPort`, and bound `CustomDomains` into standard `[[proxies]]` blocks. + +--- + +## Interaction Logic & Traffic Model + +The intranet penetration subsystem implements consistent version control and status feedback loops. + +### 1. Control Plane Publishing & Sync Flow + +```text +Admin modifies tunnel/intranet port mappings -> Click Publish -> Generate new Tunnel version & Checksum + | + v (Push or Heartbeat Pull) ++-----------------------------------------------------------------------+-----------------------------------------------------------------------+ +| | +v (Relay Side) v (Client Side) +openflare-relay heartbeat detects frps port/Token change openflared heartbeat detects tunnel_version change +Re-render local frps.toml Request full proxy configuration details +Kill and restart the frps process Re-render frpc_.toml configs +Report health status as healthy Restart or hot-reload changed frpc processes + Report application results (Apply Success/Error) +``` + +1. **Versioned Controls**: All intranet tunnel routes and mapping relationships are version-controlled, dispatching a unique `version` and `checksum` to ensure clients do not repeatedly write files or trigger redundant reloads. +2. **Closed-Loop Application Feedback**: After applying new configurations, the client reports the application result in the next heartbeat. If the intranet port is unreachable or certificate bindings fail, the client intercepts the stdout/stderr of the subprocess to report `LastError` to the Server, providing administrators with transparent error details. + +### 2. Data Plane Traffic Model +1. **Public Entrance (Agent)**: + ```nginx + server { + listen 443 ssl; + server_name intranet.example.com; + # ... TLS certificates & WAF filtering ... + location / { + proxy_pass http://127.0.0.1:18080; # Points to local frps vhost port + proxy_set_header Host $host; # Must preserve the original Host header, which frps relies on to route requests + proxy_set_header X-Real-IP $remote_addr; + } + } + ``` +2. **Relay Node (frps)**: + `frps` listens to the Vhost port `18080`. When an HTTP request arrives, it extracts `Host: intranet.example.com` from the request headers and searches its active registered tunnel registry to locate the matching encrypted TCP connection (initiated by the intranet frpc). +3. **Encrypted Tunnel Transmission (TCP)**: + `frps` encapsulates the HTTP request into the custom TCP tunnel protocol and transmits it down to the intranet `frpc` client. +4. **Intranet Client Distribution (frpc)**: + The `frpc` instance managed by `openflared` receives the payload, resolves it according to local settings (`localIP = "127.0.0.1"`, `localPort = 8080`), initiates a local TCP connection to forward the request to the intranet web service, and returns the response back through the tunnel to the public viewer. diff --git a/docs/en/design/waf-design.md b/docs/en/design/waf-design.md new file mode 100644 index 00000000..ae4fbbe3 --- /dev/null +++ b/docs/en/design/waf-design.md @@ -0,0 +1,132 @@ +# WAF Design Document + +You will learn: The core architecture of the OpenFlare edge Web Application Firewall (WAF), the dynamic IP group asynchronous differential sync model, the high-performance OpenResty Lua caching scheme, and the complete request filtering and decision logic. + +--- + +## Requirements Analysis + +In public internet environments, web applications face a wide variety of security threats (such as scanner profiling, api scraping, malicious botnets targeted at specific regions, ransomware, and CC attacks). Allowing malicious requests to pass directly to the origin server (Origin Server) results in: +1. **Origin Server Overload**: High-frequency database queries and intensive CPU computations easily exhaust server resources. +2. **Sensitive API Abuse**: APIs like login, registration, and SMS verification codes can be maliciously exploited, leading to financial and computational losses. +3. **Data Exposure Risks**: Malicious common vulnerability probing actions are not intercepted proactively. + +Therefore, OpenFlare needs to build a **high-performance, resiliently scalable WAF filtering engine** at the frontmost data plane layer (OpenResty). This engine is capable of executing deep filtering on malicious requests at the edge layer closest to users with sub-millisecond overhead. This relieves pressure on origin servers and provides core security capabilities like CC protection (PoW challenge), IP whitelisting/blacklisting, and region-level interception. + +--- + +## Core Capabilities + +OpenFlare WAF includes the following core protection dimensions: + +* **IP Interception (IP Whitelist/Blacklist)**: Supports filtering by single IP or CIDR block, and aggregating tens of thousands of IPs into IP groups for highly efficient matching. +* **Geographical Whitelist/Blacklist (GeoIP Limit)**: Integrates MaxMind databases to support precise admission controls based on countries and provinces/regions. +* **Custom Interception Responses**: Supports custom block status codes (e.g., 403, 418) and personalized HTML block pages for different filtering rules. +* **Human-Machine Challenge (PoW CC Protection)**: Supports seamless client-side PoW challenges, calculating Hash collisions to prevent automated scripts and botnets from hitting endpoints concurrently. + +--- + +## IP Group Design & Dynamic Asynchronous Sync + +IP groups are the core containers for highly efficient IP whitelisting and blacklisting. OpenFlare classifies IP groups into three types based on their update frequencies and source channels: + +### 1. IP Group Types +* **Manual**: Manually input by administrators in the control panel. Primarily used for static trusted IPs or long-term blocks. +* **Subscription**: Configured with remote text feeds (one IP/CIDR per line) or standard JSON subscription URLs. Server-side cron jobs periodically fetch and parse the remote subscription sources. Primarily used for integrating open-source threat intelligence feeds, cloud provider IP ranges, etc. +* **Automatic**: **The most resilient dynamic protection channel**. Control plane scanning jobs read access logs from all nodes, performing aggregation and analysis based on configured Expr rules (e.g., "requesting the `/api/login` endpoint over 50 times with a 401 status code in 5 minutes"). Once matched, the source IP is automatically added to a temporary block list for a specified duration. + +### 2. Asynchronous Differential Sync Design (No Nginx Reload) +In traditional Nginx WAF designs, IP blacklist updates typically require writing configurations and executing reloads. If malicious IP blocks occur at high frequencies (seconds or minutes), frequent reloads force Nginx to constantly spawn new worker processes and tear down old ones, severely degrading performance. + +OpenFlare adopts a **dynamic IP group asynchronous differential sync design**: + +```text +WAF IP member updates (Manual/Subscription/Auto-trigger) + | + v +Server updates the database and calculates the new MD5 Checksum of the IP group + | + +----------------------------------------+ + | (WebSocket Real-time Broadcast) | (Heartbeat Fallback Comparison) + v v +Server immediately pushes complete members Agent heartbeats report the local IP groups +of modified groups to all Agents checksum mapping table + | | + | v + | Server detects Checksum mismatch and dispatches + v the modified IP group members +Agent receives member data and writes it as JSON to local disk: waf_ip_groups.json + | + v (Lua Memory Awareness) +OpenResty Lua engine detects file changes via MD5 checksum in seconds and hot-updates its memory, +completely bypassing Nginx process reloads. +``` + +Through this architecture, the persistence and activation of tens of thousands of highly volatile dynamic blacklist IPs **require absolutely no Nginx reloads**, maximally protecting the high-concurrency throughput of the gateway. + +--- + +## Rule Groups & Site Bindings + +* **WAF Rule Group**: The smallest logical collection of WAF filtering policies. A single rule group can contain IP whitelists/blacklists, IP group references, regional restrictions, and CC protection configurations. +* **Global Rule Group**: When a rule group is marked as `is_global = true`, it takes effect on **all website routes** hosted on the node by default. +* **Site Binding**: Website routes (`proxy_routes`) can bind one or more non-global rule groups. During request validation, WAF evaluates the union of `Global Rule Group + Bound Rule Groups`. + +--- + +## Implementation Details & High-Performance Caching + +WAF is triggered in the OpenResty `access_by_lua` phase, implemented primarily through Lua files and local JSON configurations. + +### 1. Physical Structures +* `waf_config.json`: Contains metadata for all rule groups, geographic country/region limits, and website-to-rule-group bindings. +* `waf_ip_groups.json`: Contains all synchronized IP groups and their corresponding IP lists. +* `waf/runtime.lua`: The actual runtime engine responsible for WAF rule comparison. +* `waf/check.lua`: The entry point for the access layer, handling packages inclusion and triggering `check()`. + +### 2. Shared Memory Dictionary (ngx.shared) High-Performance Cache Design +Reading JSON files from the disk and decoding them upon every incoming web request would make disk I/O a severe performance bottleneck. + +OpenFlare leverages the **OpenResty Shared Memory Dictionary (ngx.shared.openflare_waf_config)** to implement a two-level caching mechanism: + +1. **Zero File I/O Path**: + In Lua, every time `check()` executes, it first computes the MD5 hash of the local JSON file using `ngx.md5` (which takes virtually zero time since the file is cached in the OS Page Cache). +2. **Hash Comparison & Hot Loading**: + It compares this against the cached hash key (`_config_hash`) stored in the shared memory dictionary. + * **If the hash is unchanged**: It reads the pre-decoded Lua Table configuration stored directly in shared memory. The entire verification runs purely in **shared memory**, completing in **microseconds**. + * **If the hash is mismatched**: Indicating that the Agent has just updated the WAF rules or IP groups on the disk, the Lua engine automatically reads the disk file, decodes it via `cjson.decode`, writes the decoded data and the new MD5 hash into shared memory, and makes it seamlessly readable by all subsequent worker processes. + +--- + +## Application Flow & Decision Judgment Control Logic + +When an HTTP/HTTPS request arrives at OpenResty, WAF evaluates and intercepts it step-by-step in the `access` phase according to the funnel decision chain below: + +### 1. WAF Decision Flowchart + +```mermaid +flowchart TD + A[Request enters access phase] --> B[Get Site Name of current request] + B --> C[Load all active rule groups bound to this Site in shared memory] + C --> D{Matches IP whitelist or Whitelist IP group?} + D -- Yes (Matched) --> E[Pass request - ALLOW] + D -- No --> F{Matches country/region whitelist?} + F -- Yes (Matched) --> E + F -- No --> G{Matches IP blacklist or Blacklist IP group?} + G -- Yes (Matched) --> H[Block request - BLOCK] + G -- No --> I{Matches country/region blacklist?} + I -- Yes (Matched) --> H + I -- No --> J{Is CC PoW verification enabled?} + J -- Yes --> K[Transfer to CC Protection module] + J -- No --> L[No security risks, pass normally] + + H --> M[Exit and return custom status code and block page HTML configured in the rule group] +``` + +### 2. Decision Step Details +1. **Whitelist Precedence**: + To prevent false positives and guarantee smooth passage of core back-to-source traffic (such as search engine spiders, CDN back-to-source IPs, and office egresses), WAF **prioritizes matching IP whitelists and regional whitelists**. Once a whitelist matches, it immediately bypasses all subsequent blacklist checks and CC challenges. +2. **Blacklist Aggressive Block**: + If a request is not captured by the whitelist evaluation, it enters the blacklist funnel. Once the source IP matches an IP blacklist, a referenced blacklist IP group, or lies within a prohibited country/region, the Lua engine immediately marks `ngx.ctx.openflare_waf_blocked` as `true`. +3. **Response Output**: + Upon hitting the blacklist, Lua extracts the `block_status_code` (defaults to 418 or 403) and `block_response_body` (interception HTML page) configured in the matching rule group. It outputs the response body via `ngx.say()` and gracefully terminates the request using `ngx.exit(status)` to prevent the request from passing upstream. diff --git a/docs/en/design/waf-orchestration-design.md b/docs/en/design/waf-orchestration-design.md new file mode 100644 index 00000000..97dce7c5 --- /dev/null +++ b/docs/en/design/waf-orchestration-design.md @@ -0,0 +1,118 @@ +# WAF Orchestration Rule Design + +This document defines the target architecture, data model, execution semantics, release model, and migration boundaries for reworking OpenFlare WAF from a fixed decision chain into a visual directed acyclic graph (DAG). IP group sources and membership computation still follow [WAF Design](./waf-design.md); this document only changes how rules are composed and executed. + +## Goals and Boundaries + +When a user adds a WAF rule, they only enter a name. The Server immediately creates a legal default graph `start → pass`, and the frontend enters a standalone orchestration page based on React Flow. Users build policies by adding processing units, configuring nodes, and connecting branches — no more filling in fixed-order allow/block lists and PoW forms. + +Supported nodes: + +| Node | Count Constraint | Inputs | Outputs | Config | +| --- | --- | --- | --- | --- | +| Start | exactly one per graph | none | `next` | none | +| Pass | exactly one per graph | one or more | none | none | +| Block | multiple allowed | one or more | none | HTTP status code, HTML response body | +| IP match | multiple allowed | one or more | `true`, `false` | IP, CIDR, IP group ID | +| Geo match | multiple allowed | one or more | `true`, `false` | country code, region code | +| UA check | multiple allowed | one or more | `true`, `false` | UA required, browser/OS allowlist with and/or, block crawlers/abnormal UAs (excluding crawlers)/custom regex | +| Security | multiple allowed | one or more | `true`, `false` | basic signature detection (path traversal/file inclusion on by default; SQL/XSS/command injection/SSRF/upload/XXE/CRLF toggleable); any enabled rule hit → false | +| PoW | multiple allowed | one or more | `next` | algorithm, difficulty, session TTL, challenge TTL | + +IP match, geo match, UA check, and security do not distinguish allowlist vs blocklist. `true` only means the request passed that node's judgment; `false` only means it did not; the business meaning of allow vs block is entirely determined by the wiring. UA check evaluation order: require UA → block crawlers/abnormal UAs → allowlist match. Security performs signature matching on the request Path/Query/Header/Cookie/Body (bounded). After PoW verification succeeds, execution continues along `next`; when incomplete, the challenge page takes over the request and no `false` branch is produced. + +Loops, script nodes, arbitrary expression nodes, subgraph calls, and cross-rule jumps are not implemented. + +## Control-Plane Architecture + +The rule graph uses a dual model separating control-plane edit state and data-plane runtime state: + +1. The React Flow editor submits a versioned graph JSON containing node IDs, node types, display names, coordinates, typed configs, and edges. +2. The Server performs authoritative validation of the whole graph; on success it saves the graph in a single transaction and increments the revision number. +3. On config release, the Server validates all enabled rules again, compiles the graph into a compact runtime DAG without UI fields (coordinates, labels, etc.), and collects the referenced IP group IDs. +4. The Agent atomically writes the full release snapshot and reloads OpenResty. New Workers only load and parse the rule JSON once at startup. +5. The request hot path only traverses the immutable in-memory runtime graph in the Worker — no file reads, no checksum computation, no JSON parsing. + +The edit-state JSON uses an explicit `schema_version`. Node configs use per-node-type structures; unconstrained key-value objects bypassing Server validation are not allowed. Initial safety limits: 128 nodes, 256 edges, and 256 KiB of edit-state JSON per rule; these limits are enforced by both the API and the release compiler. + +## Graph Structure Constraints + +Rule save and release must satisfy all constraints: + +* The graph is a DAG; self-loops and arbitrary cycles are forbidden. +* Exactly one start node and one pass node exist; multiple block nodes are allowed. +* The start node has no incoming edges and exactly one `next` outlet; pass and block nodes have no outlets. +* IP match, geo match, UA check, and security must each connect their `true` and `false` outlets once; PoW's `next` must connect once. +* No dangling outlets except terminal nodes; every non-start node has at least one incoming edge. +* All nodes must be reachable from start, and every executable node must be able to reach pass or block. +* Edge source ports must belong to the source node type; the same source port must not connect to multiple targets. +* Node IDs are unique within the graph, edge IDs are unique within the graph, and all referenced nodes must exist. +* Node configs must pass per-type field, range, reference-existence, and size validation. + +The frontend provides instant validation and connection restrictions to improve UX, but the Server is the only authoritative validator. When deleting a node, the frontend synchronously removes related edges and marks the rule unsaved; saving is forbidden until the graph is legal again. + +## Multi-Rule Execution Semantics + +A route can bind multiple custom rules. The binding is an ordered list following this order: + +1. Enabled global rules always execute first and do not participate in route-side ordering. +2. Enabled rules bound to the route execute in binding order. +3. When the current rule reaches a block node, it immediately outputs that node's configured response and terminates the request. +4. When the current rule reaches a pass node, that only means the current rule finished; if more rules remain, execution continues. +5. Only after all rules reach a pass node is the request truly allowed to proceed into the OpenResty/origin chain. + +The runtime graph is fully validated before release. If the Lua executor still encounters an unknown node, unknown port, missing target, or exceeds the node-step limit, it logs a rate-limited error and blocks the request, preventing a broken security config from accidentally allowing traffic. + +## IP Group Memory Refresh + +Rule topology only takes effect on release + OpenResty reload; IP group membership can still be updated independently by manual, subscription, or auto tasks without release or reload. + +IP groups use a two-level cache of coordinating worker, shared snapshot, and worker-local objects: + +1. Requests always read the IP group object in the current worker's memory — no file access or shared-dict JSON parsing. +2. Every 5 seconds only one worker holding a shared lock reads the lightweight checksum file. +3. If the checksum is unchanged, it ends immediately without reading the full `waf_ip_groups.json`. +4. On checksum change, the coordinating worker reads and validates the full JSON once, writes the raw snapshot to a dedicated 64 MiB `ngx.shared.openflare_waf_ip_groups` keyed by checksum, then updates the commit pointer. +5. Other workers detecting a shared version change fetch the snapshot from shared memory, parse it, and atomically replace their local object — no repeated disk reads. +6. On refresh failure, keep using the previous valid object, log a rate-limited error, and retry next cycle. + +The Agent must atomically replace the IP group JSON first, then atomically update the checksum last, so workers never recognize a half-written file as a new version. Server release/sync and Agent disk writes jointly enforce the 20 MiB aggregate snapshot limit; shared dict uses a safe write that never force-evicts old keys, keeping the current and previous immutable snapshots on failure. + +## API and Editor + +The create API only accepts a rule name and returns the rule detail with the default graph. Rule metadata, graph save, and route binding use separate operations, so toggling enabled state or binding does not overwrite the canvas. + +The graph detail includes `revision`. Save requests submit `revision + graph`; the Server only updates and increments the revision when it matches. On mismatch it returns a conflict, the frontend prompts a reload, and silent overwrites of another page's changes are forbidden. The route binding API accepts an ordered array of rule IDs. + +The React Flow editor page uses a full-width canvas and a fixed right property panel: + +* Top bar: back, rule name, enabled state, validation state, and save. +* The canvas uses compact height with a smaller initial fit scale; zoom, pan, box-select, delete, auto-layout, MiniMap/Controls and other necessary navigation are supported. Node dragging is handled by React Flow's local controlled state in real time; coordinates are written back to the edit graph only after the drag ends. +* "Add processing unit" offers IP match, geo match, UA check, security, PoW, and block; start and pass are provided by the default graph and cannot be deleted or duplicated. +* Selecting a normal node or edge allows deletion via the canvas delete button or Delete/Backspace; deleting a node synchronously removes associated edges. +* The right property panel is hidden by default, shown only when a node is selected; it collapses when clicking an edge or blank canvas. +* Geo-match properties use the full country and ISO 3166-2 first-level administrative division data; country options show both localized names and codes; administrative divisions support search by country name, division name, or code to avoid rendering thousands of options at once. +* Leaving the page with unsaved changes must prompt; save conflicts and Server validation errors should locate the relevant node or edge. + +The WAF list shows rule name, enabled state, node count, bound route count, and update time. The new-rule dialog only has the name field and navigates to the orchestration page immediately on success. + +## Persistence and Migration + +Rule records add a versioned graph JSON and a revision number; binding records add execution order. The graph is saved as a single aggregate (not split into node/edge tables) to keep edit operations transactional and let new node types avoid frequent DB schema extensions. + +When upgrading existing installs: + +* Keep rule names, global flags, enabled state, and route bindings. +* All rule graphs reset to `start → pass`; legacy IP/geo lists, PoW, or block-response configs are not migrated. +* Existing bindings are written into the order field in a stable sequence; global rules remain fixed in front. +* Once the new graph and runtime stabilize, remove the legacy rule fields, fixed-order compile logic, and old frontend forms — do not maintain dual executors long-term. + +This migration stops the old protection config from taking effect; the upgrade notes must prominently tell admins to re-orchestrate rules before releasing the next version. + +## Release, Failure, and Rollback + +Rule graphs only take effect on config release. If validation or compilation fails before release, the release is refused and the current active version stays unchanged. If Agent write, OpenResty config check, or reload fails, the apply flow fails and restores the previous valid released version. + +New workers only accept complete, parseable runtime rule configs. During an OpenResty graceful reload, old workers keep the old in-memory graph and new workers use the new graph, so requests never observe a half-updated state. + +When the geo database is unavailable, geo match returns `false` with a rate-limited warning, preserving current behavior. On IP group refresh failure, the old in-memory snapshot is kept. An incomplete PoW is taken over by the challenge module, not treated as an execution error; PoW node config is first written to OpenResty shared memory with a short-lived key, then passed to the internal challenge handler via explicit `ngx.exec` parameters — it cannot rely on internal redirects preserving `ngx.ctx` or implicitly inherited request params. Empty rule bindings in the release snapshot must be encoded as JSON empty arrays; at runtime, legacy `null` optional arrays in old snapshots are treated as empty arrays, so `cjson`'s `ngx.null` userdata never breaks the request. diff --git a/docs/en/design/zone-design.md b/docs/en/design/zone-design.md new file mode 100644 index 00000000..f3fab33f --- /dev/null +++ b/docs/en/design/zone-design.md @@ -0,0 +1,93 @@ +# Zone & Domain Resource Design + +## Goals + +Refactor "websites" into a Zone management experience keyed by registrable root domains. A Zone like `example.com` is a stable management boundary; users enter the Zone through a stable ID path to view and maintain its explicitly declared domains, the reverse proxy routes and certificates bound to those domains, and route-level WAF, Pages and other capabilities. + +This design replaces the concept, tables, and APIs of `managed_domains`. The Zone core does **not** include authoritative DNS record management; to point ZoneDomain A records at edge nodes, use the optional module [Cloudflare DNS Pointing](./cloudflare-pointing.md). + +## Scope and Constraints + +* Zone root domains are resolved with the Public Suffix List, e.g. `api.example.co.uk` belongs to `example.co.uk`. +* URLs use IDs: list at `/websites`, detail at `/websites/:zoneId`; domains are not used as URL parameters. +* Zone domains must be explicit FQDNs; `*.example.com` is not allowed. TLS certificates may still contain wildcard SANs and cover explicit Zone domains. +* A Zone domain is associated with at most one reverse proxy route; a route may associate with multiple Zone domains, thus sharing the same upstream, cache, rate limit, WAF and Pages config across Zones. +* The Zone model itself adds no DNS records, edge functions, preview subdomains, or tenant isolation. Creating/updating external DNS A records is handled by the separate Cloudflare pointing module and does not change the Zone / ZoneDomain table responsibilities. + +## Core Model + +```mermaid +erDiagram + ZONES ||--o{ ZONE_DOMAINS : contains + PROXY_ROUTES ||--o{ ZONE_DOMAINS : serves + TLS_CERTIFICATES ||--o{ ZONE_DOMAINS : secures + PROXY_ROUTES ||--o{ WAF_RULE_GROUP_BINDINGS : applies + PAGES_PROJECTS ||--o{ PROXY_ROUTES : backs + + ZONES { + uint id PK + string domain UK + } + ZONE_DOMAINS { + uint id PK + uint zone_id + uint proxy_route_id + string domain UK + uint cert_id + } +``` + +### `of_zones` + +Stores the root domain, created time, and updated time. Root domains are globally unique and cannot be modified in place after creation; to change one, create a new Zone and migrate the domains. Before deleting a Zone, all of its Zone domains must be cleared first. + +### `of_zone_domains` + +Stores `zone_id`, explicit `domain`, nullable `proxy_route_id`, nullable `cert_id`, and timestamps. `domain` is globally unique; all relationship fields are indexed but no physical foreign keys are created. `proxy_route_id` may be null to host historical domains that have a certificate prepared but no reverse proxy configured yet. + +`of_proxy_routes` gradually removes the domain/certificate redundancy columns `domain`, `domains`, `cert_id`, `cert_ids`, and `domain_cert_ids`. Routes must no longer specify any TLS certificate; the route name `site_name` becomes the stable human-readable identifier, and the compiler reads `server_name` and its `cert_id` from the associated Zone domains. This gives each explicit domain a single certificate source. + +## Business and API + +New Zone resources in the admin panel: + +* `GET/POST /api/v1/d/zones` +* `GET/POST /api/v1/d/zones/:id/update` +* `POST /api/v1/d/zones/:id/delete` +* `POST /api/v1/d/zones/:id/domains` (list returned via overview) +* `POST /api/v1/d/zones/:id/domains/:domainID/update` +* `POST /api/v1/d/zones/:id/domains/:domainID/delete` +* `GET /api/v1/d/zones/:id/overview` + +Reverse proxy route create/update requests switch to `zone_domain_ids` and no longer submit `domains`, `cert_id`, `cert_ids`, or `domain_cert_ids`. The server validates domain ownership, global uniqueness, and certificate SAN coverage in a transaction; failures are returned uniformly via `response.Abort*`. Deleting a Zone domain bound to a route requires unbinding or deleting the route first; deleting a Zone that still has domains must be rejected. + +WAF, Pages, upstream, and release versions remain part of `proxy_routes`. The Zone overview only aggregates the route state associated with its domains and does not copy or redefine those configs. + +## Frontend Experience + +`/websites` shows only Zone root domains with configured domain count, route count and status, plus search, create, and action menus. Clicking enters `/websites/:zoneId`. + +The detail page includes: + +* Overview: domain, route, and valid certificate statistics; domain—route—certificate summary; route-level WAF and Pages summary. +* Domains: a list of explicit FQDNs, certificate selection, and associated routes; wildcard domains are not shown or accepted. +* Routes: routes filtered to the current Zone, linking to existing route details. +* Certificates: certificates actually referenced by the current Zone's domains. +* Settings: Zone notes and a protected delete operation. + +When adding a route, select from Zone domains; users can also register domains in the Zone first, then bind a route. The global reverse proxy route entry remains but uses the same Zone domain selector. + +## Data Migration + +This rework ships in two release phases to avoid SQL using a wrong "last two labels" rule for multi-level public suffixes. Operation details: [Zone Domain Migration and Release Acceptance](../guide/zone-domain-migration.md). + +1. **Phase 1 DDL**: PostgreSQL and SQLite goose create `of_zones` / `of_zone_domains` at the same version; `of_managed_domains` and route redundancy columns are temporarily kept. +2. **Data import (automatic)**: at Server startup `migrator.Migrate()` first applies goose SQL up to `202607120002`, then automatically imports legacy route domains / `managed_domains` (registering root domains via `publicsuffix` parsing, writing `cert_id` and `proxy_route_id`), then continues with the remaining SQL. Conflicts fail startup; fixing and restarting retries idempotently. No manual command needed. +3. **Code switch**: control-plane APIs, config snapshots, rendering, and frontend all use Zone domains as the single source; route writes only use `zone_domain_ids`. +4. **Phase 2 cleanup**: goose SQL `202607130001_drop_legacy_route_domain_columns` drops `of_managed_domains` and the `of_proxy_routes` redundancy columns. Down only restores an empty dev-DB structure and does not backfill historical data. + +### Runtime Model Boundaries + +* Persistence: domains and certificates exist only in `of_zone_domains`; `of_proxy_routes` only stores route policy (upstream, cache, rate limit, WAF binding keys, etc.). +* Rendering: the config snapshot assembles temporary `Domains` / `DomainCertIDs` in memory for OpenResty rendering and does not write back to the database. +* Structure migration only uses `internal/infra/persistence/migrator/goose/{postgres,sqlite}/*.sql`; legacy domains are imported automatically at startup, and after phase 2 the old columns no longer exist so it is a no-op. diff --git a/docs/en/guide/certificates.md b/docs/en/guide/certificates.md new file mode 100644 index 00000000..54137f7f --- /dev/null +++ b/docs/en/guide/certificates.md @@ -0,0 +1,68 @@ +# TLS Certificates and Auto-Renewal + +This guide explains how to manage TLS certificates in OpenFlare. To secure traffic with HTTPS, you need to configure the corresponding certificate. OpenFlare supports **manually importing existing certificates** and **automatic issuance and managed renewal via ACME**. + +--- + +## Method 1: Manually Import an Existing Certificate + +If you have obtained a free or paid certificate from a third-party provider (such as Tencent Cloud, Alibaba Cloud, etc.), or generated a self-signed certificate locally: + +1. Log in to the admin panel, go to **「Website Management」->「TLS Certificates」** in the left navigation. +2. Click **「Import Certificate」** in the top-right corner. +3. Fill in the configuration: + * **Certificate Name**: Enter an easily recognizable alias (e.g. `my-domain-cert`). + * **Certificate Content (PEM)**: Paste the PEM-format certificate public key (usually starts with `-----BEGIN CERTIFICATE-----`). + * **Private Key (KEY)**: Paste the certificate private key (usually starts with `-----BEGIN PRIVATE KEY-----` or `-----BEGIN RSA PRIVATE KEY-----`). +4. Click **「Save」**. After a successful import, the certificate can be directly bound when configuring domains. + +--- + +## Method 2: Automatic Issuance and Auto-Renewal (ACME) + +OpenFlare has a built-in ACME client integrated with the **Asynq async task queue**. With the DNS API of your cloud DNS provider, the system can automatically complete DNS-01 challenge validation, apply for wildcard/single-domain certificates from a CA (Let's Encrypt by default), and **automatically trigger renewal 7 days before expiry**. + +### Step 1: Create a DNS API Token in Cloudflare + +To let OpenFlare automatically add TXT records under your domain for DNS validation, you need a Cloudflare API Token with specific permissions. + +> [!IMPORTANT] +> For security, **it is strongly recommended to use a permission-scoped API Token** rather than the Global API Key. + +1. Log in to the [Cloudflare dashboard](https://dash.cloudflare.com/). +2. Click the user avatar in the top-right corner and select **「My Profile」**. +3. In the left menu select **「API Tokens」**, then click **「Create Token」**. +4. Find the **「Edit Zone DNS」** template and click **「Use template」**. +5. Configure the token permissions and scope (keep defaults or restrict as needed): + * **Permissions**: + * `Zone` - `DNS` - `Edit` (required, for ACME to write TXT records) + * `Zone` - `Zone` - `Read` (required, to list and retrieve zone IDs) + * **Zone Resources**: + * Select **「Include」** -> **「All zones」**, or select **「Specific zone」** and point to the specific domain you manage. +6. Click **「Continue to summary」**, confirm, then click **「Create Token」**. +7. Copy the generated **API Token** string. It is only shown once, so save it carefully. + +### Step 2: Add a DNS Account in the Control Plane + +1. Log in to the OpenFlare admin panel, go to **「Website Management」->「DNS Accounts」**. +2. Click **「Add Account」**. +3. Fill in the configuration: + * **Account Name**: e.g. `cloudflare-main`. + * **DNS Provider**: Select `Cloudflare`. + * **API Token**: Paste the API token copied from Cloudflare (stored encrypted automatically). +4. Click **「Save」**. + +### Step 3: Submit a Certificate Application Task + +1. Go to **「Website Management」->「TLS Certificates」**, click **「Apply for Certificate」** in the top-right corner. +2. Fill in the application form: + * **Certificate Name**: Custom name (e.g. `wildcard-example-cert`). + * **Primary Domain**: The domain to apply for (wildcards supported, e.g. `example.com` or `*.example.com`). + * **Associated Domains**: Append more domains if any (wildcards supported, comma-separated). + * **DNS Account**: Select the DNS account just added from the dropdown (e.g. `cloudflare-main`). +3. Click **「Save and Apply」**. + +### Step 4: Track Application Progress and Renewal Status + +- **Real-time progress**: After saving, the system delivers a certificate renewal/application task (`of_ssl_single_renew`) to the Asynq queue. You can view detailed step-by-step logs (adding TXT records, DNS record global propagation probing, ACME validation, certificate issuance, etc.) in the admin task or node log pages. +- **Automatic renewal**: All certificates issued via ACME are automatically managed by the system. The background Scheduler scans certificate validity daily and automatically triggers renewal via async tasks 7 days before expiry — no manual maintenance needed. diff --git a/docs/en/guide/credits.md b/docs/en/guide/credits.md new file mode 100644 index 00000000..0a1fee57 --- /dev/null +++ b/docs/en/guide/credits.md @@ -0,0 +1,27 @@ +# Credits + +OpenFlare is essentially a solution integration project. During its design and implementation phases, it drew inspiration from the exceptional concepts, architectural designs, and technical achievements of numerous open-source projects. Below are the key upstream open-source projects OpenFlare relies on for its core engine, security mechanisms, and backend/frontend system frameworks, along with our sincere thanks to these projects and their active communities. + +--- + +### 1. OpenResty +* **Project Positioning**: A high-performance Web platform based on Nginx and Lua. +* **Role in OpenFlare**: Acts as the edge gateway for the global Data Plane. All public web traffic is received by OpenResty first, where high-concurrency HTTPS handshakes, WAF security rule evaluations, and PoW CC verification are performed before executing reverse proxies. +* **Project Link**: [OpenResty Official Website](https://openresty.org/) + +### 2. FRP (Fast Reverse Proxy) +* **Project Positioning**: A high-performance reverse proxy application focused on intranet penetration. +* **Role in OpenFlare**: Serves as the underlying tunnel engine for the intranet penetration subsystem. The relay-side manager `openflare-relay` is responsible for running and scheduling the `frps` engine, while the intranet client `openflared` is responsible for generating TOML configurations locally and running the multiplexed `frpc` subprocesses. +* **Project Link**: [fatedier/frp (GitHub)](https://github.com/fatedier/frp) + +--- + +### 3. Anubis (PoW Solution) +* **Project Positioning**: A lightweight human-machine verification and protection solution based on Proof of Work (PoW). +* **Role in OpenFlare**: Provides the core **seamless PoW CC challenge** capabilities for the gateway WAF. + +--- + +### 4. gin-template +* **Project Positioning**: A modern full-stack development boilerplate based on Go Gin and frontend builds. +* **Role in OpenFlare**: Provided the standard, unified backend/frontend system architecture baseline for the OpenFlare control plane (Server). diff --git a/docs/en/guide/first-site.md b/docs/en/guide/first-site.md new file mode 100644 index 00000000..4412afa3 --- /dev/null +++ b/docs/en/guide/first-site.md @@ -0,0 +1,100 @@ +# Publishing Your First Site + +You will learn: How to create your first website configuration, bind origins and certificates, publish the configuration version, and verify that the Agent applied it successfully. + +The publishing pipeline of OpenFlare centers on a complete configuration version snapshot. After modifying website configurations in the management console, you need to publish and activate the new version to let the Agent pull and apply it in the next heartbeat. + +## Pre-publish Checks + +Verify that the following conditions are met: + +| Item | Expectation | +| --- | --- | +| Server | Management console is accessible and log-in succeeds | +| Agent | At least one node is online | +| Origin | The Agent node can reach the origin server address | +| Domain | Domain is resolved to the OpenResty node, or prepared to verify via local `hosts` / `curl` Host header | +| HTTPS | If HTTPS is required, the certificate is uploaded or hosted | + +## Create Website Configuration + +A new website configuration requires at least: + +| Field | Description | +| --- | --- | +| Website Name | Business unique identifier; the primary domain is used if left blank | +| Domain | At least one domain, where the first is treated as the primary domain | +| Origin Address | A valid `http://` or `https://` upstream address | +| Enabled Status | Only enabled website configurations will participate in publishing and rendering | + +Example: + +| Field | Example | +| --- | --- | +| Website Name | `app` | +| Domain | `app.example.com` | +| Origin Address | `http://10.0.0.20:8080` | + +A single domain can belong to only one website configuration. Rate limiting, reverse proxy, and caching parameters are shared site-wide. + +## Bind Certificate + +HTTPS certificates are bound by domain. Domains without a bound certificate will not be placed into `443 ssl` server blocks automatically. + +If a website contains multiple domains, the rendering pipeline groups the HTTPS configurations by certificate while ensuring all domains belong to the same site snapshot. + +## Publish & Activate + +Standard Pipeline: + +```text +Modify rules -> Preview / Diff -> Publish -> Generate complete version -> Activate version -> Agent pulls -> Local application -> Report result +``` + +During publication, the Server reads all enabled website configurations, the main OpenResty config templates, performance and cache parameters, rendering the complete OpenResty configuration and calculating its `checksum`, saving to `config_versions`, and switching the active version. + +## Verify Results + +Verify in the management console after publishing: + +| Position | Expected Result | +| --- | --- | +| Node List | Node status is online | +| Node Details | Current version matches active version | +| Apply Logs | Most recent application succeeded | +| Version Page | The new version is currently active | + +Verify Agent logs on the node: + +```bash +journalctl -u openflare-agent -n 100 --no-pager +``` + +Access via domain: + +```bash +curl -I http://app.example.com +``` + +If the domain has not been officially resolved, you can verify by specifying the Host header against the node IP: + +```bash +curl -I -H 'Host: app.example.com' http://NODE_IP +``` + +HTTPS Validation: + +```bash +curl -I https://app.example.com +``` + +## Rollback + +If a target version application fails and triggers a rollback, the Agent blocks repeated synchronization of the same failing `version + checksum` until the active version or checksum changes on the control plane. + +Roll back to an older version: + +1. Open the Configuration Versions page. +2. Locate the last known good historic version. +3. Re-activate that version. +4. Check the node application logs to verify that the Agent successfully applied the rollback. diff --git a/docs/en/guide/index.md b/docs/en/guide/index.md new file mode 100644 index 00000000..e72964d0 --- /dev/null +++ b/docs/en/guide/index.md @@ -0,0 +1,43 @@ +# Guide Overview + +You will learn: How the OpenFlare documentation is organized, which pages to read when running it for the first time, and where to start for deployment, usage, troubleshooting, and development. + +OpenFlare is a self-hosted OpenResty control plane. It integrates reverse proxy website configurations, configuration version publishing, Agent node synchronization, TLS certificates, and basic observability into a single management console, making it ideal for a single team or organization managing multiple proxy nodes. + +## Recommended Reading Path + +If you are new to OpenFlare, read the documents in the following order: + +1. [Quick Start](./quick-start.md): Start the Server using Docker Compose, log into the management console, and connect your first Agent. +2. [Basic Usage](./usage.md): Learn common operations for website configs, origins, certificates, publishing, rollbacks, and observability. +3. [Tunnel & Intranet Penetration](./tunnel-usage.md): Learn to deploy Relay and Client to achieve secure, public IP-free reverse penetration. +4. [WAF Security Protection](./waf-usage.md): Master IP whitelisting/blacklisting, WAF auto IP group aggregation Expr rules, geographical restrictions, and PoW CC protection. +5. [WAF Auto IP Group Expressions](./waf-ip-group-expr.md): Write auto IP group Expr rules and learn keyword definitions and presets. +6. [Deployment Guide](../deployment/deployment.md): Deploy Server and Agent in closer-to-production environments. +7. [Configurations Reference](../reference/configuration.md): Check Server environment variables, runtime Options, and Agent configurations. +8. [Troubleshooting](./troubleshooting.md): Troubleshoot login, database, node sync, OpenResty application, and frontend build issues. + +## Role-Based Entrypoints + +| What do you want to do? | Recommended Entrance | +| --- | --- | +| Run the console in under 5 minutes | [Quick Start](./quick-start.md) | +| Publish your first reverse proxy configuration | [Publish First Configuration](./first-site.md) | +| Configure intranet penetration mapping | [Tunnel & Intranet Penetration](./tunnel-usage.md) | +| Configure CC protection & IP group blocking | [WAF Security Protection](./waf-usage.md) | +| Write auto IP group aggregation rules | [WAF Auto IP Group Expressions](./waf-ip-group-expr.md) | +| Connect or reinstall a node Agent | [Access Agent](../deployment/agent.md) | +| Start Server from source code | [Launch Server](../deployment/server.md) | +| Configure GitHub or OIDC SSO | [SSO Login Configuration](./sso.md) | +| Upgrade Server or Agent | [Upgrade & Maintenance](../deployment/upgrade.md) | +| Participate in development or bug fixing | [Local Development](../design/development.md) and [Development Constraints](../../guideline/Constraints.md) | +| Understand architecture and publishing | [System Architecture](../design/architecture.md) and [Agent & Publish Model](../design/agent-design.md) | +| View open-source references and credits | [Credits](./credits.md) | + +## Documentation Partitions + +`guide/` is oriented toward users and deployers, providing actionable steps from installation to daily operations. + +`reference/` collects stable facts such as configuration fields, commands, API response structures, and repository layout. + +`design/` is oriented toward maintainers and contributors, describing product boundaries, system architecture, Agent & publishing models, and engineering constraints. Before adding capabilities or changing boundaries, update the corresponding design document first. diff --git a/docs/en/guide/pages-usage.md b/docs/en/guide/pages-usage.md new file mode 100644 index 00000000..c2c0b939 --- /dev/null +++ b/docs/en/guide/pages-usage.md @@ -0,0 +1,116 @@ +# Pages Static Hosting Usage + +You will learn: how to deploy pre-built static sites via local upload, Remote URL, or public GitHub Release assets; configure SPA Fallback and API reverse proxy; and safely check for updates, auto-publish, and roll back. + +--- + +## Core Mechanics and Page Structure + +OpenFlare Pages is inspired by Cloudflare Pages' Direct Upload and deployment history interaction, but currently handles **pre-built artifacts** rather than building from repository source. The project detail is organized as "current production deployment → deployment source → deployment history": source configuration can change, while created deployments stay immutable. + +```text +Local upload ─> unified validation / upload.Ingest ─> new candidate ─> admin explicit activation ─┐ +Remote URL ── Server restricted download ─────────────┐ │ +GitHub Release asset ─ Server resolves ───────────────┴─> create/load deployment ────────────────┤ + └─> source sync atomic activation ────────┘ + | + v + Agent pulls per-project latest + | + v + OpenResty local static serving +``` + +External URLs, GitHub metadata, and auto-checks are handled only by the Server. The Agent only pulls the currently active deployment package from the control plane; it does not receive external source credentials, nor does it run `git clone`, dependency installation, or build commands. + +## Step 1: Create a Project + +1. Log in to the admin panel, go to **「Pages」**, click **「Create Project」**. +2. Fill in the project name and a unique Slug. +3. Configure the content entry: + * **Entry file name**: default `index.html`. + * **Static asset root path (RootDir)**: fill in the relative path when artifacts are in a subdirectory like `dist/`; leave empty when artifacts are at the archive root. +4. Set SPA Fallback and API proxy as needed. RootDir and entry file are project-level configs applied uniformly to all sources. + +## Step 2: Choose a Deployment Source + +### 1. Manual Upload + +Without a persistent source configured, the project stays in manual mode. Click **「Upload Deployment Package」** to select a pre-built archive; a successful upload creates a candidate deployment, which you then explicitly activate from the deployment history. Re-uploading does not modify existing deployments. + +Supported formats: `zip`, `tar.gz` / `tgz`, `tar.xz` / `txz`, `tar.bz2` / `tbz2`, `tar`, and `7z`. + +### 2. Remote URL + +In the deployment source card select **Remote URL**, fill in the HTTP(S) address and choose a network policy: + +* **public**: default policy; rejects loopback, private network, link-local addresses, DNS rebinding, self-signed TLS, and redirects to non-public targets. +* **trusted_internal**: only for explicitly trusted intranet or self-signed services; requires a second risk confirmation before saving. + +After saving, the address is only displayed masked. You don't need to re-enter it when editing other configs; only submit a new URL when choosing to change the address. Remote sources only offer **「Sync and Publish」**: the Server downloads, validates, and atomically activates each time — no "check for updates", scheduled checks, or auto-updates. + +### 3. GitHub Release + +GitHub sources only support public `github.com` repositories. Fill in: + +* A repository address in `https://github.com/{owner}/{repo}` format; +* **Latest Release** or a **fixed Tag**; +* An exact, case-sensitive Release Asset filename, default `dist.zip`. + +Both options support manual **「Check for Updates」** and **「Sync and Publish」**. Differences: + +* **latest**: supports a check interval of 5–1440 minutes, default 1440 minutes (24 hours); auto-update is off by default. When enabled, the scanner asynchronously syncs and publishes only when a new revision is found. +* **tag**: only supports manual admin checks and sync; does not participate in the scheduled scanner. + +"Check for updates" only resolves the Release/asset and advances the version cursor without downloading the deployment package; "Sync and publish" downloads, validates, creates or reuses a deployment, and activates it. If the asset under the same Release is replaced, the source enters **「Needs Confirmation」** — you must confirm the exact revision shown before publishing, to avoid silent overwrites. + +GitHub Release sources only import pre-built artifacts; they do not build from repository source. + +### 4. Switch or Delete a Source + +You can switch between Manual, Remote, and GitHub Release. Modifying or deleting a source does not delete the current production deployment or historical deployments; switching back to manual mode lets you continue uploading and explicitly activating. + +## Deployment Package Security Limits + +Deployment packages must satisfy these constraints: + +* Archive size is controlled by the system config `pages_max_package_size_mb`, default 100 MiB, configurable 1–2048 MiB. +* Expanded single-file and total size limits are "package size limit × 4", with a floor of 100 MiB; at most 1,000 regular files. +* The control plane streams regular file bodies, checking declared size against actual bytes, and validates the project entry file. +* Absolute paths, `..` path traversal, symlinks, hard links, and special files in archives are all rejected. + +The Agent also verifies SHA-256, real response byte limits, and post-extraction file count and total size on download; failures do not switch the existing `current`. + +## Step 3: Configure Advanced Routing Rules + +### 1. SPA Fallback + +When using front-end routing like React Router or Vue Router, enable **「SPA Fallback」** and set the entry path (usually `/index.html`). When a visitor accesses a physical path that doesn't exist, OpenResty falls back to the entry file for the front-end router to handle. + +### 2. API Reverse Proxy + +Pages can forward a specified prefix to a backend API under the same domain: + +* **APIProxyPath**: match prefix, e.g. `/api`. +* **APIProxyPass**: backend address, e.g. `http://10.0.0.5:8080`. +* **APIProxyRewrite**: optional path rewrite rule. + +Requests matching the API prefix go through the reverse proxy; other requests continue to be served by the static site. + +## Step 4: Bind a Route and First Publish + +1. Create or edit a proxy rule. +2. Set the origin type to **Pages** and select the Pages **project**. +3. Preview the config, then publish and activate. + +The route binds to a stable project ID, not a specific deployment. The first publish gives the Agent the project anchor; afterwards, local uploads, source syncs, auto-updates, or manual rollbacks only change the project's active deployment — the Agent converges via the latest hash reconciliation without needing to republish the main config. + +## Operations, Status, and Rollback + +* The source card shows the last check/sync time, found vs. applied revision, next check time, and security errors. While a check or sync task runs, the page polls the task status; when latest is idle, it refreshes at low frequency only near the check time. +* A failed auto-update does not replace the old active deployment; a single source failure does not block the scanner from processing other projects. +* Activating another deployment in the history is a manual rollback. The system fences in-flight source tasks and disables that source's auto-update to avoid the next latest round overwriting your manual choice; re-activating the current version is a no-op. +* The Agent downloads to a temp file, verifies SHA-256, extracts safely, then atomically switches `current`. Any failure keeps the old content; with multi-project reconciliation, a single project failure does not affect others. + +> [!TIP] +> For the source state machine, auto scanner, upload compensation, immutable deployments, and Agent atomic switching, see [Pages Static Hosting Design](../design/pages-design.md). diff --git a/docs/en/guide/proxy-config.md b/docs/en/guide/proxy-config.md new file mode 100644 index 00000000..4a8094c4 --- /dev/null +++ b/docs/en/guide/proxy-config.md @@ -0,0 +1,117 @@ +# Create a Reverse Proxy Config + +You will learn: how to create and publish a reverse proxy website configuration from scratch, step by step, in OpenFlare. This guide walks you through certificate import and application, origin definition, route rule configuration, version release, and connectivity verification. + +--- + +## Recommended Workflow + +In the gateway control plane, follow these steps to add a new reverse proxy rule: + +```text + [ Step 1. Certificate Management ] ──► [ Step 2. Origin Definition (optional) ] ──► [ Step 3. Add Website Config ] + │ + [ Step 5. Verify Access ] ◄── [ Step 4. Publish & Activate Version ] ◄───────────────┘ +``` + +--- + +## Step 1: Prepare Certificates + +Before using HTTPS-secured traffic, you need to prepare the corresponding TLS certificate (supports manually importing an existing certificate, or automatically applying from a CA via DNS validation with managed renewal). + +To keep this guide concise, the certificate details (including creating a dedicated DNS API Token in Cloudflare) have been split into a dedicated guide. First go to **[TLS Certificates & Auto-Renewal](./certificates.md)** to prepare the certificate, then come back to continue. + +--- + +## Step 2: Prepare the Upstream Origin (optional) + +An Origin represents the backend real service address being proxied. Although you can fill in an IP directly when creating a website, it is recommended to register origins in the origin library first for reuse and maintenance: + +1. Go to **「Website Management」->「Origin Addresses」** in the left navigation, click **「Add Origin」**. +2. Fill in the origin name (e.g. `production-api`). +3. Enter a valid upstream address (e.g. `http://10.0.0.10:8080`) and save. + +--- + +## Step 3: Create the Website Config + +Once the certificate and origin are ready, create the core website proxy route: + +1. Go to **「Website Management」->「Domain List」**, click **「Add Zone」**: + * **Domain**: Enter the domain bound to this site. + * **Bind Certificate**: Select the certificate prepared or applied for in Step 1. +2. Configure the request route rule: go to the **「Rule Management」** page, click **「New Rule」** or edit an existing rule: + * **Rule Name**: Enter a unique simple identifier (e.g. `app-portal-route`). + * **Domain Match**: Enter the corresponding domain (wildcards or exact domains supported; must match the registered domain above). + * In the **「Reverse Proxy」** tab below, select the origin mode as「Direct Upstream」. + * **Origin Selection**: Select the origin created in Step 2 from the dropdown; or choose manual input and fill in `http://10.0.0.20:9000`. +3. Click save to create the config. + +--- + +## Step 4: Publish and Apply the Config + +Website configs added in the admin panel are only saved in the Server database — **they do not take effect immediately**. You must generate a config version snapshot and distribute it to the Agent edge nodes: + +1. Click the **「Preview and Publish」** button in the top-right corner of the control panel. +2. Review the config file diff, confirming the newly added `server` block and certificate binding rules are correct. +3. Click **「Confirm Publish」**. +4. **Agent application mechanism**: + * The Agent node on the data plane detects the active version Checksum change in its heartbeat, and automatically pulls the full OpenResty config files and certificate bundle locally. + * It automatically runs a local config validation (similar to `openresty -t`); after confirming no syntax errors, it performs a smooth reload. + * If reload or validation fails, the Agent safely blocks and rolls back to the previous stable version to keep the node highly available. + +--- + +## Step 5: Connectivity and Rollback Verification + +### 1. Verify Access +You can verify the new config takes effect as follows: +* **Browser access**: Open `https://your-domain.com` directly in a browser and check whether it proxies the backend successfully. +* **CLI verification** (recommended): probe with `curl`: + ```bash + curl -I https://your-domain.com + ``` +* **Bypass DNS validation**: if your domain is not yet resolvable, temporarily send a `Host` header request to the Agent node's physical IP: + ```bash + curl -I -H "Host: your-domain.com" https://AGENT_NODE_IP --insecure + ``` + +### 2. One-Click Second-Level Rollback +If the released config causes an online business issue: +1. Navigate to the **「Version Release」** menu on the left. +2. Find the previous stable version before the release in the history list. +3. Click **「Activate」**. +4. All online Agent nodes will automatically reload the historical config within seconds for second-level risk avoidance. + +--- + +## Edge Cache (optional) + +The **「Cache」** page in the site details can enable edge `proxy_cache` (requires **Performance Settings → Global OpenResty Cache** to be enabled at the same time). Behavior mirrors the Cloudflare default model; see [Edge Cache Strategy Design](../design/edge-cache-design.md). + +### Recommended Settings + +| Item | Suggestion | +| --- | --- | +| Strategy | **Standard static assets** (recommended default): only css/js/map/images/fonts, **not HTML/JSON** | +| Login Cookie | Is **not** separately skipped from caching; users with sessions can still hit static assets | +| Origin | Static assets: `Cache-Control: public, max-age=…`; dynamic/personalized must be `private` or `no-store` | +| Response Set-Cookie | Is not written to the edge cache | +| No origin cache headers | Uses default Edge TTL by status code (e.g. ~120 min for 200) | + +### Advanced Strategy「All Cacheable GET」 + +Similar to Cloudflare Cache Everything: the path is no longer limited by extension. If the origin does not declare `private`/`no-store` for HTML, **personalized pages may be cached and served across users**. Use only when origin cache headers are correct or content is globally consistent. + +### How It Takes Effect + +The cache switch and strategy are written into the config snapshot. After saving the site, you must **publish and activate the config version** for the Agent to apply it. Changing the UI only without publishing leaves nodes on the old rules. + +### Quick Self-Check + +1. Global cache on, site cache on, strategy「Standard static assets」. +2. Publish the config and confirm nodes applied successfully. +3. Request the same `/assets/app.js` (or a hashed immutable path) twice with a login cookie; the `cache_status` in access logs should be **HIT** on the second request. +4. If still「not cached」: check whether the strategy matches the path extension, whether it is a non-GET request, whether the origin returns `Set-Cookie` / `private`, and whether the node applied the new version. More in [Troubleshooting · Edge Cache](./troubleshooting.md#edge-cache-hit-rate-anomalies). diff --git a/docs/en/guide/quick-start.md b/docs/en/guide/quick-start.md new file mode 100644 index 00000000..48277ba6 --- /dev/null +++ b/docs/en/guide/quick-start.md @@ -0,0 +1,219 @@ +# Quick Start + +You will learn: How to start OpenFlare Server using Docker Compose, complete your first login, connect your first Agent, and verify if a configuration has been published to the node. + +The minimum running unit of OpenFlare consists of: + +| Component | Responsibility | +| --- | --- | +| Server | Admin UI, Admin API, Agent API, configuration rendering, version publishing, and state storage. | +| Agent | Runs on the proxy node, pulls configurations, writes files for OpenResty, executes validations, and triggers reloads. | +| OpenResty | Receives actual traffic and reverse proxies it to origin servers. | + +The Agent manages the runtime through the OpenResty binary. A local deployment requires the `openresty` executable to be already present on the node; a Docker deployment can directly run the Agent image containing built-in OpenResty. + +## Environment Requirements + +| Item | Requirement | +| --- | --- | +| Docker / Docker Compose | Used to start Server and PostgreSQL; also used to run the Agent if using the Docker Agent image | +| OpenResty | Required to have the `openresty` executable when installing the Agent locally, or specify its path in the installation script | +| Reachable Ports | The Server listens on port `3000` by default; the Agent node needs to be able to reach the Server address | +| Browser | Used to access the management console | + +* **Docker**: `20.10.0+` +* **Docker Compose**: `2.0.0+` + +## 1. Start the Server + +Create a `docker-compose.yml` file in an empty directory: + +```yaml +services: + postgres: + image: postgres:17-alpine + restart: unless-stopped + environment: + POSTGRES_DB: openflare + POSTGRES_USER: openflare + POSTGRES_PASSWORD: replace-with-strong-password + volumes: + - postgres-data:/var/lib/postgresql/data + healthcheck: + test: ["CMD-SHELL", "pg_isready -U openflare -d openflare"] + interval: 10s + timeout: 5s + retries: 5 + + openflare: + image: ghcr.io/rain-kl/openflare:latest + restart: unless-stopped + depends_on: + postgres: + condition: service_healthy + ports: + - "3000:3000" + environment: + SESSION_SECRET: replace-with-a-long-random-string + DSN: postgres://openflare:replace-with-strong-password@postgres:5432/openflare?sslmode=disable + GIN_MODE: release + LOG_LEVEL: info + volumes: + - openflare-data:/data + +volumes: + postgres-data: + openflare-data: +``` + +Start the services: + +```bash +docker compose up -d +``` + +Verify that the containers are running: + +```bash +docker compose ps +docker compose logs -f openflare +``` + +Once you see `server listening` in the logs and the `openflare` container status is running, access: + +```text +http://localhost:3000 +``` + +Default credentials: + +| Username | Password | +| --- | --- | +| `root` | `123456` | + +Please change the default password immediately after your first login. + +## 2. Prepare Agent Token + +The Agent can be connected using one of two types of credentials: + +| Credential | Applicable Scenario | +| --- | --- | +| `discovery_token` | Automatically registers a node for the first time, which the Server exchanges for a node-specific Token | +| `agent_token` | Node has already been created/allocated in the management console, directly uses this node-specific Token | + +After preparing one of these credentials in the management console, proceed to the next step. + +* **`discovery_token`** path: "System Settings" -> "Auto Registration" +* **`agent_token`** path: "Node Management" -> "Add Node" + +## 3. Install/Run the Agent + +The recommended Agent deployment method is using Docker (which runs the Agent image with built-in OpenResty); deploying the Agent locally on the host using the installation script is also supported. + +### Option A: Run Agent in Docker (Recommended) + +Run the Agent image directly on the proxy node: + +```bash +docker pull ghcr.io/rain-kl/openflare-agent:latest +docker rm -f openflare-agent 2>/dev/null || true +docker run -d --name openflare-agent --restart unless-stopped \ + -p 80:80 -p 443:443 \ + -v openflare-agent-data:/data \ + -e OPENFLARE_SERVER_URL=http://your-server:3000 \ + -e OPENFLARE_AGENT_TOKEN=YOUR_AGENT_TOKEN \ + ghcr.io/rain-kl/openflare-agent:latest +``` + +### Option B: Execute Installation Script (Local Host Deployment) + +Execute the installation script on the proxy node. + +Using the `discovery_token`: + +```bash +curl -fsSL https://raw.githubusercontent.com/Rain-kl/OpenFlare/main/scripts/install-agent.sh | bash -s -- \ + --server-url http://your-server:3000 \ + --discovery-token YOUR_DISCOVERY_TOKEN +``` + +Using the node-specific `agent_token`: + +```bash +curl -fsSL https://raw.githubusercontent.com/Rain-kl/OpenFlare/main/scripts/install-agent.sh | bash -s -- \ + --server-url http://your-server:3000 \ + --agent-token YOUR_AGENT_TOKEN +``` + +The script defaults to: + +| Item | Default Value | +| --- | --- | +| Install Directory | `/opt/openflare-agent` | +| Config File | `/opt/openflare-agent/agent.json` | +| systemd Service | `openflare-agent.service` | +| OpenResty Path | Automatically detects `openresty` if unspecified | + +Verify the Agent service status: + +```bash +systemctl status openflare-agent +journalctl -u openflare-agent -f +``` + +If systemd is not available on the OS, the script outputs manual startup commands instead. + +## 4. Publish Your First Configuration + +Perform the following operations in the management console: + +1. Add a website configuration, filling in the website name, domain, and origin address. +2. Verify that the website configuration is enabled. +3. Check the preview or change summary before publishing. +4. Publish and activate the new version. +5. Wait for the Agent to detect and apply the version in the next heartbeat. + +The version number format is `YYYYMMDD-NNN`. Historic versions are immutable; rollbacks are accomplished by re-activating an older version. + +## 5. Verify Success + +Confirm in the management console: + +| Position | Expected Result | +| --- | --- | +| Node List | Agent node status is online | +| Node Details | Current version matches active version | +| Apply Logs | Most recent application succeeded | +| Version Page | The new version is currently active | + +Confirm on the Agent node: + +```bash +journalctl -u openflare-agent -n 100 --no-pager +``` + +## Common Failures + +| Symptom | Troubleshooting Direction | +| --- | --- | +| Management console fails to load in browser | Verify that the Server is running in `docker compose ps` and port `3000` is not bound by other processes | +| Data fails to save after logging in | Check the health of the PostgreSQL container, and verify the username, password, and database name in `DSN` | +| Agent fails to register | Verify that the Agent node can reach `--server-url`, and verify if the Token is typed correctly or expired | +| Agent is online but configuration is not applied | Verify that the website configuration is enabled and a version has been published and activated | +| OpenResty application fails | Review node application logs and `journalctl -u openflare-agent`, checking domains, certificates, upstreams, and port conflicts | + +For more troubleshooting details, see [Troubleshooting](./troubleshooting.md). + +--- + +## Advanced Deployment Guides + +Once you complete the quick start and familiarize yourself with the basic operations of OpenFlare, you can read the following advanced deployment documents to put components into production: + +* **Server Production Deployment**: Read [Launch Server](../deployment/server.md) to learn how to build the frontend from source, configure system environment variables, and run with Docker Compose. +* **Agent Production Integration**: Read [Deploy Agent](../deployment/agent.md) to learn about systemd-based service management, detailed local configuration parameters, and troubleshooting. +* **Tunnel Relay Deployment**: Read [Deploy Relay](../deployment/relay.md) to learn how to configure public relay nodes (frps) for penetration tunnels. +* **Tunnel Client Deployment**: Read [Deploy OpenFlared](../deployment/openflared.md) to learn how to run the penetration daemon client (frpc) on the intranet server side. +* **Production Deployment Topology**: Read [Deployment Guide](../deployment/deployment.md) to learn about high-availability production topologies and overall network planning. +* **System Upgrades & Maintenance**: Read [Upgrade & Maintenance](../deployment/upgrade.md) to learn how to upgrade the Server and individual node Agents smoothly. diff --git a/docs/en/guide/sso.md b/docs/en/guide/sso.md new file mode 100644 index 00000000..d8d59a27 --- /dev/null +++ b/docs/en/guide/sso.md @@ -0,0 +1,106 @@ +# SSO Login Configuration + +You will learn: How to configure GitHub OAuth or standard OIDC login portals for OpenFlare, how to fill in callback URLs, and how third-party accounts bind to existing local users. + +OpenFlare supports third-party logins configured via Authentication Sources. Currently, GitHub OAuth and standard OIDC Providers (e.g., Logto, authentik, Keycloak, Casdoor) are supported. + +Once an Authentication Source is configured and enabled, it displays in the third-party login section of the login page. Users can log in using their third-party accounts or bind their third-party accounts to their current local account while logged in. + +## Prerequisites + +Before starting, prepare the following: + +| Item | Description | +| --- | --- | +| OpenFlare URL | The actual URL accessed by user browsers, e.g., `https://openflare.example.com` | +| Auth Source Name | Unique internal identifier in OpenFlare, e.g., `github`, `company-oidc` | +| Client ID | Provided after creating an application in the third-party platform | +| Client Secret | Provided after creating an application in the third-party platform | +| OIDC Discovery URL | Required for OIDC only, e.g., `https://idp.example.com/.well-known/openid-configuration` | + +**Verify that "System Settings -> General Settings -> Server Address" accurately matches your domain name.** + +The Auth Source name can only contain letters, numbers, hyphens, or underscores, and must start with a letter or number. The Auth Source name will appear in the callback URL; if you modify the name after saving, you must simultaneously modify the callback URL on the third-party platform. + +## Callback URL + +The Redirect URI / Callback URL in third-party platforms is formatted as: + +```text +/oauth/ +``` + +Example: + +```text +https://openflare.example.com/oauth/github +https://openflare.example.com/oauth/company-oidc +``` + +When creating or editing an authentication source in the management console, the form automatically generates the callback URL based on your current browser URL and the Auth Source name you entered. + +## Configure GitHub Login + +1. Create an OAuth App in GitHub. +2. Fill `Homepage URL` with your OpenFlare URL. +3. Fill `Authorization callback URL` with the callback URL generated in OpenFlare, e.g., `https://openflare.example.com/oauth/github`. +4. Copy the Client ID and Client Secret provided by GitHub. +5. Log into the OpenFlare management console, go to "Settings -> System Settings -> Configure Authentication Sources". +6. Add an authentication source, choosing `GitHub` as the type. +7. Fill in the Auth Source name, display name, Client ID, and Client Secret. +8. The Scope defaults to `user:email`, which usually requires no modification. +9. Save and enable the authentication source. + +Once enabled, the corresponding GitHub login button will display on the login page. + +## Configure OIDC Login + +1. Create an application or client in your OIDC Provider. +2. Select Web / Confidential Client as the application type. +3. Fill `Redirect URI / Callback URL` with the callback URL generated in OpenFlare, e.g., `https://openflare.example.com/oauth/company-oidc`. +4. Copy the Client ID and Client Secret. +5. Retrieve the Provider's Discovery URL, which usually ends with `/.well-known/openid-configuration`. +6. Log into the OpenFlare management console, go to "Settings -> System Settings -> Configure Authentication Sources". +7. Add an authentication source, choosing `OIDC` as the type. +8. Fill in the Auth Source name, display name, Client ID, Client Secret, and OIDC Discovery URL. +9. Scope defaults to `openid profile email`. If the Provider restricts scopes, adjust to values permitted by the Provider. +10. Save and enable the authentication source. + +Once enabled, the corresponding OIDC login button will display on the login page. + +## Login & Binding Behaviors + +Once a third-party account returns to OpenFlare, it is processed according to the following rules: + +| Scenario | Behavior | +| --- | --- | +| Third-party account is already bound to a local user | Logs in directly | +| User is already logged in and initiates third-party authorization | Binds to the current local user | +| Third-party account is unbound, and registration is enabled | Automatically creates a standard user and binds | +| Third-party account is unbound, and registration is disabled | Prompts to enter an existing local username and password to complete the binding | + +If you want only existing users to use SSO, you can disable user registration. Unbound third-party accounts will then trigger the binding flow. + +## Modify Authentication Source + +When editing an authentication source, leaving the Client Secret field blank retains the existing secret; entering a new value will overwrite the saved secret. + +If you modify the Auth Source name, the callback URL changes accordingly. You must modify the Redirect URI / Callback URL on the third-party platform; otherwise, the third-party platform will deny the callback or return an error. + +## Common Problems + +### Returns `invalid_scope` + +This indicates that the third-party platform does not permit the configured Scope. OIDC defaults to `openid profile email`, and GitHub defaults to `user:email`. Adjust the Scope in the authentication source edit page or configure the third-party platform to permit the scope. + +### Callback Address Mismatch + +Verify if the Redirect URI / Callback URL configured in the third-party platform matches the prompt in the OpenFlare form exactly. The protocol, domain, port, and path must match. + +### Third-party Login Button Not Showing on Login Page + +Verify if the authentication source is enabled and confirm that the Client ID and Client Secret are saved. OpenFlare validates these fields before enabling the source. + +### Client Secret Saved but Not Displayed in Clear Text + +This is expected behavior. OpenFlare does not echo the Client Secret back via API, displaying only whether the secret is configured. diff --git a/docs/en/guide/troubleshooting.md b/docs/en/guide/troubleshooting.md new file mode 100644 index 00000000..8975e8be --- /dev/null +++ b/docs/en/guide/troubleshooting.md @@ -0,0 +1,249 @@ +# Troubleshooting + +You will learn: How to troubleshoot OpenFlare Server, database, login, Agent, OpenResty, configuration publishing, and frontend build issues by symptoms. + +During troubleshooting, first identify which layer the issue occurs in: browser, Server, database, Agent, OpenResty, origin server, or DNS. OpenFlare configurations are not written directly to nodes online; only after the active version changes will the Agent detect and apply it in heartbeats. + +## Quick Diagnostic + +| Symptom | Where to check first | +| --- | --- | +| Admin panel fails to open | Server container or process logs, port listening | +| Login anomalies | Default credentials, Session Secret, browser request payloads, Server logs | +| Data fails to save | Database connection, SQLite file permissions, PostgreSQL health | +| Agent offline | Agent logs, Token, Server URL, network connectivity | +| Node not updated after publishing | Active version, node heartbeat, application logs | +| OpenResty application failed | Application logs, Agent logs, certificates, upstream addresses, port conflicts | +| Observability analytics has no data | OpenResty container status, observability port, Agent retry logs | + +## Server Fails to Start + +1. View logs: + +```bash +docker compose logs -n 200 openflare +``` + +For source-code execution, inspect terminal outputs. + +2. Check port conflicts: + +```bash +lsof -i :3000 +``` + +3. If using PostgreSQL, verify that the database is healthy: + +```bash +docker compose ps postgres +docker compose logs -n 100 postgres +``` + +4. If using SQLite, verify that the database directory is writable: + +```bash +ls -ld "$(dirname /path/to/openflare.db)" +``` + +Common causes: + +| Log or Symptom | Action | +| --- | --- | +| Database connection failed | Check `DSN` username, password, host, port, dbname, and `sslmode` | +| SQLite fails to create files | Check if the parent directory of `SQLITE_PATH` exists and is writable | +| Port is already in use | Change `PORT` or `--port`, or stop the process binding to the port | + +## Admin Console Fails to Load or Shows Blank Page + +1. Verify that the Server is listening: + +```bash +curl -I http://127.0.0.1:3000 +``` + +2. If running from source, verify that the frontend static assets have been built: + +```bash +cd openflare-server/web +pnpm build +``` + +3. Verify if the browser URL matches your reverse proxy domain. + +4. If accessing via the frontend dev server, verify the backend proxy configuration: + +```bash +cd openflare-server/web +NEXT_DEV_BACKEND_URL=http://127.0.0.1:3000 pnpm dev +``` + +## Default Credentials Fail to Log In + +The default credentials are `root` / `123456`. If you have modified the password after your first login, use your new password. + +Troubleshooting Steps: + +1. Confirm that you are connecting to the expected database, avoiding `SQLITE_PATH` or `DSN` pointing to a different environment. +2. Check the Server log to see if it is running on `sqlite` or `postgres`. +3. If deployed in multi-replicas or behind a reverse proxy, verify that `SESSION_SECRET` is static and uniform across all instances. +4. Clear browser Cookies and try logging in again. + +### Emergency Reset of Admin Password + +If you forget the password for the `root` account, you can reset it back to `123456` by directly updating the password hash in the database (please change it immediately after logging in): + +#### 1. If using SQLite Database +Stop the Server and open the database file using the `sqlite3` client: +```bash +sqlite3 /path/to/openflare.db +``` +Execute the following SQL statement: +```sql +UPDATE users SET password_hash = '$2a$10$wN9aE3zTz83rO7R1uKlhuehJtA3c604pX4Z12B/9.5c0X337t1L4m' WHERE username = 'root'; +``` +Type `.exit` to exit and restart the Server. + +#### 2. If using PostgreSQL Database +Connect to your PostgreSQL instance using a database tool (e.g., `psql`, `pgAdmin`, or `DBeaver`), select the corresponding `openflare` database, and execute the following SQL: +```sql +UPDATE users SET password_hash = '$2a$10$wN9aE3zTz83rO7R1uKlhuehJtA3c604pX4Z12B/9.5c0X337t1L4m' WHERE username = 'root'; +``` +Once executed successfully, you can log in using the default password `123456`. + +## Agent Fails to Register or Stays Offline + +Execute on the Agent node: + +```bash +curl -I http://your-server:3000 +``` + +Inspect Agent logs: + +```bash +journalctl -u openflare-agent -n 200 --no-pager +``` + +Verify configuration parameters: + +```bash +sed -n '1,160p' /opt/openflare-agent/agent.json +``` + +Key Settings: + +| Configuration | Description | +| --- | --- | +| `server_url` | Must be the Server address reachable by the Agent node | +| `agent_token` / `discovery_token` | At least one must be provided | +| `heartbeat_interval` | Supports integer milliseconds or Go duration strings | +| `request_timeout` | Can be increased for slower network links | + +If the log warns that the Token is invalid, retrieve a new Token in the management console, update `agent.json`, and restart the Agent: + +```bash +systemctl restart openflare-agent +``` + +## Node Fails to Apply New Version after Publishing + +Verify in sequence: + +1. Confirm that the target version is activated on the Versions page. +2. Verify if the node is online and if its last heartbeat time has updated. +3. Check the Application Logs for successful, warned, or failed logs for the target version. +4. Verify if the website configuration is enabled; disabled websites do not participate in rendering. +5. Inspect Agent logs for pulls, validations, reloads, or rollback events. + +Inspect Agent logs: + +```bash +journalctl -u openflare-agent -f +``` + +Note: If a target `version + checksum` fails to apply and triggers a rollback, the Agent blocks repeated synchronization of that failing target in its local state. You must fix the configuration issues and republish to generate a new checksum, or activate an older version to trigger a rollback. + +If this is the Agent's first time applying configurations and no historic `nginx.conf` exists locally to roll back to, the failed version remains blocked but the Agent will attempt to enter the safe fallback runtime. At this point, the application logs and Agent logs will contain `fallback runtime started`. OpenResty will only listen to port `80`, returning a `503` with the body `OpenFlare: No Valid Configuration`, while retaining the local `/openflare/stub_status` health probe. After correcting the configurations and republishing, the Agent overrides the fallback config and restores normal reverse proxies. + +## OpenResty Application Fails + +Common Causes: + +| Cause | Diagnostic | +| --- | --- | +| Domain or server block conflict | Verify if the same domain is used by multiple website configurations | +| Invalid upstream address | Confirm that all upstreams are valid `http://` or `https://` URLs | +| Mismatched multi-upstream format | Multi-upstreams must be pure `scheme://host[:port]` | +| Missing cert or invalid paths | Verify if domains are bound to certs and check if the Agent cert directory is writable | +| Port already in use | Verify ports `80` and `443` on the host | + +OpenResty Configuration Validation: + +```bash +openresty -t -c /path/to/openflare/data/etc/nginx/nginx.conf +``` + +OpenResty Runtime Status: + +```bash +ps aux | grep openresty +``` + +The Agent determines OpenResty survival periodically using the local endpoint `http://127.0.0.1:/openflare/stub_status`, completely bypassing repeated `openresty -t` calls. If a node is marked as unhealthy, confirm if this local observability port is listening. If failures only occur when applying configurations (e.g., `host not found in upstream`), the failure lies in config validation or reload, not the periodic health checks. + +Actual binary paths and main configuration paths are governed by `openresty_path` and `main_config_path` in `agent.json`. + +## HTTPS Fails to Work + +1. Verify that the certificate has been uploaded or hosted. +2. Verify that the website configuration binds the certificate to the domain. +3. Confirm that the configuration version has been published and activated. +4. Check if the Application Logs indicate a success. +5. Check the certificate chain and status code using `curl`: + +```bash +curl -Iv https://your-domain +``` + +Domains without a bound certificate will not be added to the HTTPS configuration automatically; this is expected behavior. + +## Traffic Analytics Has No Data + +1. Confirm that the node has successfully applied configurations carrying observability Lua scripts. +2. Verify that OpenResty is running. +3. Check Agent logs for observability extraction or upload errors. +4. Check if `openresty_observability_port` (default is `18081`) is bound by other processes. +5. Verify if the Server database has purged data inside the time window. + +## Frontend Build Fails + +Execute: + +```bash +cd openflare-server/web +corepack enable +pnpm install +pnpm lint +pnpm typecheck +pnpm test +pnpm build +``` + +Common causes: + +| Symptom | Action | +| --- | --- | +| pnpm version mismatch | Reinstall packages after executing `corepack enable` | +| TypeScript errors | Locate detailed file bugs by running `pnpm typecheck` | +| API type mismatch | Check responses structures in `lib/api/` and `types/` | +| E2E test failures | Confirm that both the Server and frontend dev server are running | + +## Documentation Build Fails + +```bash +cd docs +pnpm install +pnpm build +``` + +If it fails on broken links, check if new pages are added to the `docs/config.ts` sidebar, or if relative markdown links point to existing markdown files. diff --git a/docs/en/guide/tunnel-usage.md b/docs/en/guide/tunnel-usage.md new file mode 100644 index 00000000..9a643e37 --- /dev/null +++ b/docs/en/guide/tunnel-usage.md @@ -0,0 +1,174 @@ +# Tunnel & Intranet Penetration + +You will learn: The design principles of OpenFlare intranet penetration tunnels, core concepts (Relay nodes and Tunnel clients), and how to safely and stably publish your intranet development environment or private cloud services to a public domain name from scratch. + +In many practical development and operations scenarios, our origin servers are deployed in local LANs, local development machines, or heavily guarded private VPCs, having no public IP address and no port mapping (NAT) configured on border firewalls or routers. + +OpenFlare provides an end-to-end solution **based on reverse relay penetration tunnels**. You only need to initiate a secure outbound connection from your intranet environment to the public relay node, without configuring any inbound ports, to smoothly route public web traffic into your intranet origin. At the same time, you benefit from automatic TLS certificate hosting and WAF security protection provided by the gateway. + +--- + +## Core Concepts + +Before using the intranet penetration features, you need to familiarize yourself with the following components and core concepts: + +| Concept | Description | Component / Operation | +| --- | --- | --- | +| **Relay Node (Relay)** | Traffic relay services deployed at the public edge, responsible for listening to intranet client persistent connections, acting as the transit bridge between the gateway Agent (OpenResty) and internal traffic. | Node of type `tunnel_relay` running the `openflare-relay` daemon | +| **Penetration Tunnel (Tunnel)** | Logical penetration client instances having a globally unique ID and secure authentication token, used to identify a specific intranet environment. | Globally unique ID generated by Server `tunnel_id` (format: `tun-<32hex>`) | +| **Tunnel Client (Client)** | A lightweight controller running in the intranet environment, automatically managing the underlying frpc tunnel subprocesses according to the configuration dispatched by the Server. | The `openflared` container or independent binary process deployed in the intranet | +| **Tunnel Upstream (Tunnel Upstream)** | A special upstream type in the website configuration. When this type is selected, the gateway forwards public traffic to the Vhost port of the local relay node, eventually reaching the intranet origin. | Upstream of type `tunnel` configured in the website details | + +--- + +## Recommended Operation Sequence + +To publish an intranet service to the public internet, we recommend doing so in the following order: + +1. Register and deploy at least one public **Relay Node (Relay)** and keep it online. +2. Create a **Penetration Tunnel (Tunnel)** in the management console and copy its dedicated Token. +3. Deploy and start the **Tunnel Client (OpenFlared)** on your intranet server. +4. Confirm that the status of the tunnel in the management console shows as "Online". +5. Add a website configuration, selecting **Intranet Penetration** as the upstream type, binding it to the corresponding tunnel, and entering the intranet port (e.g., `127.0.0.1:8080`). +6. Publish and activate the new version. +7. Access via the public domain to verify that the intranet penetration link is established. + +--- + +## Detailed Configuration Steps + +### Step 1: Prepare the Relay Node (Relay) + +Intranet traffic is routed through public relay nodes. Before starting, ensure you have a public relay server available. + +1. Log into the management console and go to **"Node Management"**. +2. Add a new node, selecting **Relay Node (tunnel_relay)** as the **Node Type**. +3. Save and copy the node-specific `agent_token`. +4. Start the `openflare-relay` process on your public server. You can run it quickly using Docker: + + ```bash + docker run -d --name openflare-relay --restart unless-stopped \ + -p 7000:7000 \ + -e OPENFLARE_SERVER_URL=http://:3000 \ + -e OPENFLARE_AGENT_TOKEN= \ + -v openflare-relay-data:/var/lib/openflare-relay \ + ghcr.io/rain-kl/openflare-relay:latest + ``` + + > [!IMPORTANT] + > Make sure to allow port `7000` (the control port for frpc client connections) in your cloud provider's security group. If your Server and Relay are deployed on the same machine, `OPENFLARE_SERVER_URL` should point to the Server's public or internal IP. + +### Step 2: Create a Penetration Tunnel in the Management Console + +1. Navigate to the **"Intranet Penetration"** section in the side navigation bar. +2. Click the **"Create Tunnel"** button and enter: + * **Tunnel Name**: Describes the intranet environment, e.g., `home-lab` or `office-dev`. + * **Description**: Optional, describes the purpose of this tunnel. +3. Click save, and the system will automatically generate a globally unique ID and a dedicated `tunnel_token` (e.g., `tun-xxxx...`). +4. Copy the **Client Deployment Command** generated in the popup window, which will be used in the next step. + +### Step 3: Deploy the Intranet Client (OpenFlared) + +Return to your intranet server and execute the copied deployment command to run the client. + +#### Option A: Deploy with Docker (Highly Recommended) + +The official `openflared` image embeds the master daemon and `frpc` runtime, working out-of-the-box with no extra dependencies: + +```bash +docker run -d --name openflared --restart unless-stopped \ + -e OPENFLARE_SERVER_URL=http://:3000 \ + -e OPENFLARE_TUNNEL_TOKEN= \ + -v openflared-data:/app/data \ + ghcr.io/rain-kl/openflared:latest +``` + +#### Option B: Host Binary Manual Execution + +If you cannot use Docker, you can download or compile the `flared` binary: + +1. Create a `flared.json` configuration file in the same directory as the executable on your intranet machine: + ```json + { + "server_url": "http://:3000", + "tunnel_token": "", + "frpc_path": "/usr/local/bin/frpc", + "data_dir": "./data" + } + ``` +2. Execute the startup command: + ```bash + ./flared -config ./flared.json + ``` + +#### Verify Online Status + +Once started successfully, the intranet client will send heartbeats through outbound networks to synchronize configurations. At this point: +1. Refresh the **"Intranet Penetration"** list in the management console; the tunnel status indicator should turn green and show **"Online"**. +2. Click tunnel details to view which public Relays the intranet client is currently connected to. + +### Step 4: Create a Website and Bind the Tunnel Upstream + +Now you can configure public reverse proxy and domain routing for your intranet service. + +1. Go to the **"Website Configuration"** page and click **"Create Website"**. +2. Enter the **Domain Name** required to access the service publicly, e.g., `nas.example.com`. +3. Critical Configuration: In the **"Upstream Configuration"** section, switch the **Upstream Type** from "Direct" to **"Intranet Penetration"**. +4. In the dropdown list, select your newly deployed **Intranet Tunnel** (e.g., `home-lab`). +5. Enter the **Intranet Target Address** (the local address and port reachable by the intranet client, e.g., `127.0.0.1:8080`) and select the **Intranet Protocol** (usually `http`). +6. Configure other standard website settings (such as TLS certificates) and click save. + +### Step 5: Publish & Activate + +To allow the gateway's OpenResty instance to match and route domain traffic correctly, we need to publish a new configuration version. + +1. Click **"Preview Config"** in the top right corner of the navigation bar to verify the generated configurations. +2. In the popup window, click **"Publish & Activate"**. +3. Now, the public edge Agent pulls the latest routing, forwarding requests for `nas.example.com` to the loopback virtual host port of `openflare-relay (frps)`. +4. The intranet client `openflared (frpc)` receives the relayed packets, securely hands them over to the local `127.0.0.1:8080` service, and returns responses back through the tunnel. +5. Access `nas.example.com` in your browser to confirm that the intranet service displays successfully! + +--- + +## Advanced Application Scenarios + +### 1. Single-Tunnel Multi-Service Multiplexing (Multi-Port Mapping) + +You do not need to deploy an `openflared` container for every single internal service. + +If you want to map multiple different services in the same intranet environment (e.g., `127.0.0.1:80` for a blog, `127.0.0.1:8080` for an API, and `192.168.1.120:9000` for a local network drive): +1. Keep this single `openflared` client online. +2. Create three independent website configurations in the management console (binding their respective public domains). +3. Set the **Upstream Type** to **the same intranet tunnel** for all three website configurations. +4. Fill in their respective "Intranet Target Addresses" (e.g., `127.0.0.1:80`, `127.0.0.1:8080`, and `192.168.1.120:9000`). +5. Publish and activate the new version to achieve single-tunnel multi-service multiplexing. + +### 2. Seamless Integration with Gateway Security Features + +Since all public traffic enters the public Agent node first, completing the HTTPS/TLS handshake and WAF filtering before traveling through the secure tunnel: + +Your intranet services **naturally benefit from the following advanced features without any code changes**: +* **One-Click HTTPS**: Select or issue SSL certificates directly in the management console, encrypting transmission end-to-end. +* **Global/Custom WAF Protections**: Enables SQL injection blocking, XSS prevention, and regional IP filtering. +* **Human-Machine Challenge (PoW CC)**: Instantly blocks brute-force CC API attacks targeting your intranet services. + +--- + +## Common Troubleshooting + +### 1. Tunnel Shows as "Offline" in the Management Console + +* **Check the Token**: Check if the `tunnel_token` configured in `flared` logs or environment variables matches the one generated in the management console. +* **Check Outbound Connectivity**: The intranet server must be able to make outbound requests to the Server address. Ensure the control plane firewall is not blocking HTTP requests from the client. +* **Relay Firewall Port Closed**: Check if port `7000` (or your custom bindPort) on the public Relay node has been allowed in the public security groups. + +### 2. Accessing the Public Domain Returns 502 Bad Gateway / 504 Gateway Timeout + +* **Intranet Service Not Running**: Verify that the service corresponding to the intranet target address is running and listening on the intranet server. +* **Target Address Unreachable**: If the intranet address is set to `127.0.0.1:8080`, ensure the service is running on the exact same host as `openflared`; if set to a LAN IP `192.168.x.x`, test connectivity to that IP inside the `openflared` container. +* **Check Client Application Logs**: View the "Apply Logs" in the management console or inspect local `flared` logs for any `LastError`. When frpc fails to connect to the intranet port, it reports the failure details to the Server. + +### 3. Multiple Relays Network Instability or Retry Failures + +* When the control plane associates multiple Relay nodes, `openflared` spawns independent frpc daemon processes for each Relay and pulls topology states periodically at `sync_interval` (default 30s) configured in `flared.json`. +* If a Relay drops frequently due to network jitter, the system triggers the backoff retry mechanism automatically. You can see `frpc process missing, starting` logs on the host, which is a normal process self-healing action and will recover within 5-10 seconds after network recovery. diff --git a/docs/en/guide/uptime-kuma.md b/docs/en/guide/uptime-kuma.md new file mode 100644 index 00000000..40041647 --- /dev/null +++ b/docs/en/guide/uptime-kuma.md @@ -0,0 +1,49 @@ +# Uptime Kuma Monitoring Sync + +You will learn: how to enable and configure the Uptime Kuma auto-sync integration, control the sync scope and heartbeat probe parameters for monitored sites, and the underlying principles of differential synchronization between OpenFlare and Uptime Kuma. + +--- + +## Feature Overview + +In edge multi-node operations, knowing the availability of each proxied site in time is critical. To avoid manually re-entering site information into a monitoring system, OpenFlare provides deep integration with the open-source monitoring service **Uptime Kuma**. + +Once enabled, OpenFlare starts a background sync scheduler that automatically syncs the proxy sites configured in the admin panel as HTTP monitor tasks in Uptime Kuma. It supports scope filtering, differential attribute updates, and automatic cleanup of decommissioned sites. + +--- + +## Step 1: Configure the Integration in System Settings + +1. Log in to the admin panel, go to **「System Settings」** in the left navigation, select the **「OpenFlare」** tab, and configure the **「Uptime Kuma Integration」** section. +2. Configure the following core connection parameters: + * **Enabled**: Turn on the integration switch. + * **Instance URL**: Your Uptime Kuma service address, e.g. `http://192.168.1.100:3001` or `https://kuma.example.com` (the protocol prefix `http://` or `https://` is required). + * **Username** and **Password**: Credentials for a Uptime Kuma account with admin privileges, used for API authentication. + +--- + +## Step 2: Control Monitor Scope and Heartbeat Parameters + +In the integration panel you can finely control the monitor scope and probe behavior: + +### 1. Monitor Scope +* **All sites**: Default option. OpenFlare automatically syncs all **enabled** proxy route sites. When a new site is created and enabled, or an old site is disabled, the monitor list is updated automatically. +* **Selected sites**: Only monitor specified sites. After selecting this mode, click the **「Select Monitored Sites」** dialog. Inside the dialog you can filter sites by search and check the ones you want. Sites that are unchecked or not checked will not be synced (and will be automatically cleaned up if they already exist). + +### 2. Probe Frequency and Heartbeat Settings +You can specify uniform probe parameters for auto-generated monitors: +* **Sync Interval**: Frequency (minutes) of automatic differential sync, default `5` minutes. The control plane compares state with Uptime Kuma every 5 minutes. +* **Heartbeat Interval**: Frequency (seconds) at which Uptime Kuma probes sites, default `60` seconds. +* **Retry**: Maximum number of retries before a failed probe is judged Down, default `0`. +* **Retry Interval**: Seconds to wait between retries, default `60` seconds. +* **Request Timeout**: Seconds after which a probe request is judged timed out, default `48` seconds. + +--- + +## Sync and Cleanup Mechanism + +* **Dedicated tag isolation**: All auto-created monitors are bound with the `OpenFlare`-specific tag (purple-blue). The sync and cleanup routines only operate on monitors with this tag, and will not interfere with or damage other monitors you created manually in Uptime Kuma. +* **Differential incremental sync**: The sync routine periodically compares monitor metadata. When a domain or heartbeat configuration change is detected, only a differential update is performed to avoid interrupting historical statistics; when a site is disabled or moved out of scope, it is automatically taken offline and cleaned up. + +> [!TIP] +> For details on the Socket.IO control flow, anti-pollution tag model, and differential comparison algorithm of Uptime Kuma monitoring sync, see [Uptime Kuma Sync Design](../design/kuma-design.md). diff --git a/docs/en/guide/waf-ip-group-expr.md b/docs/en/guide/waf-ip-group-expr.md new file mode 100644 index 00000000..ff8d24cf --- /dev/null +++ b/docs/en/guide/waf-ip-group-expr.md @@ -0,0 +1,161 @@ +# WAF Auto IP Group Expressions + +Automatic IP groups are used to aggregate metrics from request logs on a per-client-IP basis, using Expr expressions to determine if an IP should be added to the group. Automatic IP groups can be referenced by IP blacklists or whitelists in WAF rule groups; during publication, the Server only writes the referenced IP group ID to `waf_config.json`, while IP group members are synchronized independently by the Agent into the local runtime files. + +## Configuration Structure + +The configuration of an automatic IP group is a JSON object: + +```json +{ + "lookback_minutes": 60, + "rules": [ + { + "name": "Single IP High-Frequency 404 Scanning", + "expr": "request_count > 100 && status_404_ratio >= 0.8" + } + ] +} +``` + +Field Descriptions: + +| Field | Type | Role | +| --- | --- | --- | +| `lookback_minutes` | number | How many minutes of request logs to look back during execution. Defaults to 60 minutes if blank, minimum 5 minutes, maximum 43200 minutes. | +| `rules` | array | List of automatic rules. If any rule matches, the IP is added to the automatic IP group list. | +| `rules[].name` | string | Rule name, used only for UI display and error messages. | +| `rules[].expr` | string | Expr expression, must return a boolean value. | + +## Evaluation Mechanics + +Automatic rules do not evaluate logs request-by-request, but instead aggregate them by client IP first: + +1. The Server reads request logs from the past `lookback_minutes` minutes. +2. Groups them by normalized IP (`remote_addr`). +3. Computes metrics like request count, 404 count, and direct IP host count for each IP. +4. Evaluates `rules[].expr` for each IP. +5. If an IP matches any rule, it is written to the automatic IP group's IP member list. + +Whether a request is "accessing via IP directly" is determined by the `Host` field in the request logs. If the Host header is an IPv4 or IPv6 literal (e.g., `203.0.113.10`, `[2001:db8::10]`, `203.0.113.10:443`), it is counted in `ip_host_count`. + +## Available Metrics + +The following metrics are directly available in Expr expressions: + +| Keyword | Type | Role | +| --- | --- | --- | +| `ip` | string | The client IP currently being evaluated. | +| `request_count` | number | Total request count of the IP in the lookback window. | +| `status_404_count` | number | Number of 404 responses returned to the IP in the lookback window. | +| `status_404_ratio` | number | 404 request ratio, calculated as `status_404_count / request_count`. | +| `ip_host_count` | number | Number of requests from the IP using an IP address directly as the Host header. | +| `ip_host_ratio` | number | Ratio of direct IP address accesses, calculated as `ip_host_count / request_count`. | +| `client_error_count` | number | Number of requests returning 4xx status codes. | +| `server_error_count` | number | Number of requests returning 5xx status codes. | +| `last_seen_unix` | number | Unix timestamp (in seconds) of the last request from the IP in the lookback window. | + +All ratio fields are decimals between `0` and `1`. An 80% ratio should be written as `0.8`, and 50% as `0.5`. + +## Common Expr Syntax + +Automatic IP groups use the Expr syntax. The expression must return a boolean value. + +Common Operators: + +| Operator | Role | Example | +| --- | --- | --- | +| `>`, `>=`, `<`, `<=` | Numeric comparison | `request_count > 100` | +| `==`, `!=` | Equality / Inequality | `ip != "127.0.0.1"` | +| `&&` | Logical AND | `request_count > 100 && status_404_ratio >= 0.8` | +| `||` | Logical OR | `status_404_ratio >= 0.8 || server_error_count > 20` | +| `!` | Logical NOT | `!(ip == "127.0.0.1")` | +| `in` | Value is in list | `ip in ["203.0.113.10", "198.51.100.20"]` | +| `not in` | Value is not in list | `ip not in ["127.0.0.1"]` | +| `()` | Grouping controls operator priority | `(request_count > 100 && status_404_ratio >= 0.8) || server_error_count > 50` | + +## Built-in Presets + +The management console provides two built-in preset rules that can be added directly and adjusted as needed: + +```json +{ + "name": "Single IP High-Frequency 404 Scanning", + "expr": "request_count > 100 && status_404_ratio >= 0.8" +} +``` + +Meaning: A single IP requests more than 100 times in the lookback window, and the 404 status code ratio is at least 80%. + +```json +{ + "name": "Single IP Direct IP Access Mismatch", + "expr": "ip_host_count > 50 && ip_host_ratio > 0.5" +} +``` + +Meaning: A single IP accesses the server directly using an IP address as the Host header more than 50 times, and this type of access represents more than 50% of its total requests. + +## Examples + +High-frequency 404 scanning: + +```json +{ + "lookback_minutes": 60, + "rules": [ + { + "name": "High-Frequency 404 Scanning", + "expr": "request_count > 100 && status_404_ratio >= 0.8" + } + ] +} +``` + +Direct IP access mismatch: + +```json +{ + "lookback_minutes": 30, + "rules": [ + { + "name": "Direct IP Access Mismatch", + "expr": "ip_host_count > 50 && ip_host_ratio > 0.5" + } + ] +} +``` + +Capture both high 4xx and 5xx errors: + +```json +{ + "lookback_minutes": 120, + "rules": [ + { + "name": "Abnormal Error Rates", + "expr": "(client_error_count > 80 && request_count > 100) || server_error_count > 30" + } + ] +} +``` + +Exclude trusted IPs: + +```json +{ + "lookback_minutes": 60, + "rules": [ + { + "name": "404 Scanning Excluding Trusted IPs", + "expr": "ip not in [\"203.0.113.10\", \"198.51.100.20\"] && request_count > 100 && status_404_ratio >= 0.8" + } + ] +} +``` + +## Usage Recommendations + +Start with a shorter lookback window and higher thresholds to monitor matches, then adjust thresholds gradually. The IP Groups page in the management console allows you to click **"Test Rule"** before saving to view matching IPs in the current window immediately. Once an automatic IP group runs, it overwrites the list of IPs. If you want to permanently whitelist or blacklist certain IPs, add them to a manual IP group instead, and reference both manual and automatic groups in your WAF rule groups. + +Updating automatic IP groups does not require publishing configuration versions. Online Agents receive changes via WebSocket and update the local `waf_ip_groups.json` instantly. If WebSocket is unavailable, the Agent reports its local checksum in heartbeats, and the Server syncs only the mismatched IP groups. diff --git a/docs/en/guide/waf-usage.md b/docs/en/guide/waf-usage.md new file mode 100644 index 00000000..4fb30dbc --- /dev/null +++ b/docs/en/guide/waf-usage.md @@ -0,0 +1,162 @@ +# WAF Security Protection + +You will learn: How the OpenFlare edge Web Application Firewall (WAF) works, its protection dimensions, how to manage and reference the three types of IP groups (Manual, Subscription, and Expr-based Automatic IP groups), configure CC protection challenges (PoW human-machine verification) and regional filtering, and achieve sub-second hot updates of IP group members without Nginx reloads. + +--- + +## Core Concepts + +Before configuring security policies, you need to understand the core components of the WAF: + +| Concept | Description | Scope & Activation Method | +| --- | --- | --- | +| **WAF Rule Group (Rule Group)** | A logical collection of security rules, including: IP whitelists/blacklists (direct input or IP group references), country/region limits, CC protection (PoW), and custom block responses. | Supports global enablement or binding to single/multiple websites. **Modifying rule group definitions requires publishing and activating a configuration version**. | +| **IP Group (IP Group)** | A list container storing individual IPs or CIDR blocks. Divided into **Manual**, **Subscription**, and **Automatic** types. WAF rule groups reference IP groups by ID. | Belongs to dynamic resources. **IP group member updates support sub-second WebSocket hot-syncing, completely bypassing Nginx process reloads**. | +| **PoW Challenge (CC PoW)** | A human-machine verification challenge based on Proof of Work. By prompting browsers to solve hash collisions of a specified difficulty, it silently blocks malicious brute-force scripts and bots while keeping legitimate user experience smooth. | A configuration Tab in the rule group. **Modifying PoW parameters requires publishing and activating a configuration version**. | + +--- + +## Recommended Configuration Sequence + +When configuring security protections for your websites, we recommend doing so in the following order: + +1. Navigate to IP Groups, creating the required **Manual IP Groups** (e.g., developer whitelist) or **Automatic IP Groups** (e.g., auto-blocked IPs based on 404 scans). +2. Create or edit a **WAF Rule Group**: + * Bind the IP groups you want to reference or block. + * Configure regional whitelists/blacklists for countries or provinces. + * (Optional) Configure human-machine challenge parameters in the `PoW` Tab. + * Set custom status codes (e.g., 403, 418) and HTML block pages in the `Block Response` Tab. +3. Associate the rule group with the corresponding **Website Configuration**. +4. Publish and activate the configuration version to let the edge node (Agent) apply the WAF rules to filter traffic. + +--- + +## Detailed Step Guide + +### Step 1: Manage and Configure IP Groups + +IP groups are the foundations of large-scale IP filtering. OpenFlare provides three highly resilient types of IP groups: + +#### 1. Manual IP Groups (Manual) +* **Purpose**: Statically maintain a list of verified trusted IPs or long-term blocked IPs/CIDR blocks. +* **Configuration**: Click "Create IP Group" -> select type "Manual" -> enter IPs or CIDRs line-by-line (e.g., `192.168.1.100` or `10.0.0.0/24`). + +#### 2. Subscription IP Groups (Subscription) +* **Purpose**: Integrate third-party threat intelligence databases or IP ranges published by cloud providers. +* **Configuration**: Select type "Subscription" -> enter fetch URL (supports line-separated plain text or standard JSON formats). A background cron job on the Server periodically pulls the subscription source and updates the group members automatically. + +#### 3. Automatic IP Groups (Automatic) +* **Purpose**: **The most aggressive automated defense channel against scans and brute-force attacks**. +* **Configuration**: Select type "Automatic" -> write Expr log aggregation logic. You can directly select built-in presets: + * **Single IP High-Frequency 404 Scanning**: `request_count > 100 && status_404_ratio >= 0.8` (A single IP requesting over 100 times in the past hour with a 404 response ratio of at least 80%). + * **Single IP Direct IP Access Mismatch**: `ip_host_count > 50 && ip_host_ratio > 0.5` (Bypassing domains to hit the server directly using IP address host headers). +* **Test & Run**: Click **"Test Rule"** before saving to preview IPs matching the current log window. Click **"Execute Now"** after saving to aggregate logs immediately and generate the block list. + +> [!TIP] +> For the detailed syntax and available metrics of automatic IP groups, see [WAF Auto IP Group Expressions](./waf-ip-group-expr.md). + +--- + +### Step 2: Create and Configure a WAF Rule Group + +1. Navigate to the **"WAF"** section in the side menu, and click **"Create Rule Group"**. +2. Enter the rule group name (e.g., `production-api-shield`), and select if it is a "Global Rule Group". +3. Enter rule group details, and configure the tabs sequentially below: + +#### 1. Whitelist / Blacklist Configuration (Allow / Block Lists) +* **Direct IPs**: Enter individual IPs or CIDR blocks line-by-line that need temporary whitelisting or blacklisting directly in the text area. +* **IP Group Reference**: Click "Bind IP Groups", selecting the manual, automatic, or subscription IP groups you configured in Step 1. Whitelists permit traffic instantly, whereas blacklists block it. + +#### 2. Regional Restriction (GeoIP) +* **Description**: OpenFlare integrates GeoIP geolocation resolution. +* **Configuration**: Toggle the regional restriction switch, selecting "Allow Only" or "Block". +* * For example, if your service is only intended for domestic users, set the mode to "Allow Only" and check `China` in the country list. +* * Supports refining to specific provinces/regions, enabling you to block malicious traffic originating from targeted geographic zones with one click. + +#### 3. Human-Machine Challenge Configuration (PoW CC Protection) +* **Description**: Enable CC protection human-machine challenges. When a request triggers the CC protection threshold, the browser renders a silent challenge page, solving a mathematical challenge (hash collision) within several hundred milliseconds. Upon passing, it sets a Cookie and allows subsequent visits. This is seamless to actual users but blocks brute-force scripts and CC tools that do not support JS execution or mathematical computations. +* **Core Parameters**: + * **Status**: Enable / Disable. + * **Hash Difficulty**: Controls the computation difficulty (recommending `4` or `5`). + * **Cookie Expiration**: How long the verification remains valid after passing (e.g., `3600` seconds). + * **Custom Challenge HTML**: Customize the Loading page style of the challenge to match your business design. + +#### 4. Block Response (Block Response) +* **Description**: Define the behavior of the WAF when blocking malicious requests. +* **Configuration**: + * **Block Status Code**: Customize the HTTP status code returned, e.g., the standard `403` or a fun `418 (I'm a teapot)`. + * **Block Response Body**: Input custom HTML content shown to blocked attackers (e.g., "WAF Interception: Your request has been logged"). + +--- + +### Step 3: Associate the Rule Group with Websites + +Once configured, the rule group does not automatically take effect; you need to bind it to specific website configurations. + +* **Option A (Recommended)**: In the **"Bind Websites"** Tab of the rule group details, select the websites you wish to apply this rule group to and save. +* **Option B**: Return to **"Website Configuration"**, edit a specific website, and check and bind the rule group in the "Security Protection" section. + +> [!NOTE] +> If a rule group is marked as **"Global Rule Group (is_global)"**, it applies to **all websites** hosted on the gateway automatically, requiring no manual binding. + +--- + +### Step 4: Publish & Activate Configurations + +1. If you modify **rule group definitions**, **GeoIP scopes**, **PoW CC difficulties**, or **website-to-rule-group bindings**: + * Click **"Preview Config"** -> **"Publish & Activate"** in the top right corner. + * Once the Agent pulls and validates the new version, it rewrites local core OpenResty config files (`waf_config.json`, etc.) and gracefully reloads the processes to apply the policies. +2. If you only update **IP group members** (e.g., adding/deleting an IP in a manual IP group, or an automatic IP group aggregates a new set of blocked IPs periodically): + * **No publication or activation is required!** + * The Server calculates the new MD5 Checksum of the IP group immediately after updating the database. + * The control plane **broadcasts the modified IP group members in real-time to all online Agents via WebSocket**. The Agent overwrites the runtime local disk file `waf_ip_groups.json` incrementally. + * The OpenResty Lua engine calculates the file hash in microseconds when processing new requests. If it detects a Checksum change, it reloads it into the memory dictionary (`ngx.shared`) in real-time. **This entire process requires absolutely no Nginx service reloads, having zero impact on online high-concurrency operations**. + * Even if the WebSocket connection drops, the Agent reports its local Checksum in every heartbeat cycle, and the Server syncs the differential updates to guarantee synchronization. + +--- + +## WAF Evaluation Flow (Filtering Funnel) + +When an external request reaches the OpenResty data plane, the WAF runtime evaluates it in the `access` phase according to the funnel decision chain below. Once a match is made, evaluation terminates: + +```text + Request enters access phase + │ + v + Get all active rule groups bound to this site (Global + Bound Custom groups) + │ + v + 1. Matches IP whitelist / Whitelist IP group? ──────(Yes)─────► [ Allow (ALLOW) ] + │ (No) + v + 2. Matches country / province whitelist? ────────(Yes)─────► [ Allow (ALLOW) ] + │ (No) + v + 3. Matches IP blacklist / Blacklist IP group? ──────(Yes)─────► [ Block (BLOCK) ] ──► Return status & HTML block page + │ (No) + v + 4. Matches country / province blacklist? ────────(Yes)─────► [ Block (BLOCK) ] ──► Return status & HTML block page + │ (No) + v + 5. Is PoW CC protection enabled for this site? + ├───(Yes)───► [ Validate PoW Cookie ] ──(Passed)──► [ Allow (ALLOW) ] + │ │ + │ (Not Passed) + │ v + │ [ Render PoW Challenge ] ──(Solved)──► Set Cookie & Allow + v + 6. No rules triggered, legitimate traffic ─────────────────────► [ Allow (ALLOW) ] +``` + +--- + +## Best Practices & Tuning Recommendations + +* **Whitelist Precedence & Protection**: Before deploying strict blacklists or regional blocks, we strongly recommend creating a "Trusted IP Group" containing your team's office egress IPs, local development IPs, and third-party callback server IPs (e.g., WeChat or Alipay payment callback addresses), and prioritizing it in the rule group's **whitelist**. This effectively prevents accidental blockages. +* **Reasonably Fine-tune PoW Difficulty**: Human-machine CC challenge hash difficulty (`challenge_difficulty`) is a double-edged sword: + * Difficulty `3`: Computes almost instantly, providing low protection. + * Difficulty `4`: Normal phones/low-end browsers solve it in 100-300ms, providing good protection. + * Difficulty `5`: Requires 500ms-2s, providing strong protection but low-end client browsers might perceive slight loading delays. + * Difficulty `6` and above: Computes exponentially slower, easily freezing client browser CPUs. **We strongly recommend choosing `4` or `5` in production**. +* **Utilize "Test Rule"**: For automatic IP groups, always click **"Test Rule"** before saving. By inspecting the list of matching IPs in the current window, verify if your Expr expressions thresholds (such as request counts, 404 ratios, etc.) are too broad or too strict, preventing accidental blockages of legitimate users. +* **Isolate Static & Dynamic Blacklists**: Never enter static malicious IPs that require permanent blocks directly into automatic IP groups (since the aggregated list will be overwritten in the next cron cycle). You should add permanent malicious IPs into a dedicated "Manual Blacklist IP Group" and reference both the manual and automatic groups in your rule groups. diff --git a/docs/en/guide/zone-domain-migration.md b/docs/en/guide/zone-domain-migration.md new file mode 100644 index 00000000..01de31f1 --- /dev/null +++ b/docs/en/guide/zone-domain-migration.md @@ -0,0 +1,52 @@ +# Zone Domain Migration and Release Acceptance + +When migrating from the legacy `managed_domains` / inline domain columns of reverse proxy routes to the Zone + Zone Domain model, data import and table structure upgrades are both completed by the **automatic goose migration at Server startup** — no separate import command is needed. + +## What Happens During Upgrade + +When starting (or rolling-upgrading) a Server version that includes the Zone rework, **no manual command is required**; `migrator.Migrate()` automatically: + +1. Applies goose SQL: creates `of_zones` / `of_zone_domains` (if they do not yet exist). +2. **Automatically imports** the legacy route domain columns (and `of_managed_domains` when routes have no domains) as Zone / Zone Domains, binding `proxy_route_id` / `cert_id` (registering root domains via public suffix list parsing). +3. Continues goose SQL: drops the redundant domain/certificate columns from `of_managed_domains` and `of_proxy_routes`. + +The import is idempotent: existing domains are skipped or have their route binding back-filled. + +**If historical data cannot be parsed (conflicting domains, invalid root domains, missing certificates, etc.), startup fails.** Fix the data or restore a backup and start again to retry. + +## Recommended Actions + +### 1. Back Up Before Upgrading + +```bash +# PostgreSQL example +pg_dump "$DATABASE_URL" > openflare-pre-zone-$(date +%Y%m%d).sql + +# Or copy the backup volume / snapshot; for SQLite, copy the database file in the data directory +``` + +Optional: note down the current **active config version number** and checksum in the admin panel for config rollback comparison. + +### 2. Upgrade and Start the Server + +Deploy the new version and start it. Watch the goose success messages in the startup log; if "Zone migration failed (N conflicts)" appears, fix the source data according to the conflicts listed in the log and restart. + +### 3. Post-Upgrade Checks + +1. Admin panel **Websites** `/websites`: check whether Zone root domains and domain counts are reasonable. +2. Zone details: domains, certificates, associated route IDs. +3. **Reverse proxy routes**: domain bindings come from Zone Domains, not legacy hand-written fields. + +### 4. Config Preview and Release + +1. Review the config diff / preview in the admin panel. +2. Verify **per route**: `server_name` set, certificate paths, WAF Route ID, Pages references. +3. **Allow** the redundant `domain` / `domains` / `cert_ids` on routes in old snapshot JSON to disappear. +4. **Do not allow** data-plane semantic changes. +5. After the preview passes, release it; if needed, activate the pre-upgrade version in config versions for config rollback. For database rollback, use the pre-upgrade backup (down migrations do not backfill business domain data). + +## Related Docs + +* [Zone & Domain Resource Design](../design/zone-design.md) +* [Create a Reverse Proxy Config](./proxy-config.md) +* [Publish First Configuration](./first-site.md) diff --git a/docs/en/index.md b/docs/en/index.md new file mode 100644 index 00000000..7a1aa6ba --- /dev/null +++ b/docs/en/index.md @@ -0,0 +1,38 @@ +--- +layout: home + +hero: + name: OpenFlare + text: Open-source CDN Orchestration & Edge Security Platform + tagline: Supports reverse proxy, centralized configuration synchronization, Pages static hosting, secure intranet penetration (Tunnels), dynamic WAF protection, and anti-CC challenges. + actions: + - theme: brand + text: Quick Start + link: /en/guide/quick-start + - theme: alt + text: Design Boundaries + link: /en/design/ + - theme: alt + text: GitHub + link: https://github.com/Rain-kl/OpenFlare + +features: + - icon: 🛰️ + title: Centralized Config Sync + details: Sync configurations across all nodes in real time via WebSockets and heartbeats with sub-second hot reload. Instantly retrieve alerts and statuses. + - icon: 🌐 + title: Distributed CDN Orchestration + details: Orchestrate scattered and independent OpenResty nodes into a highly collaborative CDN fleet with multi-upstream load balancing. + - icon: 📄 + title: Pages Static Hosting + details: Upload pre-built frontend zip assets directly; edge nodes pull, extract, and serve them locally at high performance with API proxying. + - icon: 🚇 + title: Secure Intranet Penetration (Tunnels) + details: An open-source alternative to Cloudflare Tunnels. Expose local intranet services securely to the public network without a public IP or open inbound ports. + - icon: 🛡️ + title: Edge WAF Protection + details: Dynamic WAF rules with differential syncing of IP groups to Lua shared memory without Nginx reloads, plus country-level regional access control. + - icon: 🧩 + title: Anti-CC & Bot Defense (PoW) + details: Built-in high-performance client-side cryptographic Proof of Work challenges (similar to Turnstile) to intercept botnets and scrapers at the edge. +--- diff --git a/docs/en/reference/cli.md b/docs/en/reference/cli.md new file mode 100644 index 00000000..1e468631 --- /dev/null +++ b/docs/en/reference/cli.md @@ -0,0 +1,149 @@ +# CLI Commands + +You will learn: Common commands for starting, building, testing, installing, and uninstalling the OpenFlare Server, Admin Frontend, Agent, Swagger, and Documentation site. + +## Server + +Start from source: + +```bash +cd openflare-server +export SESSION_SECRET='replace-with-random-string' +export SQLITE_PATH='./openflare.db' +export LOG_LEVEL='info' +go run . +``` + +Specify listening port and logging directory: + +```bash +go run . --port 3000 --log-dir ./logs +``` + +Run tests: + +```bash +cd openflare-server +GOCACHE=/tmp/openflare-go-cache go test ./... +``` + +## Frontend + +Development: + +```bash +cd openflare-server/web +pnpm install +pnpm dev +``` + +Build static assets: + +```bash +cd openflare-server/web +pnpm build +``` + +Linting and testing checks: + +```bash +cd openflare-server/web +pnpm lint +pnpm typecheck +pnpm test +``` + +## Agent + +Run from source: + +```bash +cd openflare-agent +go run ./cmd/agent -config /path/to/agent.json +``` + +Compile: + +```bash +cd openflare-agent +go build -o openflare-agent ./cmd/agent +``` + +Run tests: + +```bash +cd openflare-agent +GOCACHE=/tmp/openflare-go-cache go test ./... +``` + +## Relay (Server-side) + +Run from source: + +```bash +cd openflare-relay +go run ./cmd -config /path/to/relay.json +``` + +Compile: + +```bash +cd openflare-relay +go build -o openflare-relay ./cmd +``` + +## OpenFlared (Client-side) + +Run from source: + +```bash +cd openflared +go run ./cmd -config /path/to/flared.json +``` + +Compile: + +```bash +cd openflared +go build -o openflared ./cmd +``` + +## Install Agent + +```bash +curl -fsSL https://raw.githubusercontent.com/Rain-kl/OpenFlare/main/scripts/install-agent.sh | bash -s -- \ + --server-url http://your-server:3000 \ + --agent-token YOUR_AGENT_TOKEN +``` + +## Uninstall Agent + +```bash +curl -fsSL https://raw.githubusercontent.com/Rain-kl/OpenFlare/main/scripts/uninstall-agent.sh | bash +``` + +## Swagger + +Regenerate Swagger documentation: + +```bash +go install github.com/swaggo/swag/cmd/swag@v1.16.4 +cd openflare-server +swag init -g main.go -o docs +``` + +## Docs + +Local preview: + +```bash +cd docs +pnpm dev +``` + +Build: + +```bash +cd docs +pnpm build +``` diff --git a/docs/en/reference/configuration.md b/docs/en/reference/configuration.md new file mode 100644 index 00000000..4282cc33 --- /dev/null +++ b/docs/en/reference/configuration.md @@ -0,0 +1,370 @@ +# Configuration Options + +You will learn: What configuration sources are supported by OpenFlare Server, frontend builds, and Agents; what the default configuration values are; and how to configure common deployment combinations. + +This document aggregates the currently supported configuration options for OpenFlare Server and Agent in version `1.0.0`, keeping only running parameters that are currently active. + +## Configuration Sources + +The Server supports three types of configuration sources: + +1. CLI arguments. +2. Environment variables. +3. Runtime configurations in the database `options` table. + +The Agent supports: + +1. The `-config` CLI argument. +2. The `agent.json` configuration file. +3. A small set of environment variables for overriding logs and settings. + +The Relay (Server-side) supports: + +1. The `-config` CLI argument. +2. The `relay.json` configuration file. +3. Persistent environment variables for overriding runtime flags. + +The Client (Intranet Client) supports: + +1. The `-config` CLI argument. +2. The `flared.json` configuration file. +3. Startup overrides and logging environment variables. + +## Configuration File Locations + +| Component | Default Location | Description | +| --- | --- | --- | +| Server SQLite | `openflare.db` | Can be customized via `SQLITE_PATH` | +| Agent Config | `./agent.json` | Can be specified via `-config` | +| One-Click Agent | `/opt/openflare-agent/agent.json` | Generated by the installation script by default | +| Agent Data Dir | `data` in the config folder | Can be customized via `data_dir` | +| Relay Config | `./relay.json` | Can be specified via `-config` | +| One-Click Relay | `/opt/openflare-relay/relay.json` | Generated by the installation script by default | +| Client Config | `./flared.json` | Can be specified via `-config` | +| One-Click Client | `/opt/openflared/flared.json` | Generated by the installation script by default | + +## Server CLI Arguments + +```bash +cd openflare-server +go run . --port 3000 --log-dir ./logs +``` + +| Argument | Description | Default Value | +| --- | --- | --- | +| `--port` | Port the Server listens on | `3000` | +| `--log-dir` | Directory to output logs | Empty (stdout) | +| `--version` | Outputs current version and exits | `false` | +| `--help` | Outputs help information and exits | `false` | + +## Server Environment Variables + +| Environment Variable | Description | Default Value | +| --- | --- | --- | +| `PORT` | Port the Server listens on | `3000` | +| `GIN_MODE` | Gin framework running mode | Defaults to release unless `debug` | +| `LOG_LEVEL` | Logging level | `info` | +| `SESSION_SECRET` | Session signing key | Randomly generated on startup | +| `SQLITE_PATH` | SQLite database file path | `openflare.db` | +| `DSN` | PostgreSQL DSN (takes precedence over SQLite) | Empty | +| `SQL_DSN` | Legacy PostgreSQL DSN (lower priority than `DSN`) | Empty | +| `REDIS_CONN_STRING` | Redis connection string | Empty | +| `AGENT_TOKEN` | Legacy global Agent Token | Empty | + +Notes: + +* If both `DSN` and `SQL_DSN` exist, `DSN` is prioritized. +* If either `DSN` or `SQL_DSN` coexist with `SQLITE_PATH`, PostgreSQL is prioritized. +* If the target PostgreSQL database is empty and a local SQLite file exists at `SQLITE_PATH`, the Server automatically migrates SQLite data table-by-table on startup. +* `SESSION_SECRET` must be explicitly configured in production. +* If `REDIS_CONN_STRING` is unconfigured, co-located features fall back to in-memory implementations. + +## Runtime Options + +The following options are maintained in the admin settings page and support hot reloading: + +| Parameter | Description | Default Value | +| --- | --- | --- | +| `AgentHeartbeatInterval` | Heartbeat interval for Agents (ms) | `10000` | +| `AgentWebsocketUpgradeEnabled` | Toggles WebSocket upgrades after successful HTTP heartbeat | `true` | +| `NodeOfflineThreshold` | Threshold duration to mark a node offline (ms) | `120000` | +| `AgentUpdateRepo` | GitHub repository for Agent self-updates | `Rain-kl/OpenFlare` | +| `GeoIPProvider` | Geolocation resolution provider | `ipinfo` | +| `DatabaseAutoCleanupEnabled` | Toggles daily automatic cleanup of observability logs | `false` | +| `DatabaseAutoCleanupRetentionDays` | Data retention duration in days, minimum 1 day | `30` | +| `GlobalApiRateLimitNum` / `GlobalApiRateLimitDuration` | Global API rate limit count / window | `300` / `180` | +| `GlobalWebRateLimitNum` / `GlobalWebRateLimitDuration` | Global Web rate limit count / window | `300` / `180` | +| `CriticalRateLimitNum` / `CriticalRateLimitDuration` | Sensitive API rate limit count / window | `100` / `1200` | + +Notes: + +* When `DatabaseAutoCleanupEnabled` is enabled, the Server deletes `node_access_logs`, `node_metric_snapshots`, and `node_request_reports` daily at 3:00 AM. +* `DatabaseAutoCleanupRetentionDays` must be greater than or equal to 1. +* Leaving retention days blank during a manual trigger in the console deletes all historic logs instantly. +* The GitHub Release in `AgentUpdateRepo` must contain a matching `.sha256` checksum file for every Agent binary (e.g., `openflare-agent-linux-amd64.sha256`); the Agent validates this checksum before replacing the local executable. +* Third-party logins no longer use `GitHubOAuthEnabled`, `GitHubClientId`, and `GitHubClientSecret` as main configuration entrypoints; these legacy options are used only for migrating default GitHub credentials during upgrades. +* The legacy WeChat login options are kept for backward compatibility, but the option page no longer edits them. +* Legacy Cloudflare Turnstile options and validation logic are retained and will work normally. + +## OpenResty Parameters + +OpenResty performance and caching parameters are managed in the `options` table, including: + +* `OpenRestyWorkerProcesses` +* `OpenRestyWorkerConnections` +* `OpenRestyWorkerRlimitNofile` +* `OpenRestyKeepaliveTimeout` +* `OpenRestyProxyConnectTimeout` +* `OpenRestyProxySendTimeout` +* `OpenRestyProxyReadTimeout` +* `OpenRestyProxyBufferingEnabled` +* `OpenRestyGzipEnabled` +* `OpenRestyCacheEnabled` +* `OpenRestyCachePath` +* `OpenRestyCacheMaxSize` + +These parameters must be validated, saved, and rendered structurally. + +Constraints: + +* The console no longer exposes `resolver` settings. +* Upstreams are rendered uniformly as named `upstream` blocks with keepalive enabled. +* Single upstreams carrying a base path or query have their URI correctly appended in `proxy_pass`. +* Multi-upstreams must be pure `scheme://host[:port]` using the same protocol within a single rule. +* `OpenRestyCacheEnabled` enables cache infrastructure and global defaults; the actual caching matching policies (by URL, suffix, or path) are configured per `proxy_routes`. +* The default cache key is `$scheme$host$request_uri`. +* Default `keepalive_timeout` is `20` seconds; default `proxy_connect_timeout` is `3` seconds. +* The default event model is `epoll` with `multi_accept` enabled. +* HTTPS listeners use the independent `http2 on;` directive to avoid deprecation warnings for `listen ... http2` in newer Nginx/OpenResty versions. + +## Frontend Build Environment Variables + +| Environment Variable | Description | Default Value | +| --- | --- | --- | +| `NEXT_PUBLIC_API_BASE_URL` | Base path for frontend API calls | `/api` | +| `NEXT_PUBLIC_APP_VERSION` | Application version shown in the UI | `dev` | +| `NEXT_DEV_BACKEND_URL` | Target backend proxied by the local dev server | `http://127.0.0.1:3000` | + +## Agent Environment Variables + +| Environment Variable | Description | Default Value | +| --- | --- | --- | +| `LOG_LEVEL` | Logging level for the Agent | `info` | +| `OPENFLARE_SERVER_URL` | Server URL; overrides `agent.json` | Empty | +| `OPENFLARE_AGENT_TOKEN` | Node-specific Token; overrides `agent.json` | Empty | +| `OPENFLARE_DISCOVERY_TOKEN` | Auto-registration Token; overrides `agent.json` | Empty | +| `OPENFLARE_NODE_NAME` | Node name; overrides `agent.json` | Empty | +| `OPENFLARE_NODE_IP` | Node IP; overrides `agent.json` | Empty | +| `OPENFLARE_DATA_DIR` | Agent data directory; overrides `agent.json` | Empty | +| `OPENFLARE_OPENRESTY_PATH` | Path to OpenResty binary; overrides `agent.json` | Empty | +| `OPENFLARE_HEARTBEAT_INTERVAL` | Heartbeat interval; overrides `agent.json` | Empty | +| `OPENFLARE_REQUEST_TIMEOUT` | Request timeout; overrides `agent.json` | Empty | +| `OPENFLARE_OPENRESTY_OBSERVABILITY_PORT` | Local observability port; overrides `agent.json` | Empty | +| `OPENFLARE_MMDB_PATH` | WAF GeoIP mmdb path; overrides `agent.json` | Empty | +| `OPENFLARE_MMDB_UPDATE_INTERVAL` | GeoIP mmdb update interval; overrides `agent.json` | Empty | +| `OPENFLARE_MMDB_DOWNLOAD_URL` | GeoIP mmdb download link; overrides `agent.json` | Empty | + +## Agent CLI Arguments + +| Argument | Description | Default Value | +| --- | --- | --- | +| `-config` | Path to the Agent configuration file | `./agent.json` | + +## Agent Configurations Fields + +| Field | Description | Required | Default Value / Behavior | +| --- | --- | --- | --- | +| `server_url` | Control plane URL | Yes | None | +| `agent_token` | Node-specific access Token | Mutually exclusive with discovery_token | Empty | +| `discovery_token` | Global auto-registration Token | Mutually exclusive with agent_token | Empty | +| `node_name` | Node name | No | Hostname | +| `node_ip` | Node IP | No | Auto-detect, resolves outbound public IP via realip.cc first, falls back to local adapters | +| `openresty_path` | Path to the OpenResty binary | No | `"openresty"` | +| `openresty_observability_port` | Observability port for health checks | No | `18081` | +| `data_dir` | Agent data directory | No | `data` in the config folder | +| `main_config_path` | Write path for Nginx main configuration | No | `data_dir/etc/nginx/nginx.conf` | +| `route_config_path` | Write path for route configurations | No | `data_dir/etc/nginx/conf.d/openflare_routes.conf` | +| `access_log_path` | Write path for OpenResty access logs | No | `data_dir/var/log/openflare/access.log` | +| `cert_dir` | Write directory for SSL certificates | No | `data_dir/etc/nginx/certs` | +| `openresty_cert_dir` | Read directory for certificates in Nginx | No | Same as `cert_dir` | +| `lua_dir` | Write directory for Lua scripts and assets | No | `data_dir/etc/nginx/lua` | +| `openresty_lua_dir` | Read directory for Lua scripts in Nginx | No | Same as `lua_dir` | +| `runtime_config_dir` | Write directory for Agent runtime configs | No | `data_dir/etc/openflare` | +| `mmdb_path` | WAF GeoIP database file path | No | `data_dir/etc/openflare/GeoLite2-Country.mmdb` | +| `mmdb_update_interval` | WAF GeoIP database check interval | No | `86400000` milliseconds | +| `mmdb_download_url` | WAF GeoIP database download URL | No | Built-in GeoLite2 Country URL | +| `observability_buffer_path` | Buffer path for retry metrics logs | No | `data_dir/var/lib/openflare/observability-buffer.json` | +| `observability_replay_minutes` | Lookback window for metric retries | No | `15` | +| `state_path` | Path to store local state JSON file | No | `data_dir/var/lib/openflare/agent-state.json` | +| `heartbeat_interval` | Heartbeat polling interval | No | `10000` milliseconds | +| `request_timeout` | HTTP request timeout duration | No | `10000` milliseconds | + +Notes: + +* `agent_token` and `discovery_token` cannot both be empty. +* `heartbeat_interval` and `request_timeout` support integer milliseconds or Go duration strings. +* If `AgentWebsocketUpgradeEnabled` is enabled on the Server, the Agent upgrades the HTTP heartbeat to WebSocket; it automatically falls back to HTTP heartbeats if it fails or disconnects. +* If `openresty_path` is left blank, the Agent calls `openresty` on the host. +* Periodic health checks query `http://127.0.0.1:/openflare/stub_status` instead of executing `openresty -t`; validation prior to reloads, starts, or rollbacks still runs `openresty -t -c `. +* The Agent initializes and periodically updates `mmdb_path` to support GeoIP region checks; failures to update write warnings and do not disrupt configuration synchronizations. +* The Agent boots normally if `agent.json` is missing but environment variables (`OPENFLARE_SERVER_URL` and a Token) are available; environment variables override JSON settings. +* If `node_ip` is left blank, the Agent resolves its outbound IP via `https://realip.cc` first, which is suitable for Docker/NAT networks. +* If the Agent registers a private `node_ip`, the Server prioritizes saving the public TCP connection IP, preventing NAT adapters from registering internal IPs. +* Enabling "Lock Node IP" in the console retains the manual IP; subsequent Agent registration or heartbeats do not overwrite it. + +## Relay Environment Variables + +| Environment Variable | Description | Default Value | +| --- | --- | --- | +| `LOG_LEVEL` | Logging level for the Relay | `info` | +| `OPENFLARE_SERVER_URL` | Server URL; overrides `relay.json` | Empty | +| `OPENFLARE_AGENT_TOKEN` | Node-specific Token; overrides `relay.json` | Empty | +| `OPENFLARE_DISCOVERY_TOKEN` | Auto-registration Token; overrides `relay.json` | Empty | +| `OPENFLARE_NODE_NAME` | Node name; overrides `relay.json` | Empty | +| `OPENFLARE_NODE_IP` | Node IP; overrides `relay.json` | Empty | +| `OPENFLARE_DATA_DIR` | Relay data directory; overrides `relay.json` | Empty | +| `OPENFLARE_FRPS_PATH` | frps binary path; overrides `relay.json` | Empty | + +## Relay CLI Arguments + +| Argument | Description | Default Value | +| --- | --- | --- | +| `-config` | Path to the Relay configuration file | `./relay.json` | + +## Relay Configuration Fields + +| Field | Description | Required | Default Value / Behavior | +| --- | --- | --- | --- | +| `server_url` | Control plane URL | Yes | None | +| `agent_token` | Node-specific access Token | Mutually exclusive with discovery_token | Empty | +| `discovery_token` | Global auto-registration Token | Mutually exclusive with agent_token | Empty | +| `node_name` | Node name | No | Hostname | +| `node_ip` | Relay listening IP for tunnel traffic | No | Auto-detect, prioritizes outbound public IP | +| `frps_path` | Path to the `frps` binary | No | `frps` (system PATH) | +| `data_dir` | Relay runtime data directory | No | `data` in the config folder | +| `state_path` | Path to store local state JSON file | No | `data_dir/relay-state.json` | +| `heartbeat_interval` | Heartbeat polling interval | No | `10000` milliseconds, supports Go duration strings | +| `request_timeout` | HTTP request timeout duration | No | `10000` milliseconds, supports Go duration strings | + +## OpenFlared (Client) Environment Variables + +| Environment Variable | Description | Default Value | +| --- | --- | --- | +| `LOG_LEVEL` | Logging level for the client | `info` | +| `OPENFLARE_SERVER_URL` | Server URL; overrides `flared.json` | Empty | +| `OPENFLARE_TUNNEL_TOKEN` | Tunnel access Token; overrides `flared.json` | Empty | +| `OPENFLARE_DATA_DIR` | Client data directory; overrides `flared.json` | Empty | +| `OPENFLARE_FRPC_PATH` | frpc binary path; overrides `flared.json` | Empty | + +## OpenFlared (Client) CLI Arguments + +| Argument | Description | Default Value | +| --- | --- | --- | +| `-config` | Path to the client configuration file | `./flared.json` | + +## OpenFlared (Client) Configuration Fields + +| Field | Description | Required | Default Value / Behavior | +| --- | --- | --- | --- | +| `server_url` | Control plane URL | Yes | None | +| `tunnel_token` | Tunnel dedicated access Token | Yes | None | +| `frpc_path` | Path to the `frpc` binary | No | `frpc` (system PATH) | +| `data_dir` | Client runtime data directory | No | `data` in the config folder | +| `state_path` | Path to store local state JSON file | No | `data_dir/flared-state.json` | +| `heartbeat_interval` | Heartbeat polling interval | No | `10000` milliseconds, supports Go duration strings | +| `sync_interval` | Configuration sync interval | No | `30000` milliseconds, supports Go duration strings | +| `request_timeout` | HTTP request timeout duration | No | `10000` milliseconds, supports Go duration strings | + +## Common Configuration Combos + +### Production Server + PostgreSQL + +```bash +export SESSION_SECRET='replace-with-a-long-random-string' +export DSN='postgres://openflare:replace-with-strong-password@postgres:5432/openflare?sslmode=disable' +export GIN_MODE='release' +export LOG_LEVEL='info' +``` + +### Local Server + SQLite + +```bash +export SESSION_SECRET='dev-session-secret' +export SQLITE_PATH='./openflare-dev.db' +export LOG_LEVEL='debug' +go run . +``` + +### Agent + Default OpenResty + +```json +{ + "server_url": "http://your-server:3000", + "agent_token": "replace-with-node-auth-token", + "data_dir": "/opt/openflare-agent/data", + "openresty_path": "openresty", + "heartbeat_interval": 10000, + "request_timeout": 10000 +} +``` + +### Agent + Customized OpenResty Paths + +```json +{ + "server_url": "http://your-server:3000", + "agent_token": "replace-with-node-auth-token", + "data_dir": "/var/lib/openflare-agent", + "openresty_path": "/usr/local/openresty/nginx/sbin/openresty", + "main_config_path": "/var/lib/openflare-agent/etc/nginx/nginx.conf", + "route_config_path": "/var/lib/openflare-agent/etc/nginx/conf.d/openflare_routes.conf", + "access_log_path": "/var/lib/openflare-agent/var/log/openflare/access.log", + "cert_dir": "/var/lib/openflare-agent/etc/nginx/certs", + "lua_dir": "/var/lib/openflare-agent/etc/nginx/lua", + "runtime_config_dir": "/var/lib/openflare-agent/etc/openflare", + "heartbeat_interval": 10000, + "request_timeout": 10000 +} +``` + +### Relay (Server-side) Default Configuration + +`relay.json`: + +```json +{ + "server_url": "http://your-server:3000", + "agent_token": "replace-with-relay-auth-token", + "frps_path": "frps", + "data_dir": "/opt/openflare-relay/data", + "heartbeat_interval": 10000, + "request_timeout": 10000 +} +``` + +### OpenFlared (Client-side) Default Configuration + +`flared.json`: + +```json +{ + "server_url": "http://your-server:3000", + "tunnel_token": "replace-with-tunnel-token", + "frpc_path": "frpc", + "data_dir": "/opt/openflared/data", + "heartbeat_interval": 10000, + "sync_interval": 30000, + "request_timeout": 10000 +} +``` + +## Maintenance Rules + +This document must be updated in sync when any of the following change: + +* Server CLI arguments. +* Server environment variables. +* Agent CLI arguments and configuration parameters. +* Relay CLI arguments and configuration parameters. +* Client CLI arguments and configuration parameters. +* Default values, scopes, or examples of any configuration items. diff --git a/docs/en/reference/index.md b/docs/en/reference/index.md new file mode 100644 index 00000000..009679cc --- /dev/null +++ b/docs/en/reference/index.md @@ -0,0 +1,13 @@ +# Reference Manuals + +You will learn: Which information belongs to stable reference manuals, and where to look up configurations, commands, APIs, and repository structures. + +This section collects stable information at the runtime, API, and repository layers, suitable for rapid lookup during deployment, integration, and troubleshooting. + +| Page | Content | +| --- | --- | +| [Configuration Options](./configuration.md) | Server environment variables, CLI arguments, runtime Options, and Agent configuration parameters | +| [CLI Commands](./cli.md) | Common CLI commands for starting, building, testing, installing, and uninstalling | +| [API Conventions](./api.md) | Response structures, authentication, and routing paths for Admin and Agent APIs | +| [Repository Structure](../design/repository.md) | Scope of responsibilities and folder layering of the Server, Agent, Relay, and Client | +| [Deployment & Upgrade](../deployment/) | Server and Agent deployment, configuration, and upgrade guides (dedicated section) | diff --git a/frontend/components/common/home/home-main.tsx b/frontend/components/common/home/home-main.tsx index 99ec10b0..53cbb916 100644 --- a/frontend/components/common/home/home-main.tsx +++ b/frontend/components/common/home/home-main.tsx @@ -35,7 +35,7 @@ export function HomeMain() { title: '使用文档', description: '学习如何集成 API 及日常操作帮助指南', icon: HelpCircle, - url: 'https://open-flare.pages.dev/', + url: 'https://openflare.fyrn.link/', color: 'text-purple-500', bgColor: 'bg-purple-500/10', borderColor: 'hover:border-purple-500/30', diff --git a/frontend/components/common/settings/other-tab.tsx b/frontend/components/common/settings/other-tab.tsx index acab0f85..7defd507 100644 --- a/frontend/components/common/settings/other-tab.tsx +++ b/frontend/components/common/settings/other-tab.tsx @@ -193,7 +193,7 @@ const MENU_GROUPS: MenuGroup[] = [ nameKey: 'groupDocs', items: [ { - path: 'https://open-flare.pages.dev/', + path: 'https://openflare.fyrn.link/', labelKey: 'usageDocs', descKey: 'descUsageDocs', icon: FileText, diff --git a/frontend/components/home/developer-section.tsx b/frontend/components/home/developer-section.tsx index e0ee8e22..14969f54 100644 --- a/frontend/components/home/developer-section.tsx +++ b/frontend/components/home/developer-section.tsx @@ -161,7 +161,7 @@ curl -X POST https://api.example.com/api/v1/auth/register \\
diff --git a/frontend/components/home/footer-section.tsx b/frontend/components/home/footer-section.tsx index 98c5f4bc..80db79af 100644 --- a/frontend/components/home/footer-section.tsx +++ b/frontend/components/home/footer-section.tsx @@ -75,7 +75,7 @@ export const FooterSection = React.memo(function FooterSection({

开发