feat(log): decouple log storage from ClickHouse with switchable logstore

- New internal/repository/logstore abstraction: exported domain interfaces
  (AccessLogStore/ObservabilityStore/UserAccessLogStore/StatusStore),
  config-driven provider (Active/Build/Migrating/SetConfigReader), GORM
  implementation for PostgreSQL/SQLite (incl. hourly rollups computed in
  real time, migration listers, PG partition maintenance), and a ClickHouse
  wrapper preserving the native batch path; repository facade delegates to
  logstore; import-lint test enforces apps never import analyticsrepo.
- ClickHouse is now optional: the log DB is either the main DB (postgres
  when database.enabled, else sqlite) or clickhouse; boot validation +
  first-run seed; log_database / log_db_migration are protected keys.
- New user task 切换日志数据库 (of_log_db_switch): freeze log writes,
  drain batch writers, copy all 6 raw log tables by id (preserving IDs)
  with target-partition pre-creation for PG, flip log_database on success,
  clear the freeze flag on failure.
- Per-store retention (log_retention_days_*) with expiry cleanup folded
  into the daily system_cleanup task; legacy database_auto_cleanup_* and
  of_database_auto_cleanup decommissioned.
- goose migrations: 6 log tables in PG (2 monthly-partitioned) + SQLite,
  retention config seeds, schedule cleanup; GET
  /api/v1/admin/status/log-database endpoint; frontend retention settings,
  switch-task UI and status badge; changelog and docs updated.

docs(plan): log database decoupling implementation plan

docs(design): log database decoupling design (ClickHouse optional)
This commit is contained in:
ryan
2026-08-08 11:36:56 +08:00
parent 734fe45baa
commit 7d71f1e4e1
95 changed files with 10569 additions and 3728 deletions
@@ -11,6 +11,8 @@ import (
"sync"
"sync/atomic"
"time"
"github.com/Rain-kl/Wavelet/internal/model/analytics"
)
// FlushFunc persists a batch of queued items. It is invoked from the worker goroutine.
@@ -22,14 +24,8 @@ type FlushFunc[T any] func(ctx context.Context, items []T) error
type FlushErrorHandler[T any] func(ctx context.Context, items []T, err error)
// Stats is a point-in-time snapshot of Writer queue and failure counters.
type Stats struct {
Name string `json:"name"`
Depth int `json:"depth"`
Cap int `json:"cap"`
Drops int64 `json:"drops"`
FlushErrors int64 `json:"flush_errors"`
Running bool `json:"running"`
}
// It is an alias of analyticsmodel.BatchWriterStats (moved to keep model pure data).
type Stats = analytics.BatchWriterStats
// Writer buffers items and flushes them by size or interval.
type Writer[T any] struct {
@@ -0,0 +1,115 @@
-- +goose Up
-- 节点访问日志:按月 RANGE 分区,复合主键 (id, logged_at) 满足分区键进唯一索引要求。
CREATE TABLE IF NOT EXISTS of_node_access_logs (
id BIGINT NOT NULL,
node_id VARCHAR(64) NOT NULL DEFAULT '',
logged_at TIMESTAMPTZ NOT NULL,
remote_addr VARCHAR(128) NOT NULL DEFAULT '',
region VARCHAR(128) NOT NULL DEFAULT '',
host VARCHAR(255) NOT NULL DEFAULT '',
path VARCHAR(2048) NOT NULL DEFAULT '',
user_agent TEXT NOT NULL DEFAULT '',
cache_status VARCHAR(64) NOT NULL DEFAULT '',
status_code INTEGER NOT NULL DEFAULT 0,
bytes_sent BIGINT NOT NULL DEFAULT 0,
request_length BIGINT NOT NULL DEFAULT 0,
request_time_ms INTEGER NOT NULL DEFAULT 0,
created_at TIMESTAMPTZ NOT NULL DEFAULT CURRENT_TIMESTAMP,
PRIMARY KEY (id, logged_at)
) PARTITION BY RANGE (logged_at);
CREATE INDEX IF NOT EXISTS idx_of_node_access_logs_node_id ON of_node_access_logs (node_id, logged_at DESC);
CREATE INDEX IF NOT EXISTS idx_of_node_access_logs_host ON of_node_access_logs (host, logged_at DESC);
CREATE INDEX IF NOT EXISTS idx_of_node_access_logs_remote_addr ON of_node_access_logs (remote_addr, logged_at DESC);
CREATE INDEX IF NOT EXISTS idx_of_node_access_logs_status_code ON of_node_access_logs (status_code, logged_at DESC);
-- 用户访问日志:按月分区。
CREATE TABLE IF NOT EXISTS w_user_access_logs (
id BIGINT NOT NULL,
user_id BIGINT NOT NULL DEFAULT 0,
path VARCHAR(2048) NOT NULL DEFAULT '',
method VARCHAR(16) NOT NULL DEFAULT '',
ip VARCHAR(128) NOT NULL DEFAULT '',
user_agent TEXT NOT NULL DEFAULT '',
headers TEXT NOT NULL DEFAULT '',
status INTEGER NOT NULL DEFAULT 0,
latency BIGINT NOT NULL DEFAULT 0,
created_at TIMESTAMPTZ NOT NULL DEFAULT CURRENT_TIMESTAMP,
PRIMARY KEY (id, created_at)
) PARTITION BY RANGE (created_at);
CREATE INDEX IF NOT EXISTS idx_w_user_access_logs_user_id ON w_user_access_logs (user_id, created_at DESC);
-- 可观测 4 表:普通表 + (node_id, captured_at DESC) 索引。
CREATE TABLE IF NOT EXISTS of_node_metric_snapshots (
id BIGINT NOT NULL PRIMARY KEY,
node_id VARCHAR(64) NOT NULL DEFAULT '',
captured_at TIMESTAMPTZ NOT NULL,
cpu_usage_percent DOUBLE PRECISION NOT NULL DEFAULT 0,
memory_used_bytes BIGINT NOT NULL DEFAULT 0,
memory_total_bytes BIGINT NOT NULL DEFAULT 0,
storage_used_bytes BIGINT NOT NULL DEFAULT 0,
storage_total_bytes BIGINT NOT NULL DEFAULT 0,
disk_read_bytes BIGINT NOT NULL DEFAULT 0,
disk_write_bytes BIGINT NOT NULL DEFAULT 0,
network_rx_bytes BIGINT NOT NULL DEFAULT 0,
network_tx_bytes BIGINT NOT NULL DEFAULT 0,
created_at TIMESTAMPTZ NOT NULL DEFAULT CURRENT_TIMESTAMP
);
CREATE INDEX IF NOT EXISTS idx_of_node_metric_snapshots_node ON of_node_metric_snapshots (node_id, captured_at DESC);
CREATE TABLE IF NOT EXISTS of_node_edge_health (
id BIGINT NOT NULL PRIMARY KEY,
node_id VARCHAR(64) NOT NULL DEFAULT '',
captured_at TIMESTAMPTZ NOT NULL,
status VARCHAR(64) NOT NULL DEFAULT '',
connections BIGINT NOT NULL DEFAULT 0,
created_at TIMESTAMPTZ NOT NULL DEFAULT CURRENT_TIMESTAMP
);
CREATE INDEX IF NOT EXISTS idx_of_node_edge_health_node ON of_node_edge_health (node_id, captured_at DESC);
CREATE TABLE IF NOT EXISTS of_node_obs_frps (
id BIGINT NOT NULL PRIMARY KEY,
node_id VARCHAR(64) NOT NULL DEFAULT '',
captured_at TIMESTAMPTZ NOT NULL,
frps_connections INTEGER NOT NULL DEFAULT 0,
frps_proxy_count INTEGER NOT NULL DEFAULT 0,
frps_client_count INTEGER NOT NULL DEFAULT 0,
frps_proxies TEXT NOT NULL DEFAULT '',
created_at TIMESTAMPTZ NOT NULL DEFAULT CURRENT_TIMESTAMP
);
CREATE INDEX IF NOT EXISTS idx_of_node_obs_frps_node ON of_node_obs_frps (node_id, captured_at DESC);
CREATE TABLE IF NOT EXISTS of_node_obs_frpc (
id BIGINT NOT NULL PRIMARY KEY,
node_id VARCHAR(64) NOT NULL DEFAULT '',
captured_at TIMESTAMPTZ NOT NULL,
tunnel_status VARCHAR(16) NOT NULL DEFAULT '',
connected_relays_count INTEGER NOT NULL DEFAULT 0,
created_at TIMESTAMPTZ NOT NULL DEFAULT CURRENT_TIMESTAMP
);
CREATE INDEX IF NOT EXISTS idx_of_node_obs_frpc_node ON of_node_obs_frpc (node_id, captured_at DESC);
-- 分区预建:创建当月及未来 2 个月分区(共 3 个月)。
-- +goose StatementBegin
DO $$
DECLARE
d date;
BEGIN
FOR d IN SELECT generate_series(date_trunc('month', now())::date, (date_trunc('month', now()) + interval '2 months')::date, interval '1 month')::date
LOOP
EXECUTE format('CREATE TABLE IF NOT EXISTS of_node_access_logs_%s PARTITION OF of_node_access_logs FOR VALUES FROM (%L) TO (%L)',
to_char(d, 'YYYYMM'), d, d + interval '1 month');
EXECUTE format('CREATE TABLE IF NOT EXISTS w_user_access_logs_%s PARTITION OF w_user_access_logs FOR VALUES FROM (%L) TO (%L)',
to_char(d, 'YYYYMM'), d, d + interval '1 month');
END LOOP;
END $$;
-- +goose StatementEnd
-- +goose Down
DROP TABLE IF EXISTS w_user_access_logs;
DROP TABLE IF EXISTS of_node_access_logs;
DROP TABLE IF EXISTS of_node_metric_snapshots;
DROP TABLE IF EXISTS of_node_edge_health;
DROP TABLE IF EXISTS of_node_obs_frps;
DROP TABLE IF EXISTS of_node_obs_frpc;
@@ -0,0 +1,19 @@
-- +goose Up
-- 日志保留天数配置(business),替换旧的 database_auto_cleanup_* 键。
INSERT INTO w_system_configs (key, value, type, visibility, description, created_at, updated_at)
VALUES
('log_retention_days_postgres', '90', 'business', 0, 'PostgreSQL 日志保留天数(访问日志与可观测统一)', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP),
('log_retention_days_sqlite', '90', 'business', 0, 'SQLite 日志保留天数', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP),
('log_retention_days_clickhouse','90', 'business', 0, 'ClickHouse 日志保留天数', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP)
ON CONFLICT (key) DO NOTHING;
DELETE FROM w_system_configs WHERE key IN ('database_auto_cleanup_enabled', 'database_auto_cleanup_retention_days');
-- +goose Down
INSERT INTO w_system_configs (key, value, type, visibility, description, created_at, updated_at)
VALUES
('database_auto_cleanup_enabled', 'true', 'business', 0, '数据库自动清理开关', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP),
('database_auto_cleanup_retention_days', '30', 'business', 0, '数据库保留天数', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP)
ON CONFLICT (key) DO NOTHING;
DELETE FROM w_system_configs WHERE key IN ('log_retention_days_postgres', 'log_retention_days_sqlite', 'log_retention_days_clickhouse');
@@ -0,0 +1,7 @@
-- +goose Up
DELETE FROM w_schedules WHERE task_type = 'of_database_auto_cleanup';
-- +goose Down
INSERT INTO w_schedules (id, name, task_type, cron, payload, is_active, created_at, updated_at)
VALUES (102, 'OpenFlare 可观测数据自动清理', 'of_database_auto_cleanup', '0 3 * * *', '{}', TRUE, CURRENT_TIMESTAMP, CURRENT_TIMESTAMP)
ON CONFLICT (id) DO NOTHING;
@@ -0,0 +1,93 @@
-- +goose Up
-- 节点访问日志:普通表(同 PG 语义,索引名保持一致)。
CREATE TABLE IF NOT EXISTS of_node_access_logs (
id INTEGER PRIMARY KEY AUTOINCREMENT,
node_id TEXT NOT NULL DEFAULT '',
logged_at DATETIME NOT NULL,
remote_addr TEXT NOT NULL DEFAULT '',
region TEXT NOT NULL DEFAULT '',
host TEXT NOT NULL DEFAULT '',
path TEXT NOT NULL DEFAULT '',
user_agent TEXT NOT NULL DEFAULT '',
cache_status TEXT NOT NULL DEFAULT '',
status_code INTEGER NOT NULL DEFAULT 0,
bytes_sent INTEGER NOT NULL DEFAULT 0,
request_length INTEGER NOT NULL DEFAULT 0,
request_time_ms INTEGER NOT NULL DEFAULT 0,
created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP
);
CREATE INDEX IF NOT EXISTS idx_of_node_access_logs_node_id ON of_node_access_logs (node_id, logged_at DESC);
CREATE INDEX IF NOT EXISTS idx_of_node_access_logs_host ON of_node_access_logs (host, logged_at DESC);
CREATE INDEX IF NOT EXISTS idx_of_node_access_logs_remote_addr ON of_node_access_logs (remote_addr, logged_at DESC);
CREATE INDEX IF NOT EXISTS idx_of_node_access_logs_status_code ON of_node_access_logs (status_code, logged_at DESC);
CREATE TABLE IF NOT EXISTS w_user_access_logs (
id INTEGER PRIMARY KEY AUTOINCREMENT,
user_id INTEGER NOT NULL DEFAULT 0,
path TEXT NOT NULL DEFAULT '',
method TEXT NOT NULL DEFAULT '',
ip TEXT NOT NULL DEFAULT '',
user_agent TEXT NOT NULL DEFAULT '',
headers TEXT NOT NULL DEFAULT '',
status INTEGER NOT NULL DEFAULT 0,
latency INTEGER NOT NULL DEFAULT 0,
created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP
);
CREATE INDEX IF NOT EXISTS idx_w_user_access_logs_user_id ON w_user_access_logs (user_id, created_at DESC);
CREATE TABLE IF NOT EXISTS of_node_metric_snapshots (
id INTEGER PRIMARY KEY AUTOINCREMENT,
node_id TEXT NOT NULL DEFAULT '',
captured_at DATETIME NOT NULL,
cpu_usage_percent REAL NOT NULL DEFAULT 0,
memory_used_bytes INTEGER NOT NULL DEFAULT 0,
memory_total_bytes INTEGER NOT NULL DEFAULT 0,
storage_used_bytes INTEGER NOT NULL DEFAULT 0,
storage_total_bytes INTEGER NOT NULL DEFAULT 0,
disk_read_bytes INTEGER NOT NULL DEFAULT 0,
disk_write_bytes INTEGER NOT NULL DEFAULT 0,
network_rx_bytes INTEGER NOT NULL DEFAULT 0,
network_tx_bytes INTEGER NOT NULL DEFAULT 0,
created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP
);
CREATE INDEX IF NOT EXISTS idx_of_node_metric_snapshots_node ON of_node_metric_snapshots (node_id, captured_at DESC);
CREATE TABLE IF NOT EXISTS of_node_edge_health (
id INTEGER PRIMARY KEY AUTOINCREMENT,
node_id TEXT NOT NULL DEFAULT '',
captured_at DATETIME NOT NULL,
status TEXT NOT NULL DEFAULT '',
connections INTEGER NOT NULL DEFAULT 0,
created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP
);
CREATE INDEX IF NOT EXISTS idx_of_node_edge_health_node ON of_node_edge_health (node_id, captured_at DESC);
CREATE TABLE IF NOT EXISTS of_node_obs_frps (
id INTEGER PRIMARY KEY AUTOINCREMENT,
node_id TEXT NOT NULL DEFAULT '',
captured_at DATETIME NOT NULL,
frps_connections INTEGER NOT NULL DEFAULT 0,
frps_proxy_count INTEGER NOT NULL DEFAULT 0,
frps_client_count INTEGER NOT NULL DEFAULT 0,
frps_proxies TEXT NOT NULL DEFAULT '',
created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP
);
CREATE INDEX IF NOT EXISTS idx_of_node_obs_frps_node ON of_node_obs_frps (node_id, captured_at DESC);
CREATE TABLE IF NOT EXISTS of_node_obs_frpc (
id INTEGER PRIMARY KEY AUTOINCREMENT,
node_id TEXT NOT NULL DEFAULT '',
captured_at DATETIME NOT NULL,
tunnel_status TEXT NOT NULL DEFAULT '',
connected_relays_count INTEGER NOT NULL DEFAULT 0,
created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP
);
CREATE INDEX IF NOT EXISTS idx_of_node_obs_frpc_node ON of_node_obs_frpc (node_id, captured_at DESC);
-- +goose Down
DROP TABLE IF EXISTS of_node_obs_frpc;
DROP TABLE IF EXISTS of_node_obs_frps;
DROP TABLE IF EXISTS of_node_edge_health;
DROP TABLE IF EXISTS of_node_metric_snapshots;
DROP TABLE IF EXISTS w_user_access_logs;
DROP TABLE IF EXISTS of_node_access_logs;
@@ -0,0 +1,17 @@
-- +goose Up
-- 日志保留天数配置(business),替换旧的 database_auto_cleanup_* 键。
INSERT OR IGNORE INTO w_system_configs (key, value, type, visibility, description, created_at, updated_at)
VALUES
('log_retention_days_postgres', '90', 'business', 0, 'PostgreSQL 日志保留天数(访问日志与可观测统一)', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP),
('log_retention_days_sqlite', '90', 'business', 0, 'SQLite 日志保留天数', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP),
('log_retention_days_clickhouse','90', 'business', 0, 'ClickHouse 日志保留天数', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP);
DELETE FROM w_system_configs WHERE key IN ('database_auto_cleanup_enabled', 'database_auto_cleanup_retention_days');
-- +goose Down
INSERT OR IGNORE INTO w_system_configs (key, value, type, visibility, description, created_at, updated_at)
VALUES
('database_auto_cleanup_enabled', 'true', 'business', 0, '数据库自动清理开关', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP),
('database_auto_cleanup_retention_days', '30', 'business', 0, '数据库保留天数', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP);
DELETE FROM w_system_configs WHERE key IN ('log_retention_days_postgres', 'log_retention_days_sqlite', 'log_retention_days_clickhouse');
@@ -0,0 +1,7 @@
-- +goose Up
DELETE FROM w_schedules WHERE task_type = 'of_database_auto_cleanup';
-- +goose Down
INSERT INTO w_schedules (id, name, task_type, cron, payload, is_active, created_at, updated_at)
VALUES (102, 'OpenFlare 可观测数据自动清理', 'of_database_auto_cleanup', '0 3 * * *', '{}', 1, CURRENT_TIMESTAMP, CURRENT_TIMESTAMP)
ON CONFLICT (id) DO NOTHING;
@@ -21,8 +21,10 @@ import (
// expectedMigratedSystemConfigCount 包含初始 32 项系统配置、202606220004
// 从 of_options 迁移过来的 48 项业务配置、Pages 的 2 项业务配置、
// OpenResty 默认限流的 3 项业务配置,以及单 IP 请求频率限制 1 项业务配置。
const expectedMigratedSystemConfigCount = 86
// OpenResty 默认限流的 3 项业务配置、单 IP 请求频率限制 1 项业务配置、
// 源站错误页选项 4 项(202608060001 的 3 项 + 202608060002 的 1 项),
// 以及 202608080002 新增 3 项日志保留天数业务配置并删除 2 项旧清理配置(净 +1)。
const expectedMigratedSystemConfigCount = 91
func TestMigrateInitializesSQLiteDatabase(t *testing.T) {
sqliteDB, err := gorm.Open(sqlite.Open(":memory:"), &gorm.Config{