fix(observability): fall back to raw hourly when rollup is incomplete

Materialized capacity/openresty hourly tables only hold data after the MV
exists. Preferring any non-empty rollup hid full raw history and left 24h
charts with only recent hours. Use rollup only when its earliest bucket
covers the query window start.
This commit is contained in:
ryan
2026-07-10 11:20:18 +08:00
parent bbadcca294
commit 4b11279662
4 changed files with 134 additions and 46 deletions
+56 -33
View File
@@ -1,3 +1,7 @@
// Package main is a manual smoke tool for ClickHouse app write path.
// Usage (from repo root, with config.yaml and Docker CH up):
//
// go run ./scripts/live_ch_smoke
package main
import (
@@ -11,13 +15,27 @@ import (
"github.com/Rain-kl/Wavelet/internal/model"
)
const (
flushWaitTimeout = 45 * time.Second
listLimit = 5
pollInterval = 2 * time.Second
)
func main() {
if !db.ChConnReady() {
fmt.Fprintln(os.Stderr, "ChConn not ready — check config.yaml clickhouse.enabled")
if err := run(); err != nil {
fmt.Fprintln(os.Stderr, err)
os.Exit(1)
}
}
func run() error {
if !db.ChConnReady() {
return fmt.Errorf("ChConn not ready — check config.yaml clickhouse.enabled")
}
ctx := context.Background()
chwriter.Init(ctx)
defer func() { _ = chwriter.Stop(ctx) }()
now := time.Now().UTC()
nodeID := "e2e-app-" + now.Format("150405")
if err := model.InsertOpenFlareMetricSnapshot(ctx, &model.OpenFlareMetricSnapshot{
@@ -26,44 +44,49 @@ func main() {
StorageUsedBytes: 456, StorageTotalBytes: 2000,
DiskReadBytes: 11, DiskWriteBytes: 22, NetworkRxBytes: 33, NetworkTxBytes: 44,
}); err != nil {
fmt.Fprintln(os.Stderr, "insert:", err)
os.Exit(1)
return fmt.Errorf("insert: %w", err)
}
fmt.Println("queued", nodeID)
deadline := time.Now().Add(45 * time.Second)
if err := waitForSnapshot(ctx, nodeID, now); err != nil {
return err
}
if err := assertLatestIncludes(ctx, nodeID, now); err != nil {
return err
}
for _, s := range chwriter.WriterStats() {
fmt.Printf("writer %s running=%v depth=%d drops=%d flush_err=%d\n",
s.Name, s.Running, s.Depth, s.Drops, s.FlushErrors)
}
return nil
}
func waitForSnapshot(ctx context.Context, nodeID string, now time.Time) error {
deadline := time.Now().Add(flushWaitTimeout)
for time.Now().Before(deadline) {
rows, err := model.ListOpenFlareMetricSnapshotsSince(ctx, nodeID, now.Add(-time.Minute), 5)
rows, err := model.ListOpenFlareMetricSnapshotsSince(ctx, nodeID, now.Add(-time.Minute), listLimit)
if err != nil {
fmt.Fprintln(os.Stderr, "list:", err)
os.Exit(1)
return fmt.Errorf("list: %w", err)
}
if len(rows) > 0 {
fmt.Printf("OK flushed id=%d cpu=%.1f\n", rows[0].ID, rows[0].CPUUsagePercent)
latest, err := model.ListOpenFlareLatestMetricSnapshotsSince(ctx, "", now.Add(-time.Hour))
if err != nil {
fmt.Fprintln(os.Stderr, "latest:", err)
os.Exit(1)
}
ok := false
for _, r := range latest {
if r != nil && r.NodeID == nodeID {
ok = true
}
}
if !ok {
fmt.Fprintln(os.Stderr, "FAIL latest-per-node missing node")
os.Exit(1)
}
fmt.Println("OK latest-per-node includes node")
for _, s := range chwriter.WriterStats() {
fmt.Printf("writer %s running=%v depth=%d drops=%d flush_err=%d\n",
s.Name, s.Running, s.Depth, s.Drops, s.FlushErrors)
}
_ = chwriter.Stop(ctx)
return
return nil
}
time.Sleep(2 * time.Second)
time.Sleep(pollInterval)
}
fmt.Fprintln(os.Stderr, "FAIL: not flushed within 45s")
os.Exit(1)
return fmt.Errorf("not flushed within timeout")
}
func assertLatestIncludes(ctx context.Context, nodeID string, now time.Time) error {
latest, err := model.ListOpenFlareLatestMetricSnapshotsSince(ctx, "", now.Add(-time.Hour))
if err != nil {
return fmt.Errorf("latest: %w", err)
}
for _, r := range latest {
if r != nil && r.NodeID == nodeID {
fmt.Println("OK latest-per-node includes node")
return nil
}
}
return fmt.Errorf("latest-per-node missing node %s", nodeID)
}