Files
sub2api/backend/internal/service/channel_monitor_v2_aggregator.go
T

420 lines
12 KiB
Go
Raw Normal View History

package service
import (
"context"
"database/sql"
"sync"
"time"
"github.com/Wei-Shaw/sub2api/internal/pkg/logger"
"github.com/google/uuid"
)
const (
channelMonitorV2AggregatorLockKey = "channel-monitor-v2-aggregator"
// Retention walks back to the longest stored tier (1d rollup = 90d). Per-tier
// prune in the repository drops short-lived 1m/user/hist facts earlier.
channelMonitorV2RetentionMax = 90 * 24 * time.Hour
// First tick after upgrade prioritizes the default 90m view (with small padding).
channelMonitorV2BootstrapFirst = 2 * time.Hour
// Always refresh a small trailing window so late writes land without
// re-aggregating large history every tick.
channelMonitorV2RecentOverlap = 10 * time.Minute
// Gentle backfill: small adaptive chunks, never default 24h hammering.
// Initial historical chunk after the 2h seed.
channelMonitorV2BackfillChunkInit = time.Hour
channelMonitorV2MinBackfillChunk = 15 * time.Minute
// Depth-based ceilings (product phases 90m → 1d → 7d → 30d → 90d).
channelMonitorV2MaxChunkNear1d = 2 * time.Hour
channelMonitorV2MaxChunkNear7d = 4 * time.Hour
channelMonitorV2MaxChunkFar = 6 * time.Hour
// Soft adaptive timing: grow only when a recompute is clearly cheap.
channelMonitorV2GrowChunkUnder = 15 * time.Second
channelMonitorV2MaxBackoff = 10 * time.Minute
)
// channelMonitorRuntimeSubscriber is the optional settings hook that lets the
// aggregator wake immediately when channel_monitor_enabled / mode flips.
type channelMonitorRuntimeSubscriber interface {
SubscribeChannelMonitorRuntime(listener func()) (unsubscribe func())
}
type ChannelMonitorV2Aggregator struct {
repo ChannelMonitorV2Repository
db *sql.DB
settings channelMonitorRuntimeReader
instanceID string
stopCh chan struct{}
// kickCh wakes the loop early after a settings change (buffered 1).
kickCh chan struct{}
startOnce sync.Once
stopOnce sync.Once
mu sync.Mutex
// backfillAt is the earliest minute already recomputed (mirrors DB cursor).
// Zero means "not yet loaded from durable watermark this process".
backfillAt time.Time
backfillChunk time.Duration
backfillFailures int
// nextWaitFloor is applied after runOnce when failures require backoff.
nextWaitFloor time.Duration
// cursorLoaded is true after the first successful watermark read (or init).
cursorLoaded bool
// hasAggregated is true once any recompute in this process (or durable data) exists.
hasAggregated bool
unsub func()
ctx context.Context
cancel context.CancelFunc
}
func NewChannelMonitorV2Aggregator(repo ChannelMonitorV2Repository, db *sql.DB, settings channelMonitorRuntimeReader) *ChannelMonitorV2Aggregator {
return &ChannelMonitorV2Aggregator{
repo: repo,
db: db,
settings: settings,
instanceID: uuid.NewString(),
stopCh: make(chan struct{}),
kickCh: make(chan struct{}, 1),
backfillChunk: channelMonitorV2BackfillChunkInit,
}
}
func (s *ChannelMonitorV2Aggregator) Start() {
if s == nil || s.repo == nil {
return
}
s.startOnce.Do(func() {
s.mu.Lock()
s.ctx, s.cancel = context.WithCancel(context.Background())
s.mu.Unlock()
if sub, ok := s.settings.(channelMonitorRuntimeSubscriber); ok && sub != nil {
unsub := sub.SubscribeChannelMonitorRuntime(func() {
s.kick()
})
s.mu.Lock()
stopped := s.ctx == nil
if !stopped {
select {
case <-s.ctx.Done():
stopped = true
default:
}
}
if !stopped {
s.unsub = unsub
}
s.mu.Unlock()
if stopped && unsub != nil {
unsub()
}
}
go s.loop()
})
}
func (s *ChannelMonitorV2Aggregator) Stop() {
if s == nil {
return
}
s.stopOnce.Do(func() {
s.mu.Lock()
cancel := s.cancel
unsub := s.unsub
s.cancel = nil
s.unsub = nil
s.mu.Unlock()
if cancel != nil {
cancel()
}
if unsub != nil {
unsub()
}
close(s.stopCh)
})
}
// kick wakes the aggregation loop so mode flips take effect without waiting
// for the next refresh interval.
func (s *ChannelMonitorV2Aggregator) kick() {
if s == nil {
return
}
select {
case s.kickCh <- struct{}{}:
default:
}
}
func (s *ChannelMonitorV2Aggregator) loop() {
for {
interval := time.Minute
ctx, cancel := context.WithTimeout(context.Background(), 3*time.Second)
if !s.passiveAggregationAllowed(ctx) {
cancel()
if !s.wait(interval) {
return
}
continue
}
if cfg, err := s.repo.GetConfig(ctx); err == nil {
if !cfg.Enabled {
cancel()
if !s.wait(interval) {
return
}
continue
}
if cfg.RefreshIntervalSeconds > 0 {
interval = time.Duration(cfg.RefreshIntervalSeconds) * time.Second
}
}
cancel()
s.runOnce()
// Hard gate: never compress bootstrap to multi-Hz ticks. Soft gate: on
// repeated failures raise the wait floor (exponential backoff).
s.mu.Lock()
floor := s.nextWaitFloor
s.mu.Unlock()
if floor > interval {
interval = floor
}
if !s.wait(interval) {
return
}
}
}
func (s *ChannelMonitorV2Aggregator) passiveAggregationAllowed(ctx context.Context) bool {
if s == nil || s.settings == nil {
// Fail closed without settings: do not aggregate under ambiguous mode.
return false
}
return s.settings.GetChannelMonitorRuntime(ctx).PassiveAggregationAllowed()
}
func (s *ChannelMonitorV2Aggregator) wait(interval time.Duration) bool {
timer := time.NewTimer(interval)
defer timer.Stop()
select {
case <-timer.C:
return true
case <-s.kickCh:
// Drain any coalesced kicks so a burst of settings writes only wakes once.
for {
select {
case <-s.kickCh:
default:
return true
}
}
case <-s.stopCh:
return false
}
}
func (s *ChannelMonitorV2Aggregator) runOnce() {
s.mu.Lock()
parent := s.ctx
s.mu.Unlock()
if parent == nil {
parent = context.Background()
}
ctx, cancel := context.WithTimeout(parent, 55*time.Second)
defer cancel()
release, acquired := tryAcquireSingletonLeaderLock(ctx, nil, s.db, channelMonitorV2AggregatorLockKey, s.instanceID, 2*time.Minute)
if !acquired {
return
}
if release != nil {
defer release()
}
now := time.Now().UTC().Truncate(time.Minute)
if err := s.ensureCursor(ctx, now); err != nil {
logger.LegacyPrintf("service.channel_monitor_v2", "[ChannelMonitorV2] load watermark failed: %v", err)
return
}
s.mu.Lock()
cursor := s.backfillAt
hasData := s.hasAggregated
s.mu.Unlock()
// Phase 1 (first upgrade / empty): seed the default 90m UI window quickly.
if !hasData || cursor.IsZero() {
start := now.Add(-channelMonitorV2BootstrapFirst)
started := time.Now()
if err := s.repo.RecomputeRange(ctx, start, now); err != nil {
logger.LegacyPrintf("service.channel_monitor_v2", "[ChannelMonitorV2] bootstrap recent aggregation failed: %v", err)
s.recordBackfillFailure(now, cursor)
return
}
s.recordBackfillSuccess(start, time.Since(started), now)
return
}
// Always refresh the trailing overlap so late usage/error writes land in 1m facts.
if err := s.repo.RecomputeRange(ctx, now.Add(-channelMonitorV2RecentOverlap), now); err != nil {
logger.LegacyPrintf("service.channel_monitor_v2", "[ChannelMonitorV2] overlap aggregation failed: %v", err)
return
}
// Phase 2: walk history backward at most one chunk per tick until retention max (90d).
// Product UI (30d) fills first; remaining 3090d continues silently.
retentionCutoff := now.Add(-channelMonitorV2RetentionMax)
if !cursor.After(retentionCutoff) {
return
}
end := cursor
s.mu.Lock()
chunk := s.backfillChunk
s.mu.Unlock()
if chunk <= 0 {
chunk = channelMonitorV2BackfillChunkInit
}
maxChunk := channelMonitorV2MaxChunkForDepth(now, end)
if chunk > maxChunk {
chunk = maxChunk
}
if chunk < channelMonitorV2MinBackfillChunk {
chunk = channelMonitorV2MinBackfillChunk
}
start := end.Add(-chunk)
// Once bootstrap reaches historical data, keep chunks on day boundaries so
// daily rollups never depend on 1m rows from two independently pruned chunks.
if end.Before(now.Add(-7 * 24 * time.Hour)) {
aligned := end.Add(-chunk).Truncate(24 * time.Hour)
if aligned.Before(end) {
start = aligned
}
}
if start.Before(retentionCutoff) {
start = retentionCutoff
}
if !start.Before(end) {
return
}
started := time.Now()
if err := s.repo.RecomputeRange(ctx, start, end); err != nil {
logger.LegacyPrintf("service.channel_monitor_v2", "[ChannelMonitorV2] backfill failed %s..%s: %v", start, end, err)
s.recordBackfillFailure(now, end)
return
}
s.recordBackfillSuccess(start, time.Since(started), now)
}
// channelMonitorV2MaxChunkForDepth returns the hard ceiling for a historical
// chunk ending at `end` (earliest already covered / next walk end).
func channelMonitorV2MaxChunkForDepth(now, end time.Time) time.Duration {
age := now.Sub(end)
switch {
case age < 24*time.Hour:
return channelMonitorV2MaxChunkNear1d
case age < 7*24*time.Hour:
return channelMonitorV2MaxChunkNear7d
default:
return channelMonitorV2MaxChunkFar
}
}
func (s *ChannelMonitorV2Aggregator) recordBackfillSuccess(coveredFrom time.Time, elapsed time.Duration, now time.Time) {
s.mu.Lock()
defer s.mu.Unlock()
s.backfillAt = coveredFrom
s.hasAggregated = true
s.backfillFailures = 0
s.nextWaitFloor = 0
maxChunk := channelMonitorV2MaxChunkForDepth(now, coveredFrom)
// Grow slowly only when the recompute was clearly cheap.
if elapsed > 0 && elapsed < channelMonitorV2GrowChunkUnder {
next := s.backfillChunk
if next <= 0 {
next = channelMonitorV2BackfillChunkInit
}
next = time.Duration(float64(next) * 1.5)
if next > maxChunk {
next = maxChunk
}
if next < channelMonitorV2MinBackfillChunk {
next = channelMonitorV2MinBackfillChunk
}
s.backfillChunk = next
return
}
// Keep a healthy chunk within the depth ceiling after success.
if s.backfillChunk <= 0 || s.backfillChunk > maxChunk {
s.backfillChunk = channelMonitorV2BackfillChunkInit
if s.backfillChunk > maxChunk {
s.backfillChunk = maxChunk
}
}
}
func (s *ChannelMonitorV2Aggregator) recordBackfillFailure(now, end time.Time) {
s.mu.Lock()
defer s.mu.Unlock()
s.backfillFailures++
// Shrink chunk toward the minimum so large DBs self-throttle.
if s.backfillChunk <= 0 {
s.backfillChunk = channelMonitorV2BackfillChunkInit
}
s.backfillChunk /= 2
if s.backfillChunk < channelMonitorV2MinBackfillChunk {
s.backfillChunk = channelMonitorV2MinBackfillChunk
}
maxChunk := channelMonitorV2MaxChunkForDepth(now, end)
if s.backfillChunk > maxChunk {
s.backfillChunk = maxChunk
}
// Exponential backoff on wait floor: 1m, 2m, 4m… capped at 10m.
floor := time.Minute << uint(s.backfillFailures-1)
if floor > channelMonitorV2MaxBackoff {
floor = channelMonitorV2MaxBackoff
}
if floor < time.Minute {
floor = time.Minute
}
s.nextWaitFloor = floor
}
// ensureCursor restores durable backfill_cursor after process restart so progress
// and historical walk continue instead of re-seeding only the last 2h.
func (s *ChannelMonitorV2Aggregator) ensureCursor(ctx context.Context, now time.Time) error {
s.mu.Lock()
loaded := s.cursorLoaded
s.mu.Unlock()
if loaded {
return nil
}
wm, err := s.repo.GetAggregationWatermark(ctx)
if err != nil {
return err
}
s.mu.Lock()
defer s.mu.Unlock()
if s.cursorLoaded {
return nil
}
if wm != nil {
if !wm.BackfillCursor.IsZero() {
s.backfillAt = wm.BackfillCursor.UTC().Truncate(time.Minute)
}
if wm.HasData || !wm.DataThrough.IsZero() {
s.hasAggregated = true
// Legacy rows may have data_through but null backfill_cursor (older workers).
// Infer cursor from data_through initial window so we do not re-bootstrap
// only 2h and claim zero progress forever.
if s.backfillAt.IsZero() && !wm.DataThrough.IsZero() {
inferred := wm.DataThrough.UTC().Truncate(time.Minute).Add(-channelMonitorV2BootstrapFirst)
if inferred.After(now) {
inferred = now.Add(-channelMonitorV2BootstrapFirst)
}
s.backfillAt = inferred
}
}
}
s.cursorLoaded = true
return nil
}