Add web auth, infinite-scroll, agent admission and review fixes across agent/server/web
This commit is contained in:
@@ -2,6 +2,7 @@ package scheduler
|
||||
|
||||
import (
|
||||
"context"
|
||||
"reflect"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
@@ -33,82 +34,316 @@ func (e Event) WireEvent() Event {
|
||||
}
|
||||
|
||||
type Hooks struct {
|
||||
OnSkipped func(checkID string)
|
||||
OnResult func(ev Event)
|
||||
OnSkipped func(checkID string)
|
||||
OnResult func(ev Event)
|
||||
OnEventChannelFull func()
|
||||
OnResourceLimitFailure func(checkID string, resource string)
|
||||
OnStopped func(checkID string)
|
||||
}
|
||||
|
||||
// Run starts one goroutine per check. Emits events to out. Blocks until ctx is done.
|
||||
func Run(ctx context.Context, agentID string, checks []config.CheckConfig, out chan<- Event, h Hooks) {
|
||||
var wg sync.WaitGroup
|
||||
for i := range checks {
|
||||
c := checks[i]
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
runCheck(ctx, agentID, c, out, h)
|
||||
}()
|
||||
// Manager owns per-check workers and applies hot config updates.
|
||||
type Manager struct {
|
||||
ctx context.Context
|
||||
agentID string
|
||||
out chan<- Event
|
||||
hooks Hooks
|
||||
|
||||
mu sync.Mutex
|
||||
workers map[string]*checkWorker
|
||||
draining map[string]struct{}
|
||||
pending map[string]config.CheckConfig
|
||||
all []*checkWorker
|
||||
stopped bool
|
||||
}
|
||||
|
||||
// NewManager creates a scheduler manager bound to the parent context.
|
||||
func NewManager(ctx context.Context, agentID string, out chan<- Event, h Hooks) *Manager {
|
||||
return &Manager{
|
||||
ctx: ctx,
|
||||
agentID: agentID,
|
||||
out: out,
|
||||
hooks: h,
|
||||
workers: make(map[string]*checkWorker),
|
||||
draining: make(map[string]struct{}),
|
||||
pending: make(map[string]config.CheckConfig),
|
||||
}
|
||||
wg.Wait()
|
||||
}
|
||||
|
||||
func runCheck(ctx context.Context, agentID string, c config.CheckConfig, out chan<- Event, h Hooks) {
|
||||
t := time.NewTicker(c.Interval.Duration)
|
||||
defer t.Stop()
|
||||
var running bool
|
||||
var mu sync.Mutex
|
||||
// Update applies the current check set without killing in-flight runs.
|
||||
func (m *Manager) Update(checks []config.CheckConfig) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
if m.stopped {
|
||||
return
|
||||
}
|
||||
next := make(map[string]config.CheckConfig, len(checks))
|
||||
for _, c := range checks {
|
||||
next[c.ID] = c
|
||||
if w, ok := m.workers[c.ID]; ok {
|
||||
if _, draining := m.draining[c.ID]; draining {
|
||||
m.pending[c.ID] = c
|
||||
} else {
|
||||
w.Update(c)
|
||||
}
|
||||
continue
|
||||
}
|
||||
id := c.ID
|
||||
w := newCheckWorker(m.ctx, m.agentID, c, m.out, m.hooks, func(stopped *checkWorker) {
|
||||
m.workerStopped(id, stopped)
|
||||
})
|
||||
m.workers[c.ID] = w
|
||||
m.all = append(m.all, w)
|
||||
}
|
||||
for id, w := range m.workers {
|
||||
if _, ok := next[id]; ok {
|
||||
continue
|
||||
}
|
||||
m.draining[id] = struct{}{}
|
||||
delete(m.pending, id)
|
||||
w.Stop()
|
||||
}
|
||||
}
|
||||
|
||||
exec := func() {
|
||||
mu.Lock()
|
||||
// Stop prevents new runs and lets in-flight checks finish or observe ctx cancel.
|
||||
func (m *Manager) Stop() {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
if m.stopped {
|
||||
return
|
||||
}
|
||||
m.stopped = true
|
||||
for _, w := range m.workers {
|
||||
w.Stop()
|
||||
}
|
||||
}
|
||||
|
||||
// Wait blocks until all workers have drained.
|
||||
func (m *Manager) Wait() {
|
||||
m.mu.Lock()
|
||||
workers := append([]*checkWorker(nil), m.all...)
|
||||
m.mu.Unlock()
|
||||
for _, w := range workers {
|
||||
w.Wait()
|
||||
}
|
||||
}
|
||||
|
||||
func (m *Manager) workerStopped(checkID string, w *checkWorker) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
if current, ok := m.workers[checkID]; !ok || current != w {
|
||||
return
|
||||
}
|
||||
delete(m.workers, checkID)
|
||||
delete(m.draining, checkID)
|
||||
if m.stopped {
|
||||
delete(m.pending, checkID)
|
||||
return
|
||||
}
|
||||
if cfg, ok := m.pending[checkID]; ok {
|
||||
delete(m.pending, checkID)
|
||||
next := newCheckWorker(m.ctx, m.agentID, cfg, m.out, m.hooks, func(stopped *checkWorker) {
|
||||
m.workerStopped(checkID, stopped)
|
||||
})
|
||||
m.workers[checkID] = next
|
||||
m.all = append(m.all, next)
|
||||
}
|
||||
}
|
||||
|
||||
type checkWorker struct {
|
||||
ctx context.Context
|
||||
agentID string
|
||||
out chan<- Event
|
||||
hooks Hooks
|
||||
onDone func(*checkWorker)
|
||||
|
||||
updateCh chan config.CheckConfig
|
||||
stopCh chan struct{}
|
||||
doneCh chan struct{}
|
||||
stopOnce sync.Once
|
||||
wg sync.WaitGroup
|
||||
}
|
||||
|
||||
func newCheckWorker(ctx context.Context, agentID string, c config.CheckConfig, out chan<- Event, h Hooks, onDone func(*checkWorker)) *checkWorker {
|
||||
w := &checkWorker{
|
||||
ctx: ctx,
|
||||
agentID: agentID,
|
||||
out: out,
|
||||
hooks: h,
|
||||
onDone: onDone,
|
||||
updateCh: make(chan config.CheckConfig, 1),
|
||||
stopCh: make(chan struct{}),
|
||||
doneCh: make(chan struct{}, 1),
|
||||
}
|
||||
w.wg.Add(1)
|
||||
go w.loop(c)
|
||||
return w
|
||||
}
|
||||
|
||||
func (w *checkWorker) Update(c config.CheckConfig) {
|
||||
select {
|
||||
case w.updateCh <- c:
|
||||
default:
|
||||
select {
|
||||
case <-w.updateCh:
|
||||
default:
|
||||
}
|
||||
w.updateCh <- c
|
||||
}
|
||||
}
|
||||
|
||||
func (w *checkWorker) Stop() {
|
||||
w.stopOnce.Do(func() { close(w.stopCh) })
|
||||
}
|
||||
|
||||
func (w *checkWorker) Wait() {
|
||||
w.wg.Wait()
|
||||
}
|
||||
|
||||
func (w *checkWorker) loop(initial config.CheckConfig) {
|
||||
defer func() {
|
||||
if w.onDone != nil {
|
||||
w.onDone(w)
|
||||
}
|
||||
w.wg.Done()
|
||||
}()
|
||||
cfg := initial
|
||||
t := time.NewTicker(cfg.Interval.Duration)
|
||||
defer t.Stop()
|
||||
|
||||
running := false
|
||||
stopping := false
|
||||
runAfterCurrent := false
|
||||
ctxDone := w.ctx.Done()
|
||||
stopCh := w.stopCh
|
||||
|
||||
start := func(c config.CheckConfig) {
|
||||
if running {
|
||||
mu.Unlock()
|
||||
if h.OnSkipped != nil {
|
||||
h.OnSkipped(c.ID)
|
||||
if w.hooks.OnSkipped != nil {
|
||||
w.hooks.OnSkipped(c.ID)
|
||||
}
|
||||
return
|
||||
}
|
||||
running = true
|
||||
mu.Unlock()
|
||||
defer func() {
|
||||
mu.Lock()
|
||||
running = false
|
||||
mu.Unlock()
|
||||
go func() {
|
||||
w.run(c)
|
||||
w.doneCh <- struct{}{}
|
||||
}()
|
||||
|
||||
res := runner.Run(ctx, c.Argv(), c.Timeout.Duration)
|
||||
id, err := eventid.New()
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
ev := Event{
|
||||
EventID: id,
|
||||
CheckID: c.ID,
|
||||
ObservedAt: time.Now().UTC(),
|
||||
Status: string(res.Status),
|
||||
ExitCode: res.ExitCode,
|
||||
DurationMs: res.DurationMs,
|
||||
Output: res.Output,
|
||||
OutputTruncated: res.OutputTruncated,
|
||||
NotificationsEnabled: boolPtr(c.NotificationsOn()),
|
||||
IncidentKey: incidentKey(agentID, c),
|
||||
}
|
||||
if h.OnResult != nil {
|
||||
h.OnResult(ev)
|
||||
}
|
||||
select {
|
||||
case out <- ev:
|
||||
case <-ctx.Done():
|
||||
}
|
||||
}
|
||||
|
||||
// Fire first run immediately so smoke tests don't wait a full interval.
|
||||
go exec()
|
||||
start(cfg)
|
||||
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-ctxDone:
|
||||
stopping = true
|
||||
ctxDone = nil
|
||||
if !running {
|
||||
if w.hooks.OnStopped != nil {
|
||||
w.hooks.OnStopped(cfg.ID)
|
||||
}
|
||||
return
|
||||
}
|
||||
case <-stopCh:
|
||||
stopping = true
|
||||
stopCh = nil
|
||||
if !running {
|
||||
if w.hooks.OnStopped != nil {
|
||||
w.hooks.OnStopped(cfg.ID)
|
||||
}
|
||||
return
|
||||
}
|
||||
case next := <-w.updateCh:
|
||||
if reflect.DeepEqual(cfg, next) {
|
||||
continue
|
||||
}
|
||||
cfg = next
|
||||
t.Reset(cfg.Interval.Duration)
|
||||
if running {
|
||||
runAfterCurrent = true
|
||||
} else {
|
||||
start(cfg)
|
||||
}
|
||||
case <-w.doneCh:
|
||||
running = false
|
||||
if !stopping {
|
||||
select {
|
||||
case <-stopCh:
|
||||
stopping = true
|
||||
stopCh = nil
|
||||
default:
|
||||
}
|
||||
}
|
||||
if stopping {
|
||||
if w.hooks.OnStopped != nil {
|
||||
w.hooks.OnStopped(cfg.ID)
|
||||
}
|
||||
return
|
||||
}
|
||||
if runAfterCurrent {
|
||||
runAfterCurrent = false
|
||||
for {
|
||||
select {
|
||||
case next := <-w.updateCh:
|
||||
if !reflect.DeepEqual(cfg, next) {
|
||||
cfg = next
|
||||
t.Reset(cfg.Interval.Duration)
|
||||
}
|
||||
default:
|
||||
start(cfg)
|
||||
goto nextLoop
|
||||
}
|
||||
}
|
||||
nextLoop:
|
||||
continue
|
||||
}
|
||||
case <-t.C:
|
||||
go exec()
|
||||
if stopping {
|
||||
continue
|
||||
}
|
||||
start(cfg)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (w *checkWorker) run(c config.CheckConfig) {
|
||||
res := runner.RunCommand(w.ctx, c.Command, c.Timeout.Duration, runner.ResourceLimits{
|
||||
CPUTime: c.ResourceLimits.CPUTime.Duration,
|
||||
MemoryBytes: c.ResourceLimits.Memory.Bytes,
|
||||
OpenFiles: c.ResourceLimits.OpenFiles,
|
||||
})
|
||||
if res.ResourceLimitFailure != "" && w.hooks.OnResourceLimitFailure != nil {
|
||||
w.hooks.OnResourceLimitFailure(c.ID, res.ResourceLimitFailure)
|
||||
}
|
||||
id, err := eventid.New()
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
ev := Event{
|
||||
EventID: id,
|
||||
CheckID: c.ID,
|
||||
ObservedAt: time.Now().UTC(),
|
||||
Status: string(res.Status),
|
||||
ExitCode: res.ExitCode,
|
||||
DurationMs: res.DurationMs,
|
||||
Output: res.Output,
|
||||
OutputTruncated: res.OutputTruncated,
|
||||
NotificationsEnabled: boolPtr(c.NotificationsOn()),
|
||||
IncidentKey: incidentKey(w.agentID, c),
|
||||
}
|
||||
if w.hooks.OnResult != nil {
|
||||
w.hooks.OnResult(ev)
|
||||
}
|
||||
select {
|
||||
case w.out <- ev:
|
||||
case <-w.ctx.Done():
|
||||
default:
|
||||
if w.hooks.OnEventChannelFull != nil {
|
||||
w.hooks.OnEventChannelFull()
|
||||
}
|
||||
select {
|
||||
case w.out <- ev:
|
||||
case <-w.ctx.Done():
|
||||
case <-w.stopCh:
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user