feat: publish standalone worker
Separate worker packaging and service lifecycle from the control plane.
Этот коммит содержится в:
534
app/models/monitor.go
Обычный файл
534
app/models/monitor.go
Обычный файл
@@ -0,0 +1,534 @@
|
||||
package models
|
||||
|
||||
import (
|
||||
"log"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/davecgh/go-spew/spew"
|
||||
"github.com/lib/pq"
|
||||
"github.com/pkg/errors"
|
||||
"gorm.io/gorm"
|
||||
"gorm.io/gorm/clause"
|
||||
|
||||
"rsgit.ru/rsmon/rsmon/app/models/concerns"
|
||||
)
|
||||
|
||||
const (
|
||||
stateOK = "OK"
|
||||
stateERR = "ERR"
|
||||
stateWARN = "WARN"
|
||||
stateFail = "FAIL"
|
||||
stateDegraded = "DEGRADED"
|
||||
|
||||
// Event states.
|
||||
stateEnded = "ended"
|
||||
|
||||
// Check column names used in map[string]interface{} GORM updates. Defining
|
||||
// them as constants keeps GORM column references in sync with model fields.
|
||||
colLastStart = "last_start"
|
||||
colLastEnd = "last_end"
|
||||
colState = "state"
|
||||
colWarnings = "warnings"
|
||||
colInfos = "infos"
|
||||
|
||||
// Check kinds. Used to avoid sprinkling magic strings across the codebase.
|
||||
kindHTTP = "http"
|
||||
kindSSL = "ssl"
|
||||
kindSSH = "ssh"
|
||||
kindFTP = "ftp"
|
||||
kindDNS = "dns"
|
||||
kindWhois = "whois"
|
||||
kindRKN = "rkn"
|
||||
kindBSSL = "bssl"
|
||||
kindLLM = "llm"
|
||||
kindLLMHTTP = "llm-http"
|
||||
kindPing = "ping"
|
||||
kindTCP = "tcp"
|
||||
kindUDP = "udp"
|
||||
)
|
||||
|
||||
// Monitor monitor
|
||||
type Monitor struct {
|
||||
concerns.Model
|
||||
// activity status
|
||||
Enabled bool `gorm:"not null;default:true" json:"enabled"`
|
||||
|
||||
// check state, OK - all green, ERR - some checks have failed, UNK - new or not run, FAIL - unable to check
|
||||
State string `gorm:"not null;default:'UNK'" json:"state"`
|
||||
ConfirmState string `gorm:"size:32;not null;default:'none';index" json:"confirm_state"`
|
||||
ConfirmAt *time.Time `json:"confirm_at,omitempty"`
|
||||
ConfirmedByWorkerID *int64 `gorm:"type:bigint REFERENCES worker_nodes(id) ON DELETE SET NULL" json:"confirmed_by_worker_id,omitempty"`
|
||||
|
||||
// Group ID
|
||||
GroupID int64 `gorm:"type:bigint REFERENCES groups(id)" json:"group_id,omitempty" validate:"required"`
|
||||
Group *Group `json:"group,omitempty"`
|
||||
|
||||
// Optional inventory Site join (see docs/plans/inventory-management.md §6.3).
|
||||
// Lets the operator navigate monitor → site → deployments → server in one query.
|
||||
SiteID *int64 `gorm:"type:bigint REFERENCES sites(id) ON DELETE SET NULL;index" json:"site_id,omitempty"`
|
||||
Site *Site `json:"site,omitempty"`
|
||||
|
||||
// Tags for monitor grouping/searching
|
||||
Tags pq.StringArray `gorm:"type:varchar(255)[]" json:"tags"`
|
||||
|
||||
// PreferredRegions is the optional whitelist of region codes a distributed
|
||||
// worker must be in to receive this monitor's checks. Empty/NULL means
|
||||
// "no preference" — any worker can pick it up. Used by Phase 2 of
|
||||
// docs/plans/worker-notifier-mvp.md (regional job routing); see
|
||||
// app/models/check_jobs.go applyRegionRouting.
|
||||
PreferredRegions pq.StringArray `gorm:"type:varchar(64)[]" json:"preferred_regions,omitempty"`
|
||||
|
||||
// RegionMode controls how PreferredRegions is interpreted by the
|
||||
// distributed-worker job router. Defaults to "any" so monitors without
|
||||
// explicit routing still match every worker — backwards-compatible with
|
||||
// Phase 1 deployments.
|
||||
// "any" — no region filter; legacy behavior (default)
|
||||
// "specific" — only workers whose region_code is in PreferredRegions
|
||||
// "all" — Phase 3 placeholder; today behaves like "any". The
|
||||
// multi-region quorum aggregation is not implemented yet,
|
||||
// see docs/todo.md Phase 3.
|
||||
RegionMode string `gorm:"size:16;not null;default:'any'" json:"region_mode" validate:"omitempty,oneof=any specific all"`
|
||||
|
||||
// Monitor name
|
||||
Name *string `json:"name,omitempty"`
|
||||
|
||||
// Host to monitor
|
||||
Host string `json:"host" validate:"required"`
|
||||
|
||||
// UserID specify user for this monitor (info field)
|
||||
UserID *int64 `gorm:"type:bigint REFERENCES users(id)" json:"user_id"`
|
||||
User *User `json:"user,omitempty"`
|
||||
|
||||
// Comment (info field)
|
||||
Comment *string `json:"comment"`
|
||||
|
||||
Checks []Check `json:"checks,omitempty"`
|
||||
DNSRecords []DNSRecord `json:"-"`
|
||||
|
||||
StatsData `gorm:"-:all" sql:"-" json:"stats"`
|
||||
concerns.Timestamped
|
||||
Audited
|
||||
}
|
||||
|
||||
// KINDS Check kinds
|
||||
var KINDS = []string{kindHTTP, kindSSL, kindSSH, kindFTP, kindDNS, kindWhois, kindRKN, kindBSSL, kindLLM, kindLLMHTTP, kindPing, kindTCP, kindUDP}
|
||||
|
||||
// ValidCheckKind is the single canonical allow-list for user supplied check
|
||||
// kinds. Keep it beside the kind constants so every transport validates the
|
||||
// same set before a check reaches a worker.
|
||||
func ValidCheckKind(kind string) bool {
|
||||
for _, candidate := range KINDS {
|
||||
if kind == candidate {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// Region routing mode constants used by Monitor.RegionMode and
|
||||
// app/models/check_jobs.go applyRegionRouting. Centralized so the literal
|
||||
// values are not sprinkled through the codebase.
|
||||
const (
|
||||
// RegionModeAny keeps the legacy behavior: every worker is eligible,
|
||||
// PreferredRegions is ignored. Default for newly-created monitors.
|
||||
RegionModeAny = "any"
|
||||
|
||||
// RegionModeSpecific limits eligible workers to those whose RegionCode
|
||||
// is contained in PreferredRegions. Empty PreferredRegions falls back to
|
||||
// RegionModeAny so the field is safe to leave blank in the UI.
|
||||
RegionModeSpecific = "specific"
|
||||
|
||||
// RegionModeAll is the Phase 3 placeholder: a monitor pinned to all of
|
||||
// its preferred regions for quorum aggregation. Phase 2 treats it as
|
||||
// RegionModeAny and logs a TODO marker so it is easy to grep for.
|
||||
RegionModeAll = "all"
|
||||
)
|
||||
|
||||
// RegionCodesFromSlice is a convenience wrapper so callers (mainly tests
|
||||
// and HTTP handlers) can pass a plain []string and get the pq.StringArray
|
||||
// type the model expects. nil/empty input is preserved as a nil slice so
|
||||
// the GORM column writes a SQL NULL instead of an empty array, matching
|
||||
// the column default.
|
||||
func RegionCodesFromSlice(in []string) pq.StringArray {
|
||||
if len(in) == 0 {
|
||||
return nil
|
||||
}
|
||||
out := make(pq.StringArray, len(in))
|
||||
copy(out, in)
|
||||
return out
|
||||
}
|
||||
|
||||
// Int64ArrayFromSlice mirrors RegionCodesFromSlice for bigint[] columns
|
||||
// such as status_pages.monitor_ids and status_page_maintenance.monitor_ids.
|
||||
// The GORM pq.Int64Array driver expects a non-nil slice for ordered
|
||||
// inserts; callers that always have a non-empty list (the dashboard list
|
||||
// filter, the maintenance form) can rely on this to write a stable shape.
|
||||
func Int64ArrayFromSlice(in []int64) pq.Int64Array {
|
||||
if len(in) == 0 {
|
||||
return pq.Int64Array{}
|
||||
}
|
||||
out := make(pq.Int64Array, len(in))
|
||||
copy(out, in)
|
||||
return out
|
||||
}
|
||||
|
||||
// ValidateRegionMode returns an error when RegionMode is not one of the
|
||||
// documented values ("any", "specific", "all"). Empty strings are treated as
|
||||
// "any" for backwards compatibility with monitors persisted before the field
|
||||
// existed; the DB column also defaults to "any".
|
||||
func (m *Monitor) ValidateRegionMode() error {
|
||||
switch m.RegionMode {
|
||||
case "", RegionModeAny, RegionModeSpecific, RegionModeAll:
|
||||
return nil
|
||||
default:
|
||||
return errors.Errorf("invalid region_mode %q (expected any|specific|all)", m.RegionMode)
|
||||
}
|
||||
}
|
||||
|
||||
// WantsRegion returns true when the monitor should be routed to a worker
|
||||
// operating in the given region code. Callers use this in
|
||||
// app/models/check_jobs.go to filter the eligible worker pool per check.
|
||||
//
|
||||
// - RegionModeAny: always true (no preference).
|
||||
// - RegionModeAll (Phase 3 placeholder): behaves like Any today; returns
|
||||
// true unconditionally so every region sees the check.
|
||||
// - RegionModeSpecific: true when code is contained in PreferredRegions,
|
||||
// or when PreferredRegions is empty (fall-back to Any).
|
||||
func (m *Monitor) WantsRegion(code string) bool {
|
||||
switch m.RegionMode {
|
||||
case RegionModeSpecific:
|
||||
if len(m.PreferredRegions) == 0 {
|
||||
return true
|
||||
}
|
||||
for _, r := range m.PreferredRegions {
|
||||
if r == code {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
case RegionModeAll:
|
||||
// TODO(phase3): enumerate PreferredRegions and emit one assignment
|
||||
// per region so the result aggregator can do quorum. Today we
|
||||
// behave like Any so existing workers keep getting checks.
|
||||
return true
|
||||
default:
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
// GetLabel provides functionality.
|
||||
func (m *Monitor) GetLabel() string {
|
||||
if m.Name != nil {
|
||||
return *m.Name
|
||||
}
|
||||
return m.Host
|
||||
}
|
||||
|
||||
// ProcessChecks provides functionality.
|
||||
func (m *Monitor) ProcessChecks(tx *gorm.DB) error {
|
||||
log.Println("process checks")
|
||||
checks := make([]Check, 0)
|
||||
for _, c := range m.Checks { //nolint:gocritic // range copy is acceptable here
|
||||
log.Println("maybe delete check", c.ID, c.Deleted, c.IsNew)
|
||||
if c.Deleted {
|
||||
if !c.IsNew {
|
||||
log.Println("delete check", c.ID)
|
||||
err := tx.Exec("delete from event_checks where check_id = ?", c.ID).Error
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
// First, find all message IDs for this check
|
||||
var messageIDs []int64
|
||||
err = tx.Model(&Message{}).Where("check_id = ?", c.ID).Pluck("id", &messageIDs).Error
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
// Delete event_messages (join table) first to avoid FK constraint violation
|
||||
if len(messageIDs) > 0 {
|
||||
err = tx.Exec("DELETE FROM event_messages WHERE message_id IN (?)", messageIDs).Error
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
|
||||
// Now delete the messages
|
||||
err = tx.Where("check_id = ?", c.ID).Delete(Message{}).Error
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
err = tx.Where("id = ? AND monitor_id = ?", c.ID, m.ID).Delete(Check{}).Error
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
continue
|
||||
}
|
||||
|
||||
if c.IsNew {
|
||||
c.ID = 0
|
||||
}
|
||||
err := c.ValidateSettings()
|
||||
if err != nil {
|
||||
return errors.Wrap(err, "check validation error")
|
||||
}
|
||||
checks = append(checks, c)
|
||||
}
|
||||
m.Checks = checks
|
||||
return nil
|
||||
}
|
||||
|
||||
// ActiveEvent provides functionality.
|
||||
func (m *Monitor) ActiveEvent() Event {
|
||||
evt := Event{}
|
||||
DB().Where("monitor_id = ? AND state != 'old'", m.ID).First(&evt)
|
||||
if evt.ID != 0 {
|
||||
evt.MonitorID = m.ID
|
||||
t := time.Now()
|
||||
evt.StartTime = &t
|
||||
}
|
||||
return evt
|
||||
}
|
||||
|
||||
var mutex sync.Mutex
|
||||
|
||||
// checkSeverityRank assigns an ordinal to each check state so the monitor
|
||||
// aggregator can pick the highest-severity child deterministically.
|
||||
// Severity order is FAIL > ERR > DEGRADED > WARN > OK — see docs/todo.md
|
||||
// Phase 3 for the rationale (DEGRADED = partial regional failure, sits
|
||||
// between OK and ERR). Unknown states (UNK, empty, ...) rank 0 so any
|
||||
// real check state takes precedence over them.
|
||||
func checkSeverityRank(state string) int {
|
||||
switch state {
|
||||
case stateFail:
|
||||
return 5
|
||||
case stateERR:
|
||||
return 4
|
||||
case stateDegraded:
|
||||
return 3
|
||||
case stateWARN:
|
||||
return 2
|
||||
case stateOK:
|
||||
return 1
|
||||
default:
|
||||
return 0
|
||||
}
|
||||
}
|
||||
|
||||
// UpdateStatusFromChecks updates the monitor status based on its checks.
|
||||
func (m *Monitor) UpdateStatusFromChecks() {
|
||||
mutex.Lock()
|
||||
checks := make([]Check, 0)
|
||||
|
||||
tx := DB().Begin()
|
||||
var locked Monitor
|
||||
if err := tx.Clauses(clause.Locking{Strength: "UPDATE"}).First(&locked, m.ID).Error; err != nil {
|
||||
tx.Rollback()
|
||||
mutex.Unlock()
|
||||
log.Println("UpdateStatusFromChecks lock monitor", err)
|
||||
return
|
||||
}
|
||||
m.State = locked.State
|
||||
|
||||
_ = tx.Model(m).Association("Checks").Find(&checks)
|
||||
prevState := m.State
|
||||
m.State = stateOK
|
||||
|
||||
// Pick the highest-severity enabled check. The previous implementation
|
||||
// inlined three if-statements with non-obvious precedence (a WARN that
|
||||
// appeared AFTER an ERR in the iteration would never downgrade back,
|
||||
// but a FAIL after ERR would silently get clobbered). Using a single
|
||||
// severity rank keeps the rule FAIL > ERR > DEGRADED > WARN > OK
|
||||
// independent of slice ordering — the same rule Phase 3 introduces
|
||||
// for DEGRADED, applied uniformly to the existing states too.
|
||||
bestRank := checkSeverityRank(stateOK)
|
||||
bestState := stateOK
|
||||
hasChecks := false
|
||||
for _, check := range checks { //nolint:gocritic // range copy is acceptable here
|
||||
if check.Enabled == nil || !*check.Enabled {
|
||||
continue
|
||||
}
|
||||
hasChecks = true
|
||||
if r := checkSeverityRank(check.State); r > bestRank {
|
||||
bestRank = r
|
||||
bestState = check.State
|
||||
}
|
||||
}
|
||||
m.State = bestState
|
||||
|
||||
if !hasChecks {
|
||||
m.State = stateWARN
|
||||
}
|
||||
|
||||
if m.State != prevState {
|
||||
err := tx.Model(&m).UpdateColumn("state", m.State).Error
|
||||
if err != nil {
|
||||
tx.Rollback()
|
||||
log.Println("UpdateStatusFromChecks fail update state", err)
|
||||
mutex.Unlock()
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
evt := Event{}
|
||||
if err := tx.Where("monitor_id = ? AND state = ?", m.ID, "current").First(&evt).Error; err != nil && !errors.Is(err, gorm.ErrRecordNotFound) {
|
||||
tx.Rollback()
|
||||
mutex.Unlock()
|
||||
log.Println("UpdateStatusFromChecks active event", err)
|
||||
return
|
||||
}
|
||||
// log.Println("state", m.State, "active event:", evt.ID)
|
||||
|
||||
for _, check := range checks { //nolint:gocritic // range copy is acceptable here
|
||||
if check.State == stateERR || check.State == stateFail {
|
||||
evt.ChecksDown = append(evt.ChecksDown, check.Kind)
|
||||
evt.Checks = append(evt.Checks, check)
|
||||
if check.Error != nil {
|
||||
evt.Reason = *check.Error
|
||||
} else {
|
||||
evt.Reason = "unknown error"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
switch m.State {
|
||||
case stateOK, stateWARN, stateDegraded:
|
||||
// DEGRADED is treated like OK/WARN for the event lifecycle: we do
|
||||
// NOT open a new "current" outage event for a partial regional
|
||||
// failure. Operators see DEGRADED on the check detail page and
|
||||
// the monitor list, but the existing notifier pipeline (down /
|
||||
// restore events) only fires for full ERR/FAIL. A future
|
||||
// improvement can add a separate "degraded" message kind.
|
||||
if evt.ID != 0 {
|
||||
upd := map[string]interface{}{
|
||||
"duration": time.Since(*evt.StartTime).Seconds(),
|
||||
"state": stateEnded,
|
||||
"oks": evt.Oks + 1,
|
||||
}
|
||||
if evt.EndTime == nil {
|
||||
upd["end_time"] = time.Now()
|
||||
}
|
||||
err := tx.Model(&evt).UpdateColumns(upd).Error
|
||||
if err != nil {
|
||||
tx.Rollback()
|
||||
log.Println("UpdateStatusFromChecks fail update to ended", err)
|
||||
mutex.Unlock()
|
||||
return
|
||||
}
|
||||
}
|
||||
case stateERR, stateFail:
|
||||
if evt.ID == 0 {
|
||||
tn := time.Now()
|
||||
evt.StartTime = &tn
|
||||
evt.Duration = 0
|
||||
evt.State = "current"
|
||||
evt.MonitorID = m.ID
|
||||
err := tx.Save(&evt).Error
|
||||
if err != nil {
|
||||
tx.Rollback()
|
||||
spew.Dump(evt)
|
||||
log.Println("UpdateStatusFromChecks fail create", err)
|
||||
mutex.Unlock()
|
||||
return
|
||||
}
|
||||
} else {
|
||||
upd := map[string]interface{}{
|
||||
"end_time": nil,
|
||||
"state": "current",
|
||||
"errors": evt.Errors + 1,
|
||||
}
|
||||
if evt.StartTime == nil {
|
||||
upd["start_time"] = time.Now()
|
||||
upd["duration"] = 0
|
||||
} else {
|
||||
upd["duration"] = time.Since(*evt.StartTime).Seconds()
|
||||
}
|
||||
|
||||
err := tx.Model(&evt).UpdateColumns(upd).Error
|
||||
if err != nil {
|
||||
tx.Rollback()
|
||||
spew.Dump(evt)
|
||||
spew.Dump(upd)
|
||||
log.Println("UpdateStatusFromChecks fail update to current", err)
|
||||
mutex.Unlock()
|
||||
return
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if err := m.syncStatusPageIncidentsTx(tx, &evt); err != nil {
|
||||
tx.Rollback()
|
||||
log.Println("UpdateStatusFromChecks status page incident", err)
|
||||
mutex.Unlock()
|
||||
return
|
||||
}
|
||||
err := tx.Commit().Error
|
||||
|
||||
mutex.Unlock()
|
||||
|
||||
if err != nil {
|
||||
log.Println("UpdateStatusFromChecks commit fail", err)
|
||||
return
|
||||
}
|
||||
if m.State != prevState {
|
||||
m.invalidateStatusPages()
|
||||
}
|
||||
}
|
||||
|
||||
func (m *Monitor) invalidateStatusPages() {
|
||||
var ids []int64
|
||||
if err := DB().Model(&StatusPage{}).Where("? = ANY(monitor_ids)", m.ID).Pluck("id", &ids).Error; err != nil {
|
||||
return
|
||||
}
|
||||
for _, id := range ids {
|
||||
InvalidateStatusPageCache(id)
|
||||
}
|
||||
}
|
||||
|
||||
func (m *Monitor) syncStatusPageIncidentsTx(tx *gorm.DB, event *Event) error {
|
||||
if event == nil || event.ID == 0 {
|
||||
return nil
|
||||
}
|
||||
var pages []StatusPage
|
||||
if err := tx.Where("auto_open_incidents = TRUE AND ? = ANY(monitor_ids)", m.ID).Find(&pages).Error; err != nil {
|
||||
return err
|
||||
}
|
||||
for i := range pages {
|
||||
page := &pages[i]
|
||||
var incident StatusPageIncident
|
||||
err := tx.Where("status_page_id = ? AND event_id = ?", page.ID, event.ID).First(&incident).Error
|
||||
if m.State == stateERR || m.State == "FAIL" {
|
||||
if err != nil {
|
||||
// The database uniqueness constraint makes concurrent state updates idempotent.
|
||||
incident = StatusPageIncident{StatusPageID: page.ID, EventID: &event.ID, Title: m.GetLabel() + " is unavailable", BodyMD: event.Reason, Severity: StatusPageIncidentSeverityCrit, StartedAt: time.Now()}
|
||||
if err := tx.Create(&incident).Error; err != nil {
|
||||
return err
|
||||
}
|
||||
if err := EnqueueStatusPageIncidentDeliveriesTx(tx, page, &incident, "opened"); err != nil {
|
||||
return err
|
||||
}
|
||||
} else if incident.BodyMD != event.Reason {
|
||||
if err := tx.Model(&incident).Update("body_md", event.Reason).Error; err != nil {
|
||||
return err
|
||||
}
|
||||
incident.BodyMD, incident.UpdatedAt = event.Reason, time.Now()
|
||||
if err := EnqueueStatusPageIncidentDeliveriesTx(tx, page, &incident, "updated"); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
} else if err == nil && incident.ResolvedAt == nil {
|
||||
now := time.Now()
|
||||
if err := tx.Model(&incident).Update("resolved_at", now).Error; err != nil {
|
||||
return err
|
||||
}
|
||||
incident.ResolvedAt, incident.UpdatedAt = &now, now
|
||||
if err := EnqueueStatusPageIncidentDeliveriesTx(tx, page, &incident, "resolved"); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
Ссылка в новой задаче
Block a user