392 строки
14 KiB
Go
392 строки
14 KiB
Go
package models
|
|
|
|
import (
|
|
"errors"
|
|
"fmt"
|
|
"log"
|
|
"time"
|
|
|
|
"github.com/google/uuid"
|
|
"gorm.io/gorm"
|
|
"gorm.io/gorm/clause"
|
|
)
|
|
|
|
// DefaultCheckTaskLeaseTTL is how long a leased check remains owned before the reaper
|
|
// returns it to the queue. It is deliberately larger than the worker's heartbeat
|
|
// (10s) so a healthy worker can finish a 30s check without the reaper stealing
|
|
// the lease, and deliberately smaller than the heartbeat timeout (2m) so a
|
|
// crashed worker sees its tasks reaped within one watchdog cycle.
|
|
const DefaultCheckTaskLeaseTTL = 60 * time.Second
|
|
|
|
// DefaultTaskLeaseTTL remains the check-task default for existing confirmation
|
|
// callers. Generic task selection must use TaskLeaseTTL so notification work is
|
|
// not reclaimed during its longer execution window.
|
|
const DefaultTaskLeaseTTL = DefaultCheckTaskLeaseTTL
|
|
|
|
// Notification execution is bounded by the worker runner at five minutes. The
|
|
// additional minute covers result serialization, websocket/HTTP transport, and
|
|
// a control-plane scheduling delay before the persisted lease may be reaped.
|
|
const (
|
|
DefaultNotificationExecutionTimeout = 5 * time.Minute
|
|
NotificationTaskReportMargin = time.Minute
|
|
DefaultNotificationTaskLeaseTTL = DefaultNotificationExecutionTimeout + NotificationTaskReportMargin
|
|
)
|
|
|
|
// TaskLeaseTTL returns the persisted lease lifetime for a task kind.
|
|
func TaskLeaseTTL(kind string) time.Duration {
|
|
if kind == TaskKindNotification {
|
|
return DefaultNotificationTaskLeaseTTL
|
|
}
|
|
return DefaultCheckTaskLeaseTTL
|
|
}
|
|
|
|
// DefaultTaskMaxAttempts is the retry budget for a task before it moves to dead.
|
|
const DefaultTaskMaxAttempts = 5
|
|
|
|
// DefaultNotificationTaskDeadline is assigned to manually replayed notification
|
|
// dead letters. Normal producer tasks may be deadline-free, but a replay must
|
|
// never inherit an already-expired deadline.
|
|
const DefaultNotificationTaskDeadline = 15 * time.Minute
|
|
|
|
// ErrNotificationMethodNotAuthorized is returned by EnqueueNotificationTask
|
|
// when the producer can prove no worker in the pool is authorized for the
|
|
// (method, account) pair. The caller may skip the enqueue or log + continue.
|
|
var ErrNotificationMethodNotAuthorized = errors.New("no worker authorized for method/account")
|
|
|
|
// EnqueueNotificationTaskInput is the pre-rendered envelope produced by the
|
|
// notifier producer. All slices are required; the selector never reads them.
|
|
type EnqueueNotificationTaskInput struct {
|
|
AccountID int64
|
|
NotificationID int64
|
|
ContactID int64
|
|
MessageID *int64
|
|
MonitorID *int64
|
|
CheckID *int64
|
|
EventIDs []int64
|
|
Method string // "email", "telegram", "webhook", "mattermost", "sms", "voice"
|
|
Subject string
|
|
BodyText string
|
|
BodyHTML string
|
|
BodyMarkdown string
|
|
Language string
|
|
MessageKind string // "down", "up", "exp", "test"
|
|
NotBefore time.Time
|
|
Deadline *time.Time
|
|
MaxAttempts int
|
|
Payload []byte // marshaled task-specific data
|
|
IdempotencyKey string // optional; manual test tasks use a unique key and do not have event IDs
|
|
}
|
|
|
|
// EnqueueNotificationTask writes one Task row keyed by a stable idempotency key.
|
|
// A second call with the same key (same notification/contact/event triple) is a
|
|
// no-op so the producer is safe to call more than once per pass.
|
|
//
|
|
// The capability precheck uses the same NotificationMethods + NotificationAccounts
|
|
// rule the selector does, so the producer can skip enqueueing work that no
|
|
// operated worker could ever pick up (sms/voice until phase 4).
|
|
func EnqueueNotificationTask(input *EnqueueNotificationTaskInput) (*Task, error) {
|
|
return EnqueueNotificationTaskTx(DB(), input)
|
|
}
|
|
|
|
// EnqueueNotificationTaskTx is the transactional form used by state machines
|
|
// that must commit their transition, audit event, message, and task together.
|
|
func EnqueueNotificationTaskTx(tx *gorm.DB, input *EnqueueNotificationTaskInput) (*Task, error) {
|
|
if tx == nil {
|
|
return nil, errors.New("enqueue: nil transaction")
|
|
}
|
|
if input.AccountID == 0 || input.ContactID == 0 {
|
|
return nil, errors.New("enqueue: account_id and contact_id are required")
|
|
}
|
|
|
|
idempotencyKey := input.IdempotencyKey
|
|
if idempotencyKey == "" {
|
|
if input.NotificationID == 0 || len(input.EventIDs) == 0 {
|
|
return nil, errors.New("enqueue: notification_id and event_ids are required without explicit idempotency_key")
|
|
}
|
|
idempotencyKey = notificationIdempotencyKey(input.NotificationID, input.ContactID, input.EventIDs[0])
|
|
}
|
|
|
|
// Fast path: row already exists from a previous producer tick. Returning
|
|
// the existing row is the idempotency guarantee — second calls return the
|
|
// same id, no second INSERT.
|
|
var existing Task
|
|
if err := tx.Where("idempotency_key = ?", idempotencyKey).First(&existing).Error; err == nil {
|
|
return &existing, nil
|
|
} else if !errors.Is(err, gorm.ErrRecordNotFound) {
|
|
return nil, err
|
|
}
|
|
|
|
maxAttempts := input.MaxAttempts
|
|
if maxAttempts <= 0 {
|
|
maxAttempts = DefaultTaskMaxAttempts
|
|
}
|
|
if !anyWorkerCanDeliver(input.Method, input.AccountID) {
|
|
return nil, fmt.Errorf("%w: method=%s account=%d", ErrNotificationMethodNotAuthorized, input.Method, input.AccountID)
|
|
}
|
|
|
|
notBefore := input.NotBefore
|
|
if notBefore.IsZero() {
|
|
notBefore = time.Now()
|
|
}
|
|
|
|
now := time.Now()
|
|
task := &Task{
|
|
JobID: uuid.New().String(),
|
|
Kind: TaskKindNotification,
|
|
State: TaskStateQueued,
|
|
AccountID: input.AccountID,
|
|
MessageID: input.MessageID,
|
|
ContactID: &input.ContactID,
|
|
MonitorID: input.MonitorID,
|
|
CheckID: input.CheckID,
|
|
NotBefore: notBefore,
|
|
Deadline: input.Deadline,
|
|
Attempts: 0,
|
|
MaxAttempts: maxAttempts,
|
|
IdempotencyKey: idempotencyKey,
|
|
}
|
|
if len(input.Payload) > 0 {
|
|
task.Payload = input.Payload
|
|
}
|
|
task.CreatedAt = now
|
|
task.UpdatedAt = now
|
|
|
|
// ON CONFLICT DO NOTHING so a concurrent producer tick racing with us on
|
|
// the same idempotency_key loses the race but does not duplicate the row.
|
|
if err := tx.Clauses(clause.OnConflict{DoNothing: true}).Create(task).Error; err != nil {
|
|
return nil, err
|
|
}
|
|
if task.ID == 0 {
|
|
// Lost the race. Re-read and return the winner.
|
|
if err := tx.Where("idempotency_key = ?", idempotencyKey).First(&existing).Error; err != nil {
|
|
return nil, err
|
|
}
|
|
return &existing, nil
|
|
}
|
|
return task, nil
|
|
}
|
|
|
|
// NotificationIdempotencyKey exposes the producer's idempotency key so the
|
|
// result handler and admin tooling can match a Task row back to the logical
|
|
// (notification, contact, event) tuple without re-deriving the format.
|
|
func NotificationIdempotencyKey(notificationID, contactID, eventID int64) string {
|
|
return notificationIdempotencyKey(notificationID, contactID, eventID)
|
|
}
|
|
|
|
func notificationIdempotencyKey(notificationID, contactID, eventID int64) string {
|
|
return fmt.Sprintf("notif:%d:contact:%d:event:%d", notificationID, contactID, eventID)
|
|
}
|
|
|
|
// anyWorkerCanDeliver returns true if at least one active worker in the pool is
|
|
// authorized to deliver the given (method, account) pair. Used by the producer
|
|
// to skip enqueues no worker could ever pick up.
|
|
func anyWorkerCanDeliver(method string, accountID int64) bool {
|
|
var nodes []WorkerNode
|
|
if err := DB().Where("status = ? AND last_seen > ?", "active", time.Now().Add(-WorkerHeartbeatFreshness)).Find(&nodes).Error; err != nil {
|
|
log.Printf("task_selector: cannot list workers: %v", err)
|
|
// Be permissive on lookup failure: the selector's own filter would still
|
|
// hold the lease back, so the worst case is a queued task nobody picks
|
|
// up — which the reaper eventually dead-letters.
|
|
return true
|
|
}
|
|
for i := range nodes {
|
|
if nodes[i].SupportsTaskEnvelope() && nodes[i].CanDeliverNotification(method, accountID) {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// TasksForWorker leases up to `limit` due tasks for the worker. The selection
|
|
// is one transaction so the FOR UPDATE SKIP LOCKED + UPDATE that flips state
|
|
// from queued to leased is atomic. Notification tasks are filtered by the worker's
|
|
// notification_methods + notification_accounts capability set; check tasks are
|
|
// filtered by check_types in their payload.
|
|
//
|
|
// The function is safe to call from multiple goroutines for different workers.
|
|
// Two workers that hit the DB at the same time will see disjoint task sets.
|
|
func TasksForWorker(worker *WorkerNode, limit int) ([]Task, error) {
|
|
if worker == nil {
|
|
return nil, errors.New("TasksForWorker: worker is nil")
|
|
}
|
|
if !worker.SupportsTaskEnvelope() {
|
|
return nil, nil
|
|
}
|
|
if limit <= 0 {
|
|
limit = 1
|
|
}
|
|
|
|
notifMethods := worker.NotificationMethods()
|
|
notifAccounts := worker.AccessibleAccountIDs()
|
|
hasNotif := len(notifMethods) > 0
|
|
|
|
tx := DB().Begin()
|
|
if tx.Error != nil {
|
|
return nil, tx.Error
|
|
}
|
|
defer func() {
|
|
if r := recover(); r != nil {
|
|
_ = tx.Rollback().Error
|
|
panic(r)
|
|
}
|
|
}()
|
|
|
|
now := time.Now()
|
|
var out []Task
|
|
|
|
// First pass: notification tasks the worker is authorized to deliver. We
|
|
// also bump attempts and flip state to leased in the same row so the
|
|
// outer selector+lease is atomic. The method filter is a JSONB extract on
|
|
// payload->>'method' so a single worker query can target one method list.
|
|
if hasNotif {
|
|
notifQuery := tx.Clauses(SkipLockedClause).
|
|
Where("state = ? AND kind = ?", TaskStateQueued, TaskKindNotification).
|
|
Where("not_before <= ?", now).
|
|
Where("(deadline IS NULL OR deadline > ?)", now).
|
|
Where("payload->>'method' IN (?)", notifMethods).
|
|
Where("payload->>'method' <> ''")
|
|
if len(notifAccounts) > 0 {
|
|
notifQuery = notifQuery.Where("account_id IN (?)", notifAccounts)
|
|
}
|
|
|
|
var picked []Task
|
|
if err := notifQuery.Limit(limit).Find(&picked).Error; err != nil {
|
|
_ = tx.Rollback().Error
|
|
return nil, err
|
|
}
|
|
|
|
for i := range picked {
|
|
row := picked[i]
|
|
newAttempts := row.Attempts + 1
|
|
leaseToken := uuid.NewString()
|
|
leaseUntil := now.Add(TaskLeaseTTL(row.Kind))
|
|
if err := tx.Model(&row).Updates(map[string]interface{}{
|
|
colState: TaskStateLeased,
|
|
"lease_owner": worker.WorkerID,
|
|
"lease_expires_at": leaseUntil,
|
|
"attempts": newAttempts,
|
|
"lease_token": leaseToken,
|
|
"updated_at": now,
|
|
}).Error; err != nil {
|
|
_ = tx.Rollback().Error
|
|
return nil, err
|
|
}
|
|
row.State = TaskStateLeased
|
|
row.LeaseOwner = worker.WorkerID
|
|
row.LeaseExpiresAt = &leaseUntil
|
|
row.Attempts = newAttempts
|
|
row.LeaseToken = leaseToken
|
|
out = append(out, row)
|
|
}
|
|
}
|
|
|
|
remaining := limit - len(out)
|
|
if checkTypes := worker.CheckTypes(); remaining > 0 && len(checkTypes) > 0 {
|
|
checkQuery := tx.Clauses(SkipLockedClause).
|
|
Where("state = ? AND kind = ?", TaskStateQueued, TaskKindCheck).
|
|
Where("not_before <= ?", now).
|
|
Where("(deadline IS NULL OR deadline > ?)", now).
|
|
Where("payload->>'kind' IN (?)", checkTypes)
|
|
if accounts := worker.AccessibleAccountIDs(); len(accounts) > 0 {
|
|
checkQuery = checkQuery.Where("account_id IN (?)", accounts)
|
|
}
|
|
var picked []Task
|
|
if err := checkQuery.Limit(remaining).Find(&picked).Error; err != nil {
|
|
_ = tx.Rollback().Error
|
|
return nil, err
|
|
}
|
|
for i := range picked {
|
|
row := picked[i]
|
|
newAttempts := row.Attempts + 1
|
|
leaseToken := uuid.NewString()
|
|
leaseUntil := now.Add(TaskLeaseTTL(row.Kind))
|
|
if err := tx.Model(&row).Updates(map[string]interface{}{
|
|
colState: TaskStateLeased, "lease_owner": worker.WorkerID,
|
|
"lease_expires_at": leaseUntil, "attempts": newAttempts, "lease_token": leaseToken, "updated_at": now,
|
|
}).Error; err != nil {
|
|
_ = tx.Rollback().Error
|
|
return nil, err
|
|
}
|
|
row.State, row.LeaseOwner, row.LeaseExpiresAt, row.Attempts, row.LeaseToken = TaskStateLeased, worker.WorkerID, &leaseUntil, newAttempts, leaseToken
|
|
out = append(out, row)
|
|
}
|
|
}
|
|
|
|
if err := tx.Commit().Error; err != nil {
|
|
return nil, err
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
// AvailableWorkerTaskCapacity returns unoccupied local worker slots. Durable
|
|
// unexpired leases and the worker's heartbeat-reported active/queued workload
|
|
// describe the same work from different sides, so the larger value is used to
|
|
// avoid both over-dispatching and double-counting a healthy worker.
|
|
func AvailableWorkerTaskCapacity(worker *WorkerNode) (int, error) {
|
|
if worker == nil {
|
|
return 0, errors.New("worker capacity: worker is nil")
|
|
}
|
|
concurrency := worker.Concurrency
|
|
if concurrency < 1 {
|
|
concurrency = 1
|
|
}
|
|
var leased int64
|
|
if err := DB().Model(&Task{}).Where("state = ? AND lease_owner = ? AND lease_expires_at > ?", TaskStateLeased, worker.WorkerID, time.Now()).Count(&leased).Error; err != nil {
|
|
return 0, err
|
|
}
|
|
var confirmationLeases int64
|
|
if err := DB().Model(&CheckAttempt{}).Where("worker_node_id = ? AND kind = ? AND state = ? AND lease_expires_at > ?", worker.ID, AttemptKindConfirm, AttemptStateLeased, time.Now()).Count(&confirmationLeases).Error; err != nil {
|
|
return 0, err
|
|
}
|
|
used := int(leased + confirmationLeases)
|
|
if reported := worker.ReportedWorkload(); reported > used {
|
|
used = reported
|
|
}
|
|
if used >= concurrency {
|
|
return 0, nil
|
|
}
|
|
return concurrency - used, nil
|
|
}
|
|
|
|
// GetTaskByJobID returns one task row keyed by its unique job_id. The result
|
|
// handler uses this to validate that the incoming JobID exists and matches the
|
|
// calling worker before it mutates state.
|
|
func GetTaskByJobID(jobID string) (*Task, error) {
|
|
if jobID == "" {
|
|
return nil, errors.New("GetTaskByJobID: empty job_id")
|
|
}
|
|
var task Task
|
|
if err := DB().Where("job_id = ?", jobID).First(&task).Error; err != nil {
|
|
return nil, err
|
|
}
|
|
return &task, nil
|
|
}
|
|
|
|
// TasksForWorkerTx is the variant exposed for tests so a single SELECT inside a
|
|
// caller-provided transaction can be inspected without the auto-commit wrapper.
|
|
// Production code should use TasksForWorker.
|
|
func TasksForWorkerTx(tx *gorm.DB, worker *WorkerNode, limit int) ([]Task, error) {
|
|
if tx == nil {
|
|
return nil, errors.New("TasksForWorkerTx: nil tx")
|
|
}
|
|
notifMethods := worker.NotificationMethods()
|
|
if len(notifMethods) == 0 {
|
|
return nil, nil
|
|
}
|
|
now := time.Now()
|
|
var out []Task
|
|
// Use SKIP LOCKED to avoid contention between workers (mirror of ChecksForWorker).
|
|
q := tx.Clauses(SkipLockedClause).
|
|
Where("state = ? AND kind = ?", TaskStateQueued, TaskKindNotification).
|
|
Where("not_before <= ?", now).
|
|
Where("(deadline IS NULL OR deadline > ?)", now).
|
|
Where("payload->>'method' IN (?)", notifMethods)
|
|
if accounts := worker.AccessibleAccountIDs(); len(accounts) > 0 {
|
|
q = q.Where("account_id IN (?)", accounts)
|
|
}
|
|
if err := q.Limit(limit).Find(&out).Error; err != nil {
|
|
return nil, err
|
|
}
|
|
return out, nil
|
|
}
|