292 lines
8.0 KiB
Go
292 lines
8.0 KiB
Go
// Package router assigns queued tasks to registered, reachable herdrs.
|
|
package router
|
|
|
|
import (
|
|
"encoding/json"
|
|
"errors"
|
|
"orchestra/internal/authz"
|
|
"orchestra/internal/domain"
|
|
"orchestra/internal/registry"
|
|
"orchestra/internal/store"
|
|
"sort"
|
|
"strings"
|
|
"sync"
|
|
"time"
|
|
)
|
|
|
|
type Availability interface{ Available(h registry.Herdr) bool }
|
|
|
|
// ProjectAvailability is an optional stricter availability contract used by
|
|
// federated workers, whose local checkout configuration is authoritative.
|
|
type ProjectAvailability interface {
|
|
Supports(h registry.Herdr, project string) bool
|
|
}
|
|
type AlwaysAvailable struct{}
|
|
|
|
func (AlwaysAvailable) Available(registry.Herdr) bool { return true }
|
|
|
|
// QuotaWindowLimits are the two independent caps the spec (§7.2) requires:
|
|
// the subscription pool's 5-hour rolling window and its weekly window. They
|
|
// are tracked and evaluated separately — a harness deep into its 5h window
|
|
// but fine on the week, or vice versa, must still be excluded.
|
|
type QuotaWindowLimits struct {
|
|
FiveHour float64
|
|
Weekly float64
|
|
}
|
|
|
|
const (
|
|
fiveHourWindow = 5 * time.Hour
|
|
weeklyWindow = 7 * 24 * time.Hour
|
|
// quotaConservativeFraction is the degrade-safe default from spec §7.2/§9
|
|
// item 1: since no quota pool is authoritative, treat 80% reported as
|
|
// full rather than trusting the exact number.
|
|
quotaConservativeFraction = 0.8
|
|
)
|
|
|
|
// QuotaAvailability applies the conservative 80% rule independently to the
|
|
// 5-hour rolling window and the weekly window, per harness. Receipts are
|
|
// additive across rotations; a cumulative session total must never replace
|
|
// earlier rotations' receipts (spec §5.2.1) — summing native per-report
|
|
// `consumed` deltas is what keeps this correct across rotation.
|
|
type QuotaAvailability struct {
|
|
Store *store.Store
|
|
Limits map[string]QuotaWindowLimits
|
|
Now func() time.Time
|
|
}
|
|
|
|
func (q QuotaAvailability) sumSince(harnessID string, since time.Time) (float64, bool) {
|
|
return q.Store.QuotaSince(harnessID, since)
|
|
}
|
|
|
|
func (q QuotaAvailability) Available(h registry.Herdr) bool {
|
|
if q.Store == nil {
|
|
return false
|
|
}
|
|
limits, bounded := q.Limits[h.ID]
|
|
if !bounded || (limits.FiveHour <= 0 && limits.Weekly <= 0) {
|
|
return true
|
|
}
|
|
now := time.Now()
|
|
if q.Now != nil {
|
|
now = q.Now()
|
|
}
|
|
if limits.FiveHour > 0 {
|
|
used, known := q.sumSince(h.ID, now.Add(-fiveHourWindow))
|
|
if !known || used >= limits.FiveHour*quotaConservativeFraction {
|
|
return false
|
|
}
|
|
}
|
|
if limits.Weekly > 0 {
|
|
used, known := q.sumSince(h.ID, now.Add(-weeklyWindow))
|
|
if !known || used >= limits.Weekly*quotaConservativeFraction {
|
|
return false
|
|
}
|
|
}
|
|
return true
|
|
}
|
|
|
|
type RetryPolicy struct {
|
|
MaxAttempts int
|
|
Backoff time.Duration
|
|
}
|
|
type Router struct {
|
|
Store *store.Store
|
|
Registry registry.Registry
|
|
Reachability registry.Reachability
|
|
Availability Availability
|
|
Timeout time.Duration
|
|
Retry RetryPolicy
|
|
Now func() time.Time
|
|
OnLease func(domain.Event) error
|
|
// HealthTTL bounds re-use of a successful or failed reachability probe.
|
|
// A scheduling pass always has a coherent health snapshot; this cache also
|
|
// prevents bursts of TaskCreated events from repeatedly dialing the same
|
|
// herdr between passes.
|
|
HealthTTL time.Duration
|
|
healthMu sync.Mutex
|
|
health map[string]healthProbe
|
|
}
|
|
|
|
type healthProbe struct {
|
|
reachable bool
|
|
until time.Time
|
|
}
|
|
|
|
func (r *Router) init() {
|
|
if r.Availability == nil {
|
|
r.Availability = AlwaysAvailable{}
|
|
}
|
|
if r.Now == nil {
|
|
r.Now = time.Now
|
|
}
|
|
if r.HealthTTL <= 0 {
|
|
r.HealthTTL = 5 * time.Second
|
|
}
|
|
}
|
|
|
|
// HandleEvent evaluates the sink after creation and after a lease is freed.
|
|
func (r *Router) HandleEvent(e domain.Event) ([]domain.Event, error) {
|
|
r.init()
|
|
if e.Type != "TaskCreated" && e.Type != "TaskReleased" {
|
|
return nil, nil
|
|
}
|
|
return r.AssignPending()
|
|
}
|
|
|
|
func (r *Router) AssignPending() ([]domain.Event, error) {
|
|
r.init()
|
|
if r.Store == nil {
|
|
return nil, errors.New("router: store required")
|
|
}
|
|
now := r.Now()
|
|
snapshot := r.Store.SchedulingSnapshot()
|
|
var queued []domain.Task
|
|
for _, t := range snapshot.Tasks {
|
|
if t.State == domain.StateQueued && (t.NextRetryAt.IsZero() || !now.Before(t.NextRetryAt)) {
|
|
queued = append(queued, t)
|
|
}
|
|
}
|
|
sort.SliceStable(queued, func(i, j int) bool { return importance(queued[i], now).Before(importance(queued[j], now)) })
|
|
health := r.healthSnapshot(now)
|
|
candidates := make(map[string][]registry.Herdr)
|
|
availability := make(map[string]bool)
|
|
availabilityKnown := make(map[string]bool)
|
|
projectSupport := make(map[string]bool)
|
|
projectSupportKnown := make(map[string]bool)
|
|
activeLeases := snapshot.ActiveLeases
|
|
var out []domain.Event
|
|
for _, t := range queued {
|
|
if r.Retry.MaxAttempts > 0 && t.Attempt >= r.Retry.MaxAttempts {
|
|
e, err := r.fail(t)
|
|
if err != nil {
|
|
return out, err
|
|
}
|
|
out = append(out, e)
|
|
continue
|
|
}
|
|
cs, found := candidates[t.Project]
|
|
if !found {
|
|
var err error
|
|
cs, err = r.Registry.CandidatesWithHealth(t.Project, health)
|
|
if err != nil {
|
|
continue
|
|
}
|
|
candidates[t.Project] = cs
|
|
}
|
|
for _, h := range cs {
|
|
projectKey := h.ID + "\x00" + t.Project
|
|
projectOK, checked := projectSupport[projectKey], projectSupportKnown[projectKey]
|
|
if !checked {
|
|
projectOK = true
|
|
if projects, ok := r.Availability.(ProjectAvailability); ok {
|
|
projectOK = projects.Supports(h, t.Project)
|
|
}
|
|
projectSupport[projectKey], projectSupportKnown[projectKey] = projectOK, true
|
|
}
|
|
if !projectOK || !matches(t.Capability, h.Capabilities) {
|
|
continue
|
|
}
|
|
available, checked := availability[h.ID], availabilityKnown[h.ID]
|
|
if !checked {
|
|
available = r.Availability.Available(h)
|
|
availability[h.ID], availabilityKnown[h.ID] = available, true
|
|
}
|
|
if !available || occupiedCount(activeLeases[h.ID], h.Concurrency) {
|
|
continue
|
|
}
|
|
e, err := r.Store.Lease(t.ID, h.ID, 30*time.Minute)
|
|
if err != nil {
|
|
continue
|
|
}
|
|
out = append(out, e)
|
|
activeLeases[h.ID]++
|
|
if r.OnLease != nil {
|
|
if err := r.OnLease(e); err != nil {
|
|
return out, err
|
|
}
|
|
}
|
|
break
|
|
}
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
func matches(need, have []string) bool {
|
|
set := map[string]bool{}
|
|
for _, x := range have {
|
|
set[strings.ToLower(x)] = true
|
|
}
|
|
for _, x := range need {
|
|
if !set[strings.ToLower(x)] {
|
|
return false
|
|
}
|
|
}
|
|
return true
|
|
}
|
|
func occupiedCount(count, limit int) bool {
|
|
return limit > 0 && count >= limit
|
|
}
|
|
|
|
func (r *Router) healthSnapshot(now time.Time) map[string]bool {
|
|
herdrs := r.Registry.Herdrs()
|
|
health := make(map[string]bool, len(herdrs))
|
|
if r.Reachability == nil {
|
|
for _, h := range herdrs {
|
|
health[h.ID] = true
|
|
}
|
|
return health
|
|
}
|
|
type target struct {
|
|
key string
|
|
id string
|
|
address string
|
|
}
|
|
var probes []target
|
|
r.healthMu.Lock()
|
|
if r.health == nil {
|
|
r.health = make(map[string]healthProbe)
|
|
}
|
|
for _, h := range herdrs {
|
|
key := h.ID + "\x00" + r.Registry.Endpoint(h)
|
|
if cached, ok := r.health[key]; ok && now.Before(cached.until) {
|
|
health[h.ID] = cached.reachable
|
|
continue
|
|
}
|
|
probes = append(probes, target{key: key, id: h.ID, address: r.Registry.Endpoint(h)})
|
|
}
|
|
r.healthMu.Unlock()
|
|
var wg sync.WaitGroup
|
|
var mu sync.Mutex
|
|
for _, target := range probes {
|
|
target := target
|
|
wg.Add(1)
|
|
go func() {
|
|
defer wg.Done()
|
|
reachable := r.Reachability.Reachable(target.address, r.Timeout)
|
|
mu.Lock()
|
|
health[target.id] = reachable
|
|
mu.Unlock()
|
|
}()
|
|
}
|
|
wg.Wait()
|
|
if len(probes) > 0 {
|
|
r.healthMu.Lock()
|
|
for _, target := range probes {
|
|
r.health[target.key] = healthProbe{reachable: health[target.id], until: now.Add(r.HealthTTL)}
|
|
}
|
|
r.healthMu.Unlock()
|
|
}
|
|
return health
|
|
}
|
|
func importance(t domain.Task, now time.Time) time.Time {
|
|
if t.Due != nil {
|
|
return t.Due.Add(-time.Duration(t.InherentPriority) * time.Hour)
|
|
}
|
|
return now.Add(-time.Duration(t.InherentPriority) * time.Hour)
|
|
}
|
|
func (r *Router) fail(t domain.Task) (domain.Event, error) {
|
|
b, _ := json.Marshal(map[string]any{"reason": "retry_limit", "attempts": t.Attempt, "failure_class": t.FailureClass})
|
|
e := domain.Event{ID: domain.NewID(), Type: "TaskFailed", TaskID: t.ID, Version: t.Version + 1, Payload: b, Surface: string(authz.System)}
|
|
return e, r.Store.Append(e)
|
|
}
|