Files
orchestra/internal/router/router.go
T

292 lines
8.0 KiB
Go

// Package router assigns queued tasks to registered, reachable herdrs.
package router
import (
"encoding/json"
"errors"
"orchestra/internal/authz"
"orchestra/internal/domain"
"orchestra/internal/registry"
"orchestra/internal/store"
"sort"
"strings"
"sync"
"time"
)
type Availability interface{ Available(h registry.Herdr) bool }
// ProjectAvailability is an optional stricter availability contract used by
// federated workers, whose local checkout configuration is authoritative.
type ProjectAvailability interface {
Supports(h registry.Herdr, project string) bool
}
type AlwaysAvailable struct{}
func (AlwaysAvailable) Available(registry.Herdr) bool { return true }
// QuotaWindowLimits are the two independent caps the spec (§7.2) requires:
// the subscription pool's 5-hour rolling window and its weekly window. They
// are tracked and evaluated separately — a harness deep into its 5h window
// but fine on the week, or vice versa, must still be excluded.
type QuotaWindowLimits struct {
FiveHour float64
Weekly float64
}
const (
fiveHourWindow = 5 * time.Hour
weeklyWindow = 7 * 24 * time.Hour
// quotaConservativeFraction is the degrade-safe default from spec §7.2/§9
// item 1: since no quota pool is authoritative, treat 80% reported as
// full rather than trusting the exact number.
quotaConservativeFraction = 0.8
)
// QuotaAvailability applies the conservative 80% rule independently to the
// 5-hour rolling window and the weekly window, per harness. Receipts are
// additive across rotations; a cumulative session total must never replace
// earlier rotations' receipts (spec §5.2.1) — summing native per-report
// `consumed` deltas is what keeps this correct across rotation.
type QuotaAvailability struct {
Store *store.Store
Limits map[string]QuotaWindowLimits
Now func() time.Time
}
func (q QuotaAvailability) sumSince(harnessID string, since time.Time) (float64, bool) {
return q.Store.QuotaSince(harnessID, since)
}
func (q QuotaAvailability) Available(h registry.Herdr) bool {
if q.Store == nil {
return false
}
limits, bounded := q.Limits[h.ID]
if !bounded || (limits.FiveHour <= 0 && limits.Weekly <= 0) {
return true
}
now := time.Now()
if q.Now != nil {
now = q.Now()
}
if limits.FiveHour > 0 {
used, known := q.sumSince(h.ID, now.Add(-fiveHourWindow))
if !known || used >= limits.FiveHour*quotaConservativeFraction {
return false
}
}
if limits.Weekly > 0 {
used, known := q.sumSince(h.ID, now.Add(-weeklyWindow))
if !known || used >= limits.Weekly*quotaConservativeFraction {
return false
}
}
return true
}
type RetryPolicy struct {
MaxAttempts int
Backoff time.Duration
}
type Router struct {
Store *store.Store
Registry registry.Registry
Reachability registry.Reachability
Availability Availability
Timeout time.Duration
Retry RetryPolicy
Now func() time.Time
OnLease func(domain.Event) error
// HealthTTL bounds re-use of a successful or failed reachability probe.
// A scheduling pass always has a coherent health snapshot; this cache also
// prevents bursts of TaskCreated events from repeatedly dialing the same
// herdr between passes.
HealthTTL time.Duration
healthMu sync.Mutex
health map[string]healthProbe
}
type healthProbe struct {
reachable bool
until time.Time
}
func (r *Router) init() {
if r.Availability == nil {
r.Availability = AlwaysAvailable{}
}
if r.Now == nil {
r.Now = time.Now
}
if r.HealthTTL <= 0 {
r.HealthTTL = 5 * time.Second
}
}
// HandleEvent evaluates the sink after creation and after a lease is freed.
func (r *Router) HandleEvent(e domain.Event) ([]domain.Event, error) {
r.init()
if e.Type != "TaskCreated" && e.Type != "TaskReleased" {
return nil, nil
}
return r.AssignPending()
}
func (r *Router) AssignPending() ([]domain.Event, error) {
r.init()
if r.Store == nil {
return nil, errors.New("router: store required")
}
now := r.Now()
snapshot := r.Store.SchedulingSnapshot()
var queued []domain.Task
for _, t := range snapshot.Tasks {
if t.State == domain.StateQueued && (t.NextRetryAt.IsZero() || !now.Before(t.NextRetryAt)) {
queued = append(queued, t)
}
}
sort.SliceStable(queued, func(i, j int) bool { return importance(queued[i], now).Before(importance(queued[j], now)) })
health := r.healthSnapshot(now)
candidates := make(map[string][]registry.Herdr)
availability := make(map[string]bool)
availabilityKnown := make(map[string]bool)
projectSupport := make(map[string]bool)
projectSupportKnown := make(map[string]bool)
activeLeases := snapshot.ActiveLeases
var out []domain.Event
for _, t := range queued {
if r.Retry.MaxAttempts > 0 && t.Attempt >= r.Retry.MaxAttempts {
e, err := r.fail(t)
if err != nil {
return out, err
}
out = append(out, e)
continue
}
cs, found := candidates[t.Project]
if !found {
var err error
cs, err = r.Registry.CandidatesWithHealth(t.Project, health)
if err != nil {
continue
}
candidates[t.Project] = cs
}
for _, h := range cs {
projectKey := h.ID + "\x00" + t.Project
projectOK, checked := projectSupport[projectKey], projectSupportKnown[projectKey]
if !checked {
projectOK = true
if projects, ok := r.Availability.(ProjectAvailability); ok {
projectOK = projects.Supports(h, t.Project)
}
projectSupport[projectKey], projectSupportKnown[projectKey] = projectOK, true
}
if !projectOK || !matches(t.Capability, h.Capabilities) {
continue
}
available, checked := availability[h.ID], availabilityKnown[h.ID]
if !checked {
available = r.Availability.Available(h)
availability[h.ID], availabilityKnown[h.ID] = available, true
}
if !available || occupiedCount(activeLeases[h.ID], h.Concurrency) {
continue
}
e, err := r.Store.Lease(t.ID, h.ID, 30*time.Minute)
if err != nil {
continue
}
out = append(out, e)
activeLeases[h.ID]++
if r.OnLease != nil {
if err := r.OnLease(e); err != nil {
return out, err
}
}
break
}
}
return out, nil
}
func matches(need, have []string) bool {
set := map[string]bool{}
for _, x := range have {
set[strings.ToLower(x)] = true
}
for _, x := range need {
if !set[strings.ToLower(x)] {
return false
}
}
return true
}
func occupiedCount(count, limit int) bool {
return limit > 0 && count >= limit
}
func (r *Router) healthSnapshot(now time.Time) map[string]bool {
herdrs := r.Registry.Herdrs()
health := make(map[string]bool, len(herdrs))
if r.Reachability == nil {
for _, h := range herdrs {
health[h.ID] = true
}
return health
}
type target struct {
key string
id string
address string
}
var probes []target
r.healthMu.Lock()
if r.health == nil {
r.health = make(map[string]healthProbe)
}
for _, h := range herdrs {
key := h.ID + "\x00" + r.Registry.Endpoint(h)
if cached, ok := r.health[key]; ok && now.Before(cached.until) {
health[h.ID] = cached.reachable
continue
}
probes = append(probes, target{key: key, id: h.ID, address: r.Registry.Endpoint(h)})
}
r.healthMu.Unlock()
var wg sync.WaitGroup
var mu sync.Mutex
for _, target := range probes {
target := target
wg.Add(1)
go func() {
defer wg.Done()
reachable := r.Reachability.Reachable(target.address, r.Timeout)
mu.Lock()
health[target.id] = reachable
mu.Unlock()
}()
}
wg.Wait()
if len(probes) > 0 {
r.healthMu.Lock()
for _, target := range probes {
r.health[target.key] = healthProbe{reachable: health[target.id], until: now.Add(r.HealthTTL)}
}
r.healthMu.Unlock()
}
return health
}
func importance(t domain.Task, now time.Time) time.Time {
if t.Due != nil {
return t.Due.Add(-time.Duration(t.InherentPriority) * time.Hour)
}
return now.Add(-time.Duration(t.InherentPriority) * time.Hour)
}
func (r *Router) fail(t domain.Task) (domain.Event, error) {
b, _ := json.Marshal(map[string]any{"reason": "retry_limit", "attempts": t.Attempt, "failure_class": t.FailureClass})
e := domain.Event{ID: domain.NewID(), Type: "TaskFailed", TaskID: t.ID, Version: t.Version + 1, Payload: b, Surface: string(authz.System)}
return e, r.Store.Append(e)
}