iop/apps/node/internal/bootstrap/runtime_supervisor.go
toki 0ffcb88db0 feat: provider-resource-admission-ownership alignment
- Archive provider-resource-admission-ownership milestone/SDD
- Align contract: CP-edge wire, runtime refresh, node runtime, OpenAI surface
- Update roadmap: phase state, priority queue
- Update specs: control-plane ops, OpenAI surface, edge execution, provider pool refresh
- Add node runtime supervisor bootstrapping and unit tests
- Fix control-plane edge registry handler and http_views
- Fix edge model queue admission and long context queue tests
2026-07-22 20:45:04 +09:00

216 lines
6 KiB
Go

package bootstrap
import (
"context"
"sync"
"time"
"go.uber.org/fx"
"go.uber.org/zap"
"iop/apps/node/internal/transport"
"iop/packages/go/config"
)
// runtimeSupervisor owns the node's single active edge connection across its
// entire lifetime. A single supervisor goroutine serially performs the initial
// dial, waits for the established session to end, and reconnects, so at most one
// in-flight dial and one active session exist at any time. The mutex guards
// current across the supervisor goroutine and the OnStop hook.
type runtimeSupervisor struct {
cfg *config.NodeConfig
logger *zap.Logger
dialer DialFunc
sleeper func(ctx context.Context, d time.Duration)
metricsStarter func(port int) error
shutdowner fx.Shutdowner
mu sync.Mutex
current *runtimeOwner
cancelSupervisor context.CancelFunc
done chan struct{}
}
// start launches the metrics server and the single supervisor goroutine, then
// returns immediately. Establishing the initial connection is the supervisor's
// responsibility, so a transient Edge outage at startup no longer fails the fx
// OnStart hook (SDD S16).
func (s *runtimeSupervisor) start(_ context.Context) error {
supCtx, cancel := context.WithCancel(context.Background())
s.cancelSupervisor = cancel
s.done = make(chan struct{})
go func() {
defer close(s.done)
s.run(supCtx)
}()
if s.cfg.Metrics.Port > 0 {
go func() {
if err := s.metricsStarter(s.cfg.Metrics.Port); err != nil {
s.logger.Warn("metrics server exited", zap.Error(err))
}
}()
}
return nil
}
// run drives the full connectivity lifecycle: connect (initial), wait for the
// active session to end, then reconnect, repeating until the supervisor is
// cancelled, a fatal error occurs, finite attempts are exhausted, or a local
// shutdown ends the session.
func (s *runtimeSupervisor) run(supCtx context.Context) {
owner := s.connect(supCtx, true)
if owner == nil {
return
}
for {
select {
case <-supCtx.Done():
return
case <-owner.sess.Done():
}
// Guard against supervisor cancellation racing with disconnect.
select {
case <-supCtx.Done():
return
default:
}
if owner.sess.IsLocalShutdown() {
return
}
owner = s.connect(supCtx, false)
if owner == nil {
return
}
}
}
// connect runs the bounded connect/retry loop shared by the initial dial and
// established-session reconnect. On success it installs the new owner and
// returns it. It returns nil after supervisor cancellation, a fatal
// (non-retryable) failure, or finite attempt exhaustion; the latter two request
// a node shutdown with exit code 1 before returning.
//
// The reconnect path sleeps before every attempt; the initial path issues its
// first dial immediately for fast startup and sleeps only before retries. A
// finite policy (max_attempts>0) makes exactly max_attempts attempts before
// exhausting; an unlimited policy (max_attempts=0) retries until cancelled.
func (s *runtimeSupervisor) connect(supCtx context.Context, initial bool) *runtimeOwner {
maxAttempts := s.cfg.Reconnect.MaxAttempts
intervalSec := s.cfg.Reconnect.IntervalSec
unlimited := maxAttempts == 0
for attempt := 1; unlimited || attempt <= maxAttempts; attempt++ {
if attempt > 1 || !initial {
s.sleeper(supCtx, time.Duration(intervalSec)*time.Second)
}
if supCtx.Err() != nil {
return nil
}
s.logger.Info("connecting to edge",
zap.Bool("initial", initial),
zap.Int("attempt", attempt),
zap.Int("max_attempts", maxAttempts),
zap.Bool("unlimited", unlimited),
zap.Int("interval_sec", intervalSec),
)
dialCtx, dialCancel := context.WithTimeout(supCtx, 30*time.Second)
owner, err := connectRuntime(dialCtx, s.cfg, s.logger, s.dialer)
dialCancel()
// Local shutdown wins when a dial completes concurrently with supervisor
// cancellation. A dialer may observe cancellation and still return a late
// success or fatal result; never install that owner or turn it into exit 1.
if supCtx.Err() != nil {
if owner != nil {
owner.close()
}
return nil
}
if err == nil {
s.swapOwner(owner)
return owner
}
if transport.IsFatalConnectError(err) {
s.fatal(err)
return nil
}
s.logger.Warn("connect attempt failed",
zap.Bool("initial", initial),
zap.Int("attempt", attempt),
zap.Int("max_attempts", maxAttempts),
zap.Error(err),
)
}
// A local shutdown that lands on the final finite attempt falls out of the
// loop; treat it as cancellation rather than exhaustion.
if supCtx.Err() != nil {
return nil
}
s.exhaust(maxAttempts)
return nil
}
// swapOwner installs owner as the current connection and closes the previous one.
func (s *runtimeSupervisor) swapOwner(owner *runtimeOwner) {
s.mu.Lock()
prev := s.current
s.current = owner
s.mu.Unlock()
if prev != nil {
prev.close()
}
}
// fatal clears the current owner, logs the non-retryable failure, and requests a
// node shutdown with a non-zero exit code.
func (s *runtimeSupervisor) fatal(err error) {
s.clearCurrent()
s.logger.Error("connect failed with non-retryable error, shutting down node",
zap.Error(err),
)
_ = s.shutdowner.Shutdown(fx.ExitCode(1))
}
// exhaust clears the current owner, logs the exhaustion, and requests a node
// shutdown with a non-zero exit code.
func (s *runtimeSupervisor) exhaust(maxAttempts int) {
s.clearCurrent()
s.logger.Error("reconnect exhausted, shutting down node",
zap.Int("max_attempts", maxAttempts),
)
_ = s.shutdowner.Shutdown(fx.ExitCode(1))
}
// clearCurrent detaches and closes the current owner, if any.
func (s *runtimeSupervisor) clearCurrent() {
s.mu.Lock()
prev := s.current
s.current = nil
s.mu.Unlock()
if prev != nil {
prev.close()
}
}
// stop cancels the supervisor goroutine, waits for it to exit, and closes the
// current connection. Waiting guarantees no retry sleep or dial survives the
// shutdown and no additional attempt runs afterward.
func (s *runtimeSupervisor) stop() {
if s.cancelSupervisor != nil {
s.cancelSupervisor()
}
if s.done != nil {
<-s.done
}
s.clearCurrent()
}