Fail VMs if the worker had crashed/is unhealthy (#70)

* Fail VMs if the worker had crashed/is unhealthy

* OnDiskName: properly handle cases when VM's name contains hyphens

* Worker: introduce Offline() method and check it before scheduling

* tart.List(): use Tart's JSON output

* OnDiskName: remove empty parts check

* Scheduler: move health-checking logic to a separate function

* Only fail "running" VMs

* Only fail orphaned VMs if they're in terminal state

* Integration tests

* Run healthCheckingLoopIteration() before schedulingLoopIteration()

* Worker: sync on-disk VMs only once at start
This commit is contained in:
Nikolay Edigaryev
2023-04-03 16:47:49 +04:00
committed by GitHub
parent ea1e5c8578
commit 4eafec99a5
14 changed files with 523 additions and 48 deletions
+6 -3
View File
@@ -43,14 +43,16 @@ type Controller struct {
workerNotifier *notifier.Notifier
proxy *proxy.Proxy
enableSwaggerDocs bool
workerOfflineTimeout time.Duration
rpc.UnimplementedControllerServer
}
func New(opts ...Option) (*Controller, error) {
controller := &Controller{
workerNotifier: notifier.NewNotifier(),
proxy: proxy.NewProxy(),
workerNotifier: notifier.NewNotifier(),
proxy: proxy.NewProxy(),
workerOfflineTimeout: 3 * time.Minute,
}
// Apply options
@@ -76,7 +78,8 @@ func New(opts ...Option) (*Controller, error) {
return nil, err
}
controller.store = store
controller.scheduler = scheduler.NewScheduler(store, controller.workerNotifier, controller.logger)
controller.scheduler = scheduler.NewScheduler(store, controller.workerNotifier,
controller.workerOfflineTimeout, controller.logger)
listener, err := net.Listen("tcp", controller.listenAddr)
if err != nil {