mirror of
https://github.com/psviderski/uncloud.git
synced 2026-08-26 19:13:34 +00:00
feat: add --skip-health flag to bypass health monitoring during container deployment
This commit is contained in:
@@ -15,6 +15,8 @@ type RunContainerOperation struct {
|
||||
ServiceID string
|
||||
Spec api.ServiceSpec
|
||||
MachineID string
|
||||
// SkipHealthMonitor skips the monitoring period and health checks after starting a container.
|
||||
SkipHealthMonitor bool
|
||||
}
|
||||
|
||||
func (o *RunContainerOperation) Execute(ctx context.Context, cli Client) error {
|
||||
@@ -26,6 +28,10 @@ func (o *RunContainerOperation) Execute(ctx context.Context, cli Client) error {
|
||||
return fmt.Errorf("start container: %w", err)
|
||||
}
|
||||
|
||||
if o.SkipHealthMonitor {
|
||||
return nil
|
||||
}
|
||||
|
||||
opts := api.WaitContainerHealthyOptions{MonitorPeriod: o.Spec.UpdateConfig.MonitorPeriod}
|
||||
if err = cli.WaitContainerHealthy(ctx, o.ServiceID, resp.ID, opts); err != nil {
|
||||
return fmt.Errorf("container '%s/%s' failed to become healthy: %w",
|
||||
@@ -111,6 +117,8 @@ type ReplaceContainerOperation struct {
|
||||
OldContainer api.ServiceContainer
|
||||
// Order specifies the update order: "start-first" or "stop-first".
|
||||
Order string
|
||||
// SkipHealthMonitor skips the monitoring period and health checks after starting a new container.
|
||||
SkipHealthMonitor bool
|
||||
}
|
||||
|
||||
func (o *ReplaceContainerOperation) Execute(ctx context.Context, cli Client) error {
|
||||
@@ -139,34 +147,36 @@ func (o *ReplaceContainerOperation) Execute(ctx context.Context, cli Client) err
|
||||
return fmt.Errorf("start new container: %w", err)
|
||||
}
|
||||
|
||||
opts := api.WaitContainerHealthyOptions{MonitorPeriod: o.Spec.UpdateConfig.MonitorPeriod}
|
||||
if err = cli.WaitContainerHealthy(ctx, o.ServiceID, resp.ID, opts); err != nil {
|
||||
// New container failed to become healthy. Stop it and roll back to the previous container.
|
||||
// Don't remove the new stopped container to allow users to inspect logs and state.
|
||||
// TODO: collect logs from the new container and include in the error message to speed up debugging.
|
||||
if !o.SkipHealthMonitor {
|
||||
opts := api.WaitContainerHealthyOptions{MonitorPeriod: o.Spec.UpdateConfig.MonitorPeriod}
|
||||
if err = cli.WaitContainerHealthy(ctx, o.ServiceID, resp.ID, opts); err != nil {
|
||||
// New container failed to become healthy. Stop it and roll back to the previous container.
|
||||
// Don't remove the new stopped container to allow users to inspect logs and state.
|
||||
// TODO: collect logs from the new container and include in the error message to speed up debugging.
|
||||
|
||||
// Use context without progress to not overwrite the container Unhealthy status with Stopped.
|
||||
ctxWithoutProgress := progress.WithContextWriter(ctx, nil)
|
||||
_ = cli.StopContainer(ctxWithoutProgress, o.ServiceID, resp.ID, container.StopOptions{})
|
||||
// Use context without progress to not overwrite the container Unhealthy status with Stopped.
|
||||
ctxWithoutProgress := progress.WithContextWriter(ctx, nil)
|
||||
_ = cli.StopContainer(ctxWithoutProgress, o.ServiceID, resp.ID, container.StopOptions{})
|
||||
|
||||
newCtr := fmt.Sprintf("%s/%s", o.Spec.Name, stringid.TruncateID(resp.ID))
|
||||
healthErr := fmt.Errorf(
|
||||
"new container '%s' failed to become healthy: %w. "+
|
||||
"It's stopped and available for inspection. Fetch logs with 'uc logs %s'",
|
||||
newCtr, err, o.Spec.Name,
|
||||
)
|
||||
newCtr := fmt.Sprintf("%s/%s", o.Spec.Name, stringid.TruncateID(resp.ID))
|
||||
healthErr := fmt.Errorf(
|
||||
"new container '%s' failed to become healthy: %w. "+
|
||||
"It's stopped and available for inspection. Fetch logs with 'uc logs %s'",
|
||||
newCtr, err, o.Spec.Name,
|
||||
)
|
||||
|
||||
if stopFirst && wasRunning {
|
||||
// Restart the old container only if it was running before we stopped it.
|
||||
oldCtr := fmt.Sprintf("%s/%s", o.OldContainer.ServiceSpec.Name, o.OldContainer.ShortID())
|
||||
if rollbackErr := cli.StartContainer(ctx, o.ServiceID, o.OldContainer.ID); rollbackErr != nil {
|
||||
return fmt.Errorf("%w. Rolled back to old container '%s' but failed to restart it: %w",
|
||||
healthErr, oldCtr, rollbackErr)
|
||||
if stopFirst && wasRunning {
|
||||
// Restart the old container only if it was running before we stopped it.
|
||||
oldCtr := fmt.Sprintf("%s/%s", o.OldContainer.ServiceSpec.Name, o.OldContainer.ShortID())
|
||||
if rollbackErr := cli.StartContainer(ctx, o.ServiceID, o.OldContainer.ID); rollbackErr != nil {
|
||||
return fmt.Errorf("%w. Rolled back to old container '%s' but failed to restart it: %w",
|
||||
healthErr, oldCtr, rollbackErr)
|
||||
}
|
||||
return fmt.Errorf("%w. Rolled back to old container '%s'", healthErr, oldCtr)
|
||||
}
|
||||
return fmt.Errorf("%w. Rolled back to old container '%s'", healthErr, oldCtr)
|
||||
}
|
||||
|
||||
return healthErr
|
||||
return healthErr
|
||||
}
|
||||
}
|
||||
|
||||
// For start-first, we need to stop before removing.
|
||||
|
||||
@@ -29,6 +29,8 @@ type RollingStrategy struct {
|
||||
// ForceRecreate indicates whether all containers should be recreated during the deployment,
|
||||
// regardless of whether their specifications have changed.
|
||||
ForceRecreate bool
|
||||
// SkipHealthMonitor skips the monitoring period and health checks for faster emergency deployments.
|
||||
SkipHealthMonitor bool
|
||||
|
||||
// state is the current and planned state of the cluster used for scheduling decisions.
|
||||
state *scheduler.ClusterState
|
||||
@@ -146,9 +148,10 @@ func (s *RollingStrategy) planReplicated(svc *api.Service, spec api.ServiceSpec)
|
||||
if len(containers) == 0 {
|
||||
// No more existing containers on this machine, create a new one.
|
||||
plan.Operations = append(plan.Operations, &operation.RunContainerOperation{
|
||||
ServiceID: plan.ServiceID,
|
||||
Spec: spec,
|
||||
MachineID: m.Id,
|
||||
ServiceID: plan.ServiceID,
|
||||
Spec: spec,
|
||||
MachineID: m.Id,
|
||||
SkipHealthMonitor: s.SkipHealthMonitor,
|
||||
})
|
||||
continue
|
||||
}
|
||||
@@ -166,11 +169,12 @@ func (s *RollingStrategy) planReplicated(svc *api.Service, spec api.ServiceSpec)
|
||||
// Replace the old container with a new one.
|
||||
order := determineUpdateOrder(ctr, spec)
|
||||
plan.Operations = append(plan.Operations, &operation.ReplaceContainerOperation{
|
||||
ServiceID: plan.ServiceID,
|
||||
Spec: spec,
|
||||
MachineID: m.Id,
|
||||
OldContainer: ctr,
|
||||
Order: order,
|
||||
ServiceID: plan.ServiceID,
|
||||
Spec: spec,
|
||||
MachineID: m.Id,
|
||||
OldContainer: ctr,
|
||||
Order: order,
|
||||
SkipHealthMonitor: s.SkipHealthMonitor,
|
||||
})
|
||||
}
|
||||
|
||||
@@ -216,7 +220,8 @@ func (s *RollingStrategy) planGlobal(svc *api.Service, spec api.ServiceSpec) (Pl
|
||||
|
||||
for _, m := range availableMachines {
|
||||
containers := containersOnMachine[m.Info.Id]
|
||||
ops, err := reconcileGlobalContainer(containers, spec, plan.ServiceID, m.Info.Id, s.ForceRecreate)
|
||||
ops, err := reconcileGlobalContainer(
|
||||
containers, spec, plan.ServiceID, m.Info.Id, s.ForceRecreate, s.SkipHealthMonitor)
|
||||
if err != nil {
|
||||
return plan, err
|
||||
}
|
||||
@@ -242,16 +247,18 @@ func (s *RollingStrategy) planGlobal(svc *api.Service, spec api.ServiceSpec) (Pl
|
||||
// It ensures exactly one container with the desired spec is running on the machine by creating a new container and
|
||||
// removing old ones. If there is a host port conflict, it stops the old container before starting a new one.
|
||||
func reconcileGlobalContainer(
|
||||
containers []api.MachineServiceContainer, spec api.ServiceSpec, serviceID, machineID string, forceRecreate bool,
|
||||
containers []api.MachineServiceContainer, spec api.ServiceSpec, serviceID, machineID string,
|
||||
forceRecreate, skipHealthCheck bool,
|
||||
) ([]operation.Operation, error) {
|
||||
var ops []operation.Operation
|
||||
|
||||
if len(containers) == 0 {
|
||||
// No containers on this machine, create a new one.
|
||||
ops = append(ops, &operation.RunContainerOperation{
|
||||
ServiceID: serviceID,
|
||||
Spec: spec,
|
||||
MachineID: machineID,
|
||||
ServiceID: serviceID,
|
||||
Spec: spec,
|
||||
MachineID: machineID,
|
||||
SkipHealthMonitor: skipHealthCheck,
|
||||
})
|
||||
return ops, nil
|
||||
}
|
||||
@@ -322,11 +329,12 @@ func reconcileGlobalContainer(
|
||||
// Replace the running container with a new one.
|
||||
order := determineUpdateOrder(containerToReplace.Container, spec)
|
||||
ops = append(ops, &operation.ReplaceContainerOperation{
|
||||
ServiceID: serviceID,
|
||||
Spec: spec,
|
||||
MachineID: machineID,
|
||||
OldContainer: containerToReplace.Container,
|
||||
Order: order,
|
||||
ServiceID: serviceID,
|
||||
Spec: spec,
|
||||
MachineID: machineID,
|
||||
OldContainer: containerToReplace.Container,
|
||||
Order: order,
|
||||
SkipHealthMonitor: skipHealthCheck,
|
||||
})
|
||||
|
||||
// Remove any other containers (there shouldn't be any in normal operation).
|
||||
@@ -342,9 +350,10 @@ func reconcileGlobalContainer(
|
||||
} else {
|
||||
// No running containers, create a new one and remove all stopped containers.
|
||||
ops = append(ops, &operation.RunContainerOperation{
|
||||
ServiceID: serviceID,
|
||||
Spec: spec,
|
||||
MachineID: machineID,
|
||||
ServiceID: serviceID,
|
||||
Spec: spec,
|
||||
MachineID: machineID,
|
||||
SkipHealthMonitor: skipHealthCheck,
|
||||
})
|
||||
for _, c := range containers {
|
||||
ops = append(ops, &operation.RemoveContainerOperation{
|
||||
|
||||
@@ -401,7 +401,9 @@ func TestReconcileGlobalContainer(t *testing.T) {
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
ops, err := reconcileGlobalContainer(tt.containers, tt.spec, "service-1", "machine-1", tt.forceRecreate)
|
||||
ops, err := reconcileGlobalContainer(
|
||||
tt.containers, tt.spec, "service-1", "machine-1", tt.forceRecreate, false,
|
||||
)
|
||||
assert.NoError(t, err)
|
||||
assertOperationsEqual(t, tt.expectedOps, ops)
|
||||
})
|
||||
|
||||
Reference in New Issue
Block a user