mirror of
https://github.com/psviderski/uncloud.git
synced 2026-08-28 12:03:33 +00:00
feat: add --skip-health flag to bypass health monitoring during container deployment
This commit is contained in:
@@ -26,6 +26,7 @@ type deployOptions struct {
|
|||||||
services []string
|
services []string
|
||||||
noBuild bool
|
noBuild bool
|
||||||
recreate bool
|
recreate bool
|
||||||
|
skipHealth bool
|
||||||
yes bool
|
yes bool
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -61,6 +62,10 @@ func NewDeployCommand() *cobra.Command {
|
|||||||
"One or more Compose profiles to enable.")
|
"One or more Compose profiles to enable.")
|
||||||
cmd.Flags().BoolVar(&opts.recreate, "recreate", false,
|
cmd.Flags().BoolVar(&opts.recreate, "recreate", false,
|
||||||
"Recreate containers even if their configuration and image haven't changed.")
|
"Recreate containers even if their configuration and image haven't changed.")
|
||||||
|
cmd.Flags().BoolVar(&opts.skipHealth, "skip-health", false,
|
||||||
|
"Skip the monitoring period and health checks after starting new containers. Useful for faster emergency "+
|
||||||
|
"deployments.\n"+
|
||||||
|
"Warning: This may cause downtime if new containers fail to start properly.")
|
||||||
cmd.Flags().BoolVarP(&opts.yes, "yes", "y", false,
|
cmd.Flags().BoolVarP(&opts.yes, "yes", "y", false,
|
||||||
"Auto-confirm deployment plan. Should be explicitly set when running non-interactively,\n"+
|
"Auto-confirm deployment plan. Should be explicitly set when running non-interactively,\n"+
|
||||||
"e.g., in CI/CD pipelines. [$UNCLOUD_AUTO_CONFIRM]")
|
"e.g., in CI/CD pipelines. [$UNCLOUD_AUTO_CONFIRM]")
|
||||||
@@ -149,9 +154,9 @@ func runDeploy(ctx context.Context, uncli *cli.CLI, opts deployOptions) error {
|
|||||||
fmt.Println()
|
fmt.Println()
|
||||||
}
|
}
|
||||||
|
|
||||||
var strategy deploy.Strategy
|
strategy := &deploy.RollingStrategy{
|
||||||
if opts.recreate {
|
ForceRecreate: opts.recreate,
|
||||||
strategy = &deploy.RollingStrategy{ForceRecreate: true}
|
SkipHealthMonitor: opts.skipHealth,
|
||||||
}
|
}
|
||||||
composeDeploy, err := compose.NewDeploymentWithStrategy(ctx, clusterClient, project, strategy)
|
composeDeploy, err := compose.NewDeploymentWithStrategy(ctx, clusterClient, project, strategy)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
|
|||||||
@@ -15,6 +15,8 @@ type RunContainerOperation struct {
|
|||||||
ServiceID string
|
ServiceID string
|
||||||
Spec api.ServiceSpec
|
Spec api.ServiceSpec
|
||||||
MachineID string
|
MachineID string
|
||||||
|
// SkipHealthMonitor skips the monitoring period and health checks after starting a container.
|
||||||
|
SkipHealthMonitor bool
|
||||||
}
|
}
|
||||||
|
|
||||||
func (o *RunContainerOperation) Execute(ctx context.Context, cli Client) error {
|
func (o *RunContainerOperation) Execute(ctx context.Context, cli Client) error {
|
||||||
@@ -26,6 +28,10 @@ func (o *RunContainerOperation) Execute(ctx context.Context, cli Client) error {
|
|||||||
return fmt.Errorf("start container: %w", err)
|
return fmt.Errorf("start container: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if o.SkipHealthMonitor {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
opts := api.WaitContainerHealthyOptions{MonitorPeriod: o.Spec.UpdateConfig.MonitorPeriod}
|
opts := api.WaitContainerHealthyOptions{MonitorPeriod: o.Spec.UpdateConfig.MonitorPeriod}
|
||||||
if err = cli.WaitContainerHealthy(ctx, o.ServiceID, resp.ID, opts); err != nil {
|
if err = cli.WaitContainerHealthy(ctx, o.ServiceID, resp.ID, opts); err != nil {
|
||||||
return fmt.Errorf("container '%s/%s' failed to become healthy: %w",
|
return fmt.Errorf("container '%s/%s' failed to become healthy: %w",
|
||||||
@@ -111,6 +117,8 @@ type ReplaceContainerOperation struct {
|
|||||||
OldContainer api.ServiceContainer
|
OldContainer api.ServiceContainer
|
||||||
// Order specifies the update order: "start-first" or "stop-first".
|
// Order specifies the update order: "start-first" or "stop-first".
|
||||||
Order string
|
Order string
|
||||||
|
// SkipHealthMonitor skips the monitoring period and health checks after starting a new container.
|
||||||
|
SkipHealthMonitor bool
|
||||||
}
|
}
|
||||||
|
|
||||||
func (o *ReplaceContainerOperation) Execute(ctx context.Context, cli Client) error {
|
func (o *ReplaceContainerOperation) Execute(ctx context.Context, cli Client) error {
|
||||||
@@ -139,6 +147,7 @@ func (o *ReplaceContainerOperation) Execute(ctx context.Context, cli Client) err
|
|||||||
return fmt.Errorf("start new container: %w", err)
|
return fmt.Errorf("start new container: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if !o.SkipHealthMonitor {
|
||||||
opts := api.WaitContainerHealthyOptions{MonitorPeriod: o.Spec.UpdateConfig.MonitorPeriod}
|
opts := api.WaitContainerHealthyOptions{MonitorPeriod: o.Spec.UpdateConfig.MonitorPeriod}
|
||||||
if err = cli.WaitContainerHealthy(ctx, o.ServiceID, resp.ID, opts); err != nil {
|
if err = cli.WaitContainerHealthy(ctx, o.ServiceID, resp.ID, opts); err != nil {
|
||||||
// New container failed to become healthy. Stop it and roll back to the previous container.
|
// New container failed to become healthy. Stop it and roll back to the previous container.
|
||||||
@@ -168,6 +177,7 @@ func (o *ReplaceContainerOperation) Execute(ctx context.Context, cli Client) err
|
|||||||
|
|
||||||
return healthErr
|
return healthErr
|
||||||
}
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// For start-first, we need to stop before removing.
|
// For start-first, we need to stop before removing.
|
||||||
// For stop-first, the container is already stopped.
|
// For stop-first, the container is already stopped.
|
||||||
|
|||||||
@@ -29,6 +29,8 @@ type RollingStrategy struct {
|
|||||||
// ForceRecreate indicates whether all containers should be recreated during the deployment,
|
// ForceRecreate indicates whether all containers should be recreated during the deployment,
|
||||||
// regardless of whether their specifications have changed.
|
// regardless of whether their specifications have changed.
|
||||||
ForceRecreate bool
|
ForceRecreate bool
|
||||||
|
// SkipHealthMonitor skips the monitoring period and health checks for faster emergency deployments.
|
||||||
|
SkipHealthMonitor bool
|
||||||
|
|
||||||
// state is the current and planned state of the cluster used for scheduling decisions.
|
// state is the current and planned state of the cluster used for scheduling decisions.
|
||||||
state *scheduler.ClusterState
|
state *scheduler.ClusterState
|
||||||
@@ -149,6 +151,7 @@ func (s *RollingStrategy) planReplicated(svc *api.Service, spec api.ServiceSpec)
|
|||||||
ServiceID: plan.ServiceID,
|
ServiceID: plan.ServiceID,
|
||||||
Spec: spec,
|
Spec: spec,
|
||||||
MachineID: m.Id,
|
MachineID: m.Id,
|
||||||
|
SkipHealthMonitor: s.SkipHealthMonitor,
|
||||||
})
|
})
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
@@ -171,6 +174,7 @@ func (s *RollingStrategy) planReplicated(svc *api.Service, spec api.ServiceSpec)
|
|||||||
MachineID: m.Id,
|
MachineID: m.Id,
|
||||||
OldContainer: ctr,
|
OldContainer: ctr,
|
||||||
Order: order,
|
Order: order,
|
||||||
|
SkipHealthMonitor: s.SkipHealthMonitor,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -216,7 +220,8 @@ func (s *RollingStrategy) planGlobal(svc *api.Service, spec api.ServiceSpec) (Pl
|
|||||||
|
|
||||||
for _, m := range availableMachines {
|
for _, m := range availableMachines {
|
||||||
containers := containersOnMachine[m.Info.Id]
|
containers := containersOnMachine[m.Info.Id]
|
||||||
ops, err := reconcileGlobalContainer(containers, spec, plan.ServiceID, m.Info.Id, s.ForceRecreate)
|
ops, err := reconcileGlobalContainer(
|
||||||
|
containers, spec, plan.ServiceID, m.Info.Id, s.ForceRecreate, s.SkipHealthMonitor)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return plan, err
|
return plan, err
|
||||||
}
|
}
|
||||||
@@ -242,7 +247,8 @@ func (s *RollingStrategy) planGlobal(svc *api.Service, spec api.ServiceSpec) (Pl
|
|||||||
// It ensures exactly one container with the desired spec is running on the machine by creating a new container and
|
// It ensures exactly one container with the desired spec is running on the machine by creating a new container and
|
||||||
// removing old ones. If there is a host port conflict, it stops the old container before starting a new one.
|
// removing old ones. If there is a host port conflict, it stops the old container before starting a new one.
|
||||||
func reconcileGlobalContainer(
|
func reconcileGlobalContainer(
|
||||||
containers []api.MachineServiceContainer, spec api.ServiceSpec, serviceID, machineID string, forceRecreate bool,
|
containers []api.MachineServiceContainer, spec api.ServiceSpec, serviceID, machineID string,
|
||||||
|
forceRecreate, skipHealthCheck bool,
|
||||||
) ([]operation.Operation, error) {
|
) ([]operation.Operation, error) {
|
||||||
var ops []operation.Operation
|
var ops []operation.Operation
|
||||||
|
|
||||||
@@ -252,6 +258,7 @@ func reconcileGlobalContainer(
|
|||||||
ServiceID: serviceID,
|
ServiceID: serviceID,
|
||||||
Spec: spec,
|
Spec: spec,
|
||||||
MachineID: machineID,
|
MachineID: machineID,
|
||||||
|
SkipHealthMonitor: skipHealthCheck,
|
||||||
})
|
})
|
||||||
return ops, nil
|
return ops, nil
|
||||||
}
|
}
|
||||||
@@ -327,6 +334,7 @@ func reconcileGlobalContainer(
|
|||||||
MachineID: machineID,
|
MachineID: machineID,
|
||||||
OldContainer: containerToReplace.Container,
|
OldContainer: containerToReplace.Container,
|
||||||
Order: order,
|
Order: order,
|
||||||
|
SkipHealthMonitor: skipHealthCheck,
|
||||||
})
|
})
|
||||||
|
|
||||||
// Remove any other containers (there shouldn't be any in normal operation).
|
// Remove any other containers (there shouldn't be any in normal operation).
|
||||||
@@ -345,6 +353,7 @@ func reconcileGlobalContainer(
|
|||||||
ServiceID: serviceID,
|
ServiceID: serviceID,
|
||||||
Spec: spec,
|
Spec: spec,
|
||||||
MachineID: machineID,
|
MachineID: machineID,
|
||||||
|
SkipHealthMonitor: skipHealthCheck,
|
||||||
})
|
})
|
||||||
for _, c := range containers {
|
for _, c := range containers {
|
||||||
ops = append(ops, &operation.RemoveContainerOperation{
|
ops = append(ops, &operation.RemoveContainerOperation{
|
||||||
|
|||||||
@@ -401,7 +401,9 @@ func TestReconcileGlobalContainer(t *testing.T) {
|
|||||||
|
|
||||||
for _, tt := range tests {
|
for _, tt := range tests {
|
||||||
t.Run(tt.name, func(t *testing.T) {
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
ops, err := reconcileGlobalContainer(tt.containers, tt.spec, "service-1", "machine-1", tt.forceRecreate)
|
ops, err := reconcileGlobalContainer(
|
||||||
|
tt.containers, tt.spec, "service-1", "machine-1", tt.forceRecreate, false,
|
||||||
|
)
|
||||||
assert.NoError(t, err)
|
assert.NoError(t, err)
|
||||||
assertOperationsEqual(t, tt.expectedOps, ops)
|
assertOperationsEqual(t, tt.expectedOps, ops)
|
||||||
})
|
})
|
||||||
|
|||||||
Reference in New Issue
Block a user