feat: support compose stop_grace_period to change timeout before sending SIGKILL

This commit is contained in:
Pasha Sviderski
2026-03-04 14:07:22 +10:00
parent 1f328e99d5
commit d9934b9dba
7 changed files with 125 additions and 90 deletions
+28 -12
View File
@@ -3,6 +3,7 @@ package operation
import (
"context"
"fmt"
"time"
"github.com/docker/compose/v2/pkg/progress"
"github.com/docker/docker/api/types/container"
@@ -53,13 +54,14 @@ func (o *RunContainerOperation) String() string {
// StopContainerOperation stops a container on a specific machine.
type StopContainerOperation struct {
ServiceID string
ContainerID string
MachineID string
ServiceID string
ContainerID string
MachineID string
StopGracePeriod *time.Duration
}
func (o *StopContainerOperation) Execute(ctx context.Context, cli Client) error {
if err := cli.StopContainer(ctx, o.ServiceID, o.ContainerID, container.StopOptions{}); err != nil {
if err := cli.StopContainer(ctx, o.ServiceID, o.ContainerID, stopOptions(o.StopGracePeriod)); err != nil {
return fmt.Errorf("stop container: %w", err)
}
return nil
@@ -78,15 +80,18 @@ func (o *StopContainerOperation) String() string {
// RemoveContainerOperation stops and removes a container from a specific machine.
type RemoveContainerOperation struct {
MachineID string
Container api.ServiceContainer
MachineID string
Container api.ServiceContainer
StopGracePeriod *time.Duration
}
func (o *RemoveContainerOperation) Execute(ctx context.Context, cli Client) error {
if err := cli.StopContainer(ctx, o.Container.ServiceID(), o.Container.ID, container.StopOptions{}); err != nil {
err := cli.StopContainer(ctx, o.Container.ServiceID(), o.Container.ID, stopOptions(o.StopGracePeriod))
if err != nil {
return fmt.Errorf("stop container: %w", err)
}
if err := cli.RemoveContainer(ctx, o.Container.ServiceID(), o.Container.ID, container.RemoveOptions{
if err = cli.RemoveContainer(ctx, o.Container.ServiceID(), o.Container.ID, container.RemoveOptions{
// Remove anonymous volumes created by the container.
RemoveVolumes: true,
}); err != nil {
@@ -119,6 +124,7 @@ type ReplaceContainerOperation struct {
Order string
// SkipHealthMonitor skips the monitoring period and health checks after starting a new container.
SkipHealthMonitor bool
StopGracePeriod *time.Duration
}
func (o *ReplaceContainerOperation) Execute(ctx context.Context, cli Client) error {
@@ -133,7 +139,8 @@ func (o *ReplaceContainerOperation) Execute(ctx context.Context, cli Client) err
}
wasRunning = ctr.Container.State.Running
if wasRunning {
if err = cli.StopContainer(ctx, o.ServiceID, o.OldContainer.ID, container.StopOptions{}); err != nil {
err = cli.StopContainer(ctx, o.ServiceID, o.OldContainer.ID, stopOptions(o.StopGracePeriod))
if err != nil {
return fmt.Errorf("stop old container: %w", err)
}
}
@@ -156,7 +163,7 @@ func (o *ReplaceContainerOperation) Execute(ctx context.Context, cli Client) err
// Use context without progress to not overwrite the container Unhealthy status with Stopped.
ctxWithoutProgress := progress.WithContextWriter(ctx, nil)
_ = cli.StopContainer(ctxWithoutProgress, o.ServiceID, resp.ID, container.StopOptions{})
_ = cli.StopContainer(ctxWithoutProgress, o.ServiceID, resp.ID, stopOptions(o.StopGracePeriod))
newCtr := fmt.Sprintf("%s/%s", o.Spec.Name, stringid.TruncateID(resp.ID))
healthErr := fmt.Errorf(
@@ -186,12 +193,12 @@ func (o *ReplaceContainerOperation) Execute(ctx context.Context, cli Client) err
// There still might be a brief downtime (for a 1 replica service) when Caddy doesn't know about
// the new container but we're stopping the old container. We should somehow ensure Caddy is updated
// with the new container before we stop the old one to avoid this downtime.
if err := cli.StopContainer(ctx, o.ServiceID, o.OldContainer.ID, container.StopOptions{}); err != nil {
if err = cli.StopContainer(ctx, o.ServiceID, o.OldContainer.ID, stopOptions(o.StopGracePeriod)); err != nil {
return fmt.Errorf("stop old container: %w", err)
}
}
if err := cli.RemoveContainer(ctx, o.ServiceID, o.OldContainer.ID, container.RemoveOptions{
if err = cli.RemoveContainer(ctx, o.ServiceID, o.OldContainer.ID, container.RemoveOptions{
RemoveVolumes: true,
}); err != nil {
return fmt.Errorf("remove old container: %w", err)
@@ -209,3 +216,12 @@ func (o *ReplaceContainerOperation) String() string {
return fmt.Sprintf("ReplaceContainerOperation[machine_id=%s service_id=%s old_container_id=%s order=%s]",
o.MachineID, o.ServiceID, o.OldContainer.ID, o.Order)
}
// stopOptions converts a stop grace period duration to Docker container stop options.
func stopOptions(gracePeriod *time.Duration) container.StopOptions {
if gracePeriod == nil {
return container.StopOptions{}
}
t := int(gracePeriod.Seconds())
return container.StopOptions{Timeout: &t}
}
+21 -13
View File
@@ -175,6 +175,7 @@ func (s *RollingStrategy) planReplicated(svc *api.Service, spec api.ServiceSpec)
OldContainer: ctr,
Order: order,
SkipHealthMonitor: s.SkipHealthMonitor,
StopGracePeriod: spec.StopGracePeriod,
})
}
@@ -182,8 +183,9 @@ func (s *RollingStrategy) planReplicated(svc *api.Service, spec api.ServiceSpec)
for mid, containers := range containersOnMachine {
for _, c := range containers {
plan.Operations = append(plan.Operations, &operation.RemoveContainerOperation{
MachineID: mid,
Container: c,
MachineID: mid,
Container: c,
StopGracePeriod: spec.StopGracePeriod,
})
}
}
@@ -234,8 +236,9 @@ func (s *RollingStrategy) planGlobal(svc *api.Service, spec api.ServiceSpec) (Pl
for _, containers := range containersOnMachine {
for _, c := range containers {
plan.Operations = append(plan.Operations, &operation.RemoveContainerOperation{
MachineID: c.MachineID,
Container: c.Container,
MachineID: c.MachineID,
Container: c.Container,
StopGracePeriod: spec.StopGracePeriod,
})
}
}
@@ -286,8 +289,9 @@ func reconcileGlobalContainer(
continue
}
ops = append(ops, &operation.RemoveContainerOperation{
MachineID: old.MachineID,
Container: old.Container,
MachineID: old.MachineID,
Container: old.Container,
StopGracePeriod: spec.StopGracePeriod,
})
}
break
@@ -319,9 +323,10 @@ func reconcileGlobalContainer(
conflictingPorts, err := c.Container.ConflictingServicePorts(spec.Ports)
if err != nil || len(conflictingPorts) > 0 {
ops = append(ops, &operation.StopContainerOperation{
ServiceID: serviceID,
ContainerID: c.Container.ID,
MachineID: machineID,
ServiceID: serviceID,
ContainerID: c.Container.ID,
MachineID: machineID,
StopGracePeriod: spec.StopGracePeriod,
})
}
}
@@ -335,6 +340,7 @@ func reconcileGlobalContainer(
OldContainer: containerToReplace.Container,
Order: order,
SkipHealthMonitor: skipHealthCheck,
StopGracePeriod: spec.StopGracePeriod,
})
// Remove any other containers (there shouldn't be any in normal operation).
@@ -343,8 +349,9 @@ func reconcileGlobalContainer(
continue
}
ops = append(ops, &operation.RemoveContainerOperation{
MachineID: c.MachineID,
Container: c.Container,
MachineID: c.MachineID,
Container: c.Container,
StopGracePeriod: spec.StopGracePeriod,
})
}
} else {
@@ -357,8 +364,9 @@ func reconcileGlobalContainer(
})
for _, c := range containers {
ops = append(ops, &operation.RemoveContainerOperation{
MachineID: c.MachineID,
Container: c.Container,
MachineID: c.MachineID,
Container: c.Container,
StopGracePeriod: spec.StopGracePeriod,
})
}
}