mirror of
https://github.com/moby/moby.git
synced 2026-08-02 22:26:52 +00:00
libnetwork: Always clean up endpoint driver info in sbLeave
In Swarm overlay networks, remote nodes program VXLAN FDB and neighbor entries from the overlay driver's NetworkDB table (overlay_peer_table). Those entries are installed as static/permanent neighbors and are removed only when the endpoint's driver table entries are deleted from NetworkDB and the delete event is gossiped to peers. Endpoint.sbLeave() updates the local endpoint object in the store via storeEndpoint(). If that store update fails (notably datastore.ErrKeyModified from a boltDB CAS conflict during concurrent endpoint operations), sbLeave() returned early without calling deleteDriverInfoFromCluster(). This left the endpoint's overlay_peer_table record behind in NetworkDB, so no DELETE gossip event was emitted, and remote nodes never removed the corresponding neighbor/FDB state. This causes stale permanent ARP/FDB entries to accumulate in overlay network namespaces, eventually blackholing TCP traffic and producing intermittent 502/504 until the service is redeployed. Fix this by removing the early return on storeEndpoint failure. Instead, accumulate errors and continue the full leave operation. Assisted-by: GPT-5.2 (AI) Signed-off-by: Paweł Gronowski <pawel.gronowski@docker.com>
This commit is contained in:
@@ -841,17 +841,23 @@ func (ep *Endpoint) sbLeave(ctx context.Context, sb *Sandbox, n *Network, force
|
||||
}
|
||||
}
|
||||
|
||||
var errs []error
|
||||
|
||||
// Update the store about the sandbox detach only after we
|
||||
// have completed sb.clearNetworkResources above to avoid
|
||||
// spurious logs when cleaning up the sandbox when the daemon
|
||||
// ungracefully exits and restarts before completing sandbox
|
||||
// detach but after store has been updated.
|
||||
if err := n.getController().storeEndpoint(ctx, ep); err != nil {
|
||||
return err
|
||||
errs = append(errs, fmt.Errorf("failed to store endpoint: %w", err))
|
||||
// Do not return early and continue with deleting driver info from the
|
||||
// cluster (NetworkDB) so that remote nodes learn about the endpoint
|
||||
// deletion and don't accumulate stale PERMANENT ARP/FDB entries in
|
||||
// overlay network namespaces.
|
||||
}
|
||||
|
||||
if e := ep.deleteDriverInfoFromCluster(); e != nil {
|
||||
log.G(ctx).WithError(e).Error("Failed to delete endpoint state for endpoint from cluster")
|
||||
if err := ep.deleteDriverInfoFromCluster(); err != nil {
|
||||
log.G(ctx).WithError(err).Error("Failed to delete endpoint state for endpoint from cluster")
|
||||
}
|
||||
|
||||
// When a container is connected to a network, it gets /etc/hosts
|
||||
@@ -871,7 +877,9 @@ func (ep *Endpoint) sbLeave(ctx context.Context, sb *Sandbox, n *Network, force
|
||||
sb.deleteHostsEntries(etcHostsAddrs)
|
||||
|
||||
if !sbInDelete && sb.needDefaultGW() && sb.getEndpointInGWNetwork() == nil {
|
||||
return sb.setupDefaultGW()
|
||||
if err := sb.setupDefaultGW(); err != nil {
|
||||
errs = append(errs, fmt.Errorf("failed to set default gateway: %w", err))
|
||||
}
|
||||
}
|
||||
|
||||
// Disable upstream forwarding if the sandbox lost external connectivity.
|
||||
@@ -902,7 +910,7 @@ func (ep *Endpoint) sbLeave(ctx context.Context, sb *Sandbox, n *Network, force
|
||||
}
|
||||
}
|
||||
|
||||
return nil
|
||||
return errors.Join(errs...)
|
||||
}
|
||||
|
||||
// Delete deletes and detaches this endpoint from the network.
|
||||
|
||||
Reference in New Issue
Block a user