cri: skip failed container instead of dropping entire sandbox metrics

When collectContainerMetrics fails for a single container, the
goroutine was returning nil which exited the loop entirely. The
sandbox metrics, including pod-level network metrics and any
successfully collected container metrics, were never appended to
the response.

Replace return nil with continue so that individual container
failures only skip that container.

Signed-off-by: Damien Grisonnet <dgrisonn@redhat.com>
This commit is contained in:
Damien Grisonnet
2026-08-03 09:06:46 +02:00
parent 56bc534e7e
commit 34524e8a68

View File

@@ -109,7 +109,7 @@ func (c *criService) ListPodSandboxMetrics(ctx context.Context, r *runtime.ListP
case errdefs.IsUnavailable(err), errdefs.IsNotFound(err):
log.G(gctx).WithField("podsandboxid", sandbox.ID).WithField("containerid", container.ID).WithError(err).Error("failed to get container metrics, this is likely a transient error")
// Don't return error for transient issues, just log and continue
return nil
continue
case errdefs.IsCanceled(err):
log.G(gctx).WithField("podsandboxid", sandbox.ID).WithField("containerid", container.ID).WithError(err).Debug("metrics collection cancelled")
// Return the cancellation error to stop other goroutines
@@ -117,7 +117,7 @@ func (c *criService) ListPodSandboxMetrics(ctx context.Context, r *runtime.ListP
default:
log.G(gctx).WithField("podsandboxid", sandbox.ID).WithField("containerid", container.ID).WithError(err).Error("failed to collect container metrics")
// Don't return error for individual failures, just log and continue
return nil
continue
}
}