mirror of
https://github.com/containerd/containerd.git
synced 2026-08-09 01:21:15 +00:00
Add support for EROFS fsmerge feature
EROFS has supported a tiny metadata-only image to reference external blobs since Linux 5.16. This eliminates the need to mount each EROFS layer one by one and is also useful for VM-based containers (e.g. nerdbox and Kata containers.) Similar to LCOW/CimFS, `snapshots.UnpackKeyPrefix` is used to trigger fsmerge generation (typically < 100 ms) on demand in Prepare(). In the future, we can also generate fsmeta in Commit() of the final unpacking layer (by introducing an annotation to keep the chainID). However, in the case of intermediate layer reuse, the Prepare() handling will still be required. ```toml [plugins."io.containerd.snapshotter.v1.erofs"] max_unmerged_layers = 1 # enable fsmerge if image layers >= 2 ``` Signed-off-by: Gao Xiang <hsiangkao@linux.alibaba.com>
This commit is contained in:
@@ -239,8 +239,6 @@ For the EROFS differ:
|
|||||||
|
|
||||||
## TODO
|
## TODO
|
||||||
|
|
||||||
- EROFS Flatten filesystem support (EROFS fsmerge feature);
|
|
||||||
|
|
||||||
- ID-mapped mount spport;
|
- ID-mapped mount spport;
|
||||||
|
|
||||||
- DMVerity support.
|
- DMVerity support.
|
||||||
|
|||||||
@@ -20,8 +20,11 @@ import (
|
|||||||
"context"
|
"context"
|
||||||
"fmt"
|
"fmt"
|
||||||
"os"
|
"os"
|
||||||
|
"os/exec"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"runtime"
|
"runtime"
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
|
||||||
"github.com/containerd/continuity/fs"
|
"github.com/containerd/continuity/fs"
|
||||||
"github.com/containerd/errdefs"
|
"github.com/containerd/errdefs"
|
||||||
@@ -43,6 +46,8 @@ type SnapshotterConfig struct {
|
|||||||
setImmutable bool
|
setImmutable bool
|
||||||
// defaultSize creates a default size writable layer for active snapshots
|
// defaultSize creates a default size writable layer for active snapshots
|
||||||
defaultSize int64
|
defaultSize int64
|
||||||
|
// fsMergeThreshold (>0) enables fsmerge when the number of image layers exceeds this value
|
||||||
|
fsMergeThreshold uint
|
||||||
}
|
}
|
||||||
|
|
||||||
// Opt is an option to configure the erofs snapshotter
|
// Opt is an option to configure the erofs snapshotter
|
||||||
@@ -76,6 +81,13 @@ func WithDefaultSize(size int64) Opt {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// WithFsMergeThreshold (>0) enables fsmerge when the number of image layers exceeds this value
|
||||||
|
func WithFsMergeThreshold(v uint) Opt {
|
||||||
|
return func(config *SnapshotterConfig) {
|
||||||
|
config.fsMergeThreshold = v
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
type MetaStore interface {
|
type MetaStore interface {
|
||||||
TransactionContext(ctx context.Context, writable bool) (context.Context, storage.Transactor, error)
|
TransactionContext(ctx context.Context, writable bool) (context.Context, storage.Transactor, error)
|
||||||
WithTransaction(ctx context.Context, writable bool, fn storage.TransactionCallback) error
|
WithTransaction(ctx context.Context, writable bool, fn storage.TransactionCallback) error
|
||||||
@@ -83,13 +95,14 @@ type MetaStore interface {
|
|||||||
}
|
}
|
||||||
|
|
||||||
type snapshotter struct {
|
type snapshotter struct {
|
||||||
root string
|
root string
|
||||||
ms *storage.MetaStore
|
ms *storage.MetaStore
|
||||||
ovlOptions []string
|
ovlOptions []string
|
||||||
enableFsverity bool
|
enableFsverity bool
|
||||||
setImmutable bool
|
setImmutable bool
|
||||||
defaultWritable int64
|
defaultWritable int64
|
||||||
blockMode bool
|
blockMode bool
|
||||||
|
fsMergeThreshold uint
|
||||||
}
|
}
|
||||||
|
|
||||||
// NewSnapshotter returns a Snapshotter which uses EROFS+OverlayFS. The layers
|
// NewSnapshotter returns a Snapshotter which uses EROFS+OverlayFS. The layers
|
||||||
@@ -139,13 +152,14 @@ func NewSnapshotter(root string, opts ...Opt) (snapshots.Snapshotter, error) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
return &snapshotter{
|
return &snapshotter{
|
||||||
root: root,
|
root: root,
|
||||||
ms: ms,
|
ms: ms,
|
||||||
ovlOptions: config.ovlOptions,
|
ovlOptions: config.ovlOptions,
|
||||||
enableFsverity: config.enableFsverity,
|
enableFsverity: config.enableFsverity,
|
||||||
setImmutable: config.setImmutable,
|
setImmutable: config.setImmutable,
|
||||||
defaultWritable: config.defaultSize,
|
defaultWritable: config.defaultSize,
|
||||||
blockMode: config.defaultSize > 0,
|
blockMode: config.defaultSize > 0,
|
||||||
|
fsMergeThreshold: config.fsMergeThreshold,
|
||||||
}, nil
|
}, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -171,6 +185,10 @@ func (s *snapshotter) layerBlobPath(id string) string {
|
|||||||
return filepath.Join(s.root, "snapshots", id, "layer.erofs")
|
return filepath.Join(s.root, "snapshots", id, "layer.erofs")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func (s *snapshotter) fsMetaPath(id string) string {
|
||||||
|
return filepath.Join(s.root, "snapshots", id, "fsmeta.erofs")
|
||||||
|
}
|
||||||
|
|
||||||
func (s *snapshotter) lowerPath(id string) (string, error) {
|
func (s *snapshotter) lowerPath(id string) (string, error) {
|
||||||
layerBlob := s.layerBlobPath(id)
|
layerBlob := s.layerBlobPath(id)
|
||||||
if _, err := os.Stat(layerBlob); err != nil {
|
if _, err := os.Stat(layerBlob); err != nil {
|
||||||
@@ -205,6 +223,24 @@ func (s *snapshotter) prepareDirectory(ctx context.Context, snapshotDir string,
|
|||||||
return td, nil
|
return td, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func (s *snapshotter) mountFsMeta(snap storage.Snapshot, id int) (mount.Mount, bool) {
|
||||||
|
mergedMeta := s.fsMetaPath(snap.ParentIDs[id])
|
||||||
|
if fi, err := os.Stat(mergedMeta); err != nil || fi.Size() == 0 {
|
||||||
|
return mount.Mount{}, false
|
||||||
|
}
|
||||||
|
|
||||||
|
m := mount.Mount{
|
||||||
|
Source: mergedMeta,
|
||||||
|
Type: "erofs",
|
||||||
|
Options: []string{"ro", "loop"},
|
||||||
|
}
|
||||||
|
for j := len(snap.ParentIDs) - 1; j >= id; j-- {
|
||||||
|
path := s.layerBlobPath(snap.ParentIDs[j])
|
||||||
|
m.Options = append(m.Options, "device="+path)
|
||||||
|
}
|
||||||
|
return m, true
|
||||||
|
}
|
||||||
|
|
||||||
func (s *snapshotter) mounts(snap storage.Snapshot, _ snapshots.Info) ([]mount.Mount, error) {
|
func (s *snapshotter) mounts(snap storage.Snapshot, _ snapshots.Info) ([]mount.Mount, error) {
|
||||||
var options []string
|
var options []string
|
||||||
|
|
||||||
@@ -313,6 +349,16 @@ func (s *snapshotter) mounts(snap storage.Snapshot, _ snapshots.Info) ([]mount.M
|
|||||||
|
|
||||||
first := len(mounts)
|
first := len(mounts)
|
||||||
for i := range snap.ParentIDs {
|
for i := range snap.ParentIDs {
|
||||||
|
// If a merged fsmeta is valid for this layer, skip the remaining bottom layers.
|
||||||
|
// Why? Because bottom layers have been flattened with the thin fsmeta.
|
||||||
|
if s.fsMergeThreshold > 0 {
|
||||||
|
if m, ok := s.mountFsMeta(snap, i); ok {
|
||||||
|
mounts = append(mounts, m)
|
||||||
|
first = len(mounts) - 1
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
layerBlob, err := s.lowerPath(snap.ParentIDs[i])
|
layerBlob, err := s.lowerPath(snap.ParentIDs[i])
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
@@ -370,7 +416,6 @@ func (s *snapshotter) createSnapshot(ctx context.Context, kind snapshots.Kind, k
|
|||||||
}
|
}
|
||||||
|
|
||||||
if err := s.ms.WithTransaction(ctx, true, func(ctx context.Context) (err error) {
|
if err := s.ms.WithTransaction(ctx, true, func(ctx context.Context) (err error) {
|
||||||
|
|
||||||
snap, err = storage.CreateSnapshot(ctx, kind, key, parent, opts...)
|
snap, err = storage.CreateSnapshot(ctx, kind, key, parent, opts...)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("failed to create snapshot: %w", err)
|
return fmt.Errorf("failed to create snapshot: %w", err)
|
||||||
@@ -392,11 +437,16 @@ func (s *snapshotter) createSnapshot(ctx context.Context, kind snapshots.Kind, k
|
|||||||
return fmt.Errorf("failed to rename: %w", err)
|
return fmt.Errorf("failed to rename: %w", err)
|
||||||
}
|
}
|
||||||
td = ""
|
td = ""
|
||||||
|
|
||||||
return nil
|
return nil
|
||||||
}); err != nil {
|
}); err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Generate fsmeta outside of the transaction since it's unnecessary.
|
||||||
|
// Also ignore all errors since it's a nice-to-have stuff.
|
||||||
|
if !strings.Contains(key, snapshots.UnpackKeyPrefix) {
|
||||||
|
s.generateFsMeta(ctx, snap.ParentIDs)
|
||||||
|
}
|
||||||
return s.mounts(snap, info)
|
return s.mounts(snap, info)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -445,6 +495,47 @@ func (s *snapshotter) commitBlock(ctx context.Context, layerBlob string, id stri
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// generate a metadata-only EROFS fsmeta.erofs if all EROFS layer blobs are valid
|
||||||
|
func (s *snapshotter) generateFsMeta(ctx context.Context, snapIDs []string) {
|
||||||
|
var blobs []string
|
||||||
|
|
||||||
|
if s.fsMergeThreshold == 0 || uint(len(snapIDs)) <= s.fsMergeThreshold {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
t1 := time.Now()
|
||||||
|
mergedMeta := s.fsMetaPath(snapIDs[0])
|
||||||
|
// If the empty placeholder cannot be created (mainly due to os.IsExist), just return
|
||||||
|
if _, err := os.OpenFile(mergedMeta, os.O_CREATE|os.O_EXCL, 0644); err != nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
for i := len(snapIDs) - 1; i >= 0; i-- {
|
||||||
|
blob := s.layerBlobPath(snapIDs[i])
|
||||||
|
if _, err := os.Stat(blob); err != nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
blobs = append(blobs, blob)
|
||||||
|
}
|
||||||
|
tmpMergedMeta := mergedMeta + ".tmp"
|
||||||
|
args := append([]string{"--aufs", "--ovlfs-strip=1", "--quiet", tmpMergedMeta}, blobs...)
|
||||||
|
log.G(ctx).Infof("merging layers with mkfs.erofs %v", args)
|
||||||
|
cmd := exec.CommandContext(ctx, "mkfs.erofs", args...)
|
||||||
|
out, err := cmd.CombinedOutput()
|
||||||
|
if err != nil {
|
||||||
|
log.G(ctx).Warnf("failed to generate merged fsmeta for %v: %q: %v", snapIDs[0], string(out), err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
// Atomically replace the fsmeta with the generated file
|
||||||
|
if err = os.Rename(tmpMergedMeta, mergedMeta); err != nil {
|
||||||
|
log.G(ctx).Errorf("failed to rename fsmeta: %v", err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
log.G(ctx).WithFields(log.Fields{
|
||||||
|
"d": time.Since(t1),
|
||||||
|
}).Infof("merged fsmeta for %v generated", snapIDs[0])
|
||||||
|
}
|
||||||
|
|
||||||
func (s *snapshotter) Commit(ctx context.Context, name, key string, opts ...snapshots.Opt) error {
|
func (s *snapshotter) Commit(ctx context.Context, name, key string, opts ...snapshots.Opt) error {
|
||||||
var layerBlob string
|
var layerBlob string
|
||||||
var id string
|
var id string
|
||||||
|
|||||||
@@ -46,6 +46,9 @@ type Config struct {
|
|||||||
|
|
||||||
// DefaultSize is the default size of a writable layer in string
|
// DefaultSize is the default size of a writable layer in string
|
||||||
DefaultSize string `toml:"default_size"`
|
DefaultSize string `toml:"default_size"`
|
||||||
|
|
||||||
|
// MaxUnmergedLayers (>0) enables fsmerge when the number of image layers exceeds this value.
|
||||||
|
MaxUnmergedLayers uint `toml:"max_unmerged_layers"`
|
||||||
}
|
}
|
||||||
|
|
||||||
func init() {
|
func init() {
|
||||||
@@ -87,6 +90,10 @@ func init() {
|
|||||||
opts = append(opts, erofs.WithDefaultSize(size))
|
opts = append(opts, erofs.WithDefaultSize(size))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if config.MaxUnmergedLayers > 0 {
|
||||||
|
opts = append(opts, erofs.WithFsMergeThreshold(config.MaxUnmergedLayers))
|
||||||
|
}
|
||||||
|
|
||||||
ic.Meta.Exports[plugins.SnapshotterRootDir] = root
|
ic.Meta.Exports[plugins.SnapshotterRootDir] = root
|
||||||
ic.Meta.Capabilities = append(ic.Meta.Capabilities, "rebase")
|
ic.Meta.Capabilities = append(ic.Meta.Capabilities, "rebase")
|
||||||
return erofs.NewSnapshotter(root, opts...)
|
return erofs.NewSnapshotter(root, opts...)
|
||||||
|
|||||||
Reference in New Issue
Block a user