Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 0 additions & 4 deletions .ko.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -23,7 +23,3 @@ baseImageOverrides:
# atelet needs glibc to run the node's kubelet credential provider, which
# some platforms ship dynamically linked (e.g. AKS, GKE Ubuntu).
github.com/agent-substrate/substrate/cmd/atelet: gcr.io/distroless/base-debian13:latest@sha256:389cad21f73e4c37b94ffe5b13736d5a92bd5bd3c6c6b38c2be1c881e14ba2bd
# ateom-microvm needs glibc (for the fetched cloud-hypervisor binary) and mount/umount
# (to bind the image into the virtiofsd shared dir) — both in debian:13-slim but
# not in the distroless static default.
github.com/agent-substrate/substrate/cmd/ateom-microvm: debian:13-slim@sha256:d7e12182ce18b85b93007c1dedf31f2d29e01ccf3182cc4017c709b6259bc132
110 changes: 106 additions & 4 deletions cmd/ateom-microvm/internal/ch/merge.go
Original file line number Diff line number Diff line change
Expand Up @@ -17,15 +17,14 @@
package ch

import (
"bytes"
"context"
"errors"
"fmt"
"io"
"os"
"os/exec"
"syscall"

"github.com/agent-substrate/substrate/cmd/ateom-microvm/internal/reaper"
"golang.org/x/sys/unix"
)

Expand All @@ -51,8 +50,8 @@ func MergeSparseOverlay(ctx context.Context, baseFile, deltaFile, outFile string
// outFile := sparse copy of baseFile (preserves holes so it stays sparse).
tmp := outFile + ".merge.tmp"
_ = os.Remove(tmp)
if o, err := reaper.RunCombined(exec.CommandContext(ctx, "cp", "--sparse=always", baseFile, tmp)); err != nil {
return fmt.Errorf("cp base->tmp: %w: %s", err, o)
if err := copySparseFile(ctx, baseFile, tmp); err != nil {
return fmt.Errorf("sparse copy base->tmp: %w", err)
}

d, err := os.Open(deltaFile)
Expand Down Expand Up @@ -238,3 +237,106 @@ func copySparseRegions(src, dst *os.File) (copied int64, err error) {
}
return copied, nil
}

// copySparseFile creates dstPath as a sparse copy of srcPath, like
// cp --sparse=always: holes in srcPath stay holes, and so does any all-zero
// block inside its data regions.
func copySparseFile(ctx context.Context, srcPath, dstPath string) error {
s, err := os.Open(srcPath)
if err != nil {
return err
}
defer s.Close()
si, err := s.Stat()
if err != nil {
return err
}
d, err := os.OpenFile(dstPath, os.O_CREATE|os.O_TRUNC|os.O_WRONLY, 0o600)
if err != nil {
return err
}
defer d.Close()
// A fresh file of the full size reads as zeros, so a skipped block needs no write.
if err := d.Truncate(si.Size()); err != nil {
return err
}
if err := copyNonZeroRegions(ctx, s, d, si.Size()); err != nil {
return err
}
return d.Close()
}

// sparseBlock is the granularity at which copyNonZeroRegions leaves zeros as holes.
const sparseBlock = 4096

// copyNonZeroRegions writes every block of src's data regions that is not all
// zeros to the same offset in dst. dst must read as zeros wherever nothing is
// written, as a freshly truncated file does, so unlike copySparseRegions this
// cannot overlay onto existing data: a skipped zero block would leave the old
// bytes in place.
func copyNonZeroRegions(ctx context.Context, src, dst *os.File, size int64) error {
sfd := int(src.Fd())
buf := make([]byte, 1<<20)
for off := int64(0); off < size; {
ds, err := unix.Seek(sfd, off, unix.SEEK_DATA)
if errors.Is(err, unix.ENXIO) {
return nil // no more data
}
if err != nil {
return fmt.Errorf("SEEK_DATA: %w", err)
}
de, err := unix.Seek(sfd, ds, unix.SEEK_HOLE)
if err != nil {
return fmt.Errorf("SEEK_HOLE: %w", err)
}
for pos := ds; pos < de; {
if err := ctx.Err(); err != nil {
return err
}
n := min(int64(len(buf)), de-pos)
if _, err := src.ReadAt(buf[:n], pos); err != nil {
return fmt.Errorf("reading data region: %w", err)
}
if err := writeNonZero(dst, buf[:n], pos); err != nil {
return err
}
pos += n
}
off = de
}
return nil
}

// writeNonZero writes p to dst at off, skipping all-zero blocks and coalescing
// the rest into as few writes as possible.
func writeNonZero(dst *os.File, p []byte, off int64) error {
start := -1 // start of the current run of non-zero blocks
for i := 0; i < len(p); i += sparseBlock {
if !allZero(p[i:min(i+sparseBlock, len(p))]) {
if start < 0 {
start = i
}
continue
}
if start >= 0 {
if _, err := dst.WriteAt(p[start:i], off+int64(start)); err != nil {
return err
}
start = -1
}
}
if start >= 0 {
if _, err := dst.WriteAt(p[start:], off+int64(start)); err != nil {
return err
}
}
return nil
}

// zeroBlock is what allZero compares against: bytes.Equal is vectorized, so it
// beats checking byte by byte.
var zeroBlock = make([]byte, sparseBlock)

func allZero(b []byte) bool {
return bytes.Equal(b, zeroBlock[:len(b)])
}
31 changes: 31 additions & 0 deletions cmd/ateom-microvm/internal/ch/merge_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -248,3 +248,34 @@ func TestMergeDeltaIntoBaseSizeMismatch(t *testing.T) {
t.Errorf("base should be intact after a refused merge: %v", err)
}
}

// Zero runs inside a data region come out as holes, as with cp --sparse=always.
func TestCopySparseFileLeavesZeroBlocksAsHoles(t *testing.T) {
dir := t.TempDir()
// 1 MiB of written (allocated) zeros with one non-zero page in the middle.
data := make([]byte, 1<<20)
copy(data[512<<10:], bytes.Repeat([]byte{0xab}, 4096))
src := filepath.Join(dir, "src")
if err := os.WriteFile(src, data, 0o600); err != nil {
t.Fatal(err)
}
dst := filepath.Join(dir, "dst")
if err := copySparseFile(context.Background(), src, dst); err != nil {
t.Fatal(err)
}
got, err := os.ReadFile(dst)
if err != nil {
t.Fatal(err)
}
if !bytes.Equal(got, data) {
t.Fatal("copy differs from source")
}
var st syscall.Stat_t
if err := syscall.Stat(dst, &st); err != nil {
t.Fatal(err)
}
// st.Blocks counts 512-byte units; allow slack for the filesystem's block size.
if allocated := st.Blocks * 512; allocated > 64<<10 {
t.Errorf("dst allocates %d bytes, want the zero runs left as holes", allocated)
}
}
79 changes: 40 additions & 39 deletions cmd/ateom-microvm/internal/kata/overlay_linux.go
Original file line number Diff line number Diff line change
Expand Up @@ -35,14 +35,13 @@ import (
"os"
"os/exec"
"path/filepath"
"strings"
"syscall"
"time"

"github.com/agent-substrate/substrate/cmd/ateom-microvm/internal/reaper"
"github.com/agent-substrate/substrate/cmd/ateom-microvm/internal/third_party/kata/agentpb"
"github.com/agent-substrate/substrate/internal/ocispec"
specs "github.com/opencontainers/runtime-spec/specs-go"
"golang.org/x/sys/unix"
)

const (
Expand Down Expand Up @@ -178,6 +177,31 @@ func waitForSocket(ctx context.Context, path string, timeout time.Duration) erro
}
}

// Unmount drops a mount at dst, falling back to lazy unmount if busy.
func Unmount(dst string) {
if err := unix.Unmount(dst, 0); err != nil {
if errors.Is(err, syscall.EINVAL) || errors.Is(err, syscall.ENOENT) {
return
}
_ = unix.Unmount(dst, unix.MNT_DETACH)
}
}

// RemountReadOnly remounts a bind mount at dst as read-only, preserving existing
// per-mount flags (nosuid, nodev, noexec, etc.) like libmount's remount,bind,ro.
func RemountReadOnly(dst string) error {
var st unix.Statfs_t
if err := unix.Statfs(dst, &st); err != nil {
return fmt.Errorf("statfs %q: %w", dst, err)
}
flags := unix.MS_BIND | unix.MS_REMOUNT | unix.MS_RDONLY |
int(st.Flags&(unix.ST_NOSUID|unix.ST_NODEV|unix.ST_NOEXEC|unix.ST_NOATIME|unix.ST_NODIRATIME))

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I think we might need to do some explicit mapping between stat and mount flags in the future but it can be a follow-up

if err := unix.Mount("", dst, "", uintptr(flags), ""); err != nil {
return err
}
return nil
}

// StageImageVolume bind-mounts one composed image volume read-only at
// <cid>/volumes/<name> under SharedDir(id), so virtiofsd exposes it to the
// guest.
Expand All @@ -191,11 +215,8 @@ func StageImageVolume(ctx context.Context, src, id, cid, volumeName string) erro
// Read-only is this volume type's contract (the image is someone else's,
// mounted for its contents); the other subtree consumers stay writable.
dst := SharedVolumeDir(id, cid, volumeName)
ro := exec.CommandContext(ctx, "mount", "-o", "remount,bind,ro", dst)
var roErr strings.Builder
ro.Stderr = &roErr
if err := reaper.Run(ro); err != nil {
return fmt.Errorf("remounting image volume %q read-only: %w (%s)", dst, err, strings.TrimSpace(roErr.String()))
if err := RemountReadOnly(dst); err != nil {
return fmt.Errorf("remounting image volume %q read-only: %w", dst, err)
}
return nil
}
Expand All @@ -217,9 +238,7 @@ func StageMergedRootfs(ctx context.Context, bundleRootfs, upperBase, restoreID,
dst := filepath.Join(SharedDir(restoreID), cid, "rootfs")
upper, work := UpperWorkDirs(upperBase, cid)
// Drop any stale mount first (lazy if busy), then ensure clean mountpoints.
if err := reaper.Run(exec.Command("umount", dst)); err != nil {
_ = reaper.Run(exec.Command("umount", "-l", dst))
}
Unmount(dst)
// The workdir is scratch: wipe it so a volatile mount is never refused by a
// dirty marker left behind by the previous activation.
if err := os.RemoveAll(work); err != nil {
Expand Down Expand Up @@ -247,11 +266,8 @@ func StageMergedRootfs(ctx context.Context, bundleRootfs, upperBase, restoreID,
// a previous volatile mount, hence the wipe above.
opts := "lowerdir=" + bundleRootfs + ",upperdir=" + upper + ",workdir=" + work +
",metacopy=off,index=off,volatile"
cmd := exec.CommandContext(ctx, "mount", "-t", "overlay", "overlay", "-o", opts, dst)
var stderr strings.Builder
cmd.Stderr = &stderr
if err := reaper.Run(cmd); err != nil {
return fmt.Errorf("mounting merged rootfs overlay at %q: %w (%s)", dst, err, strings.TrimSpace(stderr.String()))
if err := unix.Mount("overlay", dst, "overlay", 0, opts); err != nil {
return fmt.Errorf("mounting merged rootfs overlay at %q: %w", dst, err)
}
// Ensure the standard OCI mountpoints exist even for minimal images: the container
// mounts /proc,/sys,/dev over them, and find-paths re-opens the tree by path on
Expand Down Expand Up @@ -288,9 +304,7 @@ func ensureOCIMountpoints(rootfs string) error {
// CleanupSandboxState's sweep catches stragglers on the next boot.
func UnmountMergedRootfs(restoreID, cid string) {
dst := filepath.Join(SharedDir(restoreID), cid, "rootfs")
if err := reaper.Run(exec.Command("umount", dst)); err != nil {
_ = reaper.Run(exec.Command("umount", "-l", dst))
}
Unmount(dst)
}

// BindIntoShare bind-mounts a host directory at SharedDir(id)/<name>, so the
Expand Down Expand Up @@ -319,20 +333,15 @@ func BindIntoShare(ctx context.Context, src, id, rel string) error {
}
dst := filepath.Join(SharedDir(id), rel)
// Drop any stale bind first (lazy if busy), then ensure a clean mountpoint.
if err := reaper.Run(exec.Command("umount", dst)); err != nil {
_ = reaper.Run(exec.Command("umount", "-l", dst))
}
Unmount(dst)
if err := os.MkdirAll(dst, 0o755); err != nil {
return fmt.Errorf("creating share subdir %q: %w", dst, err)
}
// --rbind, not --bind: a source that is itself a mount (a composed image
// volume) comes along either way, but only rbind carries mounts nested
// beneath it; for the plain-directory sources it is the same operation.
cmd := exec.CommandContext(ctx, "mount", "--rbind", src, dst)
var stderr strings.Builder
cmd.Stderr = &stderr
if err := reaper.Run(cmd); err != nil {
return fmt.Errorf("bind-mounting %q into the shared tree at %q: %w (%s)", src, dst, err, strings.TrimSpace(stderr.String()))
if err := unix.Mount(src, dst, "", unix.MS_BIND|unix.MS_REC, ""); err != nil {
return fmt.Errorf("bind-mounting %q into the shared tree at %q: %w", src, dst, err)
}
return nil
}
Expand All @@ -350,17 +359,12 @@ func ReconstructSharedDirFromImage(ctx context.Context, bundleRootfs, restoreID,
dst := filepath.Join(SharedDir(restoreID), cid, "rootfs")
// Drop any stale bind first (lazy if busy), then ensure a clean mountpoint. Not
// RemoveAll: that would chase a live bind into bundleRootfs.
if err := reaper.Run(exec.Command("umount", dst)); err != nil {
_ = reaper.Run(exec.Command("umount", "-l", dst))
}
Unmount(dst)
if err := os.MkdirAll(dst, 0o755); err != nil {
return fmt.Errorf("creating shared dir %q: %w", dst, err)
}
cmd := exec.CommandContext(ctx, "mount", "--bind", bundleRootfs, dst)
var stderr strings.Builder
cmd.Stderr = &stderr
if err := reaper.Run(cmd); err != nil {
return fmt.Errorf("bind-mounting image rootfs %q -> %q: %w (%s)", bundleRootfs, dst, err, strings.TrimSpace(stderr.String()))
if err := unix.Mount(bundleRootfs, dst, "", unix.MS_BIND, ""); err != nil {
return fmt.Errorf("bind-mounting image rootfs %q -> %q: %w", bundleRootfs, dst, err)
}
// Ensure the standard OCI mountpoints exist even for minimal images: the container
// mounts /proc,/sys,/dev over them, and find-paths re-opens the lower by path on
Expand All @@ -370,11 +374,8 @@ func ReconstructSharedDirFromImage(ctx context.Context, bundleRootfs, restoreID,
}
// Remount read-only: the lower is immutable, so all writes go to the overlay upper
// and it stays byte-identical across reconstructions (required by find-paths migration).
ro := exec.CommandContext(ctx, "mount", "-o", "remount,bind,ro", dst)
var roErr strings.Builder
ro.Stderr = &roErr
if err := reaper.Run(ro); err != nil {
return fmt.Errorf("remounting overlay lower read-only %q: %w (%s)", dst, err, strings.TrimSpace(roErr.String()))
if err := RemountReadOnly(dst); err != nil {
return fmt.Errorf("remounting overlay lower read-only %q: %w", dst, err)
}
return nil
}
Expand Down
Loading
Loading