From 6c92fbf60ef5b6e77433a86334e17b121146c72d Mon Sep 17 00:00:00 2001 From: SaladDay <1203511142@qq.com> Date: Wed, 30 Sep 2026 23:00:14 +0000 Subject: [PATCH] Add the file access protocol, Linux file service and client MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit internal/sandboxfs defines the Runtime–file service protocol on sandboxwire framing: node and handle operations shaped like FUSE, typed failures with effects, capabilities with admission, golden fixtures, a decoder fuzz target and a generic client and server. Request IDs follow sandboxwire.RequestSequence. Describe reports the service identity, and Attach is limited to the Link export grants. apps/sandboxio/internal/fileservice implements it on Linux. It serves the one export world from a root the caller isolates, keys nodes by mount ID, device and inode, keeps symlinks and proc magic links opaque, and provides handles that survive unlink, atomic append, renameat2 modes, getdents cookies, flock that interoperates with native processes and incarnation checks. Add docs/file-access-protocol.md and its AGENTS.md and development.md rows. --- AGENTS.md | 1 + apps/sandboxio/internal/fileservice/errors.go | 123 + apps/sandboxio/internal/fileservice/files.go | 454 ++++ .../internal/fileservice/namespace.go | 442 ++++ .../sandboxio/internal/fileservice/readdir.go | 146 ++ .../sandboxio/internal/fileservice/service.go | 570 +++++ .../internal/fileservice/service_test.go | 590 +++++ docs/development.md | 2 + docs/file-access-protocol.md | 328 +++ internal/sandboxfs/client.go | 391 +++ internal/sandboxfs/client_test.go | 33 + internal/sandboxfs/protocol.go | 2212 +++++++++++++++++ internal/sandboxfs/protocol_test.go | 297 +++ internal/sandboxfs/server.go | 139 ++ internal/sandboxfs/server_test.go | 83 + .../sandboxfs/testdata/create_request.hex | 11 + .../sandboxfs/testdata/describe_response.hex | 28 + .../sandboxfs/testdata/failure_response.hex | 10 + .../sandboxfs/testdata/readdir_request.hex | 9 + .../sandboxfs/testdata/readdir_response.hex | 18 + .../sandboxfs/testdata/rename_request.hex | 10 + internal/sandboxfs/testdata/walk_request.hex | 9 + internal/sandboxfs/testdata/walk_response.hex | 20 + .../sandboxfs/testdata/write_response.hex | 12 + 24 files changed, 5938 insertions(+) create mode 100644 apps/sandboxio/internal/fileservice/errors.go create mode 100644 apps/sandboxio/internal/fileservice/files.go create mode 100644 apps/sandboxio/internal/fileservice/namespace.go create mode 100644 apps/sandboxio/internal/fileservice/readdir.go create mode 100644 apps/sandboxio/internal/fileservice/service.go create mode 100644 apps/sandboxio/internal/fileservice/service_test.go create mode 100644 docs/file-access-protocol.md create mode 100644 internal/sandboxfs/client.go create mode 100644 internal/sandboxfs/client_test.go create mode 100644 internal/sandboxfs/protocol.go create mode 100644 internal/sandboxfs/protocol_test.go create mode 100644 internal/sandboxfs/server.go create mode 100644 internal/sandboxfs/server_test.go create mode 100644 internal/sandboxfs/testdata/create_request.hex create mode 100644 internal/sandboxfs/testdata/describe_response.hex create mode 100644 internal/sandboxfs/testdata/failure_response.hex create mode 100644 internal/sandboxfs/testdata/readdir_request.hex create mode 100644 internal/sandboxfs/testdata/readdir_response.hex create mode 100644 internal/sandboxfs/testdata/rename_request.hex create mode 100644 internal/sandboxfs/testdata/walk_request.hex create mode 100644 internal/sandboxfs/testdata/walk_response.hex create mode 100644 internal/sandboxfs/testdata/write_response.hex diff --git a/AGENTS.md b/AGENTS.md index 7099527d..c752b094 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -23,6 +23,7 @@ OpenAgentCore is protocol-first and modular. Core orchestrates operations that p | Provider–Runtime startup | `internal/runtimebootstrap/bootstrap.go` | [Runtime bootstrap](docs/runtime-bootstrap.md) | | Runtime and Sandbox I/O service–relay (Link) | `internal/sandboxlink/protocol.go` | [Sandbox link protocol](docs/sandbox-link-protocol.md) | | Provider–Sandbox I/O startup | `internal/sandboxbootstrap/bootstrap.go` | [Sandbox bootstrap](docs/sandbox-bootstrap.md) | +| Runtime–file service | `internal/sandboxfs/protocol.go` | [File access protocol](docs/file-access-protocol.md) | | Core–Runtime wire | `internal/agentdaemon/proto/` | [Core–Runtime protocol](docs/runtime-protocol.md) | | Runtime–Harness | `apps/daemon/internal/agent/harness.go` | [Harness onboarding](contracts/agents-api/harness-onboarding.md) | | Harness–Model provider | `internal/modelprovider/config.go` | [Model execution](contracts/agents-api/model-execution.md) | diff --git a/apps/sandboxio/internal/fileservice/errors.go b/apps/sandboxio/internal/fileservice/errors.go new file mode 100644 index 00000000..9507a650 --- /dev/null +++ b/apps/sandboxio/internal/fileservice/errors.go @@ -0,0 +1,123 @@ +//go:build linux + +package fileservice + +import ( + "errors" + "os" + "strconv" + + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxfs" + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxwire" + "golang.org/x/sys/unix" +) + +var errnos = map[unix.Errno]sandboxfs.Errno{ + unix.EACCES: sandboxfs.ErrnoPermissionDenied, + unix.EPERM: sandboxfs.ErrnoOperationNotPermitted, + unix.ENOENT: sandboxfs.ErrnoNotFound, + unix.EEXIST: sandboxfs.ErrnoExists, + unix.ENOTDIR: sandboxfs.ErrnoNotDirectory, + unix.EISDIR: sandboxfs.ErrnoIsDirectory, + unix.ENOTEMPTY: sandboxfs.ErrnoDirectoryNotEmpty, + unix.EINVAL: sandboxfs.ErrnoInvalidArgument, + unix.EBADF: sandboxfs.ErrnoBadDescriptor, + unix.EMFILE: sandboxfs.ErrnoTooManyOpenFiles, + unix.ENFILE: sandboxfs.ErrnoTooManyOpenFiles, + unix.ENOSPC: sandboxfs.ErrnoNoSpace, + unix.EDQUOT: sandboxfs.ErrnoQuotaExceeded, + unix.EROFS: sandboxfs.ErrnoReadOnlyFilesystem, + unix.EXDEV: sandboxfs.ErrnoCrossDevice, + unix.ENAMETOOLONG: sandboxfs.ErrnoNameTooLong, + unix.ELOOP: sandboxfs.ErrnoSymlinkLoop, + unix.EFBIG: sandboxfs.ErrnoFileTooLarge, + unix.EOVERFLOW: sandboxfs.ErrnoOverflow, + unix.EBUSY: sandboxfs.ErrnoBusy, + unix.ETXTBSY: sandboxfs.ErrnoBusy, + unix.EAGAIN: sandboxfs.ErrnoAgain, + unix.EINTR: sandboxfs.ErrnoInterrupted, + unix.EIO: sandboxfs.ErrnoIO, + unix.ENODEV: sandboxfs.ErrnoNoDevice, + unix.ENXIO: sandboxfs.ErrnoNoSuchDeviceOrAddress, + unix.EPIPE: sandboxfs.ErrnoBrokenPipe, + unix.EOPNOTSUPP: sandboxfs.ErrnoNotSupported, + unix.ENOSYS: sandboxfs.ErrnoNotSupported, + unix.ENOLCK: sandboxfs.ErrnoNoLocks, + unix.EDEADLK: sandboxfs.ErrnoDeadlock, +} + +// failure converts err to the *Failure a Service method returns. A Linux +// errno without an entry becomes IO. EffectPossible overrides the EffectNone +// of a typed failure, because a step before it may have changed state. +func failure(err error, effect sandboxwire.Effect) error { + var f *sandboxfs.Failure + if errors.As(err, &f) { + if effect == sandboxwire.EffectPossible && f.Effect != effect { + promoted := *f + promoted.Effect = effect + return &promoted + } + return f + } + var e unix.Errno + if errors.As(err, &e) { + errno, ok := errnos[e] + if !ok { + errno = sandboxfs.ErrnoIO + } + return sandboxfs.NewErrnoFailure(errno, effect, e.Error()) + } + return sandboxfs.NewErrnoFailure(sandboxfs.ErrnoIO, effect, err.Error()) +} + +func unsupported(what string) error { + return sandboxfs.NewFailure(sandboxfs.CodeUnsupported, sandboxwire.EffectNone, what+" is not supported") +} + +func errnoFailure(errno sandboxfs.Errno, message string) error { + return sandboxfs.NewErrnoFailure(errno, sandboxwire.EffectNone, message) +} + +// use runs fn with f's descriptor. f cannot be closed while fn runs, so the +// descriptor number is never reused under fn. A closed f yields stale(). +func use(f *os.File, stale func() error, fn func(fd int) error) error { + rc, err := f.SyscallConn() + if err != nil { + return stale() + } + var inner error + if err := rc.Control(func(fd uintptr) { inner = fn(int(fd)) }); err != nil { + return stale() + } + return inner +} + +// viaProc runs fn with the /proc/self/fd directory and fd's name in it; fd +// is always a descriptor the service opened and holds. Resolving that name +// reaches the object fd refers to, even after it was renamed or unlinked, +// and stops at a symlink or proc magic link the object is. +func (s *Service) viaProc(fd int, fn func(dir int, name string) error) error { + return use(s.proc, errStaleAttachment, func(dir int) error { return fn(dir, strconv.Itoa(fd)) }) +} + +// eintr retries a system call interrupted by a signal. +func eintr[T any](call func() (T, error)) (T, error) { + for { + v, err := call() + if err != unix.EINTR { + return v, err + } + } +} + +func openat(dir int, name string, flags int, mode uint32) (int, error) { + return eintr(func() (int, error) { return unix.Openat(dir, name, flags|unix.O_CLOEXEC, mode) }) +} + +func openFile(dir int, name string, flags int) (*os.File, error) { + fd, err := openat(dir, name, flags, 0) + if err != nil { + return nil, err + } + return os.NewFile(uintptr(fd), name), nil +} diff --git a/apps/sandboxio/internal/fileservice/files.go b/apps/sandboxio/internal/fileservice/files.go new file mode 100644 index 00000000..594e7121 --- /dev/null +++ b/apps/sandboxio/internal/fileservice/files.go @@ -0,0 +1,454 @@ +//go:build linux + +package fileservice + +import ( + "context" + "math" + "os" + "time" + + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxfs" + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxwire" + "golang.org/x/sys/unix" +) + +// openFlags converts an access mode and open flags. The service never opens +// a controlling terminal and never follows a symlink it opens. +func openFlags(access sandboxfs.AccessMode, flags sandboxfs.OpenFlags) int { + f := map[sandboxfs.AccessMode]int{sandboxfs.AccessRead: unix.O_RDONLY, sandboxfs.AccessWrite: unix.O_WRONLY, sandboxfs.AccessReadWrite: unix.O_RDWR}[access] + f |= unix.O_NOCTTY + for bit, o := range map[sandboxfs.OpenFlags]int{ + sandboxfs.OpenAppend: unix.O_APPEND, sandboxfs.OpenTruncate: unix.O_TRUNC, sandboxfs.OpenNoFollow: unix.O_NOFOLLOW, + sandboxfs.OpenSync: unix.O_SYNC, sandboxfs.OpenDataSync: unix.O_DSYNC, + } { + if flags&bit != 0 { + f |= o + } + } + return f +} + +// openable reports why a node of type typ cannot be opened as a file. +func openable(typ uint32) error { + switch typ { + case sandboxfs.ModeRegular: + return nil + case sandboxfs.ModeDirectory: + return unix.EISDIR + case sandboxfs.ModeSymlink: + return unix.ELOOP + } + return unsupported("opening a special file") +} + +// reopen opens the object of fd, a descriptor the service opened itself, +// through its /proc/self/fd name, after checking that the object has type +// typ. That name resolves to the object fd holds, so reopening never reaches +// a symlink's target. +func (s *Service) reopen(fd int, typ uint32, flags int) (fd2 int, err error) { + var sb unix.Stat_t + if err := unix.Fstat(fd, &sb); err != nil { + return -1, err + } + switch got := sb.Mode & sandboxfs.ModeType; { + case got == typ: + case typ == sandboxfs.ModeDirectory: + return -1, unix.ENOTDIR + default: + return -1, openable(got) + } + err = s.viaProc(fd, func(dir int, name string) (err error) { + fd2, err = openat(dir, name, flags&^unix.O_NOFOLLOW, 0) + return err + }) + return fd2, err +} + +func (s *Service) Open(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.OpenRequest) (*sandboxfs.OpenResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + n, err := st.node(r.Node) + if err != nil { + return nil, err + } + if err := openable(n.typ); err != nil { + return nil, failure(err, none) + } + if err := st.reserve(); err != nil { + return nil, err + } + var fd int + err = use(n.f, errStaleNode, func(pfd int) (err error) { + fd, err = s.reopen(pfd, sandboxfs.ModeRegular, openFlags(r.Access, r.Flags)) + return err + }) + if err != nil { + st.unreserve() + return nil, failure(err, none) + } + id, err := st.addHandle(&handle{f: os.NewFile(uintptr(fd), ""), append: r.Flags&sandboxfs.OpenAppend != 0}) + if err != nil { + return nil, failure(err, possible) + } + return &sandboxfs.OpenResponse{Handle: id}, nil +} + +// Create creates the entry with O_EXCL first, so it never opens a special +// file or follows a symlink. Without Exclusive an existing regular file is +// opened instead. +func (s *Service) Create(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.CreateRequest) (*sandboxfs.CreateResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + parent, err := st.dir(r.Parent) + if err != nil { + return nil, err + } + if err := st.reserve(); err != nil { + return nil, err + } + flags := openFlags(r.Access, r.Flags) | unix.O_NOFOLLOW + effect := none + var fd, pathFD int + var sb unix.Stat_t + err = use(parent.f, errStaleNode, func(dir int) error { + var err error + for range 3 { + fd, err = openat(dir, string(r.Name), flags|unix.O_CREAT|unix.O_EXCL, r.Mode) + if err != unix.EEXIST || r.Exclusive { + break + } + if fd, err = s.openExisting(dir, r.Name, flags); err != unix.ENOENT { + break + } + } + if err != nil { + return err + } + effect = possible + if pathFD, err = s.reopen(fd, sandboxfs.ModeRegular, unix.O_PATH); err == nil { + if err = unix.Fstat(pathFD, &sb); err != nil { + unix.Close(pathFD) + } + } + if err != nil { + unix.Close(fd) + } + return err + }) + if err != nil { + st.unreserve() + return nil, failure(err, effect) + } + n, err := st.addNode(pathFD, &sb) + if err != nil { + st.unreserve() + unix.Close(fd) + return nil, failure(err, possible) + } + id, err := st.addHandle(&handle{f: os.NewFile(uintptr(fd), ""), append: r.Flags&sandboxfs.OpenAppend != 0}) + if err != nil { + return nil, failure(err, possible) + } + return &sandboxfs.CreateResponse{Entry: sandboxfs.Entry{Node: n.ref, Attr: st.attr(&sb)}, Handle: id}, nil +} + +// openExisting opens an existing regular file without following a symlink. +func (s *Service) openExisting(dir int, name []byte, flags int) (int, error) { + pfd, err := openat(dir, string(name), unix.O_PATH|unix.O_NOFOLLOW, 0) + if err != nil { + return -1, err + } + defer unix.Close(pfd) + return s.reopen(pfd, sandboxfs.ModeRegular, flags) +} + +// fileHandle returns a handle that must not be a directory handle. +func (st *state) fileHandle(id sandboxfs.HandleID, dirErr error) (*handle, error) { + h, err := st.handle(id) + if err != nil { + return nil, err + } + if h.dir != nil { + return nil, failure(dirErr, none) + } + return h, nil +} + +func (s *Service) Read(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.ReadRequest) (*sandboxfs.ReadResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + h, err := st.fileHandle(r.Handle, unix.EISDIR) + if err != nil { + return nil, err + } + buf := make([]byte, r.Size) + total := 0 + err = use(h.f, errStaleHandle, func(fd int) error { + for total < len(buf) { + n, err := eintr(func() (int, error) { return unix.Pread(fd, buf[total:], int64(r.Offset)+int64(total)) }) + if err != nil { + return err + } + if n == 0 { + break + } + total += n + } + return nil + }) + if err != nil { + return nil, failure(err, none) + } + return &sandboxfs.ReadResponse{Data: buf[:total]}, nil +} + +// Write writes at the offset, or appends atomically with one write(2) on an +// append handle. It reports the written prefix and the error that stopped it. +func (s *Service) Write(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.WriteRequest) (*sandboxfs.WriteResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + h, err := st.fileHandle(r.Handle, unix.EISDIR) + if err != nil { + return nil, err + } + if !h.append && r.Offset > math.MaxInt64-uint64(len(r.Data)) { + return nil, sandboxfs.NewFailure(sandboxfs.CodeInvalidArgument, none, "write ends beyond 2^63-1") + } + total := 0 + err = use(h.f, errStaleHandle, func(fd int) error { + if h.append { + n, err := eintr(func() (int, error) { return unix.Write(fd, r.Data) }) + total = max(n, 0) + return err + } + for total < len(r.Data) { + n, err := eintr(func() (int, error) { return unix.Pwrite(fd, r.Data[total:], int64(r.Offset)+int64(total)) }) + if err != nil { + return err + } + if n == 0 { + break + } + total += n + } + return nil + }) + switch { + case err == nil: + return &sandboxfs.WriteResponse{Written: uint32(total)}, nil + case total == 0: + return nil, failure(err, none) + } + return &sandboxfs.WriteResponse{Written: uint32(total), Failure: failure(err, none).(*sandboxfs.Failure)}, nil +} + +// Flush closes a duplicate of the handle's descriptor, which is what a close +// of one descriptor does to the file. +func (s *Service) Flush(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.FlushRequest) (*sandboxfs.FlushResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + h, err := st.fileHandle(r.Handle, unix.EBADF) + if err != nil { + return nil, err + } + effect := none + err = use(h.f, errStaleHandle, func(fd int) error { + dup, err := unix.FcntlInt(uintptr(fd), unix.F_DUPFD_CLOEXEC, 0) + if err != nil { + return err + } + effect = possible + return unix.Close(dup) + }) + if err != nil { + return nil, failure(err, effect) + } + return &sandboxfs.FlushResponse{}, nil +} + +func (s *Service) Fsync(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.FsyncRequest) (*sandboxfs.FsyncResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + h, err := st.handle(r.Handle) + if err != nil { + return nil, err + } + sync := unix.Fsync + if r.DataOnly { + sync = unix.Fdatasync + } + effect := none + err = use(h.f, errStaleHandle, func(fd int) error { + effect = possible + _, err := eintr(func() (struct{}, error) { return struct{}{}, sync(fd) }) + return err + }) + if err != nil { + return nil, failure(err, effect) + } + return &sandboxfs.FsyncResponse{}, nil +} + +func (s *Service) Release(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.ReleaseRequest) (*sandboxfs.ReleaseResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + if err := st.release(r.Handle, false); err != nil { + return nil, failure(err, none) + } + return &sandboxfs.ReleaseResponse{}, nil +} + +func (s *Service) OpenDir(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.OpenDirRequest) (*sandboxfs.OpenDirResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + n, err := st.dir(r.Node) + if err != nil { + return nil, err + } + if err := st.reserve(); err != nil { + return nil, err + } + var fd int + var sb unix.Stat_t + err = use(n.f, errStaleNode, func(pfd int) (err error) { + if err = unix.Fstat(pfd, &sb); err == nil { + fd, err = s.reopen(pfd, sandboxfs.ModeDirectory, unix.O_RDONLY|unix.O_DIRECTORY) + } + return err + }) + if err != nil { + st.unreserve() + return nil, failure(err, none) + } + id, err := st.addHandle(&handle{f: os.NewFile(uintptr(fd), ""), dir: &cursor{dev: uint64(sb.Dev)}}) + if err != nil { + return nil, err + } + return &sandboxfs.OpenDirResponse{Handle: id}, nil +} + +func (s *Service) ReadDir(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.ReadDirRequest) (*sandboxfs.ReadDirResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + h, err := st.handle(r.Handle) + if err != nil { + return nil, err + } + if h.dir == nil { + return nil, failure(unix.ENOTDIR, none) + } + var resp *sandboxfs.ReadDirResponse + err = use(h.f, errStaleHandle, func(fd int) (err error) { + resp, err = h.dir.read(st, fd, r) + return err + }) + if err != nil { + return nil, failure(err, none) + } + return resp, nil +} + +func (s *Service) ReleaseDir(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.ReleaseDirRequest) (*sandboxfs.ReleaseDirResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + if err := st.release(r.Handle, true); err != nil { + return nil, failure(err, none) + } + return &sandboxfs.ReleaseDirResponse{}, nil +} + +// GetLock queries POSIX locks, which this service does not declare. +func (s *Service) GetLock(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.GetLockRequest) (*sandboxfs.GetLockResponse, error) { + if _, err := s.enter(a, r); err != nil { + return nil, err + } + return nil, unsupported("POSIX locks") +} + +// SetLock takes flock locks on the handle's open file description, the same +// lock native flock(2) takes, so both exclude each other. A waiting request +// polls, holding the handle's lock state only during each attempt, so other +// requests on the handle proceed and cancelling the context ends the wait. +func (s *Service) SetLock(ctx context.Context, a sandboxfs.Attachment, r *sandboxfs.SetLockRequest) (*sandboxfs.SetLockResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + h, err := st.handle(r.Handle) + if err != nil { + return nil, err + } + effect := none + for delay := time.Millisecond; ; delay = min(2*delay, 50*time.Millisecond) { + if err := ctx.Err(); err != nil { + return nil, contextFailure(err, effect) + } + dropped, err := h.tryLock(r.Lock.Mode) + if dropped { + effect = possible + } + switch { + case err == nil: + return &sandboxfs.SetLockResponse{}, nil + case err != unix.EWOULDBLOCK || !r.Wait: + return nil, failure(err, effect) + } + wait := time.NewTimer(delay) + select { + case <-ctx.Done(): + wait.Stop() + case <-wait.C: + } + } +} + +var flockOps = map[sandboxfs.LockMode]int{sandboxfs.LockRead: unix.LOCK_SH, sandboxfs.LockWrite: unix.LOCK_EX, sandboxfs.LockUnlock: unix.LOCK_UN} + +// tryLock makes one non-blocking flock attempt. Converting a held lock drops +// it before taking the new one, so a failed conversion leaves the handle +// unlocked and reports dropped. +func (h *handle) tryLock(mode sandboxfs.LockMode) (dropped bool, err error) { + h.lockMu.Lock() + defer h.lockMu.Unlock() + converting := h.flock != 0 && h.flock != mode && mode != sandboxfs.LockUnlock + err = use(h.f, errStaleHandle, func(fd int) error { return unix.Flock(fd, flockOps[mode]|unix.LOCK_NB) }) + switch { + case err == nil && mode == sandboxfs.LockUnlock: + h.flock = 0 + case err == nil: + h.flock = mode + case converting: + h.flock = 0 + return true, err + } + return false, err +} + +func contextFailure(err error, effect sandboxwire.Effect) error { + code := sandboxfs.CodeCancelled + if err == context.DeadlineExceeded { + code = sandboxfs.CodeDeadlineExceeded + } + return sandboxfs.NewFailure(code, effect, err.Error()) +} diff --git a/apps/sandboxio/internal/fileservice/namespace.go b/apps/sandboxio/internal/fileservice/namespace.go new file mode 100644 index 00000000..b73238a6 --- /dev/null +++ b/apps/sandboxio/internal/fileservice/namespace.go @@ -0,0 +1,442 @@ +//go:build linux + +package fileservice + +import ( + "context" + "errors" + "os" + "time" + + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxfs" + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxwire" + "golang.org/x/sys/unix" +) + +const ( + none = sandboxwire.EffectNone + possible = sandboxwire.EffectPossible +) + +// lookup opens name in directory dir without following a symlink and +// acquires one reference on its node. Names never contain "/" and are never +// "." or "..", so a lookup resolves exactly one entry of dir, and a symlink, +// including a proc magic link, is returned as itself. +func (st *state) lookup(dir int, name []byte) (*node, sandboxfs.Entry, error) { + fd, err := openat(dir, string(name), unix.O_PATH|unix.O_NOFOLLOW, 0) + if err != nil { + return nil, sandboxfs.Entry{}, err + } + var sb unix.Stat_t + if err := unix.Fstat(fd, &sb); err != nil { + unix.Close(fd) + return nil, sandboxfs.Entry{}, err + } + n, err := st.addNode(fd, &sb) + if err != nil { + return nil, sandboxfs.Entry{}, err + } + return n, sandboxfs.Entry{Node: n.ref, Attr: st.attr(&sb)}, nil +} + +func (st *state) lookupIn(parent *node, name []byte) (n *node, e sandboxfs.Entry, err error) { + err = use(parent.f, errStaleNode, func(dir int) error { + n, e, err = st.lookup(dir, name) + return err + }) + return n, e, err +} + +// withNode runs fn with the descriptor of the node ref names. +func (st *state) withNode(ref sandboxfs.NodeRef, fn func(n *node, fd int) error) error { + n, err := st.node(ref) + if err != nil { + return err + } + return use(n.f, errStaleNode, func(fd int) error { return fn(n, fd) }) +} + +// dir returns the node ref names, which must be a directory. A symlink is +// never a directory, so no request resolves through one. +func (st *state) dir(ref sandboxfs.NodeRef) (*node, error) { + n, err := st.node(ref) + if err != nil { + return nil, err + } + if n.typ != sandboxfs.ModeDirectory { + return nil, failure(unix.ENOTDIR, none) + } + return n, nil +} + +// withDir runs fn with the descriptor of the directory ref names. +func (st *state) withDir(ref sandboxfs.NodeRef, fn func(dir int) error) error { + n, err := st.dir(ref) + if err != nil { + return err + } + return use(n.f, errStaleNode, fn) +} + +func (s *Service) Lookup(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.LookupRequest) (*sandboxfs.LookupResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + parent, err := st.dir(r.Parent) + if err != nil { + return nil, err + } + _, e, err := st.lookupIn(parent, r.Name) + if err != nil { + return nil, failure(err, none) + } + return &sandboxfs.LookupResponse{Entry: e}, nil +} + +// Walk looks up each name in turn and stops after a symlink or at the first +// failure, which it reports after the entries already walked. +func (s *Service) Walk(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.WalkRequest) (*sandboxfs.WalkResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + cur, err := st.dir(r.Parent) + if err != nil { + return nil, err + } + resp := &sandboxfs.WalkResponse{} + for _, name := range r.Names { + n, e, err := st.lookupIn(cur, name) + if err != nil { + if len(resp.Entries) == 0 { + return nil, failure(err, none) + } + resp.Failure = failure(err, none).(*sandboxfs.Failure) + break + } + resp.Entries = append(resp.Entries, e) + if n.typ == sandboxfs.ModeSymlink { + break + } + cur = n + } + return resp, nil +} + +// target resolves a node or handle Target to its descriptor holder. typ is +// the node's type, or zero for a handle. +func (st *state) target(t sandboxfs.Target) (f *os.File, typ uint32, stale func() error, err error) { + if t.Kind == sandboxfs.TargetHandle { + h, err := st.handle(t.Handle) + if err != nil { + return nil, 0, nil, err + } + return h.f, 0, errStaleHandle, nil + } + n, err := st.node(t.Node) + if err != nil { + return nil, 0, nil, err + } + return n.f, n.typ, errStaleNode, nil +} + +func (s *Service) GetAttr(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.GetAttrRequest) (*sandboxfs.GetAttrResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + f, _, stale, err := st.target(r.Target) + if err != nil { + return nil, err + } + var sb unix.Stat_t + if err := use(f, stale, func(fd int) error { return unix.Fstat(fd, &sb) }); err != nil { + return nil, failure(err, none) + } + return &sandboxfs.GetAttrResponse{Attr: st.attr(&sb)}, nil +} + +// SetAttr applies the owner, then the mode, the size and the times. A +// failure after the first applied change carries EffectPossible. +func (s *Service) SetAttr(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.SetAttrRequest) (*sandboxfs.SetAttrResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + f, typ, stale, err := st.target(r.Target) + if err != nil { + return nil, err + } + isHandle := r.Target.Kind == sandboxfs.TargetHandle + applied := false + var sb unix.Stat_t + err = use(f, stale, func(fd int) error { + apply := func(err error) error { + applied = applied || err == nil + return err + } + if r.Set&(sandboxfs.AttrUID|sandboxfs.AttrGID) != 0 { + uid, gid := -1, -1 + if r.Set&sandboxfs.AttrUID != 0 { + uid = int(r.UID) + } + if r.Set&sandboxfs.AttrGID != 0 { + gid = int(r.GID) + } + if err := apply(unix.Fchownat(fd, "", uid, gid, unix.AT_EMPTY_PATH)); err != nil { + return err + } + } + if r.Set&sandboxfs.AttrMode != 0 { + var err error + switch { + case isHandle: + err = unix.Fchmod(fd, r.Mode) + case typ == sandboxfs.ModeSymlink: + err = unix.EOPNOTSUPP + default: + err = s.viaProc(fd, func(dir int, name string) error { return unix.Fchmodat(dir, name, r.Mode, 0) }) + } + if err := apply(err); err != nil { + return err + } + } + if r.Set&sandboxfs.AttrSize != 0 { + if err := apply(s.truncate(fd, isHandle, typ, int64(r.Size))); err != nil { + return err + } + } + if r.Set&(sandboxfs.AttrAtime|sandboxfs.AttrMtime|sandboxfs.AttrAtimeNow|sandboxfs.AttrMtimeNow) != 0 { + atime, aerr := timespec(r.Set, sandboxfs.AttrAtime, sandboxfs.AttrAtimeNow, r.Atime) + mtime, merr := timespec(r.Set, sandboxfs.AttrMtime, sandboxfs.AttrMtimeNow, r.Mtime) + if err := errors.Join(aerr, merr); err != nil { + return err + } + err := s.viaProc(fd, func(dir int, name string) error { + return unix.UtimesNanoAt(dir, name, []unix.Timespec{atime, mtime}, 0) + }) + if err := apply(err); err != nil { + return err + } + } + return unix.Fstat(fd, &sb) + }) + if err != nil { + effect := none + if applied { + effect = possible + } + return nil, failure(err, effect) + } + return &sandboxfs.SetAttrResponse{Attr: st.attr(&sb)}, nil +} + +func (s *Service) truncate(fd int, isHandle bool, typ uint32, size int64) error { + switch { + case isHandle: + _, err := eintr(func() (struct{}, error) { return struct{}{}, unix.Ftruncate(fd, size) }) + return err + case typ == sandboxfs.ModeDirectory: + return unix.EISDIR + case typ != sandboxfs.ModeRegular: + return unix.EINVAL + } + w, err := s.reopen(fd, sandboxfs.ModeRegular, unix.O_WRONLY|unix.O_NOCTTY) + if err != nil { + return err + } + defer unix.Close(w) + _, err = eintr(func() (struct{}, error) { return struct{}{}, unix.Ftruncate(w, size) }) + return err +} + +func timespec(set, value, now sandboxfs.AttrMask, t sandboxfs.Timestamp) (unix.Timespec, error) { + switch { + case set&value != 0: + return unix.TimeToTimespec(time.Unix(t.Sec, int64(t.Nsec))) + case set&now != 0: + return unix.Timespec{Nsec: unix.UTIME_NOW}, nil + } + return unix.Timespec{Nsec: unix.UTIME_OMIT}, nil +} + +func (s *Service) Access(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.AccessRequest) (*sandboxfs.AccessResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + var mode uint32 + for bit, m := range map[sandboxfs.AccessMask]uint32{sandboxfs.MayRead: unix.R_OK, sandboxfs.MayWrite: unix.W_OK, sandboxfs.MayExecute: unix.X_OK} { + if r.Mask&bit != 0 { + mode |= m + } + } + err = st.withNode(r.Node, func(_ *node, fd int) error { + return s.viaProc(fd, func(dir int, name string) error { return unix.Faccessat(dir, name, mode, 0) }) + }) + if err != nil { + return nil, failure(err, none) + } + return &sandboxfs.AccessResponse{}, nil +} + +// create runs make in the parent directory and then looks the new entry up. +// A failure after make succeeded carries EffectPossible. +func (st *state) create(parent sandboxfs.NodeRef, name []byte, make func(dir int) error) (sandboxfs.Entry, error) { + var e sandboxfs.Entry + made := false + err := st.withDir(parent, func(dir int) (err error) { + if err := make(dir); err != nil { + return err + } + made = true + _, e, err = st.lookup(dir, name) + return err + }) + if err != nil { + effect := none + if made { + effect = possible + } + return e, failure(err, effect) + } + return e, nil +} + +func (s *Service) Mkdir(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.MkdirRequest) (*sandboxfs.MkdirResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + e, err := st.create(r.Parent, r.Name, func(dir int) error { return unix.Mkdirat(dir, string(r.Name), r.Mode) }) + if err != nil { + return nil, err + } + return &sandboxfs.MkdirResponse{Entry: e}, nil +} + +func (s *Service) Symlink(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.SymlinkRequest) (*sandboxfs.SymlinkResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + e, err := st.create(r.Parent, r.Name, func(dir int) error { return unix.Symlinkat(string(r.Target), dir, string(r.Name)) }) + if err != nil { + return nil, err + } + return &sandboxfs.SymlinkResponse{Entry: e}, nil +} + +func (s *Service) Link(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.LinkRequest) (*sandboxfs.LinkResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + var e sandboxfs.Entry + err = st.withNode(r.Node, func(_ *node, fd int) error { + e, err = st.create(r.NewParent, r.NewName, func(dir int) error { + return s.viaProc(fd, func(proc int, name string) error { + return unix.Linkat(proc, name, dir, string(r.NewName), unix.AT_SYMLINK_FOLLOW) + }) + }) + return err + }) + if err != nil { + return nil, failure(err, none) + } + return &sandboxfs.LinkResponse{Entry: e}, nil +} + +func (s *Service) Unlink(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.UnlinkRequest) (*sandboxfs.UnlinkResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + if err := st.withDir(r.Parent, func(dir int) error { return unix.Unlinkat(dir, string(r.Name), 0) }); err != nil { + return nil, failure(err, none) + } + return &sandboxfs.UnlinkResponse{}, nil +} + +func (s *Service) Rmdir(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.RmdirRequest) (*sandboxfs.RmdirResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + if err := st.withDir(r.Parent, func(dir int) error { return unix.Unlinkat(dir, string(r.Name), unix.AT_REMOVEDIR) }); err != nil { + return nil, failure(err, none) + } + return &sandboxfs.RmdirResponse{}, nil +} + +var renameFlags = map[sandboxfs.RenameMode]uint{ + sandboxfs.RenameReplace: 0, + sandboxfs.RenameNoReplace: unix.RENAME_NOREPLACE, + sandboxfs.RenameExchange: unix.RENAME_EXCHANGE, +} + +func (s *Service) Rename(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.RenameRequest) (*sandboxfs.RenameResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + err = st.withDir(r.Parent, func(dir int) error { + return st.withDir(r.NewParent, func(newDir int) error { + return unix.Renameat2(dir, string(r.Name), newDir, string(r.NewName), renameFlags[r.Mode]) + }) + }) + if err != nil { + return nil, failure(err, none) + } + return &sandboxfs.RenameResponse{}, nil +} + +func (s *Service) Readlink(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.ReadlinkRequest) (*sandboxfs.ReadlinkResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + buf := make([]byte, s.caps.MaxPathBytes+1) + var n int + err = st.withNode(r.Node, func(nd *node, fd int) (err error) { + if nd.typ != sandboxfs.ModeSymlink { + return unix.EINVAL // as readlink(2) answers for a non-symlink + } + n, err = unix.Readlinkat(fd, "", buf) + return err + }) + switch { + case err != nil: + return nil, failure(err, none) + case n > int(s.caps.MaxPathBytes): + return nil, errnoFailure(sandboxfs.ErrnoNameTooLong, "symlink target exceeds MaxPathBytes") + } + return &sandboxfs.ReadlinkResponse{Target: buf[:n]}, nil +} + +func (s *Service) StatFS(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.StatFSRequest) (*sandboxfs.StatFSResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + var sf unix.Statfs_t + if err := st.withNode(r.Node, func(_ *node, fd int) error { return unix.Fstatfs(fd, &sf) }); err != nil { + return nil, failure(err, none) + } + return &sandboxfs.StatFSResponse{ + Blocks: sf.Blocks, BlocksFree: sf.Bfree, BlocksAvailable: sf.Bavail, Files: sf.Files, FilesFree: sf.Ffree, + BlockSize: uint32(sf.Bsize), FragmentSize: uint32(sf.Frsize), NameMax: uint32(sf.Namelen), + }, nil +} + +func (s *Service) Forget(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.ForgetRequest) (*sandboxfs.ForgetResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + if err := st.forget(r.Entries); err != nil { + return nil, err + } + return &sandboxfs.ForgetResponse{}, nil +} diff --git a/apps/sandboxio/internal/fileservice/readdir.go b/apps/sandboxio/internal/fileservice/readdir.go new file mode 100644 index 00000000..72f0ec65 --- /dev/null +++ b/apps/sandboxio/internal/fileservice/readdir.go @@ -0,0 +1,146 @@ +//go:build linux + +package fileservice + +import ( + "bytes" + "encoding/binary" + "sync" + + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxfs" + "golang.org/x/sys/unix" +) + +// cursor is a directory handle's getdents position. A cookie is the kernel's +// d_off of the entry it follows, so resuming at another cookie is one lseek. +type cursor struct { + dev uint64 // the directory's device, for composing entry inode numbers + + mu sync.Mutex + buf []byte + rest []byte // entries read from the kernel and not yet returned + pos uint64 // cookie of the last returned or skipped entry +} + +type rawEntry struct { + ino uint64 + off uint64 + typ uint8 + name []byte +} + +// next parses one linux_dirent64 from c.rest. +func (c *cursor) next() (rawEntry, []byte, error) { + const header = 19 // d_ino, d_off, d_reclen, d_type + if len(c.rest) < header { + return rawEntry{}, nil, unix.EIO + } + reclen := int(binary.NativeEndian.Uint16(c.rest[16:18])) + if reclen < header || reclen > len(c.rest) { + return rawEntry{}, nil, unix.EIO + } + name := c.rest[header:reclen] + if i := bytes.IndexByte(name, 0); i >= 0 { + name = name[:i] + } + e := rawEntry{ino: binary.NativeEndian.Uint64(c.rest[0:8]), off: binary.NativeEndian.Uint64(c.rest[8:16]), typ: c.rest[18], name: name} + return e, c.rest[reclen:], nil +} + +var direntTypes = map[uint8]uint32{ + unix.DT_REG: sandboxfs.ModeRegular, unix.DT_DIR: sandboxfs.ModeDirectory, unix.DT_LNK: sandboxfs.ModeSymlink, + unix.DT_FIFO: sandboxfs.ModeFIFO, unix.DT_SOCK: sandboxfs.ModeSocket, unix.DT_CHR: sandboxfs.ModeCharDevice, + unix.DT_BLK: sandboxfs.ModeBlockDevice, +} + +// read returns the entries after r.Cookie that fit r.Limit. It skips "." and +// "..", and entries removed before they could be described. A directory that +// changes while it is read can repeat a name or a cookie, so a page ends +// before a repeat. +func (c *cursor) read(st *state, fd int, r *sandboxfs.ReadDirRequest) (*sandboxfs.ReadDirResponse, error) { + c.mu.Lock() + defer c.mu.Unlock() + if r.Cookie != c.pos { + if _, err := unix.Seek(fd, int64(r.Cookie), unix.SEEK_SET); err != nil { + return nil, err + } + c.rest, c.pos = nil, r.Cookie + } + resp := &sandboxfs.ReadDirResponse{} + size := 0 + names, cookies := map[string]bool{}, map[uint64]bool{} + for { + if len(c.rest) == 0 { + if c.buf == nil { + c.buf = make([]byte, 32<<10) + } + n, err := eintr(func() (int, error) { return unix.Getdents(fd, c.buf) }) + if err != nil { + if len(resp.Entries) > 0 { + return resp, nil + } + return nil, err + } + if n == 0 { + resp.End = true + return resp, nil + } + c.rest = c.buf[:n] + } + raw, rest, err := c.next() + if err != nil { + c.rest = nil + return nil, err + } + skip := func() { c.rest, c.pos = rest, raw.off } + if string(raw.name) == "." || string(raw.name) == ".." || raw.typ == unix.DT_WHT { + skip() + continue + } + if names[string(raw.name)] || cookies[raw.off] { + return resp, nil + } + e := sandboxfs.DirEntry{Name: bytes.Clone(raw.name), Ino: st.ino(c.dev, raw.ino), Type: direntTypes[raw.typ], Cookie: raw.off} + if r.WithAttrs { + e.Entry = &sandboxfs.Entry{} + } + if size+e.WireSize() > int(r.Limit) { + if len(resp.Entries) == 0 { + return nil, unix.EINVAL + } + return resp, nil + } + switch { + case r.WithAttrs: + _, entry, err := st.lookup(fd, raw.name) + if err == unix.ENOENT { + skip() + continue + } + if err != nil { + if len(resp.Entries) > 0 { + return resp, nil + } + return nil, err + } + *e.Entry = entry + e.Ino, e.Type = entry.Attr.Ino, entry.Attr.Mode&sandboxfs.ModeType + case e.Type == 0: + var sb unix.Stat_t + if err := unix.Fstatat(fd, string(raw.name), &sb, unix.AT_SYMLINK_NOFOLLOW); err == unix.ENOENT { + skip() + continue + } else if err != nil { + if len(resp.Entries) > 0 { + return resp, nil + } + return nil, err + } + e.Type = sb.Mode & sandboxfs.ModeType + } + resp.Entries = append(resp.Entries, e) + names[string(e.Name)], cookies[e.Cookie] = true, true + size += e.WireSize() + skip() + } +} diff --git a/apps/sandboxio/internal/fileservice/service.go b/apps/sandboxio/internal/fileservice/service.go new file mode 100644 index 00000000..33c722ea --- /dev/null +++ b/apps/sandboxio/internal/fileservice/service.go @@ -0,0 +1,570 @@ +//go:build linux + +// Package fileservice is the sandbox file service: it implements +// sandboxfs.Service over the one export world with the Linux *at system +// calls. It acts as its own process identity and never impersonates. +package fileservice + +import ( + "context" + "errors" + "fmt" + "math/rand/v2" + "os" + "path/filepath" + "strconv" + "strings" + "sync" + + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxfs" + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxlink" + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxwire" + "golang.org/x/sys/unix" +) + +// Export is the one export the service serves. +const Export sandboxlink.ExportID = "world" + +// Service serves File requests for any number of attachments. Its node and +// handle tables live in memory, so each Service has its own ServerInstanceID. +type Service struct { + instance sandboxwire.ID + identity sandboxfs.Identity + caps sandboxfs.Capabilities + root *os.File // O_PATH directory of the export + rootDev uint64 + proc *os.File // /proc/self/fd, for reopening a held descriptor + fdinfo *os.File // /proc/self/fdinfo, for mount IDs statx does not report + + mu sync.Mutex + atts map[sandboxwire.ID]*state + closed bool +} + +// New serves the absolute directory root as the export world. oac-sandbox-io +// passes "/", the sandbox's root after the Provider's namespace setup; the +// caller owns the isolation of everything under root, because the service +// neither detects nor enforces a boundary inside it. New sets the process +// umask to zero, because the service applies every requested mode +// explicitly. +func New(root string) (*Service, error) { + if !filepath.IsAbs(root) { + return nil, fmt.Errorf("fileservice: root %q is not absolute", root) + } + unix.Umask(0) + s := &Service{ + instance: sandboxwire.NewID(), + identity: sandboxfs.Identity{UID: uint32(unix.Geteuid()), GID: uint32(unix.Getegid())}, + atts: map[sandboxwire.ID]*state{}, + } + for _, d := range []struct { + f **os.File + path string + }{{&s.proc, "/proc/self/fd"}, {&s.fdinfo, "/proc/self/fdinfo"}, {&s.root, root}} { + f, err := openFile(unix.AT_FDCWD, d.path, unix.O_PATH|unix.O_DIRECTORY) + if err != nil { + s.Close() + return nil, fmt.Errorf("fileservice: open %s: %w", d.path, err) + } + *d.f = f + } + var sb unix.Stat_t + if err := unix.Fstat(int(s.root.Fd()), &sb); err != nil { + s.Close() + return nil, fmt.Errorf("fileservice: root %s: %w", root, err) + } + s.rootDev = uint64(sb.Dev) + noReplace, exchange := probeRename() + s.caps = sandboxfs.Capabilities{ + PathProfile: sandboxfs.PathProfileLinuxBytes, + CacheProfile: sandboxfs.CacheProfileUncached, + Durability: sandboxfs.DurabilityFsyncRequired, + MaxNameBytes: 255, + MaxPathBytes: 4095, + MaxReadBytes: sandboxwire.MaxChunk, + MaxWriteBytes: sandboxwire.MaxChunk, + MaxWalkComponents: 256, + MaxReadDirBytes: 64 << 10, + MaxOpenHandles: 4096, + AtomicAppend: true, + AtomicRename: true, + RenameNoReplace: noReplace, + RenameExchange: exchange, + HardLinks: true, + Symlinks: true, + SetMode: true, + SetOwner: true, + SetTimes: true, + DirectoryFsync: true, + ReadDirPlus: true, + Flock: true, + } + return s, nil +} + +// probeRename reports which renameat2 modes the kernel supports, by trying +// them in a temporary directory. +func probeRename() (noReplace, exchange bool) { + dir, err := os.MkdirTemp("", "oac-fileservice-") + if err != nil { + return false, false + } + defer os.RemoveAll(dir) + a, b := filepath.Join(dir, "a"), filepath.Join(dir, "b") + if os.WriteFile(a, nil, 0o600) != nil || os.WriteFile(b, nil, 0o600) != nil { + return false, false + } + noReplace = unix.Renameat2(unix.AT_FDCWD, a, unix.AT_FDCWD, b, unix.RENAME_NOREPLACE) == unix.EEXIST + exchange = unix.Renameat2(unix.AT_FDCWD, a, unix.AT_FDCWD, b, unix.RENAME_EXCHANGE) == nil + return noReplace, exchange +} + +// InstanceID is the service incarnation. Link binds each stream to it, and +// the Attachment a stream passes to sandboxfs.Serve carries it. +func (s *Service) InstanceID() sandboxwire.ID { return s.instance } + +// Close releases every attachment and the export root. Requests still +// running fail as stale. +func (s *Service) Close() error { + s.mu.Lock() + s.closed = true + atts := s.atts + s.atts = map[sandboxwire.ID]*state{} + s.mu.Unlock() + for _, st := range atts { + st.close() + } + for _, f := range []*os.File{s.root, s.proc, s.fdinfo} { + if f != nil { + f.Close() + } + } + return nil +} + +var ( + errStaleAttachment = func() error { + return sandboxfs.NewFailure(sandboxfs.CodeStaleAttachment, sandboxwire.EffectNone, "attachment is not attached") + } + errStaleNode = func() error { + return sandboxfs.NewFailure(sandboxfs.CodeStaleNode, sandboxwire.EffectNone, "node is not held by this attachment") + } + errStaleHandle = func() error { + return sandboxfs.NewFailure(sandboxfs.CodeStaleHandle, sandboxwire.EffectNone, "handle is not open in this attachment") + } +) + +// check admits a request for attachment a without requiring it to be attached. +func (s *Service) check(a sandboxfs.Attachment) error { + if a.ServerInstanceID != s.instance { + return sandboxfs.NewFailure(sandboxfs.CodeInstanceChanged, sandboxwire.EffectNone, "stream is bound to another service instance") + } + if a.Lease == nil { + return errStaleAttachment() + } + select { + case <-a.Lease.Done(): + return errStaleAttachment() + default: + return nil + } +} + +// enter admits r for an attached attachment and returns its state. +func (s *Service) enter(a sandboxfs.Attachment, r sandboxfs.Request) (*state, error) { + if err := s.check(a); err != nil { + return nil, err + } + s.mu.Lock() + st := s.atts[a.ID] + s.mu.Unlock() + if st == nil { + return nil, errStaleAttachment() + } + if f := s.caps.Admit(r, st.readOnly); f != nil { + return nil, f + } + return st, nil +} + +// Describe lists world when the attachment is granted it. +func (s *Service) Describe(_ context.Context, a sandboxfs.Attachment, _ *sandboxfs.DescribeRequest) (*sandboxfs.DescribeResponse, error) { + if err := s.check(a); err != nil { + return nil, err + } + var exports []sandboxlink.ExportID + if _, ok := a.Grant(Export); ok { + exports = []sandboxlink.ExportID{Export} + } + return &sandboxfs.DescribeResponse{ServerInstanceID: s.instance, Identity: s.identity, Capabilities: s.caps, Exports: exports}, nil +} + +func (s *Service) Attach(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.AttachRequest) (*sandboxfs.AttachResponse, error) { + if err := s.check(a); err != nil { + return nil, err + } + switch g, ok := a.Grant(r.Export); { + case !ok: + return nil, sandboxfs.NewFailure(sandboxfs.CodeUnauthorized, sandboxwire.EffectNone, "export "+string(r.Export)+" is not granted") + case g.ReadOnly && !r.ReadOnly: + return nil, sandboxfs.NewFailure(sandboxfs.CodeUnauthorized, sandboxwire.EffectNone, "export "+string(r.Export)+" is granted read-only") + } + if f := s.caps.Admit(r, r.ReadOnly); f != nil { + return nil, f + } + if r.Export != Export { + return nil, sandboxfs.NewFailure(sandboxfs.CodeInvalidArgument, sandboxwire.EffectNone, "unknown export "+string(r.Export)) + } + var fd int + err := use(s.root, errStaleAttachment, func(root int) (err error) { + fd, err = unix.FcntlInt(uintptr(root), unix.F_DUPFD_CLOEXEC, 0) + return err + }) + if err != nil { + return nil, failure(err, sandboxwire.EffectNone) + } + var sb unix.Stat_t + if err := unix.Fstat(fd, &sb); err != nil { + unix.Close(fd) + return nil, failure(err, sandboxwire.EffectNone) + } + st := newState(s, r.ReadOnly) + s.mu.Lock() + switch { + case s.closed: + s.mu.Unlock() + unix.Close(fd) + return nil, errStaleAttachment() + case s.atts[a.ID] != nil: + s.mu.Unlock() + unix.Close(fd) + return nil, sandboxfs.NewFailure(sandboxfs.CodeInvalidArgument, sandboxwire.EffectNone, "attachment is already attached") + } + s.atts[a.ID] = st + s.mu.Unlock() + go s.watch(a, st) + root, err := st.addNode(fd, &sb) + if err != nil { + s.detach(a.ID, st) + return nil, failure(err, sandboxwire.EffectNone) + } + return &sandboxfs.AttachResponse{Root: sandboxfs.Entry{Node: root.ref, Attr: st.attr(&sb)}}, nil +} + +// watch detaches st when its lease ends. +func (s *Service) watch(a sandboxfs.Attachment, st *state) { + select { + case <-a.Lease.Done(): + s.detach(a.ID, st) + case <-st.done: + } +} + +func (s *Service) detach(id sandboxwire.ID, st *state) { + s.mu.Lock() + if s.atts[id] == st { + delete(s.atts, id) + } + s.mu.Unlock() + st.close() +} + +func (s *Service) Detach(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.DetachRequest) (*sandboxfs.DetachResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + s.detach(a.ID, st) + return &sandboxfs.DetachResponse{}, nil +} + +// state is one attachment: its node and handle tables. +type state struct { + svc *Service + readOnly bool + done chan struct{} + + mu sync.Mutex + closed bool + nodes []*node // index ID-1 + free []int + inodes map[inodeKey]*node + generation uint64 + handles map[sandboxfs.HandleID]*handle + reserved int + nextHandle uint64 +} + +// inodeKey identifies a node. The mount ID keeps the same inode reached +// through two mounts, such as a bind mount and its source, in two nodes. +type inodeKey struct{ mnt, dev, ino uint64 } + +// mountID returns the ID of the mount the object of fd is reached through: +// statx reports it from Linux 5.8, and the mnt_id line of +// /proc/self/fdinfo/ before that. +func (s *Service) mountID(fd int) (uint64, error) { + var stx unix.Statx_t + if unix.Statx(fd, "", unix.AT_EMPTY_PATH|unix.AT_SYMLINK_NOFOLLOW, unix.STATX_MNT_ID, &stx) == nil && stx.Mask&unix.STATX_MNT_ID != 0 { + return stx.Mnt_id, nil + } + return s.fdinfoMountID(fd) +} + +func (s *Service) fdinfoMountID(fd int) (uint64, error) { + var info []byte + err := use(s.fdinfo, errStaleAttachment, func(dir int) error { + f, err := openat(dir, strconv.Itoa(fd), unix.O_RDONLY, 0) + if err != nil { + return err + } + defer unix.Close(f) + buf := make([]byte, 4096) + for { + n, err := eintr(func() (int, error) { return unix.Read(f, buf) }) + if n <= 0 || err != nil { + return err + } + info = append(info, buf[:n]...) + } + }) + if err != nil { + return 0, err + } + for line := range strings.Lines(string(info)) { + if v, ok := strings.CutPrefix(line, "mnt_id:"); ok { + return strconv.ParseUint(strings.TrimSpace(v), 10, 64) + } + } + return 0, errors.New("fdinfo has no mnt_id") +} + +// node is a file-system object held by an O_PATH descriptor that never +// follows a symlink. +type node struct { + ref sandboxfs.NodeRef + f *os.File + key inodeKey + typ uint32 // sandboxfs mode type bits; an inode never changes type + refs uint64 +} + +type handle struct { + f *os.File + append bool + dir *cursor // set for directory handles + + lockMu sync.Mutex + flock sandboxfs.LockMode // held flock mode; zero when none +} + +func newState(s *Service, readOnly bool) *state { + // Random bases make a NodeRef or HandleID from another attachment or + // incarnation miss instead of naming a live object. + return &state{ + svc: s, readOnly: readOnly, done: make(chan struct{}), + inodes: map[inodeKey]*node{}, handles: map[sandboxfs.HandleID]*handle{}, + generation: rand.Uint64() >> 2, nextHandle: rand.Uint64() >> 2, + } +} + +// close releases every node and handle; closing a handle releases its locks. +func (st *state) close() { + st.mu.Lock() + if st.closed { + st.mu.Unlock() + return + } + st.closed = true + nodes, handles := st.nodes, st.handles + st.nodes, st.inodes, st.handles = nil, nil, nil + close(st.done) + st.mu.Unlock() + for _, n := range nodes { + if n != nil { + n.f.Close() + } + } + for _, h := range handles { + h.f.Close() + } +} + +// addNode takes ownership of fd and acquires one reference on its node. +func (st *state) addNode(fd int, sb *unix.Stat_t) (*node, error) { + mnt, err := st.svc.mountID(fd) + if err != nil { + unix.Close(fd) + return nil, err + } + key := inodeKey{mnt, uint64(sb.Dev), sb.Ino} + st.mu.Lock() + defer st.mu.Unlock() + if st.closed { + unix.Close(fd) + return nil, errStaleAttachment() + } + if n := st.inodes[key]; n != nil { + n.refs++ + unix.Close(fd) + return n, nil + } + i := len(st.nodes) + if k := len(st.free); k > 0 { + i, st.free = st.free[k-1], st.free[:k-1] + } else { + st.nodes = append(st.nodes, nil) + } + st.generation++ + n := &node{ + ref: sandboxfs.NodeRef{ID: uint64(i) + 1, Generation: st.generation}, + f: os.NewFile(uintptr(fd), ""), key: key, typ: sb.Mode & sandboxfs.ModeType, refs: 1, + } + st.nodes[i], st.inodes[key] = n, n + return n, nil +} + +func (st *state) node(ref sandboxfs.NodeRef) (*node, error) { + st.mu.Lock() + defer st.mu.Unlock() + if st.closed { + return nil, errStaleAttachment() + } + if i := ref.ID - 1; i < uint64(len(st.nodes)) { + if n := st.nodes[i]; n != nil && n.ref == ref { + return n, nil + } + } + return nil, errStaleNode() +} + +// forget applies every entry or none. +func (st *state) forget(entries []sandboxfs.ForgetEntry) error { + st.mu.Lock() + var drop []*node + defer func() { + st.mu.Unlock() + for _, n := range drop { + n.f.Close() + } + }() + if st.closed { + return errStaleAttachment() + } + for _, e := range entries { + i := e.Node.ID - 1 + if i >= uint64(len(st.nodes)) || st.nodes[i] == nil || st.nodes[i].ref != e.Node { + return errStaleNode() + } + if e.Count > st.nodes[i].refs { + return sandboxfs.NewFailure(sandboxfs.CodeInvalidArgument, sandboxwire.EffectNone, "forget exceeds the node's references") + } + } + for _, e := range entries { + i := e.Node.ID - 1 + n := st.nodes[i] + if n.refs -= e.Count; n.refs == 0 { + st.nodes[i] = nil + st.free = append(st.free, int(i)) + delete(st.inodes, n.key) + drop = append(drop, n) + } + } + return nil +} + +// reserve claims a handle slot before a handle is opened, so exhaustion +// fails before anything happens. +func (st *state) reserve() error { + st.mu.Lock() + defer st.mu.Unlock() + if st.closed { + return errStaleAttachment() + } + if len(st.handles)+st.reserved >= int(st.svc.caps.MaxOpenHandles) { + return sandboxfs.NewFailure(sandboxfs.CodeResourceExhausted, sandboxwire.EffectNone, "too many open handles") + } + st.reserved++ + return nil +} + +func (st *state) unreserve() { + st.mu.Lock() + st.reserved-- + st.mu.Unlock() +} + +// addHandle registers h in a reserved slot. +func (st *state) addHandle(h *handle) (sandboxfs.HandleID, error) { + st.mu.Lock() + defer st.mu.Unlock() + st.reserved-- + if st.closed { + h.f.Close() + return 0, errStaleAttachment() + } + st.nextHandle++ + id := sandboxfs.HandleID(st.nextHandle) + st.handles[id] = h + return id, nil +} + +func (st *state) handle(id sandboxfs.HandleID) (*handle, error) { + st.mu.Lock() + defer st.mu.Unlock() + if st.closed { + return nil, errStaleAttachment() + } + if h := st.handles[id]; h != nil { + return h, nil + } + return nil, errStaleHandle() +} + +// release closes handle id if it is a directory handle exactly when dir is +// set. +func (st *state) release(id sandboxfs.HandleID, dir bool) error { + st.mu.Lock() + h := st.handles[id] + switch { + case st.closed: + st.mu.Unlock() + return errStaleAttachment() + case h == nil: + st.mu.Unlock() + return errStaleHandle() + case (h.dir != nil) != dir: + st.mu.Unlock() + return sandboxfs.NewErrnoFailure(sandboxfs.ErrnoBadDescriptor, sandboxwire.EffectNone, "wrong handle kind") + } + delete(st.handles, id) + st.mu.Unlock() + return h.f.Close() +} + +// attr converts a stat. Ino combines the device with the inode number, as +// go-fuse's loopback does, so files on different mounts stay distinct. +func (st *state) attr(sb *unix.Stat_t) sandboxfs.Attr { + return sandboxfs.Attr{ + Ino: st.ino(uint64(sb.Dev), sb.Ino), + Mode: sb.Mode, + Nlink: uint32(sb.Nlink), + UID: sb.Uid, + GID: sb.Gid, + Rdev: uint64(sb.Rdev), + Size: uint64(sb.Size), + Blocks: uint64(sb.Blocks), + Blksize: uint32(sb.Blksize), + Atime: timestamp(sb.Atim), + Mtime: timestamp(sb.Mtim), + Ctime: timestamp(sb.Ctim), + } +} + +func (st *state) ino(dev, ino uint64) uint64 { + swap := func(d uint64) uint64 { return d<<32 | d>>32 } + return swap(dev) ^ swap(st.svc.rootDev) ^ ino +} + +func timestamp(t unix.Timespec) sandboxfs.Timestamp { + return sandboxfs.Timestamp{Sec: int64(t.Sec), Nsec: uint32(t.Nsec)} +} diff --git a/apps/sandboxio/internal/fileservice/service_test.go b/apps/sandboxio/internal/fileservice/service_test.go new file mode 100644 index 00000000..8c3156e0 --- /dev/null +++ b/apps/sandboxio/internal/fileservice/service_test.go @@ -0,0 +1,590 @@ +//go:build linux + +package fileservice + +import ( + "bytes" + "context" + "errors" + "fmt" + "math" + "net" + "os" + "os/exec" + "path/filepath" + "slices" + "strconv" + "strings" + "sync" + "syscall" + "testing" + "time" + + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxfs" + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxlink" + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxwire" +) + +type fixture struct { + t *testing.T + dir string + svc *Service + att sandboxfs.Attachment + c *sandboxfs.Client + root sandboxfs.NodeRef +} + +// newFixture serves a temporary directory as the export world and attaches +// to it. +func newFixture(t *testing.T) *fixture { + return attachRoot(t, t.TempDir()) +} + +// attachRoot serves dir as the export world and attaches to it. +func attachRoot(t *testing.T, dir string) *fixture { + svc, err := New(dir) + if err != nil { + t.Fatal(err) + } + t.Cleanup(func() { svc.Close() }) + f := &fixture{t: t, dir: dir, svc: svc, att: attachment(svc, sandboxlink.ExportGrant{ID: "world"})} + f.c, _, _ = f.connect(f.svc, f.att) + resp, err := f.c.Attach(context.Background(), &sandboxfs.AttachRequest{Export: "world"}) + if err != nil { + t.Fatal(err) + } + f.root = resp.Root.Node + return f +} + +// connect opens a stream to svc for attachment a. It returns the client, the +// server end of the stream and the channel Serve's result arrives on. +func (f *fixture) connect(svc *Service, a sandboxfs.Attachment) (*sandboxfs.Client, net.Conn, chan error) { + cc, sc := net.Pipe() + served := make(chan error, 1) + go func() { served <- sandboxfs.Serve(context.Background(), sc, svc, a) }() + c := sandboxfs.NewClient(cc) + f.t.Cleanup(func() { c.Close() }) + return c, sc, served +} + +// attachment returns a fresh attachment to svc with grants. +func attachment(svc *Service, grants ...sandboxlink.ExportGrant) sandboxfs.Attachment { + return sandboxfs.Attachment{ID: sandboxwire.NewID(), ServerInstanceID: svc.InstanceID(), Lease: context.Background(), Exports: grants} +} + +func (f *fixture) lookup(parent sandboxfs.NodeRef, name string) sandboxfs.Entry { + f.t.Helper() + r, err := f.c.Lookup(context.Background(), &sandboxfs.LookupRequest{Parent: parent, Name: []byte(name)}) + if err != nil { + f.t.Fatalf("lookup %q: %v", name, err) + } + return r.Entry +} + +func (f *fixture) create(name string, access sandboxfs.AccessMode, flags sandboxfs.OpenFlags) (sandboxfs.Entry, sandboxfs.HandleID) { + f.t.Helper() + r, err := f.c.Create(context.Background(), &sandboxfs.CreateRequest{Parent: f.root, Name: []byte(name), Mode: 0o640, Access: access, Flags: flags, Exclusive: true}) + if err != nil { + f.t.Fatalf("create %q: %v", name, err) + } + return r.Entry, r.Handle +} + +func (f *fixture) write(h sandboxfs.HandleID, off uint64, data string) { + f.t.Helper() + r, err := f.c.Write(context.Background(), &sandboxfs.WriteRequest{Handle: h, Offset: off, Data: []byte(data)}) + if err != nil || r.Written != uint32(len(data)) || r.Failure != nil { + f.t.Fatalf("write: %+v, %v", r, err) + } +} + +func (f *fixture) read(h sandboxfs.HandleID) string { + f.t.Helper() + r, err := f.c.Read(context.Background(), &sandboxfs.ReadRequest{Handle: h, Size: 4096}) + if err != nil { + f.t.Fatalf("read: %v", err) + } + return string(r.Data) +} + +// wantFailure checks err is a failure with code, errno and effect. +func wantFailure(t *testing.T, err error, code sandboxfs.ErrorCode, errno sandboxfs.Errno, effect sandboxwire.Effect) *sandboxfs.Failure { + t.Helper() + var f *sandboxfs.Failure + if !errors.As(err, &f) || f.Code != code || f.Errno != errno || f.Effect != effect { + t.Fatalf("got %v, want %s %s effect %d", err, code, errno, effect) + } + return f +} + +func TestCreateWriteRead(t *testing.T) { + f := newFixture(t) + d, err := f.c.Describe(context.Background(), &sandboxfs.DescribeRequest{}) + if err != nil { + t.Fatal(err) + } + if d.Identity != (sandboxfs.Identity{UID: uint32(os.Geteuid()), GID: uint32(os.Getegid())}) || !slices.Equal(d.Exports, []sandboxlink.ExportID{"world"}) || d.Capabilities.POSIXLocks || !d.Capabilities.Flock { + t.Fatalf("describe %+v", d) + } + + e, h := f.create("notes", sandboxfs.AccessReadWrite, 0) + f.write(h, 0, "hello world") + f.write(h, 6, "there") + if got := f.read(h); got != "hello there" { + t.Fatalf("read %q", got) + } + info, err := os.Lstat(filepath.Join(f.dir, "notes")) + if err != nil || info.Mode().Perm() != 0o640 || info.Size() != 11 { + t.Fatalf("on disk %v, %v", info.Mode(), err) + } + a, err := f.c.GetAttr(context.Background(), &sandboxfs.GetAttrRequest{Target: sandboxfs.Target{Kind: sandboxfs.TargetNode, Node: e.Node}}) + if err != nil || a.Attr.Size != 11 || a.Attr.Mode != sandboxfs.ModeRegular|0o640 { + t.Fatalf("getattr %+v, %v", a, err) + } + if _, err := f.c.Release(context.Background(), &sandboxfs.ReleaseRequest{Handle: h}); err != nil { + t.Fatal(err) + } + _, err = f.c.Read(context.Background(), &sandboxfs.ReadRequest{Handle: h, Size: 1}) + wantFailure(t, err, sandboxfs.CodeStaleHandle, 0, sandboxwire.EffectNone) +} + +func TestExclusiveCreateConflict(t *testing.T) { + f := newFixture(t) + f.create("x", sandboxfs.AccessWrite, 0) + _, err := f.c.Create(context.Background(), &sandboxfs.CreateRequest{Parent: f.root, Name: []byte("x"), Mode: 0o600, Access: sandboxfs.AccessWrite, Exclusive: true}) + wantFailure(t, err, sandboxfs.CodeErrno, sandboxfs.ErrnoExists, sandboxwire.EffectNone) + if _, err := f.c.Create(context.Background(), &sandboxfs.CreateRequest{Parent: f.root, Name: []byte("x"), Mode: 0o600, Access: sandboxfs.AccessWrite}); err != nil { + t.Fatalf("non-exclusive create of an existing file: %v", err) + } +} + +func TestAtomicAppendFromTwoHandles(t *testing.T) { + f := newFixture(t) + _, h1 := f.create("log", sandboxfs.AccessWrite, sandboxfs.OpenAppend) + open, err := f.c.Open(context.Background(), &sandboxfs.OpenRequest{Node: f.lookup(f.root, "log").Node, Access: sandboxfs.AccessWrite, Flags: sandboxfs.OpenAppend}) + if err != nil { + t.Fatal(err) + } + const records = 200 + var wg sync.WaitGroup + for i, h := range []sandboxfs.HandleID{h1, open.Handle} { + wg.Add(1) + go func() { + defer wg.Done() + for j := range records { + // Every write names offset zero; append ignores it. + rec := fmt.Sprintf("%d:%03d:%s\n", i, j, strings.Repeat("x", 64)) + if r, err := f.c.Write(context.Background(), &sandboxfs.WriteRequest{Handle: h, Data: []byte(rec)}); err != nil || int(r.Written) != len(rec) { + t.Errorf("append: %+v, %v", r, err) + return + } + } + }() + } + wg.Wait() + data, err := os.ReadFile(filepath.Join(f.dir, "log")) + if err != nil { + t.Fatal(err) + } + lines := strings.Split(strings.TrimSuffix(string(data), "\n"), "\n") + if len(lines) != 2*records { + t.Fatalf("%d records, want %d", len(lines), 2*records) + } + for _, l := range lines { + if len(l) != 6+64 || !strings.HasSuffix(l, strings.Repeat("x", 64)) { + t.Fatalf("torn record %q", l) + } + } +} + +func TestReadOpenHandleAfterUnlink(t *testing.T) { + f := newFixture(t) + _, h := f.create("gone", sandboxfs.AccessReadWrite, 0) + f.write(h, 0, "still here") + if _, err := f.c.Unlink(context.Background(), &sandboxfs.UnlinkRequest{Parent: f.root, Name: []byte("gone")}); err != nil { + t.Fatal(err) + } + if got := f.read(h); got != "still here" { + t.Fatalf("read %q", got) + } + a, err := f.c.GetAttr(context.Background(), &sandboxfs.GetAttrRequest{Target: sandboxfs.Target{Kind: sandboxfs.TargetHandle, Handle: h}}) + if err != nil || a.Attr.Nlink != 0 { + t.Fatalf("getattr %+v, %v", a, err) + } +} + +func TestRenameModes(t *testing.T) { + f := newFixture(t) + d, _ := f.c.Describe(context.Background(), &sandboxfs.DescribeRequest{}) + if !d.Capabilities.RenameNoReplace || !d.Capabilities.RenameExchange { + t.Skip("the kernel lacks renameat2 modes") + } + for name, content := range map[string]string{"a": "A", "b": "B"} { + if err := os.WriteFile(filepath.Join(f.dir, name), []byte(content), 0o600); err != nil { + t.Fatal(err) + } + } + rename := func(mode sandboxfs.RenameMode) error { + _, err := f.c.Rename(context.Background(), &sandboxfs.RenameRequest{Parent: f.root, Name: []byte("a"), NewParent: f.root, NewName: []byte("b"), Mode: mode}) + return err + } + content := func(name string) string { + b, _ := os.ReadFile(filepath.Join(f.dir, name)) + return string(b) + } + + wantFailure(t, rename(sandboxfs.RenameNoReplace), sandboxfs.CodeErrno, sandboxfs.ErrnoExists, sandboxwire.EffectNone) + if err := rename(sandboxfs.RenameExchange); err != nil || content("a") != "B" || content("b") != "A" { + t.Fatalf("exchange: %v, a=%q b=%q", err, content("a"), content("b")) + } + if err := rename(sandboxfs.RenameReplace); err != nil || content("b") != "B" { + t.Fatalf("replace: %v, b=%q", err, content("b")) + } + if _, err := os.Lstat(filepath.Join(f.dir, "a")); !os.IsNotExist(err) { + t.Fatalf("a after replace: %v", err) + } + wantFailure(t, rename(sandboxfs.RenameNoReplace), sandboxfs.CodeErrno, sandboxfs.ErrnoNotFound, sandboxwire.EffectNone) +} + +func TestReadDirPagesWithCookies(t *testing.T) { + f := newFixture(t) + var want []string + for i := range 40 { + name := fmt.Sprintf("file-%02d", i) + want = append(want, name) + if err := os.WriteFile(filepath.Join(f.dir, name), nil, 0o600); err != nil { + t.Fatal(err) + } + } + dir, err := f.c.OpenDir(context.Background(), &sandboxfs.OpenDirRequest{Node: f.root}) + if err != nil { + t.Fatal(err) + } + readAll := func(cookie uint64) (names []string, cookies []uint64) { + for { + r, err := f.c.ReadDir(context.Background(), &sandboxfs.ReadDirRequest{Handle: dir.Handle, Cookie: cookie, Limit: 200}) + if err != nil { + t.Fatal(err) + } + if !r.End && len(r.Entries) == 0 { + t.Fatal("empty page before the end") + } + for _, e := range r.Entries { + names = append(names, string(e.Name)) + cookies = append(cookies, e.Cookie) + cookie = e.Cookie + } + if r.End { + return names, cookies + } + } + } + names, cookies := readAll(0) + if got := slices.Sorted(slices.Values(names)); !slices.Equal(got, want) { + t.Fatalf("names %v", got) + } + // Resuming at an earlier cookie continues after that entry. + again, _ := readAll(cookies[9]) + if !slices.Equal(again, names[10:]) { + t.Fatalf("resumed at cookie 10: %v, want %v", again, names[10:]) + } + + r, err := f.c.ReadDir(context.Background(), &sandboxfs.ReadDirRequest{Handle: dir.Handle, Limit: 1024, WithAttrs: true}) + if err != nil || len(r.Entries) == 0 || r.Entries[0].Entry == nil || r.Entries[0].Entry.Attr.Mode&sandboxfs.ModeType != sandboxfs.ModeRegular { + t.Fatalf("readdir with attributes: %+v, %v", r, err) + } + for _, e := range r.Entries { + if _, err := f.c.Forget(context.Background(), &sandboxfs.ForgetRequest{Entries: []sandboxfs.ForgetEntry{{Node: e.Entry.Node, Count: 1}}}); err != nil { + t.Fatal(err) + } + } +} + +func TestWalkStopsAtSymlink(t *testing.T) { + f := newFixture(t) + if err := os.MkdirAll(filepath.Join(f.dir, "d", "sub"), 0o755); err != nil { + t.Fatal(err) + } + if err := os.Symlink("sub", filepath.Join(f.dir, "d", "link")); err != nil { + t.Fatal(err) + } + walk := func(names ...string) *sandboxfs.WalkResponse { + r := &sandboxfs.WalkRequest{Parent: f.root} + for _, n := range names { + r.Names = append(r.Names, []byte(n)) + } + resp, err := f.c.Walk(context.Background(), r) + if err != nil { + t.Fatal(err) + } + return resp + } + r := walk("d", "link", "x") + if len(r.Entries) != 2 || r.Failure != nil || r.Entries[1].Attr.Mode&sandboxfs.ModeType != sandboxfs.ModeSymlink { + t.Fatalf("walk through symlink: %+v", r) + } + r = walk("d", "missing", "x") + if len(r.Entries) != 1 || r.Failure == nil || r.Failure.Errno != sandboxfs.ErrnoNotFound { + t.Fatalf("walk to a missing name: %+v", r) + } +} + +func TestRootEscapeIsRefused(t *testing.T) { + f := newFixture(t) + _, err := f.c.Lookup(context.Background(), &sandboxfs.LookupRequest{Parent: f.root, Name: []byte("..")}) + wantFailure(t, err, sandboxfs.CodeInvalidArgument, 0, sandboxwire.EffectNone) + + // A client that skips validation gets the same answer from the server. + cc, sc := net.Pipe() + go sandboxfs.Serve(context.Background(), sc, f.svc, f.att) + defer cc.Close() + var e sandboxwire.Encoder + e.U64(f.root.ID) + e.U64(f.root.Generation) + e.Bytes([]byte("..")) + if err := sandboxwire.WriteFrame(cc, sandboxwire.Frame{Type: uint16(sandboxfs.OpLookup), RequestID: 1, Payload: e.Payload()}); err != nil { + t.Fatal(err) + } + resp, err := sandboxwire.ReadFrame(cc, sandboxwire.MaxPayload) + if err != nil || resp.Type != sandboxwire.ResponseType(uint16(sandboxfs.OpLookup)) || !bytes.HasPrefix(resp.Payload, []byte{0, 2, 0, byte(sandboxfs.CodeInvalidArgument)}) { + t.Fatalf("raw lookup of \"..\": %x, %v", resp.Payload, err) + } + + // A symlink to "/" is a leaf: nothing resolves through it. + if err := os.Symlink("/", filepath.Join(f.dir, "escape")); err != nil { + t.Fatal(err) + } + link := f.lookup(f.root, "escape").Node + _, err = f.c.Lookup(context.Background(), &sandboxfs.LookupRequest{Parent: link, Name: []byte("etc")}) + wantFailure(t, err, sandboxfs.CodeErrno, sandboxfs.ErrnoNotDirectory, sandboxwire.EffectNone) + _, err = f.c.Open(context.Background(), &sandboxfs.OpenRequest{Node: link, Access: sandboxfs.AccessRead}) + wantFailure(t, err, sandboxfs.CodeErrno, sandboxfs.ErrnoSymlinkLoop, sandboxwire.EffectNone) + _, err = f.c.OpenDir(context.Background(), &sandboxfs.OpenDirRequest{Node: link}) + wantFailure(t, err, sandboxfs.CodeErrno, sandboxfs.ErrnoNotDirectory, sandboxwire.EffectNone) + _, err = f.c.Mkdir(context.Background(), &sandboxfs.MkdirRequest{Parent: link, Name: []byte("x"), Mode: 0o755}) + wantFailure(t, err, sandboxfs.CodeErrno, sandboxfs.ErrnoNotDirectory, sandboxwire.EffectNone) +} + +func TestFlockInteroperatesWithNativeFlock(t *testing.T) { + f := newFixture(t) + _, h := f.create("lock", sandboxfs.AccessReadWrite, 0) + native, err := os.Open(filepath.Join(f.dir, "lock")) + if err != nil { + t.Fatal(err) + } + defer native.Close() + whole := sandboxfs.Lock{Mode: sandboxfs.LockWrite, End: math.MaxInt64} + setLock := func(ctx context.Context, mode sandboxfs.LockMode, wait bool) error { + l := whole + l.Mode = mode + _, err := f.c.SetLock(ctx, &sandboxfs.SetLockRequest{Handle: h, Kind: sandboxfs.LockFlock, Owner: 1, Lock: l, Wait: wait}) + return err + } + nativeTry := func() error { return syscall.Flock(int(native.Fd()), syscall.LOCK_EX|syscall.LOCK_NB) } + + if err := setLock(context.Background(), sandboxfs.LockWrite, false); err != nil { + t.Fatal(err) + } + if err := nativeTry(); err != syscall.EWOULDBLOCK { + t.Fatalf("native flock while the service holds it: %v", err) + } + if err := setLock(context.Background(), sandboxfs.LockUnlock, false); err != nil { + t.Fatal(err) + } + if err := nativeTry(); err != nil { + t.Fatalf("native flock after unlock: %v", err) + } + wantFailure(t, setLock(context.Background(), sandboxfs.LockWrite, false), sandboxfs.CodeErrno, sandboxfs.ErrnoAgain, sandboxwire.EffectNone) + + // A waiting lock the client abandons is cancelled on the server. + ctx, cancel := context.WithTimeout(context.Background(), 100*time.Millisecond) + defer cancel() + wantFailure(t, setLock(ctx, sandboxfs.LockWrite, true), sandboxfs.CodeDeadlineExceeded, 0, sandboxwire.EffectPossible) + syscall.Flock(int(native.Fd()), syscall.LOCK_UN) + time.Sleep(200 * time.Millisecond) + if err := nativeTry(); err != nil { + t.Fatalf("native flock after the cancelled wait: %v", err) + } + + _, err = f.c.SetLock(context.Background(), &sandboxfs.SetLockRequest{Handle: h, Kind: sandboxfs.LockPOSIX, Lock: whole}) + wantFailure(t, err, sandboxfs.CodeUnsupported, 0, sandboxwire.EffectNone) +} + +func TestIncarnationChange(t *testing.T) { + f := newFixture(t) + _, h := f.create("f", sandboxfs.AccessRead, 0) + next, err := New(f.dir) + if err != nil { + t.Fatal(err) + } + defer next.Close() + + // A stream bound to the old incarnation. + old, _, _ := f.connect(next, f.att) + _, err = old.Lookup(context.Background(), &sandboxfs.LookupRequest{Parent: f.root, Name: []byte("f")}) + wantFailure(t, err, sandboxfs.CodeInstanceChanged, 0, sandboxwire.EffectNone) + + // The same attachment rebound to the new incarnation holds nothing. + a := f.att + a.ServerInstanceID = next.InstanceID() + c, _, _ := f.connect(next, a) + _, err = c.Lookup(context.Background(), &sandboxfs.LookupRequest{Parent: f.root, Name: []byte("f")}) + wantFailure(t, err, sandboxfs.CodeStaleAttachment, 0, sandboxwire.EffectNone) + if _, err := c.Attach(context.Background(), &sandboxfs.AttachRequest{Export: "world"}); err != nil { + t.Fatal(err) + } + _, err = c.Lookup(context.Background(), &sandboxfs.LookupRequest{Parent: f.root, Name: []byte("f")}) + wantFailure(t, err, sandboxfs.CodeStaleNode, 0, sandboxwire.EffectNone) + _, err = c.Read(context.Background(), &sandboxfs.ReadRequest{Handle: h, Size: 1}) + wantFailure(t, err, sandboxfs.CodeStaleHandle, 0, sandboxwire.EffectNone) +} + +func TestAttachFollowsExportGrants(t *testing.T) { + f := newFixture(t) + ctx := context.Background() + + c, _, _ := f.connect(f.svc, attachment(f.svc, sandboxlink.ExportGrant{ID: "logs"})) + _, err := c.Attach(ctx, &sandboxfs.AttachRequest{Export: "world"}) + wantFailure(t, err, sandboxfs.CodeUnauthorized, 0, sandboxwire.EffectNone) + + c, _, _ = f.connect(f.svc, attachment(f.svc, sandboxlink.ExportGrant{ID: "world", ReadOnly: true})) + _, err = c.Attach(ctx, &sandboxfs.AttachRequest{Export: "world"}) + wantFailure(t, err, sandboxfs.CodeUnauthorized, 0, sandboxwire.EffectNone) + r, err := c.Attach(ctx, &sandboxfs.AttachRequest{Export: "world", ReadOnly: true}) + if err != nil { + t.Fatal(err) + } + _, err = c.Create(ctx, &sandboxfs.CreateRequest{Parent: r.Root.Node, Name: []byte("f"), Mode: 0o644, Access: sandboxfs.AccessWrite}) + wantFailure(t, err, sandboxfs.CodeErrno, sandboxfs.ErrnoReadOnlyFilesystem, sandboxwire.EffectNone) +} + +func TestDescribeListsGrantedExports(t *testing.T) { + f := newFixture(t) + c, _, _ := f.connect(f.svc, attachment(f.svc, sandboxlink.ExportGrant{ID: "logs"})) + d, err := c.Describe(context.Background(), &sandboxfs.DescribeRequest{}) + if err != nil || len(d.Exports) != 0 { + t.Fatalf("describe without a world grant: %+v, %v", d, err) + } +} + +// The service never resolves through a proc magic link: oac-sandbox-io +// serves "/", where /proc//cwd links to another directory. +func TestProcMagicLinkIsOpaque(t *testing.T) { + f := attachRoot(t, "/") + ctx := context.Background() + w, err := f.c.Walk(ctx, &sandboxfs.WalkRequest{Parent: f.root, Names: [][]byte{[]byte("proc"), []byte(strconv.Itoa(os.Getpid())), []byte("cwd")}}) + if err != nil || len(w.Entries) != 3 || w.Entries[2].Attr.Mode&sandboxfs.ModeType != sandboxfs.ModeSymlink { + t.Fatalf("walk to /proc//cwd: %+v, %v", w, err) + } + cwd := w.Entries[2].Node + _, err = f.c.Lookup(ctx, &sandboxfs.LookupRequest{Parent: cwd, Name: []byte("x")}) + wantFailure(t, err, sandboxfs.CodeErrno, sandboxfs.ErrnoNotDirectory, sandboxwire.EffectNone) + _, err = f.c.OpenDir(ctx, &sandboxfs.OpenDirRequest{Node: cwd}) + wantFailure(t, err, sandboxfs.CodeErrno, sandboxfs.ErrnoNotDirectory, sandboxwire.EffectNone) + _, err = f.c.Open(ctx, &sandboxfs.OpenRequest{Node: cwd, Access: sandboxfs.AccessRead}) + wantFailure(t, err, sandboxfs.CodeErrno, sandboxfs.ErrnoSymlinkLoop, sandboxwire.EffectNone) +} + +// A bind mount and its source are one inode on two mounts, and stay two +// nodes. Bind mounts need a mount namespace, so the test reruns itself under +// unshare when unprivileged user namespaces are available. +func TestBindAliasesStayDistinct(t *testing.T) { + if os.Getenv("OAC_TEST_IN_USERNS") == "" { + unshare, err := exec.LookPath("unshare") + if err != nil || exec.Command(unshare, "-Urm", "true").Run() != nil { + t.Skip("unprivileged user and mount namespaces are unavailable") + } + cmd := exec.Command(unshare, "-Urm", os.Args[0], "-test.run=^TestBindAliasesStayDistinct$", "-test.v") + cmd.Env = append(os.Environ(), "OAC_TEST_IN_USERNS=1") + if out, err := cmd.CombinedOutput(); err != nil || !bytes.Contains(out, []byte("--- PASS")) { + t.Fatalf("in a user namespace: %v\n%s", err, out) + } + return + } + dir := t.TempDir() + src, alias := filepath.Join(dir, "src"), filepath.Join(dir, "alias") + for _, d := range []string{src, alias} { + if err := os.Mkdir(d, 0o755); err != nil { + t.Fatal(err) + } + } + if err := syscall.Mount(src, alias, "", syscall.MS_BIND, ""); err != nil { + t.Fatal(err) + } + defer syscall.Unmount(alias, syscall.MNT_DETACH) + f := attachRoot(t, dir) + a, b := f.lookup(f.root, "src"), f.lookup(f.root, "alias") + if a.Node == b.Node || a.Attr.Ino != b.Attr.Ino { + t.Fatalf("source %+v and bind alias %+v", a, b) + } +} + +// The fdinfo fallback reports the mount ID statx does. +func TestMountIDMechanismsAgree(t *testing.T) { + f := newFixture(t) + fd := int(f.svc.root.Fd()) + viaStatx, err := f.svc.mountID(fd) + if err != nil { + t.Fatal(err) + } + if viaFdinfo, err := f.svc.fdinfoMountID(fd); err != nil || viaFdinfo != viaStatx { + t.Fatalf("fdinfo mount ID %d, %v; statx %d", viaFdinfo, err, viaStatx) + } +} + +func TestFailureAfterAChangeIsEffectPossible(t *testing.T) { + wantFailure(t, failure(errStaleAttachment(), possible), sandboxfs.CodeStaleAttachment, 0, possible) +} + +// wroteConn reports each completed write. +type wroteConn struct { + net.Conn + wrote chan struct{} +} + +func (c wroteConn) Write(p []byte) (int, error) { + n, err := c.Conn.Write(p) + c.wrote <- struct{}{} + return n, err +} + +func TestTransportLoss(t *testing.T) { + f := newFixture(t) + _, h := f.create("busy", sandboxfs.AccessRead, 0) + native, err := os.Open(filepath.Join(f.dir, "busy")) + if err != nil { + t.Fatal(err) + } + defer native.Close() + if err := syscall.Flock(int(native.Fd()), syscall.LOCK_EX); err != nil { + t.Fatal(err) + } + + // A second stream of the same attachment reuses its handles. + cc, sc := net.Pipe() + served := make(chan error, 1) + go func() { served <- sandboxfs.Serve(context.Background(), sc, f.svc, f.att) }() + wrote := make(chan struct{}, 1) + c := sandboxfs.NewClient(wroteConn{cc, wrote}) + defer c.Close() + result := make(chan error, 1) + go func() { + _, err := c.SetLock(context.Background(), &sandboxfs.SetLockRequest{Handle: h, Kind: sandboxfs.LockFlock, Lock: sandboxfs.Lock{Mode: sandboxfs.LockWrite, End: math.MaxInt64}, Wait: true}) + result <- err + }() + <-wrote // the server has read the request + sc.Close() + + fail := wantFailure(t, <-result, sandboxfs.CodeUnknown, 0, sandboxwire.EffectPossible) + if !errors.Is(fail, sandboxfs.ErrTransport) { + t.Fatalf("%v is not a transport failure", fail) + } + _, err = c.GetAttr(context.Background(), &sandboxfs.GetAttrRequest{Target: sandboxfs.Target{Kind: sandboxfs.TargetHandle, Handle: h}}) + wantFailure(t, err, sandboxfs.CodeUnknown, 0, sandboxwire.EffectNone) + select { + case <-served: + case <-time.After(5 * time.Second): + t.Fatal("Serve did not return after the transport closed") + } +} diff --git a/docs/development.md b/docs/development.md index d33ada6f..c9d86154 100644 --- a/docs/development.md +++ b/docs/development.md @@ -66,6 +66,8 @@ For frontend development, run `pnpm dev:web` using the fixture or Core connectio | `internal/sandboxwire` | Frame header, primitive encoding and request ID sequence shared by the sandbox I/O protocols | [Framing](sandbox-link-protocol.md#framing) | | `internal/sandboxlink` | Link protocol, peer libraries and relay core | [Sandbox link protocol](sandbox-link-protocol.md) | | `internal/sandboxbootstrap` | Provider-to-Sandbox I/O service startup input | [Sandbox bootstrap](sandbox-bootstrap.md) | +| `internal/sandboxfs` | Runtime–file service wire types, validators, client and server | [File access protocol](file-access-protocol.md) | +| `apps/sandboxio` | Sandbox I/O service and its Linux protocol services | [File access protocol](file-access-protocol.md#the-linux-service) | | `apps/daemon/internal/dispatch` | Runtime preparation, Executor reuse, Turn and cleanup ownership | [Harness lifecycle](../contracts/agents-api/harness-onboarding.md#required-adapter-interfaces) | | `apps/daemon/internal/agent` | Native harness adapters | [Native references](../contracts/agents-api/harness-onboarding.md#native-references) | | `services/core/internal/sandbox` | Provider interfaces and managed compute lifecycle | [Provider onboarding](sandbox-provider.md) | diff --git a/docs/file-access-protocol.md b/docs/file-access-protocol.md new file mode 100644 index 00000000..1496f95d --- /dev/null +++ b/docs/file-access-protocol.md @@ -0,0 +1,328 @@ +# File access protocol + +The File access protocol is how a Runtime reads and changes the files of a sandbox. The Sandbox I/O service in the sandbox serves it, and the Runtime is its client. It is a node and handle protocol shaped like the FUSE low-level operations: lookups acquire node references, opens return handles, reads and writes take offsets, directory reads resume at cookies, and locks work against the sandbox's own processes. Phase 1 serves the [Uncached](#uncached-profile) profile only, with no change stream. + +[`internal/sandboxfs/protocol.go`](../internal/sandboxfs/protocol.go) is the authored definition: message tags, payload layouts, validators and the `Service` interface. The same package holds the generic client and server. [`apps/sandboxio/internal/fileservice`](../apps/sandboxio/internal/fileservice) is the Linux service. Frames use the shared [framing](sandbox-link-protocol.md#framing), and the Link layer supplies the authenticated attachment of each stream. + +## Streams and attachments + +- A stream belongs to one attachment, which Link authenticates and hands to the server as a `sandboxfs.Attachment`: its ID, the `ServerInstanceID` the stream was bound to, its lease and the exports it is granted. No request names an attachment, an OS user or a credential. +- Request IDs follow the [framing](sandbox-link-protocol.md#framing) rule, and a response carries its request's RequestID. Requests on one stream run concurrently, so responses can arrive in any order. The protocol has no events. +- An attachment attaches to one export. Its node references, handles and locks live in the service until it detaches, its lease ends or the service restarts. A new stream of the same attachment in the same incarnation continues with them. +- A service incarnation is named by `ServerInstanceID`. A request on a stream bound to another incarnation fails with `InstanceChanged`, and nothing is reopened automatically. + +## Implement a client + +The Go client is `sandboxfs.NewClient(stream)`. It has one method per operation, is safe for concurrent use, and returns a `*sandboxfs.Failure` for every failure. Its methods map one to one onto the go-fuse node operations, so a FUSE frontend turns each kernel request into one call. + +1. Call `Describe`. Keep `ServerInstanceID`, and check each request against `Capabilities` before sending it; the service rejects anything the capabilities do not declare. +2. `Attach` an export and keep the root `NodeRef`. +3. `Lookup`, `Walk`, `Create`, `Mkdir`, `Symlink`, `Link` and `ReadDir` with `WithAttrs` each acquire one reference on every node they return. Release references with `Forget` when the kernel forgets them. +4. `Walk` stops after a symlink. Resolve the link yourself, relative to the view, with `Readlink` and further walks. +5. `Open`, `Create` and `OpenDir` return handles. Send `Flush` on each close of a descriptor for the handle, and `Release` or `ReleaseDir` when the last one closes. +6. Read the `Effect` of every failure. After `EffectPossible`, the request may have taken effect: never replay a mutation automatically. Report the failure, or inspect the state with `GetAttr` or `Lookup` first. + +Cancelling a call's context returns at once with `Cancelled` or `DeadlineExceeded`. A call cancelled before its request is written fails with `EffectNone`. After the request is written, the client sends `CancelRequest` for it and discards the late response, and the failure is `EffectNone` for a [side-effect-free request](#effects-and-cancellation) and `EffectPossible` for every other one. Cancelling a call while its request is being written fails the stream instead, because a partial frame cannot be withdrawn. A cancel that races the completion of a write may still fail the stream, and the requests in flight then fail with `EffectPossible`. + +When the stream fails, every request in flight fails with `Unknown` and `EffectPossible`, and later calls fail with `EffectNone`; `errors.Is(err, sandboxfs.ErrTransport)` matches both. `Client.Done` closes and `Client.Err` returns the cause. To continue, open a new stream for the same attachment with `ExpectedServerInstanceID` set, as the [Sandbox link protocol](sandbox-link-protocol.md) describes. `InstanceChanged` then means every node, handle and lock of the attachment is gone. + +## Implement a service + +Implement `sandboxfs.Service` and serve each stream with `sandboxfs.Serve(ctx, stream, service, attachment)`. `Serve`: + +- refuses an attachment without an ID, a `ServerInstanceID`, a lease or at least one valid export grant with unique IDs; +- decodes and validates each request, and answers a malformed payload with `InvalidArgument` and `EffectNone`; +- ends the stream on a framing violation: an unknown tag, a frame that is not a request, or a RequestID that does not increase; +- holds up to `sandboxfs.MaxInFlight` (256) requests, each from admission until its response is written, and answers any more with `ResourceExhausted` and `EffectNone`, so a `CancelRequest` arrives while the client reads responses; +- answers `CancelRequest` itself by cancelling the target's context; +- returns a method's `*Failure` as the typed failure, and reports any other error, or a response that fails validation, as `Unknown` with `EffectPossible`; +- cancels every request's context when the stream ends. + +A service must: + +- generate a new `ServerInstanceID` whenever it loses its node and handle tables, and answer a request whose `Attachment.ServerInstanceID` is not its own with `InstanceChanged`; +- answer `StaleAttachment` while the attachment is not attached or after its lease ends, and `StaleNode` or `StaleHandle` for a reference or handle the attachment does not hold. A node ID is reused only with a new generation; +- call `Capabilities.Admit(request, readOnly)` before running a request and return the failure it reports. Limits that depend on service state, such as `MaxOpenHandles`, stay with the service; +- advertise only what it enforces, and advertise locks only when they interoperate with native processes in the sandbox; +- act as its own process identity, never as an identity a request supplies, and apply requested permission bits exactly; +- resolve each name as one entry of a directory node and never traverse a symlink; +- report `EffectPossible` once any part of a mutation may have applied. + +### The Linux service + +`fileservice.New(root)` serves the absolute directory `root` as the one export `world`, and `Describe` lists `world` only to an attachment granted it. `oac-sandbox-io` passes `/`; the Provider's sandbox setup owns the isolation of everything under it, as the [Sandbox bootstrap](sandbox-bootstrap.md#responsibilities) states, and the service enforces no boundary inside the export. `New` sets the process umask to zero and reports its effective UID and GID as `Identity`. `InstanceID` returns the `ServerInstanceID` to give Link, and `Close` releases every attachment. + +- Each node holds an `O_PATH|O_NOFOLLOW` descriptor. In an attachment a node is one mount ID, device and inode, so hard links share a node while a bind mount and its source stay two. The mount ID comes from `statx` with `STATX_MNT_ID`, or from the `mnt_id` line of `/proc/self/fdinfo/` on kernels older than 5.8. +- A lookup opens one component with `openat` and `O_NOFOLLOW` on its parent's descriptor. A symlink, including a proc magic link such as `/proc//cwd`, is a node of its own and is never traversed: `Lookup` and `Readlink` return the link itself, a directory operation on it fails with `Errno` `NotDirectory`, and `Open` fails with `SymlinkLoop`. +- Operations on a node, such as opening, truncating, changing its mode or times and linking it, go through the `/proc/self/fd` name of the descriptor the service holds for it. That name resolves to the descriptor's own object, never to a symlink's target, and an open through it first checks the object's file type. +- An open handle holds its own descriptor, so it keeps working after its file is unlinked or renamed. +- `Rename` uses `renameat2`. The service declares `RenameNoReplace` and `RenameExchange` only when a probe at start succeeds. +- `LockFlock` locks the handle's descriptor, so it interoperates with native `flock`. `POSIXLocks` is false, and `GetLock` and `LockPOSIX` return `Unsupported`. +- `ReadDir` cookies are the kernel's directory offsets. `Attr.Ino` combines the device and the inode number as go-fuse's loopback does. +- It declares `MaxNameBytes` 255, `MaxPathBytes` 4095, `MaxReadBytes` and `MaxWriteBytes` 64 KiB, `MaxWalkComponents` 256, `MaxReadDirBytes` 64 KiB and `MaxOpenHandles` 4096, and every flag except `ReadOnly` and `POSIXLocks`, with the rename modes as probed. + +## Reference + +### Messages + +Requests use tags 1 to 30; the response to tag `t` uses `t | 0x8000`. + +| Tag | Request | Fields | Response | Meaning | +| --- | --- | --- | --- | --- | +| 1 | `Describe` | – | `ServerInstanceID`, `Identity`, `Capabilities`, `Exports` | The incarnation, identity, capabilities and granted exports | +| 2 | `Attach` | `Export`, `ReadOnly` | `Root` (`Entry`) | Attach to a granted export. The root is a directory | +| 3 | `Detach` | – | – | Release every node, handle and lock of the attachment | +| 4 | `Lookup` | `Parent`, `Name` | `Entry` | Look up one entry | +| 5 | `Walk` | `Parent`, `Names` | `Entries`, optional `Failure` | Look up components in turn. Stops after a symlink, or at a failure that `Failure` reports; a failure at the first name fails the request | +| 6 | `GetAttr` | `Target` | `Attr` | Attributes of a node or open handle | +| 7 | `SetAttr` | `Target`, `Set`, selected values | `Attr` | Change the selected attributes | +| 8 | `Access` | `Node`, `Mask` | – | Check permissions as the service's identity | +| 9 | `Open` | `Node`, `Access`, `Flags` | `Handle` | Open a regular file | +| 10 | `Create` | `Parent`, `Name`, `Mode`, `Access`, `Flags`, `Exclusive` | `Entry`, `Handle` | Create and open a regular file | +| 11 | `Read` | `Handle`, `Offset`, `Size` | `Data` | Fewer bytes than asked means end of file | +| 12 | `Write` | `Handle`, `Offset`, `Data` | `Written`, optional `Failure` | Write at `Offset`, or append on an append-opened handle | +| 13 | `Flush` | `Handle`, `Owner` | – | One descriptor for the handle closed | +| 14 | `Fsync` | `Handle`, `DataOnly` | – | Make the file's data, and its metadata unless `DataOnly`, durable | +| 15 | `Release` | `Handle` | – | Close a file handle | +| 16 | `OpenDir` | `Node` | `Handle` | Open a directory | +| 17 | `ReadDir` | `Handle`, `Cookie`, `Limit`, `WithAttrs` | `Entries`, `End` | Read entries after `Cookie` | +| 18 | `ReleaseDir` | `Handle` | – | Close a directory handle | +| 19 | `Mkdir` | `Parent`, `Name`, `Mode` | `Entry` | Create a directory | +| 20 | `Unlink` | `Parent`, `Name` | – | Remove a non-directory entry | +| 21 | `Rmdir` | `Parent`, `Name` | – | Remove an empty directory | +| 22 | `Rename` | `Parent`, `Name`, `NewParent`, `NewName`, `Mode` | – | Rename in one [mode](#rename) | +| 23 | `Link` | `Node`, `NewParent`, `NewName` | `Entry` | Create a hard link | +| 24 | `Symlink` | `Parent`, `Name`, `Target` | `Entry` | Create a symlink | +| 25 | `Readlink` | `Node` | `Target` | Read a symlink | +| 26 | `StatFS` | `Node` | File-system statistics | Statistics of the file system holding the node | +| 27 | `Forget` | `Entries` | – | Release lookup references, all or none | +| 28 | `GetLock` | `Handle`, `Owner`, `Lock` | optional `Conflict` | A POSIX lock that would conflict | +| 29 | `SetLock` | `Handle`, `Kind`, `Owner`, `Lock`, `Wait` | – | Acquire, convert or release a lock | +| 30 | `CancelRequest` | `Target` (a RequestID) | – | Ask to cancel an outstanding request | + +A response payload begins with a uint16 result: 1 for success, followed by the response's fields, or 2 for failure, followed by a [`Failure`](#failures). Payloads list their fields in this order: + +```text +Describe (no fields) +DescribeResponse ServerInstanceID ID, Identity {UID u32, GID u32}, Capabilities, Exports count 0..64 of bytes +Attach Export bytes, ReadOnly bool +AttachResponse Root Entry +Detach (no fields); response (no fields) +Lookup Parent NodeRef, Name bytes +LookupResponse Entry +Walk Parent NodeRef, Names count 1..1024 of bytes +WalkResponse Entries count 1..1024 of Entry, Failure optional Failure +GetAttr Target +GetAttrResponse Attr +SetAttr Target, Set u32, then for each selected bit in order: Size u64, Mode u32, UID u32, GID u32, Atime Timestamp, Mtime Timestamp +SetAttrResponse Attr +Access Node NodeRef, Mask u32; response (no fields) +Open Node NodeRef, Access enum, Flags u32 +OpenResponse Handle u64 +Create Parent NodeRef, Name bytes, Mode u32, Access enum, Flags u32, Exclusive bool +CreateResponse Entry, Handle u64 +Read Handle u64, Offset u64, Size u32 +ReadResponse Data bytes +Write Handle u64, Offset u64, Data bytes +WriteResponse Written u32, Failure optional Failure +Flush Handle u64, Owner u64; response (no fields) +Fsync Handle u64, DataOnly bool; response (no fields) +Release Handle u64; response (no fields) +OpenDir Node NodeRef +OpenDirResponse Handle u64 +ReadDir Handle u64, Cookie u64, Limit u32, WithAttrs bool +ReadDirResponse Entries count of DirEntry, End bool +ReleaseDir Handle u64; response (no fields) +Mkdir Parent NodeRef, Name bytes, Mode u32 +MkdirResponse Entry +Unlink, Rmdir Parent NodeRef, Name bytes; response (no fields) +Rename Parent NodeRef, Name bytes, NewParent NodeRef, NewName bytes, Mode enum; response (no fields) +Link Node NodeRef, NewParent NodeRef, NewName bytes +LinkResponse Entry +Symlink Parent NodeRef, Name bytes, Target bytes +SymlinkResponse Entry +Readlink Node NodeRef +ReadlinkResponse Target bytes +StatFS Node NodeRef +StatFSResponse Blocks u64, BlocksFree u64, BlocksAvailable u64, Files u64, FilesFree u64, BlockSize u32, FragmentSize u32, NameMax u32 +Forget Entries count 1..4096 of {Node NodeRef, Count u64}; response (no fields) +GetLock Handle u64, Owner u64, Lock +GetLockResponse Conflict optional Lock +SetLock Handle u64, Kind enum, Owner u64, Lock, Wait bool; response (no fields) +CancelRequest Target u64; response (no fields) +``` + +[`testdata`](../internal/sandboxfs/testdata) holds annotated golden frames of `Describe`, `Walk`, `Create`, a short `Write`, `ReadDir` with a cookie, `Rename` and a failure. + +### Shared types + +```text +NodeRef ID u64, Generation u64 // both nonzero +HandleID u64 // nonzero, opaque +Timestamp Sec i64, Nsec u32 // Nsec below one billion +Attr Ino u64, Mode u32, Nlink u32, UID u32, GID u32, Rdev u64, Size u64, Blocks u64, Blksize u32, Atime Timestamp, Mtime Timestamp, Ctime Timestamp +Entry Node NodeRef, Attr +Target Kind enum (TargetNode = 1, TargetHandle = 2), then Node NodeRef or Handle u64 +DirEntry Name bytes, Ino u64, Type u32, Cookie u64, Entry optional Entry +Lock Mode enum (LockRead = 1, LockWrite = 2, LockUnlock = 3), Start u64, End u64 +``` + +- A `NodeRef` and a `HandleID` are scoped to the attachment and the service incarnation. A node ID is reused only after its object is forgotten, and then with another generation, so a stale `NodeRef` never names another object. +- `Attr.Mode` holds exactly one file type and the permission bits (`0o7777`) in the Linux `st_mode` layout. `Size` is at most 2^63−1, and `Blocks` counts 512-byte blocks. `Ino` identifies the file within the export. +- A `DirEntry`'s `Type` is one file type in the same layout. When `Entry` is present, its `Attr.Ino` and file type equal the entry's. +- `Blocks`, `Files` and the other `StatFSResponse` fields follow `statfs`: `BlocksAvailable` is what an unprivileged process may use. + +### Describe and capabilities + +`DescribeResponse` carries the incarnation, the identity, the capabilities and the declared exports the attachment is granted: 0 to 64 unique `ExportID`s, with the grammar of Link's [export grants](sandbox-link-protocol.md#opening-a-stream). `Identity` is the effective UID and GID the service runs as, which own the files it creates; it is informational, and no request carries or selects an identity. + +`Capabilities` encodes every field, in this order: + +| Field | Type | Meaning | +| --- | --- | --- | +| `PathProfile` | enum | `LinuxBytes` (1): names and symlink targets are Linux byte strings | +| `CacheProfile` | enum | `Uncached` (1), the [Uncached profile](#uncached-profile) | +| `Durability` | enum | `FsyncRequired` (1): data is durable only after `Fsync` succeeds | +| `MaxNameBytes` | u32 | Longest entry name, 1 to 1024 | +| `MaxPathBytes` | u32 | Longest symlink target, 1 to 4096 | +| `MaxReadBytes`, `MaxWriteBytes` | u32 | Largest `Read` size and `Write` data, 1 to 64 KiB | +| `MaxWalkComponents` | u32 | Most `Walk` names, 1 to 1024 | +| `MaxReadDirBytes` | u32 | Largest `ReadDir` limit, 1 to 256 KiB | +| `MaxOpenHandles` | u32 | Most open handles per attachment, at least 1 | +| `ReadOnly` | bool | The service accepts only read-only attachments | +| `AtomicAppend` | bool | Appends from several handles never interleave within a write | +| `AtomicRename` | bool | `RenameReplace` replaces the destination atomically | +| `RenameNoReplace`, `RenameExchange` | bool | The rename mode is supported | +| `HardLinks`, `Symlinks` | bool | `Link` and `Symlink` are supported | +| `SetMode`, `SetOwner`, `SetTimes` | bool | `SetAttr` may set the mode, the owner and the times | +| `DirectoryFsync` | bool | `Fsync` of a directory handle makes its entries durable | +| `ReadDirPlus` | bool | `ReadDir` supports `WithAttrs` | +| `Flock` | bool | `SetLock` supports `LockFlock` | +| `POSIXLocks` | bool | `GetLock` and `SetLock` support `LockPOSIX` | + +A writable service, one without `ReadOnly`, declares `AtomicAppend`, `AtomicRename`, `HardLinks` and `Symlinks`. A request beyond a declared limit fails with `InvalidArgument`, or `Errno` `NameTooLong` for a name or target, and a request for an undeclared feature fails with `Unsupported`. + +### Attach + +`Attach` selects one export by `ExportID`; it never takes a server path. The export must be one that the attachment's Link binding grants in `Attachment.Exports` (see the [Sandbox link protocol](sandbox-link-protocol.md)), and a read-only grant allows only `ReadOnly` attaches; otherwise `Attach` fails with `Unauthorized`. A granted export the service does not declare fails with `InvalidArgument`, as does a second `Attach` before `Detach`. Every request that changes files on a read-only attachment fails with `Errno` `ReadOnlyFilesystem`: `SetAttr`, `Create`, `Write`, `Mkdir`, `Unlink`, `Rmdir`, `Rename`, `Link`, `Symlink`, and `Open` for writing or with `OpenTruncate`. + +### Names and paths + +- Names and symlink targets are bytes, not UTF-8 strings. +- An entry name is 1 to 1024 bytes without NUL or `/`, and is never `.` or `..`. `ReadDir` never returns `.` or `..`. +- A symlink target is 1 to 4096 bytes without NUL. The service stores and returns it unchanged; the client resolves it. +- No request resolves a path or traverses a symlink. Every lookup names one entry of a directory node, and a request that needs a directory fails with `Errno` `NotDirectory` on any other node. + +### Attributes + +`SetAttr` addresses a node or an open handle and changes only the attributes its `Set` mask selects: + +| Bit | Name | Value | +| --- | --- | --- | +| `0x01` | `AttrSize` | `Size`: truncate or extend a regular file | +| `0x02` | `AttrMode` | `Mode`: permission bits only | +| `0x04` | `AttrUID` | `UID` | +| `0x08` | `AttrGID` | `GID` | +| `0x10` | `AttrAtime` | `Atime` | +| `0x20` | `AttrMtime` | `Mtime` | +| `0x40` | `AttrAtimeNow` | None: the access time becomes the service's current time | +| `0x80` | `AttrMtimeNow` | None: the modification time becomes the service's current time | + +Unknown bits are rejected, `AttrAtime` excludes `AttrAtimeNow` and `AttrMtime` excludes `AttrMtimeNow`, and an unselected value must be zero. A selected `UID` or `GID` of 4294967295, which `chown` reads as no change, is rejected. Setting the mode of a symlink fails with `Errno` `NotSupported`; times set on a symlink apply to the link itself. The response carries the attributes after the change. + +`Access` checks the `Mask` bits `MayExecute` (1), `MayWrite` (2) and `MayRead` (4) as the service's identity; a zero mask checks existence. + +### Files + +- `AccessMode` is `AccessRead` (1), `AccessWrite` (2) or `AccessReadWrite` (3). +- `OpenFlags` are `OpenAppend` (1), `OpenTruncate` (2), `OpenNoFollow` (4), `OpenSync` (8) and `OpenDataSync` (16); unknown flags are rejected. +- `Open` opens a regular file node. A node is an object, not a path, so `Open` never follows a symlink: a directory fails with `Errno` `IsDirectory`, a symlink with `SymlinkLoop`, and a special file with `Unsupported`. +- `Create` creates a regular file with exactly the requested permission bits; no umask applies. With `Exclusive`, an existing entry fails with `Errno` `Exists`. Without it, an existing regular file is opened, and `OpenTruncate` truncates it. +- `Read` takes an offset up to 2^63−1. A positioned `Write` must end at or before 2^63−1, or it fails with `InvalidArgument`. A handle opened with `OpenAppend` appends each write atomically and ignores `Offset`. +- A successful `WriteResponse` carries the exact number of bytes written. When a write stops after a nonzero prefix, `Written` counts the prefix and `Failure` says why it stopped; a write that wrote nothing is a failure response. +- `Flush` is neither `Fsync` nor `Release`. It reports the errors of closing one descriptor for the handle, and `Owner` names the closing lock owner. +- A file operation on a directory handle, or a directory operation on a file handle, fails with `Errno` `IsDirectory`, `NotDirectory` or `BadDescriptor`. + +### Directories + +- `OpenDir` opens a directory node, and `ReadDir` reads its entries after `Cookie`; cookie 0 is the start. Each `DirEntry.Cookie` is the position after that entry and is otherwise opaque. +- `Limit` bounds the sum of the returned entries' encoded sizes: 25 bytes plus the name for each entry, plus 104 when it carries an `Entry`. A limit too small for the first entry fails with `Errno` `InvalidArgument`. +- With `WithAttrs`, each returned entry carries an `Entry` with one lookup reference. +- `End` is true when no entries remain, and a page without entries always has it set. No name or cookie repeats within a page. `ReadDir` promises no snapshot of a directory that changes while it is read. + +### Rename + +`Rename` takes exactly one mode: `RenameReplace` (1) replaces an existing destination, `RenameNoReplace` (2) fails with `Errno` `Exists` when the destination exists, and `RenameExchange` (3) swaps two existing entries. + +### Locks + +- A `Lock` covers `Start` through `End` inclusive; `End` 2^63−1 extends to the end of the file. +- `SetLock` takes `LockPOSIX` (1) or `LockFlock` (2), an attachment-scoped `LockOwner`, a range, a mode and `Wait`. A flock lock covers the whole file, so its range is 0 to 2^63−1. Without `Wait`, a conflicting lock fails with `Errno` `Again`; with it, the request waits until the lock is free or the request is cancelled. +- `GetLock` asks for a POSIX lock that would conflict with a read or write `Lock`, and returns it in `Conflict`. +- A service advertises a lock kind only when its locks interoperate with native processes in the sandbox. A client-local lock table is not enough. + +### Failures + +```text +Failure + Code enum + Errno optional enum // present exactly when Code is Errno + Effect Effect + Message bytes // at most 1024 bytes, for people +``` + +| Code | Name | Returned when | +| --- | --- | --- | +| 1 | `InvalidArgument` | A payload fails validation, a request exceeds a declared limit, the export is unknown, the attachment is already attached, or `Forget` exceeds the references held | +| 2 | `Unsupported` | The capabilities do not declare the operation or option, or the request needs a feature phase 1 excludes | +| 3 | `Unauthorized` | `Attach` names an export the Link binding does not grant, or asks for write access to a read-only grant | +| 4 | `StaleAttachment` | The attachment is not attached, has detached or its lease ended | +| 5 | `InstanceChanged` | The stream is bound to another service incarnation | +| 6 | `StaleNode` | The attachment holds no such `NodeRef` | +| 7 | `StaleHandle` | The attachment has no such open handle | +| 8 | `ResourceExhausted` | The stream holds `MaxInFlight` requests, or the attachment has `MaxOpenHandles` handles | +| 9 | `Cancelled` | The request was cancelled | +| 10 | `DeadlineExceeded` | The caller's deadline passed | +| 11 | `Errno` | A file-system call failed; `Errno` says how | +| 12 | `Unknown` | The stream failed, or the service failed without a typed error | + +`Errno` is a semantic enum, numbered from 1 in this order: `PermissionDenied`, `OperationNotPermitted`, `NotFound`, `Exists`, `NotDirectory`, `IsDirectory`, `DirectoryNotEmpty`, `InvalidArgument`, `BadDescriptor`, `TooManyOpenFiles`, `NoSpace`, `QuotaExceeded`, `ReadOnlyFilesystem`, `CrossDevice`, `NameTooLong`, `SymlinkLoop`, `FileTooLarge`, `Overflow`, `Busy`, `Again`, `Interrupted`, `IO`, `NoDevice`, `NoSuchDeviceOrAddress`, `BrokenPipe`, `NotSupported`, `NoLocks`, `Deadlock`. Each service converts its native errors; an unknown native error is `IO`, and success is never fabricated. + +### Effects and cancellation + +Every failure carries `EffectNone`, when the request certainly changed nothing, or `EffectPossible`. Transport loss after a request was written is `EffectPossible` unless the server later establishes the result. + +`Describe`, `GetAttr`, `Access`, `Read`, `Readlink`, `StatFS`, `GetLock` and `ReadDir` without `WithAttrs` leave no state behind, so abandoning one is `EffectNone`. Every other request may create, change or acquire something. + +An ambiguous mutation is never replayed automatically. This includes `Create`, a truncating `Open`, `SetAttr`, `Write` and a lock acquisition. + +`CancelRequest` asks the server to stop an outstanding request. Its acknowledgement is not the target's result and never proves that a mutation did not happen; the target's own response still follows. + +### Uncached profile + +`Uncached` means zero entry, attribute and negative TTLs, direct file I/O, no writeback cache and no directory listing retained by the client. It does not prohibit the server kernel from buffering data before `Fsync`. + +### Not in phase 1 + +Phase 1 has no operations and no negotiated support for mmap, extended attributes, allocation or hole punching, copy-range, creating or opening special files, and change notifications. The client returns the matching unsupported outcome for these. Attributes still describe an existing special file. + +### Limits + +| Limit | Value | +| --- | --- | +| Frame payload | 1 MiB | +| `Read` size, `Write` data | 64 KiB | +| Entry name | 1024 bytes | +| Symlink target | 4096 bytes | +| `Walk` names | 1024 | +| `ReadDir` limit | 256 KiB | +| `Forget` entries | 4096 | +| Declared exports | 64 | +| Failure message | 1024 bytes | +| Requests in flight per stream | 256 | + +Capabilities may declare smaller limits. + +## Verification + +`go test ./internal/sandboxfs` covers the golden frames, a round trip of every message, decode rejection and admission while responses go unread. `go test -run '^$' -fuzz FuzzDecode ./internal/sandboxfs` fuzzes the decoder. `go test ./apps/sandboxio/...` runs the Linux service over an in-memory stream against a temporary export, including exclusive create, atomic append, rename modes, directory paging, the root-escape attempts, opaque symlinks and proc magic links, bind-mount aliases, flock against native `flock`, export grants, an incarnation change and transport loss. The bind-mount test reruns itself under `unshare -Urm` and is skipped where unprivileged user namespaces are unavailable. diff --git a/internal/sandboxfs/client.go b/internal/sandboxfs/client.go new file mode 100644 index 00000000..7e41e393 --- /dev/null +++ b/internal/sandboxfs/client.go @@ -0,0 +1,391 @@ +package sandboxfs + +import ( + "context" + "errors" + "fmt" + "io" + "sync" + "sync/atomic" + + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxwire" +) + +// ErrTransport is the cause of a failure the client reports because the +// stream failed or closed. errors.Is finds it through the *Failure. +var ErrTransport = errors.New("sandboxfs: transport lost") + +// Client issues File requests over one stream. Its methods are safe for +// concurrent use; each returns a *Failure on failure. +type Client struct { + conn io.ReadWriteCloser + wsem chan struct{} // held while a RequestID is allocated and its frame written + seq sandboxwire.RequestSequence // used only while wsem is held + + mu sync.Mutex + pending map[uint64]*call + abandoned map[uint64]Op // requests whose response is discarded + err error + done chan struct{} + + afterWrite func() // test seam: runs once a frame is recorded as written +} + +type call struct { + op Op + ch chan outcome +} + +type outcome struct { + msg message + fail *Failure +} + +// NewClient starts a client on conn. The client owns conn and closes it on +// Close or when the stream fails. +func NewClient(conn io.ReadWriteCloser) *Client { + c := &Client{conn: conn, wsem: make(chan struct{}, 1), pending: map[uint64]*call{}, abandoned: map[uint64]Op{}, done: make(chan struct{})} + go c.readLoop() + return c +} + +// Close closes the stream. Requests in flight fail with EffectPossible. +func (c *Client) Close() error { + c.shutdown(errClientClosed) + return nil +} + +var errClientClosed = errors.New("client closed") + +// Done is closed once the stream has failed or closed. +func (c *Client) Done() <-chan struct{} { return c.done } + +// Err returns why the stream ended, or nil while it is open. +func (c *Client) Err() error { + c.mu.Lock() + defer c.mu.Unlock() + return c.err +} + +func transportFailure(effect sandboxwire.Effect, cause error) *Failure { + f := NewFailure(CodeUnknown, effect, "transport lost: "+cause.Error()) + f.cause = fmt.Errorf("%w: %w", ErrTransport, cause) + return f +} + +func (c *Client) shutdown(cause error) { + c.mu.Lock() + if c.err != nil { + c.mu.Unlock() + return + } + c.err = cause + pending := c.pending + c.pending, c.abandoned = nil, nil + close(c.done) + c.mu.Unlock() + c.conn.Close() + for _, cl := range pending { + cl.ch <- outcome{fail: transportFailure(sandboxwire.EffectPossible, cause)} + } +} + +// send registers a request and writes it. A request that ends before its +// write starts fails with EffectNone. Cancelling ctx during the write fails +// the stream, since a partial frame cannot be taken back, and the request +// then fails with EffectPossible. Once the write is recorded as finished, a +// cancellation is left to roundTrip, which sends CancelRequest. +func (c *Client) send(ctx context.Context, op Op, payload []byte) (uint64, *call, *Failure) { + if fail := c.acquire(ctx); fail != nil { + return 0, nil, fail + } + defer c.release() + if err := ctx.Err(); err != nil { + return 0, nil, contextFailure(err, sandboxwire.EffectNone) + } + c.mu.Lock() + if c.err != nil { + err := c.err + c.mu.Unlock() + return 0, nil, transportFailure(sandboxwire.EffectNone, err) + } + id, cl := c.seq.Next(), &call{op: op, ch: make(chan outcome, 1)} + c.pending[id] = cl + c.mu.Unlock() + // The write and the cancellation callback race for state; whichever + // moves it from writing decides. The callback is settled before the + // write slot is released, so it never acts on a later write. + const writing, written, interrupted = 0, 1, 2 + var state atomic.Int32 + settled := make(chan struct{}) + stop := context.AfterFunc(ctx, func() { + defer close(settled) + if state.CompareAndSwap(writing, interrupted) { + c.shutdown(fmt.Errorf("request cancelled while being written: %w", context.Cause(ctx))) + } + }) + err := sandboxwire.WriteFrame(c.conn, sandboxwire.Frame{Type: uint16(op), RequestID: id, Payload: payload}) + state.CompareAndSwap(writing, written) + if c.afterWrite != nil { + c.afterWrite() + } + if !stop() { + <-settled + } + if err != nil { + c.shutdown(err) + } + return id, cl, nil +} + +// acquire takes the write turn unless ctx ends or the stream fails first. +func (c *Client) acquire(ctx context.Context) *Failure { + select { + case c.wsem <- struct{}{}: + return nil + case <-ctx.Done(): + return contextFailure(ctx.Err(), sandboxwire.EffectNone) + case <-c.done: + return transportFailure(sandboxwire.EffectNone, c.Err()) + } +} + +func (c *Client) release() { <-c.wsem } + +// abandon stops waiting for request id. It reports false when the outcome +// has already been delivered. +func (c *Client) abandon(id uint64, cl *call) bool { + c.mu.Lock() + if c.pending[id] != cl { + c.mu.Unlock() + return false + } + delete(c.pending, id) + c.abandoned[id] = cl.op + c.mu.Unlock() + go c.cancel(id) + return true +} + +func (c *Client) cancel(target uint64) { + payload, err := encodeRequest(&CancelRequestRequest{Target: target}) + if err != nil || c.acquire(context.Background()) != nil { + return + } + defer c.release() + c.mu.Lock() + if c.err != nil { + c.mu.Unlock() + return + } + id := c.seq.Next() + c.abandoned[id] = OpCancelRequest + c.mu.Unlock() + if err := sandboxwire.WriteFrame(c.conn, sandboxwire.Frame{Type: uint16(OpCancelRequest), RequestID: id, Payload: payload}); err != nil { + c.shutdown(err) + } +} + +func (c *Client) readLoop() { + for { + f, err := sandboxwire.ReadFrame(c.conn, sandboxwire.MaxPayload) + if err == nil { + err = c.deliver(f) + } + if err != nil { + c.shutdown(err) + return + } + } +} + +// deliver routes one response. Anything but a well-formed response to an +// outstanding request is a protocol violation that ends the stream. +func (c *Client) deliver(f sandboxwire.Frame) error { + kind, err := tags.Classify(f.Type) + if err != nil { + return err + } + if kind != sandboxwire.KindResponse { + return malformed("message type %#04x from the server", f.Type) + } + op := Op(f.Type ^ sandboxwire.ResponseType(0)) + msg, fail, err := decodeResponse(op, f.Payload) + if err != nil { + return err + } + c.mu.Lock() + if cl, ok := c.pending[f.RequestID]; ok && cl.op == op { + delete(c.pending, f.RequestID) + c.mu.Unlock() + cl.ch <- outcome{msg: msg, fail: fail} + return nil + } + if abandoned, ok := c.abandoned[f.RequestID]; ok && abandoned == op { + delete(c.abandoned, f.RequestID) + c.mu.Unlock() + return nil + } + c.mu.Unlock() + return malformed("unexpected %s response to request %d", op, f.RequestID) +} + +func roundTrip[R message](ctx context.Context, c *Client, q Request) (R, error) { + var zero R + if err := ctx.Err(); err != nil { + return zero, contextFailure(err, sandboxwire.EffectNone) + } + payload, err := encodeRequest(q) + if err != nil { + f := NewFailure(CodeInvalidArgument, sandboxwire.EffectNone, err.Error()) + f.cause = err + return zero, f + } + id, cl, fail := c.send(ctx, q.Op(), payload) + if fail != nil { + return zero, fail + } + var out outcome + select { + case out = <-cl.ch: + case <-ctx.Done(): + if c.abandon(id, cl) { + effect := sandboxwire.EffectPossible + if sideEffectFree(q) { + effect = sandboxwire.EffectNone + } + return zero, contextFailure(ctx.Err(), effect) + } + out = <-cl.ch + } + if out.fail != nil { + return zero, out.fail + } + return out.msg.(R), nil +} + +func contextFailure(err error, effect sandboxwire.Effect) *Failure { + code := CodeCancelled + if errors.Is(err, context.DeadlineExceeded) { + code = CodeDeadlineExceeded + } + f := NewFailure(code, effect, err.Error()) + f.cause = err + return f +} + +func (c *Client) Describe(ctx context.Context, r *DescribeRequest) (*DescribeResponse, error) { + return roundTrip[*DescribeResponse](ctx, c, r) +} + +func (c *Client) Attach(ctx context.Context, r *AttachRequest) (*AttachResponse, error) { + return roundTrip[*AttachResponse](ctx, c, r) +} + +func (c *Client) Detach(ctx context.Context, r *DetachRequest) (*DetachResponse, error) { + return roundTrip[*DetachResponse](ctx, c, r) +} + +func (c *Client) Lookup(ctx context.Context, r *LookupRequest) (*LookupResponse, error) { + return roundTrip[*LookupResponse](ctx, c, r) +} + +func (c *Client) Walk(ctx context.Context, r *WalkRequest) (*WalkResponse, error) { + return roundTrip[*WalkResponse](ctx, c, r) +} + +func (c *Client) GetAttr(ctx context.Context, r *GetAttrRequest) (*GetAttrResponse, error) { + return roundTrip[*GetAttrResponse](ctx, c, r) +} + +func (c *Client) SetAttr(ctx context.Context, r *SetAttrRequest) (*SetAttrResponse, error) { + return roundTrip[*SetAttrResponse](ctx, c, r) +} + +func (c *Client) Access(ctx context.Context, r *AccessRequest) (*AccessResponse, error) { + return roundTrip[*AccessResponse](ctx, c, r) +} + +func (c *Client) Open(ctx context.Context, r *OpenRequest) (*OpenResponse, error) { + return roundTrip[*OpenResponse](ctx, c, r) +} + +func (c *Client) Create(ctx context.Context, r *CreateRequest) (*CreateResponse, error) { + return roundTrip[*CreateResponse](ctx, c, r) +} + +func (c *Client) Read(ctx context.Context, r *ReadRequest) (*ReadResponse, error) { + return roundTrip[*ReadResponse](ctx, c, r) +} + +func (c *Client) Write(ctx context.Context, r *WriteRequest) (*WriteResponse, error) { + return roundTrip[*WriteResponse](ctx, c, r) +} + +func (c *Client) Flush(ctx context.Context, r *FlushRequest) (*FlushResponse, error) { + return roundTrip[*FlushResponse](ctx, c, r) +} + +func (c *Client) Fsync(ctx context.Context, r *FsyncRequest) (*FsyncResponse, error) { + return roundTrip[*FsyncResponse](ctx, c, r) +} + +func (c *Client) Release(ctx context.Context, r *ReleaseRequest) (*ReleaseResponse, error) { + return roundTrip[*ReleaseResponse](ctx, c, r) +} + +func (c *Client) OpenDir(ctx context.Context, r *OpenDirRequest) (*OpenDirResponse, error) { + return roundTrip[*OpenDirResponse](ctx, c, r) +} + +func (c *Client) ReadDir(ctx context.Context, r *ReadDirRequest) (*ReadDirResponse, error) { + return roundTrip[*ReadDirResponse](ctx, c, r) +} + +func (c *Client) ReleaseDir(ctx context.Context, r *ReleaseDirRequest) (*ReleaseDirResponse, error) { + return roundTrip[*ReleaseDirResponse](ctx, c, r) +} + +func (c *Client) Mkdir(ctx context.Context, r *MkdirRequest) (*MkdirResponse, error) { + return roundTrip[*MkdirResponse](ctx, c, r) +} + +func (c *Client) Unlink(ctx context.Context, r *UnlinkRequest) (*UnlinkResponse, error) { + return roundTrip[*UnlinkResponse](ctx, c, r) +} + +func (c *Client) Rmdir(ctx context.Context, r *RmdirRequest) (*RmdirResponse, error) { + return roundTrip[*RmdirResponse](ctx, c, r) +} + +func (c *Client) Rename(ctx context.Context, r *RenameRequest) (*RenameResponse, error) { + return roundTrip[*RenameResponse](ctx, c, r) +} + +func (c *Client) Link(ctx context.Context, r *LinkRequest) (*LinkResponse, error) { + return roundTrip[*LinkResponse](ctx, c, r) +} + +func (c *Client) Symlink(ctx context.Context, r *SymlinkRequest) (*SymlinkResponse, error) { + return roundTrip[*SymlinkResponse](ctx, c, r) +} + +func (c *Client) Readlink(ctx context.Context, r *ReadlinkRequest) (*ReadlinkResponse, error) { + return roundTrip[*ReadlinkResponse](ctx, c, r) +} + +func (c *Client) StatFS(ctx context.Context, r *StatFSRequest) (*StatFSResponse, error) { + return roundTrip[*StatFSResponse](ctx, c, r) +} + +func (c *Client) Forget(ctx context.Context, r *ForgetRequest) (*ForgetResponse, error) { + return roundTrip[*ForgetResponse](ctx, c, r) +} + +func (c *Client) GetLock(ctx context.Context, r *GetLockRequest) (*GetLockResponse, error) { + return roundTrip[*GetLockResponse](ctx, c, r) +} + +func (c *Client) SetLock(ctx context.Context, r *SetLockRequest) (*SetLockResponse, error) { + return roundTrip[*SetLockResponse](ctx, c, r) +} diff --git a/internal/sandboxfs/client_test.go b/internal/sandboxfs/client_test.go new file mode 100644 index 00000000..ee4534a0 --- /dev/null +++ b/internal/sandboxfs/client_test.go @@ -0,0 +1,33 @@ +package sandboxfs + +import ( + "context" + "errors" + "net" + "testing" + + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxlink" + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxwire" +) + +// A cancellation that arrives after the frame is written cancels the request +// and leaves the stream open. +func TestCancelAfterWriteKeepsStream(t *testing.T) { + cc, sc := net.Pipe() + a := Attachment{ID: sandboxwire.NewID(), ServerInstanceID: testInstance, Lease: context.Background(), Exports: []sandboxlink.ExportGrant{{ID: "world"}}} + go Serve(context.Background(), sc, &describer{}, a) + c := NewClient(cc) + defer c.Close() + + ctx, cancel := context.WithCancel(context.Background()) + c.afterWrite = cancel + _, err := c.Describe(ctx, &DescribeRequest{}) + var f *Failure + if err != nil && !(errors.As(err, &f) && f.Code == CodeCancelled) { + t.Fatalf("cancelled describe: %v", err) + } + c.afterWrite = nil + if _, err := c.Describe(context.Background(), &DescribeRequest{}); err != nil || c.Err() != nil { + t.Fatalf("describe after the cancellation: %v; stream: %v", err, c.Err()) + } +} diff --git a/internal/sandboxfs/protocol.go b/internal/sandboxfs/protocol.go new file mode 100644 index 00000000..1dae79ab --- /dev/null +++ b/internal/sandboxfs/protocol.go @@ -0,0 +1,2212 @@ +// Package sandboxfs is the File access protocol between a Runtime and a sandbox +// file service. This file is its one authored definition: the request tags, +// message types, wire layouts, validators and the Service interface a file +// service implements. client.go and server.go carry the protocol over one +// stream. docs/file-access-protocol.md describes it. +package sandboxfs + +import ( + "context" + "errors" + "fmt" + "math" + + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxlink" + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxwire" +) + +// Version is the protocol version. Link matches it exactly when it opens a +// file stream. +const Version uint16 = 1 + +// Op is a request tag. Each response carries its request's tag with +// sandboxwire.ResponseType. The protocol has no events. +type Op uint16 + +const ( + OpDescribe Op = iota + 1 + OpAttach + OpDetach + OpLookup + OpWalk + OpGetAttr + OpSetAttr + OpAccess + OpOpen + OpCreate + OpRead + OpWrite + OpFlush + OpFsync + OpRelease + OpOpenDir + OpReadDir + OpReleaseDir + OpMkdir + OpUnlink + OpRmdir + OpRename + OpLink + OpSymlink + OpReadlink + OpStatFS + OpForget + OpGetLock + OpSetLock + OpCancelRequest +) + +var opNames = [...]string{ + OpDescribe: "Describe", OpAttach: "Attach", OpDetach: "Detach", + OpLookup: "Lookup", OpWalk: "Walk", OpGetAttr: "GetAttr", OpSetAttr: "SetAttr", OpAccess: "Access", + OpOpen: "Open", OpCreate: "Create", OpRead: "Read", OpWrite: "Write", OpFlush: "Flush", OpFsync: "Fsync", OpRelease: "Release", + OpOpenDir: "OpenDir", OpReadDir: "ReadDir", OpReleaseDir: "ReleaseDir", + OpMkdir: "Mkdir", OpUnlink: "Unlink", OpRmdir: "Rmdir", OpRename: "Rename", OpLink: "Link", OpSymlink: "Symlink", OpReadlink: "Readlink", + OpStatFS: "StatFS", OpForget: "Forget", OpGetLock: "GetLock", OpSetLock: "SetLock", OpCancelRequest: "CancelRequest", +} + +func (o Op) String() string { + if o >= 1 && int(o) < len(opNames) { + return opNames[o] + } + return fmt.Sprintf("Op(%d)", uint16(o)) +} + +var tags = sandboxwire.Tags{Requests: uint16(OpCancelRequest)} + +// Hard limits. Capabilities may advertise smaller ones; nothing on the wire +// exceeds these. +const ( + maxNameBytes = 1024 // an entry name, the FUSE name limit + maxTargetBytes = 4096 // a symlink target + maxWalkNames = 1024 + maxReadDirBytes = 256 << 10 + maxForgetEntries = 4096 + maxMessageBytes = 1024 + maxOffset = math.MaxInt64 +) + +// Attachment is the authenticated context Link establishes for a stream. File +// requests never carry it: the server passes it to every Service call. +type Attachment struct { + ID sandboxwire.ID + // ServerInstanceID is the service incarnation the stream was opened + // against. A service answers InstanceChanged when it differs from its own. + ServerInstanceID sandboxwire.ID + Lease Lease + // Exports are the exports the Link binding grants the attachment: at + // least one, each ID once. Attach can select only these. + Exports []sandboxlink.ExportGrant +} + +// Grant returns the attachment's grant of export id. +func (a *Attachment) Grant(id sandboxlink.ExportID) (sandboxlink.ExportGrant, bool) { + for _, g := range a.Exports { + if g.ID == id { + return g, true + } + } + return sandboxlink.ExportGrant{}, false +} + +func (a *Attachment) validate() error { + if a.ID.IsZero() || a.ServerInstanceID.IsZero() || a.Lease == nil { + return errors.New("sandboxfs: attachment without an ID, server instance or lease") + } + if len(a.Exports) == 0 { + return errors.New("sandboxfs: attachment grants no export") + } + seen := make(map[sandboxlink.ExportID]bool, len(a.Exports)) + for _, g := range a.Exports { + if !g.ID.Valid() || seen[g.ID] { + return fmt.Errorf("sandboxfs: attachment grants export %q", string(g.ID)) + } + seen[g.ID] = true + } + return nil +} + +// Lease is an attachment's authority. Done closes when the lease lapses or is +// revoked and never reopens. Losing a transport does not end a lease, so +// every stream of one attachment shares one Lease. A context.Context +// satisfies it. +type Lease interface { + Done() <-chan struct{} +} + +// Service is implemented by a file service, one method per operation. +// CancelRequest is not a method: the server answers it by cancelling the +// target's context. A method returns a *Failure for a typed failure; the server +// reports any other error as Unknown with EffectPossible. +type Service interface { + Describe(context.Context, Attachment, *DescribeRequest) (*DescribeResponse, error) + Attach(context.Context, Attachment, *AttachRequest) (*AttachResponse, error) + Detach(context.Context, Attachment, *DetachRequest) (*DetachResponse, error) + Lookup(context.Context, Attachment, *LookupRequest) (*LookupResponse, error) + Walk(context.Context, Attachment, *WalkRequest) (*WalkResponse, error) + GetAttr(context.Context, Attachment, *GetAttrRequest) (*GetAttrResponse, error) + SetAttr(context.Context, Attachment, *SetAttrRequest) (*SetAttrResponse, error) + Access(context.Context, Attachment, *AccessRequest) (*AccessResponse, error) + Open(context.Context, Attachment, *OpenRequest) (*OpenResponse, error) + Create(context.Context, Attachment, *CreateRequest) (*CreateResponse, error) + Read(context.Context, Attachment, *ReadRequest) (*ReadResponse, error) + Write(context.Context, Attachment, *WriteRequest) (*WriteResponse, error) + Flush(context.Context, Attachment, *FlushRequest) (*FlushResponse, error) + Fsync(context.Context, Attachment, *FsyncRequest) (*FsyncResponse, error) + Release(context.Context, Attachment, *ReleaseRequest) (*ReleaseResponse, error) + OpenDir(context.Context, Attachment, *OpenDirRequest) (*OpenDirResponse, error) + ReadDir(context.Context, Attachment, *ReadDirRequest) (*ReadDirResponse, error) + ReleaseDir(context.Context, Attachment, *ReleaseDirRequest) (*ReleaseDirResponse, error) + Mkdir(context.Context, Attachment, *MkdirRequest) (*MkdirResponse, error) + Unlink(context.Context, Attachment, *UnlinkRequest) (*UnlinkResponse, error) + Rmdir(context.Context, Attachment, *RmdirRequest) (*RmdirResponse, error) + Rename(context.Context, Attachment, *RenameRequest) (*RenameResponse, error) + Link(context.Context, Attachment, *LinkRequest) (*LinkResponse, error) + Symlink(context.Context, Attachment, *SymlinkRequest) (*SymlinkResponse, error) + Readlink(context.Context, Attachment, *ReadlinkRequest) (*ReadlinkResponse, error) + StatFS(context.Context, Attachment, *StatFSRequest) (*StatFSResponse, error) + Forget(context.Context, Attachment, *ForgetRequest) (*ForgetResponse, error) + GetLock(context.Context, Attachment, *GetLockRequest) (*GetLockResponse, error) + SetLock(context.Context, Attachment, *SetLockRequest) (*SetLockResponse, error) +} + +// Enumerations. Each is a uint16 on the wire, and zero is never valid. + +// Result is the discriminator every response payload begins with. +type Result uint16 + +const ( + ResultSuccess Result = 1 + ResultFailure Result = 2 +) + +type ErrorCode uint16 + +const ( + CodeInvalidArgument ErrorCode = iota + 1 + CodeUnsupported + CodeUnauthorized + CodeStaleAttachment + CodeInstanceChanged + CodeStaleNode + CodeStaleHandle + CodeResourceExhausted + CodeCancelled + CodeDeadlineExceeded + CodeErrno + CodeUnknown +) + +var codeNames = [...]string{ + CodeInvalidArgument: "InvalidArgument", CodeUnsupported: "Unsupported", CodeUnauthorized: "Unauthorized", + CodeStaleAttachment: "StaleAttachment", CodeInstanceChanged: "InstanceChanged", CodeStaleNode: "StaleNode", + CodeStaleHandle: "StaleHandle", CodeResourceExhausted: "ResourceExhausted", CodeCancelled: "Cancelled", + CodeDeadlineExceeded: "DeadlineExceeded", CodeErrno: "Errno", CodeUnknown: "Unknown", +} + +func (c ErrorCode) Valid() bool { return c >= 1 && int(c) < len(codeNames) } +func (c ErrorCode) String() string { return enumName(codeNames[:], uint16(c), "ErrorCode") } + +// Errno is the semantic error of a failed file-system call. Each adapter +// converts its native errors; an unknown native error is ErrnoIO. +type Errno uint16 + +const ( + ErrnoPermissionDenied Errno = iota + 1 + ErrnoOperationNotPermitted + ErrnoNotFound + ErrnoExists + ErrnoNotDirectory + ErrnoIsDirectory + ErrnoDirectoryNotEmpty + ErrnoInvalidArgument + ErrnoBadDescriptor + ErrnoTooManyOpenFiles + ErrnoNoSpace + ErrnoQuotaExceeded + ErrnoReadOnlyFilesystem + ErrnoCrossDevice + ErrnoNameTooLong + ErrnoSymlinkLoop + ErrnoFileTooLarge + ErrnoOverflow + ErrnoBusy + ErrnoAgain + ErrnoInterrupted + ErrnoIO + ErrnoNoDevice + ErrnoNoSuchDeviceOrAddress + ErrnoBrokenPipe + ErrnoNotSupported + ErrnoNoLocks + ErrnoDeadlock +) + +var errnoNames = [...]string{ + ErrnoPermissionDenied: "PermissionDenied", ErrnoOperationNotPermitted: "OperationNotPermitted", + ErrnoNotFound: "NotFound", ErrnoExists: "Exists", ErrnoNotDirectory: "NotDirectory", + ErrnoIsDirectory: "IsDirectory", ErrnoDirectoryNotEmpty: "DirectoryNotEmpty", + ErrnoInvalidArgument: "InvalidArgument", ErrnoBadDescriptor: "BadDescriptor", + ErrnoTooManyOpenFiles: "TooManyOpenFiles", ErrnoNoSpace: "NoSpace", ErrnoQuotaExceeded: "QuotaExceeded", + ErrnoReadOnlyFilesystem: "ReadOnlyFilesystem", ErrnoCrossDevice: "CrossDevice", + ErrnoNameTooLong: "NameTooLong", ErrnoSymlinkLoop: "SymlinkLoop", ErrnoFileTooLarge: "FileTooLarge", + ErrnoOverflow: "Overflow", ErrnoBusy: "Busy", ErrnoAgain: "Again", ErrnoInterrupted: "Interrupted", + ErrnoIO: "IO", ErrnoNoDevice: "NoDevice", ErrnoNoSuchDeviceOrAddress: "NoSuchDeviceOrAddress", + ErrnoBrokenPipe: "BrokenPipe", ErrnoNotSupported: "NotSupported", ErrnoNoLocks: "NoLocks", + ErrnoDeadlock: "Deadlock", +} + +func (e Errno) Valid() bool { return e >= 1 && int(e) < len(errnoNames) } +func (e Errno) String() string { return enumName(errnoNames[:], uint16(e), "Errno") } + +func enumName(names []string, v uint16, kind string) string { + if v >= 1 && int(v) < len(names) { + return names[v] + } + return fmt.Sprintf("%s(%d)", kind, v) +} + +// TargetKind selects what GetAttr and SetAttr address. +type TargetKind uint16 + +const ( + TargetNode TargetKind = 1 + TargetHandle TargetKind = 2 +) + +type AccessMode uint16 + +const ( + AccessRead AccessMode = 1 + AccessWrite AccessMode = 2 + AccessReadWrite AccessMode = 3 +) + +// Writes reports whether the mode opens for writing. +func (m AccessMode) Writes() bool { return m == AccessWrite || m == AccessReadWrite } + +type RenameMode uint16 + +const ( + RenameReplace RenameMode = 1 + RenameNoReplace RenameMode = 2 + RenameExchange RenameMode = 3 +) + +type LockKind uint16 + +const ( + LockPOSIX LockKind = 1 + LockFlock LockKind = 2 +) + +type LockMode uint16 + +const ( + LockRead LockMode = 1 + LockWrite LockMode = 2 + LockUnlock LockMode = 3 +) + +type PathProfile uint16 + +// PathProfileLinuxBytes: names and symlink targets are Linux byte strings. +const PathProfileLinuxBytes PathProfile = 1 + +type CacheProfile uint16 + +// CacheProfileUncached: zero entry, attribute and negative TTLs, direct I/O, +// no writeback cache and no retained directory listing on the client. +const CacheProfileUncached CacheProfile = 1 + +type Durability uint16 + +// DurabilityFsyncRequired: data is durable only after Fsync succeeds. +const DurabilityFsyncRequired Durability = 1 + +// Flag sets. Each is a uint32 on the wire; unknown bits are rejected. + +type OpenFlags uint32 + +const ( + OpenAppend OpenFlags = 1 << iota + OpenTruncate + OpenNoFollow + OpenSync + OpenDataSync + + openFlagsAll = OpenAppend | OpenTruncate | OpenNoFollow | OpenSync | OpenDataSync +) + +// AttrMask selects the attributes SetAttr changes. AttrAtimeNow and +// AttrMtimeNow set the time from the service's clock and exclude AttrAtime +// and AttrMtime respectively. +type AttrMask uint32 + +const ( + AttrSize AttrMask = 1 << iota + AttrMode + AttrUID + AttrGID + AttrAtime + AttrMtime + AttrAtimeNow + AttrMtimeNow + + attrMaskAll = AttrMtimeNow<<1 - 1 +) + +// AccessMask selects the permissions Access checks. Zero checks existence. +type AccessMask uint32 + +const ( + MayExecute AccessMask = 1 << iota + MayWrite + MayRead + + accessMaskAll = MayExecute | MayWrite | MayRead +) + +// Mode bits of Attr.Mode and DirEntry.Type, in the Linux st_mode layout. +const ( + ModeType uint32 = 0o170000 + ModeSocket uint32 = 0o140000 + ModeSymlink uint32 = 0o120000 + ModeRegular uint32 = 0o100000 + ModeBlockDevice uint32 = 0o060000 + ModeDirectory uint32 = 0o040000 + ModeCharDevice uint32 = 0o020000 + ModeFIFO uint32 = 0o010000 + ModePerm uint32 = 0o7777 +) + +func validFileType(t uint32) bool { + switch t { + case ModeSocket, ModeSymlink, ModeRegular, ModeBlockDevice, ModeDirectory, ModeCharDevice, ModeFIFO: + return true + } + return false +} + +// NodeRef names a file-system object the attachment holds lookup references +// on. An ID is reused only after its object is forgotten, and Generation then +// differs, so a stale NodeRef never names another object. Both are nonzero. +type NodeRef struct { + ID uint64 + Generation uint64 +} + +// HandleID names an open file or directory handle. It is opaque and nonzero. +type HandleID uint64 + +// LockOwner identifies a lock owner within an attachment. +type LockOwner uint64 + +type Timestamp struct { + Sec int64 + Nsec uint32 // below one billion +} + +// Attr is a file's attributes. Ino identifies the file within its export. +type Attr struct { + Ino uint64 + Mode uint32 // file type and permission bits + Nlink uint32 + UID uint32 + GID uint32 + Rdev uint64 + Size uint64 + Blocks uint64 // 512-byte blocks + Blksize uint32 + Atime Timestamp + Mtime Timestamp + Ctime Timestamp +} + +// Entry is a node the response acquired one lookup reference on, with its +// attributes. +type Entry struct { + Node NodeRef + Attr Attr +} + +// Identity is the uid and gid the service acts as. Files it creates carry +// them. It is informational: requests never select an identity. +type Identity struct { + UID uint32 + GID uint32 +} + +// Capabilities declares what a service supports. Every field is encoded. +type Capabilities struct { + PathProfile PathProfile + CacheProfile CacheProfile + Durability Durability + MaxNameBytes uint32 + MaxPathBytes uint32 // longest symlink target + MaxReadBytes uint32 + MaxWriteBytes uint32 + MaxWalkComponents uint32 + MaxReadDirBytes uint32 + MaxOpenHandles uint32 // per attachment + ReadOnly bool + AtomicAppend bool + AtomicRename bool + RenameNoReplace bool + RenameExchange bool + HardLinks bool + Symlinks bool + SetMode bool + SetOwner bool + SetTimes bool + DirectoryFsync bool + ReadDirPlus bool + Flock bool + POSIXLocks bool +} + +// Target is the node or open handle GetAttr and SetAttr address. Only the +// field Kind selects is set. +type Target struct { + Kind TargetKind + Node NodeRef + Handle HandleID +} + +// DirEntry is one directory entry. Cookie is the position after it: ReadDir +// resumes there. Entry is set when ReadDir asked WithAttrs, and then carries +// one lookup reference. +type DirEntry struct { + Name []byte + Ino uint64 + Type uint32 // one ModeType value + Cookie uint64 + Entry *Entry +} + +const ( + attrWireSize = 8 + 4 + 4 + 4 + 4 + 8 + 8 + 8 + 4 + 3*(8+4) + entryWireSize = 16 + attrWireSize + dirEntryWireSize = 4 + 8 + 4 + 8 + 1 +) + +// WireSize is the entry's encoded length. ReadDir's Limit bounds the sum of +// the returned entries' sizes. +func (e *DirEntry) WireSize() int { + n := dirEntryWireSize + len(e.Name) + if e.Entry != nil { + n += entryWireSize + } + return n +} + +type ForgetEntry struct { + Node NodeRef + Count uint64 +} + +// Lock is a byte range lock: Start through End inclusive, where End +// math.MaxInt64 extends to the end of the file. +type Lock struct { + Mode LockMode + Start uint64 + End uint64 +} + +// Failure is a typed failure. Errno is set exactly when Code is CodeErrno. +// Effect says whether the failed request may have taken effect. +type Failure struct { + Code ErrorCode + Errno Errno + Effect sandboxwire.Effect + Message string + + cause error // set on failures the client produces; never encoded +} + +// NewFailure returns a failure with code, effect and a message cut to the +// protocol limit. +func NewFailure(code ErrorCode, effect sandboxwire.Effect, message string) *Failure { + return &Failure{Code: code, Effect: effect, Message: clip(message)} +} + +// NewErrnoFailure returns a CodeErrno failure. +func NewErrnoFailure(errno Errno, effect sandboxwire.Effect, message string) *Failure { + return &Failure{Code: CodeErrno, Errno: errno, Effect: effect, Message: clip(message)} +} + +func clip(s string) string { + if len(s) > maxMessageBytes { + return s[:maxMessageBytes] + } + return s +} + +func (f *Failure) Error() string { + s := "sandboxfs: " + f.Code.String() + if f.Code == CodeErrno { + s += " " + f.Errno.String() + } + if f.Message != "" { + s += ": " + f.Message + } + if f.Effect == sandboxwire.EffectPossible { + s += " (effect possible)" + } + return s +} + +func (f *Failure) Unwrap() error { return f.cause } + +// Requests and responses, in tag order. + +type DescribeRequest struct{} + +type DescribeResponse struct { + ServerInstanceID sandboxwire.ID + Identity Identity + Capabilities Capabilities + Exports []sandboxlink.ExportID +} + +// AttachRequest selects a declared export that the attachment's Link grant +// includes. A read-only grant requires ReadOnly. An attachment attaches once +// until it detaches. +type AttachRequest struct { + Export sandboxlink.ExportID + ReadOnly bool +} + +type AttachResponse struct { + Root Entry +} + +// DetachRequest releases every node, handle and lock of the attachment. +type DetachRequest struct{} + +type DetachResponse struct{} + +type LookupRequest struct { + Parent NodeRef + Name []byte +} + +type LookupResponse struct { + Entry Entry +} + +// WalkRequest looks up Names in turn from Parent. The walk stops after a +// symlink, so fewer entries than names with no Failure means the last entry +// is a symlink. +type WalkRequest struct { + Parent NodeRef + Names [][]byte +} + +// WalkResponse holds the entries walked, at least one. Failure is the error +// that stopped the walk after them. +type WalkResponse struct { + Entries []Entry + Failure *Failure +} + +type GetAttrRequest struct { + Target Target +} + +type GetAttrResponse struct { + Attr Attr +} + +// SetAttrRequest changes the attributes Set selects; the other fields are +// zero. On the wire each selected value follows the mask in field order. +type SetAttrRequest struct { + Target Target + Set AttrMask + Size uint64 + Mode uint32 // permission bits + UID uint32 + GID uint32 + Atime Timestamp + Mtime Timestamp +} + +type SetAttrResponse struct { + Attr Attr +} + +type AccessRequest struct { + Node NodeRef + Mask AccessMask +} + +type AccessResponse struct{} + +type OpenRequest struct { + Node NodeRef + Access AccessMode + Flags OpenFlags +} + +type OpenResponse struct { + Handle HandleID +} + +type CreateRequest struct { + Parent NodeRef + Name []byte + Mode uint32 // permission bits, applied as given + Access AccessMode + Flags OpenFlags + Exclusive bool +} + +type CreateResponse struct { + Entry Entry + Handle HandleID +} + +type ReadRequest struct { + Handle HandleID + Offset uint64 + Size uint32 +} + +// ReadResponse holds the bytes read; fewer than asked means end of file. +type ReadResponse struct { + Data []byte +} + +// WriteRequest writes Data at Offset. An append-opened handle appends +// atomically and ignores Offset. +type WriteRequest struct { + Handle HandleID + Offset uint64 + Data []byte +} + +// WriteResponse reports the bytes written. Failure is the error that stopped +// the write after that nonzero prefix. +type WriteResponse struct { + Written uint32 + Failure *Failure +} + +// FlushRequest runs on each close of a descriptor for the handle. Owner is the +// closing lock owner, whose POSIX locks on the file it releases. +type FlushRequest struct { + Handle HandleID + Owner LockOwner +} + +type FlushResponse struct{} + +type FsyncRequest struct { + Handle HandleID + DataOnly bool +} + +type FsyncResponse struct{} + +type ReleaseRequest struct { + Handle HandleID +} + +type ReleaseResponse struct{} + +type OpenDirRequest struct { + Node NodeRef +} + +type OpenDirResponse struct { + Handle HandleID +} + +// ReadDirRequest reads entries after Cookie; cookie zero is the start. Limit +// bounds the entries' WireSize sum. "." and ".." are never returned. +type ReadDirRequest struct { + Handle HandleID + Cookie uint64 + Limit uint32 + WithAttrs bool +} + +type ReadDirResponse struct { + Entries []DirEntry + End bool +} + +type ReleaseDirRequest struct { + Handle HandleID +} + +type ReleaseDirResponse struct{} + +type MkdirRequest struct { + Parent NodeRef + Name []byte + Mode uint32 // permission bits, applied as given +} + +type MkdirResponse struct { + Entry Entry +} + +type UnlinkRequest struct { + Parent NodeRef + Name []byte +} + +type UnlinkResponse struct{} + +type RmdirRequest struct { + Parent NodeRef + Name []byte +} + +type RmdirResponse struct{} + +type RenameRequest struct { + Parent NodeRef + Name []byte + NewParent NodeRef + NewName []byte + Mode RenameMode +} + +type RenameResponse struct{} + +type LinkRequest struct { + Node NodeRef + NewParent NodeRef + NewName []byte +} + +type LinkResponse struct { + Entry Entry +} + +type SymlinkRequest struct { + Parent NodeRef + Name []byte + Target []byte +} + +type SymlinkResponse struct { + Entry Entry +} + +type ReadlinkRequest struct { + Node NodeRef +} + +type ReadlinkResponse struct { + Target []byte +} + +type StatFSRequest struct { + Node NodeRef +} + +type StatFSResponse struct { + Blocks uint64 + BlocksFree uint64 + BlocksAvailable uint64 + Files uint64 + FilesFree uint64 + BlockSize uint32 + FragmentSize uint32 + NameMax uint32 +} + +// ForgetRequest releases lookup references. It applies entirely or not at +// all. +type ForgetRequest struct { + Entries []ForgetEntry +} + +type ForgetResponse struct{} + +// GetLockRequest asks for a POSIX lock that would conflict with Lock. +type GetLockRequest struct { + Handle HandleID + Owner LockOwner + Lock Lock +} + +// GetLockResponse holds the conflicting lock, if any. +type GetLockResponse struct { + Conflict *Lock +} + +// SetLockRequest acquires or releases a lock. A LockFlock lock covers the +// whole file. Wait blocks until the lock is available or the request is +// cancelled. +type SetLockRequest struct { + Handle HandleID + Kind LockKind + Owner LockOwner + Lock Lock + Wait bool +} + +type SetLockResponse struct{} + +// CancelRequestRequest asks to cancel the outstanding request Target. Its +// acknowledgement says nothing about the target's outcome. +type CancelRequestRequest struct { + Target uint64 +} + +type CancelRequestResponse struct{} + +func (*DescribeRequest) Op() Op { return OpDescribe } +func (*AttachRequest) Op() Op { return OpAttach } +func (*DetachRequest) Op() Op { return OpDetach } +func (*LookupRequest) Op() Op { return OpLookup } +func (*WalkRequest) Op() Op { return OpWalk } +func (*GetAttrRequest) Op() Op { return OpGetAttr } +func (*SetAttrRequest) Op() Op { return OpSetAttr } +func (*AccessRequest) Op() Op { return OpAccess } +func (*OpenRequest) Op() Op { return OpOpen } +func (*CreateRequest) Op() Op { return OpCreate } +func (*ReadRequest) Op() Op { return OpRead } +func (*WriteRequest) Op() Op { return OpWrite } +func (*FlushRequest) Op() Op { return OpFlush } +func (*FsyncRequest) Op() Op { return OpFsync } +func (*ReleaseRequest) Op() Op { return OpRelease } +func (*OpenDirRequest) Op() Op { return OpOpenDir } +func (*ReadDirRequest) Op() Op { return OpReadDir } +func (*ReleaseDirRequest) Op() Op { return OpReleaseDir } +func (*MkdirRequest) Op() Op { return OpMkdir } +func (*UnlinkRequest) Op() Op { return OpUnlink } +func (*RmdirRequest) Op() Op { return OpRmdir } +func (*RenameRequest) Op() Op { return OpRename } +func (*LinkRequest) Op() Op { return OpLink } +func (*SymlinkRequest) Op() Op { return OpSymlink } +func (*ReadlinkRequest) Op() Op { return OpReadlink } +func (*StatFSRequest) Op() Op { return OpStatFS } +func (*ForgetRequest) Op() Op { return OpForget } +func (*GetLockRequest) Op() Op { return OpGetLock } +func (*SetLockRequest) Op() Op { return OpSetLock } +func (*CancelRequestRequest) Op() Op { return OpCancelRequest } + +// Request is implemented by every request type. +type Request interface { + message + Op() Op +} + +type message interface { + encode(*sandboxwire.Encoder) + decode(*decoder) + validate() error +} + +// opSpec binds an operation's request, response and Service method. +type opSpec struct { + newRequest func() Request + newResponse func() message + serve func(context.Context, Service, Attachment, Request) (message, error) +} + +var errNilResponse = errors.New("service returned no response") + +func bind[QT, RT any, Q interface { + *QT + Request +}, R interface { + *RT + message +}](method func(Service, context.Context, Attachment, Q) (R, error)) opSpec { + return opSpec{ + newRequest: func() Request { return Q(new(QT)) }, + newResponse: func() message { return R(new(RT)) }, + serve: func(ctx context.Context, s Service, a Attachment, q Request) (message, error) { + r, err := method(s, ctx, a, q.(Q)) + if err != nil { + return nil, err + } + if r == nil { + return nil, errNilResponse + } + return r, nil + }, + } +} + +// opSpecs is indexed by Op; each request type's Op method places its entry. +var opSpecs = func() (t [OpCancelRequest + 1]opSpec) { + for _, s := range []opSpec{ + bind(Service.Describe), bind(Service.Attach), bind(Service.Detach), + bind(Service.Lookup), bind(Service.Walk), bind(Service.GetAttr), bind(Service.SetAttr), bind(Service.Access), + bind(Service.Open), bind(Service.Create), bind(Service.Read), bind(Service.Write), + bind(Service.Flush), bind(Service.Fsync), bind(Service.Release), + bind(Service.OpenDir), bind(Service.ReadDir), bind(Service.ReleaseDir), + bind(Service.Mkdir), bind(Service.Unlink), bind(Service.Rmdir), bind(Service.Rename), + bind(Service.Link), bind(Service.Symlink), bind(Service.Readlink), + bind(Service.StatFS), bind(Service.Forget), bind(Service.GetLock), bind(Service.SetLock), + { + newRequest: func() Request { return new(CancelRequestRequest) }, + newResponse: func() message { return new(CancelRequestResponse) }, + }, + } { + t[s.newRequest().Op()] = s + } + return t +}() + +// sideEffectFree reports whether r leaves no state behind, so abandoning it +// cannot have had an effect. Lookups and ReadDir WithAttrs acquire references. +func sideEffectFree(r Request) bool { + switch r := r.(type) { + case *DescribeRequest, *GetAttrRequest, *AccessRequest, *ReadRequest, *ReadlinkRequest, *StatFSRequest, *GetLockRequest: + return true + case *ReadDirRequest: + return !r.WithAttrs + } + return false +} + +// modifiesFiles reports whether r changes the file system, which a read-only +// attachment refuses. +func modifiesFiles(r Request) bool { + switch r := r.(type) { + case *SetAttrRequest, *CreateRequest, *WriteRequest, *MkdirRequest, *UnlinkRequest, *RmdirRequest, + *RenameRequest, *LinkRequest, *SymlinkRequest: + return true + case *OpenRequest: + return r.Access.Writes() || r.Flags&OpenTruncate != 0 + } + return false +} + +// Admit checks r against the declared capabilities and the attachment's +// read-only choice. A service calls it before running a request and returns +// the failure it reports. Limits a request can only be judged against service +// state, such as MaxOpenHandles, stay with the service. +func (c *Capabilities) Admit(r Request, readOnly bool) *Failure { + unsupported := func(what string) *Failure { + return NewFailure(CodeUnsupported, sandboxwire.EffectNone, what+" is not supported") + } + tooLong := func(n []byte, max uint32) bool { return uint64(len(n)) > uint64(max) } + nameTooLong := NewErrnoFailure(ErrnoNameTooLong, sandboxwire.EffectNone, "name exceeds MaxNameBytes") + if readOnly && modifiesFiles(r) { + return NewErrnoFailure(ErrnoReadOnlyFilesystem, sandboxwire.EffectNone, "attachment is read-only") + } + switch r := r.(type) { + case *AttachRequest: + if !r.ReadOnly && c.ReadOnly { + return unsupported("a writable attachment") + } + case *LookupRequest: + if tooLong(r.Name, c.MaxNameBytes) { + return nameTooLong + } + case *WalkRequest: + if uint64(len(r.Names)) > uint64(c.MaxWalkComponents) { + return NewFailure(CodeInvalidArgument, sandboxwire.EffectNone, "walk exceeds MaxWalkComponents") + } + for _, n := range r.Names { + if tooLong(n, c.MaxNameBytes) { + return nameTooLong + } + } + case *SetAttrRequest: + switch { + case r.Set&AttrMode != 0 && !c.SetMode: + return unsupported("setting the mode") + case r.Set&(AttrUID|AttrGID) != 0 && !c.SetOwner: + return unsupported("setting the owner") + case r.Set&(AttrAtime|AttrMtime|AttrAtimeNow|AttrMtimeNow) != 0 && !c.SetTimes: + return unsupported("setting times") + } + case *CreateRequest: + if tooLong(r.Name, c.MaxNameBytes) { + return nameTooLong + } + case *ReadRequest: + if r.Size > c.MaxReadBytes { + return NewFailure(CodeInvalidArgument, sandboxwire.EffectNone, "read exceeds MaxReadBytes") + } + case *WriteRequest: + if tooLong(r.Data, c.MaxWriteBytes) { + return NewFailure(CodeInvalidArgument, sandboxwire.EffectNone, "write exceeds MaxWriteBytes") + } + case *ReadDirRequest: + if r.Limit > c.MaxReadDirBytes { + return NewFailure(CodeInvalidArgument, sandboxwire.EffectNone, "limit exceeds MaxReadDirBytes") + } + if r.WithAttrs && !c.ReadDirPlus { + return unsupported("ReadDir WithAttrs") + } + case *MkdirRequest: + if tooLong(r.Name, c.MaxNameBytes) { + return nameTooLong + } + case *UnlinkRequest: + if tooLong(r.Name, c.MaxNameBytes) { + return nameTooLong + } + case *RmdirRequest: + if tooLong(r.Name, c.MaxNameBytes) { + return nameTooLong + } + case *RenameRequest: + switch { + case tooLong(r.Name, c.MaxNameBytes) || tooLong(r.NewName, c.MaxNameBytes): + return nameTooLong + case r.Mode == RenameNoReplace && !c.RenameNoReplace: + return unsupported("RenameNoReplace") + case r.Mode == RenameExchange && !c.RenameExchange: + return unsupported("RenameExchange") + } + case *LinkRequest: + if !c.HardLinks { + return unsupported("Link") + } + if tooLong(r.NewName, c.MaxNameBytes) { + return nameTooLong + } + case *SymlinkRequest: + switch { + case !c.Symlinks: + return unsupported("Symlink") + case tooLong(r.Name, c.MaxNameBytes): + return nameTooLong + case tooLong(r.Target, c.MaxPathBytes): + return NewErrnoFailure(ErrnoNameTooLong, sandboxwire.EffectNone, "target exceeds MaxPathBytes") + } + case *GetLockRequest: + if !c.POSIXLocks { + return unsupported("POSIX locks") + } + case *SetLockRequest: + if r.Kind == LockPOSIX && !c.POSIXLocks { + return unsupported("POSIX locks") + } + if r.Kind == LockFlock && !c.Flock { + return unsupported("flock") + } + } + return nil +} + +// Codec. Every message has an ordered layout of sandboxwire primitives. +// Decoding is structural; validate then applies the protocol rules, and +// encoding validates first, so both directions enforce them. + +func malformed(format string, args ...any) error { + return fmt.Errorf("%w: "+format, append([]any{sandboxwire.ErrMalformed}, args...)...) +} + +// decoder keeps the first error and reads nothing after it. +type decoder struct { + d *sandboxwire.Decoder + err error +} + +func read[T any](d *decoder, f func() (T, error)) T { + var v T + if d.err == nil { + v, d.err = f() + } + return v +} + +func (d *decoder) u16() uint16 { return read(d, d.d.U16) } +func (d *decoder) u32() uint32 { return read(d, d.d.U32) } +func (d *decoder) u64() uint64 { return read(d, d.d.U64) } +func (d *decoder) i64() int64 { return read(d, d.d.I64) } +func (d *decoder) bool() bool { return read(d, d.d.Bool) } +func (d *decoder) bytes() []byte { return read(d, d.d.Bytes) } +func (d *decoder) present() bool { return read(d, d.d.Present) } +func (d *decoder) id() sandboxwire.ID { return read(d, d.d.ID) } + +func (d *decoder) count(max uint32) int { + return read(d, func() (int, error) { return d.d.Count(max) }) +} + +func decodeMessage(m message, payload []byte) error { + d := &decoder{d: sandboxwire.NewDecoder(payload)} + m.decode(d) + if d.err != nil { + return d.err + } + if err := d.d.Finish(); err != nil { + return err + } + return m.validate() +} + +func encodeMessage(e *sandboxwire.Encoder, m message) error { + if err := m.validate(); err != nil { + return err + } + m.encode(e) + return nil +} + +func checkPayload(p []byte) ([]byte, error) { + if len(p) > sandboxwire.MaxPayload { + return nil, malformed("payload of %d bytes exceeds the frame limit", len(p)) + } + return p, nil +} + +func encodeRequest(r Request) ([]byte, error) { + var e sandboxwire.Encoder + if err := encodeMessage(&e, r); err != nil { + return nil, err + } + return checkPayload(e.Payload()) +} + +func decodeRequest(op Op, payload []byte) (Request, error) { + if op < 1 || op > OpCancelRequest { + return nil, malformed("unknown request %d", uint16(op)) + } + r := opSpecs[op].newRequest() + return r, decodeMessage(r, payload) +} + +// encodeResponse encodes a success carrying r, or the failure f when it is +// set. +func encodeResponse(r message, f *Failure) ([]byte, error) { + var e sandboxwire.Encoder + if f != nil { + if err := f.validate(); err != nil { + return nil, err + } + e.Enum(uint16(ResultFailure)) + f.encode(&e) + } else { + if err := r.validate(); err != nil { + return nil, err + } + e.Enum(uint16(ResultSuccess)) + r.encode(&e) + } + return checkPayload(e.Payload()) +} + +func decodeResponse(op Op, payload []byte) (message, *Failure, error) { + if op < 1 || op > OpCancelRequest { + return nil, nil, malformed("unknown request %d", uint16(op)) + } + d := &decoder{d: sandboxwire.NewDecoder(payload)} + switch result := Result(d.u16()); { + case d.err != nil: + return nil, nil, d.err + case result == ResultSuccess: + r := opSpecs[op].newResponse() + return r, nil, decodeRest(d, r) + case result == ResultFailure: + f := new(Failure) + return nil, f, decodeRest(d, f) + default: + return nil, nil, malformed("result %d", result) + } +} + +func decodeRest(d *decoder, m message) error { + m.decode(d) + if d.err != nil { + return d.err + } + if err := d.d.Finish(); err != nil { + return err + } + return m.validate() +} + +// Shared values. + +func (r NodeRef) encode(e *sandboxwire.Encoder) { e.U64(r.ID); e.U64(r.Generation) } +func (r *NodeRef) decode(d *decoder) { r.ID = d.u64(); r.Generation = d.u64() } +func (r NodeRef) validate() error { + if r.ID == 0 || r.Generation == 0 { + return malformed("node reference %d/%d", r.ID, r.Generation) + } + return nil +} + +func (h HandleID) validate() error { + if h == 0 { + return malformed("zero handle") + } + return nil +} + +func (t Timestamp) encode(e *sandboxwire.Encoder) { e.I64(t.Sec); e.U32(t.Nsec) } +func (t *Timestamp) decode(d *decoder) { t.Sec = d.i64(); t.Nsec = d.u32() } +func (t Timestamp) validate() error { + if t.Nsec >= 1e9 { + return malformed("nanoseconds %d", t.Nsec) + } + return nil +} + +func (a *Attr) encode(e *sandboxwire.Encoder) { + e.U64(a.Ino) + e.U32(a.Mode) + e.U32(a.Nlink) + e.U32(a.UID) + e.U32(a.GID) + e.U64(a.Rdev) + e.U64(a.Size) + e.U64(a.Blocks) + e.U32(a.Blksize) + a.Atime.encode(e) + a.Mtime.encode(e) + a.Ctime.encode(e) +} + +func (a *Attr) decode(d *decoder) { + a.Ino = d.u64() + a.Mode = d.u32() + a.Nlink = d.u32() + a.UID = d.u32() + a.GID = d.u32() + a.Rdev = d.u64() + a.Size = d.u64() + a.Blocks = d.u64() + a.Blksize = d.u32() + a.Atime.decode(d) + a.Mtime.decode(d) + a.Ctime.decode(d) +} + +func (a *Attr) validate() error { + if !validFileType(a.Mode&ModeType) || a.Mode&^(ModeType|ModePerm) != 0 { + return malformed("mode %#o", a.Mode) + } + if a.Size > maxOffset { + return malformed("size %d", a.Size) + } + return errors.Join(a.Atime.validate(), a.Mtime.validate(), a.Ctime.validate()) +} + +func (en *Entry) encode(e *sandboxwire.Encoder) { en.Node.encode(e); en.Attr.encode(e) } +func (en *Entry) decode(d *decoder) { en.Node.decode(d); en.Attr.decode(d) } +func (en *Entry) validate() error { return errors.Join(en.Node.validate(), en.Attr.validate()) } + +func (t *Target) encode(e *sandboxwire.Encoder) { + e.Enum(uint16(t.Kind)) + if t.Kind == TargetNode { + t.Node.encode(e) + } else { + e.U64(uint64(t.Handle)) + } +} + +func (t *Target) decode(d *decoder) { + switch t.Kind = TargetKind(d.u16()); t.Kind { + case TargetNode: + t.Node.decode(d) + case TargetHandle: + t.Handle = HandleID(d.u64()) + } +} + +func (t *Target) validate() error { + switch t.Kind { + case TargetNode: + if t.Handle != 0 { + return malformed("node target with a handle") + } + return t.Node.validate() + case TargetHandle: + if t.Node != (NodeRef{}) { + return malformed("handle target with a node") + } + return t.Handle.validate() + } + return malformed("target kind %d", t.Kind) +} + +func (l *Lock) encode(e *sandboxwire.Encoder) { e.Enum(uint16(l.Mode)); e.U64(l.Start); e.U64(l.End) } +func (l *Lock) decode(d *decoder) { l.Mode = LockMode(d.u16()); l.Start = d.u64(); l.End = d.u64() } +func (l *Lock) validate() error { + if l.Mode < LockRead || l.Mode > LockUnlock { + return malformed("lock mode %d", l.Mode) + } + if l.Start > l.End || l.End > maxOffset { + return malformed("lock range %d-%d", l.Start, l.End) + } + return nil +} + +func (f *Failure) encode(e *sandboxwire.Encoder) { + e.Enum(uint16(f.Code)) + e.Present(f.Code == CodeErrno) + if f.Code == CodeErrno { + e.Enum(uint16(f.Errno)) + } + e.Effect(f.Effect) + e.Bytes([]byte(f.Message)) +} + +func (f *Failure) decode(d *decoder) { + f.Code = ErrorCode(d.u16()) + if d.present() { + if f.Errno = Errno(d.u16()); f.Errno == 0 && d.err == nil { + d.err = malformed("zero errno") + } + } + f.Effect = sandboxwire.Effect(d.u16()) + f.Message = string(d.bytes()) +} + +func (f *Failure) validate() error { + switch { + case !f.Code.Valid(): + return malformed("error code %d", f.Code) + case (f.Code == CodeErrno) != (f.Errno != 0): + return malformed("errno %d with code %s", f.Errno, f.Code) + case f.Errno != 0 && !f.Errno.Valid(): + return malformed("errno %d", f.Errno) + case !f.Effect.Valid(): + return malformed("effect %d", f.Effect) + case len(f.Message) > maxMessageBytes: + return malformed("message of %d bytes", len(f.Message)) + } + return nil +} + +func encodeOptionalFailure(e *sandboxwire.Encoder, f *Failure) { + e.Present(f != nil) + if f != nil { + f.encode(e) + } +} + +func decodeOptionalFailure(d *decoder) *Failure { + if !d.present() { + return nil + } + f := new(Failure) + f.decode(d) + return f +} + +func validName(n []byte) error { + if len(n) == 0 || len(n) > maxNameBytes { + return malformed("name of %d bytes", len(n)) + } + if string(n) == "." || string(n) == ".." { + return malformed("name %q", n) + } + for _, c := range n { + if c == 0 || c == '/' { + return malformed("name contains %q", c) + } + } + return nil +} + +func validTarget(t []byte) error { + if len(t) == 0 || len(t) > maxTargetBytes { + return malformed("symlink target of %d bytes", len(t)) + } + for _, c := range t { + if c == 0 { + return malformed("symlink target contains NUL") + } + } + return nil +} + +func validPerm(m uint32) error { + if m&^ModePerm != 0 { + return malformed("permission bits %#o", m) + } + return nil +} + +func validAccess(m AccessMode) error { + if m < AccessRead || m > AccessReadWrite { + return malformed("access mode %d", m) + } + return nil +} + +func validOpenFlags(f OpenFlags) error { + if f&^openFlagsAll != 0 { + return malformed("open flags %#x", uint32(f)) + } + return nil +} + +func (c *Capabilities) encode(e *sandboxwire.Encoder) { + e.Enum(uint16(c.PathProfile)) + e.Enum(uint16(c.CacheProfile)) + e.Enum(uint16(c.Durability)) + for _, v := range []uint32{c.MaxNameBytes, c.MaxPathBytes, c.MaxReadBytes, c.MaxWriteBytes, c.MaxWalkComponents, c.MaxReadDirBytes, c.MaxOpenHandles} { + e.U32(v) + } + for _, v := range c.flags() { + e.Bool(*v) + } +} + +func (c *Capabilities) decode(d *decoder) { + c.PathProfile = PathProfile(d.u16()) + c.CacheProfile = CacheProfile(d.u16()) + c.Durability = Durability(d.u16()) + for _, v := range []*uint32{&c.MaxNameBytes, &c.MaxPathBytes, &c.MaxReadBytes, &c.MaxWriteBytes, &c.MaxWalkComponents, &c.MaxReadDirBytes, &c.MaxOpenHandles} { + *v = d.u32() + } + for _, v := range c.flags() { + *v = d.bool() + } +} + +// flags lists the boolean capabilities in wire order. +func (c *Capabilities) flags() []*bool { + return []*bool{&c.ReadOnly, &c.AtomicAppend, &c.AtomicRename, &c.RenameNoReplace, &c.RenameExchange, + &c.HardLinks, &c.Symlinks, &c.SetMode, &c.SetOwner, &c.SetTimes, &c.DirectoryFsync, &c.ReadDirPlus, + &c.Flock, &c.POSIXLocks} +} + +func (c *Capabilities) validate() error { + within := func(v, max uint32) bool { return v >= 1 && v <= max } + switch { + case c.PathProfile != PathProfileLinuxBytes: + return malformed("path profile %d", c.PathProfile) + case c.CacheProfile != CacheProfileUncached: + return malformed("cache profile %d", c.CacheProfile) + case c.Durability != DurabilityFsyncRequired: + return malformed("durability %d", c.Durability) + case !within(c.MaxNameBytes, maxNameBytes), !within(c.MaxPathBytes, maxTargetBytes), + !within(c.MaxReadBytes, sandboxwire.MaxChunk), !within(c.MaxWriteBytes, sandboxwire.MaxChunk), + !within(c.MaxWalkComponents, maxWalkNames), !within(c.MaxReadDirBytes, maxReadDirBytes), + c.MaxOpenHandles == 0: + return malformed("capability limit out of range") + case !c.ReadOnly && !(c.AtomicAppend && c.AtomicRename && c.HardLinks && c.Symlinks): + return malformed("a writable service requires AtomicAppend, AtomicRename, HardLinks and Symlinks") + } + return nil +} + +// Messages. + +func (*DescribeRequest) encode(*sandboxwire.Encoder) {} +func (*DescribeRequest) decode(*decoder) {} +func (*DescribeRequest) validate() error { return nil } + +func (r *DescribeResponse) encode(e *sandboxwire.Encoder) { + e.ID(r.ServerInstanceID) + e.U32(r.Identity.UID) + e.U32(r.Identity.GID) + r.Capabilities.encode(e) + e.Count(len(r.Exports)) + for _, x := range r.Exports { + e.Bytes([]byte(x)) + } +} + +func (r *DescribeResponse) decode(d *decoder) { + r.ServerInstanceID = d.id() + r.Identity.UID = d.u32() + r.Identity.GID = d.u32() + r.Capabilities.decode(d) + r.Exports = make([]sandboxlink.ExportID, d.count(sandboxlink.MaxExports)) + for i := range r.Exports { + r.Exports[i] = sandboxlink.ExportID(d.bytes()) + } +} + +func (r *DescribeResponse) validate() error { + if r.ServerInstanceID.IsZero() { + return malformed("zero server instance") + } + if len(r.Exports) > sandboxlink.MaxExports { + return malformed("%d exports", len(r.Exports)) + } + seen := make(map[sandboxlink.ExportID]bool, len(r.Exports)) + for _, x := range r.Exports { + if !x.Valid() || seen[x] { + return malformed("export %q", string(x)) + } + seen[x] = true + } + return r.Capabilities.validate() +} + +func (r *AttachRequest) encode(e *sandboxwire.Encoder) { e.Bytes([]byte(r.Export)); e.Bool(r.ReadOnly) } +func (r *AttachRequest) decode(d *decoder) { + r.Export = sandboxlink.ExportID(d.bytes()) + r.ReadOnly = d.bool() +} +func (r *AttachRequest) validate() error { + if !r.Export.Valid() { + return malformed("export %q", string(r.Export)) + } + return nil +} + +func (r *AttachResponse) encode(e *sandboxwire.Encoder) { r.Root.encode(e) } +func (r *AttachResponse) decode(d *decoder) { r.Root.decode(d) } +func (r *AttachResponse) validate() error { + if r.Root.Attr.Mode&ModeType != ModeDirectory { + return malformed("export root is not a directory") + } + return r.Root.validate() +} + +func (*DetachRequest) encode(*sandboxwire.Encoder) {} +func (*DetachRequest) decode(*decoder) {} +func (*DetachRequest) validate() error { return nil } +func (*DetachResponse) encode(*sandboxwire.Encoder) {} +func (*DetachResponse) decode(*decoder) {} +func (*DetachResponse) validate() error { return nil } + +func (r *LookupRequest) encode(e *sandboxwire.Encoder) { r.Parent.encode(e); e.Bytes(r.Name) } +func (r *LookupRequest) decode(d *decoder) { r.Parent.decode(d); r.Name = d.bytes() } +func (r *LookupRequest) validate() error { return errors.Join(r.Parent.validate(), validName(r.Name)) } + +func (r *LookupResponse) encode(e *sandboxwire.Encoder) { r.Entry.encode(e) } +func (r *LookupResponse) decode(d *decoder) { r.Entry.decode(d) } +func (r *LookupResponse) validate() error { return r.Entry.validate() } + +func (r *WalkRequest) encode(e *sandboxwire.Encoder) { + r.Parent.encode(e) + e.Count(len(r.Names)) + for _, n := range r.Names { + e.Bytes(n) + } +} + +func (r *WalkRequest) decode(d *decoder) { + r.Parent.decode(d) + r.Names = make([][]byte, d.count(maxWalkNames)) + for i := range r.Names { + r.Names[i] = d.bytes() + } +} + +func (r *WalkRequest) validate() error { + if len(r.Names) == 0 || len(r.Names) > maxWalkNames { + return malformed("walk of %d names", len(r.Names)) + } + for _, n := range r.Names { + if err := validName(n); err != nil { + return err + } + } + return r.Parent.validate() +} + +func (r *WalkResponse) encode(e *sandboxwire.Encoder) { + e.Count(len(r.Entries)) + for i := range r.Entries { + r.Entries[i].encode(e) + } + encodeOptionalFailure(e, r.Failure) +} + +func (r *WalkResponse) decode(d *decoder) { + r.Entries = make([]Entry, d.count(maxWalkNames)) + for i := range r.Entries { + r.Entries[i].decode(d) + } + r.Failure = decodeOptionalFailure(d) +} + +func (r *WalkResponse) validate() error { + if len(r.Entries) == 0 || len(r.Entries) > maxWalkNames { + return malformed("walk of %d entries", len(r.Entries)) + } + for i := range r.Entries { + if err := r.Entries[i].validate(); err != nil { + return err + } + } + if r.Failure != nil { + return r.Failure.validate() + } + return nil +} + +func (r *GetAttrRequest) encode(e *sandboxwire.Encoder) { r.Target.encode(e) } +func (r *GetAttrRequest) decode(d *decoder) { r.Target.decode(d) } +func (r *GetAttrRequest) validate() error { return r.Target.validate() } + +func (r *GetAttrResponse) encode(e *sandboxwire.Encoder) { r.Attr.encode(e) } +func (r *GetAttrResponse) decode(d *decoder) { r.Attr.decode(d) } +func (r *GetAttrResponse) validate() error { return r.Attr.validate() } + +func (r *SetAttrRequest) encode(e *sandboxwire.Encoder) { + r.Target.encode(e) + e.U32(uint32(r.Set)) + if r.Set&AttrSize != 0 { + e.U64(r.Size) + } + if r.Set&AttrMode != 0 { + e.U32(r.Mode) + } + if r.Set&AttrUID != 0 { + e.U32(r.UID) + } + if r.Set&AttrGID != 0 { + e.U32(r.GID) + } + if r.Set&AttrAtime != 0 { + r.Atime.encode(e) + } + if r.Set&AttrMtime != 0 { + r.Mtime.encode(e) + } +} + +func (r *SetAttrRequest) decode(d *decoder) { + r.Target.decode(d) + r.Set = AttrMask(d.u32()) + if r.Set&AttrSize != 0 { + r.Size = d.u64() + } + if r.Set&AttrMode != 0 { + r.Mode = d.u32() + } + if r.Set&AttrUID != 0 { + r.UID = d.u32() + } + if r.Set&AttrGID != 0 { + r.GID = d.u32() + } + if r.Set&AttrAtime != 0 { + r.Atime.decode(d) + } + if r.Set&AttrMtime != 0 { + r.Mtime.decode(d) + } +} + +func (r *SetAttrRequest) validate() error { + unset := func(bit AttrMask, zero bool) bool { return r.Set&bit == 0 && !zero } + switch { + case r.Set&^attrMaskAll != 0: + return malformed("attribute mask %#x", uint32(r.Set)) + case r.Set&AttrAtime != 0 && r.Set&AttrAtimeNow != 0, r.Set&AttrMtime != 0 && r.Set&AttrMtimeNow != 0: + return malformed("a time is both given and now") + case unset(AttrSize, r.Size == 0), unset(AttrMode, r.Mode == 0), unset(AttrUID, r.UID == 0), + unset(AttrGID, r.GID == 0), unset(AttrAtime, r.Atime == Timestamp{}), unset(AttrMtime, r.Mtime == Timestamp{}): + return malformed("an unselected attribute is set") + case r.Size > maxOffset: + return malformed("size %d", r.Size) + case r.Set&AttrUID != 0 && r.UID == math.MaxUint32, r.Set&AttrGID != 0 && r.GID == math.MaxUint32: + return malformed("owner 4294967295 means no change") + } + return errors.Join(r.Target.validate(), validPerm(r.Mode), r.Atime.validate(), r.Mtime.validate()) +} + +func (r *SetAttrResponse) encode(e *sandboxwire.Encoder) { r.Attr.encode(e) } +func (r *SetAttrResponse) decode(d *decoder) { r.Attr.decode(d) } +func (r *SetAttrResponse) validate() error { return r.Attr.validate() } + +func (r *AccessRequest) encode(e *sandboxwire.Encoder) { r.Node.encode(e); e.U32(uint32(r.Mask)) } +func (r *AccessRequest) decode(d *decoder) { r.Node.decode(d); r.Mask = AccessMask(d.u32()) } +func (r *AccessRequest) validate() error { + if r.Mask&^accessMaskAll != 0 { + return malformed("access mask %#x", uint32(r.Mask)) + } + return r.Node.validate() +} + +func (*AccessResponse) encode(*sandboxwire.Encoder) {} +func (*AccessResponse) decode(*decoder) {} +func (*AccessResponse) validate() error { return nil } + +func (r *OpenRequest) encode(e *sandboxwire.Encoder) { + r.Node.encode(e) + e.Enum(uint16(r.Access)) + e.U32(uint32(r.Flags)) +} + +func (r *OpenRequest) decode(d *decoder) { + r.Node.decode(d) + r.Access = AccessMode(d.u16()) + r.Flags = OpenFlags(d.u32()) +} + +func (r *OpenRequest) validate() error { + return errors.Join(r.Node.validate(), validAccess(r.Access), validOpenFlags(r.Flags)) +} + +func (r *OpenResponse) encode(e *sandboxwire.Encoder) { e.U64(uint64(r.Handle)) } +func (r *OpenResponse) decode(d *decoder) { r.Handle = HandleID(d.u64()) } +func (r *OpenResponse) validate() error { return r.Handle.validate() } + +func (r *CreateRequest) encode(e *sandboxwire.Encoder) { + r.Parent.encode(e) + e.Bytes(r.Name) + e.U32(r.Mode) + e.Enum(uint16(r.Access)) + e.U32(uint32(r.Flags)) + e.Bool(r.Exclusive) +} + +func (r *CreateRequest) decode(d *decoder) { + r.Parent.decode(d) + r.Name = d.bytes() + r.Mode = d.u32() + r.Access = AccessMode(d.u16()) + r.Flags = OpenFlags(d.u32()) + r.Exclusive = d.bool() +} + +func (r *CreateRequest) validate() error { + return errors.Join(r.Parent.validate(), validName(r.Name), validPerm(r.Mode), validAccess(r.Access), validOpenFlags(r.Flags)) +} + +func (r *CreateResponse) encode(e *sandboxwire.Encoder) { r.Entry.encode(e); e.U64(uint64(r.Handle)) } +func (r *CreateResponse) decode(d *decoder) { r.Entry.decode(d); r.Handle = HandleID(d.u64()) } +func (r *CreateResponse) validate() error { + return errors.Join(r.Entry.validate(), r.Handle.validate()) +} + +func (r *ReadRequest) encode(e *sandboxwire.Encoder) { + e.U64(uint64(r.Handle)) + e.U64(r.Offset) + e.U32(r.Size) +} +func (r *ReadRequest) decode(d *decoder) { + r.Handle = HandleID(d.u64()) + r.Offset = d.u64() + r.Size = d.u32() +} + +func (r *ReadRequest) validate() error { + if r.Offset > maxOffset || r.Size > sandboxwire.MaxChunk { + return malformed("read of %d bytes at %d", r.Size, r.Offset) + } + return r.Handle.validate() +} + +func (r *ReadResponse) encode(e *sandboxwire.Encoder) { e.Bytes(r.Data) } +func (r *ReadResponse) decode(d *decoder) { r.Data = d.bytes() } +func (r *ReadResponse) validate() error { + if len(r.Data) > sandboxwire.MaxChunk { + return malformed("read of %d bytes", len(r.Data)) + } + return nil +} + +func (r *WriteRequest) encode(e *sandboxwire.Encoder) { + e.U64(uint64(r.Handle)) + e.U64(r.Offset) + e.Bytes(r.Data) +} +func (r *WriteRequest) decode(d *decoder) { + r.Handle = HandleID(d.u64()) + r.Offset = d.u64() + r.Data = d.bytes() +} + +func (r *WriteRequest) validate() error { + if len(r.Data) > sandboxwire.MaxChunk { + return malformed("write of %d bytes", len(r.Data)) + } + return r.Handle.validate() +} + +func (r *WriteResponse) encode(e *sandboxwire.Encoder) { + e.U32(r.Written) + encodeOptionalFailure(e, r.Failure) +} +func (r *WriteResponse) decode(d *decoder) { r.Written = d.u32(); r.Failure = decodeOptionalFailure(d) } +func (r *WriteResponse) validate() error { + if r.Written > sandboxwire.MaxChunk { + return malformed("wrote %d bytes", r.Written) + } + if r.Failure != nil { + if r.Written == 0 { + return malformed("a write failure after no bytes belongs in a failure response") + } + return r.Failure.validate() + } + return nil +} + +func (r *FlushRequest) encode(e *sandboxwire.Encoder) { + e.U64(uint64(r.Handle)) + e.U64(uint64(r.Owner)) +} +func (r *FlushRequest) decode(d *decoder) { r.Handle = HandleID(d.u64()); r.Owner = LockOwner(d.u64()) } +func (r *FlushRequest) validate() error { return r.Handle.validate() } + +func (*FlushResponse) encode(*sandboxwire.Encoder) {} +func (*FlushResponse) decode(*decoder) {} +func (*FlushResponse) validate() error { return nil } + +func (r *FsyncRequest) encode(e *sandboxwire.Encoder) { e.U64(uint64(r.Handle)); e.Bool(r.DataOnly) } +func (r *FsyncRequest) decode(d *decoder) { r.Handle = HandleID(d.u64()); r.DataOnly = d.bool() } +func (r *FsyncRequest) validate() error { return r.Handle.validate() } + +func (*FsyncResponse) encode(*sandboxwire.Encoder) {} +func (*FsyncResponse) decode(*decoder) {} +func (*FsyncResponse) validate() error { return nil } + +func (r *ReleaseRequest) encode(e *sandboxwire.Encoder) { e.U64(uint64(r.Handle)) } +func (r *ReleaseRequest) decode(d *decoder) { r.Handle = HandleID(d.u64()) } +func (r *ReleaseRequest) validate() error { return r.Handle.validate() } + +func (*ReleaseResponse) encode(*sandboxwire.Encoder) {} +func (*ReleaseResponse) decode(*decoder) {} +func (*ReleaseResponse) validate() error { return nil } + +func (r *OpenDirRequest) encode(e *sandboxwire.Encoder) { r.Node.encode(e) } +func (r *OpenDirRequest) decode(d *decoder) { r.Node.decode(d) } +func (r *OpenDirRequest) validate() error { return r.Node.validate() } + +func (r *OpenDirResponse) encode(e *sandboxwire.Encoder) { e.U64(uint64(r.Handle)) } +func (r *OpenDirResponse) decode(d *decoder) { r.Handle = HandleID(d.u64()) } +func (r *OpenDirResponse) validate() error { return r.Handle.validate() } + +func (r *ReadDirRequest) encode(e *sandboxwire.Encoder) { + e.U64(uint64(r.Handle)) + e.U64(r.Cookie) + e.U32(r.Limit) + e.Bool(r.WithAttrs) +} + +func (r *ReadDirRequest) decode(d *decoder) { + r.Handle = HandleID(d.u64()) + r.Cookie = d.u64() + r.Limit = d.u32() + r.WithAttrs = d.bool() +} + +func (r *ReadDirRequest) validate() error { + if r.Limit == 0 || r.Limit > maxReadDirBytes { + return malformed("limit %d", r.Limit) + } + return r.Handle.validate() +} + +func (r *ReadDirResponse) encode(e *sandboxwire.Encoder) { + e.Count(len(r.Entries)) + for i := range r.Entries { + x := &r.Entries[i] + e.Bytes(x.Name) + e.U64(x.Ino) + e.U32(x.Type) + e.U64(x.Cookie) + e.Present(x.Entry != nil) + if x.Entry != nil { + x.Entry.encode(e) + } + } + e.Bool(r.End) +} + +func (r *ReadDirResponse) decode(d *decoder) { + r.Entries = make([]DirEntry, d.count(maxReadDirBytes/dirEntryWireSize)) + for i := range r.Entries { + x := &r.Entries[i] + x.Name = d.bytes() + x.Ino = d.u64() + x.Type = d.u32() + x.Cookie = d.u64() + if d.present() { + x.Entry = new(Entry) + x.Entry.decode(d) + } + } + r.End = d.bool() +} + +func (r *ReadDirResponse) validate() error { + if len(r.Entries) == 0 && !r.End { + return malformed("an empty page before the end") + } + size := 0 + names := make(map[string]bool, len(r.Entries)) + cookies := make(map[uint64]bool, len(r.Entries)) + for i := range r.Entries { + x := &r.Entries[i] + if err := validName(x.Name); err != nil { + return err + } + if names[string(x.Name)] || cookies[x.Cookie] { + return malformed("entry %q repeats a name or cookie in the page", x.Name) + } + names[string(x.Name)], cookies[x.Cookie] = true, true + if !validFileType(x.Type) { + return malformed("entry type %#o", x.Type) + } + if x.Entry != nil { + if err := x.Entry.validate(); err != nil { + return err + } + if x.Entry.Attr.Ino != x.Ino || x.Entry.Attr.Mode&ModeType != x.Type { + return malformed("entry attributes disagree with the entry") + } + } + size += x.WireSize() + } + if size > maxReadDirBytes { + return malformed("entries of %d bytes", size) + } + return nil +} + +func (r *ReleaseDirRequest) encode(e *sandboxwire.Encoder) { e.U64(uint64(r.Handle)) } +func (r *ReleaseDirRequest) decode(d *decoder) { r.Handle = HandleID(d.u64()) } +func (r *ReleaseDirRequest) validate() error { return r.Handle.validate() } + +func (*ReleaseDirResponse) encode(*sandboxwire.Encoder) {} +func (*ReleaseDirResponse) decode(*decoder) {} +func (*ReleaseDirResponse) validate() error { return nil } + +func (r *MkdirRequest) encode(e *sandboxwire.Encoder) { + r.Parent.encode(e) + e.Bytes(r.Name) + e.U32(r.Mode) +} +func (r *MkdirRequest) decode(d *decoder) { r.Parent.decode(d); r.Name = d.bytes(); r.Mode = d.u32() } +func (r *MkdirRequest) validate() error { + return errors.Join(r.Parent.validate(), validName(r.Name), validPerm(r.Mode)) +} + +func (r *MkdirResponse) encode(e *sandboxwire.Encoder) { r.Entry.encode(e) } +func (r *MkdirResponse) decode(d *decoder) { r.Entry.decode(d) } +func (r *MkdirResponse) validate() error { return r.Entry.validate() } + +func (r *UnlinkRequest) encode(e *sandboxwire.Encoder) { r.Parent.encode(e); e.Bytes(r.Name) } +func (r *UnlinkRequest) decode(d *decoder) { r.Parent.decode(d); r.Name = d.bytes() } +func (r *UnlinkRequest) validate() error { return errors.Join(r.Parent.validate(), validName(r.Name)) } + +func (*UnlinkResponse) encode(*sandboxwire.Encoder) {} +func (*UnlinkResponse) decode(*decoder) {} +func (*UnlinkResponse) validate() error { return nil } + +func (r *RmdirRequest) encode(e *sandboxwire.Encoder) { r.Parent.encode(e); e.Bytes(r.Name) } +func (r *RmdirRequest) decode(d *decoder) { r.Parent.decode(d); r.Name = d.bytes() } +func (r *RmdirRequest) validate() error { return errors.Join(r.Parent.validate(), validName(r.Name)) } + +func (*RmdirResponse) encode(*sandboxwire.Encoder) {} +func (*RmdirResponse) decode(*decoder) {} +func (*RmdirResponse) validate() error { return nil } + +func (r *RenameRequest) encode(e *sandboxwire.Encoder) { + r.Parent.encode(e) + e.Bytes(r.Name) + r.NewParent.encode(e) + e.Bytes(r.NewName) + e.Enum(uint16(r.Mode)) +} + +func (r *RenameRequest) decode(d *decoder) { + r.Parent.decode(d) + r.Name = d.bytes() + r.NewParent.decode(d) + r.NewName = d.bytes() + r.Mode = RenameMode(d.u16()) +} + +func (r *RenameRequest) validate() error { + if r.Mode < RenameReplace || r.Mode > RenameExchange { + return malformed("rename mode %d", r.Mode) + } + return errors.Join(r.Parent.validate(), validName(r.Name), r.NewParent.validate(), validName(r.NewName)) +} + +func (*RenameResponse) encode(*sandboxwire.Encoder) {} +func (*RenameResponse) decode(*decoder) {} +func (*RenameResponse) validate() error { return nil } + +func (r *LinkRequest) encode(e *sandboxwire.Encoder) { + r.Node.encode(e) + r.NewParent.encode(e) + e.Bytes(r.NewName) +} +func (r *LinkRequest) decode(d *decoder) { + r.Node.decode(d) + r.NewParent.decode(d) + r.NewName = d.bytes() +} +func (r *LinkRequest) validate() error { + return errors.Join(r.Node.validate(), r.NewParent.validate(), validName(r.NewName)) +} + +func (r *LinkResponse) encode(e *sandboxwire.Encoder) { r.Entry.encode(e) } +func (r *LinkResponse) decode(d *decoder) { r.Entry.decode(d) } +func (r *LinkResponse) validate() error { return r.Entry.validate() } + +func (r *SymlinkRequest) encode(e *sandboxwire.Encoder) { + r.Parent.encode(e) + e.Bytes(r.Name) + e.Bytes(r.Target) +} +func (r *SymlinkRequest) decode(d *decoder) { + r.Parent.decode(d) + r.Name = d.bytes() + r.Target = d.bytes() +} +func (r *SymlinkRequest) validate() error { + return errors.Join(r.Parent.validate(), validName(r.Name), validTarget(r.Target)) +} + +func (r *SymlinkResponse) encode(e *sandboxwire.Encoder) { r.Entry.encode(e) } +func (r *SymlinkResponse) decode(d *decoder) { r.Entry.decode(d) } +func (r *SymlinkResponse) validate() error { return r.Entry.validate() } + +func (r *ReadlinkRequest) encode(e *sandboxwire.Encoder) { r.Node.encode(e) } +func (r *ReadlinkRequest) decode(d *decoder) { r.Node.decode(d) } +func (r *ReadlinkRequest) validate() error { return r.Node.validate() } + +func (r *ReadlinkResponse) encode(e *sandboxwire.Encoder) { e.Bytes(r.Target) } +func (r *ReadlinkResponse) decode(d *decoder) { r.Target = d.bytes() } +func (r *ReadlinkResponse) validate() error { return validTarget(r.Target) } + +func (r *StatFSRequest) encode(e *sandboxwire.Encoder) { r.Node.encode(e) } +func (r *StatFSRequest) decode(d *decoder) { r.Node.decode(d) } +func (r *StatFSRequest) validate() error { return r.Node.validate() } + +func (r *StatFSResponse) encode(e *sandboxwire.Encoder) { + for _, v := range []uint64{r.Blocks, r.BlocksFree, r.BlocksAvailable, r.Files, r.FilesFree} { + e.U64(v) + } + e.U32(r.BlockSize) + e.U32(r.FragmentSize) + e.U32(r.NameMax) +} + +func (r *StatFSResponse) decode(d *decoder) { + for _, v := range []*uint64{&r.Blocks, &r.BlocksFree, &r.BlocksAvailable, &r.Files, &r.FilesFree} { + *v = d.u64() + } + r.BlockSize = d.u32() + r.FragmentSize = d.u32() + r.NameMax = d.u32() +} + +func (*StatFSResponse) validate() error { return nil } + +func (r *ForgetRequest) encode(e *sandboxwire.Encoder) { + e.Count(len(r.Entries)) + for _, f := range r.Entries { + f.Node.encode(e) + e.U64(f.Count) + } +} + +func (r *ForgetRequest) decode(d *decoder) { + r.Entries = make([]ForgetEntry, d.count(maxForgetEntries)) + for i := range r.Entries { + r.Entries[i].Node.decode(d) + r.Entries[i].Count = d.u64() + } +} + +func (r *ForgetRequest) validate() error { + if len(r.Entries) == 0 || len(r.Entries) > maxForgetEntries { + return malformed("forget of %d entries", len(r.Entries)) + } + seen := make(map[uint64]bool, len(r.Entries)) + for _, f := range r.Entries { + if err := f.Node.validate(); err != nil { + return err + } + if f.Count == 0 || seen[f.Node.ID] { + return malformed("forget entry %d", f.Node.ID) + } + seen[f.Node.ID] = true + } + return nil +} + +func (*ForgetResponse) encode(*sandboxwire.Encoder) {} +func (*ForgetResponse) decode(*decoder) {} +func (*ForgetResponse) validate() error { return nil } + +func (r *GetLockRequest) encode(e *sandboxwire.Encoder) { + e.U64(uint64(r.Handle)) + e.U64(uint64(r.Owner)) + r.Lock.encode(e) +} + +func (r *GetLockRequest) decode(d *decoder) { + r.Handle = HandleID(d.u64()) + r.Owner = LockOwner(d.u64()) + r.Lock.decode(d) +} + +func (r *GetLockRequest) validate() error { + if r.Lock.Mode == LockUnlock { + return malformed("GetLock of an unlock") + } + return errors.Join(r.Handle.validate(), r.Lock.validate()) +} + +func (r *GetLockResponse) encode(e *sandboxwire.Encoder) { + e.Present(r.Conflict != nil) + if r.Conflict != nil { + r.Conflict.encode(e) + } +} + +func (r *GetLockResponse) decode(d *decoder) { + if d.present() { + r.Conflict = new(Lock) + r.Conflict.decode(d) + } +} + +func (r *GetLockResponse) validate() error { + if r.Conflict == nil { + return nil + } + if r.Conflict.Mode == LockUnlock { + return malformed("conflict with an unlock") + } + return r.Conflict.validate() +} + +func (r *SetLockRequest) encode(e *sandboxwire.Encoder) { + e.U64(uint64(r.Handle)) + e.Enum(uint16(r.Kind)) + e.U64(uint64(r.Owner)) + r.Lock.encode(e) + e.Bool(r.Wait) +} + +func (r *SetLockRequest) decode(d *decoder) { + r.Handle = HandleID(d.u64()) + r.Kind = LockKind(d.u16()) + r.Owner = LockOwner(d.u64()) + r.Lock.decode(d) + r.Wait = d.bool() +} + +func (r *SetLockRequest) validate() error { + switch r.Kind { + case LockPOSIX: + case LockFlock: + if r.Lock.Start != 0 || r.Lock.End != maxOffset { + return malformed("a flock lock covers the whole file") + } + default: + return malformed("lock kind %d", r.Kind) + } + return errors.Join(r.Handle.validate(), r.Lock.validate()) +} + +func (*SetLockResponse) encode(*sandboxwire.Encoder) {} +func (*SetLockResponse) decode(*decoder) {} +func (*SetLockResponse) validate() error { return nil } + +func (r *CancelRequestRequest) encode(e *sandboxwire.Encoder) { e.U64(r.Target) } +func (r *CancelRequestRequest) decode(d *decoder) { r.Target = d.u64() } +func (r *CancelRequestRequest) validate() error { + if !sandboxwire.ValidRequestID(r.Target) { + return malformed("cancel of request 0") + } + return nil +} + +func (*CancelRequestResponse) encode(*sandboxwire.Encoder) {} +func (*CancelRequestResponse) decode(*decoder) {} +func (*CancelRequestResponse) validate() error { return nil } diff --git a/internal/sandboxfs/protocol_test.go b/internal/sandboxfs/protocol_test.go new file mode 100644 index 00000000..ed8ad24a --- /dev/null +++ b/internal/sandboxfs/protocol_test.go @@ -0,0 +1,297 @@ +package sandboxfs + +import ( + "bytes" + "encoding/hex" + "errors" + "math" + "os" + "path/filepath" + "reflect" + "strings" + "testing" + + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxlink" + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxwire" +) + +func readHexFixture(t testing.TB, name string) []byte { + t.Helper() + raw, err := os.ReadFile(filepath.Join("testdata", name)) + if err != nil { + t.Fatal(err) + } + var digits strings.Builder + for _, line := range strings.Split(string(raw), "\n") { + line, _, _ = strings.Cut(line, "#") + digits.WriteString(strings.Join(strings.Fields(line), "")) + } + b, err := hex.DecodeString(digits.String()) + if err != nil { + t.Fatal(err) + } + return b +} + +var ( + testNode = NodeRef{ID: 1, Generation: 7} + testNode2 = NodeRef{ID: 2, Generation: 9} + testAttr = Attr{Ino: 0x102, Mode: ModeRegular | 0o644, Nlink: 1, UID: 1000, GID: 1000, Size: 5, Blocks: 8, Blksize: 4096, Atime: Timestamp{1, 2}, Mtime: Timestamp{3, 4}, Ctime: Timestamp{-5, 6}} + testDirAttr = Attr{Ino: 0x101, Mode: ModeDirectory | 0o755, Nlink: 2, Blksize: 4096} + testEntry = Entry{Node: testNode2, Attr: testAttr} + testCaps = Capabilities{ + PathProfile: PathProfileLinuxBytes, CacheProfile: CacheProfileUncached, Durability: DurabilityFsyncRequired, + MaxNameBytes: 255, MaxPathBytes: 4095, MaxReadBytes: 65536, MaxWriteBytes: 65536, MaxWalkComponents: 256, + MaxReadDirBytes: 65536, MaxOpenHandles: 4096, AtomicAppend: true, AtomicRename: true, RenameNoReplace: true, + RenameExchange: true, HardLinks: true, Symlinks: true, SetMode: true, SetOwner: true, SetTimes: true, + DirectoryFsync: true, ReadDirPlus: true, Flock: true, + } + testInstance = sandboxwire.ID{0x00, 0x11, 0x22, 0x33, 0x44, 0x55, 0x66, 0x77, 0x88, 0x99, 0xaa, 0xbb, 0xcc, 0xdd, 0xee, 0xff} +) + +func TestGoldenFixtures(t *testing.T) { + symlink := Attr{Ino: 0x102, Mode: ModeSymlink | 0o777, Nlink: 1, UID: 1000, GID: 1000, Size: 1, Blksize: 4096, + Atime: Timestamp{0x65000000, 1}, Mtime: Timestamp{0x65000000, 2}, Ctime: Timestamp{0x65000000, 3}} + for _, tc := range []struct { + file string + op Op + id uint64 + req Request + resp message + fail *Failure + }{ + {file: "describe_response.hex", op: OpDescribe, id: 1, resp: &DescribeResponse{ + ServerInstanceID: testInstance, Identity: Identity{UID: 1000, GID: 1000}, Capabilities: testCaps, Exports: []sandboxlink.ExportID{"world"}}}, + {file: "walk_request.hex", op: OpWalk, id: 2, req: &WalkRequest{Parent: testNode, Names: [][]byte{[]byte("link"), []byte("x")}}}, + {file: "walk_response.hex", op: OpWalk, id: 2, resp: &WalkResponse{Entries: []Entry{{Node: NodeRef{ID: 2, Generation: 7}, Attr: symlink}}}}, + {file: "create_request.hex", op: OpCreate, id: 3, req: &CreateRequest{ + Parent: testNode, Name: []byte("notes.md"), Mode: 0o644, Access: AccessReadWrite, Flags: OpenAppend, Exclusive: true}}, + {file: "write_response.hex", op: OpWrite, id: 4, resp: &WriteResponse{ + Written: 4096, Failure: &Failure{Code: CodeErrno, Errno: ErrnoNoSpace, Effect: sandboxwire.EffectNone, Message: "disk full"}}}, + {file: "readdir_request.hex", op: OpReadDir, id: 5, req: &ReadDirRequest{Handle: 0x42, Cookie: 0x1c6a3e5f0b9d2471, Limit: 65536}}, + {file: "readdir_response.hex", op: OpReadDir, id: 5, resp: &ReadDirResponse{Entries: []DirEntry{ + {Name: []byte("a.txt"), Ino: 0x103, Type: ModeRegular, Cookie: 0x2f0e4b6c7d8a9b10}, + {Name: []byte("src"), Ino: 0x104, Type: ModeDirectory, Cookie: 0x3a5c7e9f1b2d4f60}, + }}}, + {file: "rename_request.hex", op: OpRename, id: 6, req: &RenameRequest{ + Parent: testNode, Name: []byte("old"), NewParent: testNode2, NewName: []byte("new"), Mode: RenameExchange}}, + {file: "failure_response.hex", op: OpLookup, id: 7, fail: &Failure{ + Code: CodeErrno, Errno: ErrnoNotFound, Effect: sandboxwire.EffectNone, Message: "missing"}}, + } { + t.Run(tc.file, func(t *testing.T) { + want := readHexFixture(t, tc.file) + typ := uint16(tc.op) + var payload []byte + var err error + if tc.req != nil { + payload, err = encodeRequest(tc.req) + } else { + typ = sandboxwire.ResponseType(typ) + payload, err = encodeResponse(tc.resp, tc.fail) + } + if err != nil { + t.Fatal(err) + } + var got bytes.Buffer + if err := sandboxwire.WriteFrame(&got, sandboxwire.Frame{Type: typ, RequestID: tc.id, Payload: payload}); err != nil { + t.Fatal(err) + } + if !bytes.Equal(got.Bytes(), want) { + t.Fatalf("encoded\n got %x\nwant %x", got.Bytes(), want) + } + + f, err := sandboxwire.ReadFrame(bytes.NewReader(want), sandboxwire.MaxPayload) + if err != nil || f.Type != typ || f.RequestID != tc.id { + t.Fatalf("frame %+v, %v", f, err) + } + if tc.req != nil { + req, err := decodeRequest(tc.op, f.Payload) + if err != nil || !reflect.DeepEqual(req, tc.req) { + t.Fatalf("decoded %+v, %v", req, err) + } + return + } + resp, fail, err := decodeResponse(tc.op, f.Payload) + if err != nil || !reflect.DeepEqual(fail, tc.fail) || (tc.resp != nil && !reflect.DeepEqual(resp, tc.resp)) { + t.Fatalf("decoded %+v %+v, %v", resp, fail, err) + } + }) + } +} + +// samples holds one valid request and response per operation, in tag order. +func samples() []struct { + req Request + resp message +} { + flock := Lock{Mode: LockWrite, Start: 0, End: math.MaxInt64} + return []struct { + req Request + resp message + }{ + {&DescribeRequest{}, &DescribeResponse{ServerInstanceID: testInstance, Identity: Identity{UID: 1, GID: 2}, Capabilities: testCaps, Exports: []sandboxlink.ExportID{"world", "home"}}}, + {&AttachRequest{Export: "world", ReadOnly: true}, &AttachResponse{Root: Entry{Node: testNode, Attr: testDirAttr}}}, + {&DetachRequest{}, &DetachResponse{}}, + {&LookupRequest{Parent: testNode, Name: []byte("a\xff")}, &LookupResponse{Entry: testEntry}}, + {&WalkRequest{Parent: testNode, Names: [][]byte{[]byte("a"), []byte("b")}}, &WalkResponse{Entries: []Entry{testEntry}, Failure: NewErrnoFailure(ErrnoNotFound, sandboxwire.EffectNone, "b")}}, + {&GetAttrRequest{Target: Target{Kind: TargetHandle, Handle: 3}}, &GetAttrResponse{Attr: testAttr}}, + {&SetAttrRequest{Target: Target{Kind: TargetNode, Node: testNode}, Set: AttrSize | AttrMode | AttrUID | AttrGID | AttrAtime | AttrMtimeNow, Size: 9, Mode: 0o4755, UID: 5, GID: 6, Atime: Timestamp{7, 8}}, &SetAttrResponse{Attr: testAttr}}, + {&AccessRequest{Node: testNode, Mask: MayRead | MayExecute}, &AccessResponse{}}, + {&OpenRequest{Node: testNode, Access: AccessWrite, Flags: OpenTruncate | OpenSync}, &OpenResponse{Handle: 4}}, + {&CreateRequest{Parent: testNode, Name: []byte("n"), Mode: 0o600, Access: AccessRead, Flags: OpenDataSync}, &CreateResponse{Entry: testEntry, Handle: 5}}, + {&ReadRequest{Handle: 4, Offset: 10, Size: 65536}, &ReadResponse{Data: []byte("data")}}, + {&WriteRequest{Handle: 4, Offset: 10, Data: []byte("data")}, &WriteResponse{Written: 4}}, + {&FlushRequest{Handle: 4, Owner: 11}, &FlushResponse{}}, + {&FsyncRequest{Handle: 4, DataOnly: true}, &FsyncResponse{}}, + {&ReleaseRequest{Handle: 4}, &ReleaseResponse{}}, + {&OpenDirRequest{Node: testNode}, &OpenDirResponse{Handle: 6}}, + {&ReadDirRequest{Handle: 6, Cookie: 1, Limit: 4096, WithAttrs: true}, &ReadDirResponse{Entries: []DirEntry{{Name: []byte("f"), Ino: testAttr.Ino, Type: ModeRegular, Cookie: 2, Entry: &testEntry}}, End: true}}, + {&ReleaseDirRequest{Handle: 6}, &ReleaseDirResponse{}}, + {&MkdirRequest{Parent: testNode, Name: []byte("d"), Mode: 0o1777}, &MkdirResponse{Entry: testEntry}}, + {&UnlinkRequest{Parent: testNode, Name: []byte("f")}, &UnlinkResponse{}}, + {&RmdirRequest{Parent: testNode, Name: []byte("d")}, &RmdirResponse{}}, + {&RenameRequest{Parent: testNode, Name: []byte("a"), NewParent: testNode2, NewName: []byte("b"), Mode: RenameNoReplace}, &RenameResponse{}}, + {&LinkRequest{Node: testNode2, NewParent: testNode, NewName: []byte("h")}, &LinkResponse{Entry: testEntry}}, + {&SymlinkRequest{Parent: testNode, Name: []byte("s"), Target: []byte("../t")}, &SymlinkResponse{Entry: testEntry}}, + {&ReadlinkRequest{Node: testNode2}, &ReadlinkResponse{Target: []byte("../t")}}, + {&StatFSRequest{Node: testNode}, &StatFSResponse{Blocks: 1, BlocksFree: 2, BlocksAvailable: 3, Files: 4, FilesFree: 5, BlockSize: 4096, FragmentSize: 4096, NameMax: 255}}, + {&ForgetRequest{Entries: []ForgetEntry{{Node: testNode, Count: 1}, {Node: testNode2, Count: 3}}}, &ForgetResponse{}}, + {&GetLockRequest{Handle: 4, Owner: 12, Lock: Lock{Mode: LockRead, Start: 0, End: 99}}, &GetLockResponse{Conflict: &Lock{Mode: LockWrite, Start: 50, End: 60}}}, + {&SetLockRequest{Handle: 4, Kind: LockFlock, Owner: 12, Lock: flock, Wait: true}, &SetLockResponse{}}, + {&CancelRequestRequest{Target: 9}, &CancelRequestResponse{}}, + } +} + +func TestRoundTripEveryMessage(t *testing.T) { + all := samples() + if len(all) != int(OpCancelRequest) { + t.Fatalf("%d samples for %d operations", len(all), OpCancelRequest) + } + for i, s := range all { + op := Op(i + 1) + if s.req.Op() != op { + t.Fatalf("sample %d is %s", i, s.req.Op()) + } + payload, err := encodeRequest(s.req) + if err != nil { + t.Fatalf("%s request: %v", op, err) + } + if req, err := decodeRequest(op, payload); err != nil || !reflect.DeepEqual(req, s.req) { + t.Fatalf("%s request decoded %+v, %v", op, req, err) + } + if payload, err = encodeResponse(s.resp, nil); err != nil { + t.Fatalf("%s response: %v", op, err) + } + if resp, fail, err := decodeResponse(op, payload); err != nil || fail != nil || !reflect.DeepEqual(resp, s.resp) { + t.Fatalf("%s response decoded %+v %v, %v", op, resp, fail, err) + } + } +} + +func TestDecodeRejects(t *testing.T) { + encode := func(m message) []byte { + var e sandboxwire.Encoder + m.encode(&e) + return e.Payload() + } + for name, tc := range map[string]struct { + op Op + payload []byte + }{ + "dot-dot name": {OpLookup, encode(&LookupRequest{Parent: testNode, Name: []byte("..")})}, + "slash in name": {OpMkdir, encode(&MkdirRequest{Parent: testNode, Name: []byte("a/b")})}, + "zero generation": {OpOpenDir, encode(&OpenDirRequest{Node: NodeRef{ID: 1}})}, + "unknown open flag": {OpOpen, encode(&OpenRequest{Node: testNode, Access: AccessRead, Flags: 1 << 5})}, + "zero access mode": {OpOpen, encode(&OpenRequest{Node: testNode})}, + "duplicate forget": {OpForget, encode(&ForgetRequest{Entries: []ForgetEntry{{testNode, 1}, {testNode, 1}}})}, + "partial flock range": {OpSetLock, encode(&SetLockRequest{Handle: 1, Kind: LockFlock, Lock: Lock{Mode: LockRead, End: 10}})}, + "trailing byte": {OpReleaseDir, append(encode(&ReleaseDirRequest{Handle: 1}), 0)}, + "owner sentinel": {OpSetAttr, encode(&SetAttrRequest{Target: Target{Kind: TargetHandle, Handle: 1}, Set: AttrUID, UID: math.MaxUint32})}, + "uppercase export": {OpAttach, encode(&AttachRequest{Export: "World"})}, + "dot in export": {OpAttach, encode(&AttachRequest{Export: "a.b"})}, + } { + if _, err := decodeRequest(tc.op, tc.payload); !errors.Is(err, sandboxwire.ErrMalformed) { + t.Errorf("%s: %v", name, err) + } + } + entry := func(name string, cookie uint64) DirEntry { + return DirEntry{Name: []byte(name), Ino: 1, Type: ModeRegular, Cookie: cookie} + } + for name, page := range map[string]*ReadDirResponse{ + "repeated name": {Entries: []DirEntry{entry("a", 1), entry("a", 2)}, End: true}, + "repeated cookie": {Entries: []DirEntry{entry("a", 1), entry("b", 1)}, End: true}, + "empty page": {}, + } { + var e sandboxwire.Encoder + e.Enum(uint16(ResultSuccess)) + page.encode(&e) + if _, _, err := decodeResponse(OpReadDir, e.Payload()); !errors.Is(err, sandboxwire.ErrMalformed) { + t.Errorf("%s: %v", name, err) + } + } +} + +func FuzzDecode(f *testing.F) { + entries, _ := os.ReadDir("testdata") + for _, e := range entries { + if strings.HasSuffix(e.Name(), ".hex") { + f.Add(readHexFixture(f, e.Name())) + } + } + for i, s := range samples() { + op := uint16(i + 1) + for _, m := range []struct { + typ uint16 + msg message + }{{op, s.req}, {sandboxwire.ResponseType(op), s.resp}} { + var e sandboxwire.Encoder + if m.typ == op { + m.msg.encode(&e) + } else { + e.Enum(uint16(ResultSuccess)) + m.msg.encode(&e) + } + var w bytes.Buffer + sandboxwire.WriteFrame(&w, sandboxwire.Frame{Type: m.typ, RequestID: 1, Payload: e.Payload()}) + f.Add(w.Bytes()) + } + } + f.Fuzz(func(t *testing.T, data []byte) { + fr, err := sandboxwire.ReadFrame(bytes.NewReader(data), sandboxwire.MaxPayload) + if err != nil { + return + } + kind, err := tags.Classify(fr.Type) + if err != nil { + return + } + var again []byte + switch kind { + case sandboxwire.KindRequest: + req, err := decodeRequest(Op(fr.Type), fr.Payload) + if err != nil { + if !errors.Is(err, sandboxwire.ErrMalformed) { + t.Fatalf("request error %v is not ErrMalformed", err) + } + return + } + if again, err = encodeRequest(req); err != nil { + t.Fatalf("decoded request does not encode: %v", err) + } + case sandboxwire.KindResponse: + resp, fail, err := decodeResponse(Op(fr.Type^sandboxwire.ResponseType(0)), fr.Payload) + if err != nil { + if !errors.Is(err, sandboxwire.ErrMalformed) { + t.Fatalf("response error %v is not ErrMalformed", err) + } + return + } + if again, err = encodeResponse(resp, fail); err != nil { + t.Fatalf("decoded response does not encode: %v", err) + } + } + if !bytes.Equal(again, fr.Payload) { + t.Fatalf("round trip changed the payload:\n got %x\nwant %x", again, fr.Payload) + } + }) +} diff --git a/internal/sandboxfs/server.go b/internal/sandboxfs/server.go new file mode 100644 index 00000000..52735d53 --- /dev/null +++ b/internal/sandboxfs/server.go @@ -0,0 +1,139 @@ +package sandboxfs + +import ( + "context" + "errors" + "io" + "slices" + "sync" + + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxwire" +) + +// MaxInFlight is how many requests one stream holds at once, from admission +// until the response is written. The server answers a request beyond it with +// ResourceExhausted and EffectNone, so it keeps reading and a CancelRequest +// always gets through while the client reads responses. +const MaxInFlight = 256 + +// Serve answers the requests on conn with svc until the stream ends or ctx is +// done. a is the attachment Link authenticated for the stream; Serve refuses +// one without an ID, server instance, lease or valid export grants. Serve owns +// conn and closes it. On return every request context is cancelled and every +// handler has finished. A stream that ends cleanly returns nil. +func Serve(ctx context.Context, conn io.ReadWriteCloser, svc Service, a Attachment) error { + if err := a.validate(); err != nil { + conn.Close() + return err + } + a.Exports = slices.Clone(a.Exports) + ctx, cancel := context.WithCancel(ctx) + s := &server{conn: conn, svc: svc, a: a, inflight: map[uint64]context.CancelFunc{}} + stop := context.AfterFunc(ctx, func() { conn.Close() }) + defer func() { + cancel() + stop() + conn.Close() + s.wg.Wait() + }() + for { + f, err := sandboxwire.ReadFrame(conn, sandboxwire.MaxPayload) + if err == nil { + err = s.dispatch(ctx, f) + } + switch { + case err == nil: + case ctx.Err() != nil: + return ctx.Err() + case errors.Is(err, io.EOF): + return nil + default: + return err + } + } +} + +type server struct { + conn io.ReadWriteCloser + svc Service + a Attachment + seq sandboxwire.RequestSequence // read loop only + wmu sync.Mutex + wg sync.WaitGroup + + mu sync.Mutex + inflight map[uint64]context.CancelFunc // running handlers, for CancelRequest + held int // admitted requests whose response is not yet written +} + +// dispatch starts one request. A frame that is not a request, or whose +// RequestID does not increase, ends the stream. +func (s *server) dispatch(ctx context.Context, f sandboxwire.Frame) error { + kind, err := tags.Classify(f.Type) + if err != nil { + return err + } + if kind != sandboxwire.KindRequest { + return malformed("message type %#04x from the client", f.Type) + } + if !s.seq.Admit(f.RequestID) { + return malformed("request ID %d does not increase", f.RequestID) + } + op := Op(f.Type) + req, err := decodeRequest(op, f.Payload) + s.mu.Lock() + switch { + case err != nil: + s.mu.Unlock() + return s.reply(f.RequestID, op, nil, NewFailure(CodeInvalidArgument, sandboxwire.EffectNone, err.Error())) + case op == OpCancelRequest: + if cancel := s.inflight[req.(*CancelRequestRequest).Target]; cancel != nil { + cancel() + } + s.mu.Unlock() + return s.reply(f.RequestID, op, &CancelRequestResponse{}, nil) + case s.held >= MaxInFlight: + s.mu.Unlock() + return s.reply(f.RequestID, op, nil, NewFailure(CodeResourceExhausted, sandboxwire.EffectNone, "too many requests in flight")) + } + rctx, cancel := context.WithCancel(ctx) + s.inflight[f.RequestID] = cancel + s.held++ + s.mu.Unlock() + s.wg.Add(1) + go func() { + defer s.wg.Done() + resp, err := opSpecs[op].serve(rctx, s.svc, s.a, req) + s.mu.Lock() + delete(s.inflight, f.RequestID) + s.mu.Unlock() + cancel() + var fail *Failure + if err != nil && !errors.As(err, &fail) { + fail = NewFailure(CodeUnknown, sandboxwire.EffectPossible, err.Error()) + } + // The request keeps its slot until its response is written, so a + // client that stops reading stops admission instead of piling up + // finished requests. + werr := s.reply(f.RequestID, op, resp, fail) + s.mu.Lock() + s.held-- + s.mu.Unlock() + if werr != nil { + s.conn.Close() + } + }() + return nil +} + +// reply writes a response. A response the service built wrongly becomes +// Unknown with EffectPossible, since the request may have run. +func (s *server) reply(id uint64, op Op, resp message, fail *Failure) error { + payload, err := encodeResponse(resp, fail) + if err != nil { + payload, _ = encodeResponse(nil, NewFailure(CodeUnknown, sandboxwire.EffectPossible, "service response: "+err.Error())) + } + s.wmu.Lock() + defer s.wmu.Unlock() + return sandboxwire.WriteFrame(s.conn, sandboxwire.Frame{Type: sandboxwire.ResponseType(uint16(op)), RequestID: id, Payload: payload}) +} diff --git a/internal/sandboxfs/server_test.go b/internal/sandboxfs/server_test.go new file mode 100644 index 00000000..f397aad7 --- /dev/null +++ b/internal/sandboxfs/server_test.go @@ -0,0 +1,83 @@ +package sandboxfs + +import ( + "context" + "errors" + "net" + "sync/atomic" + "testing" + "time" + + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxlink" + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxwire" +) + +// describer answers Describe and counts the calls. +type describer struct { + Service + calls atomic.Int32 +} + +func (d *describer) Describe(context.Context, Attachment, *DescribeRequest) (*DescribeResponse, error) { + d.calls.Add(1) + return &DescribeResponse{ServerInstanceID: testInstance, Capabilities: testCaps}, nil +} + +// A request holds its slot until its response is written, so a client that +// stops reading responses stops admission. +func TestUnreadResponsesStopAdmission(t *testing.T) { + cc, sc := net.Pipe() + svc := &describer{} + served := make(chan error, 1) + a := Attachment{ID: sandboxwire.NewID(), ServerInstanceID: testInstance, Lease: context.Background(), Exports: []sandboxlink.ExportGrant{{ID: "world"}}} + go func() { served <- Serve(context.Background(), sc, svc, a) }() + var read atomic.Int32 // net.Pipe completes a write once the server has read it + go func() { + for id := uint64(1); id <= 2*MaxInFlight; id++ { + if sandboxwire.WriteFrame(cc, sandboxwire.Frame{Type: uint16(OpDescribe), RequestID: id}) != nil { + return + } + read.Add(1) + } + }() + // MaxInFlight requests run; the next one blocks the server writing its + // ResourceExhausted answer, so nothing further is read. + settled := func() bool { return svc.calls.Load() == MaxInFlight && read.Load() == MaxInFlight+1 } + for deadline := time.Now().Add(5 * time.Second); !settled() && time.Now().Before(deadline); { + time.Sleep(time.Millisecond) + } + time.Sleep(100 * time.Millisecond) + if !settled() { + t.Fatalf("%d requests served and %d read while no response was read", svc.calls.Load(), read.Load()) + } + cc.Close() + <-served +} + +// A RequestID that does not increase ends the stream without dispatching. +func TestRequestIDsMustIncrease(t *testing.T) { + cc, sc := net.Pipe() + defer cc.Close() + svc := &describer{} + served := make(chan error, 1) + a := Attachment{ID: sandboxwire.NewID(), ServerInstanceID: testInstance, Lease: context.Background(), Exports: []sandboxlink.ExportGrant{{ID: "world"}}} + go func() { served <- Serve(context.Background(), sc, svc, a) }() + describe := sandboxwire.Frame{Type: uint16(OpDescribe), RequestID: 5} + if err := sandboxwire.WriteFrame(cc, describe); err != nil { + t.Fatal(err) + } + if _, err := sandboxwire.ReadFrame(cc, sandboxwire.MaxPayload); err != nil { + t.Fatal(err) + } + if err := sandboxwire.WriteFrame(cc, describe); err != nil { + t.Fatal(err) + } + select { + case err := <-served: + if !errors.Is(err, sandboxwire.ErrMalformed) || svc.calls.Load() != 1 { + t.Fatalf("Serve returned %v after %d calls", err, svc.calls.Load()) + } + case <-time.After(5 * time.Second): + t.Fatalf("a repeated RequestID kept the stream after %d calls", svc.calls.Load()) + } +} diff --git a/internal/sandboxfs/testdata/create_request.hex b/internal/sandboxfs/testdata/create_request.hex new file mode 100644 index 00000000..ff011c93 --- /dev/null +++ b/internal/sandboxfs/testdata/create_request.hex @@ -0,0 +1,11 @@ +# Exclusive Create of an append-mode file. +00000027 # PayloadLength 39 +000a # MessageType: Create (10) +0000 # Flags +0000000000000003 # RequestID 3 +0000000000000001 0000000000000007 # Parent: ID 1, Generation 7 +00000008 6e6f7465732e6d64 # Name "notes.md" +000001a4 # Mode 0644 +0003 # Access: ReadWrite +00000001 # Flags: OpenAppend +01 # Exclusive diff --git a/internal/sandboxfs/testdata/describe_response.hex b/internal/sandboxfs/testdata/describe_response.hex new file mode 100644 index 00000000..a0b444f5 --- /dev/null +++ b/internal/sandboxfs/testdata/describe_response.hex @@ -0,0 +1,28 @@ +# Describe response. Hex bytes; text after # is a comment. +00000057 # PayloadLength 87 +8001 # MessageType: response to Describe (1) +0000 # Flags +0000000000000001 # RequestID 1 +0001 # Result: success +00112233445566778899aabbccddeeff # ServerInstanceID +000003e8 # Identity.UID 1000 +000003e8 # Identity.GID 1000 +0001 # PathProfile: LinuxBytes +0001 # CacheProfile: Uncached +0001 # Durability: FsyncRequired +000000ff # MaxNameBytes 255 +00000fff # MaxPathBytes 4095 +00010000 # MaxReadBytes 65536 +00010000 # MaxWriteBytes 65536 +00000100 # MaxWalkComponents 256 +00010000 # MaxReadDirBytes 65536 +00001000 # MaxOpenHandles 4096 +00 # ReadOnly +01 01 # AtomicAppend, AtomicRename +01 01 # RenameNoReplace, RenameExchange +01 01 # HardLinks, Symlinks +01 01 01 # SetMode, SetOwner, SetTimes +01 01 # DirectoryFsync, ReadDirPlus +01 00 # Flock, POSIXLocks +00000001 # Exports: count 1 +00000005 776f726c64 # "world" diff --git a/internal/sandboxfs/testdata/failure_response.hex b/internal/sandboxfs/testdata/failure_response.hex new file mode 100644 index 00000000..29c2a443 --- /dev/null +++ b/internal/sandboxfs/testdata/failure_response.hex @@ -0,0 +1,10 @@ +# Lookup failure: the entry does not exist. +00000014 # PayloadLength 20 +8004 # MessageType: response to Lookup (4) +0000 # Flags +0000000000000007 # RequestID 7 +0002 # Result: failure +000b # Code: Errno +01 0003 # Errno: present, NotFound +0001 # Effect: None +00000007 6d697373696e67 # Message "missing" diff --git a/internal/sandboxfs/testdata/readdir_request.hex b/internal/sandboxfs/testdata/readdir_request.hex new file mode 100644 index 00000000..639fcdd9 --- /dev/null +++ b/internal/sandboxfs/testdata/readdir_request.hex @@ -0,0 +1,9 @@ +# ReadDir resuming after a cookie. +00000015 # PayloadLength 21 +0011 # MessageType: ReadDir (17) +0000 # Flags +0000000000000005 # RequestID 5 +0000000000000042 # Handle +1c6a3e5f0b9d2471 # Cookie +00010000 # Limit 65536 +00 # WithAttrs diff --git a/internal/sandboxfs/testdata/readdir_response.hex b/internal/sandboxfs/testdata/readdir_response.hex new file mode 100644 index 00000000..be457520 --- /dev/null +++ b/internal/sandboxfs/testdata/readdir_response.hex @@ -0,0 +1,18 @@ +# ReadDir page of two entries, not the end. +00000041 # PayloadLength 65 +8011 # MessageType: response to ReadDir (17) +0000 # Flags +0000000000000005 # RequestID 5 +0001 # Result: success +00000002 # Entries: count 2 +00000005 612e747874 # Name "a.txt" +0000000000000103 # Ino +00008000 # Type: regular +2f0e4b6c7d8a9b10 # Cookie +00 # Entry: absent +00000003 737263 # Name "src" +0000000000000104 # Ino +00004000 # Type: directory +3a5c7e9f1b2d4f60 # Cookie +00 # Entry: absent +00 # End: false diff --git a/internal/sandboxfs/testdata/rename_request.hex b/internal/sandboxfs/testdata/rename_request.hex new file mode 100644 index 00000000..5c8babda --- /dev/null +++ b/internal/sandboxfs/testdata/rename_request.hex @@ -0,0 +1,10 @@ +# Rename exchanging two entries. +00000030 # PayloadLength 48 +0016 # MessageType: Rename (22) +0000 # Flags +0000000000000006 # RequestID 6 +0000000000000001 0000000000000007 # Parent: ID 1, Generation 7 +00000003 6f6c64 # Name "old" +0000000000000002 0000000000000009 # NewParent: ID 2, Generation 9 +00000003 6e6577 # NewName "new" +0003 # Mode: RenameExchange diff --git a/internal/sandboxfs/testdata/walk_request.hex b/internal/sandboxfs/testdata/walk_request.hex new file mode 100644 index 00000000..4cbbfca0 --- /dev/null +++ b/internal/sandboxfs/testdata/walk_request.hex @@ -0,0 +1,9 @@ +# Walk request. +00000021 # PayloadLength 33 +0005 # MessageType: Walk (5) +0000 # Flags +0000000000000002 # RequestID 2 +0000000000000001 0000000000000007 # Parent: ID 1, Generation 7 +00000002 # Names: count 2 +00000004 6c696e6b # "link" +00000001 78 # "x" diff --git a/internal/sandboxfs/testdata/walk_response.hex b/internal/sandboxfs/testdata/walk_response.hex new file mode 100644 index 00000000..5c76488d --- /dev/null +++ b/internal/sandboxfs/testdata/walk_response.hex @@ -0,0 +1,20 @@ +# Walk response that stopped at the symlink "link": one entry, no failure. +0000006f # PayloadLength 111 +8005 # MessageType: response to Walk (5) +0000 # Flags +0000000000000002 # RequestID 2 +0001 # Result: success +00000001 # Entries: count 1 +0000000000000002 0000000000000007 # Node: ID 2, Generation 7 +0000000000000102 # Attr.Ino +0000a1ff # Attr.Mode: symlink 0777 +00000001 # Attr.Nlink +000003e8 000003e8 # Attr.UID, Attr.GID +0000000000000000 # Attr.Rdev +0000000000000001 # Attr.Size +0000000000000000 # Attr.Blocks +00001000 # Attr.Blksize +0000000065000000 00000001 # Attr.Atime +0000000065000000 00000002 # Attr.Mtime +0000000065000000 00000003 # Attr.Ctime +00 # Failure: absent diff --git a/internal/sandboxfs/testdata/write_response.hex b/internal/sandboxfs/testdata/write_response.hex new file mode 100644 index 00000000..e78d62c5 --- /dev/null +++ b/internal/sandboxfs/testdata/write_response.hex @@ -0,0 +1,12 @@ +# Short write: 4096 bytes written, then the device filled. +0000001b # PayloadLength 27 +800c # MessageType: response to Write (12) +0000 # Flags +0000000000000004 # RequestID 4 +0001 # Result: success +00001000 # Written 4096 +01 # Failure: present +000b # Code: Errno +01 000b # Errno: present, NoSpace +0001 # Effect: None +00000009 6469736b2066756c6c # Message "disk full"