diff --git a/AGENTS.md b/AGENTS.md index 7099527d..c752b094 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -23,6 +23,7 @@ OpenAgentCore is protocol-first and modular. Core orchestrates operations that p | Provider–Runtime startup | `internal/runtimebootstrap/bootstrap.go` | [Runtime bootstrap](docs/runtime-bootstrap.md) | | Runtime and Sandbox I/O service–relay (Link) | `internal/sandboxlink/protocol.go` | [Sandbox link protocol](docs/sandbox-link-protocol.md) | | Provider–Sandbox I/O startup | `internal/sandboxbootstrap/bootstrap.go` | [Sandbox bootstrap](docs/sandbox-bootstrap.md) | +| Runtime–file service | `internal/sandboxfs/protocol.go` | [File access protocol](docs/file-access-protocol.md) | | Core–Runtime wire | `internal/agentdaemon/proto/` | [Core–Runtime protocol](docs/runtime-protocol.md) | | Runtime–Harness | `apps/daemon/internal/agent/harness.go` | [Harness onboarding](contracts/agents-api/harness-onboarding.md) | | Harness–Model provider | `internal/modelprovider/config.go` | [Model execution](contracts/agents-api/model-execution.md) | diff --git a/apps/sandboxio/internal/fileservice/errors.go b/apps/sandboxio/internal/fileservice/errors.go new file mode 100644 index 00000000..9507a650 --- /dev/null +++ b/apps/sandboxio/internal/fileservice/errors.go @@ -0,0 +1,123 @@ +//go:build linux + +package fileservice + +import ( + "errors" + "os" + "strconv" + + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxfs" + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxwire" + "golang.org/x/sys/unix" +) + +var errnos = map[unix.Errno]sandboxfs.Errno{ + unix.EACCES: sandboxfs.ErrnoPermissionDenied, + unix.EPERM: sandboxfs.ErrnoOperationNotPermitted, + unix.ENOENT: sandboxfs.ErrnoNotFound, + unix.EEXIST: sandboxfs.ErrnoExists, + unix.ENOTDIR: sandboxfs.ErrnoNotDirectory, + unix.EISDIR: sandboxfs.ErrnoIsDirectory, + unix.ENOTEMPTY: sandboxfs.ErrnoDirectoryNotEmpty, + unix.EINVAL: sandboxfs.ErrnoInvalidArgument, + unix.EBADF: sandboxfs.ErrnoBadDescriptor, + unix.EMFILE: sandboxfs.ErrnoTooManyOpenFiles, + unix.ENFILE: sandboxfs.ErrnoTooManyOpenFiles, + unix.ENOSPC: sandboxfs.ErrnoNoSpace, + unix.EDQUOT: sandboxfs.ErrnoQuotaExceeded, + unix.EROFS: sandboxfs.ErrnoReadOnlyFilesystem, + unix.EXDEV: sandboxfs.ErrnoCrossDevice, + unix.ENAMETOOLONG: sandboxfs.ErrnoNameTooLong, + unix.ELOOP: sandboxfs.ErrnoSymlinkLoop, + unix.EFBIG: sandboxfs.ErrnoFileTooLarge, + unix.EOVERFLOW: sandboxfs.ErrnoOverflow, + unix.EBUSY: sandboxfs.ErrnoBusy, + unix.ETXTBSY: sandboxfs.ErrnoBusy, + unix.EAGAIN: sandboxfs.ErrnoAgain, + unix.EINTR: sandboxfs.ErrnoInterrupted, + unix.EIO: sandboxfs.ErrnoIO, + unix.ENODEV: sandboxfs.ErrnoNoDevice, + unix.ENXIO: sandboxfs.ErrnoNoSuchDeviceOrAddress, + unix.EPIPE: sandboxfs.ErrnoBrokenPipe, + unix.EOPNOTSUPP: sandboxfs.ErrnoNotSupported, + unix.ENOSYS: sandboxfs.ErrnoNotSupported, + unix.ENOLCK: sandboxfs.ErrnoNoLocks, + unix.EDEADLK: sandboxfs.ErrnoDeadlock, +} + +// failure converts err to the *Failure a Service method returns. A Linux +// errno without an entry becomes IO. EffectPossible overrides the EffectNone +// of a typed failure, because a step before it may have changed state. +func failure(err error, effect sandboxwire.Effect) error { + var f *sandboxfs.Failure + if errors.As(err, &f) { + if effect == sandboxwire.EffectPossible && f.Effect != effect { + promoted := *f + promoted.Effect = effect + return &promoted + } + return f + } + var e unix.Errno + if errors.As(err, &e) { + errno, ok := errnos[e] + if !ok { + errno = sandboxfs.ErrnoIO + } + return sandboxfs.NewErrnoFailure(errno, effect, e.Error()) + } + return sandboxfs.NewErrnoFailure(sandboxfs.ErrnoIO, effect, err.Error()) +} + +func unsupported(what string) error { + return sandboxfs.NewFailure(sandboxfs.CodeUnsupported, sandboxwire.EffectNone, what+" is not supported") +} + +func errnoFailure(errno sandboxfs.Errno, message string) error { + return sandboxfs.NewErrnoFailure(errno, sandboxwire.EffectNone, message) +} + +// use runs fn with f's descriptor. f cannot be closed while fn runs, so the +// descriptor number is never reused under fn. A closed f yields stale(). +func use(f *os.File, stale func() error, fn func(fd int) error) error { + rc, err := f.SyscallConn() + if err != nil { + return stale() + } + var inner error + if err := rc.Control(func(fd uintptr) { inner = fn(int(fd)) }); err != nil { + return stale() + } + return inner +} + +// viaProc runs fn with the /proc/self/fd directory and fd's name in it; fd +// is always a descriptor the service opened and holds. Resolving that name +// reaches the object fd refers to, even after it was renamed or unlinked, +// and stops at a symlink or proc magic link the object is. +func (s *Service) viaProc(fd int, fn func(dir int, name string) error) error { + return use(s.proc, errStaleAttachment, func(dir int) error { return fn(dir, strconv.Itoa(fd)) }) +} + +// eintr retries a system call interrupted by a signal. +func eintr[T any](call func() (T, error)) (T, error) { + for { + v, err := call() + if err != unix.EINTR { + return v, err + } + } +} + +func openat(dir int, name string, flags int, mode uint32) (int, error) { + return eintr(func() (int, error) { return unix.Openat(dir, name, flags|unix.O_CLOEXEC, mode) }) +} + +func openFile(dir int, name string, flags int) (*os.File, error) { + fd, err := openat(dir, name, flags, 0) + if err != nil { + return nil, err + } + return os.NewFile(uintptr(fd), name), nil +} diff --git a/apps/sandboxio/internal/fileservice/files.go b/apps/sandboxio/internal/fileservice/files.go new file mode 100644 index 00000000..594e7121 --- /dev/null +++ b/apps/sandboxio/internal/fileservice/files.go @@ -0,0 +1,454 @@ +//go:build linux + +package fileservice + +import ( + "context" + "math" + "os" + "time" + + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxfs" + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxwire" + "golang.org/x/sys/unix" +) + +// openFlags converts an access mode and open flags. The service never opens +// a controlling terminal and never follows a symlink it opens. +func openFlags(access sandboxfs.AccessMode, flags sandboxfs.OpenFlags) int { + f := map[sandboxfs.AccessMode]int{sandboxfs.AccessRead: unix.O_RDONLY, sandboxfs.AccessWrite: unix.O_WRONLY, sandboxfs.AccessReadWrite: unix.O_RDWR}[access] + f |= unix.O_NOCTTY + for bit, o := range map[sandboxfs.OpenFlags]int{ + sandboxfs.OpenAppend: unix.O_APPEND, sandboxfs.OpenTruncate: unix.O_TRUNC, sandboxfs.OpenNoFollow: unix.O_NOFOLLOW, + sandboxfs.OpenSync: unix.O_SYNC, sandboxfs.OpenDataSync: unix.O_DSYNC, + } { + if flags&bit != 0 { + f |= o + } + } + return f +} + +// openable reports why a node of type typ cannot be opened as a file. +func openable(typ uint32) error { + switch typ { + case sandboxfs.ModeRegular: + return nil + case sandboxfs.ModeDirectory: + return unix.EISDIR + case sandboxfs.ModeSymlink: + return unix.ELOOP + } + return unsupported("opening a special file") +} + +// reopen opens the object of fd, a descriptor the service opened itself, +// through its /proc/self/fd name, after checking that the object has type +// typ. That name resolves to the object fd holds, so reopening never reaches +// a symlink's target. +func (s *Service) reopen(fd int, typ uint32, flags int) (fd2 int, err error) { + var sb unix.Stat_t + if err := unix.Fstat(fd, &sb); err != nil { + return -1, err + } + switch got := sb.Mode & sandboxfs.ModeType; { + case got == typ: + case typ == sandboxfs.ModeDirectory: + return -1, unix.ENOTDIR + default: + return -1, openable(got) + } + err = s.viaProc(fd, func(dir int, name string) (err error) { + fd2, err = openat(dir, name, flags&^unix.O_NOFOLLOW, 0) + return err + }) + return fd2, err +} + +func (s *Service) Open(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.OpenRequest) (*sandboxfs.OpenResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + n, err := st.node(r.Node) + if err != nil { + return nil, err + } + if err := openable(n.typ); err != nil { + return nil, failure(err, none) + } + if err := st.reserve(); err != nil { + return nil, err + } + var fd int + err = use(n.f, errStaleNode, func(pfd int) (err error) { + fd, err = s.reopen(pfd, sandboxfs.ModeRegular, openFlags(r.Access, r.Flags)) + return err + }) + if err != nil { + st.unreserve() + return nil, failure(err, none) + } + id, err := st.addHandle(&handle{f: os.NewFile(uintptr(fd), ""), append: r.Flags&sandboxfs.OpenAppend != 0}) + if err != nil { + return nil, failure(err, possible) + } + return &sandboxfs.OpenResponse{Handle: id}, nil +} + +// Create creates the entry with O_EXCL first, so it never opens a special +// file or follows a symlink. Without Exclusive an existing regular file is +// opened instead. +func (s *Service) Create(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.CreateRequest) (*sandboxfs.CreateResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + parent, err := st.dir(r.Parent) + if err != nil { + return nil, err + } + if err := st.reserve(); err != nil { + return nil, err + } + flags := openFlags(r.Access, r.Flags) | unix.O_NOFOLLOW + effect := none + var fd, pathFD int + var sb unix.Stat_t + err = use(parent.f, errStaleNode, func(dir int) error { + var err error + for range 3 { + fd, err = openat(dir, string(r.Name), flags|unix.O_CREAT|unix.O_EXCL, r.Mode) + if err != unix.EEXIST || r.Exclusive { + break + } + if fd, err = s.openExisting(dir, r.Name, flags); err != unix.ENOENT { + break + } + } + if err != nil { + return err + } + effect = possible + if pathFD, err = s.reopen(fd, sandboxfs.ModeRegular, unix.O_PATH); err == nil { + if err = unix.Fstat(pathFD, &sb); err != nil { + unix.Close(pathFD) + } + } + if err != nil { + unix.Close(fd) + } + return err + }) + if err != nil { + st.unreserve() + return nil, failure(err, effect) + } + n, err := st.addNode(pathFD, &sb) + if err != nil { + st.unreserve() + unix.Close(fd) + return nil, failure(err, possible) + } + id, err := st.addHandle(&handle{f: os.NewFile(uintptr(fd), ""), append: r.Flags&sandboxfs.OpenAppend != 0}) + if err != nil { + return nil, failure(err, possible) + } + return &sandboxfs.CreateResponse{Entry: sandboxfs.Entry{Node: n.ref, Attr: st.attr(&sb)}, Handle: id}, nil +} + +// openExisting opens an existing regular file without following a symlink. +func (s *Service) openExisting(dir int, name []byte, flags int) (int, error) { + pfd, err := openat(dir, string(name), unix.O_PATH|unix.O_NOFOLLOW, 0) + if err != nil { + return -1, err + } + defer unix.Close(pfd) + return s.reopen(pfd, sandboxfs.ModeRegular, flags) +} + +// fileHandle returns a handle that must not be a directory handle. +func (st *state) fileHandle(id sandboxfs.HandleID, dirErr error) (*handle, error) { + h, err := st.handle(id) + if err != nil { + return nil, err + } + if h.dir != nil { + return nil, failure(dirErr, none) + } + return h, nil +} + +func (s *Service) Read(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.ReadRequest) (*sandboxfs.ReadResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + h, err := st.fileHandle(r.Handle, unix.EISDIR) + if err != nil { + return nil, err + } + buf := make([]byte, r.Size) + total := 0 + err = use(h.f, errStaleHandle, func(fd int) error { + for total < len(buf) { + n, err := eintr(func() (int, error) { return unix.Pread(fd, buf[total:], int64(r.Offset)+int64(total)) }) + if err != nil { + return err + } + if n == 0 { + break + } + total += n + } + return nil + }) + if err != nil { + return nil, failure(err, none) + } + return &sandboxfs.ReadResponse{Data: buf[:total]}, nil +} + +// Write writes at the offset, or appends atomically with one write(2) on an +// append handle. It reports the written prefix and the error that stopped it. +func (s *Service) Write(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.WriteRequest) (*sandboxfs.WriteResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + h, err := st.fileHandle(r.Handle, unix.EISDIR) + if err != nil { + return nil, err + } + if !h.append && r.Offset > math.MaxInt64-uint64(len(r.Data)) { + return nil, sandboxfs.NewFailure(sandboxfs.CodeInvalidArgument, none, "write ends beyond 2^63-1") + } + total := 0 + err = use(h.f, errStaleHandle, func(fd int) error { + if h.append { + n, err := eintr(func() (int, error) { return unix.Write(fd, r.Data) }) + total = max(n, 0) + return err + } + for total < len(r.Data) { + n, err := eintr(func() (int, error) { return unix.Pwrite(fd, r.Data[total:], int64(r.Offset)+int64(total)) }) + if err != nil { + return err + } + if n == 0 { + break + } + total += n + } + return nil + }) + switch { + case err == nil: + return &sandboxfs.WriteResponse{Written: uint32(total)}, nil + case total == 0: + return nil, failure(err, none) + } + return &sandboxfs.WriteResponse{Written: uint32(total), Failure: failure(err, none).(*sandboxfs.Failure)}, nil +} + +// Flush closes a duplicate of the handle's descriptor, which is what a close +// of one descriptor does to the file. +func (s *Service) Flush(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.FlushRequest) (*sandboxfs.FlushResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + h, err := st.fileHandle(r.Handle, unix.EBADF) + if err != nil { + return nil, err + } + effect := none + err = use(h.f, errStaleHandle, func(fd int) error { + dup, err := unix.FcntlInt(uintptr(fd), unix.F_DUPFD_CLOEXEC, 0) + if err != nil { + return err + } + effect = possible + return unix.Close(dup) + }) + if err != nil { + return nil, failure(err, effect) + } + return &sandboxfs.FlushResponse{}, nil +} + +func (s *Service) Fsync(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.FsyncRequest) (*sandboxfs.FsyncResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + h, err := st.handle(r.Handle) + if err != nil { + return nil, err + } + sync := unix.Fsync + if r.DataOnly { + sync = unix.Fdatasync + } + effect := none + err = use(h.f, errStaleHandle, func(fd int) error { + effect = possible + _, err := eintr(func() (struct{}, error) { return struct{}{}, sync(fd) }) + return err + }) + if err != nil { + return nil, failure(err, effect) + } + return &sandboxfs.FsyncResponse{}, nil +} + +func (s *Service) Release(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.ReleaseRequest) (*sandboxfs.ReleaseResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + if err := st.release(r.Handle, false); err != nil { + return nil, failure(err, none) + } + return &sandboxfs.ReleaseResponse{}, nil +} + +func (s *Service) OpenDir(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.OpenDirRequest) (*sandboxfs.OpenDirResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + n, err := st.dir(r.Node) + if err != nil { + return nil, err + } + if err := st.reserve(); err != nil { + return nil, err + } + var fd int + var sb unix.Stat_t + err = use(n.f, errStaleNode, func(pfd int) (err error) { + if err = unix.Fstat(pfd, &sb); err == nil { + fd, err = s.reopen(pfd, sandboxfs.ModeDirectory, unix.O_RDONLY|unix.O_DIRECTORY) + } + return err + }) + if err != nil { + st.unreserve() + return nil, failure(err, none) + } + id, err := st.addHandle(&handle{f: os.NewFile(uintptr(fd), ""), dir: &cursor{dev: uint64(sb.Dev)}}) + if err != nil { + return nil, err + } + return &sandboxfs.OpenDirResponse{Handle: id}, nil +} + +func (s *Service) ReadDir(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.ReadDirRequest) (*sandboxfs.ReadDirResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + h, err := st.handle(r.Handle) + if err != nil { + return nil, err + } + if h.dir == nil { + return nil, failure(unix.ENOTDIR, none) + } + var resp *sandboxfs.ReadDirResponse + err = use(h.f, errStaleHandle, func(fd int) (err error) { + resp, err = h.dir.read(st, fd, r) + return err + }) + if err != nil { + return nil, failure(err, none) + } + return resp, nil +} + +func (s *Service) ReleaseDir(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.ReleaseDirRequest) (*sandboxfs.ReleaseDirResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + if err := st.release(r.Handle, true); err != nil { + return nil, failure(err, none) + } + return &sandboxfs.ReleaseDirResponse{}, nil +} + +// GetLock queries POSIX locks, which this service does not declare. +func (s *Service) GetLock(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.GetLockRequest) (*sandboxfs.GetLockResponse, error) { + if _, err := s.enter(a, r); err != nil { + return nil, err + } + return nil, unsupported("POSIX locks") +} + +// SetLock takes flock locks on the handle's open file description, the same +// lock native flock(2) takes, so both exclude each other. A waiting request +// polls, holding the handle's lock state only during each attempt, so other +// requests on the handle proceed and cancelling the context ends the wait. +func (s *Service) SetLock(ctx context.Context, a sandboxfs.Attachment, r *sandboxfs.SetLockRequest) (*sandboxfs.SetLockResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + h, err := st.handle(r.Handle) + if err != nil { + return nil, err + } + effect := none + for delay := time.Millisecond; ; delay = min(2*delay, 50*time.Millisecond) { + if err := ctx.Err(); err != nil { + return nil, contextFailure(err, effect) + } + dropped, err := h.tryLock(r.Lock.Mode) + if dropped { + effect = possible + } + switch { + case err == nil: + return &sandboxfs.SetLockResponse{}, nil + case err != unix.EWOULDBLOCK || !r.Wait: + return nil, failure(err, effect) + } + wait := time.NewTimer(delay) + select { + case <-ctx.Done(): + wait.Stop() + case <-wait.C: + } + } +} + +var flockOps = map[sandboxfs.LockMode]int{sandboxfs.LockRead: unix.LOCK_SH, sandboxfs.LockWrite: unix.LOCK_EX, sandboxfs.LockUnlock: unix.LOCK_UN} + +// tryLock makes one non-blocking flock attempt. Converting a held lock drops +// it before taking the new one, so a failed conversion leaves the handle +// unlocked and reports dropped. +func (h *handle) tryLock(mode sandboxfs.LockMode) (dropped bool, err error) { + h.lockMu.Lock() + defer h.lockMu.Unlock() + converting := h.flock != 0 && h.flock != mode && mode != sandboxfs.LockUnlock + err = use(h.f, errStaleHandle, func(fd int) error { return unix.Flock(fd, flockOps[mode]|unix.LOCK_NB) }) + switch { + case err == nil && mode == sandboxfs.LockUnlock: + h.flock = 0 + case err == nil: + h.flock = mode + case converting: + h.flock = 0 + return true, err + } + return false, err +} + +func contextFailure(err error, effect sandboxwire.Effect) error { + code := sandboxfs.CodeCancelled + if err == context.DeadlineExceeded { + code = sandboxfs.CodeDeadlineExceeded + } + return sandboxfs.NewFailure(code, effect, err.Error()) +} diff --git a/apps/sandboxio/internal/fileservice/namespace.go b/apps/sandboxio/internal/fileservice/namespace.go new file mode 100644 index 00000000..b73238a6 --- /dev/null +++ b/apps/sandboxio/internal/fileservice/namespace.go @@ -0,0 +1,442 @@ +//go:build linux + +package fileservice + +import ( + "context" + "errors" + "os" + "time" + + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxfs" + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxwire" + "golang.org/x/sys/unix" +) + +const ( + none = sandboxwire.EffectNone + possible = sandboxwire.EffectPossible +) + +// lookup opens name in directory dir without following a symlink and +// acquires one reference on its node. Names never contain "/" and are never +// "." or "..", so a lookup resolves exactly one entry of dir, and a symlink, +// including a proc magic link, is returned as itself. +func (st *state) lookup(dir int, name []byte) (*node, sandboxfs.Entry, error) { + fd, err := openat(dir, string(name), unix.O_PATH|unix.O_NOFOLLOW, 0) + if err != nil { + return nil, sandboxfs.Entry{}, err + } + var sb unix.Stat_t + if err := unix.Fstat(fd, &sb); err != nil { + unix.Close(fd) + return nil, sandboxfs.Entry{}, err + } + n, err := st.addNode(fd, &sb) + if err != nil { + return nil, sandboxfs.Entry{}, err + } + return n, sandboxfs.Entry{Node: n.ref, Attr: st.attr(&sb)}, nil +} + +func (st *state) lookupIn(parent *node, name []byte) (n *node, e sandboxfs.Entry, err error) { + err = use(parent.f, errStaleNode, func(dir int) error { + n, e, err = st.lookup(dir, name) + return err + }) + return n, e, err +} + +// withNode runs fn with the descriptor of the node ref names. +func (st *state) withNode(ref sandboxfs.NodeRef, fn func(n *node, fd int) error) error { + n, err := st.node(ref) + if err != nil { + return err + } + return use(n.f, errStaleNode, func(fd int) error { return fn(n, fd) }) +} + +// dir returns the node ref names, which must be a directory. A symlink is +// never a directory, so no request resolves through one. +func (st *state) dir(ref sandboxfs.NodeRef) (*node, error) { + n, err := st.node(ref) + if err != nil { + return nil, err + } + if n.typ != sandboxfs.ModeDirectory { + return nil, failure(unix.ENOTDIR, none) + } + return n, nil +} + +// withDir runs fn with the descriptor of the directory ref names. +func (st *state) withDir(ref sandboxfs.NodeRef, fn func(dir int) error) error { + n, err := st.dir(ref) + if err != nil { + return err + } + return use(n.f, errStaleNode, fn) +} + +func (s *Service) Lookup(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.LookupRequest) (*sandboxfs.LookupResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + parent, err := st.dir(r.Parent) + if err != nil { + return nil, err + } + _, e, err := st.lookupIn(parent, r.Name) + if err != nil { + return nil, failure(err, none) + } + return &sandboxfs.LookupResponse{Entry: e}, nil +} + +// Walk looks up each name in turn and stops after a symlink or at the first +// failure, which it reports after the entries already walked. +func (s *Service) Walk(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.WalkRequest) (*sandboxfs.WalkResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + cur, err := st.dir(r.Parent) + if err != nil { + return nil, err + } + resp := &sandboxfs.WalkResponse{} + for _, name := range r.Names { + n, e, err := st.lookupIn(cur, name) + if err != nil { + if len(resp.Entries) == 0 { + return nil, failure(err, none) + } + resp.Failure = failure(err, none).(*sandboxfs.Failure) + break + } + resp.Entries = append(resp.Entries, e) + if n.typ == sandboxfs.ModeSymlink { + break + } + cur = n + } + return resp, nil +} + +// target resolves a node or handle Target to its descriptor holder. typ is +// the node's type, or zero for a handle. +func (st *state) target(t sandboxfs.Target) (f *os.File, typ uint32, stale func() error, err error) { + if t.Kind == sandboxfs.TargetHandle { + h, err := st.handle(t.Handle) + if err != nil { + return nil, 0, nil, err + } + return h.f, 0, errStaleHandle, nil + } + n, err := st.node(t.Node) + if err != nil { + return nil, 0, nil, err + } + return n.f, n.typ, errStaleNode, nil +} + +func (s *Service) GetAttr(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.GetAttrRequest) (*sandboxfs.GetAttrResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + f, _, stale, err := st.target(r.Target) + if err != nil { + return nil, err + } + var sb unix.Stat_t + if err := use(f, stale, func(fd int) error { return unix.Fstat(fd, &sb) }); err != nil { + return nil, failure(err, none) + } + return &sandboxfs.GetAttrResponse{Attr: st.attr(&sb)}, nil +} + +// SetAttr applies the owner, then the mode, the size and the times. A +// failure after the first applied change carries EffectPossible. +func (s *Service) SetAttr(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.SetAttrRequest) (*sandboxfs.SetAttrResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + f, typ, stale, err := st.target(r.Target) + if err != nil { + return nil, err + } + isHandle := r.Target.Kind == sandboxfs.TargetHandle + applied := false + var sb unix.Stat_t + err = use(f, stale, func(fd int) error { + apply := func(err error) error { + applied = applied || err == nil + return err + } + if r.Set&(sandboxfs.AttrUID|sandboxfs.AttrGID) != 0 { + uid, gid := -1, -1 + if r.Set&sandboxfs.AttrUID != 0 { + uid = int(r.UID) + } + if r.Set&sandboxfs.AttrGID != 0 { + gid = int(r.GID) + } + if err := apply(unix.Fchownat(fd, "", uid, gid, unix.AT_EMPTY_PATH)); err != nil { + return err + } + } + if r.Set&sandboxfs.AttrMode != 0 { + var err error + switch { + case isHandle: + err = unix.Fchmod(fd, r.Mode) + case typ == sandboxfs.ModeSymlink: + err = unix.EOPNOTSUPP + default: + err = s.viaProc(fd, func(dir int, name string) error { return unix.Fchmodat(dir, name, r.Mode, 0) }) + } + if err := apply(err); err != nil { + return err + } + } + if r.Set&sandboxfs.AttrSize != 0 { + if err := apply(s.truncate(fd, isHandle, typ, int64(r.Size))); err != nil { + return err + } + } + if r.Set&(sandboxfs.AttrAtime|sandboxfs.AttrMtime|sandboxfs.AttrAtimeNow|sandboxfs.AttrMtimeNow) != 0 { + atime, aerr := timespec(r.Set, sandboxfs.AttrAtime, sandboxfs.AttrAtimeNow, r.Atime) + mtime, merr := timespec(r.Set, sandboxfs.AttrMtime, sandboxfs.AttrMtimeNow, r.Mtime) + if err := errors.Join(aerr, merr); err != nil { + return err + } + err := s.viaProc(fd, func(dir int, name string) error { + return unix.UtimesNanoAt(dir, name, []unix.Timespec{atime, mtime}, 0) + }) + if err := apply(err); err != nil { + return err + } + } + return unix.Fstat(fd, &sb) + }) + if err != nil { + effect := none + if applied { + effect = possible + } + return nil, failure(err, effect) + } + return &sandboxfs.SetAttrResponse{Attr: st.attr(&sb)}, nil +} + +func (s *Service) truncate(fd int, isHandle bool, typ uint32, size int64) error { + switch { + case isHandle: + _, err := eintr(func() (struct{}, error) { return struct{}{}, unix.Ftruncate(fd, size) }) + return err + case typ == sandboxfs.ModeDirectory: + return unix.EISDIR + case typ != sandboxfs.ModeRegular: + return unix.EINVAL + } + w, err := s.reopen(fd, sandboxfs.ModeRegular, unix.O_WRONLY|unix.O_NOCTTY) + if err != nil { + return err + } + defer unix.Close(w) + _, err = eintr(func() (struct{}, error) { return struct{}{}, unix.Ftruncate(w, size) }) + return err +} + +func timespec(set, value, now sandboxfs.AttrMask, t sandboxfs.Timestamp) (unix.Timespec, error) { + switch { + case set&value != 0: + return unix.TimeToTimespec(time.Unix(t.Sec, int64(t.Nsec))) + case set&now != 0: + return unix.Timespec{Nsec: unix.UTIME_NOW}, nil + } + return unix.Timespec{Nsec: unix.UTIME_OMIT}, nil +} + +func (s *Service) Access(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.AccessRequest) (*sandboxfs.AccessResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + var mode uint32 + for bit, m := range map[sandboxfs.AccessMask]uint32{sandboxfs.MayRead: unix.R_OK, sandboxfs.MayWrite: unix.W_OK, sandboxfs.MayExecute: unix.X_OK} { + if r.Mask&bit != 0 { + mode |= m + } + } + err = st.withNode(r.Node, func(_ *node, fd int) error { + return s.viaProc(fd, func(dir int, name string) error { return unix.Faccessat(dir, name, mode, 0) }) + }) + if err != nil { + return nil, failure(err, none) + } + return &sandboxfs.AccessResponse{}, nil +} + +// create runs make in the parent directory and then looks the new entry up. +// A failure after make succeeded carries EffectPossible. +func (st *state) create(parent sandboxfs.NodeRef, name []byte, make func(dir int) error) (sandboxfs.Entry, error) { + var e sandboxfs.Entry + made := false + err := st.withDir(parent, func(dir int) (err error) { + if err := make(dir); err != nil { + return err + } + made = true + _, e, err = st.lookup(dir, name) + return err + }) + if err != nil { + effect := none + if made { + effect = possible + } + return e, failure(err, effect) + } + return e, nil +} + +func (s *Service) Mkdir(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.MkdirRequest) (*sandboxfs.MkdirResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + e, err := st.create(r.Parent, r.Name, func(dir int) error { return unix.Mkdirat(dir, string(r.Name), r.Mode) }) + if err != nil { + return nil, err + } + return &sandboxfs.MkdirResponse{Entry: e}, nil +} + +func (s *Service) Symlink(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.SymlinkRequest) (*sandboxfs.SymlinkResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + e, err := st.create(r.Parent, r.Name, func(dir int) error { return unix.Symlinkat(string(r.Target), dir, string(r.Name)) }) + if err != nil { + return nil, err + } + return &sandboxfs.SymlinkResponse{Entry: e}, nil +} + +func (s *Service) Link(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.LinkRequest) (*sandboxfs.LinkResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + var e sandboxfs.Entry + err = st.withNode(r.Node, func(_ *node, fd int) error { + e, err = st.create(r.NewParent, r.NewName, func(dir int) error { + return s.viaProc(fd, func(proc int, name string) error { + return unix.Linkat(proc, name, dir, string(r.NewName), unix.AT_SYMLINK_FOLLOW) + }) + }) + return err + }) + if err != nil { + return nil, failure(err, none) + } + return &sandboxfs.LinkResponse{Entry: e}, nil +} + +func (s *Service) Unlink(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.UnlinkRequest) (*sandboxfs.UnlinkResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + if err := st.withDir(r.Parent, func(dir int) error { return unix.Unlinkat(dir, string(r.Name), 0) }); err != nil { + return nil, failure(err, none) + } + return &sandboxfs.UnlinkResponse{}, nil +} + +func (s *Service) Rmdir(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.RmdirRequest) (*sandboxfs.RmdirResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + if err := st.withDir(r.Parent, func(dir int) error { return unix.Unlinkat(dir, string(r.Name), unix.AT_REMOVEDIR) }); err != nil { + return nil, failure(err, none) + } + return &sandboxfs.RmdirResponse{}, nil +} + +var renameFlags = map[sandboxfs.RenameMode]uint{ + sandboxfs.RenameReplace: 0, + sandboxfs.RenameNoReplace: unix.RENAME_NOREPLACE, + sandboxfs.RenameExchange: unix.RENAME_EXCHANGE, +} + +func (s *Service) Rename(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.RenameRequest) (*sandboxfs.RenameResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + err = st.withDir(r.Parent, func(dir int) error { + return st.withDir(r.NewParent, func(newDir int) error { + return unix.Renameat2(dir, string(r.Name), newDir, string(r.NewName), renameFlags[r.Mode]) + }) + }) + if err != nil { + return nil, failure(err, none) + } + return &sandboxfs.RenameResponse{}, nil +} + +func (s *Service) Readlink(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.ReadlinkRequest) (*sandboxfs.ReadlinkResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + buf := make([]byte, s.caps.MaxPathBytes+1) + var n int + err = st.withNode(r.Node, func(nd *node, fd int) (err error) { + if nd.typ != sandboxfs.ModeSymlink { + return unix.EINVAL // as readlink(2) answers for a non-symlink + } + n, err = unix.Readlinkat(fd, "", buf) + return err + }) + switch { + case err != nil: + return nil, failure(err, none) + case n > int(s.caps.MaxPathBytes): + return nil, errnoFailure(sandboxfs.ErrnoNameTooLong, "symlink target exceeds MaxPathBytes") + } + return &sandboxfs.ReadlinkResponse{Target: buf[:n]}, nil +} + +func (s *Service) StatFS(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.StatFSRequest) (*sandboxfs.StatFSResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + var sf unix.Statfs_t + if err := st.withNode(r.Node, func(_ *node, fd int) error { return unix.Fstatfs(fd, &sf) }); err != nil { + return nil, failure(err, none) + } + return &sandboxfs.StatFSResponse{ + Blocks: sf.Blocks, BlocksFree: sf.Bfree, BlocksAvailable: sf.Bavail, Files: sf.Files, FilesFree: sf.Ffree, + BlockSize: uint32(sf.Bsize), FragmentSize: uint32(sf.Frsize), NameMax: uint32(sf.Namelen), + }, nil +} + +func (s *Service) Forget(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.ForgetRequest) (*sandboxfs.ForgetResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + if err := st.forget(r.Entries); err != nil { + return nil, err + } + return &sandboxfs.ForgetResponse{}, nil +} diff --git a/apps/sandboxio/internal/fileservice/readdir.go b/apps/sandboxio/internal/fileservice/readdir.go new file mode 100644 index 00000000..72f0ec65 --- /dev/null +++ b/apps/sandboxio/internal/fileservice/readdir.go @@ -0,0 +1,146 @@ +//go:build linux + +package fileservice + +import ( + "bytes" + "encoding/binary" + "sync" + + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxfs" + "golang.org/x/sys/unix" +) + +// cursor is a directory handle's getdents position. A cookie is the kernel's +// d_off of the entry it follows, so resuming at another cookie is one lseek. +type cursor struct { + dev uint64 // the directory's device, for composing entry inode numbers + + mu sync.Mutex + buf []byte + rest []byte // entries read from the kernel and not yet returned + pos uint64 // cookie of the last returned or skipped entry +} + +type rawEntry struct { + ino uint64 + off uint64 + typ uint8 + name []byte +} + +// next parses one linux_dirent64 from c.rest. +func (c *cursor) next() (rawEntry, []byte, error) { + const header = 19 // d_ino, d_off, d_reclen, d_type + if len(c.rest) < header { + return rawEntry{}, nil, unix.EIO + } + reclen := int(binary.NativeEndian.Uint16(c.rest[16:18])) + if reclen < header || reclen > len(c.rest) { + return rawEntry{}, nil, unix.EIO + } + name := c.rest[header:reclen] + if i := bytes.IndexByte(name, 0); i >= 0 { + name = name[:i] + } + e := rawEntry{ino: binary.NativeEndian.Uint64(c.rest[0:8]), off: binary.NativeEndian.Uint64(c.rest[8:16]), typ: c.rest[18], name: name} + return e, c.rest[reclen:], nil +} + +var direntTypes = map[uint8]uint32{ + unix.DT_REG: sandboxfs.ModeRegular, unix.DT_DIR: sandboxfs.ModeDirectory, unix.DT_LNK: sandboxfs.ModeSymlink, + unix.DT_FIFO: sandboxfs.ModeFIFO, unix.DT_SOCK: sandboxfs.ModeSocket, unix.DT_CHR: sandboxfs.ModeCharDevice, + unix.DT_BLK: sandboxfs.ModeBlockDevice, +} + +// read returns the entries after r.Cookie that fit r.Limit. It skips "." and +// "..", and entries removed before they could be described. A directory that +// changes while it is read can repeat a name or a cookie, so a page ends +// before a repeat. +func (c *cursor) read(st *state, fd int, r *sandboxfs.ReadDirRequest) (*sandboxfs.ReadDirResponse, error) { + c.mu.Lock() + defer c.mu.Unlock() + if r.Cookie != c.pos { + if _, err := unix.Seek(fd, int64(r.Cookie), unix.SEEK_SET); err != nil { + return nil, err + } + c.rest, c.pos = nil, r.Cookie + } + resp := &sandboxfs.ReadDirResponse{} + size := 0 + names, cookies := map[string]bool{}, map[uint64]bool{} + for { + if len(c.rest) == 0 { + if c.buf == nil { + c.buf = make([]byte, 32<<10) + } + n, err := eintr(func() (int, error) { return unix.Getdents(fd, c.buf) }) + if err != nil { + if len(resp.Entries) > 0 { + return resp, nil + } + return nil, err + } + if n == 0 { + resp.End = true + return resp, nil + } + c.rest = c.buf[:n] + } + raw, rest, err := c.next() + if err != nil { + c.rest = nil + return nil, err + } + skip := func() { c.rest, c.pos = rest, raw.off } + if string(raw.name) == "." || string(raw.name) == ".." || raw.typ == unix.DT_WHT { + skip() + continue + } + if names[string(raw.name)] || cookies[raw.off] { + return resp, nil + } + e := sandboxfs.DirEntry{Name: bytes.Clone(raw.name), Ino: st.ino(c.dev, raw.ino), Type: direntTypes[raw.typ], Cookie: raw.off} + if r.WithAttrs { + e.Entry = &sandboxfs.Entry{} + } + if size+e.WireSize() > int(r.Limit) { + if len(resp.Entries) == 0 { + return nil, unix.EINVAL + } + return resp, nil + } + switch { + case r.WithAttrs: + _, entry, err := st.lookup(fd, raw.name) + if err == unix.ENOENT { + skip() + continue + } + if err != nil { + if len(resp.Entries) > 0 { + return resp, nil + } + return nil, err + } + *e.Entry = entry + e.Ino, e.Type = entry.Attr.Ino, entry.Attr.Mode&sandboxfs.ModeType + case e.Type == 0: + var sb unix.Stat_t + if err := unix.Fstatat(fd, string(raw.name), &sb, unix.AT_SYMLINK_NOFOLLOW); err == unix.ENOENT { + skip() + continue + } else if err != nil { + if len(resp.Entries) > 0 { + return resp, nil + } + return nil, err + } + e.Type = sb.Mode & sandboxfs.ModeType + } + resp.Entries = append(resp.Entries, e) + names[string(e.Name)], cookies[e.Cookie] = true, true + size += e.WireSize() + skip() + } +} diff --git a/apps/sandboxio/internal/fileservice/service.go b/apps/sandboxio/internal/fileservice/service.go new file mode 100644 index 00000000..33c722ea --- /dev/null +++ b/apps/sandboxio/internal/fileservice/service.go @@ -0,0 +1,570 @@ +//go:build linux + +// Package fileservice is the sandbox file service: it implements +// sandboxfs.Service over the one export world with the Linux *at system +// calls. It acts as its own process identity and never impersonates. +package fileservice + +import ( + "context" + "errors" + "fmt" + "math/rand/v2" + "os" + "path/filepath" + "strconv" + "strings" + "sync" + + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxfs" + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxlink" + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxwire" + "golang.org/x/sys/unix" +) + +// Export is the one export the service serves. +const Export sandboxlink.ExportID = "world" + +// Service serves File requests for any number of attachments. Its node and +// handle tables live in memory, so each Service has its own ServerInstanceID. +type Service struct { + instance sandboxwire.ID + identity sandboxfs.Identity + caps sandboxfs.Capabilities + root *os.File // O_PATH directory of the export + rootDev uint64 + proc *os.File // /proc/self/fd, for reopening a held descriptor + fdinfo *os.File // /proc/self/fdinfo, for mount IDs statx does not report + + mu sync.Mutex + atts map[sandboxwire.ID]*state + closed bool +} + +// New serves the absolute directory root as the export world. oac-sandbox-io +// passes "/", the sandbox's root after the Provider's namespace setup; the +// caller owns the isolation of everything under root, because the service +// neither detects nor enforces a boundary inside it. New sets the process +// umask to zero, because the service applies every requested mode +// explicitly. +func New(root string) (*Service, error) { + if !filepath.IsAbs(root) { + return nil, fmt.Errorf("fileservice: root %q is not absolute", root) + } + unix.Umask(0) + s := &Service{ + instance: sandboxwire.NewID(), + identity: sandboxfs.Identity{UID: uint32(unix.Geteuid()), GID: uint32(unix.Getegid())}, + atts: map[sandboxwire.ID]*state{}, + } + for _, d := range []struct { + f **os.File + path string + }{{&s.proc, "/proc/self/fd"}, {&s.fdinfo, "/proc/self/fdinfo"}, {&s.root, root}} { + f, err := openFile(unix.AT_FDCWD, d.path, unix.O_PATH|unix.O_DIRECTORY) + if err != nil { + s.Close() + return nil, fmt.Errorf("fileservice: open %s: %w", d.path, err) + } + *d.f = f + } + var sb unix.Stat_t + if err := unix.Fstat(int(s.root.Fd()), &sb); err != nil { + s.Close() + return nil, fmt.Errorf("fileservice: root %s: %w", root, err) + } + s.rootDev = uint64(sb.Dev) + noReplace, exchange := probeRename() + s.caps = sandboxfs.Capabilities{ + PathProfile: sandboxfs.PathProfileLinuxBytes, + CacheProfile: sandboxfs.CacheProfileUncached, + Durability: sandboxfs.DurabilityFsyncRequired, + MaxNameBytes: 255, + MaxPathBytes: 4095, + MaxReadBytes: sandboxwire.MaxChunk, + MaxWriteBytes: sandboxwire.MaxChunk, + MaxWalkComponents: 256, + MaxReadDirBytes: 64 << 10, + MaxOpenHandles: 4096, + AtomicAppend: true, + AtomicRename: true, + RenameNoReplace: noReplace, + RenameExchange: exchange, + HardLinks: true, + Symlinks: true, + SetMode: true, + SetOwner: true, + SetTimes: true, + DirectoryFsync: true, + ReadDirPlus: true, + Flock: true, + } + return s, nil +} + +// probeRename reports which renameat2 modes the kernel supports, by trying +// them in a temporary directory. +func probeRename() (noReplace, exchange bool) { + dir, err := os.MkdirTemp("", "oac-fileservice-") + if err != nil { + return false, false + } + defer os.RemoveAll(dir) + a, b := filepath.Join(dir, "a"), filepath.Join(dir, "b") + if os.WriteFile(a, nil, 0o600) != nil || os.WriteFile(b, nil, 0o600) != nil { + return false, false + } + noReplace = unix.Renameat2(unix.AT_FDCWD, a, unix.AT_FDCWD, b, unix.RENAME_NOREPLACE) == unix.EEXIST + exchange = unix.Renameat2(unix.AT_FDCWD, a, unix.AT_FDCWD, b, unix.RENAME_EXCHANGE) == nil + return noReplace, exchange +} + +// InstanceID is the service incarnation. Link binds each stream to it, and +// the Attachment a stream passes to sandboxfs.Serve carries it. +func (s *Service) InstanceID() sandboxwire.ID { return s.instance } + +// Close releases every attachment and the export root. Requests still +// running fail as stale. +func (s *Service) Close() error { + s.mu.Lock() + s.closed = true + atts := s.atts + s.atts = map[sandboxwire.ID]*state{} + s.mu.Unlock() + for _, st := range atts { + st.close() + } + for _, f := range []*os.File{s.root, s.proc, s.fdinfo} { + if f != nil { + f.Close() + } + } + return nil +} + +var ( + errStaleAttachment = func() error { + return sandboxfs.NewFailure(sandboxfs.CodeStaleAttachment, sandboxwire.EffectNone, "attachment is not attached") + } + errStaleNode = func() error { + return sandboxfs.NewFailure(sandboxfs.CodeStaleNode, sandboxwire.EffectNone, "node is not held by this attachment") + } + errStaleHandle = func() error { + return sandboxfs.NewFailure(sandboxfs.CodeStaleHandle, sandboxwire.EffectNone, "handle is not open in this attachment") + } +) + +// check admits a request for attachment a without requiring it to be attached. +func (s *Service) check(a sandboxfs.Attachment) error { + if a.ServerInstanceID != s.instance { + return sandboxfs.NewFailure(sandboxfs.CodeInstanceChanged, sandboxwire.EffectNone, "stream is bound to another service instance") + } + if a.Lease == nil { + return errStaleAttachment() + } + select { + case <-a.Lease.Done(): + return errStaleAttachment() + default: + return nil + } +} + +// enter admits r for an attached attachment and returns its state. +func (s *Service) enter(a sandboxfs.Attachment, r sandboxfs.Request) (*state, error) { + if err := s.check(a); err != nil { + return nil, err + } + s.mu.Lock() + st := s.atts[a.ID] + s.mu.Unlock() + if st == nil { + return nil, errStaleAttachment() + } + if f := s.caps.Admit(r, st.readOnly); f != nil { + return nil, f + } + return st, nil +} + +// Describe lists world when the attachment is granted it. +func (s *Service) Describe(_ context.Context, a sandboxfs.Attachment, _ *sandboxfs.DescribeRequest) (*sandboxfs.DescribeResponse, error) { + if err := s.check(a); err != nil { + return nil, err + } + var exports []sandboxlink.ExportID + if _, ok := a.Grant(Export); ok { + exports = []sandboxlink.ExportID{Export} + } + return &sandboxfs.DescribeResponse{ServerInstanceID: s.instance, Identity: s.identity, Capabilities: s.caps, Exports: exports}, nil +} + +func (s *Service) Attach(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.AttachRequest) (*sandboxfs.AttachResponse, error) { + if err := s.check(a); err != nil { + return nil, err + } + switch g, ok := a.Grant(r.Export); { + case !ok: + return nil, sandboxfs.NewFailure(sandboxfs.CodeUnauthorized, sandboxwire.EffectNone, "export "+string(r.Export)+" is not granted") + case g.ReadOnly && !r.ReadOnly: + return nil, sandboxfs.NewFailure(sandboxfs.CodeUnauthorized, sandboxwire.EffectNone, "export "+string(r.Export)+" is granted read-only") + } + if f := s.caps.Admit(r, r.ReadOnly); f != nil { + return nil, f + } + if r.Export != Export { + return nil, sandboxfs.NewFailure(sandboxfs.CodeInvalidArgument, sandboxwire.EffectNone, "unknown export "+string(r.Export)) + } + var fd int + err := use(s.root, errStaleAttachment, func(root int) (err error) { + fd, err = unix.FcntlInt(uintptr(root), unix.F_DUPFD_CLOEXEC, 0) + return err + }) + if err != nil { + return nil, failure(err, sandboxwire.EffectNone) + } + var sb unix.Stat_t + if err := unix.Fstat(fd, &sb); err != nil { + unix.Close(fd) + return nil, failure(err, sandboxwire.EffectNone) + } + st := newState(s, r.ReadOnly) + s.mu.Lock() + switch { + case s.closed: + s.mu.Unlock() + unix.Close(fd) + return nil, errStaleAttachment() + case s.atts[a.ID] != nil: + s.mu.Unlock() + unix.Close(fd) + return nil, sandboxfs.NewFailure(sandboxfs.CodeInvalidArgument, sandboxwire.EffectNone, "attachment is already attached") + } + s.atts[a.ID] = st + s.mu.Unlock() + go s.watch(a, st) + root, err := st.addNode(fd, &sb) + if err != nil { + s.detach(a.ID, st) + return nil, failure(err, sandboxwire.EffectNone) + } + return &sandboxfs.AttachResponse{Root: sandboxfs.Entry{Node: root.ref, Attr: st.attr(&sb)}}, nil +} + +// watch detaches st when its lease ends. +func (s *Service) watch(a sandboxfs.Attachment, st *state) { + select { + case <-a.Lease.Done(): + s.detach(a.ID, st) + case <-st.done: + } +} + +func (s *Service) detach(id sandboxwire.ID, st *state) { + s.mu.Lock() + if s.atts[id] == st { + delete(s.atts, id) + } + s.mu.Unlock() + st.close() +} + +func (s *Service) Detach(_ context.Context, a sandboxfs.Attachment, r *sandboxfs.DetachRequest) (*sandboxfs.DetachResponse, error) { + st, err := s.enter(a, r) + if err != nil { + return nil, err + } + s.detach(a.ID, st) + return &sandboxfs.DetachResponse{}, nil +} + +// state is one attachment: its node and handle tables. +type state struct { + svc *Service + readOnly bool + done chan struct{} + + mu sync.Mutex + closed bool + nodes []*node // index ID-1 + free []int + inodes map[inodeKey]*node + generation uint64 + handles map[sandboxfs.HandleID]*handle + reserved int + nextHandle uint64 +} + +// inodeKey identifies a node. The mount ID keeps the same inode reached +// through two mounts, such as a bind mount and its source, in two nodes. +type inodeKey struct{ mnt, dev, ino uint64 } + +// mountID returns the ID of the mount the object of fd is reached through: +// statx reports it from Linux 5.8, and the mnt_id line of +// /proc/self/fdinfo/ before that. +func (s *Service) mountID(fd int) (uint64, error) { + var stx unix.Statx_t + if unix.Statx(fd, "", unix.AT_EMPTY_PATH|unix.AT_SYMLINK_NOFOLLOW, unix.STATX_MNT_ID, &stx) == nil && stx.Mask&unix.STATX_MNT_ID != 0 { + return stx.Mnt_id, nil + } + return s.fdinfoMountID(fd) +} + +func (s *Service) fdinfoMountID(fd int) (uint64, error) { + var info []byte + err := use(s.fdinfo, errStaleAttachment, func(dir int) error { + f, err := openat(dir, strconv.Itoa(fd), unix.O_RDONLY, 0) + if err != nil { + return err + } + defer unix.Close(f) + buf := make([]byte, 4096) + for { + n, err := eintr(func() (int, error) { return unix.Read(f, buf) }) + if n <= 0 || err != nil { + return err + } + info = append(info, buf[:n]...) + } + }) + if err != nil { + return 0, err + } + for line := range strings.Lines(string(info)) { + if v, ok := strings.CutPrefix(line, "mnt_id:"); ok { + return strconv.ParseUint(strings.TrimSpace(v), 10, 64) + } + } + return 0, errors.New("fdinfo has no mnt_id") +} + +// node is a file-system object held by an O_PATH descriptor that never +// follows a symlink. +type node struct { + ref sandboxfs.NodeRef + f *os.File + key inodeKey + typ uint32 // sandboxfs mode type bits; an inode never changes type + refs uint64 +} + +type handle struct { + f *os.File + append bool + dir *cursor // set for directory handles + + lockMu sync.Mutex + flock sandboxfs.LockMode // held flock mode; zero when none +} + +func newState(s *Service, readOnly bool) *state { + // Random bases make a NodeRef or HandleID from another attachment or + // incarnation miss instead of naming a live object. + return &state{ + svc: s, readOnly: readOnly, done: make(chan struct{}), + inodes: map[inodeKey]*node{}, handles: map[sandboxfs.HandleID]*handle{}, + generation: rand.Uint64() >> 2, nextHandle: rand.Uint64() >> 2, + } +} + +// close releases every node and handle; closing a handle releases its locks. +func (st *state) close() { + st.mu.Lock() + if st.closed { + st.mu.Unlock() + return + } + st.closed = true + nodes, handles := st.nodes, st.handles + st.nodes, st.inodes, st.handles = nil, nil, nil + close(st.done) + st.mu.Unlock() + for _, n := range nodes { + if n != nil { + n.f.Close() + } + } + for _, h := range handles { + h.f.Close() + } +} + +// addNode takes ownership of fd and acquires one reference on its node. +func (st *state) addNode(fd int, sb *unix.Stat_t) (*node, error) { + mnt, err := st.svc.mountID(fd) + if err != nil { + unix.Close(fd) + return nil, err + } + key := inodeKey{mnt, uint64(sb.Dev), sb.Ino} + st.mu.Lock() + defer st.mu.Unlock() + if st.closed { + unix.Close(fd) + return nil, errStaleAttachment() + } + if n := st.inodes[key]; n != nil { + n.refs++ + unix.Close(fd) + return n, nil + } + i := len(st.nodes) + if k := len(st.free); k > 0 { + i, st.free = st.free[k-1], st.free[:k-1] + } else { + st.nodes = append(st.nodes, nil) + } + st.generation++ + n := &node{ + ref: sandboxfs.NodeRef{ID: uint64(i) + 1, Generation: st.generation}, + f: os.NewFile(uintptr(fd), ""), key: key, typ: sb.Mode & sandboxfs.ModeType, refs: 1, + } + st.nodes[i], st.inodes[key] = n, n + return n, nil +} + +func (st *state) node(ref sandboxfs.NodeRef) (*node, error) { + st.mu.Lock() + defer st.mu.Unlock() + if st.closed { + return nil, errStaleAttachment() + } + if i := ref.ID - 1; i < uint64(len(st.nodes)) { + if n := st.nodes[i]; n != nil && n.ref == ref { + return n, nil + } + } + return nil, errStaleNode() +} + +// forget applies every entry or none. +func (st *state) forget(entries []sandboxfs.ForgetEntry) error { + st.mu.Lock() + var drop []*node + defer func() { + st.mu.Unlock() + for _, n := range drop { + n.f.Close() + } + }() + if st.closed { + return errStaleAttachment() + } + for _, e := range entries { + i := e.Node.ID - 1 + if i >= uint64(len(st.nodes)) || st.nodes[i] == nil || st.nodes[i].ref != e.Node { + return errStaleNode() + } + if e.Count > st.nodes[i].refs { + return sandboxfs.NewFailure(sandboxfs.CodeInvalidArgument, sandboxwire.EffectNone, "forget exceeds the node's references") + } + } + for _, e := range entries { + i := e.Node.ID - 1 + n := st.nodes[i] + if n.refs -= e.Count; n.refs == 0 { + st.nodes[i] = nil + st.free = append(st.free, int(i)) + delete(st.inodes, n.key) + drop = append(drop, n) + } + } + return nil +} + +// reserve claims a handle slot before a handle is opened, so exhaustion +// fails before anything happens. +func (st *state) reserve() error { + st.mu.Lock() + defer st.mu.Unlock() + if st.closed { + return errStaleAttachment() + } + if len(st.handles)+st.reserved >= int(st.svc.caps.MaxOpenHandles) { + return sandboxfs.NewFailure(sandboxfs.CodeResourceExhausted, sandboxwire.EffectNone, "too many open handles") + } + st.reserved++ + return nil +} + +func (st *state) unreserve() { + st.mu.Lock() + st.reserved-- + st.mu.Unlock() +} + +// addHandle registers h in a reserved slot. +func (st *state) addHandle(h *handle) (sandboxfs.HandleID, error) { + st.mu.Lock() + defer st.mu.Unlock() + st.reserved-- + if st.closed { + h.f.Close() + return 0, errStaleAttachment() + } + st.nextHandle++ + id := sandboxfs.HandleID(st.nextHandle) + st.handles[id] = h + return id, nil +} + +func (st *state) handle(id sandboxfs.HandleID) (*handle, error) { + st.mu.Lock() + defer st.mu.Unlock() + if st.closed { + return nil, errStaleAttachment() + } + if h := st.handles[id]; h != nil { + return h, nil + } + return nil, errStaleHandle() +} + +// release closes handle id if it is a directory handle exactly when dir is +// set. +func (st *state) release(id sandboxfs.HandleID, dir bool) error { + st.mu.Lock() + h := st.handles[id] + switch { + case st.closed: + st.mu.Unlock() + return errStaleAttachment() + case h == nil: + st.mu.Unlock() + return errStaleHandle() + case (h.dir != nil) != dir: + st.mu.Unlock() + return sandboxfs.NewErrnoFailure(sandboxfs.ErrnoBadDescriptor, sandboxwire.EffectNone, "wrong handle kind") + } + delete(st.handles, id) + st.mu.Unlock() + return h.f.Close() +} + +// attr converts a stat. Ino combines the device with the inode number, as +// go-fuse's loopback does, so files on different mounts stay distinct. +func (st *state) attr(sb *unix.Stat_t) sandboxfs.Attr { + return sandboxfs.Attr{ + Ino: st.ino(uint64(sb.Dev), sb.Ino), + Mode: sb.Mode, + Nlink: uint32(sb.Nlink), + UID: sb.Uid, + GID: sb.Gid, + Rdev: uint64(sb.Rdev), + Size: uint64(sb.Size), + Blocks: uint64(sb.Blocks), + Blksize: uint32(sb.Blksize), + Atime: timestamp(sb.Atim), + Mtime: timestamp(sb.Mtim), + Ctime: timestamp(sb.Ctim), + } +} + +func (st *state) ino(dev, ino uint64) uint64 { + swap := func(d uint64) uint64 { return d<<32 | d>>32 } + return swap(dev) ^ swap(st.svc.rootDev) ^ ino +} + +func timestamp(t unix.Timespec) sandboxfs.Timestamp { + return sandboxfs.Timestamp{Sec: int64(t.Sec), Nsec: uint32(t.Nsec)} +} diff --git a/apps/sandboxio/internal/fileservice/service_test.go b/apps/sandboxio/internal/fileservice/service_test.go new file mode 100644 index 00000000..8c3156e0 --- /dev/null +++ b/apps/sandboxio/internal/fileservice/service_test.go @@ -0,0 +1,590 @@ +//go:build linux + +package fileservice + +import ( + "bytes" + "context" + "errors" + "fmt" + "math" + "net" + "os" + "os/exec" + "path/filepath" + "slices" + "strconv" + "strings" + "sync" + "syscall" + "testing" + "time" + + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxfs" + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxlink" + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxwire" +) + +type fixture struct { + t *testing.T + dir string + svc *Service + att sandboxfs.Attachment + c *sandboxfs.Client + root sandboxfs.NodeRef +} + +// newFixture serves a temporary directory as the export world and attaches +// to it. +func newFixture(t *testing.T) *fixture { + return attachRoot(t, t.TempDir()) +} + +// attachRoot serves dir as the export world and attaches to it. +func attachRoot(t *testing.T, dir string) *fixture { + svc, err := New(dir) + if err != nil { + t.Fatal(err) + } + t.Cleanup(func() { svc.Close() }) + f := &fixture{t: t, dir: dir, svc: svc, att: attachment(svc, sandboxlink.ExportGrant{ID: "world"})} + f.c, _, _ = f.connect(f.svc, f.att) + resp, err := f.c.Attach(context.Background(), &sandboxfs.AttachRequest{Export: "world"}) + if err != nil { + t.Fatal(err) + } + f.root = resp.Root.Node + return f +} + +// connect opens a stream to svc for attachment a. It returns the client, the +// server end of the stream and the channel Serve's result arrives on. +func (f *fixture) connect(svc *Service, a sandboxfs.Attachment) (*sandboxfs.Client, net.Conn, chan error) { + cc, sc := net.Pipe() + served := make(chan error, 1) + go func() { served <- sandboxfs.Serve(context.Background(), sc, svc, a) }() + c := sandboxfs.NewClient(cc) + f.t.Cleanup(func() { c.Close() }) + return c, sc, served +} + +// attachment returns a fresh attachment to svc with grants. +func attachment(svc *Service, grants ...sandboxlink.ExportGrant) sandboxfs.Attachment { + return sandboxfs.Attachment{ID: sandboxwire.NewID(), ServerInstanceID: svc.InstanceID(), Lease: context.Background(), Exports: grants} +} + +func (f *fixture) lookup(parent sandboxfs.NodeRef, name string) sandboxfs.Entry { + f.t.Helper() + r, err := f.c.Lookup(context.Background(), &sandboxfs.LookupRequest{Parent: parent, Name: []byte(name)}) + if err != nil { + f.t.Fatalf("lookup %q: %v", name, err) + } + return r.Entry +} + +func (f *fixture) create(name string, access sandboxfs.AccessMode, flags sandboxfs.OpenFlags) (sandboxfs.Entry, sandboxfs.HandleID) { + f.t.Helper() + r, err := f.c.Create(context.Background(), &sandboxfs.CreateRequest{Parent: f.root, Name: []byte(name), Mode: 0o640, Access: access, Flags: flags, Exclusive: true}) + if err != nil { + f.t.Fatalf("create %q: %v", name, err) + } + return r.Entry, r.Handle +} + +func (f *fixture) write(h sandboxfs.HandleID, off uint64, data string) { + f.t.Helper() + r, err := f.c.Write(context.Background(), &sandboxfs.WriteRequest{Handle: h, Offset: off, Data: []byte(data)}) + if err != nil || r.Written != uint32(len(data)) || r.Failure != nil { + f.t.Fatalf("write: %+v, %v", r, err) + } +} + +func (f *fixture) read(h sandboxfs.HandleID) string { + f.t.Helper() + r, err := f.c.Read(context.Background(), &sandboxfs.ReadRequest{Handle: h, Size: 4096}) + if err != nil { + f.t.Fatalf("read: %v", err) + } + return string(r.Data) +} + +// wantFailure checks err is a failure with code, errno and effect. +func wantFailure(t *testing.T, err error, code sandboxfs.ErrorCode, errno sandboxfs.Errno, effect sandboxwire.Effect) *sandboxfs.Failure { + t.Helper() + var f *sandboxfs.Failure + if !errors.As(err, &f) || f.Code != code || f.Errno != errno || f.Effect != effect { + t.Fatalf("got %v, want %s %s effect %d", err, code, errno, effect) + } + return f +} + +func TestCreateWriteRead(t *testing.T) { + f := newFixture(t) + d, err := f.c.Describe(context.Background(), &sandboxfs.DescribeRequest{}) + if err != nil { + t.Fatal(err) + } + if d.Identity != (sandboxfs.Identity{UID: uint32(os.Geteuid()), GID: uint32(os.Getegid())}) || !slices.Equal(d.Exports, []sandboxlink.ExportID{"world"}) || d.Capabilities.POSIXLocks || !d.Capabilities.Flock { + t.Fatalf("describe %+v", d) + } + + e, h := f.create("notes", sandboxfs.AccessReadWrite, 0) + f.write(h, 0, "hello world") + f.write(h, 6, "there") + if got := f.read(h); got != "hello there" { + t.Fatalf("read %q", got) + } + info, err := os.Lstat(filepath.Join(f.dir, "notes")) + if err != nil || info.Mode().Perm() != 0o640 || info.Size() != 11 { + t.Fatalf("on disk %v, %v", info.Mode(), err) + } + a, err := f.c.GetAttr(context.Background(), &sandboxfs.GetAttrRequest{Target: sandboxfs.Target{Kind: sandboxfs.TargetNode, Node: e.Node}}) + if err != nil || a.Attr.Size != 11 || a.Attr.Mode != sandboxfs.ModeRegular|0o640 { + t.Fatalf("getattr %+v, %v", a, err) + } + if _, err := f.c.Release(context.Background(), &sandboxfs.ReleaseRequest{Handle: h}); err != nil { + t.Fatal(err) + } + _, err = f.c.Read(context.Background(), &sandboxfs.ReadRequest{Handle: h, Size: 1}) + wantFailure(t, err, sandboxfs.CodeStaleHandle, 0, sandboxwire.EffectNone) +} + +func TestExclusiveCreateConflict(t *testing.T) { + f := newFixture(t) + f.create("x", sandboxfs.AccessWrite, 0) + _, err := f.c.Create(context.Background(), &sandboxfs.CreateRequest{Parent: f.root, Name: []byte("x"), Mode: 0o600, Access: sandboxfs.AccessWrite, Exclusive: true}) + wantFailure(t, err, sandboxfs.CodeErrno, sandboxfs.ErrnoExists, sandboxwire.EffectNone) + if _, err := f.c.Create(context.Background(), &sandboxfs.CreateRequest{Parent: f.root, Name: []byte("x"), Mode: 0o600, Access: sandboxfs.AccessWrite}); err != nil { + t.Fatalf("non-exclusive create of an existing file: %v", err) + } +} + +func TestAtomicAppendFromTwoHandles(t *testing.T) { + f := newFixture(t) + _, h1 := f.create("log", sandboxfs.AccessWrite, sandboxfs.OpenAppend) + open, err := f.c.Open(context.Background(), &sandboxfs.OpenRequest{Node: f.lookup(f.root, "log").Node, Access: sandboxfs.AccessWrite, Flags: sandboxfs.OpenAppend}) + if err != nil { + t.Fatal(err) + } + const records = 200 + var wg sync.WaitGroup + for i, h := range []sandboxfs.HandleID{h1, open.Handle} { + wg.Add(1) + go func() { + defer wg.Done() + for j := range records { + // Every write names offset zero; append ignores it. + rec := fmt.Sprintf("%d:%03d:%s\n", i, j, strings.Repeat("x", 64)) + if r, err := f.c.Write(context.Background(), &sandboxfs.WriteRequest{Handle: h, Data: []byte(rec)}); err != nil || int(r.Written) != len(rec) { + t.Errorf("append: %+v, %v", r, err) + return + } + } + }() + } + wg.Wait() + data, err := os.ReadFile(filepath.Join(f.dir, "log")) + if err != nil { + t.Fatal(err) + } + lines := strings.Split(strings.TrimSuffix(string(data), "\n"), "\n") + if len(lines) != 2*records { + t.Fatalf("%d records, want %d", len(lines), 2*records) + } + for _, l := range lines { + if len(l) != 6+64 || !strings.HasSuffix(l, strings.Repeat("x", 64)) { + t.Fatalf("torn record %q", l) + } + } +} + +func TestReadOpenHandleAfterUnlink(t *testing.T) { + f := newFixture(t) + _, h := f.create("gone", sandboxfs.AccessReadWrite, 0) + f.write(h, 0, "still here") + if _, err := f.c.Unlink(context.Background(), &sandboxfs.UnlinkRequest{Parent: f.root, Name: []byte("gone")}); err != nil { + t.Fatal(err) + } + if got := f.read(h); got != "still here" { + t.Fatalf("read %q", got) + } + a, err := f.c.GetAttr(context.Background(), &sandboxfs.GetAttrRequest{Target: sandboxfs.Target{Kind: sandboxfs.TargetHandle, Handle: h}}) + if err != nil || a.Attr.Nlink != 0 { + t.Fatalf("getattr %+v, %v", a, err) + } +} + +func TestRenameModes(t *testing.T) { + f := newFixture(t) + d, _ := f.c.Describe(context.Background(), &sandboxfs.DescribeRequest{}) + if !d.Capabilities.RenameNoReplace || !d.Capabilities.RenameExchange { + t.Skip("the kernel lacks renameat2 modes") + } + for name, content := range map[string]string{"a": "A", "b": "B"} { + if err := os.WriteFile(filepath.Join(f.dir, name), []byte(content), 0o600); err != nil { + t.Fatal(err) + } + } + rename := func(mode sandboxfs.RenameMode) error { + _, err := f.c.Rename(context.Background(), &sandboxfs.RenameRequest{Parent: f.root, Name: []byte("a"), NewParent: f.root, NewName: []byte("b"), Mode: mode}) + return err + } + content := func(name string) string { + b, _ := os.ReadFile(filepath.Join(f.dir, name)) + return string(b) + } + + wantFailure(t, rename(sandboxfs.RenameNoReplace), sandboxfs.CodeErrno, sandboxfs.ErrnoExists, sandboxwire.EffectNone) + if err := rename(sandboxfs.RenameExchange); err != nil || content("a") != "B" || content("b") != "A" { + t.Fatalf("exchange: %v, a=%q b=%q", err, content("a"), content("b")) + } + if err := rename(sandboxfs.RenameReplace); err != nil || content("b") != "B" { + t.Fatalf("replace: %v, b=%q", err, content("b")) + } + if _, err := os.Lstat(filepath.Join(f.dir, "a")); !os.IsNotExist(err) { + t.Fatalf("a after replace: %v", err) + } + wantFailure(t, rename(sandboxfs.RenameNoReplace), sandboxfs.CodeErrno, sandboxfs.ErrnoNotFound, sandboxwire.EffectNone) +} + +func TestReadDirPagesWithCookies(t *testing.T) { + f := newFixture(t) + var want []string + for i := range 40 { + name := fmt.Sprintf("file-%02d", i) + want = append(want, name) + if err := os.WriteFile(filepath.Join(f.dir, name), nil, 0o600); err != nil { + t.Fatal(err) + } + } + dir, err := f.c.OpenDir(context.Background(), &sandboxfs.OpenDirRequest{Node: f.root}) + if err != nil { + t.Fatal(err) + } + readAll := func(cookie uint64) (names []string, cookies []uint64) { + for { + r, err := f.c.ReadDir(context.Background(), &sandboxfs.ReadDirRequest{Handle: dir.Handle, Cookie: cookie, Limit: 200}) + if err != nil { + t.Fatal(err) + } + if !r.End && len(r.Entries) == 0 { + t.Fatal("empty page before the end") + } + for _, e := range r.Entries { + names = append(names, string(e.Name)) + cookies = append(cookies, e.Cookie) + cookie = e.Cookie + } + if r.End { + return names, cookies + } + } + } + names, cookies := readAll(0) + if got := slices.Sorted(slices.Values(names)); !slices.Equal(got, want) { + t.Fatalf("names %v", got) + } + // Resuming at an earlier cookie continues after that entry. + again, _ := readAll(cookies[9]) + if !slices.Equal(again, names[10:]) { + t.Fatalf("resumed at cookie 10: %v, want %v", again, names[10:]) + } + + r, err := f.c.ReadDir(context.Background(), &sandboxfs.ReadDirRequest{Handle: dir.Handle, Limit: 1024, WithAttrs: true}) + if err != nil || len(r.Entries) == 0 || r.Entries[0].Entry == nil || r.Entries[0].Entry.Attr.Mode&sandboxfs.ModeType != sandboxfs.ModeRegular { + t.Fatalf("readdir with attributes: %+v, %v", r, err) + } + for _, e := range r.Entries { + if _, err := f.c.Forget(context.Background(), &sandboxfs.ForgetRequest{Entries: []sandboxfs.ForgetEntry{{Node: e.Entry.Node, Count: 1}}}); err != nil { + t.Fatal(err) + } + } +} + +func TestWalkStopsAtSymlink(t *testing.T) { + f := newFixture(t) + if err := os.MkdirAll(filepath.Join(f.dir, "d", "sub"), 0o755); err != nil { + t.Fatal(err) + } + if err := os.Symlink("sub", filepath.Join(f.dir, "d", "link")); err != nil { + t.Fatal(err) + } + walk := func(names ...string) *sandboxfs.WalkResponse { + r := &sandboxfs.WalkRequest{Parent: f.root} + for _, n := range names { + r.Names = append(r.Names, []byte(n)) + } + resp, err := f.c.Walk(context.Background(), r) + if err != nil { + t.Fatal(err) + } + return resp + } + r := walk("d", "link", "x") + if len(r.Entries) != 2 || r.Failure != nil || r.Entries[1].Attr.Mode&sandboxfs.ModeType != sandboxfs.ModeSymlink { + t.Fatalf("walk through symlink: %+v", r) + } + r = walk("d", "missing", "x") + if len(r.Entries) != 1 || r.Failure == nil || r.Failure.Errno != sandboxfs.ErrnoNotFound { + t.Fatalf("walk to a missing name: %+v", r) + } +} + +func TestRootEscapeIsRefused(t *testing.T) { + f := newFixture(t) + _, err := f.c.Lookup(context.Background(), &sandboxfs.LookupRequest{Parent: f.root, Name: []byte("..")}) + wantFailure(t, err, sandboxfs.CodeInvalidArgument, 0, sandboxwire.EffectNone) + + // A client that skips validation gets the same answer from the server. + cc, sc := net.Pipe() + go sandboxfs.Serve(context.Background(), sc, f.svc, f.att) + defer cc.Close() + var e sandboxwire.Encoder + e.U64(f.root.ID) + e.U64(f.root.Generation) + e.Bytes([]byte("..")) + if err := sandboxwire.WriteFrame(cc, sandboxwire.Frame{Type: uint16(sandboxfs.OpLookup), RequestID: 1, Payload: e.Payload()}); err != nil { + t.Fatal(err) + } + resp, err := sandboxwire.ReadFrame(cc, sandboxwire.MaxPayload) + if err != nil || resp.Type != sandboxwire.ResponseType(uint16(sandboxfs.OpLookup)) || !bytes.HasPrefix(resp.Payload, []byte{0, 2, 0, byte(sandboxfs.CodeInvalidArgument)}) { + t.Fatalf("raw lookup of \"..\": %x, %v", resp.Payload, err) + } + + // A symlink to "/" is a leaf: nothing resolves through it. + if err := os.Symlink("/", filepath.Join(f.dir, "escape")); err != nil { + t.Fatal(err) + } + link := f.lookup(f.root, "escape").Node + _, err = f.c.Lookup(context.Background(), &sandboxfs.LookupRequest{Parent: link, Name: []byte("etc")}) + wantFailure(t, err, sandboxfs.CodeErrno, sandboxfs.ErrnoNotDirectory, sandboxwire.EffectNone) + _, err = f.c.Open(context.Background(), &sandboxfs.OpenRequest{Node: link, Access: sandboxfs.AccessRead}) + wantFailure(t, err, sandboxfs.CodeErrno, sandboxfs.ErrnoSymlinkLoop, sandboxwire.EffectNone) + _, err = f.c.OpenDir(context.Background(), &sandboxfs.OpenDirRequest{Node: link}) + wantFailure(t, err, sandboxfs.CodeErrno, sandboxfs.ErrnoNotDirectory, sandboxwire.EffectNone) + _, err = f.c.Mkdir(context.Background(), &sandboxfs.MkdirRequest{Parent: link, Name: []byte("x"), Mode: 0o755}) + wantFailure(t, err, sandboxfs.CodeErrno, sandboxfs.ErrnoNotDirectory, sandboxwire.EffectNone) +} + +func TestFlockInteroperatesWithNativeFlock(t *testing.T) { + f := newFixture(t) + _, h := f.create("lock", sandboxfs.AccessReadWrite, 0) + native, err := os.Open(filepath.Join(f.dir, "lock")) + if err != nil { + t.Fatal(err) + } + defer native.Close() + whole := sandboxfs.Lock{Mode: sandboxfs.LockWrite, End: math.MaxInt64} + setLock := func(ctx context.Context, mode sandboxfs.LockMode, wait bool) error { + l := whole + l.Mode = mode + _, err := f.c.SetLock(ctx, &sandboxfs.SetLockRequest{Handle: h, Kind: sandboxfs.LockFlock, Owner: 1, Lock: l, Wait: wait}) + return err + } + nativeTry := func() error { return syscall.Flock(int(native.Fd()), syscall.LOCK_EX|syscall.LOCK_NB) } + + if err := setLock(context.Background(), sandboxfs.LockWrite, false); err != nil { + t.Fatal(err) + } + if err := nativeTry(); err != syscall.EWOULDBLOCK { + t.Fatalf("native flock while the service holds it: %v", err) + } + if err := setLock(context.Background(), sandboxfs.LockUnlock, false); err != nil { + t.Fatal(err) + } + if err := nativeTry(); err != nil { + t.Fatalf("native flock after unlock: %v", err) + } + wantFailure(t, setLock(context.Background(), sandboxfs.LockWrite, false), sandboxfs.CodeErrno, sandboxfs.ErrnoAgain, sandboxwire.EffectNone) + + // A waiting lock the client abandons is cancelled on the server. + ctx, cancel := context.WithTimeout(context.Background(), 100*time.Millisecond) + defer cancel() + wantFailure(t, setLock(ctx, sandboxfs.LockWrite, true), sandboxfs.CodeDeadlineExceeded, 0, sandboxwire.EffectPossible) + syscall.Flock(int(native.Fd()), syscall.LOCK_UN) + time.Sleep(200 * time.Millisecond) + if err := nativeTry(); err != nil { + t.Fatalf("native flock after the cancelled wait: %v", err) + } + + _, err = f.c.SetLock(context.Background(), &sandboxfs.SetLockRequest{Handle: h, Kind: sandboxfs.LockPOSIX, Lock: whole}) + wantFailure(t, err, sandboxfs.CodeUnsupported, 0, sandboxwire.EffectNone) +} + +func TestIncarnationChange(t *testing.T) { + f := newFixture(t) + _, h := f.create("f", sandboxfs.AccessRead, 0) + next, err := New(f.dir) + if err != nil { + t.Fatal(err) + } + defer next.Close() + + // A stream bound to the old incarnation. + old, _, _ := f.connect(next, f.att) + _, err = old.Lookup(context.Background(), &sandboxfs.LookupRequest{Parent: f.root, Name: []byte("f")}) + wantFailure(t, err, sandboxfs.CodeInstanceChanged, 0, sandboxwire.EffectNone) + + // The same attachment rebound to the new incarnation holds nothing. + a := f.att + a.ServerInstanceID = next.InstanceID() + c, _, _ := f.connect(next, a) + _, err = c.Lookup(context.Background(), &sandboxfs.LookupRequest{Parent: f.root, Name: []byte("f")}) + wantFailure(t, err, sandboxfs.CodeStaleAttachment, 0, sandboxwire.EffectNone) + if _, err := c.Attach(context.Background(), &sandboxfs.AttachRequest{Export: "world"}); err != nil { + t.Fatal(err) + } + _, err = c.Lookup(context.Background(), &sandboxfs.LookupRequest{Parent: f.root, Name: []byte("f")}) + wantFailure(t, err, sandboxfs.CodeStaleNode, 0, sandboxwire.EffectNone) + _, err = c.Read(context.Background(), &sandboxfs.ReadRequest{Handle: h, Size: 1}) + wantFailure(t, err, sandboxfs.CodeStaleHandle, 0, sandboxwire.EffectNone) +} + +func TestAttachFollowsExportGrants(t *testing.T) { + f := newFixture(t) + ctx := context.Background() + + c, _, _ := f.connect(f.svc, attachment(f.svc, sandboxlink.ExportGrant{ID: "logs"})) + _, err := c.Attach(ctx, &sandboxfs.AttachRequest{Export: "world"}) + wantFailure(t, err, sandboxfs.CodeUnauthorized, 0, sandboxwire.EffectNone) + + c, _, _ = f.connect(f.svc, attachment(f.svc, sandboxlink.ExportGrant{ID: "world", ReadOnly: true})) + _, err = c.Attach(ctx, &sandboxfs.AttachRequest{Export: "world"}) + wantFailure(t, err, sandboxfs.CodeUnauthorized, 0, sandboxwire.EffectNone) + r, err := c.Attach(ctx, &sandboxfs.AttachRequest{Export: "world", ReadOnly: true}) + if err != nil { + t.Fatal(err) + } + _, err = c.Create(ctx, &sandboxfs.CreateRequest{Parent: r.Root.Node, Name: []byte("f"), Mode: 0o644, Access: sandboxfs.AccessWrite}) + wantFailure(t, err, sandboxfs.CodeErrno, sandboxfs.ErrnoReadOnlyFilesystem, sandboxwire.EffectNone) +} + +func TestDescribeListsGrantedExports(t *testing.T) { + f := newFixture(t) + c, _, _ := f.connect(f.svc, attachment(f.svc, sandboxlink.ExportGrant{ID: "logs"})) + d, err := c.Describe(context.Background(), &sandboxfs.DescribeRequest{}) + if err != nil || len(d.Exports) != 0 { + t.Fatalf("describe without a world grant: %+v, %v", d, err) + } +} + +// The service never resolves through a proc magic link: oac-sandbox-io +// serves "/", where /proc//cwd links to another directory. +func TestProcMagicLinkIsOpaque(t *testing.T) { + f := attachRoot(t, "/") + ctx := context.Background() + w, err := f.c.Walk(ctx, &sandboxfs.WalkRequest{Parent: f.root, Names: [][]byte{[]byte("proc"), []byte(strconv.Itoa(os.Getpid())), []byte("cwd")}}) + if err != nil || len(w.Entries) != 3 || w.Entries[2].Attr.Mode&sandboxfs.ModeType != sandboxfs.ModeSymlink { + t.Fatalf("walk to /proc//cwd: %+v, %v", w, err) + } + cwd := w.Entries[2].Node + _, err = f.c.Lookup(ctx, &sandboxfs.LookupRequest{Parent: cwd, Name: []byte("x")}) + wantFailure(t, err, sandboxfs.CodeErrno, sandboxfs.ErrnoNotDirectory, sandboxwire.EffectNone) + _, err = f.c.OpenDir(ctx, &sandboxfs.OpenDirRequest{Node: cwd}) + wantFailure(t, err, sandboxfs.CodeErrno, sandboxfs.ErrnoNotDirectory, sandboxwire.EffectNone) + _, err = f.c.Open(ctx, &sandboxfs.OpenRequest{Node: cwd, Access: sandboxfs.AccessRead}) + wantFailure(t, err, sandboxfs.CodeErrno, sandboxfs.ErrnoSymlinkLoop, sandboxwire.EffectNone) +} + +// A bind mount and its source are one inode on two mounts, and stay two +// nodes. Bind mounts need a mount namespace, so the test reruns itself under +// unshare when unprivileged user namespaces are available. +func TestBindAliasesStayDistinct(t *testing.T) { + if os.Getenv("OAC_TEST_IN_USERNS") == "" { + unshare, err := exec.LookPath("unshare") + if err != nil || exec.Command(unshare, "-Urm", "true").Run() != nil { + t.Skip("unprivileged user and mount namespaces are unavailable") + } + cmd := exec.Command(unshare, "-Urm", os.Args[0], "-test.run=^TestBindAliasesStayDistinct$", "-test.v") + cmd.Env = append(os.Environ(), "OAC_TEST_IN_USERNS=1") + if out, err := cmd.CombinedOutput(); err != nil || !bytes.Contains(out, []byte("--- PASS")) { + t.Fatalf("in a user namespace: %v\n%s", err, out) + } + return + } + dir := t.TempDir() + src, alias := filepath.Join(dir, "src"), filepath.Join(dir, "alias") + for _, d := range []string{src, alias} { + if err := os.Mkdir(d, 0o755); err != nil { + t.Fatal(err) + } + } + if err := syscall.Mount(src, alias, "", syscall.MS_BIND, ""); err != nil { + t.Fatal(err) + } + defer syscall.Unmount(alias, syscall.MNT_DETACH) + f := attachRoot(t, dir) + a, b := f.lookup(f.root, "src"), f.lookup(f.root, "alias") + if a.Node == b.Node || a.Attr.Ino != b.Attr.Ino { + t.Fatalf("source %+v and bind alias %+v", a, b) + } +} + +// The fdinfo fallback reports the mount ID statx does. +func TestMountIDMechanismsAgree(t *testing.T) { + f := newFixture(t) + fd := int(f.svc.root.Fd()) + viaStatx, err := f.svc.mountID(fd) + if err != nil { + t.Fatal(err) + } + if viaFdinfo, err := f.svc.fdinfoMountID(fd); err != nil || viaFdinfo != viaStatx { + t.Fatalf("fdinfo mount ID %d, %v; statx %d", viaFdinfo, err, viaStatx) + } +} + +func TestFailureAfterAChangeIsEffectPossible(t *testing.T) { + wantFailure(t, failure(errStaleAttachment(), possible), sandboxfs.CodeStaleAttachment, 0, possible) +} + +// wroteConn reports each completed write. +type wroteConn struct { + net.Conn + wrote chan struct{} +} + +func (c wroteConn) Write(p []byte) (int, error) { + n, err := c.Conn.Write(p) + c.wrote <- struct{}{} + return n, err +} + +func TestTransportLoss(t *testing.T) { + f := newFixture(t) + _, h := f.create("busy", sandboxfs.AccessRead, 0) + native, err := os.Open(filepath.Join(f.dir, "busy")) + if err != nil { + t.Fatal(err) + } + defer native.Close() + if err := syscall.Flock(int(native.Fd()), syscall.LOCK_EX); err != nil { + t.Fatal(err) + } + + // A second stream of the same attachment reuses its handles. + cc, sc := net.Pipe() + served := make(chan error, 1) + go func() { served <- sandboxfs.Serve(context.Background(), sc, f.svc, f.att) }() + wrote := make(chan struct{}, 1) + c := sandboxfs.NewClient(wroteConn{cc, wrote}) + defer c.Close() + result := make(chan error, 1) + go func() { + _, err := c.SetLock(context.Background(), &sandboxfs.SetLockRequest{Handle: h, Kind: sandboxfs.LockFlock, Lock: sandboxfs.Lock{Mode: sandboxfs.LockWrite, End: math.MaxInt64}, Wait: true}) + result <- err + }() + <-wrote // the server has read the request + sc.Close() + + fail := wantFailure(t, <-result, sandboxfs.CodeUnknown, 0, sandboxwire.EffectPossible) + if !errors.Is(fail, sandboxfs.ErrTransport) { + t.Fatalf("%v is not a transport failure", fail) + } + _, err = c.GetAttr(context.Background(), &sandboxfs.GetAttrRequest{Target: sandboxfs.Target{Kind: sandboxfs.TargetHandle, Handle: h}}) + wantFailure(t, err, sandboxfs.CodeUnknown, 0, sandboxwire.EffectNone) + select { + case <-served: + case <-time.After(5 * time.Second): + t.Fatal("Serve did not return after the transport closed") + } +} diff --git a/docs/development.md b/docs/development.md index d33ada6f..c9d86154 100644 --- a/docs/development.md +++ b/docs/development.md @@ -66,6 +66,8 @@ For frontend development, run `pnpm dev:web` using the fixture or Core connectio | `internal/sandboxwire` | Frame header, primitive encoding and request ID sequence shared by the sandbox I/O protocols | [Framing](sandbox-link-protocol.md#framing) | | `internal/sandboxlink` | Link protocol, peer libraries and relay core | [Sandbox link protocol](sandbox-link-protocol.md) | | `internal/sandboxbootstrap` | Provider-to-Sandbox I/O service startup input | [Sandbox bootstrap](sandbox-bootstrap.md) | +| `internal/sandboxfs` | Runtime–file service wire types, validators, client and server | [File access protocol](file-access-protocol.md) | +| `apps/sandboxio` | Sandbox I/O service and its Linux protocol services | [File access protocol](file-access-protocol.md#the-linux-service) | | `apps/daemon/internal/dispatch` | Runtime preparation, Executor reuse, Turn and cleanup ownership | [Harness lifecycle](../contracts/agents-api/harness-onboarding.md#required-adapter-interfaces) | | `apps/daemon/internal/agent` | Native harness adapters | [Native references](../contracts/agents-api/harness-onboarding.md#native-references) | | `services/core/internal/sandbox` | Provider interfaces and managed compute lifecycle | [Provider onboarding](sandbox-provider.md) | diff --git a/docs/file-access-protocol.md b/docs/file-access-protocol.md new file mode 100644 index 00000000..1496f95d --- /dev/null +++ b/docs/file-access-protocol.md @@ -0,0 +1,328 @@ +# File access protocol + +The File access protocol is how a Runtime reads and changes the files of a sandbox. The Sandbox I/O service in the sandbox serves it, and the Runtime is its client. It is a node and handle protocol shaped like the FUSE low-level operations: lookups acquire node references, opens return handles, reads and writes take offsets, directory reads resume at cookies, and locks work against the sandbox's own processes. Phase 1 serves the [Uncached](#uncached-profile) profile only, with no change stream. + +[`internal/sandboxfs/protocol.go`](../internal/sandboxfs/protocol.go) is the authored definition: message tags, payload layouts, validators and the `Service` interface. The same package holds the generic client and server. [`apps/sandboxio/internal/fileservice`](../apps/sandboxio/internal/fileservice) is the Linux service. Frames use the shared [framing](sandbox-link-protocol.md#framing), and the Link layer supplies the authenticated attachment of each stream. + +## Streams and attachments + +- A stream belongs to one attachment, which Link authenticates and hands to the server as a `sandboxfs.Attachment`: its ID, the `ServerInstanceID` the stream was bound to, its lease and the exports it is granted. No request names an attachment, an OS user or a credential. +- Request IDs follow the [framing](sandbox-link-protocol.md#framing) rule, and a response carries its request's RequestID. Requests on one stream run concurrently, so responses can arrive in any order. The protocol has no events. +- An attachment attaches to one export. Its node references, handles and locks live in the service until it detaches, its lease ends or the service restarts. A new stream of the same attachment in the same incarnation continues with them. +- A service incarnation is named by `ServerInstanceID`. A request on a stream bound to another incarnation fails with `InstanceChanged`, and nothing is reopened automatically. + +## Implement a client + +The Go client is `sandboxfs.NewClient(stream)`. It has one method per operation, is safe for concurrent use, and returns a `*sandboxfs.Failure` for every failure. Its methods map one to one onto the go-fuse node operations, so a FUSE frontend turns each kernel request into one call. + +1. Call `Describe`. Keep `ServerInstanceID`, and check each request against `Capabilities` before sending it; the service rejects anything the capabilities do not declare. +2. `Attach` an export and keep the root `NodeRef`. +3. `Lookup`, `Walk`, `Create`, `Mkdir`, `Symlink`, `Link` and `ReadDir` with `WithAttrs` each acquire one reference on every node they return. Release references with `Forget` when the kernel forgets them. +4. `Walk` stops after a symlink. Resolve the link yourself, relative to the view, with `Readlink` and further walks. +5. `Open`, `Create` and `OpenDir` return handles. Send `Flush` on each close of a descriptor for the handle, and `Release` or `ReleaseDir` when the last one closes. +6. Read the `Effect` of every failure. After `EffectPossible`, the request may have taken effect: never replay a mutation automatically. Report the failure, or inspect the state with `GetAttr` or `Lookup` first. + +Cancelling a call's context returns at once with `Cancelled` or `DeadlineExceeded`. A call cancelled before its request is written fails with `EffectNone`. After the request is written, the client sends `CancelRequest` for it and discards the late response, and the failure is `EffectNone` for a [side-effect-free request](#effects-and-cancellation) and `EffectPossible` for every other one. Cancelling a call while its request is being written fails the stream instead, because a partial frame cannot be withdrawn. A cancel that races the completion of a write may still fail the stream, and the requests in flight then fail with `EffectPossible`. + +When the stream fails, every request in flight fails with `Unknown` and `EffectPossible`, and later calls fail with `EffectNone`; `errors.Is(err, sandboxfs.ErrTransport)` matches both. `Client.Done` closes and `Client.Err` returns the cause. To continue, open a new stream for the same attachment with `ExpectedServerInstanceID` set, as the [Sandbox link protocol](sandbox-link-protocol.md) describes. `InstanceChanged` then means every node, handle and lock of the attachment is gone. + +## Implement a service + +Implement `sandboxfs.Service` and serve each stream with `sandboxfs.Serve(ctx, stream, service, attachment)`. `Serve`: + +- refuses an attachment without an ID, a `ServerInstanceID`, a lease or at least one valid export grant with unique IDs; +- decodes and validates each request, and answers a malformed payload with `InvalidArgument` and `EffectNone`; +- ends the stream on a framing violation: an unknown tag, a frame that is not a request, or a RequestID that does not increase; +- holds up to `sandboxfs.MaxInFlight` (256) requests, each from admission until its response is written, and answers any more with `ResourceExhausted` and `EffectNone`, so a `CancelRequest` arrives while the client reads responses; +- answers `CancelRequest` itself by cancelling the target's context; +- returns a method's `*Failure` as the typed failure, and reports any other error, or a response that fails validation, as `Unknown` with `EffectPossible`; +- cancels every request's context when the stream ends. + +A service must: + +- generate a new `ServerInstanceID` whenever it loses its node and handle tables, and answer a request whose `Attachment.ServerInstanceID` is not its own with `InstanceChanged`; +- answer `StaleAttachment` while the attachment is not attached or after its lease ends, and `StaleNode` or `StaleHandle` for a reference or handle the attachment does not hold. A node ID is reused only with a new generation; +- call `Capabilities.Admit(request, readOnly)` before running a request and return the failure it reports. Limits that depend on service state, such as `MaxOpenHandles`, stay with the service; +- advertise only what it enforces, and advertise locks only when they interoperate with native processes in the sandbox; +- act as its own process identity, never as an identity a request supplies, and apply requested permission bits exactly; +- resolve each name as one entry of a directory node and never traverse a symlink; +- report `EffectPossible` once any part of a mutation may have applied. + +### The Linux service + +`fileservice.New(root)` serves the absolute directory `root` as the one export `world`, and `Describe` lists `world` only to an attachment granted it. `oac-sandbox-io` passes `/`; the Provider's sandbox setup owns the isolation of everything under it, as the [Sandbox bootstrap](sandbox-bootstrap.md#responsibilities) states, and the service enforces no boundary inside the export. `New` sets the process umask to zero and reports its effective UID and GID as `Identity`. `InstanceID` returns the `ServerInstanceID` to give Link, and `Close` releases every attachment. + +- Each node holds an `O_PATH|O_NOFOLLOW` descriptor. In an attachment a node is one mount ID, device and inode, so hard links share a node while a bind mount and its source stay two. The mount ID comes from `statx` with `STATX_MNT_ID`, or from the `mnt_id` line of `/proc/self/fdinfo/` on kernels older than 5.8. +- A lookup opens one component with `openat` and `O_NOFOLLOW` on its parent's descriptor. A symlink, including a proc magic link such as `/proc//cwd`, is a node of its own and is never traversed: `Lookup` and `Readlink` return the link itself, a directory operation on it fails with `Errno` `NotDirectory`, and `Open` fails with `SymlinkLoop`. +- Operations on a node, such as opening, truncating, changing its mode or times and linking it, go through the `/proc/self/fd` name of the descriptor the service holds for it. That name resolves to the descriptor's own object, never to a symlink's target, and an open through it first checks the object's file type. +- An open handle holds its own descriptor, so it keeps working after its file is unlinked or renamed. +- `Rename` uses `renameat2`. The service declares `RenameNoReplace` and `RenameExchange` only when a probe at start succeeds. +- `LockFlock` locks the handle's descriptor, so it interoperates with native `flock`. `POSIXLocks` is false, and `GetLock` and `LockPOSIX` return `Unsupported`. +- `ReadDir` cookies are the kernel's directory offsets. `Attr.Ino` combines the device and the inode number as go-fuse's loopback does. +- It declares `MaxNameBytes` 255, `MaxPathBytes` 4095, `MaxReadBytes` and `MaxWriteBytes` 64 KiB, `MaxWalkComponents` 256, `MaxReadDirBytes` 64 KiB and `MaxOpenHandles` 4096, and every flag except `ReadOnly` and `POSIXLocks`, with the rename modes as probed. + +## Reference + +### Messages + +Requests use tags 1 to 30; the response to tag `t` uses `t | 0x8000`. + +| Tag | Request | Fields | Response | Meaning | +| --- | --- | --- | --- | --- | +| 1 | `Describe` | – | `ServerInstanceID`, `Identity`, `Capabilities`, `Exports` | The incarnation, identity, capabilities and granted exports | +| 2 | `Attach` | `Export`, `ReadOnly` | `Root` (`Entry`) | Attach to a granted export. The root is a directory | +| 3 | `Detach` | – | – | Release every node, handle and lock of the attachment | +| 4 | `Lookup` | `Parent`, `Name` | `Entry` | Look up one entry | +| 5 | `Walk` | `Parent`, `Names` | `Entries`, optional `Failure` | Look up components in turn. Stops after a symlink, or at a failure that `Failure` reports; a failure at the first name fails the request | +| 6 | `GetAttr` | `Target` | `Attr` | Attributes of a node or open handle | +| 7 | `SetAttr` | `Target`, `Set`, selected values | `Attr` | Change the selected attributes | +| 8 | `Access` | `Node`, `Mask` | – | Check permissions as the service's identity | +| 9 | `Open` | `Node`, `Access`, `Flags` | `Handle` | Open a regular file | +| 10 | `Create` | `Parent`, `Name`, `Mode`, `Access`, `Flags`, `Exclusive` | `Entry`, `Handle` | Create and open a regular file | +| 11 | `Read` | `Handle`, `Offset`, `Size` | `Data` | Fewer bytes than asked means end of file | +| 12 | `Write` | `Handle`, `Offset`, `Data` | `Written`, optional `Failure` | Write at `Offset`, or append on an append-opened handle | +| 13 | `Flush` | `Handle`, `Owner` | – | One descriptor for the handle closed | +| 14 | `Fsync` | `Handle`, `DataOnly` | – | Make the file's data, and its metadata unless `DataOnly`, durable | +| 15 | `Release` | `Handle` | – | Close a file handle | +| 16 | `OpenDir` | `Node` | `Handle` | Open a directory | +| 17 | `ReadDir` | `Handle`, `Cookie`, `Limit`, `WithAttrs` | `Entries`, `End` | Read entries after `Cookie` | +| 18 | `ReleaseDir` | `Handle` | – | Close a directory handle | +| 19 | `Mkdir` | `Parent`, `Name`, `Mode` | `Entry` | Create a directory | +| 20 | `Unlink` | `Parent`, `Name` | – | Remove a non-directory entry | +| 21 | `Rmdir` | `Parent`, `Name` | – | Remove an empty directory | +| 22 | `Rename` | `Parent`, `Name`, `NewParent`, `NewName`, `Mode` | – | Rename in one [mode](#rename) | +| 23 | `Link` | `Node`, `NewParent`, `NewName` | `Entry` | Create a hard link | +| 24 | `Symlink` | `Parent`, `Name`, `Target` | `Entry` | Create a symlink | +| 25 | `Readlink` | `Node` | `Target` | Read a symlink | +| 26 | `StatFS` | `Node` | File-system statistics | Statistics of the file system holding the node | +| 27 | `Forget` | `Entries` | – | Release lookup references, all or none | +| 28 | `GetLock` | `Handle`, `Owner`, `Lock` | optional `Conflict` | A POSIX lock that would conflict | +| 29 | `SetLock` | `Handle`, `Kind`, `Owner`, `Lock`, `Wait` | – | Acquire, convert or release a lock | +| 30 | `CancelRequest` | `Target` (a RequestID) | – | Ask to cancel an outstanding request | + +A response payload begins with a uint16 result: 1 for success, followed by the response's fields, or 2 for failure, followed by a [`Failure`](#failures). Payloads list their fields in this order: + +```text +Describe (no fields) +DescribeResponse ServerInstanceID ID, Identity {UID u32, GID u32}, Capabilities, Exports count 0..64 of bytes +Attach Export bytes, ReadOnly bool +AttachResponse Root Entry +Detach (no fields); response (no fields) +Lookup Parent NodeRef, Name bytes +LookupResponse Entry +Walk Parent NodeRef, Names count 1..1024 of bytes +WalkResponse Entries count 1..1024 of Entry, Failure optional Failure +GetAttr Target +GetAttrResponse Attr +SetAttr Target, Set u32, then for each selected bit in order: Size u64, Mode u32, UID u32, GID u32, Atime Timestamp, Mtime Timestamp +SetAttrResponse Attr +Access Node NodeRef, Mask u32; response (no fields) +Open Node NodeRef, Access enum, Flags u32 +OpenResponse Handle u64 +Create Parent NodeRef, Name bytes, Mode u32, Access enum, Flags u32, Exclusive bool +CreateResponse Entry, Handle u64 +Read Handle u64, Offset u64, Size u32 +ReadResponse Data bytes +Write Handle u64, Offset u64, Data bytes +WriteResponse Written u32, Failure optional Failure +Flush Handle u64, Owner u64; response (no fields) +Fsync Handle u64, DataOnly bool; response (no fields) +Release Handle u64; response (no fields) +OpenDir Node NodeRef +OpenDirResponse Handle u64 +ReadDir Handle u64, Cookie u64, Limit u32, WithAttrs bool +ReadDirResponse Entries count of DirEntry, End bool +ReleaseDir Handle u64; response (no fields) +Mkdir Parent NodeRef, Name bytes, Mode u32 +MkdirResponse Entry +Unlink, Rmdir Parent NodeRef, Name bytes; response (no fields) +Rename Parent NodeRef, Name bytes, NewParent NodeRef, NewName bytes, Mode enum; response (no fields) +Link Node NodeRef, NewParent NodeRef, NewName bytes +LinkResponse Entry +Symlink Parent NodeRef, Name bytes, Target bytes +SymlinkResponse Entry +Readlink Node NodeRef +ReadlinkResponse Target bytes +StatFS Node NodeRef +StatFSResponse Blocks u64, BlocksFree u64, BlocksAvailable u64, Files u64, FilesFree u64, BlockSize u32, FragmentSize u32, NameMax u32 +Forget Entries count 1..4096 of {Node NodeRef, Count u64}; response (no fields) +GetLock Handle u64, Owner u64, Lock +GetLockResponse Conflict optional Lock +SetLock Handle u64, Kind enum, Owner u64, Lock, Wait bool; response (no fields) +CancelRequest Target u64; response (no fields) +``` + +[`testdata`](../internal/sandboxfs/testdata) holds annotated golden frames of `Describe`, `Walk`, `Create`, a short `Write`, `ReadDir` with a cookie, `Rename` and a failure. + +### Shared types + +```text +NodeRef ID u64, Generation u64 // both nonzero +HandleID u64 // nonzero, opaque +Timestamp Sec i64, Nsec u32 // Nsec below one billion +Attr Ino u64, Mode u32, Nlink u32, UID u32, GID u32, Rdev u64, Size u64, Blocks u64, Blksize u32, Atime Timestamp, Mtime Timestamp, Ctime Timestamp +Entry Node NodeRef, Attr +Target Kind enum (TargetNode = 1, TargetHandle = 2), then Node NodeRef or Handle u64 +DirEntry Name bytes, Ino u64, Type u32, Cookie u64, Entry optional Entry +Lock Mode enum (LockRead = 1, LockWrite = 2, LockUnlock = 3), Start u64, End u64 +``` + +- A `NodeRef` and a `HandleID` are scoped to the attachment and the service incarnation. A node ID is reused only after its object is forgotten, and then with another generation, so a stale `NodeRef` never names another object. +- `Attr.Mode` holds exactly one file type and the permission bits (`0o7777`) in the Linux `st_mode` layout. `Size` is at most 2^63−1, and `Blocks` counts 512-byte blocks. `Ino` identifies the file within the export. +- A `DirEntry`'s `Type` is one file type in the same layout. When `Entry` is present, its `Attr.Ino` and file type equal the entry's. +- `Blocks`, `Files` and the other `StatFSResponse` fields follow `statfs`: `BlocksAvailable` is what an unprivileged process may use. + +### Describe and capabilities + +`DescribeResponse` carries the incarnation, the identity, the capabilities and the declared exports the attachment is granted: 0 to 64 unique `ExportID`s, with the grammar of Link's [export grants](sandbox-link-protocol.md#opening-a-stream). `Identity` is the effective UID and GID the service runs as, which own the files it creates; it is informational, and no request carries or selects an identity. + +`Capabilities` encodes every field, in this order: + +| Field | Type | Meaning | +| --- | --- | --- | +| `PathProfile` | enum | `LinuxBytes` (1): names and symlink targets are Linux byte strings | +| `CacheProfile` | enum | `Uncached` (1), the [Uncached profile](#uncached-profile) | +| `Durability` | enum | `FsyncRequired` (1): data is durable only after `Fsync` succeeds | +| `MaxNameBytes` | u32 | Longest entry name, 1 to 1024 | +| `MaxPathBytes` | u32 | Longest symlink target, 1 to 4096 | +| `MaxReadBytes`, `MaxWriteBytes` | u32 | Largest `Read` size and `Write` data, 1 to 64 KiB | +| `MaxWalkComponents` | u32 | Most `Walk` names, 1 to 1024 | +| `MaxReadDirBytes` | u32 | Largest `ReadDir` limit, 1 to 256 KiB | +| `MaxOpenHandles` | u32 | Most open handles per attachment, at least 1 | +| `ReadOnly` | bool | The service accepts only read-only attachments | +| `AtomicAppend` | bool | Appends from several handles never interleave within a write | +| `AtomicRename` | bool | `RenameReplace` replaces the destination atomically | +| `RenameNoReplace`, `RenameExchange` | bool | The rename mode is supported | +| `HardLinks`, `Symlinks` | bool | `Link` and `Symlink` are supported | +| `SetMode`, `SetOwner`, `SetTimes` | bool | `SetAttr` may set the mode, the owner and the times | +| `DirectoryFsync` | bool | `Fsync` of a directory handle makes its entries durable | +| `ReadDirPlus` | bool | `ReadDir` supports `WithAttrs` | +| `Flock` | bool | `SetLock` supports `LockFlock` | +| `POSIXLocks` | bool | `GetLock` and `SetLock` support `LockPOSIX` | + +A writable service, one without `ReadOnly`, declares `AtomicAppend`, `AtomicRename`, `HardLinks` and `Symlinks`. A request beyond a declared limit fails with `InvalidArgument`, or `Errno` `NameTooLong` for a name or target, and a request for an undeclared feature fails with `Unsupported`. + +### Attach + +`Attach` selects one export by `ExportID`; it never takes a server path. The export must be one that the attachment's Link binding grants in `Attachment.Exports` (see the [Sandbox link protocol](sandbox-link-protocol.md)), and a read-only grant allows only `ReadOnly` attaches; otherwise `Attach` fails with `Unauthorized`. A granted export the service does not declare fails with `InvalidArgument`, as does a second `Attach` before `Detach`. Every request that changes files on a read-only attachment fails with `Errno` `ReadOnlyFilesystem`: `SetAttr`, `Create`, `Write`, `Mkdir`, `Unlink`, `Rmdir`, `Rename`, `Link`, `Symlink`, and `Open` for writing or with `OpenTruncate`. + +### Names and paths + +- Names and symlink targets are bytes, not UTF-8 strings. +- An entry name is 1 to 1024 bytes without NUL or `/`, and is never `.` or `..`. `ReadDir` never returns `.` or `..`. +- A symlink target is 1 to 4096 bytes without NUL. The service stores and returns it unchanged; the client resolves it. +- No request resolves a path or traverses a symlink. Every lookup names one entry of a directory node, and a request that needs a directory fails with `Errno` `NotDirectory` on any other node. + +### Attributes + +`SetAttr` addresses a node or an open handle and changes only the attributes its `Set` mask selects: + +| Bit | Name | Value | +| --- | --- | --- | +| `0x01` | `AttrSize` | `Size`: truncate or extend a regular file | +| `0x02` | `AttrMode` | `Mode`: permission bits only | +| `0x04` | `AttrUID` | `UID` | +| `0x08` | `AttrGID` | `GID` | +| `0x10` | `AttrAtime` | `Atime` | +| `0x20` | `AttrMtime` | `Mtime` | +| `0x40` | `AttrAtimeNow` | None: the access time becomes the service's current time | +| `0x80` | `AttrMtimeNow` | None: the modification time becomes the service's current time | + +Unknown bits are rejected, `AttrAtime` excludes `AttrAtimeNow` and `AttrMtime` excludes `AttrMtimeNow`, and an unselected value must be zero. A selected `UID` or `GID` of 4294967295, which `chown` reads as no change, is rejected. Setting the mode of a symlink fails with `Errno` `NotSupported`; times set on a symlink apply to the link itself. The response carries the attributes after the change. + +`Access` checks the `Mask` bits `MayExecute` (1), `MayWrite` (2) and `MayRead` (4) as the service's identity; a zero mask checks existence. + +### Files + +- `AccessMode` is `AccessRead` (1), `AccessWrite` (2) or `AccessReadWrite` (3). +- `OpenFlags` are `OpenAppend` (1), `OpenTruncate` (2), `OpenNoFollow` (4), `OpenSync` (8) and `OpenDataSync` (16); unknown flags are rejected. +- `Open` opens a regular file node. A node is an object, not a path, so `Open` never follows a symlink: a directory fails with `Errno` `IsDirectory`, a symlink with `SymlinkLoop`, and a special file with `Unsupported`. +- `Create` creates a regular file with exactly the requested permission bits; no umask applies. With `Exclusive`, an existing entry fails with `Errno` `Exists`. Without it, an existing regular file is opened, and `OpenTruncate` truncates it. +- `Read` takes an offset up to 2^63−1. A positioned `Write` must end at or before 2^63−1, or it fails with `InvalidArgument`. A handle opened with `OpenAppend` appends each write atomically and ignores `Offset`. +- A successful `WriteResponse` carries the exact number of bytes written. When a write stops after a nonzero prefix, `Written` counts the prefix and `Failure` says why it stopped; a write that wrote nothing is a failure response. +- `Flush` is neither `Fsync` nor `Release`. It reports the errors of closing one descriptor for the handle, and `Owner` names the closing lock owner. +- A file operation on a directory handle, or a directory operation on a file handle, fails with `Errno` `IsDirectory`, `NotDirectory` or `BadDescriptor`. + +### Directories + +- `OpenDir` opens a directory node, and `ReadDir` reads its entries after `Cookie`; cookie 0 is the start. Each `DirEntry.Cookie` is the position after that entry and is otherwise opaque. +- `Limit` bounds the sum of the returned entries' encoded sizes: 25 bytes plus the name for each entry, plus 104 when it carries an `Entry`. A limit too small for the first entry fails with `Errno` `InvalidArgument`. +- With `WithAttrs`, each returned entry carries an `Entry` with one lookup reference. +- `End` is true when no entries remain, and a page without entries always has it set. No name or cookie repeats within a page. `ReadDir` promises no snapshot of a directory that changes while it is read. + +### Rename + +`Rename` takes exactly one mode: `RenameReplace` (1) replaces an existing destination, `RenameNoReplace` (2) fails with `Errno` `Exists` when the destination exists, and `RenameExchange` (3) swaps two existing entries. + +### Locks + +- A `Lock` covers `Start` through `End` inclusive; `End` 2^63−1 extends to the end of the file. +- `SetLock` takes `LockPOSIX` (1) or `LockFlock` (2), an attachment-scoped `LockOwner`, a range, a mode and `Wait`. A flock lock covers the whole file, so its range is 0 to 2^63−1. Without `Wait`, a conflicting lock fails with `Errno` `Again`; with it, the request waits until the lock is free or the request is cancelled. +- `GetLock` asks for a POSIX lock that would conflict with a read or write `Lock`, and returns it in `Conflict`. +- A service advertises a lock kind only when its locks interoperate with native processes in the sandbox. A client-local lock table is not enough. + +### Failures + +```text +Failure + Code enum + Errno optional enum // present exactly when Code is Errno + Effect Effect + Message bytes // at most 1024 bytes, for people +``` + +| Code | Name | Returned when | +| --- | --- | --- | +| 1 | `InvalidArgument` | A payload fails validation, a request exceeds a declared limit, the export is unknown, the attachment is already attached, or `Forget` exceeds the references held | +| 2 | `Unsupported` | The capabilities do not declare the operation or option, or the request needs a feature phase 1 excludes | +| 3 | `Unauthorized` | `Attach` names an export the Link binding does not grant, or asks for write access to a read-only grant | +| 4 | `StaleAttachment` | The attachment is not attached, has detached or its lease ended | +| 5 | `InstanceChanged` | The stream is bound to another service incarnation | +| 6 | `StaleNode` | The attachment holds no such `NodeRef` | +| 7 | `StaleHandle` | The attachment has no such open handle | +| 8 | `ResourceExhausted` | The stream holds `MaxInFlight` requests, or the attachment has `MaxOpenHandles` handles | +| 9 | `Cancelled` | The request was cancelled | +| 10 | `DeadlineExceeded` | The caller's deadline passed | +| 11 | `Errno` | A file-system call failed; `Errno` says how | +| 12 | `Unknown` | The stream failed, or the service failed without a typed error | + +`Errno` is a semantic enum, numbered from 1 in this order: `PermissionDenied`, `OperationNotPermitted`, `NotFound`, `Exists`, `NotDirectory`, `IsDirectory`, `DirectoryNotEmpty`, `InvalidArgument`, `BadDescriptor`, `TooManyOpenFiles`, `NoSpace`, `QuotaExceeded`, `ReadOnlyFilesystem`, `CrossDevice`, `NameTooLong`, `SymlinkLoop`, `FileTooLarge`, `Overflow`, `Busy`, `Again`, `Interrupted`, `IO`, `NoDevice`, `NoSuchDeviceOrAddress`, `BrokenPipe`, `NotSupported`, `NoLocks`, `Deadlock`. Each service converts its native errors; an unknown native error is `IO`, and success is never fabricated. + +### Effects and cancellation + +Every failure carries `EffectNone`, when the request certainly changed nothing, or `EffectPossible`. Transport loss after a request was written is `EffectPossible` unless the server later establishes the result. + +`Describe`, `GetAttr`, `Access`, `Read`, `Readlink`, `StatFS`, `GetLock` and `ReadDir` without `WithAttrs` leave no state behind, so abandoning one is `EffectNone`. Every other request may create, change or acquire something. + +An ambiguous mutation is never replayed automatically. This includes `Create`, a truncating `Open`, `SetAttr`, `Write` and a lock acquisition. + +`CancelRequest` asks the server to stop an outstanding request. Its acknowledgement is not the target's result and never proves that a mutation did not happen; the target's own response still follows. + +### Uncached profile + +`Uncached` means zero entry, attribute and negative TTLs, direct file I/O, no writeback cache and no directory listing retained by the client. It does not prohibit the server kernel from buffering data before `Fsync`. + +### Not in phase 1 + +Phase 1 has no operations and no negotiated support for mmap, extended attributes, allocation or hole punching, copy-range, creating or opening special files, and change notifications. The client returns the matching unsupported outcome for these. Attributes still describe an existing special file. + +### Limits + +| Limit | Value | +| --- | --- | +| Frame payload | 1 MiB | +| `Read` size, `Write` data | 64 KiB | +| Entry name | 1024 bytes | +| Symlink target | 4096 bytes | +| `Walk` names | 1024 | +| `ReadDir` limit | 256 KiB | +| `Forget` entries | 4096 | +| Declared exports | 64 | +| Failure message | 1024 bytes | +| Requests in flight per stream | 256 | + +Capabilities may declare smaller limits. + +## Verification + +`go test ./internal/sandboxfs` covers the golden frames, a round trip of every message, decode rejection and admission while responses go unread. `go test -run '^$' -fuzz FuzzDecode ./internal/sandboxfs` fuzzes the decoder. `go test ./apps/sandboxio/...` runs the Linux service over an in-memory stream against a temporary export, including exclusive create, atomic append, rename modes, directory paging, the root-escape attempts, opaque symlinks and proc magic links, bind-mount aliases, flock against native `flock`, export grants, an incarnation change and transport loss. The bind-mount test reruns itself under `unshare -Urm` and is skipped where unprivileged user namespaces are unavailable. diff --git a/internal/sandboxfs/client.go b/internal/sandboxfs/client.go new file mode 100644 index 00000000..7e41e393 --- /dev/null +++ b/internal/sandboxfs/client.go @@ -0,0 +1,391 @@ +package sandboxfs + +import ( + "context" + "errors" + "fmt" + "io" + "sync" + "sync/atomic" + + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxwire" +) + +// ErrTransport is the cause of a failure the client reports because the +// stream failed or closed. errors.Is finds it through the *Failure. +var ErrTransport = errors.New("sandboxfs: transport lost") + +// Client issues File requests over one stream. Its methods are safe for +// concurrent use; each returns a *Failure on failure. +type Client struct { + conn io.ReadWriteCloser + wsem chan struct{} // held while a RequestID is allocated and its frame written + seq sandboxwire.RequestSequence // used only while wsem is held + + mu sync.Mutex + pending map[uint64]*call + abandoned map[uint64]Op // requests whose response is discarded + err error + done chan struct{} + + afterWrite func() // test seam: runs once a frame is recorded as written +} + +type call struct { + op Op + ch chan outcome +} + +type outcome struct { + msg message + fail *Failure +} + +// NewClient starts a client on conn. The client owns conn and closes it on +// Close or when the stream fails. +func NewClient(conn io.ReadWriteCloser) *Client { + c := &Client{conn: conn, wsem: make(chan struct{}, 1), pending: map[uint64]*call{}, abandoned: map[uint64]Op{}, done: make(chan struct{})} + go c.readLoop() + return c +} + +// Close closes the stream. Requests in flight fail with EffectPossible. +func (c *Client) Close() error { + c.shutdown(errClientClosed) + return nil +} + +var errClientClosed = errors.New("client closed") + +// Done is closed once the stream has failed or closed. +func (c *Client) Done() <-chan struct{} { return c.done } + +// Err returns why the stream ended, or nil while it is open. +func (c *Client) Err() error { + c.mu.Lock() + defer c.mu.Unlock() + return c.err +} + +func transportFailure(effect sandboxwire.Effect, cause error) *Failure { + f := NewFailure(CodeUnknown, effect, "transport lost: "+cause.Error()) + f.cause = fmt.Errorf("%w: %w", ErrTransport, cause) + return f +} + +func (c *Client) shutdown(cause error) { + c.mu.Lock() + if c.err != nil { + c.mu.Unlock() + return + } + c.err = cause + pending := c.pending + c.pending, c.abandoned = nil, nil + close(c.done) + c.mu.Unlock() + c.conn.Close() + for _, cl := range pending { + cl.ch <- outcome{fail: transportFailure(sandboxwire.EffectPossible, cause)} + } +} + +// send registers a request and writes it. A request that ends before its +// write starts fails with EffectNone. Cancelling ctx during the write fails +// the stream, since a partial frame cannot be taken back, and the request +// then fails with EffectPossible. Once the write is recorded as finished, a +// cancellation is left to roundTrip, which sends CancelRequest. +func (c *Client) send(ctx context.Context, op Op, payload []byte) (uint64, *call, *Failure) { + if fail := c.acquire(ctx); fail != nil { + return 0, nil, fail + } + defer c.release() + if err := ctx.Err(); err != nil { + return 0, nil, contextFailure(err, sandboxwire.EffectNone) + } + c.mu.Lock() + if c.err != nil { + err := c.err + c.mu.Unlock() + return 0, nil, transportFailure(sandboxwire.EffectNone, err) + } + id, cl := c.seq.Next(), &call{op: op, ch: make(chan outcome, 1)} + c.pending[id] = cl + c.mu.Unlock() + // The write and the cancellation callback race for state; whichever + // moves it from writing decides. The callback is settled before the + // write slot is released, so it never acts on a later write. + const writing, written, interrupted = 0, 1, 2 + var state atomic.Int32 + settled := make(chan struct{}) + stop := context.AfterFunc(ctx, func() { + defer close(settled) + if state.CompareAndSwap(writing, interrupted) { + c.shutdown(fmt.Errorf("request cancelled while being written: %w", context.Cause(ctx))) + } + }) + err := sandboxwire.WriteFrame(c.conn, sandboxwire.Frame{Type: uint16(op), RequestID: id, Payload: payload}) + state.CompareAndSwap(writing, written) + if c.afterWrite != nil { + c.afterWrite() + } + if !stop() { + <-settled + } + if err != nil { + c.shutdown(err) + } + return id, cl, nil +} + +// acquire takes the write turn unless ctx ends or the stream fails first. +func (c *Client) acquire(ctx context.Context) *Failure { + select { + case c.wsem <- struct{}{}: + return nil + case <-ctx.Done(): + return contextFailure(ctx.Err(), sandboxwire.EffectNone) + case <-c.done: + return transportFailure(sandboxwire.EffectNone, c.Err()) + } +} + +func (c *Client) release() { <-c.wsem } + +// abandon stops waiting for request id. It reports false when the outcome +// has already been delivered. +func (c *Client) abandon(id uint64, cl *call) bool { + c.mu.Lock() + if c.pending[id] != cl { + c.mu.Unlock() + return false + } + delete(c.pending, id) + c.abandoned[id] = cl.op + c.mu.Unlock() + go c.cancel(id) + return true +} + +func (c *Client) cancel(target uint64) { + payload, err := encodeRequest(&CancelRequestRequest{Target: target}) + if err != nil || c.acquire(context.Background()) != nil { + return + } + defer c.release() + c.mu.Lock() + if c.err != nil { + c.mu.Unlock() + return + } + id := c.seq.Next() + c.abandoned[id] = OpCancelRequest + c.mu.Unlock() + if err := sandboxwire.WriteFrame(c.conn, sandboxwire.Frame{Type: uint16(OpCancelRequest), RequestID: id, Payload: payload}); err != nil { + c.shutdown(err) + } +} + +func (c *Client) readLoop() { + for { + f, err := sandboxwire.ReadFrame(c.conn, sandboxwire.MaxPayload) + if err == nil { + err = c.deliver(f) + } + if err != nil { + c.shutdown(err) + return + } + } +} + +// deliver routes one response. Anything but a well-formed response to an +// outstanding request is a protocol violation that ends the stream. +func (c *Client) deliver(f sandboxwire.Frame) error { + kind, err := tags.Classify(f.Type) + if err != nil { + return err + } + if kind != sandboxwire.KindResponse { + return malformed("message type %#04x from the server", f.Type) + } + op := Op(f.Type ^ sandboxwire.ResponseType(0)) + msg, fail, err := decodeResponse(op, f.Payload) + if err != nil { + return err + } + c.mu.Lock() + if cl, ok := c.pending[f.RequestID]; ok && cl.op == op { + delete(c.pending, f.RequestID) + c.mu.Unlock() + cl.ch <- outcome{msg: msg, fail: fail} + return nil + } + if abandoned, ok := c.abandoned[f.RequestID]; ok && abandoned == op { + delete(c.abandoned, f.RequestID) + c.mu.Unlock() + return nil + } + c.mu.Unlock() + return malformed("unexpected %s response to request %d", op, f.RequestID) +} + +func roundTrip[R message](ctx context.Context, c *Client, q Request) (R, error) { + var zero R + if err := ctx.Err(); err != nil { + return zero, contextFailure(err, sandboxwire.EffectNone) + } + payload, err := encodeRequest(q) + if err != nil { + f := NewFailure(CodeInvalidArgument, sandboxwire.EffectNone, err.Error()) + f.cause = err + return zero, f + } + id, cl, fail := c.send(ctx, q.Op(), payload) + if fail != nil { + return zero, fail + } + var out outcome + select { + case out = <-cl.ch: + case <-ctx.Done(): + if c.abandon(id, cl) { + effect := sandboxwire.EffectPossible + if sideEffectFree(q) { + effect = sandboxwire.EffectNone + } + return zero, contextFailure(ctx.Err(), effect) + } + out = <-cl.ch + } + if out.fail != nil { + return zero, out.fail + } + return out.msg.(R), nil +} + +func contextFailure(err error, effect sandboxwire.Effect) *Failure { + code := CodeCancelled + if errors.Is(err, context.DeadlineExceeded) { + code = CodeDeadlineExceeded + } + f := NewFailure(code, effect, err.Error()) + f.cause = err + return f +} + +func (c *Client) Describe(ctx context.Context, r *DescribeRequest) (*DescribeResponse, error) { + return roundTrip[*DescribeResponse](ctx, c, r) +} + +func (c *Client) Attach(ctx context.Context, r *AttachRequest) (*AttachResponse, error) { + return roundTrip[*AttachResponse](ctx, c, r) +} + +func (c *Client) Detach(ctx context.Context, r *DetachRequest) (*DetachResponse, error) { + return roundTrip[*DetachResponse](ctx, c, r) +} + +func (c *Client) Lookup(ctx context.Context, r *LookupRequest) (*LookupResponse, error) { + return roundTrip[*LookupResponse](ctx, c, r) +} + +func (c *Client) Walk(ctx context.Context, r *WalkRequest) (*WalkResponse, error) { + return roundTrip[*WalkResponse](ctx, c, r) +} + +func (c *Client) GetAttr(ctx context.Context, r *GetAttrRequest) (*GetAttrResponse, error) { + return roundTrip[*GetAttrResponse](ctx, c, r) +} + +func (c *Client) SetAttr(ctx context.Context, r *SetAttrRequest) (*SetAttrResponse, error) { + return roundTrip[*SetAttrResponse](ctx, c, r) +} + +func (c *Client) Access(ctx context.Context, r *AccessRequest) (*AccessResponse, error) { + return roundTrip[*AccessResponse](ctx, c, r) +} + +func (c *Client) Open(ctx context.Context, r *OpenRequest) (*OpenResponse, error) { + return roundTrip[*OpenResponse](ctx, c, r) +} + +func (c *Client) Create(ctx context.Context, r *CreateRequest) (*CreateResponse, error) { + return roundTrip[*CreateResponse](ctx, c, r) +} + +func (c *Client) Read(ctx context.Context, r *ReadRequest) (*ReadResponse, error) { + return roundTrip[*ReadResponse](ctx, c, r) +} + +func (c *Client) Write(ctx context.Context, r *WriteRequest) (*WriteResponse, error) { + return roundTrip[*WriteResponse](ctx, c, r) +} + +func (c *Client) Flush(ctx context.Context, r *FlushRequest) (*FlushResponse, error) { + return roundTrip[*FlushResponse](ctx, c, r) +} + +func (c *Client) Fsync(ctx context.Context, r *FsyncRequest) (*FsyncResponse, error) { + return roundTrip[*FsyncResponse](ctx, c, r) +} + +func (c *Client) Release(ctx context.Context, r *ReleaseRequest) (*ReleaseResponse, error) { + return roundTrip[*ReleaseResponse](ctx, c, r) +} + +func (c *Client) OpenDir(ctx context.Context, r *OpenDirRequest) (*OpenDirResponse, error) { + return roundTrip[*OpenDirResponse](ctx, c, r) +} + +func (c *Client) ReadDir(ctx context.Context, r *ReadDirRequest) (*ReadDirResponse, error) { + return roundTrip[*ReadDirResponse](ctx, c, r) +} + +func (c *Client) ReleaseDir(ctx context.Context, r *ReleaseDirRequest) (*ReleaseDirResponse, error) { + return roundTrip[*ReleaseDirResponse](ctx, c, r) +} + +func (c *Client) Mkdir(ctx context.Context, r *MkdirRequest) (*MkdirResponse, error) { + return roundTrip[*MkdirResponse](ctx, c, r) +} + +func (c *Client) Unlink(ctx context.Context, r *UnlinkRequest) (*UnlinkResponse, error) { + return roundTrip[*UnlinkResponse](ctx, c, r) +} + +func (c *Client) Rmdir(ctx context.Context, r *RmdirRequest) (*RmdirResponse, error) { + return roundTrip[*RmdirResponse](ctx, c, r) +} + +func (c *Client) Rename(ctx context.Context, r *RenameRequest) (*RenameResponse, error) { + return roundTrip[*RenameResponse](ctx, c, r) +} + +func (c *Client) Link(ctx context.Context, r *LinkRequest) (*LinkResponse, error) { + return roundTrip[*LinkResponse](ctx, c, r) +} + +func (c *Client) Symlink(ctx context.Context, r *SymlinkRequest) (*SymlinkResponse, error) { + return roundTrip[*SymlinkResponse](ctx, c, r) +} + +func (c *Client) Readlink(ctx context.Context, r *ReadlinkRequest) (*ReadlinkResponse, error) { + return roundTrip[*ReadlinkResponse](ctx, c, r) +} + +func (c *Client) StatFS(ctx context.Context, r *StatFSRequest) (*StatFSResponse, error) { + return roundTrip[*StatFSResponse](ctx, c, r) +} + +func (c *Client) Forget(ctx context.Context, r *ForgetRequest) (*ForgetResponse, error) { + return roundTrip[*ForgetResponse](ctx, c, r) +} + +func (c *Client) GetLock(ctx context.Context, r *GetLockRequest) (*GetLockResponse, error) { + return roundTrip[*GetLockResponse](ctx, c, r) +} + +func (c *Client) SetLock(ctx context.Context, r *SetLockRequest) (*SetLockResponse, error) { + return roundTrip[*SetLockResponse](ctx, c, r) +} diff --git a/internal/sandboxfs/client_test.go b/internal/sandboxfs/client_test.go new file mode 100644 index 00000000..ee4534a0 --- /dev/null +++ b/internal/sandboxfs/client_test.go @@ -0,0 +1,33 @@ +package sandboxfs + +import ( + "context" + "errors" + "net" + "testing" + + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxlink" + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxwire" +) + +// A cancellation that arrives after the frame is written cancels the request +// and leaves the stream open. +func TestCancelAfterWriteKeepsStream(t *testing.T) { + cc, sc := net.Pipe() + a := Attachment{ID: sandboxwire.NewID(), ServerInstanceID: testInstance, Lease: context.Background(), Exports: []sandboxlink.ExportGrant{{ID: "world"}}} + go Serve(context.Background(), sc, &describer{}, a) + c := NewClient(cc) + defer c.Close() + + ctx, cancel := context.WithCancel(context.Background()) + c.afterWrite = cancel + _, err := c.Describe(ctx, &DescribeRequest{}) + var f *Failure + if err != nil && !(errors.As(err, &f) && f.Code == CodeCancelled) { + t.Fatalf("cancelled describe: %v", err) + } + c.afterWrite = nil + if _, err := c.Describe(context.Background(), &DescribeRequest{}); err != nil || c.Err() != nil { + t.Fatalf("describe after the cancellation: %v; stream: %v", err, c.Err()) + } +} diff --git a/internal/sandboxfs/protocol.go b/internal/sandboxfs/protocol.go new file mode 100644 index 00000000..1dae79ab --- /dev/null +++ b/internal/sandboxfs/protocol.go @@ -0,0 +1,2212 @@ +// Package sandboxfs is the File access protocol between a Runtime and a sandbox +// file service. This file is its one authored definition: the request tags, +// message types, wire layouts, validators and the Service interface a file +// service implements. client.go and server.go carry the protocol over one +// stream. docs/file-access-protocol.md describes it. +package sandboxfs + +import ( + "context" + "errors" + "fmt" + "math" + + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxlink" + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxwire" +) + +// Version is the protocol version. Link matches it exactly when it opens a +// file stream. +const Version uint16 = 1 + +// Op is a request tag. Each response carries its request's tag with +// sandboxwire.ResponseType. The protocol has no events. +type Op uint16 + +const ( + OpDescribe Op = iota + 1 + OpAttach + OpDetach + OpLookup + OpWalk + OpGetAttr + OpSetAttr + OpAccess + OpOpen + OpCreate + OpRead + OpWrite + OpFlush + OpFsync + OpRelease + OpOpenDir + OpReadDir + OpReleaseDir + OpMkdir + OpUnlink + OpRmdir + OpRename + OpLink + OpSymlink + OpReadlink + OpStatFS + OpForget + OpGetLock + OpSetLock + OpCancelRequest +) + +var opNames = [...]string{ + OpDescribe: "Describe", OpAttach: "Attach", OpDetach: "Detach", + OpLookup: "Lookup", OpWalk: "Walk", OpGetAttr: "GetAttr", OpSetAttr: "SetAttr", OpAccess: "Access", + OpOpen: "Open", OpCreate: "Create", OpRead: "Read", OpWrite: "Write", OpFlush: "Flush", OpFsync: "Fsync", OpRelease: "Release", + OpOpenDir: "OpenDir", OpReadDir: "ReadDir", OpReleaseDir: "ReleaseDir", + OpMkdir: "Mkdir", OpUnlink: "Unlink", OpRmdir: "Rmdir", OpRename: "Rename", OpLink: "Link", OpSymlink: "Symlink", OpReadlink: "Readlink", + OpStatFS: "StatFS", OpForget: "Forget", OpGetLock: "GetLock", OpSetLock: "SetLock", OpCancelRequest: "CancelRequest", +} + +func (o Op) String() string { + if o >= 1 && int(o) < len(opNames) { + return opNames[o] + } + return fmt.Sprintf("Op(%d)", uint16(o)) +} + +var tags = sandboxwire.Tags{Requests: uint16(OpCancelRequest)} + +// Hard limits. Capabilities may advertise smaller ones; nothing on the wire +// exceeds these. +const ( + maxNameBytes = 1024 // an entry name, the FUSE name limit + maxTargetBytes = 4096 // a symlink target + maxWalkNames = 1024 + maxReadDirBytes = 256 << 10 + maxForgetEntries = 4096 + maxMessageBytes = 1024 + maxOffset = math.MaxInt64 +) + +// Attachment is the authenticated context Link establishes for a stream. File +// requests never carry it: the server passes it to every Service call. +type Attachment struct { + ID sandboxwire.ID + // ServerInstanceID is the service incarnation the stream was opened + // against. A service answers InstanceChanged when it differs from its own. + ServerInstanceID sandboxwire.ID + Lease Lease + // Exports are the exports the Link binding grants the attachment: at + // least one, each ID once. Attach can select only these. + Exports []sandboxlink.ExportGrant +} + +// Grant returns the attachment's grant of export id. +func (a *Attachment) Grant(id sandboxlink.ExportID) (sandboxlink.ExportGrant, bool) { + for _, g := range a.Exports { + if g.ID == id { + return g, true + } + } + return sandboxlink.ExportGrant{}, false +} + +func (a *Attachment) validate() error { + if a.ID.IsZero() || a.ServerInstanceID.IsZero() || a.Lease == nil { + return errors.New("sandboxfs: attachment without an ID, server instance or lease") + } + if len(a.Exports) == 0 { + return errors.New("sandboxfs: attachment grants no export") + } + seen := make(map[sandboxlink.ExportID]bool, len(a.Exports)) + for _, g := range a.Exports { + if !g.ID.Valid() || seen[g.ID] { + return fmt.Errorf("sandboxfs: attachment grants export %q", string(g.ID)) + } + seen[g.ID] = true + } + return nil +} + +// Lease is an attachment's authority. Done closes when the lease lapses or is +// revoked and never reopens. Losing a transport does not end a lease, so +// every stream of one attachment shares one Lease. A context.Context +// satisfies it. +type Lease interface { + Done() <-chan struct{} +} + +// Service is implemented by a file service, one method per operation. +// CancelRequest is not a method: the server answers it by cancelling the +// target's context. A method returns a *Failure for a typed failure; the server +// reports any other error as Unknown with EffectPossible. +type Service interface { + Describe(context.Context, Attachment, *DescribeRequest) (*DescribeResponse, error) + Attach(context.Context, Attachment, *AttachRequest) (*AttachResponse, error) + Detach(context.Context, Attachment, *DetachRequest) (*DetachResponse, error) + Lookup(context.Context, Attachment, *LookupRequest) (*LookupResponse, error) + Walk(context.Context, Attachment, *WalkRequest) (*WalkResponse, error) + GetAttr(context.Context, Attachment, *GetAttrRequest) (*GetAttrResponse, error) + SetAttr(context.Context, Attachment, *SetAttrRequest) (*SetAttrResponse, error) + Access(context.Context, Attachment, *AccessRequest) (*AccessResponse, error) + Open(context.Context, Attachment, *OpenRequest) (*OpenResponse, error) + Create(context.Context, Attachment, *CreateRequest) (*CreateResponse, error) + Read(context.Context, Attachment, *ReadRequest) (*ReadResponse, error) + Write(context.Context, Attachment, *WriteRequest) (*WriteResponse, error) + Flush(context.Context, Attachment, *FlushRequest) (*FlushResponse, error) + Fsync(context.Context, Attachment, *FsyncRequest) (*FsyncResponse, error) + Release(context.Context, Attachment, *ReleaseRequest) (*ReleaseResponse, error) + OpenDir(context.Context, Attachment, *OpenDirRequest) (*OpenDirResponse, error) + ReadDir(context.Context, Attachment, *ReadDirRequest) (*ReadDirResponse, error) + ReleaseDir(context.Context, Attachment, *ReleaseDirRequest) (*ReleaseDirResponse, error) + Mkdir(context.Context, Attachment, *MkdirRequest) (*MkdirResponse, error) + Unlink(context.Context, Attachment, *UnlinkRequest) (*UnlinkResponse, error) + Rmdir(context.Context, Attachment, *RmdirRequest) (*RmdirResponse, error) + Rename(context.Context, Attachment, *RenameRequest) (*RenameResponse, error) + Link(context.Context, Attachment, *LinkRequest) (*LinkResponse, error) + Symlink(context.Context, Attachment, *SymlinkRequest) (*SymlinkResponse, error) + Readlink(context.Context, Attachment, *ReadlinkRequest) (*ReadlinkResponse, error) + StatFS(context.Context, Attachment, *StatFSRequest) (*StatFSResponse, error) + Forget(context.Context, Attachment, *ForgetRequest) (*ForgetResponse, error) + GetLock(context.Context, Attachment, *GetLockRequest) (*GetLockResponse, error) + SetLock(context.Context, Attachment, *SetLockRequest) (*SetLockResponse, error) +} + +// Enumerations. Each is a uint16 on the wire, and zero is never valid. + +// Result is the discriminator every response payload begins with. +type Result uint16 + +const ( + ResultSuccess Result = 1 + ResultFailure Result = 2 +) + +type ErrorCode uint16 + +const ( + CodeInvalidArgument ErrorCode = iota + 1 + CodeUnsupported + CodeUnauthorized + CodeStaleAttachment + CodeInstanceChanged + CodeStaleNode + CodeStaleHandle + CodeResourceExhausted + CodeCancelled + CodeDeadlineExceeded + CodeErrno + CodeUnknown +) + +var codeNames = [...]string{ + CodeInvalidArgument: "InvalidArgument", CodeUnsupported: "Unsupported", CodeUnauthorized: "Unauthorized", + CodeStaleAttachment: "StaleAttachment", CodeInstanceChanged: "InstanceChanged", CodeStaleNode: "StaleNode", + CodeStaleHandle: "StaleHandle", CodeResourceExhausted: "ResourceExhausted", CodeCancelled: "Cancelled", + CodeDeadlineExceeded: "DeadlineExceeded", CodeErrno: "Errno", CodeUnknown: "Unknown", +} + +func (c ErrorCode) Valid() bool { return c >= 1 && int(c) < len(codeNames) } +func (c ErrorCode) String() string { return enumName(codeNames[:], uint16(c), "ErrorCode") } + +// Errno is the semantic error of a failed file-system call. Each adapter +// converts its native errors; an unknown native error is ErrnoIO. +type Errno uint16 + +const ( + ErrnoPermissionDenied Errno = iota + 1 + ErrnoOperationNotPermitted + ErrnoNotFound + ErrnoExists + ErrnoNotDirectory + ErrnoIsDirectory + ErrnoDirectoryNotEmpty + ErrnoInvalidArgument + ErrnoBadDescriptor + ErrnoTooManyOpenFiles + ErrnoNoSpace + ErrnoQuotaExceeded + ErrnoReadOnlyFilesystem + ErrnoCrossDevice + ErrnoNameTooLong + ErrnoSymlinkLoop + ErrnoFileTooLarge + ErrnoOverflow + ErrnoBusy + ErrnoAgain + ErrnoInterrupted + ErrnoIO + ErrnoNoDevice + ErrnoNoSuchDeviceOrAddress + ErrnoBrokenPipe + ErrnoNotSupported + ErrnoNoLocks + ErrnoDeadlock +) + +var errnoNames = [...]string{ + ErrnoPermissionDenied: "PermissionDenied", ErrnoOperationNotPermitted: "OperationNotPermitted", + ErrnoNotFound: "NotFound", ErrnoExists: "Exists", ErrnoNotDirectory: "NotDirectory", + ErrnoIsDirectory: "IsDirectory", ErrnoDirectoryNotEmpty: "DirectoryNotEmpty", + ErrnoInvalidArgument: "InvalidArgument", ErrnoBadDescriptor: "BadDescriptor", + ErrnoTooManyOpenFiles: "TooManyOpenFiles", ErrnoNoSpace: "NoSpace", ErrnoQuotaExceeded: "QuotaExceeded", + ErrnoReadOnlyFilesystem: "ReadOnlyFilesystem", ErrnoCrossDevice: "CrossDevice", + ErrnoNameTooLong: "NameTooLong", ErrnoSymlinkLoop: "SymlinkLoop", ErrnoFileTooLarge: "FileTooLarge", + ErrnoOverflow: "Overflow", ErrnoBusy: "Busy", ErrnoAgain: "Again", ErrnoInterrupted: "Interrupted", + ErrnoIO: "IO", ErrnoNoDevice: "NoDevice", ErrnoNoSuchDeviceOrAddress: "NoSuchDeviceOrAddress", + ErrnoBrokenPipe: "BrokenPipe", ErrnoNotSupported: "NotSupported", ErrnoNoLocks: "NoLocks", + ErrnoDeadlock: "Deadlock", +} + +func (e Errno) Valid() bool { return e >= 1 && int(e) < len(errnoNames) } +func (e Errno) String() string { return enumName(errnoNames[:], uint16(e), "Errno") } + +func enumName(names []string, v uint16, kind string) string { + if v >= 1 && int(v) < len(names) { + return names[v] + } + return fmt.Sprintf("%s(%d)", kind, v) +} + +// TargetKind selects what GetAttr and SetAttr address. +type TargetKind uint16 + +const ( + TargetNode TargetKind = 1 + TargetHandle TargetKind = 2 +) + +type AccessMode uint16 + +const ( + AccessRead AccessMode = 1 + AccessWrite AccessMode = 2 + AccessReadWrite AccessMode = 3 +) + +// Writes reports whether the mode opens for writing. +func (m AccessMode) Writes() bool { return m == AccessWrite || m == AccessReadWrite } + +type RenameMode uint16 + +const ( + RenameReplace RenameMode = 1 + RenameNoReplace RenameMode = 2 + RenameExchange RenameMode = 3 +) + +type LockKind uint16 + +const ( + LockPOSIX LockKind = 1 + LockFlock LockKind = 2 +) + +type LockMode uint16 + +const ( + LockRead LockMode = 1 + LockWrite LockMode = 2 + LockUnlock LockMode = 3 +) + +type PathProfile uint16 + +// PathProfileLinuxBytes: names and symlink targets are Linux byte strings. +const PathProfileLinuxBytes PathProfile = 1 + +type CacheProfile uint16 + +// CacheProfileUncached: zero entry, attribute and negative TTLs, direct I/O, +// no writeback cache and no retained directory listing on the client. +const CacheProfileUncached CacheProfile = 1 + +type Durability uint16 + +// DurabilityFsyncRequired: data is durable only after Fsync succeeds. +const DurabilityFsyncRequired Durability = 1 + +// Flag sets. Each is a uint32 on the wire; unknown bits are rejected. + +type OpenFlags uint32 + +const ( + OpenAppend OpenFlags = 1 << iota + OpenTruncate + OpenNoFollow + OpenSync + OpenDataSync + + openFlagsAll = OpenAppend | OpenTruncate | OpenNoFollow | OpenSync | OpenDataSync +) + +// AttrMask selects the attributes SetAttr changes. AttrAtimeNow and +// AttrMtimeNow set the time from the service's clock and exclude AttrAtime +// and AttrMtime respectively. +type AttrMask uint32 + +const ( + AttrSize AttrMask = 1 << iota + AttrMode + AttrUID + AttrGID + AttrAtime + AttrMtime + AttrAtimeNow + AttrMtimeNow + + attrMaskAll = AttrMtimeNow<<1 - 1 +) + +// AccessMask selects the permissions Access checks. Zero checks existence. +type AccessMask uint32 + +const ( + MayExecute AccessMask = 1 << iota + MayWrite + MayRead + + accessMaskAll = MayExecute | MayWrite | MayRead +) + +// Mode bits of Attr.Mode and DirEntry.Type, in the Linux st_mode layout. +const ( + ModeType uint32 = 0o170000 + ModeSocket uint32 = 0o140000 + ModeSymlink uint32 = 0o120000 + ModeRegular uint32 = 0o100000 + ModeBlockDevice uint32 = 0o060000 + ModeDirectory uint32 = 0o040000 + ModeCharDevice uint32 = 0o020000 + ModeFIFO uint32 = 0o010000 + ModePerm uint32 = 0o7777 +) + +func validFileType(t uint32) bool { + switch t { + case ModeSocket, ModeSymlink, ModeRegular, ModeBlockDevice, ModeDirectory, ModeCharDevice, ModeFIFO: + return true + } + return false +} + +// NodeRef names a file-system object the attachment holds lookup references +// on. An ID is reused only after its object is forgotten, and Generation then +// differs, so a stale NodeRef never names another object. Both are nonzero. +type NodeRef struct { + ID uint64 + Generation uint64 +} + +// HandleID names an open file or directory handle. It is opaque and nonzero. +type HandleID uint64 + +// LockOwner identifies a lock owner within an attachment. +type LockOwner uint64 + +type Timestamp struct { + Sec int64 + Nsec uint32 // below one billion +} + +// Attr is a file's attributes. Ino identifies the file within its export. +type Attr struct { + Ino uint64 + Mode uint32 // file type and permission bits + Nlink uint32 + UID uint32 + GID uint32 + Rdev uint64 + Size uint64 + Blocks uint64 // 512-byte blocks + Blksize uint32 + Atime Timestamp + Mtime Timestamp + Ctime Timestamp +} + +// Entry is a node the response acquired one lookup reference on, with its +// attributes. +type Entry struct { + Node NodeRef + Attr Attr +} + +// Identity is the uid and gid the service acts as. Files it creates carry +// them. It is informational: requests never select an identity. +type Identity struct { + UID uint32 + GID uint32 +} + +// Capabilities declares what a service supports. Every field is encoded. +type Capabilities struct { + PathProfile PathProfile + CacheProfile CacheProfile + Durability Durability + MaxNameBytes uint32 + MaxPathBytes uint32 // longest symlink target + MaxReadBytes uint32 + MaxWriteBytes uint32 + MaxWalkComponents uint32 + MaxReadDirBytes uint32 + MaxOpenHandles uint32 // per attachment + ReadOnly bool + AtomicAppend bool + AtomicRename bool + RenameNoReplace bool + RenameExchange bool + HardLinks bool + Symlinks bool + SetMode bool + SetOwner bool + SetTimes bool + DirectoryFsync bool + ReadDirPlus bool + Flock bool + POSIXLocks bool +} + +// Target is the node or open handle GetAttr and SetAttr address. Only the +// field Kind selects is set. +type Target struct { + Kind TargetKind + Node NodeRef + Handle HandleID +} + +// DirEntry is one directory entry. Cookie is the position after it: ReadDir +// resumes there. Entry is set when ReadDir asked WithAttrs, and then carries +// one lookup reference. +type DirEntry struct { + Name []byte + Ino uint64 + Type uint32 // one ModeType value + Cookie uint64 + Entry *Entry +} + +const ( + attrWireSize = 8 + 4 + 4 + 4 + 4 + 8 + 8 + 8 + 4 + 3*(8+4) + entryWireSize = 16 + attrWireSize + dirEntryWireSize = 4 + 8 + 4 + 8 + 1 +) + +// WireSize is the entry's encoded length. ReadDir's Limit bounds the sum of +// the returned entries' sizes. +func (e *DirEntry) WireSize() int { + n := dirEntryWireSize + len(e.Name) + if e.Entry != nil { + n += entryWireSize + } + return n +} + +type ForgetEntry struct { + Node NodeRef + Count uint64 +} + +// Lock is a byte range lock: Start through End inclusive, where End +// math.MaxInt64 extends to the end of the file. +type Lock struct { + Mode LockMode + Start uint64 + End uint64 +} + +// Failure is a typed failure. Errno is set exactly when Code is CodeErrno. +// Effect says whether the failed request may have taken effect. +type Failure struct { + Code ErrorCode + Errno Errno + Effect sandboxwire.Effect + Message string + + cause error // set on failures the client produces; never encoded +} + +// NewFailure returns a failure with code, effect and a message cut to the +// protocol limit. +func NewFailure(code ErrorCode, effect sandboxwire.Effect, message string) *Failure { + return &Failure{Code: code, Effect: effect, Message: clip(message)} +} + +// NewErrnoFailure returns a CodeErrno failure. +func NewErrnoFailure(errno Errno, effect sandboxwire.Effect, message string) *Failure { + return &Failure{Code: CodeErrno, Errno: errno, Effect: effect, Message: clip(message)} +} + +func clip(s string) string { + if len(s) > maxMessageBytes { + return s[:maxMessageBytes] + } + return s +} + +func (f *Failure) Error() string { + s := "sandboxfs: " + f.Code.String() + if f.Code == CodeErrno { + s += " " + f.Errno.String() + } + if f.Message != "" { + s += ": " + f.Message + } + if f.Effect == sandboxwire.EffectPossible { + s += " (effect possible)" + } + return s +} + +func (f *Failure) Unwrap() error { return f.cause } + +// Requests and responses, in tag order. + +type DescribeRequest struct{} + +type DescribeResponse struct { + ServerInstanceID sandboxwire.ID + Identity Identity + Capabilities Capabilities + Exports []sandboxlink.ExportID +} + +// AttachRequest selects a declared export that the attachment's Link grant +// includes. A read-only grant requires ReadOnly. An attachment attaches once +// until it detaches. +type AttachRequest struct { + Export sandboxlink.ExportID + ReadOnly bool +} + +type AttachResponse struct { + Root Entry +} + +// DetachRequest releases every node, handle and lock of the attachment. +type DetachRequest struct{} + +type DetachResponse struct{} + +type LookupRequest struct { + Parent NodeRef + Name []byte +} + +type LookupResponse struct { + Entry Entry +} + +// WalkRequest looks up Names in turn from Parent. The walk stops after a +// symlink, so fewer entries than names with no Failure means the last entry +// is a symlink. +type WalkRequest struct { + Parent NodeRef + Names [][]byte +} + +// WalkResponse holds the entries walked, at least one. Failure is the error +// that stopped the walk after them. +type WalkResponse struct { + Entries []Entry + Failure *Failure +} + +type GetAttrRequest struct { + Target Target +} + +type GetAttrResponse struct { + Attr Attr +} + +// SetAttrRequest changes the attributes Set selects; the other fields are +// zero. On the wire each selected value follows the mask in field order. +type SetAttrRequest struct { + Target Target + Set AttrMask + Size uint64 + Mode uint32 // permission bits + UID uint32 + GID uint32 + Atime Timestamp + Mtime Timestamp +} + +type SetAttrResponse struct { + Attr Attr +} + +type AccessRequest struct { + Node NodeRef + Mask AccessMask +} + +type AccessResponse struct{} + +type OpenRequest struct { + Node NodeRef + Access AccessMode + Flags OpenFlags +} + +type OpenResponse struct { + Handle HandleID +} + +type CreateRequest struct { + Parent NodeRef + Name []byte + Mode uint32 // permission bits, applied as given + Access AccessMode + Flags OpenFlags + Exclusive bool +} + +type CreateResponse struct { + Entry Entry + Handle HandleID +} + +type ReadRequest struct { + Handle HandleID + Offset uint64 + Size uint32 +} + +// ReadResponse holds the bytes read; fewer than asked means end of file. +type ReadResponse struct { + Data []byte +} + +// WriteRequest writes Data at Offset. An append-opened handle appends +// atomically and ignores Offset. +type WriteRequest struct { + Handle HandleID + Offset uint64 + Data []byte +} + +// WriteResponse reports the bytes written. Failure is the error that stopped +// the write after that nonzero prefix. +type WriteResponse struct { + Written uint32 + Failure *Failure +} + +// FlushRequest runs on each close of a descriptor for the handle. Owner is the +// closing lock owner, whose POSIX locks on the file it releases. +type FlushRequest struct { + Handle HandleID + Owner LockOwner +} + +type FlushResponse struct{} + +type FsyncRequest struct { + Handle HandleID + DataOnly bool +} + +type FsyncResponse struct{} + +type ReleaseRequest struct { + Handle HandleID +} + +type ReleaseResponse struct{} + +type OpenDirRequest struct { + Node NodeRef +} + +type OpenDirResponse struct { + Handle HandleID +} + +// ReadDirRequest reads entries after Cookie; cookie zero is the start. Limit +// bounds the entries' WireSize sum. "." and ".." are never returned. +type ReadDirRequest struct { + Handle HandleID + Cookie uint64 + Limit uint32 + WithAttrs bool +} + +type ReadDirResponse struct { + Entries []DirEntry + End bool +} + +type ReleaseDirRequest struct { + Handle HandleID +} + +type ReleaseDirResponse struct{} + +type MkdirRequest struct { + Parent NodeRef + Name []byte + Mode uint32 // permission bits, applied as given +} + +type MkdirResponse struct { + Entry Entry +} + +type UnlinkRequest struct { + Parent NodeRef + Name []byte +} + +type UnlinkResponse struct{} + +type RmdirRequest struct { + Parent NodeRef + Name []byte +} + +type RmdirResponse struct{} + +type RenameRequest struct { + Parent NodeRef + Name []byte + NewParent NodeRef + NewName []byte + Mode RenameMode +} + +type RenameResponse struct{} + +type LinkRequest struct { + Node NodeRef + NewParent NodeRef + NewName []byte +} + +type LinkResponse struct { + Entry Entry +} + +type SymlinkRequest struct { + Parent NodeRef + Name []byte + Target []byte +} + +type SymlinkResponse struct { + Entry Entry +} + +type ReadlinkRequest struct { + Node NodeRef +} + +type ReadlinkResponse struct { + Target []byte +} + +type StatFSRequest struct { + Node NodeRef +} + +type StatFSResponse struct { + Blocks uint64 + BlocksFree uint64 + BlocksAvailable uint64 + Files uint64 + FilesFree uint64 + BlockSize uint32 + FragmentSize uint32 + NameMax uint32 +} + +// ForgetRequest releases lookup references. It applies entirely or not at +// all. +type ForgetRequest struct { + Entries []ForgetEntry +} + +type ForgetResponse struct{} + +// GetLockRequest asks for a POSIX lock that would conflict with Lock. +type GetLockRequest struct { + Handle HandleID + Owner LockOwner + Lock Lock +} + +// GetLockResponse holds the conflicting lock, if any. +type GetLockResponse struct { + Conflict *Lock +} + +// SetLockRequest acquires or releases a lock. A LockFlock lock covers the +// whole file. Wait blocks until the lock is available or the request is +// cancelled. +type SetLockRequest struct { + Handle HandleID + Kind LockKind + Owner LockOwner + Lock Lock + Wait bool +} + +type SetLockResponse struct{} + +// CancelRequestRequest asks to cancel the outstanding request Target. Its +// acknowledgement says nothing about the target's outcome. +type CancelRequestRequest struct { + Target uint64 +} + +type CancelRequestResponse struct{} + +func (*DescribeRequest) Op() Op { return OpDescribe } +func (*AttachRequest) Op() Op { return OpAttach } +func (*DetachRequest) Op() Op { return OpDetach } +func (*LookupRequest) Op() Op { return OpLookup } +func (*WalkRequest) Op() Op { return OpWalk } +func (*GetAttrRequest) Op() Op { return OpGetAttr } +func (*SetAttrRequest) Op() Op { return OpSetAttr } +func (*AccessRequest) Op() Op { return OpAccess } +func (*OpenRequest) Op() Op { return OpOpen } +func (*CreateRequest) Op() Op { return OpCreate } +func (*ReadRequest) Op() Op { return OpRead } +func (*WriteRequest) Op() Op { return OpWrite } +func (*FlushRequest) Op() Op { return OpFlush } +func (*FsyncRequest) Op() Op { return OpFsync } +func (*ReleaseRequest) Op() Op { return OpRelease } +func (*OpenDirRequest) Op() Op { return OpOpenDir } +func (*ReadDirRequest) Op() Op { return OpReadDir } +func (*ReleaseDirRequest) Op() Op { return OpReleaseDir } +func (*MkdirRequest) Op() Op { return OpMkdir } +func (*UnlinkRequest) Op() Op { return OpUnlink } +func (*RmdirRequest) Op() Op { return OpRmdir } +func (*RenameRequest) Op() Op { return OpRename } +func (*LinkRequest) Op() Op { return OpLink } +func (*SymlinkRequest) Op() Op { return OpSymlink } +func (*ReadlinkRequest) Op() Op { return OpReadlink } +func (*StatFSRequest) Op() Op { return OpStatFS } +func (*ForgetRequest) Op() Op { return OpForget } +func (*GetLockRequest) Op() Op { return OpGetLock } +func (*SetLockRequest) Op() Op { return OpSetLock } +func (*CancelRequestRequest) Op() Op { return OpCancelRequest } + +// Request is implemented by every request type. +type Request interface { + message + Op() Op +} + +type message interface { + encode(*sandboxwire.Encoder) + decode(*decoder) + validate() error +} + +// opSpec binds an operation's request, response and Service method. +type opSpec struct { + newRequest func() Request + newResponse func() message + serve func(context.Context, Service, Attachment, Request) (message, error) +} + +var errNilResponse = errors.New("service returned no response") + +func bind[QT, RT any, Q interface { + *QT + Request +}, R interface { + *RT + message +}](method func(Service, context.Context, Attachment, Q) (R, error)) opSpec { + return opSpec{ + newRequest: func() Request { return Q(new(QT)) }, + newResponse: func() message { return R(new(RT)) }, + serve: func(ctx context.Context, s Service, a Attachment, q Request) (message, error) { + r, err := method(s, ctx, a, q.(Q)) + if err != nil { + return nil, err + } + if r == nil { + return nil, errNilResponse + } + return r, nil + }, + } +} + +// opSpecs is indexed by Op; each request type's Op method places its entry. +var opSpecs = func() (t [OpCancelRequest + 1]opSpec) { + for _, s := range []opSpec{ + bind(Service.Describe), bind(Service.Attach), bind(Service.Detach), + bind(Service.Lookup), bind(Service.Walk), bind(Service.GetAttr), bind(Service.SetAttr), bind(Service.Access), + bind(Service.Open), bind(Service.Create), bind(Service.Read), bind(Service.Write), + bind(Service.Flush), bind(Service.Fsync), bind(Service.Release), + bind(Service.OpenDir), bind(Service.ReadDir), bind(Service.ReleaseDir), + bind(Service.Mkdir), bind(Service.Unlink), bind(Service.Rmdir), bind(Service.Rename), + bind(Service.Link), bind(Service.Symlink), bind(Service.Readlink), + bind(Service.StatFS), bind(Service.Forget), bind(Service.GetLock), bind(Service.SetLock), + { + newRequest: func() Request { return new(CancelRequestRequest) }, + newResponse: func() message { return new(CancelRequestResponse) }, + }, + } { + t[s.newRequest().Op()] = s + } + return t +}() + +// sideEffectFree reports whether r leaves no state behind, so abandoning it +// cannot have had an effect. Lookups and ReadDir WithAttrs acquire references. +func sideEffectFree(r Request) bool { + switch r := r.(type) { + case *DescribeRequest, *GetAttrRequest, *AccessRequest, *ReadRequest, *ReadlinkRequest, *StatFSRequest, *GetLockRequest: + return true + case *ReadDirRequest: + return !r.WithAttrs + } + return false +} + +// modifiesFiles reports whether r changes the file system, which a read-only +// attachment refuses. +func modifiesFiles(r Request) bool { + switch r := r.(type) { + case *SetAttrRequest, *CreateRequest, *WriteRequest, *MkdirRequest, *UnlinkRequest, *RmdirRequest, + *RenameRequest, *LinkRequest, *SymlinkRequest: + return true + case *OpenRequest: + return r.Access.Writes() || r.Flags&OpenTruncate != 0 + } + return false +} + +// Admit checks r against the declared capabilities and the attachment's +// read-only choice. A service calls it before running a request and returns +// the failure it reports. Limits a request can only be judged against service +// state, such as MaxOpenHandles, stay with the service. +func (c *Capabilities) Admit(r Request, readOnly bool) *Failure { + unsupported := func(what string) *Failure { + return NewFailure(CodeUnsupported, sandboxwire.EffectNone, what+" is not supported") + } + tooLong := func(n []byte, max uint32) bool { return uint64(len(n)) > uint64(max) } + nameTooLong := NewErrnoFailure(ErrnoNameTooLong, sandboxwire.EffectNone, "name exceeds MaxNameBytes") + if readOnly && modifiesFiles(r) { + return NewErrnoFailure(ErrnoReadOnlyFilesystem, sandboxwire.EffectNone, "attachment is read-only") + } + switch r := r.(type) { + case *AttachRequest: + if !r.ReadOnly && c.ReadOnly { + return unsupported("a writable attachment") + } + case *LookupRequest: + if tooLong(r.Name, c.MaxNameBytes) { + return nameTooLong + } + case *WalkRequest: + if uint64(len(r.Names)) > uint64(c.MaxWalkComponents) { + return NewFailure(CodeInvalidArgument, sandboxwire.EffectNone, "walk exceeds MaxWalkComponents") + } + for _, n := range r.Names { + if tooLong(n, c.MaxNameBytes) { + return nameTooLong + } + } + case *SetAttrRequest: + switch { + case r.Set&AttrMode != 0 && !c.SetMode: + return unsupported("setting the mode") + case r.Set&(AttrUID|AttrGID) != 0 && !c.SetOwner: + return unsupported("setting the owner") + case r.Set&(AttrAtime|AttrMtime|AttrAtimeNow|AttrMtimeNow) != 0 && !c.SetTimes: + return unsupported("setting times") + } + case *CreateRequest: + if tooLong(r.Name, c.MaxNameBytes) { + return nameTooLong + } + case *ReadRequest: + if r.Size > c.MaxReadBytes { + return NewFailure(CodeInvalidArgument, sandboxwire.EffectNone, "read exceeds MaxReadBytes") + } + case *WriteRequest: + if tooLong(r.Data, c.MaxWriteBytes) { + return NewFailure(CodeInvalidArgument, sandboxwire.EffectNone, "write exceeds MaxWriteBytes") + } + case *ReadDirRequest: + if r.Limit > c.MaxReadDirBytes { + return NewFailure(CodeInvalidArgument, sandboxwire.EffectNone, "limit exceeds MaxReadDirBytes") + } + if r.WithAttrs && !c.ReadDirPlus { + return unsupported("ReadDir WithAttrs") + } + case *MkdirRequest: + if tooLong(r.Name, c.MaxNameBytes) { + return nameTooLong + } + case *UnlinkRequest: + if tooLong(r.Name, c.MaxNameBytes) { + return nameTooLong + } + case *RmdirRequest: + if tooLong(r.Name, c.MaxNameBytes) { + return nameTooLong + } + case *RenameRequest: + switch { + case tooLong(r.Name, c.MaxNameBytes) || tooLong(r.NewName, c.MaxNameBytes): + return nameTooLong + case r.Mode == RenameNoReplace && !c.RenameNoReplace: + return unsupported("RenameNoReplace") + case r.Mode == RenameExchange && !c.RenameExchange: + return unsupported("RenameExchange") + } + case *LinkRequest: + if !c.HardLinks { + return unsupported("Link") + } + if tooLong(r.NewName, c.MaxNameBytes) { + return nameTooLong + } + case *SymlinkRequest: + switch { + case !c.Symlinks: + return unsupported("Symlink") + case tooLong(r.Name, c.MaxNameBytes): + return nameTooLong + case tooLong(r.Target, c.MaxPathBytes): + return NewErrnoFailure(ErrnoNameTooLong, sandboxwire.EffectNone, "target exceeds MaxPathBytes") + } + case *GetLockRequest: + if !c.POSIXLocks { + return unsupported("POSIX locks") + } + case *SetLockRequest: + if r.Kind == LockPOSIX && !c.POSIXLocks { + return unsupported("POSIX locks") + } + if r.Kind == LockFlock && !c.Flock { + return unsupported("flock") + } + } + return nil +} + +// Codec. Every message has an ordered layout of sandboxwire primitives. +// Decoding is structural; validate then applies the protocol rules, and +// encoding validates first, so both directions enforce them. + +func malformed(format string, args ...any) error { + return fmt.Errorf("%w: "+format, append([]any{sandboxwire.ErrMalformed}, args...)...) +} + +// decoder keeps the first error and reads nothing after it. +type decoder struct { + d *sandboxwire.Decoder + err error +} + +func read[T any](d *decoder, f func() (T, error)) T { + var v T + if d.err == nil { + v, d.err = f() + } + return v +} + +func (d *decoder) u16() uint16 { return read(d, d.d.U16) } +func (d *decoder) u32() uint32 { return read(d, d.d.U32) } +func (d *decoder) u64() uint64 { return read(d, d.d.U64) } +func (d *decoder) i64() int64 { return read(d, d.d.I64) } +func (d *decoder) bool() bool { return read(d, d.d.Bool) } +func (d *decoder) bytes() []byte { return read(d, d.d.Bytes) } +func (d *decoder) present() bool { return read(d, d.d.Present) } +func (d *decoder) id() sandboxwire.ID { return read(d, d.d.ID) } + +func (d *decoder) count(max uint32) int { + return read(d, func() (int, error) { return d.d.Count(max) }) +} + +func decodeMessage(m message, payload []byte) error { + d := &decoder{d: sandboxwire.NewDecoder(payload)} + m.decode(d) + if d.err != nil { + return d.err + } + if err := d.d.Finish(); err != nil { + return err + } + return m.validate() +} + +func encodeMessage(e *sandboxwire.Encoder, m message) error { + if err := m.validate(); err != nil { + return err + } + m.encode(e) + return nil +} + +func checkPayload(p []byte) ([]byte, error) { + if len(p) > sandboxwire.MaxPayload { + return nil, malformed("payload of %d bytes exceeds the frame limit", len(p)) + } + return p, nil +} + +func encodeRequest(r Request) ([]byte, error) { + var e sandboxwire.Encoder + if err := encodeMessage(&e, r); err != nil { + return nil, err + } + return checkPayload(e.Payload()) +} + +func decodeRequest(op Op, payload []byte) (Request, error) { + if op < 1 || op > OpCancelRequest { + return nil, malformed("unknown request %d", uint16(op)) + } + r := opSpecs[op].newRequest() + return r, decodeMessage(r, payload) +} + +// encodeResponse encodes a success carrying r, or the failure f when it is +// set. +func encodeResponse(r message, f *Failure) ([]byte, error) { + var e sandboxwire.Encoder + if f != nil { + if err := f.validate(); err != nil { + return nil, err + } + e.Enum(uint16(ResultFailure)) + f.encode(&e) + } else { + if err := r.validate(); err != nil { + return nil, err + } + e.Enum(uint16(ResultSuccess)) + r.encode(&e) + } + return checkPayload(e.Payload()) +} + +func decodeResponse(op Op, payload []byte) (message, *Failure, error) { + if op < 1 || op > OpCancelRequest { + return nil, nil, malformed("unknown request %d", uint16(op)) + } + d := &decoder{d: sandboxwire.NewDecoder(payload)} + switch result := Result(d.u16()); { + case d.err != nil: + return nil, nil, d.err + case result == ResultSuccess: + r := opSpecs[op].newResponse() + return r, nil, decodeRest(d, r) + case result == ResultFailure: + f := new(Failure) + return nil, f, decodeRest(d, f) + default: + return nil, nil, malformed("result %d", result) + } +} + +func decodeRest(d *decoder, m message) error { + m.decode(d) + if d.err != nil { + return d.err + } + if err := d.d.Finish(); err != nil { + return err + } + return m.validate() +} + +// Shared values. + +func (r NodeRef) encode(e *sandboxwire.Encoder) { e.U64(r.ID); e.U64(r.Generation) } +func (r *NodeRef) decode(d *decoder) { r.ID = d.u64(); r.Generation = d.u64() } +func (r NodeRef) validate() error { + if r.ID == 0 || r.Generation == 0 { + return malformed("node reference %d/%d", r.ID, r.Generation) + } + return nil +} + +func (h HandleID) validate() error { + if h == 0 { + return malformed("zero handle") + } + return nil +} + +func (t Timestamp) encode(e *sandboxwire.Encoder) { e.I64(t.Sec); e.U32(t.Nsec) } +func (t *Timestamp) decode(d *decoder) { t.Sec = d.i64(); t.Nsec = d.u32() } +func (t Timestamp) validate() error { + if t.Nsec >= 1e9 { + return malformed("nanoseconds %d", t.Nsec) + } + return nil +} + +func (a *Attr) encode(e *sandboxwire.Encoder) { + e.U64(a.Ino) + e.U32(a.Mode) + e.U32(a.Nlink) + e.U32(a.UID) + e.U32(a.GID) + e.U64(a.Rdev) + e.U64(a.Size) + e.U64(a.Blocks) + e.U32(a.Blksize) + a.Atime.encode(e) + a.Mtime.encode(e) + a.Ctime.encode(e) +} + +func (a *Attr) decode(d *decoder) { + a.Ino = d.u64() + a.Mode = d.u32() + a.Nlink = d.u32() + a.UID = d.u32() + a.GID = d.u32() + a.Rdev = d.u64() + a.Size = d.u64() + a.Blocks = d.u64() + a.Blksize = d.u32() + a.Atime.decode(d) + a.Mtime.decode(d) + a.Ctime.decode(d) +} + +func (a *Attr) validate() error { + if !validFileType(a.Mode&ModeType) || a.Mode&^(ModeType|ModePerm) != 0 { + return malformed("mode %#o", a.Mode) + } + if a.Size > maxOffset { + return malformed("size %d", a.Size) + } + return errors.Join(a.Atime.validate(), a.Mtime.validate(), a.Ctime.validate()) +} + +func (en *Entry) encode(e *sandboxwire.Encoder) { en.Node.encode(e); en.Attr.encode(e) } +func (en *Entry) decode(d *decoder) { en.Node.decode(d); en.Attr.decode(d) } +func (en *Entry) validate() error { return errors.Join(en.Node.validate(), en.Attr.validate()) } + +func (t *Target) encode(e *sandboxwire.Encoder) { + e.Enum(uint16(t.Kind)) + if t.Kind == TargetNode { + t.Node.encode(e) + } else { + e.U64(uint64(t.Handle)) + } +} + +func (t *Target) decode(d *decoder) { + switch t.Kind = TargetKind(d.u16()); t.Kind { + case TargetNode: + t.Node.decode(d) + case TargetHandle: + t.Handle = HandleID(d.u64()) + } +} + +func (t *Target) validate() error { + switch t.Kind { + case TargetNode: + if t.Handle != 0 { + return malformed("node target with a handle") + } + return t.Node.validate() + case TargetHandle: + if t.Node != (NodeRef{}) { + return malformed("handle target with a node") + } + return t.Handle.validate() + } + return malformed("target kind %d", t.Kind) +} + +func (l *Lock) encode(e *sandboxwire.Encoder) { e.Enum(uint16(l.Mode)); e.U64(l.Start); e.U64(l.End) } +func (l *Lock) decode(d *decoder) { l.Mode = LockMode(d.u16()); l.Start = d.u64(); l.End = d.u64() } +func (l *Lock) validate() error { + if l.Mode < LockRead || l.Mode > LockUnlock { + return malformed("lock mode %d", l.Mode) + } + if l.Start > l.End || l.End > maxOffset { + return malformed("lock range %d-%d", l.Start, l.End) + } + return nil +} + +func (f *Failure) encode(e *sandboxwire.Encoder) { + e.Enum(uint16(f.Code)) + e.Present(f.Code == CodeErrno) + if f.Code == CodeErrno { + e.Enum(uint16(f.Errno)) + } + e.Effect(f.Effect) + e.Bytes([]byte(f.Message)) +} + +func (f *Failure) decode(d *decoder) { + f.Code = ErrorCode(d.u16()) + if d.present() { + if f.Errno = Errno(d.u16()); f.Errno == 0 && d.err == nil { + d.err = malformed("zero errno") + } + } + f.Effect = sandboxwire.Effect(d.u16()) + f.Message = string(d.bytes()) +} + +func (f *Failure) validate() error { + switch { + case !f.Code.Valid(): + return malformed("error code %d", f.Code) + case (f.Code == CodeErrno) != (f.Errno != 0): + return malformed("errno %d with code %s", f.Errno, f.Code) + case f.Errno != 0 && !f.Errno.Valid(): + return malformed("errno %d", f.Errno) + case !f.Effect.Valid(): + return malformed("effect %d", f.Effect) + case len(f.Message) > maxMessageBytes: + return malformed("message of %d bytes", len(f.Message)) + } + return nil +} + +func encodeOptionalFailure(e *sandboxwire.Encoder, f *Failure) { + e.Present(f != nil) + if f != nil { + f.encode(e) + } +} + +func decodeOptionalFailure(d *decoder) *Failure { + if !d.present() { + return nil + } + f := new(Failure) + f.decode(d) + return f +} + +func validName(n []byte) error { + if len(n) == 0 || len(n) > maxNameBytes { + return malformed("name of %d bytes", len(n)) + } + if string(n) == "." || string(n) == ".." { + return malformed("name %q", n) + } + for _, c := range n { + if c == 0 || c == '/' { + return malformed("name contains %q", c) + } + } + return nil +} + +func validTarget(t []byte) error { + if len(t) == 0 || len(t) > maxTargetBytes { + return malformed("symlink target of %d bytes", len(t)) + } + for _, c := range t { + if c == 0 { + return malformed("symlink target contains NUL") + } + } + return nil +} + +func validPerm(m uint32) error { + if m&^ModePerm != 0 { + return malformed("permission bits %#o", m) + } + return nil +} + +func validAccess(m AccessMode) error { + if m < AccessRead || m > AccessReadWrite { + return malformed("access mode %d", m) + } + return nil +} + +func validOpenFlags(f OpenFlags) error { + if f&^openFlagsAll != 0 { + return malformed("open flags %#x", uint32(f)) + } + return nil +} + +func (c *Capabilities) encode(e *sandboxwire.Encoder) { + e.Enum(uint16(c.PathProfile)) + e.Enum(uint16(c.CacheProfile)) + e.Enum(uint16(c.Durability)) + for _, v := range []uint32{c.MaxNameBytes, c.MaxPathBytes, c.MaxReadBytes, c.MaxWriteBytes, c.MaxWalkComponents, c.MaxReadDirBytes, c.MaxOpenHandles} { + e.U32(v) + } + for _, v := range c.flags() { + e.Bool(*v) + } +} + +func (c *Capabilities) decode(d *decoder) { + c.PathProfile = PathProfile(d.u16()) + c.CacheProfile = CacheProfile(d.u16()) + c.Durability = Durability(d.u16()) + for _, v := range []*uint32{&c.MaxNameBytes, &c.MaxPathBytes, &c.MaxReadBytes, &c.MaxWriteBytes, &c.MaxWalkComponents, &c.MaxReadDirBytes, &c.MaxOpenHandles} { + *v = d.u32() + } + for _, v := range c.flags() { + *v = d.bool() + } +} + +// flags lists the boolean capabilities in wire order. +func (c *Capabilities) flags() []*bool { + return []*bool{&c.ReadOnly, &c.AtomicAppend, &c.AtomicRename, &c.RenameNoReplace, &c.RenameExchange, + &c.HardLinks, &c.Symlinks, &c.SetMode, &c.SetOwner, &c.SetTimes, &c.DirectoryFsync, &c.ReadDirPlus, + &c.Flock, &c.POSIXLocks} +} + +func (c *Capabilities) validate() error { + within := func(v, max uint32) bool { return v >= 1 && v <= max } + switch { + case c.PathProfile != PathProfileLinuxBytes: + return malformed("path profile %d", c.PathProfile) + case c.CacheProfile != CacheProfileUncached: + return malformed("cache profile %d", c.CacheProfile) + case c.Durability != DurabilityFsyncRequired: + return malformed("durability %d", c.Durability) + case !within(c.MaxNameBytes, maxNameBytes), !within(c.MaxPathBytes, maxTargetBytes), + !within(c.MaxReadBytes, sandboxwire.MaxChunk), !within(c.MaxWriteBytes, sandboxwire.MaxChunk), + !within(c.MaxWalkComponents, maxWalkNames), !within(c.MaxReadDirBytes, maxReadDirBytes), + c.MaxOpenHandles == 0: + return malformed("capability limit out of range") + case !c.ReadOnly && !(c.AtomicAppend && c.AtomicRename && c.HardLinks && c.Symlinks): + return malformed("a writable service requires AtomicAppend, AtomicRename, HardLinks and Symlinks") + } + return nil +} + +// Messages. + +func (*DescribeRequest) encode(*sandboxwire.Encoder) {} +func (*DescribeRequest) decode(*decoder) {} +func (*DescribeRequest) validate() error { return nil } + +func (r *DescribeResponse) encode(e *sandboxwire.Encoder) { + e.ID(r.ServerInstanceID) + e.U32(r.Identity.UID) + e.U32(r.Identity.GID) + r.Capabilities.encode(e) + e.Count(len(r.Exports)) + for _, x := range r.Exports { + e.Bytes([]byte(x)) + } +} + +func (r *DescribeResponse) decode(d *decoder) { + r.ServerInstanceID = d.id() + r.Identity.UID = d.u32() + r.Identity.GID = d.u32() + r.Capabilities.decode(d) + r.Exports = make([]sandboxlink.ExportID, d.count(sandboxlink.MaxExports)) + for i := range r.Exports { + r.Exports[i] = sandboxlink.ExportID(d.bytes()) + } +} + +func (r *DescribeResponse) validate() error { + if r.ServerInstanceID.IsZero() { + return malformed("zero server instance") + } + if len(r.Exports) > sandboxlink.MaxExports { + return malformed("%d exports", len(r.Exports)) + } + seen := make(map[sandboxlink.ExportID]bool, len(r.Exports)) + for _, x := range r.Exports { + if !x.Valid() || seen[x] { + return malformed("export %q", string(x)) + } + seen[x] = true + } + return r.Capabilities.validate() +} + +func (r *AttachRequest) encode(e *sandboxwire.Encoder) { e.Bytes([]byte(r.Export)); e.Bool(r.ReadOnly) } +func (r *AttachRequest) decode(d *decoder) { + r.Export = sandboxlink.ExportID(d.bytes()) + r.ReadOnly = d.bool() +} +func (r *AttachRequest) validate() error { + if !r.Export.Valid() { + return malformed("export %q", string(r.Export)) + } + return nil +} + +func (r *AttachResponse) encode(e *sandboxwire.Encoder) { r.Root.encode(e) } +func (r *AttachResponse) decode(d *decoder) { r.Root.decode(d) } +func (r *AttachResponse) validate() error { + if r.Root.Attr.Mode&ModeType != ModeDirectory { + return malformed("export root is not a directory") + } + return r.Root.validate() +} + +func (*DetachRequest) encode(*sandboxwire.Encoder) {} +func (*DetachRequest) decode(*decoder) {} +func (*DetachRequest) validate() error { return nil } +func (*DetachResponse) encode(*sandboxwire.Encoder) {} +func (*DetachResponse) decode(*decoder) {} +func (*DetachResponse) validate() error { return nil } + +func (r *LookupRequest) encode(e *sandboxwire.Encoder) { r.Parent.encode(e); e.Bytes(r.Name) } +func (r *LookupRequest) decode(d *decoder) { r.Parent.decode(d); r.Name = d.bytes() } +func (r *LookupRequest) validate() error { return errors.Join(r.Parent.validate(), validName(r.Name)) } + +func (r *LookupResponse) encode(e *sandboxwire.Encoder) { r.Entry.encode(e) } +func (r *LookupResponse) decode(d *decoder) { r.Entry.decode(d) } +func (r *LookupResponse) validate() error { return r.Entry.validate() } + +func (r *WalkRequest) encode(e *sandboxwire.Encoder) { + r.Parent.encode(e) + e.Count(len(r.Names)) + for _, n := range r.Names { + e.Bytes(n) + } +} + +func (r *WalkRequest) decode(d *decoder) { + r.Parent.decode(d) + r.Names = make([][]byte, d.count(maxWalkNames)) + for i := range r.Names { + r.Names[i] = d.bytes() + } +} + +func (r *WalkRequest) validate() error { + if len(r.Names) == 0 || len(r.Names) > maxWalkNames { + return malformed("walk of %d names", len(r.Names)) + } + for _, n := range r.Names { + if err := validName(n); err != nil { + return err + } + } + return r.Parent.validate() +} + +func (r *WalkResponse) encode(e *sandboxwire.Encoder) { + e.Count(len(r.Entries)) + for i := range r.Entries { + r.Entries[i].encode(e) + } + encodeOptionalFailure(e, r.Failure) +} + +func (r *WalkResponse) decode(d *decoder) { + r.Entries = make([]Entry, d.count(maxWalkNames)) + for i := range r.Entries { + r.Entries[i].decode(d) + } + r.Failure = decodeOptionalFailure(d) +} + +func (r *WalkResponse) validate() error { + if len(r.Entries) == 0 || len(r.Entries) > maxWalkNames { + return malformed("walk of %d entries", len(r.Entries)) + } + for i := range r.Entries { + if err := r.Entries[i].validate(); err != nil { + return err + } + } + if r.Failure != nil { + return r.Failure.validate() + } + return nil +} + +func (r *GetAttrRequest) encode(e *sandboxwire.Encoder) { r.Target.encode(e) } +func (r *GetAttrRequest) decode(d *decoder) { r.Target.decode(d) } +func (r *GetAttrRequest) validate() error { return r.Target.validate() } + +func (r *GetAttrResponse) encode(e *sandboxwire.Encoder) { r.Attr.encode(e) } +func (r *GetAttrResponse) decode(d *decoder) { r.Attr.decode(d) } +func (r *GetAttrResponse) validate() error { return r.Attr.validate() } + +func (r *SetAttrRequest) encode(e *sandboxwire.Encoder) { + r.Target.encode(e) + e.U32(uint32(r.Set)) + if r.Set&AttrSize != 0 { + e.U64(r.Size) + } + if r.Set&AttrMode != 0 { + e.U32(r.Mode) + } + if r.Set&AttrUID != 0 { + e.U32(r.UID) + } + if r.Set&AttrGID != 0 { + e.U32(r.GID) + } + if r.Set&AttrAtime != 0 { + r.Atime.encode(e) + } + if r.Set&AttrMtime != 0 { + r.Mtime.encode(e) + } +} + +func (r *SetAttrRequest) decode(d *decoder) { + r.Target.decode(d) + r.Set = AttrMask(d.u32()) + if r.Set&AttrSize != 0 { + r.Size = d.u64() + } + if r.Set&AttrMode != 0 { + r.Mode = d.u32() + } + if r.Set&AttrUID != 0 { + r.UID = d.u32() + } + if r.Set&AttrGID != 0 { + r.GID = d.u32() + } + if r.Set&AttrAtime != 0 { + r.Atime.decode(d) + } + if r.Set&AttrMtime != 0 { + r.Mtime.decode(d) + } +} + +func (r *SetAttrRequest) validate() error { + unset := func(bit AttrMask, zero bool) bool { return r.Set&bit == 0 && !zero } + switch { + case r.Set&^attrMaskAll != 0: + return malformed("attribute mask %#x", uint32(r.Set)) + case r.Set&AttrAtime != 0 && r.Set&AttrAtimeNow != 0, r.Set&AttrMtime != 0 && r.Set&AttrMtimeNow != 0: + return malformed("a time is both given and now") + case unset(AttrSize, r.Size == 0), unset(AttrMode, r.Mode == 0), unset(AttrUID, r.UID == 0), + unset(AttrGID, r.GID == 0), unset(AttrAtime, r.Atime == Timestamp{}), unset(AttrMtime, r.Mtime == Timestamp{}): + return malformed("an unselected attribute is set") + case r.Size > maxOffset: + return malformed("size %d", r.Size) + case r.Set&AttrUID != 0 && r.UID == math.MaxUint32, r.Set&AttrGID != 0 && r.GID == math.MaxUint32: + return malformed("owner 4294967295 means no change") + } + return errors.Join(r.Target.validate(), validPerm(r.Mode), r.Atime.validate(), r.Mtime.validate()) +} + +func (r *SetAttrResponse) encode(e *sandboxwire.Encoder) { r.Attr.encode(e) } +func (r *SetAttrResponse) decode(d *decoder) { r.Attr.decode(d) } +func (r *SetAttrResponse) validate() error { return r.Attr.validate() } + +func (r *AccessRequest) encode(e *sandboxwire.Encoder) { r.Node.encode(e); e.U32(uint32(r.Mask)) } +func (r *AccessRequest) decode(d *decoder) { r.Node.decode(d); r.Mask = AccessMask(d.u32()) } +func (r *AccessRequest) validate() error { + if r.Mask&^accessMaskAll != 0 { + return malformed("access mask %#x", uint32(r.Mask)) + } + return r.Node.validate() +} + +func (*AccessResponse) encode(*sandboxwire.Encoder) {} +func (*AccessResponse) decode(*decoder) {} +func (*AccessResponse) validate() error { return nil } + +func (r *OpenRequest) encode(e *sandboxwire.Encoder) { + r.Node.encode(e) + e.Enum(uint16(r.Access)) + e.U32(uint32(r.Flags)) +} + +func (r *OpenRequest) decode(d *decoder) { + r.Node.decode(d) + r.Access = AccessMode(d.u16()) + r.Flags = OpenFlags(d.u32()) +} + +func (r *OpenRequest) validate() error { + return errors.Join(r.Node.validate(), validAccess(r.Access), validOpenFlags(r.Flags)) +} + +func (r *OpenResponse) encode(e *sandboxwire.Encoder) { e.U64(uint64(r.Handle)) } +func (r *OpenResponse) decode(d *decoder) { r.Handle = HandleID(d.u64()) } +func (r *OpenResponse) validate() error { return r.Handle.validate() } + +func (r *CreateRequest) encode(e *sandboxwire.Encoder) { + r.Parent.encode(e) + e.Bytes(r.Name) + e.U32(r.Mode) + e.Enum(uint16(r.Access)) + e.U32(uint32(r.Flags)) + e.Bool(r.Exclusive) +} + +func (r *CreateRequest) decode(d *decoder) { + r.Parent.decode(d) + r.Name = d.bytes() + r.Mode = d.u32() + r.Access = AccessMode(d.u16()) + r.Flags = OpenFlags(d.u32()) + r.Exclusive = d.bool() +} + +func (r *CreateRequest) validate() error { + return errors.Join(r.Parent.validate(), validName(r.Name), validPerm(r.Mode), validAccess(r.Access), validOpenFlags(r.Flags)) +} + +func (r *CreateResponse) encode(e *sandboxwire.Encoder) { r.Entry.encode(e); e.U64(uint64(r.Handle)) } +func (r *CreateResponse) decode(d *decoder) { r.Entry.decode(d); r.Handle = HandleID(d.u64()) } +func (r *CreateResponse) validate() error { + return errors.Join(r.Entry.validate(), r.Handle.validate()) +} + +func (r *ReadRequest) encode(e *sandboxwire.Encoder) { + e.U64(uint64(r.Handle)) + e.U64(r.Offset) + e.U32(r.Size) +} +func (r *ReadRequest) decode(d *decoder) { + r.Handle = HandleID(d.u64()) + r.Offset = d.u64() + r.Size = d.u32() +} + +func (r *ReadRequest) validate() error { + if r.Offset > maxOffset || r.Size > sandboxwire.MaxChunk { + return malformed("read of %d bytes at %d", r.Size, r.Offset) + } + return r.Handle.validate() +} + +func (r *ReadResponse) encode(e *sandboxwire.Encoder) { e.Bytes(r.Data) } +func (r *ReadResponse) decode(d *decoder) { r.Data = d.bytes() } +func (r *ReadResponse) validate() error { + if len(r.Data) > sandboxwire.MaxChunk { + return malformed("read of %d bytes", len(r.Data)) + } + return nil +} + +func (r *WriteRequest) encode(e *sandboxwire.Encoder) { + e.U64(uint64(r.Handle)) + e.U64(r.Offset) + e.Bytes(r.Data) +} +func (r *WriteRequest) decode(d *decoder) { + r.Handle = HandleID(d.u64()) + r.Offset = d.u64() + r.Data = d.bytes() +} + +func (r *WriteRequest) validate() error { + if len(r.Data) > sandboxwire.MaxChunk { + return malformed("write of %d bytes", len(r.Data)) + } + return r.Handle.validate() +} + +func (r *WriteResponse) encode(e *sandboxwire.Encoder) { + e.U32(r.Written) + encodeOptionalFailure(e, r.Failure) +} +func (r *WriteResponse) decode(d *decoder) { r.Written = d.u32(); r.Failure = decodeOptionalFailure(d) } +func (r *WriteResponse) validate() error { + if r.Written > sandboxwire.MaxChunk { + return malformed("wrote %d bytes", r.Written) + } + if r.Failure != nil { + if r.Written == 0 { + return malformed("a write failure after no bytes belongs in a failure response") + } + return r.Failure.validate() + } + return nil +} + +func (r *FlushRequest) encode(e *sandboxwire.Encoder) { + e.U64(uint64(r.Handle)) + e.U64(uint64(r.Owner)) +} +func (r *FlushRequest) decode(d *decoder) { r.Handle = HandleID(d.u64()); r.Owner = LockOwner(d.u64()) } +func (r *FlushRequest) validate() error { return r.Handle.validate() } + +func (*FlushResponse) encode(*sandboxwire.Encoder) {} +func (*FlushResponse) decode(*decoder) {} +func (*FlushResponse) validate() error { return nil } + +func (r *FsyncRequest) encode(e *sandboxwire.Encoder) { e.U64(uint64(r.Handle)); e.Bool(r.DataOnly) } +func (r *FsyncRequest) decode(d *decoder) { r.Handle = HandleID(d.u64()); r.DataOnly = d.bool() } +func (r *FsyncRequest) validate() error { return r.Handle.validate() } + +func (*FsyncResponse) encode(*sandboxwire.Encoder) {} +func (*FsyncResponse) decode(*decoder) {} +func (*FsyncResponse) validate() error { return nil } + +func (r *ReleaseRequest) encode(e *sandboxwire.Encoder) { e.U64(uint64(r.Handle)) } +func (r *ReleaseRequest) decode(d *decoder) { r.Handle = HandleID(d.u64()) } +func (r *ReleaseRequest) validate() error { return r.Handle.validate() } + +func (*ReleaseResponse) encode(*sandboxwire.Encoder) {} +func (*ReleaseResponse) decode(*decoder) {} +func (*ReleaseResponse) validate() error { return nil } + +func (r *OpenDirRequest) encode(e *sandboxwire.Encoder) { r.Node.encode(e) } +func (r *OpenDirRequest) decode(d *decoder) { r.Node.decode(d) } +func (r *OpenDirRequest) validate() error { return r.Node.validate() } + +func (r *OpenDirResponse) encode(e *sandboxwire.Encoder) { e.U64(uint64(r.Handle)) } +func (r *OpenDirResponse) decode(d *decoder) { r.Handle = HandleID(d.u64()) } +func (r *OpenDirResponse) validate() error { return r.Handle.validate() } + +func (r *ReadDirRequest) encode(e *sandboxwire.Encoder) { + e.U64(uint64(r.Handle)) + e.U64(r.Cookie) + e.U32(r.Limit) + e.Bool(r.WithAttrs) +} + +func (r *ReadDirRequest) decode(d *decoder) { + r.Handle = HandleID(d.u64()) + r.Cookie = d.u64() + r.Limit = d.u32() + r.WithAttrs = d.bool() +} + +func (r *ReadDirRequest) validate() error { + if r.Limit == 0 || r.Limit > maxReadDirBytes { + return malformed("limit %d", r.Limit) + } + return r.Handle.validate() +} + +func (r *ReadDirResponse) encode(e *sandboxwire.Encoder) { + e.Count(len(r.Entries)) + for i := range r.Entries { + x := &r.Entries[i] + e.Bytes(x.Name) + e.U64(x.Ino) + e.U32(x.Type) + e.U64(x.Cookie) + e.Present(x.Entry != nil) + if x.Entry != nil { + x.Entry.encode(e) + } + } + e.Bool(r.End) +} + +func (r *ReadDirResponse) decode(d *decoder) { + r.Entries = make([]DirEntry, d.count(maxReadDirBytes/dirEntryWireSize)) + for i := range r.Entries { + x := &r.Entries[i] + x.Name = d.bytes() + x.Ino = d.u64() + x.Type = d.u32() + x.Cookie = d.u64() + if d.present() { + x.Entry = new(Entry) + x.Entry.decode(d) + } + } + r.End = d.bool() +} + +func (r *ReadDirResponse) validate() error { + if len(r.Entries) == 0 && !r.End { + return malformed("an empty page before the end") + } + size := 0 + names := make(map[string]bool, len(r.Entries)) + cookies := make(map[uint64]bool, len(r.Entries)) + for i := range r.Entries { + x := &r.Entries[i] + if err := validName(x.Name); err != nil { + return err + } + if names[string(x.Name)] || cookies[x.Cookie] { + return malformed("entry %q repeats a name or cookie in the page", x.Name) + } + names[string(x.Name)], cookies[x.Cookie] = true, true + if !validFileType(x.Type) { + return malformed("entry type %#o", x.Type) + } + if x.Entry != nil { + if err := x.Entry.validate(); err != nil { + return err + } + if x.Entry.Attr.Ino != x.Ino || x.Entry.Attr.Mode&ModeType != x.Type { + return malformed("entry attributes disagree with the entry") + } + } + size += x.WireSize() + } + if size > maxReadDirBytes { + return malformed("entries of %d bytes", size) + } + return nil +} + +func (r *ReleaseDirRequest) encode(e *sandboxwire.Encoder) { e.U64(uint64(r.Handle)) } +func (r *ReleaseDirRequest) decode(d *decoder) { r.Handle = HandleID(d.u64()) } +func (r *ReleaseDirRequest) validate() error { return r.Handle.validate() } + +func (*ReleaseDirResponse) encode(*sandboxwire.Encoder) {} +func (*ReleaseDirResponse) decode(*decoder) {} +func (*ReleaseDirResponse) validate() error { return nil } + +func (r *MkdirRequest) encode(e *sandboxwire.Encoder) { + r.Parent.encode(e) + e.Bytes(r.Name) + e.U32(r.Mode) +} +func (r *MkdirRequest) decode(d *decoder) { r.Parent.decode(d); r.Name = d.bytes(); r.Mode = d.u32() } +func (r *MkdirRequest) validate() error { + return errors.Join(r.Parent.validate(), validName(r.Name), validPerm(r.Mode)) +} + +func (r *MkdirResponse) encode(e *sandboxwire.Encoder) { r.Entry.encode(e) } +func (r *MkdirResponse) decode(d *decoder) { r.Entry.decode(d) } +func (r *MkdirResponse) validate() error { return r.Entry.validate() } + +func (r *UnlinkRequest) encode(e *sandboxwire.Encoder) { r.Parent.encode(e); e.Bytes(r.Name) } +func (r *UnlinkRequest) decode(d *decoder) { r.Parent.decode(d); r.Name = d.bytes() } +func (r *UnlinkRequest) validate() error { return errors.Join(r.Parent.validate(), validName(r.Name)) } + +func (*UnlinkResponse) encode(*sandboxwire.Encoder) {} +func (*UnlinkResponse) decode(*decoder) {} +func (*UnlinkResponse) validate() error { return nil } + +func (r *RmdirRequest) encode(e *sandboxwire.Encoder) { r.Parent.encode(e); e.Bytes(r.Name) } +func (r *RmdirRequest) decode(d *decoder) { r.Parent.decode(d); r.Name = d.bytes() } +func (r *RmdirRequest) validate() error { return errors.Join(r.Parent.validate(), validName(r.Name)) } + +func (*RmdirResponse) encode(*sandboxwire.Encoder) {} +func (*RmdirResponse) decode(*decoder) {} +func (*RmdirResponse) validate() error { return nil } + +func (r *RenameRequest) encode(e *sandboxwire.Encoder) { + r.Parent.encode(e) + e.Bytes(r.Name) + r.NewParent.encode(e) + e.Bytes(r.NewName) + e.Enum(uint16(r.Mode)) +} + +func (r *RenameRequest) decode(d *decoder) { + r.Parent.decode(d) + r.Name = d.bytes() + r.NewParent.decode(d) + r.NewName = d.bytes() + r.Mode = RenameMode(d.u16()) +} + +func (r *RenameRequest) validate() error { + if r.Mode < RenameReplace || r.Mode > RenameExchange { + return malformed("rename mode %d", r.Mode) + } + return errors.Join(r.Parent.validate(), validName(r.Name), r.NewParent.validate(), validName(r.NewName)) +} + +func (*RenameResponse) encode(*sandboxwire.Encoder) {} +func (*RenameResponse) decode(*decoder) {} +func (*RenameResponse) validate() error { return nil } + +func (r *LinkRequest) encode(e *sandboxwire.Encoder) { + r.Node.encode(e) + r.NewParent.encode(e) + e.Bytes(r.NewName) +} +func (r *LinkRequest) decode(d *decoder) { + r.Node.decode(d) + r.NewParent.decode(d) + r.NewName = d.bytes() +} +func (r *LinkRequest) validate() error { + return errors.Join(r.Node.validate(), r.NewParent.validate(), validName(r.NewName)) +} + +func (r *LinkResponse) encode(e *sandboxwire.Encoder) { r.Entry.encode(e) } +func (r *LinkResponse) decode(d *decoder) { r.Entry.decode(d) } +func (r *LinkResponse) validate() error { return r.Entry.validate() } + +func (r *SymlinkRequest) encode(e *sandboxwire.Encoder) { + r.Parent.encode(e) + e.Bytes(r.Name) + e.Bytes(r.Target) +} +func (r *SymlinkRequest) decode(d *decoder) { + r.Parent.decode(d) + r.Name = d.bytes() + r.Target = d.bytes() +} +func (r *SymlinkRequest) validate() error { + return errors.Join(r.Parent.validate(), validName(r.Name), validTarget(r.Target)) +} + +func (r *SymlinkResponse) encode(e *sandboxwire.Encoder) { r.Entry.encode(e) } +func (r *SymlinkResponse) decode(d *decoder) { r.Entry.decode(d) } +func (r *SymlinkResponse) validate() error { return r.Entry.validate() } + +func (r *ReadlinkRequest) encode(e *sandboxwire.Encoder) { r.Node.encode(e) } +func (r *ReadlinkRequest) decode(d *decoder) { r.Node.decode(d) } +func (r *ReadlinkRequest) validate() error { return r.Node.validate() } + +func (r *ReadlinkResponse) encode(e *sandboxwire.Encoder) { e.Bytes(r.Target) } +func (r *ReadlinkResponse) decode(d *decoder) { r.Target = d.bytes() } +func (r *ReadlinkResponse) validate() error { return validTarget(r.Target) } + +func (r *StatFSRequest) encode(e *sandboxwire.Encoder) { r.Node.encode(e) } +func (r *StatFSRequest) decode(d *decoder) { r.Node.decode(d) } +func (r *StatFSRequest) validate() error { return r.Node.validate() } + +func (r *StatFSResponse) encode(e *sandboxwire.Encoder) { + for _, v := range []uint64{r.Blocks, r.BlocksFree, r.BlocksAvailable, r.Files, r.FilesFree} { + e.U64(v) + } + e.U32(r.BlockSize) + e.U32(r.FragmentSize) + e.U32(r.NameMax) +} + +func (r *StatFSResponse) decode(d *decoder) { + for _, v := range []*uint64{&r.Blocks, &r.BlocksFree, &r.BlocksAvailable, &r.Files, &r.FilesFree} { + *v = d.u64() + } + r.BlockSize = d.u32() + r.FragmentSize = d.u32() + r.NameMax = d.u32() +} + +func (*StatFSResponse) validate() error { return nil } + +func (r *ForgetRequest) encode(e *sandboxwire.Encoder) { + e.Count(len(r.Entries)) + for _, f := range r.Entries { + f.Node.encode(e) + e.U64(f.Count) + } +} + +func (r *ForgetRequest) decode(d *decoder) { + r.Entries = make([]ForgetEntry, d.count(maxForgetEntries)) + for i := range r.Entries { + r.Entries[i].Node.decode(d) + r.Entries[i].Count = d.u64() + } +} + +func (r *ForgetRequest) validate() error { + if len(r.Entries) == 0 || len(r.Entries) > maxForgetEntries { + return malformed("forget of %d entries", len(r.Entries)) + } + seen := make(map[uint64]bool, len(r.Entries)) + for _, f := range r.Entries { + if err := f.Node.validate(); err != nil { + return err + } + if f.Count == 0 || seen[f.Node.ID] { + return malformed("forget entry %d", f.Node.ID) + } + seen[f.Node.ID] = true + } + return nil +} + +func (*ForgetResponse) encode(*sandboxwire.Encoder) {} +func (*ForgetResponse) decode(*decoder) {} +func (*ForgetResponse) validate() error { return nil } + +func (r *GetLockRequest) encode(e *sandboxwire.Encoder) { + e.U64(uint64(r.Handle)) + e.U64(uint64(r.Owner)) + r.Lock.encode(e) +} + +func (r *GetLockRequest) decode(d *decoder) { + r.Handle = HandleID(d.u64()) + r.Owner = LockOwner(d.u64()) + r.Lock.decode(d) +} + +func (r *GetLockRequest) validate() error { + if r.Lock.Mode == LockUnlock { + return malformed("GetLock of an unlock") + } + return errors.Join(r.Handle.validate(), r.Lock.validate()) +} + +func (r *GetLockResponse) encode(e *sandboxwire.Encoder) { + e.Present(r.Conflict != nil) + if r.Conflict != nil { + r.Conflict.encode(e) + } +} + +func (r *GetLockResponse) decode(d *decoder) { + if d.present() { + r.Conflict = new(Lock) + r.Conflict.decode(d) + } +} + +func (r *GetLockResponse) validate() error { + if r.Conflict == nil { + return nil + } + if r.Conflict.Mode == LockUnlock { + return malformed("conflict with an unlock") + } + return r.Conflict.validate() +} + +func (r *SetLockRequest) encode(e *sandboxwire.Encoder) { + e.U64(uint64(r.Handle)) + e.Enum(uint16(r.Kind)) + e.U64(uint64(r.Owner)) + r.Lock.encode(e) + e.Bool(r.Wait) +} + +func (r *SetLockRequest) decode(d *decoder) { + r.Handle = HandleID(d.u64()) + r.Kind = LockKind(d.u16()) + r.Owner = LockOwner(d.u64()) + r.Lock.decode(d) + r.Wait = d.bool() +} + +func (r *SetLockRequest) validate() error { + switch r.Kind { + case LockPOSIX: + case LockFlock: + if r.Lock.Start != 0 || r.Lock.End != maxOffset { + return malformed("a flock lock covers the whole file") + } + default: + return malformed("lock kind %d", r.Kind) + } + return errors.Join(r.Handle.validate(), r.Lock.validate()) +} + +func (*SetLockResponse) encode(*sandboxwire.Encoder) {} +func (*SetLockResponse) decode(*decoder) {} +func (*SetLockResponse) validate() error { return nil } + +func (r *CancelRequestRequest) encode(e *sandboxwire.Encoder) { e.U64(r.Target) } +func (r *CancelRequestRequest) decode(d *decoder) { r.Target = d.u64() } +func (r *CancelRequestRequest) validate() error { + if !sandboxwire.ValidRequestID(r.Target) { + return malformed("cancel of request 0") + } + return nil +} + +func (*CancelRequestResponse) encode(*sandboxwire.Encoder) {} +func (*CancelRequestResponse) decode(*decoder) {} +func (*CancelRequestResponse) validate() error { return nil } diff --git a/internal/sandboxfs/protocol_test.go b/internal/sandboxfs/protocol_test.go new file mode 100644 index 00000000..ed8ad24a --- /dev/null +++ b/internal/sandboxfs/protocol_test.go @@ -0,0 +1,297 @@ +package sandboxfs + +import ( + "bytes" + "encoding/hex" + "errors" + "math" + "os" + "path/filepath" + "reflect" + "strings" + "testing" + + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxlink" + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxwire" +) + +func readHexFixture(t testing.TB, name string) []byte { + t.Helper() + raw, err := os.ReadFile(filepath.Join("testdata", name)) + if err != nil { + t.Fatal(err) + } + var digits strings.Builder + for _, line := range strings.Split(string(raw), "\n") { + line, _, _ = strings.Cut(line, "#") + digits.WriteString(strings.Join(strings.Fields(line), "")) + } + b, err := hex.DecodeString(digits.String()) + if err != nil { + t.Fatal(err) + } + return b +} + +var ( + testNode = NodeRef{ID: 1, Generation: 7} + testNode2 = NodeRef{ID: 2, Generation: 9} + testAttr = Attr{Ino: 0x102, Mode: ModeRegular | 0o644, Nlink: 1, UID: 1000, GID: 1000, Size: 5, Blocks: 8, Blksize: 4096, Atime: Timestamp{1, 2}, Mtime: Timestamp{3, 4}, Ctime: Timestamp{-5, 6}} + testDirAttr = Attr{Ino: 0x101, Mode: ModeDirectory | 0o755, Nlink: 2, Blksize: 4096} + testEntry = Entry{Node: testNode2, Attr: testAttr} + testCaps = Capabilities{ + PathProfile: PathProfileLinuxBytes, CacheProfile: CacheProfileUncached, Durability: DurabilityFsyncRequired, + MaxNameBytes: 255, MaxPathBytes: 4095, MaxReadBytes: 65536, MaxWriteBytes: 65536, MaxWalkComponents: 256, + MaxReadDirBytes: 65536, MaxOpenHandles: 4096, AtomicAppend: true, AtomicRename: true, RenameNoReplace: true, + RenameExchange: true, HardLinks: true, Symlinks: true, SetMode: true, SetOwner: true, SetTimes: true, + DirectoryFsync: true, ReadDirPlus: true, Flock: true, + } + testInstance = sandboxwire.ID{0x00, 0x11, 0x22, 0x33, 0x44, 0x55, 0x66, 0x77, 0x88, 0x99, 0xaa, 0xbb, 0xcc, 0xdd, 0xee, 0xff} +) + +func TestGoldenFixtures(t *testing.T) { + symlink := Attr{Ino: 0x102, Mode: ModeSymlink | 0o777, Nlink: 1, UID: 1000, GID: 1000, Size: 1, Blksize: 4096, + Atime: Timestamp{0x65000000, 1}, Mtime: Timestamp{0x65000000, 2}, Ctime: Timestamp{0x65000000, 3}} + for _, tc := range []struct { + file string + op Op + id uint64 + req Request + resp message + fail *Failure + }{ + {file: "describe_response.hex", op: OpDescribe, id: 1, resp: &DescribeResponse{ + ServerInstanceID: testInstance, Identity: Identity{UID: 1000, GID: 1000}, Capabilities: testCaps, Exports: []sandboxlink.ExportID{"world"}}}, + {file: "walk_request.hex", op: OpWalk, id: 2, req: &WalkRequest{Parent: testNode, Names: [][]byte{[]byte("link"), []byte("x")}}}, + {file: "walk_response.hex", op: OpWalk, id: 2, resp: &WalkResponse{Entries: []Entry{{Node: NodeRef{ID: 2, Generation: 7}, Attr: symlink}}}}, + {file: "create_request.hex", op: OpCreate, id: 3, req: &CreateRequest{ + Parent: testNode, Name: []byte("notes.md"), Mode: 0o644, Access: AccessReadWrite, Flags: OpenAppend, Exclusive: true}}, + {file: "write_response.hex", op: OpWrite, id: 4, resp: &WriteResponse{ + Written: 4096, Failure: &Failure{Code: CodeErrno, Errno: ErrnoNoSpace, Effect: sandboxwire.EffectNone, Message: "disk full"}}}, + {file: "readdir_request.hex", op: OpReadDir, id: 5, req: &ReadDirRequest{Handle: 0x42, Cookie: 0x1c6a3e5f0b9d2471, Limit: 65536}}, + {file: "readdir_response.hex", op: OpReadDir, id: 5, resp: &ReadDirResponse{Entries: []DirEntry{ + {Name: []byte("a.txt"), Ino: 0x103, Type: ModeRegular, Cookie: 0x2f0e4b6c7d8a9b10}, + {Name: []byte("src"), Ino: 0x104, Type: ModeDirectory, Cookie: 0x3a5c7e9f1b2d4f60}, + }}}, + {file: "rename_request.hex", op: OpRename, id: 6, req: &RenameRequest{ + Parent: testNode, Name: []byte("old"), NewParent: testNode2, NewName: []byte("new"), Mode: RenameExchange}}, + {file: "failure_response.hex", op: OpLookup, id: 7, fail: &Failure{ + Code: CodeErrno, Errno: ErrnoNotFound, Effect: sandboxwire.EffectNone, Message: "missing"}}, + } { + t.Run(tc.file, func(t *testing.T) { + want := readHexFixture(t, tc.file) + typ := uint16(tc.op) + var payload []byte + var err error + if tc.req != nil { + payload, err = encodeRequest(tc.req) + } else { + typ = sandboxwire.ResponseType(typ) + payload, err = encodeResponse(tc.resp, tc.fail) + } + if err != nil { + t.Fatal(err) + } + var got bytes.Buffer + if err := sandboxwire.WriteFrame(&got, sandboxwire.Frame{Type: typ, RequestID: tc.id, Payload: payload}); err != nil { + t.Fatal(err) + } + if !bytes.Equal(got.Bytes(), want) { + t.Fatalf("encoded\n got %x\nwant %x", got.Bytes(), want) + } + + f, err := sandboxwire.ReadFrame(bytes.NewReader(want), sandboxwire.MaxPayload) + if err != nil || f.Type != typ || f.RequestID != tc.id { + t.Fatalf("frame %+v, %v", f, err) + } + if tc.req != nil { + req, err := decodeRequest(tc.op, f.Payload) + if err != nil || !reflect.DeepEqual(req, tc.req) { + t.Fatalf("decoded %+v, %v", req, err) + } + return + } + resp, fail, err := decodeResponse(tc.op, f.Payload) + if err != nil || !reflect.DeepEqual(fail, tc.fail) || (tc.resp != nil && !reflect.DeepEqual(resp, tc.resp)) { + t.Fatalf("decoded %+v %+v, %v", resp, fail, err) + } + }) + } +} + +// samples holds one valid request and response per operation, in tag order. +func samples() []struct { + req Request + resp message +} { + flock := Lock{Mode: LockWrite, Start: 0, End: math.MaxInt64} + return []struct { + req Request + resp message + }{ + {&DescribeRequest{}, &DescribeResponse{ServerInstanceID: testInstance, Identity: Identity{UID: 1, GID: 2}, Capabilities: testCaps, Exports: []sandboxlink.ExportID{"world", "home"}}}, + {&AttachRequest{Export: "world", ReadOnly: true}, &AttachResponse{Root: Entry{Node: testNode, Attr: testDirAttr}}}, + {&DetachRequest{}, &DetachResponse{}}, + {&LookupRequest{Parent: testNode, Name: []byte("a\xff")}, &LookupResponse{Entry: testEntry}}, + {&WalkRequest{Parent: testNode, Names: [][]byte{[]byte("a"), []byte("b")}}, &WalkResponse{Entries: []Entry{testEntry}, Failure: NewErrnoFailure(ErrnoNotFound, sandboxwire.EffectNone, "b")}}, + {&GetAttrRequest{Target: Target{Kind: TargetHandle, Handle: 3}}, &GetAttrResponse{Attr: testAttr}}, + {&SetAttrRequest{Target: Target{Kind: TargetNode, Node: testNode}, Set: AttrSize | AttrMode | AttrUID | AttrGID | AttrAtime | AttrMtimeNow, Size: 9, Mode: 0o4755, UID: 5, GID: 6, Atime: Timestamp{7, 8}}, &SetAttrResponse{Attr: testAttr}}, + {&AccessRequest{Node: testNode, Mask: MayRead | MayExecute}, &AccessResponse{}}, + {&OpenRequest{Node: testNode, Access: AccessWrite, Flags: OpenTruncate | OpenSync}, &OpenResponse{Handle: 4}}, + {&CreateRequest{Parent: testNode, Name: []byte("n"), Mode: 0o600, Access: AccessRead, Flags: OpenDataSync}, &CreateResponse{Entry: testEntry, Handle: 5}}, + {&ReadRequest{Handle: 4, Offset: 10, Size: 65536}, &ReadResponse{Data: []byte("data")}}, + {&WriteRequest{Handle: 4, Offset: 10, Data: []byte("data")}, &WriteResponse{Written: 4}}, + {&FlushRequest{Handle: 4, Owner: 11}, &FlushResponse{}}, + {&FsyncRequest{Handle: 4, DataOnly: true}, &FsyncResponse{}}, + {&ReleaseRequest{Handle: 4}, &ReleaseResponse{}}, + {&OpenDirRequest{Node: testNode}, &OpenDirResponse{Handle: 6}}, + {&ReadDirRequest{Handle: 6, Cookie: 1, Limit: 4096, WithAttrs: true}, &ReadDirResponse{Entries: []DirEntry{{Name: []byte("f"), Ino: testAttr.Ino, Type: ModeRegular, Cookie: 2, Entry: &testEntry}}, End: true}}, + {&ReleaseDirRequest{Handle: 6}, &ReleaseDirResponse{}}, + {&MkdirRequest{Parent: testNode, Name: []byte("d"), Mode: 0o1777}, &MkdirResponse{Entry: testEntry}}, + {&UnlinkRequest{Parent: testNode, Name: []byte("f")}, &UnlinkResponse{}}, + {&RmdirRequest{Parent: testNode, Name: []byte("d")}, &RmdirResponse{}}, + {&RenameRequest{Parent: testNode, Name: []byte("a"), NewParent: testNode2, NewName: []byte("b"), Mode: RenameNoReplace}, &RenameResponse{}}, + {&LinkRequest{Node: testNode2, NewParent: testNode, NewName: []byte("h")}, &LinkResponse{Entry: testEntry}}, + {&SymlinkRequest{Parent: testNode, Name: []byte("s"), Target: []byte("../t")}, &SymlinkResponse{Entry: testEntry}}, + {&ReadlinkRequest{Node: testNode2}, &ReadlinkResponse{Target: []byte("../t")}}, + {&StatFSRequest{Node: testNode}, &StatFSResponse{Blocks: 1, BlocksFree: 2, BlocksAvailable: 3, Files: 4, FilesFree: 5, BlockSize: 4096, FragmentSize: 4096, NameMax: 255}}, + {&ForgetRequest{Entries: []ForgetEntry{{Node: testNode, Count: 1}, {Node: testNode2, Count: 3}}}, &ForgetResponse{}}, + {&GetLockRequest{Handle: 4, Owner: 12, Lock: Lock{Mode: LockRead, Start: 0, End: 99}}, &GetLockResponse{Conflict: &Lock{Mode: LockWrite, Start: 50, End: 60}}}, + {&SetLockRequest{Handle: 4, Kind: LockFlock, Owner: 12, Lock: flock, Wait: true}, &SetLockResponse{}}, + {&CancelRequestRequest{Target: 9}, &CancelRequestResponse{}}, + } +} + +func TestRoundTripEveryMessage(t *testing.T) { + all := samples() + if len(all) != int(OpCancelRequest) { + t.Fatalf("%d samples for %d operations", len(all), OpCancelRequest) + } + for i, s := range all { + op := Op(i + 1) + if s.req.Op() != op { + t.Fatalf("sample %d is %s", i, s.req.Op()) + } + payload, err := encodeRequest(s.req) + if err != nil { + t.Fatalf("%s request: %v", op, err) + } + if req, err := decodeRequest(op, payload); err != nil || !reflect.DeepEqual(req, s.req) { + t.Fatalf("%s request decoded %+v, %v", op, req, err) + } + if payload, err = encodeResponse(s.resp, nil); err != nil { + t.Fatalf("%s response: %v", op, err) + } + if resp, fail, err := decodeResponse(op, payload); err != nil || fail != nil || !reflect.DeepEqual(resp, s.resp) { + t.Fatalf("%s response decoded %+v %v, %v", op, resp, fail, err) + } + } +} + +func TestDecodeRejects(t *testing.T) { + encode := func(m message) []byte { + var e sandboxwire.Encoder + m.encode(&e) + return e.Payload() + } + for name, tc := range map[string]struct { + op Op + payload []byte + }{ + "dot-dot name": {OpLookup, encode(&LookupRequest{Parent: testNode, Name: []byte("..")})}, + "slash in name": {OpMkdir, encode(&MkdirRequest{Parent: testNode, Name: []byte("a/b")})}, + "zero generation": {OpOpenDir, encode(&OpenDirRequest{Node: NodeRef{ID: 1}})}, + "unknown open flag": {OpOpen, encode(&OpenRequest{Node: testNode, Access: AccessRead, Flags: 1 << 5})}, + "zero access mode": {OpOpen, encode(&OpenRequest{Node: testNode})}, + "duplicate forget": {OpForget, encode(&ForgetRequest{Entries: []ForgetEntry{{testNode, 1}, {testNode, 1}}})}, + "partial flock range": {OpSetLock, encode(&SetLockRequest{Handle: 1, Kind: LockFlock, Lock: Lock{Mode: LockRead, End: 10}})}, + "trailing byte": {OpReleaseDir, append(encode(&ReleaseDirRequest{Handle: 1}), 0)}, + "owner sentinel": {OpSetAttr, encode(&SetAttrRequest{Target: Target{Kind: TargetHandle, Handle: 1}, Set: AttrUID, UID: math.MaxUint32})}, + "uppercase export": {OpAttach, encode(&AttachRequest{Export: "World"})}, + "dot in export": {OpAttach, encode(&AttachRequest{Export: "a.b"})}, + } { + if _, err := decodeRequest(tc.op, tc.payload); !errors.Is(err, sandboxwire.ErrMalformed) { + t.Errorf("%s: %v", name, err) + } + } + entry := func(name string, cookie uint64) DirEntry { + return DirEntry{Name: []byte(name), Ino: 1, Type: ModeRegular, Cookie: cookie} + } + for name, page := range map[string]*ReadDirResponse{ + "repeated name": {Entries: []DirEntry{entry("a", 1), entry("a", 2)}, End: true}, + "repeated cookie": {Entries: []DirEntry{entry("a", 1), entry("b", 1)}, End: true}, + "empty page": {}, + } { + var e sandboxwire.Encoder + e.Enum(uint16(ResultSuccess)) + page.encode(&e) + if _, _, err := decodeResponse(OpReadDir, e.Payload()); !errors.Is(err, sandboxwire.ErrMalformed) { + t.Errorf("%s: %v", name, err) + } + } +} + +func FuzzDecode(f *testing.F) { + entries, _ := os.ReadDir("testdata") + for _, e := range entries { + if strings.HasSuffix(e.Name(), ".hex") { + f.Add(readHexFixture(f, e.Name())) + } + } + for i, s := range samples() { + op := uint16(i + 1) + for _, m := range []struct { + typ uint16 + msg message + }{{op, s.req}, {sandboxwire.ResponseType(op), s.resp}} { + var e sandboxwire.Encoder + if m.typ == op { + m.msg.encode(&e) + } else { + e.Enum(uint16(ResultSuccess)) + m.msg.encode(&e) + } + var w bytes.Buffer + sandboxwire.WriteFrame(&w, sandboxwire.Frame{Type: m.typ, RequestID: 1, Payload: e.Payload()}) + f.Add(w.Bytes()) + } + } + f.Fuzz(func(t *testing.T, data []byte) { + fr, err := sandboxwire.ReadFrame(bytes.NewReader(data), sandboxwire.MaxPayload) + if err != nil { + return + } + kind, err := tags.Classify(fr.Type) + if err != nil { + return + } + var again []byte + switch kind { + case sandboxwire.KindRequest: + req, err := decodeRequest(Op(fr.Type), fr.Payload) + if err != nil { + if !errors.Is(err, sandboxwire.ErrMalformed) { + t.Fatalf("request error %v is not ErrMalformed", err) + } + return + } + if again, err = encodeRequest(req); err != nil { + t.Fatalf("decoded request does not encode: %v", err) + } + case sandboxwire.KindResponse: + resp, fail, err := decodeResponse(Op(fr.Type^sandboxwire.ResponseType(0)), fr.Payload) + if err != nil { + if !errors.Is(err, sandboxwire.ErrMalformed) { + t.Fatalf("response error %v is not ErrMalformed", err) + } + return + } + if again, err = encodeResponse(resp, fail); err != nil { + t.Fatalf("decoded response does not encode: %v", err) + } + } + if !bytes.Equal(again, fr.Payload) { + t.Fatalf("round trip changed the payload:\n got %x\nwant %x", again, fr.Payload) + } + }) +} diff --git a/internal/sandboxfs/server.go b/internal/sandboxfs/server.go new file mode 100644 index 00000000..52735d53 --- /dev/null +++ b/internal/sandboxfs/server.go @@ -0,0 +1,139 @@ +package sandboxfs + +import ( + "context" + "errors" + "io" + "slices" + "sync" + + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxwire" +) + +// MaxInFlight is how many requests one stream holds at once, from admission +// until the response is written. The server answers a request beyond it with +// ResourceExhausted and EffectNone, so it keeps reading and a CancelRequest +// always gets through while the client reads responses. +const MaxInFlight = 256 + +// Serve answers the requests on conn with svc until the stream ends or ctx is +// done. a is the attachment Link authenticated for the stream; Serve refuses +// one without an ID, server instance, lease or valid export grants. Serve owns +// conn and closes it. On return every request context is cancelled and every +// handler has finished. A stream that ends cleanly returns nil. +func Serve(ctx context.Context, conn io.ReadWriteCloser, svc Service, a Attachment) error { + if err := a.validate(); err != nil { + conn.Close() + return err + } + a.Exports = slices.Clone(a.Exports) + ctx, cancel := context.WithCancel(ctx) + s := &server{conn: conn, svc: svc, a: a, inflight: map[uint64]context.CancelFunc{}} + stop := context.AfterFunc(ctx, func() { conn.Close() }) + defer func() { + cancel() + stop() + conn.Close() + s.wg.Wait() + }() + for { + f, err := sandboxwire.ReadFrame(conn, sandboxwire.MaxPayload) + if err == nil { + err = s.dispatch(ctx, f) + } + switch { + case err == nil: + case ctx.Err() != nil: + return ctx.Err() + case errors.Is(err, io.EOF): + return nil + default: + return err + } + } +} + +type server struct { + conn io.ReadWriteCloser + svc Service + a Attachment + seq sandboxwire.RequestSequence // read loop only + wmu sync.Mutex + wg sync.WaitGroup + + mu sync.Mutex + inflight map[uint64]context.CancelFunc // running handlers, for CancelRequest + held int // admitted requests whose response is not yet written +} + +// dispatch starts one request. A frame that is not a request, or whose +// RequestID does not increase, ends the stream. +func (s *server) dispatch(ctx context.Context, f sandboxwire.Frame) error { + kind, err := tags.Classify(f.Type) + if err != nil { + return err + } + if kind != sandboxwire.KindRequest { + return malformed("message type %#04x from the client", f.Type) + } + if !s.seq.Admit(f.RequestID) { + return malformed("request ID %d does not increase", f.RequestID) + } + op := Op(f.Type) + req, err := decodeRequest(op, f.Payload) + s.mu.Lock() + switch { + case err != nil: + s.mu.Unlock() + return s.reply(f.RequestID, op, nil, NewFailure(CodeInvalidArgument, sandboxwire.EffectNone, err.Error())) + case op == OpCancelRequest: + if cancel := s.inflight[req.(*CancelRequestRequest).Target]; cancel != nil { + cancel() + } + s.mu.Unlock() + return s.reply(f.RequestID, op, &CancelRequestResponse{}, nil) + case s.held >= MaxInFlight: + s.mu.Unlock() + return s.reply(f.RequestID, op, nil, NewFailure(CodeResourceExhausted, sandboxwire.EffectNone, "too many requests in flight")) + } + rctx, cancel := context.WithCancel(ctx) + s.inflight[f.RequestID] = cancel + s.held++ + s.mu.Unlock() + s.wg.Add(1) + go func() { + defer s.wg.Done() + resp, err := opSpecs[op].serve(rctx, s.svc, s.a, req) + s.mu.Lock() + delete(s.inflight, f.RequestID) + s.mu.Unlock() + cancel() + var fail *Failure + if err != nil && !errors.As(err, &fail) { + fail = NewFailure(CodeUnknown, sandboxwire.EffectPossible, err.Error()) + } + // The request keeps its slot until its response is written, so a + // client that stops reading stops admission instead of piling up + // finished requests. + werr := s.reply(f.RequestID, op, resp, fail) + s.mu.Lock() + s.held-- + s.mu.Unlock() + if werr != nil { + s.conn.Close() + } + }() + return nil +} + +// reply writes a response. A response the service built wrongly becomes +// Unknown with EffectPossible, since the request may have run. +func (s *server) reply(id uint64, op Op, resp message, fail *Failure) error { + payload, err := encodeResponse(resp, fail) + if err != nil { + payload, _ = encodeResponse(nil, NewFailure(CodeUnknown, sandboxwire.EffectPossible, "service response: "+err.Error())) + } + s.wmu.Lock() + defer s.wmu.Unlock() + return sandboxwire.WriteFrame(s.conn, sandboxwire.Frame{Type: sandboxwire.ResponseType(uint16(op)), RequestID: id, Payload: payload}) +} diff --git a/internal/sandboxfs/server_test.go b/internal/sandboxfs/server_test.go new file mode 100644 index 00000000..f397aad7 --- /dev/null +++ b/internal/sandboxfs/server_test.go @@ -0,0 +1,83 @@ +package sandboxfs + +import ( + "context" + "errors" + "net" + "sync/atomic" + "testing" + "time" + + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxlink" + "github.com/MiniMax-AI/OpenAgentCore/internal/sandboxwire" +) + +// describer answers Describe and counts the calls. +type describer struct { + Service + calls atomic.Int32 +} + +func (d *describer) Describe(context.Context, Attachment, *DescribeRequest) (*DescribeResponse, error) { + d.calls.Add(1) + return &DescribeResponse{ServerInstanceID: testInstance, Capabilities: testCaps}, nil +} + +// A request holds its slot until its response is written, so a client that +// stops reading responses stops admission. +func TestUnreadResponsesStopAdmission(t *testing.T) { + cc, sc := net.Pipe() + svc := &describer{} + served := make(chan error, 1) + a := Attachment{ID: sandboxwire.NewID(), ServerInstanceID: testInstance, Lease: context.Background(), Exports: []sandboxlink.ExportGrant{{ID: "world"}}} + go func() { served <- Serve(context.Background(), sc, svc, a) }() + var read atomic.Int32 // net.Pipe completes a write once the server has read it + go func() { + for id := uint64(1); id <= 2*MaxInFlight; id++ { + if sandboxwire.WriteFrame(cc, sandboxwire.Frame{Type: uint16(OpDescribe), RequestID: id}) != nil { + return + } + read.Add(1) + } + }() + // MaxInFlight requests run; the next one blocks the server writing its + // ResourceExhausted answer, so nothing further is read. + settled := func() bool { return svc.calls.Load() == MaxInFlight && read.Load() == MaxInFlight+1 } + for deadline := time.Now().Add(5 * time.Second); !settled() && time.Now().Before(deadline); { + time.Sleep(time.Millisecond) + } + time.Sleep(100 * time.Millisecond) + if !settled() { + t.Fatalf("%d requests served and %d read while no response was read", svc.calls.Load(), read.Load()) + } + cc.Close() + <-served +} + +// A RequestID that does not increase ends the stream without dispatching. +func TestRequestIDsMustIncrease(t *testing.T) { + cc, sc := net.Pipe() + defer cc.Close() + svc := &describer{} + served := make(chan error, 1) + a := Attachment{ID: sandboxwire.NewID(), ServerInstanceID: testInstance, Lease: context.Background(), Exports: []sandboxlink.ExportGrant{{ID: "world"}}} + go func() { served <- Serve(context.Background(), sc, svc, a) }() + describe := sandboxwire.Frame{Type: uint16(OpDescribe), RequestID: 5} + if err := sandboxwire.WriteFrame(cc, describe); err != nil { + t.Fatal(err) + } + if _, err := sandboxwire.ReadFrame(cc, sandboxwire.MaxPayload); err != nil { + t.Fatal(err) + } + if err := sandboxwire.WriteFrame(cc, describe); err != nil { + t.Fatal(err) + } + select { + case err := <-served: + if !errors.Is(err, sandboxwire.ErrMalformed) || svc.calls.Load() != 1 { + t.Fatalf("Serve returned %v after %d calls", err, svc.calls.Load()) + } + case <-time.After(5 * time.Second): + t.Fatalf("a repeated RequestID kept the stream after %d calls", svc.calls.Load()) + } +} diff --git a/internal/sandboxfs/testdata/create_request.hex b/internal/sandboxfs/testdata/create_request.hex new file mode 100644 index 00000000..ff011c93 --- /dev/null +++ b/internal/sandboxfs/testdata/create_request.hex @@ -0,0 +1,11 @@ +# Exclusive Create of an append-mode file. +00000027 # PayloadLength 39 +000a # MessageType: Create (10) +0000 # Flags +0000000000000003 # RequestID 3 +0000000000000001 0000000000000007 # Parent: ID 1, Generation 7 +00000008 6e6f7465732e6d64 # Name "notes.md" +000001a4 # Mode 0644 +0003 # Access: ReadWrite +00000001 # Flags: OpenAppend +01 # Exclusive diff --git a/internal/sandboxfs/testdata/describe_response.hex b/internal/sandboxfs/testdata/describe_response.hex new file mode 100644 index 00000000..a0b444f5 --- /dev/null +++ b/internal/sandboxfs/testdata/describe_response.hex @@ -0,0 +1,28 @@ +# Describe response. Hex bytes; text after # is a comment. +00000057 # PayloadLength 87 +8001 # MessageType: response to Describe (1) +0000 # Flags +0000000000000001 # RequestID 1 +0001 # Result: success +00112233445566778899aabbccddeeff # ServerInstanceID +000003e8 # Identity.UID 1000 +000003e8 # Identity.GID 1000 +0001 # PathProfile: LinuxBytes +0001 # CacheProfile: Uncached +0001 # Durability: FsyncRequired +000000ff # MaxNameBytes 255 +00000fff # MaxPathBytes 4095 +00010000 # MaxReadBytes 65536 +00010000 # MaxWriteBytes 65536 +00000100 # MaxWalkComponents 256 +00010000 # MaxReadDirBytes 65536 +00001000 # MaxOpenHandles 4096 +00 # ReadOnly +01 01 # AtomicAppend, AtomicRename +01 01 # RenameNoReplace, RenameExchange +01 01 # HardLinks, Symlinks +01 01 01 # SetMode, SetOwner, SetTimes +01 01 # DirectoryFsync, ReadDirPlus +01 00 # Flock, POSIXLocks +00000001 # Exports: count 1 +00000005 776f726c64 # "world" diff --git a/internal/sandboxfs/testdata/failure_response.hex b/internal/sandboxfs/testdata/failure_response.hex new file mode 100644 index 00000000..29c2a443 --- /dev/null +++ b/internal/sandboxfs/testdata/failure_response.hex @@ -0,0 +1,10 @@ +# Lookup failure: the entry does not exist. +00000014 # PayloadLength 20 +8004 # MessageType: response to Lookup (4) +0000 # Flags +0000000000000007 # RequestID 7 +0002 # Result: failure +000b # Code: Errno +01 0003 # Errno: present, NotFound +0001 # Effect: None +00000007 6d697373696e67 # Message "missing" diff --git a/internal/sandboxfs/testdata/readdir_request.hex b/internal/sandboxfs/testdata/readdir_request.hex new file mode 100644 index 00000000..639fcdd9 --- /dev/null +++ b/internal/sandboxfs/testdata/readdir_request.hex @@ -0,0 +1,9 @@ +# ReadDir resuming after a cookie. +00000015 # PayloadLength 21 +0011 # MessageType: ReadDir (17) +0000 # Flags +0000000000000005 # RequestID 5 +0000000000000042 # Handle +1c6a3e5f0b9d2471 # Cookie +00010000 # Limit 65536 +00 # WithAttrs diff --git a/internal/sandboxfs/testdata/readdir_response.hex b/internal/sandboxfs/testdata/readdir_response.hex new file mode 100644 index 00000000..be457520 --- /dev/null +++ b/internal/sandboxfs/testdata/readdir_response.hex @@ -0,0 +1,18 @@ +# ReadDir page of two entries, not the end. +00000041 # PayloadLength 65 +8011 # MessageType: response to ReadDir (17) +0000 # Flags +0000000000000005 # RequestID 5 +0001 # Result: success +00000002 # Entries: count 2 +00000005 612e747874 # Name "a.txt" +0000000000000103 # Ino +00008000 # Type: regular +2f0e4b6c7d8a9b10 # Cookie +00 # Entry: absent +00000003 737263 # Name "src" +0000000000000104 # Ino +00004000 # Type: directory +3a5c7e9f1b2d4f60 # Cookie +00 # Entry: absent +00 # End: false diff --git a/internal/sandboxfs/testdata/rename_request.hex b/internal/sandboxfs/testdata/rename_request.hex new file mode 100644 index 00000000..5c8babda --- /dev/null +++ b/internal/sandboxfs/testdata/rename_request.hex @@ -0,0 +1,10 @@ +# Rename exchanging two entries. +00000030 # PayloadLength 48 +0016 # MessageType: Rename (22) +0000 # Flags +0000000000000006 # RequestID 6 +0000000000000001 0000000000000007 # Parent: ID 1, Generation 7 +00000003 6f6c64 # Name "old" +0000000000000002 0000000000000009 # NewParent: ID 2, Generation 9 +00000003 6e6577 # NewName "new" +0003 # Mode: RenameExchange diff --git a/internal/sandboxfs/testdata/walk_request.hex b/internal/sandboxfs/testdata/walk_request.hex new file mode 100644 index 00000000..4cbbfca0 --- /dev/null +++ b/internal/sandboxfs/testdata/walk_request.hex @@ -0,0 +1,9 @@ +# Walk request. +00000021 # PayloadLength 33 +0005 # MessageType: Walk (5) +0000 # Flags +0000000000000002 # RequestID 2 +0000000000000001 0000000000000007 # Parent: ID 1, Generation 7 +00000002 # Names: count 2 +00000004 6c696e6b # "link" +00000001 78 # "x" diff --git a/internal/sandboxfs/testdata/walk_response.hex b/internal/sandboxfs/testdata/walk_response.hex new file mode 100644 index 00000000..5c76488d --- /dev/null +++ b/internal/sandboxfs/testdata/walk_response.hex @@ -0,0 +1,20 @@ +# Walk response that stopped at the symlink "link": one entry, no failure. +0000006f # PayloadLength 111 +8005 # MessageType: response to Walk (5) +0000 # Flags +0000000000000002 # RequestID 2 +0001 # Result: success +00000001 # Entries: count 1 +0000000000000002 0000000000000007 # Node: ID 2, Generation 7 +0000000000000102 # Attr.Ino +0000a1ff # Attr.Mode: symlink 0777 +00000001 # Attr.Nlink +000003e8 000003e8 # Attr.UID, Attr.GID +0000000000000000 # Attr.Rdev +0000000000000001 # Attr.Size +0000000000000000 # Attr.Blocks +00001000 # Attr.Blksize +0000000065000000 00000001 # Attr.Atime +0000000065000000 00000002 # Attr.Mtime +0000000065000000 00000003 # Attr.Ctime +00 # Failure: absent diff --git a/internal/sandboxfs/testdata/write_response.hex b/internal/sandboxfs/testdata/write_response.hex new file mode 100644 index 00000000..e78d62c5 --- /dev/null +++ b/internal/sandboxfs/testdata/write_response.hex @@ -0,0 +1,12 @@ +# Short write: 4096 bytes written, then the device filled. +0000001b # PayloadLength 27 +800c # MessageType: response to Write (12) +0000 # Flags +0000000000000004 # RequestID 4 +0001 # Result: success +00001000 # Written 4096 +01 # Failure: present +000b # Code: Errno +01 000b # Errno: present, NoSpace +0001 # Effect: None +00000009 6469736b2066756c6c # Message "disk full"