From 457b1d0e85c94d63154777622fa30d2b2d307443 Mon Sep 17 00:00:00 2001 From: SaladDay <1203511142@qq.com> Date: Tue, 6 Oct 2026 17:39:19 +0000 Subject: [PATCH 1/5] Run a process in the live view ViewSession.Spawn runs a LocalExec binary as another process in the Session's live view while the Harness runs. The view's launcher starts it from its restricted thread on a control-socket request, with stdio passed by SCM_RIGHTS, and reports its exit over the same socket. Launch and Spawn return ErrNotLocalExec; Spawn also returns ErrNoLiveView and ErrViewEnded. --- apps/daemon/internal/agent/harness.go | 21 +++ apps/daemon/internal/agenthost/agenthost.go | 2 +- apps/daemon/internal/agenthost/doc.go | 13 +- .../daemon/internal/agenthost/launch_linux.go | 56 +++++++- apps/daemon/internal/agenthost/run_linux.go | 1 + .../internal/agenthost/session_linux_test.go | 17 +++ .../internal/sessionview/control_linux.go | 24 ++-- apps/daemon/internal/sessionview/doc.go | 4 +- apps/daemon/internal/sessionview/errors.go | 4 +- .../internal/sessionview/launcher_linux.go | 126 ++++++++++++++---- .../daemon/internal/sessionview/view_linux.go | 124 ++++++++++++++--- .../internal/sessionview/view_linux_test.go | 65 +++++++++ contracts/agents-api/harness-onboarding.md | 10 +- contracts/agents-api/zh/harness-onboarding.md | 12 +- 14 files changed, 402 insertions(+), 77 deletions(-) diff --git a/apps/daemon/internal/agent/harness.go b/apps/daemon/internal/agent/harness.go index 050fa4b46..a6eb76ed2 100644 --- a/apps/daemon/internal/agent/harness.go +++ b/apps/daemon/internal/agent/harness.go @@ -150,6 +150,16 @@ var ErrInvalidView = errors.New("agent: invalid view declaration") // a connection option. var ErrViewHandoff = errors.New("agent: view request carries a connection outside the Session's gateway") +// ViewSession.Launch and ViewSession.Spawn outcomes. +var ( + // ErrNotLocalExec is a Launch or Spawn whose Binary is not a LocalExec path. + ErrNotLocalExec = errors.New("agent: binary is not a LocalExec path") + // ErrNoLiveView is a Spawn while the Session has no running view. + ErrNoLiveView = errors.New("agent: the Session has no live view") + // ErrViewEnded is a Spawn after the view's Harness exited or its view closed. + ErrViewEnded = errors.New("agent: the view has ended") +) + // View declares how the Harness runs in an agent-host Session view. View // paths are absolute and clean. The closure, Exec overlays and the shim are // the only executable mounts, and all are read-only; the world, the home and @@ -250,6 +260,17 @@ type ViewSession struct { // TERM unless Cancel already did, and ends once they exit or KillTimeout // passes from the first TERM. Launch func(clirunner.StartOptions) (*clirunner.Process, error) + // Spawn runs Binary, a LocalExec path, as another process in the live + // view while its Harness runs, with StartOptions as Launch takes them. The + // process runs as the Harness does: as the same user, in the same + // namespaces, view cgroup, world and network, with no capabilities, + // no_new_privs and the same seccomp filter, in a process group of its own. + // Cancel sends TERM to that group and kills it after KillTimeout, and the + // group is killed once the process exits. The view's end ends it too: a + // Cancel of the Harness reaches it, and when the Harness exits it is one + // of the processes that remain. Spawn returns ErrNotLocalExec, + // ErrNoLiveView or ErrViewEnded. + Spawn func(clirunner.StartOptions) (*clirunner.Process, error) } // checkViewHandoff enforces, before the factory runs, that the view request diff --git a/apps/daemon/internal/agenthost/agenthost.go b/apps/daemon/internal/agenthost/agenthost.go index a36551661..90131ce69 100644 --- a/apps/daemon/internal/agenthost/agenthost.go +++ b/apps/daemon/internal/agenthost/agenthost.go @@ -125,7 +125,7 @@ var ( // ErrWorld is a world that no longer shows the sandbox faithfully, or // that cannot show that its attachment holds nothing. ErrWorld = errors.New("agenthost: world lost") - // ErrLaunch is a view that could not be launched. + // ErrLaunch is a view, or a process in a view, that could not be started. ErrLaunch = errors.New("agenthost: launch failed") // ErrProcessBroker is a view's process broker that could not start, or // whose process relay was lost while the view ran diff --git a/apps/daemon/internal/agenthost/doc.go b/apps/daemon/internal/agenthost/doc.go index 4e3baa4b8..f374136c2 100644 --- a/apps/daemon/internal/agenthost/doc.go +++ b/apps/daemon/internal/agenthost/doc.go @@ -40,12 +40,13 @@ // and calls the view's Executor factory. Each ViewSession.Launch builds one // sessionview view, of which one at a time is live, over the world that // worldfs serves from the attachment's File service, with the gateway -// listening in the view's network namespace. A view with a shim gets its own -// process broker, started once the view runs and closed once it has ended. The -// broker runs the shims' commands over the attachment's Process service in the -// strongest scope the service declares, with the view's ForwardEnv and the -// Session's Environment, and cancels a forwarded process whose shim is lost -// with the launch's kill timeout as its grace. +// listening in the view's network namespace. ViewSession.Spawn runs another +// process in the live view (sessionview.View.Spawn). A view with a shim gets +// its own process broker, started once the view runs and closed once it has +// ended. The broker runs the shims' commands over the attachment's Process +// service in the strongest scope the service declares, with the view's +// ForwardEnv and the Session's Environment, and cancels a forwarded process +// whose shim is lost with the launch's kill timeout as its grace. // // Each view presents the closure directories read-only and executable, the // Session home read-write and noexec, the agent host's /etc/passwd, group, diff --git a/apps/daemon/internal/agenthost/launch_linux.go b/apps/daemon/internal/agenthost/launch_linux.go index 3b53be029..0b7cd2848 100644 --- a/apps/daemon/internal/agenthost/launch_linux.go +++ b/apps/daemon/internal/agenthost/launch_linux.go @@ -38,6 +38,7 @@ type runningView interface { Wait() (sessionview.Exit, error) Close() error Relay() *os.File + Spawn(path string, args, env []string, dir string, stdio [3]*os.File) (*sessionview.Spawned, error) } // viewWorld is the part of *worldfs.World the Session watches. @@ -61,15 +62,23 @@ func (s *session) closeLive() { } } -// launch is ViewSession.Launch: it builds one view and runs opts.Binary in it. -func (s *session) launch(opts clirunner.StartOptions) (*clirunner.Process, error) { +// checkStart checks the options of Launch and Spawn. +func (s *session) checkStart(opts clirunner.StartOptions) error { switch { case !slices.Contains(s.plan.view.LocalExec, opts.Binary): - return nil, &Error{Kind: ErrLaunch, Err: fmt.Errorf("%q is not a LocalExec path", opts.Binary)} + return &Error{Kind: ErrLaunch, Err: fmt.Errorf("%w: %q", agent.ErrNotLocalExec, opts.Binary)} case !isViewPath(opts.Dir): - return nil, &Error{Kind: ErrLaunch, Err: fmt.Errorf("directory %q is not absolute and clean", opts.Dir)} + return &Error{Kind: ErrLaunch, Err: fmt.Errorf("directory %q is not absolute and clean", opts.Dir)} case !opts.OwnProcessGroup: - return nil, &Error{Kind: ErrLaunch, Err: errors.New("a view process runs in its own process group")} + return &Error{Kind: ErrLaunch, Err: errors.New("a view process runs in its own process group")} + } + return nil +} + +// launch is ViewSession.Launch: it builds one view and runs opts.Binary in it. +func (s *session) launch(opts clirunner.StartOptions) (*clirunner.Process, error) { + if err := s.checkStart(opts); err != nil { + return nil, err } if opts.Parent == nil { opts.Parent = context.Background() @@ -93,6 +102,43 @@ func (s *session) launch(opts clirunner.StartOptions) (*clirunner.Process, error return s.start(lv, opts) } +// spawn is ViewSession.Spawn: it runs opts.Binary in the live view. +func (s *session) spawn(opts clirunner.StartOptions) (*clirunner.Process, error) { + if err := s.checkStart(opts); err != nil { + return nil, err + } + var v runningView + s.mu.Lock() + if s.live != nil { + v = s.live.view + } + s.mu.Unlock() + if v == nil { + return nil, &Error{Kind: ErrLaunch, Op: "spawn", Err: agent.ErrNoLiveView} + } + ends, err := newStdio(opts.NeedStdin) + if err != nil { + return nil, &Error{Kind: ErrLaunch, Op: "stdio", Err: err} + } + p, err := v.Spawn(opts.Binary, append([]string{opts.Binary}, opts.Args...), opts.Env, opts.Dir, ends.child) + ends.closeChild() + var process *clirunner.Process + if err == nil { + if process, err = clirunner.FromHandle(p, clirunner.HandleOptions{Parent: opts.Parent, Stdin: ends.stdin(), + Stdout: ends.parent[1], Stderr: ends.parent[2], KillTimeout: opts.KillTimeout}); err != nil { + p.Close() + } + } + if err != nil { + ends.closeParent() + if errors.Is(err, sessionview.ErrExited) || errors.Is(err, sessionview.ErrClosed) || errors.Is(err, sessionview.ErrLauncher) { + err = fmt.Errorf("%w: %w", agent.ErrViewEnded, err) + } + return nil, &Error{Kind: ErrLaunch, Op: "spawn", Err: err} + } + return process, nil +} + // release frees the view slot and ends the launch's count. func (s *session) release(lv *liveView) { s.mu.Lock() diff --git a/apps/daemon/internal/agenthost/run_linux.go b/apps/daemon/internal/agenthost/run_linux.go index d51fde195..7f8bd4041 100644 --- a/apps/daemon/internal/agenthost/run_linux.go +++ b/apps/daemon/internal/agenthost/run_linux.go @@ -104,6 +104,7 @@ func run(ctx context.Context, cfg Config, in Session, d deps) error { Proxy: s.plan.proxy, MCP: s.plan.mcp, Launch: s.launch, + Spawn: s.spawn, }) if err != nil { err = executorError(err) diff --git a/apps/daemon/internal/agenthost/session_linux_test.go b/apps/daemon/internal/agenthost/session_linux_test.go index 9ba453393..7cdf7fa38 100644 --- a/apps/daemon/internal/agenthost/session_linux_test.go +++ b/apps/daemon/internal/agenthost/session_linux_test.go @@ -77,6 +77,19 @@ func TestViewEndReleasesTheSlotBeforeTheProcessEnds(t *testing.T) { } } +func TestSpawnNeedsARunningView(t *testing.T) { + s := newOwnerSession(t) + s.plan = &plan{view: agent.View{LocalExec: []string{"/.oac/tool/tool"}}} + opts := clirunner.StartOptions{Binary: "/.oac/tool/tool", Dir: "/", OwnProcessGroup: true} + if _, err := s.spawn(opts); !errors.Is(err, agent.ErrNoLiveView) { + t.Fatalf("Spawn without a view = %v, want ErrNoLiveView", err) + } + s.live = &liveView{view: &fakeView{exit: make(chan struct{})}} + if _, err := s.spawn(opts); !errors.Is(err, agent.ErrViewEnded) { + t.Fatalf("Spawn in a closed view = %v, want ErrViewEnded", err) + } +} + func TestFailureDuringTeardownCounts(t *testing.T) { s := newOwnerSession(t) // The relay revokes the attachment while Executor.Close waits. @@ -344,6 +357,10 @@ func (v *fakeView) Signal(syscall.Signal) error { return nil } func (v *fakeView) Relay() *os.File { return nil } +func (v *fakeView) Spawn(string, []string, []string, string, [3]*os.File) (*sessionview.Spawned, error) { + return nil, sessionview.ErrClosed +} + func (v *fakeView) Wait() (sessionview.Exit, error) { <-v.exit return sessionview.Exit{}, nil diff --git a/apps/daemon/internal/sessionview/control_linux.go b/apps/daemon/internal/sessionview/control_linux.go index 8a7411dd3..c4476268d 100644 --- a/apps/daemon/internal/sessionview/control_linux.go +++ b/apps/daemon/internal/sessionview/control_linux.go @@ -33,16 +33,21 @@ type launchSpec struct { Private []PrivateDir Overlays []Overlay Shim Shim - Path string - Args []string - Env []string - Dir string + Command command UID uint32 GID uint32 Groups []uint32 Grace time.Duration } +// command is what a process of the view runs. +type command struct { + Path string + Args []string + Env []string + Dir string +} + type msgKind uint8 const ( @@ -50,9 +55,11 @@ const ( msgProceed // daemon: the world serves and the network is set up; carries the mount targets msgStarted // launcher: the process runs msgFailed // launcher: construction failed - msgExited // launcher: the process ended - msgSignal // daemon: signal the process - msgSignaled // launcher: whether msgSignal reached the running process + msgExited // launcher: the process, or the spawned process Pid, ended + msgSignal // daemon: signal the process, or the spawned process Pid + msgSignaled // launcher: whether msgSignal reached it + msgSpawn // daemon: start Command as another process; carries its stdin, stdout and stderr + msgSpawned // launcher: the spawned process's Pid, or why none started ) type message struct { @@ -63,6 +70,7 @@ type message struct { Exit Exit Fail failure Targets map[string]string // each mountpoint's view path to the path the world presents it at + Command command } // failure carries a launcher *Error across the control socket. @@ -146,7 +154,7 @@ func (c *control) send(m message, fds ...int) error { // recv returns the next message and the files it carries. It returns io.EOF once the peer has closed its end. func (c *control) recv() (message, []*os.File, error) { buf := make([]byte, 64<<10) - oob := make([]byte, unix.CmsgSpace(2*4)) + oob := make([]byte, unix.CmsgSpace(3*4)) n, oobn, flags, _, err := c.conn.ReadMsgUnix(buf, oob) if err != nil { return message{}, nil, err diff --git a/apps/daemon/internal/sessionview/doc.go b/apps/daemon/internal/sessionview/doc.go index cc13e2585..d68a121aa 100644 --- a/apps/daemon/internal/sessionview/doc.go +++ b/apps/daemon/internal/sessionview/doc.go @@ -1,4 +1,4 @@ -// Package sessionview runs one process inside a per-Session view on the agent host. +// Package sessionview runs a process, and others spawned beside it, inside a per-Session view on the agent host. // // A view is a private mount, PID and network namespace whose root is the Session's world: a FUSE file system that the daemon serves over a /dev/fuse connection. The launcher adds the local pieces on top of the world: private directories under /.oac, trusted overlays, the command shim, a fresh /proc and a minimal /dev. The world presents a mountpoint for each piece and reports where, following the sandbox's symlinks, and the launcher mounts at those paths without following any symlink itself. The process starts with no capabilities, no_new_privs, a seccomp filter and only stdin, stdout and stderr open. Its network namespace has only loopback up. // @@ -6,6 +6,8 @@ // // The daemon calls [Init] first thing in main. [Start] re-executes the daemon binary as the launcher, which becomes PID 1 of the view: it builds the view, starts the process, delivers signals to every process in the view, reaps orphans and exits with the process status. Once the process has exited, Signal reports [ErrExited] and delivers nothing. When the process exits while others remain, the launcher sends them TERM unless one already went to the view, and waits for them up to [Process].Grace from the first TERM. Its exit kills what remains and tears the view down. // +// While the process runs, [View.Spawn] has the launcher start another process in the view, with the stdio descriptors it passes over the control socket. The launcher starts it from the thread that carries the restrictions, as it started the process, so it runs as the same user, in the view's namespaces and cgroup, under the same restrictions and in a session of its own. A spawned process counts like any other process in the view: a signal to the view reaches it, and when the process exits it is among those the launcher sends TERM and waits for. [Spawned].Signal reaches only its process group. Once the launcher reaps a spawned process, it reports the exit and kills what remains of the process group; the view's end ends it in any case. +// // A view that declares a [Shim] also runs the Session's process relay, the shim binary in relay mode (package processshim). The launcher starts it before the process, under the same restrictions and as the same user, with a listening socket at processshim.SocketPath on a read-only mount and its end of a socket pair whose other end is [View.Relay]. The relay is the only process that receives the descriptors a shim hands over; the broker outside the view holds only its end of the pair. The launcher's drain ignores the relay, which ends with the view. // // Each view runs in a cgroup v2 of its own, which Start creates in [Spec].CgroupParent. The launcher is cloned into it with CLONE_INTO_CGROUP, so no process of the view ever runs outside it. The cgroup hierarchy is the only record of what the views own: [Recover] ends every cgroup an earlier owner of the parent left, with all its processes. diff --git a/apps/daemon/internal/sessionview/errors.go b/apps/daemon/internal/sessionview/errors.go index 1bc15eddc..8c0466de9 100644 --- a/apps/daemon/internal/sessionview/errors.go +++ b/apps/daemon/internal/sessionview/errors.go @@ -25,12 +25,12 @@ var ( ErrCgroup = errors.New("sessionview: view cgroup unavailable") // ErrCleanup reports a teardown or recovery that did not finish within its bound, or a cgroup that could not be ended and removed. The cgroup stays for Recover, and what remains finishes in the background if it can. ErrCleanup = errors.New("sessionview: cleanup incomplete") - // ErrExited is Signal's result once the process has exited. It also matches os.ErrProcessDone. + // ErrExited is Signal's and Spawn's result once the process has exited. Signal's also matches os.ErrProcessDone. ErrExited = errors.New("sessionview: process exited") ) // errorKinds fixes the wire code of each kind the launcher reports. -var errorKinds = []error{ErrLauncher, ErrNoFUSE, ErrMountDenied, ErrMountTarget, ErrNetwork, ErrRestrict, ErrExec} +var errorKinds = []error{ErrLauncher, ErrNoFUSE, ErrMountDenied, ErrMountTarget, ErrNetwork, ErrRestrict, ErrExec, ErrExited} // Error is a typed sessionview failure. It matches Kind and, when present, Err. type Error struct { diff --git a/apps/daemon/internal/sessionview/launcher_linux.go b/apps/daemon/internal/sessionview/launcher_linux.go index f1ed252e8..5242a8c59 100644 --- a/apps/daemon/internal/sessionview/launcher_linux.go +++ b/apps/daemon/internal/sessionview/launcher_linux.go @@ -39,10 +39,15 @@ type launcher struct { proc int // the view's /proc relay int // the relay's pid, or 0 - // mu orders signals against the process's exit. running holds from the process's start until it is reaped; termAt is when TERM first went to the view. + // spawns carries each spawn to the restricted thread, which runs them until reaped closes once the process has been reaped. + spawns chan func() + reaped chan struct{} + + // mu orders signals and spawns against the exits. running holds from the process's start until it is reaped; termAt is when TERM first went to the view; spawned holds each spawned process until it is reaped. mu sync.Mutex running bool termAt time.Time + spawned map[int]bool } func runLauncher() int { @@ -51,7 +56,7 @@ func runLauncher() int { fmt.Fprintf(os.Stderr, "sessionview launcher: %v\n", err) return 1 } - l := &launcher{ctl: ctl, proceed: make(chan struct{})} + l := &launcher{ctl: ctl, proceed: make(chan struct{}), spawns: make(chan func()), reaped: make(chan struct{}), spawned: map[int]bool{}} code, err := l.run() if err != nil { _ = ctl.send(message{Kind: msgFailed, Fail: failureOf(err)}) @@ -65,7 +70,7 @@ func (l *launcher) run() (int, error) { if err != nil { return 0, err } - go l.serveControl() + go l.serveControl(spec) if err := unix.Mount("", "/", "", unix.MS_REC|unix.MS_PRIVATE, ""); err != nil { return 0, mountError("make-rprivate", "/", err) } @@ -98,7 +103,7 @@ func (l *launcher) run() (int, error) { return 0, err } } - pid, err := startProcess(spec) + pid, err := startProcess(spec, spec.Command, []uintptr{stdinFD, stdoutFD, stderrFD}) if err != nil { return 0, err } @@ -112,11 +117,23 @@ func (l *launcher) run() (int, error) { return 0, &Error{Kind: ErrLauncher, Op: "report start", Err: err} } l.forwardSignals() - code, err := l.reap(pid) - if err == nil { - l.drain(spec.Grace) + var code int + go func() { + code, err = l.reap(pid) + close(l.reaped) + }() + // Only this thread carries the restrictions a process inherits, so spawns start here. + for { + select { + case spawn := <-l.spawns: + spawn() + case <-l.reaped: + if err == nil { + l.drain(spec.Grace) + } + return code, err + } } - return code, err } func readSpec() (*launchSpec, error) { @@ -130,10 +147,9 @@ func readSpec() (*launchSpec, error) { } // serveControl handles daemon messages. When the daemon goes away the view goes with it. -func (l *launcher) serveControl() { +func (l *launcher) serveControl(spec *launchSpec) { for { m, files, err := l.ctl.recv() - closeFiles(files) if err != nil { os.Exit(1) } @@ -144,15 +160,46 @@ func (l *launcher) serveControl() { close(l.proceed) }) case msgSignal: - _ = l.ctl.send(message{Kind: msgSignaled, Delivered: l.signal(m.Signal)}) + _ = l.ctl.send(message{Kind: msgSignaled, Delivered: l.signal(m.Pid, m.Signal)}) + case msgSpawn: + spawn := func() { l.spawn(spec, m.Command, files) } + select { + case l.spawns <- spawn: + case <-l.reaped: + // The process has been reaped, so spawn starts nothing and needs no restricted thread. + spawn() + } + continue } + closeFiles(files) } } -// signal delivers sig to every process in the view while the process runs and reports whether it did. As PID 1 of the view, the launcher reaches them all with kill(-1) and is itself spared. Once the process has been reaped, only the drain signals what remains. -func (l *launcher) signal(sig syscall.Signal) bool { +// spawn starts c with files as its stdin, stdout and stderr, as the process's user and in a session of its own, and answers with its pid. Holding mu until the answer is sent keeps the report of its exit behind the answer. +func (l *launcher) spawn(spec *launchSpec, c command, files []*os.File) { + defer closeFiles(files) l.mu.Lock() defer l.mu.Unlock() + m := message{Kind: msgSpawned} + err := error(&Error{Kind: ErrExited, Op: "spawn"}) + if l.running { + m.Pid, err = startProcess(spec, c, []uintptr{files[0].Fd(), files[1].Fd(), files[2].Fd()}) + } + if err != nil { + m.Fail = failureOf(err) + } else { + l.spawned[m.Pid] = true + } + _ = l.ctl.send(m) +} + +// signal delivers sig and reports whether it did. With pid 0 it signals every process in the view while the process runs: as PID 1 of the view, the launcher reaches them all with kill(-1) and is itself spared. Once the process has been reaped, only the drain signals what remains. Otherwise it signals the process group of the spawned process pid until that is reaped. +func (l *launcher) signal(pid int, sig syscall.Signal) bool { + l.mu.Lock() + defer l.mu.Unlock() + if pid != 0 { + return l.spawned[pid] && unix.Kill(-pid, sig) == nil + } if !l.running { return false } @@ -168,7 +215,7 @@ func (l *launcher) forwardSignals() { signal.Notify(sigs, unix.SIGHUP, unix.SIGINT, unix.SIGQUIT, unix.SIGTERM, unix.SIGUSR1, unix.SIGUSR2, unix.SIGWINCH) go func() { for s := range sigs { - l.signal(s.(syscall.Signal)) + l.signal(0, s.(syscall.Signal)) } }() } @@ -189,9 +236,14 @@ func (l *launcher) drain(grace time.Duration) { defer close(reaped) // The last process other than the relay to exit is a child of the launcher by then, so its exit ends the Wait4. for l.othersRemain() { - if _, err := unix.Wait4(-1, nil, 0, nil); err != nil && err != unix.EINTR { + var ws unix.WaitStatus + pid, err := unix.Wait4(-1, &ws, 0, nil) + if err != nil && err != unix.EINTR { return } + if err == nil { + l.spawnExited(pid, ws) + } } }() timer := time.NewTimer(grace) @@ -234,24 +286,39 @@ func (l *launcher) reap(pid int) (int, error) { return 0, &Error{Kind: ErrLauncher, Op: "wait", Err: err} } if wpid != pid { + l.spawnExited(wpid, ws) continue } l.mu.Lock() l.running = false l.mu.Unlock() - var exit Exit - code := ws.ExitStatus() - if ws.Signaled() { - exit.Signal, exit.CoreDumped = ws.Signal(), ws.CoreDump() - code = 128 + int(exit.Signal) - } else { - exit.Code = code - } + exit, code := exitOf(ws) _ = l.ctl.send(message{Kind: msgExited, Exit: exit}) return code, nil } } +// spawnExited reports the exit of the spawned process pid, which a wait collected, and kills what remains of its process group. Any other pid is one of the view's orphans. +func (l *launcher) spawnExited(pid int, ws unix.WaitStatus) { + l.mu.Lock() + defer l.mu.Unlock() + if !l.spawned[pid] { + return + } + delete(l.spawned, pid) + _ = unix.Kill(-pid, unix.SIGKILL) + exit, _ := exitOf(ws) + _ = l.ctl.send(message{Kind: msgExited, Pid: pid, Exit: exit}) +} + +// exitOf describes how a process ended, with the code the launcher exits with for it. +func exitOf(ws unix.WaitStatus) (Exit, int) { + if ws.Signaled() { + return Exit{Signal: ws.Signal(), CoreDumped: ws.CoreDump()}, 128 + int(ws.Signal()) + } + return Exit{Code: ws.ExitStatus()}, ws.ExitStatus() +} + func (l *launcher) mountWorld(staging string) error { dev, err := unix.Open("/dev/fuse", unix.O_RDWR|unix.O_CLOEXEC, 0) if err != nil { @@ -312,18 +379,19 @@ func startRelay(spec *launchSpec, listener int) (int, error) { return pid, nil } -func startProcess(spec *launchSpec) (int, error) { - pid, err := syscall.ForkExec(spec.Path, spec.Args, &syscall.ProcAttr{ - Dir: spec.Dir, - Env: spec.Env, - Files: []uintptr{stdinFD, stdoutFD, stderrFD}, +// startProcess starts c as the process's user, in a session of its own, with stdio as its standard descriptors. +func startProcess(spec *launchSpec, c command, stdio []uintptr) (int, error) { + pid, err := syscall.ForkExec(c.Path, c.Args, &syscall.ProcAttr{ + Dir: c.Dir, + Env: c.Env, + Files: stdio, Sys: &syscall.SysProcAttr{ Setsid: true, Credential: &syscall.Credential{Uid: spec.UID, Gid: spec.GID, Groups: spec.Groups}, }, }) if err != nil { - return 0, &Error{Kind: ErrExec, Op: "exec", Path: spec.Path, Err: err} + return 0, &Error{Kind: ErrExec, Op: "exec", Path: c.Path, Err: err} } return pid, nil } diff --git a/apps/daemon/internal/sessionview/view_linux.go b/apps/daemon/internal/sessionview/view_linux.go index a2df720bf..5f00d65d0 100644 --- a/apps/daemon/internal/sessionview/view_linux.go +++ b/apps/daemon/internal/sessionview/view_linux.go @@ -54,8 +54,8 @@ type View struct { waited chan struct{} // closed once the launcher has been reaped waitErr error cleanupErr error // set before done closes - signalMu sync.Mutex - signaled chan bool + requestMu sync.Mutex + replies chan reply exited atomic.Bool closing atomic.Bool closeOnce sync.Once @@ -72,7 +72,7 @@ func Start(ctx context.Context, spec Spec) (*View, error) { if err := Probe(); err != nil { return nil, err } - v := &View{done: make(chan struct{}), signaled: make(chan bool, 1)} + v := &View{done: make(chan struct{}), replies: make(chan reply, 1)} if err := v.launch(&spec); err != nil { return nil, v.abort(err) } @@ -174,8 +174,8 @@ func (v *View) launch(spec *Spec) error { }() ls := &launchSpec{ Staging: v.staging, Private: spec.Private, Overlays: spec.Overlays, Shim: spec.Shim, - Path: spec.Process.Path, Args: spec.Process.Args, Env: spec.Process.Env, Dir: spec.Process.Dir, - UID: spec.Process.UID, GID: spec.Process.GID, Groups: spec.Process.Groups, Grace: spec.Process.Grace, + Command: command{Path: spec.Process.Path, Args: spec.Process.Args, Env: spec.Process.Env, Dir: spec.Process.Dir}, + UID: spec.Process.UID, GID: spec.Process.GID, Groups: spec.Process.Groups, Grace: spec.Process.Grace, } // A launcher that dies early breaks the pipe; the handshake reports that. go func() { @@ -391,10 +391,17 @@ func (v *View) closePipes() { } } +// reply is the launcher's answer to a request, with the process a spawn started. +type reply struct { + message + spawned *Spawned +} + func (v *View) watch() { defer close(v.done) var exit *Exit var failed error + spawned := map[int]*Spawned{} for { m, files, err := v.ctl.recv() closeFiles(files) @@ -406,15 +413,33 @@ func (v *View) watch() { } switch m.Kind { case msgExited: + if s := spawned[m.Pid]; s != nil { + delete(spawned, m.Pid) + s.exit = m.Exit + close(s.done) + break + } exit = &m.Exit v.exited.Store(true) case msgSignaled: - v.signaled <- m.Delivered + v.replies <- reply{message: m} + case msgSpawned: + r := reply{message: m} + if m.Pid != 0 { + // Registered before the next message, which may report its exit. + r.spawned = &Spawned{v: v, pid: m.Pid, done: make(chan struct{})} + spawned[m.Pid] = r.spawned + } + v.replies <- r case msgFailed: failed = m.Fail.err() } } werr, terr := v.teardown() + for _, s := range spawned { + s.err = ErrClosed + close(s.done) + } switch { case exit != nil: v.exit = *exit @@ -439,36 +464,91 @@ func (v *View) Wait() (Exit, error) { // Presentation reports how the world presented the view's mountpoints. func (v *View) Presentation() Presentation { return v.present } +// errSignalExited is Signal's result once the process it signals has exited. +var errSignalExited = &Error{Kind: ErrExited, Op: "signal", Err: os.ErrProcessDone} + // Signal delivers sig to every process in the view while the process runs. Once the process has exited it delivers nothing and returns ErrExited, even while the processes it left still drain. func (v *View) Signal(sig syscall.Signal) error { - v.signalMu.Lock() - defer v.signalMu.Unlock() - exited := &Error{Kind: ErrExited, Op: "signal", Err: os.ErrProcessDone} if v.exited.Load() { - return exited + return errSignalExited } + r, err := v.request(message{Kind: msgSignal, Signal: sig}) + if (err == ErrClosed && v.exited.Load()) || (err == nil && !r.Delivered) { + return errSignalExited + } + return err +} + +// request sends m with the descriptors of files and returns the launcher's reply, or ErrClosed once the view has ended. The launcher answers requests in order, one at a time. +func (v *View) request(m message, files ...*os.File) (reply, error) { + v.requestMu.Lock() + defer v.requestMu.Unlock() select { case <-v.done: - return ErrClosed + return reply{}, ErrClosed default: } - if err := v.ctl.send(message{Kind: msgSignal, Signal: sig}); err != nil { - return &Error{Kind: ErrLauncher, Op: "signal", Err: err} + fds := make([]int, len(files)) + for i, f := range files { + fds[i] = int(f.Fd()) + } + if err := v.ctl.send(m, fds...); err != nil { + return reply{}, &Error{Kind: ErrLauncher, Op: "request", Err: err} } select { - case delivered := <-v.signaled: - if !delivered { - return exited - } - return nil + case r := <-v.replies: + return r, nil case <-v.done: - if v.exited.Load() { - return exited - } - return ErrClosed + return reply{}, ErrClosed } } +// Spawn starts path with args, env and dir as another process in the view while the view's process runs, with stdio as its stdin, stdout and stderr; the caller keeps its files. The package documentation says how a spawned process runs and ends. Spawn returns ErrExited once the view's process has exited, ErrClosed once the view has ended, ErrLauncher when the launcher is lost and ErrExec when path did not start. +func (v *View) Spawn(path string, args, env []string, dir string, stdio [3]*os.File) (*Spawned, error) { + r, err := v.request(message{Kind: msgSpawn, Command: command{Path: path, Args: args, Env: env, Dir: dir}}, stdio[:]...) + switch { + case err != nil: + return nil, err + case r.spawned == nil: + return nil, r.Fail.err() + } + return r.spawned, nil +} + +// Spawned is a process that Spawn started. It is a clirunner.Handle. +type Spawned struct { + v *View + pid int + done chan struct{} // closed once it has been reaped or the view has ended + exit Exit + err error +} + +// Signal delivers sig to the process's group until the process has exited, and then returns ErrExited. +func (s *Spawned) Signal(sig syscall.Signal) error { + r, err := s.v.request(message{Kind: msgSignal, Pid: s.pid, Signal: sig}) + // The view's end has ended the process. + if err == ErrClosed || (err == nil && !r.Delivered) { + return errSignalExited + } + return err +} + +// Wait returns the process's exit code, or -1 when a signal ended it, once it has been reaped, which also kills what remains of its process group. It returns ErrClosed when the view ended first. +func (s *Spawned) Wait() (int, error) { + <-s.done + if s.exit.Signal != 0 { + return -1, s.err + } + return s.exit.Code, s.err +} + +// Close kills the process's group. The view's end kills it in any case. +func (s *Spawned) Close() error { + _ = s.Signal(syscall.SIGKILL) + return nil +} + // Close kills the view, waits for its teardown and closes the pipes it created. It returns ErrCleanup when the teardown did not finish within its bound, and nil otherwise. func (v *View) Close() error { v.closeOnce.Do(func() { diff --git a/apps/daemon/internal/sessionview/view_linux_test.go b/apps/daemon/internal/sessionview/view_linux_test.go index 5d9d193a4..3b305f7e1 100644 --- a/apps/daemon/internal/sessionview/view_linux_test.go +++ b/apps/daemon/internal/sessionview/view_linux_test.go @@ -183,6 +183,67 @@ func TestViewDescendantsKeepTheGrace(t *testing.T) { } } +// TestSpawnRunsInTheView checks that a spawned process runs as the process's user, unprivileged and in the view's cgroup, that its output and exit come back, and that the view's end ends it. +func TestSpawnRunsInTheView(t *testing.T) { + requireView(t) + f := newFixture(t) + w := &loopbackWorld{dir: f.world} + v, err := Start(context.Background(), f.spec(w, "wait", "OAC_VIEW_TOKEN=oac-unused")) + if err != nil { + t.Fatalf("Start: %v", err) + } + defer v.Close() + if line, err := bufio.NewReader(v.Stdout()).ReadString('\n'); err != nil || line != "ready\n" { + t.Fatalf("harness said %q, %v", line, err) + } + null, err := os.Open(os.DevNull) + if err != nil { + t.Fatal(err) + } + defer null.Close() + token := fmt.Sprintf("oac-spawned-%d", time.Now().UnixNano()) + spawn := func(mode string, stdout *os.File) *Spawned { + t.Helper() + s, err := v.Spawn("/.oac/harness/harness", []string{"harness", token}, []string{helperEnv + "=" + mode}, "/data", [3]*os.File{null, stdout, stdout}) + if err != nil { + t.Fatalf("Spawn %s: %v", mode, err) + } + return s + } + r, wr, err := os.Pipe() + if err != nil { + t.Fatal(err) + } + report := spawn("report", wr) + wr.Close() + out, err := io.ReadAll(r) + r.Close() + if err != nil { + t.Fatal(err) + } + if code, err := report.Wait(); err != nil || code != 3 { + t.Fatalf("spawned Wait = %d, %v; want exit code 3", code, err) + } + cgroups := sessionviewtest.Cgroups(t, f.cgroups) + want := fmt.Sprintf("%d %d 0::", viewID, viewID) + if len(cgroups) != 1 || !strings.HasPrefix(string(out), want) || !strings.HasSuffix(string(out), "/"+filepath.Base(cgroups[0])+"\n") { + t.Fatalf("spawned process reported %q in cgroups %v, want %q and the view's cgroup", out, cgroups, want) + } + sleeper := spawn("sleep", null) + if n := processesWith(t, token); n != 1 { + t.Fatalf("%d spawned processes running, want 1", n) + } + if err := v.Close(); err != nil { + t.Fatalf("Close: %v", err) + } + if _, err := sleeper.Wait(); !errors.Is(err, ErrClosed) { + t.Errorf("spawned Wait after the view ended = %v, want ErrClosed", err) + } + if n := processesWith(t, token); n != 0 { + t.Errorf("%d spawned processes survived the view", n) + } +} + // TestTeardownIsBounded checks that a world server that never ends the request the view's process is blocked on fails the teardown with ErrCleanup within the bound instead of hanging it, and that the view's cgroup stays for Recover after its processes end past the bound. func TestTeardownIsBounded(t *testing.T) { requireView(t) @@ -748,6 +809,10 @@ func runHelper(mode string) int { return 0 case "noop": return 0 + case "report": + cgroup, err := os.ReadFile("/proc/self/cgroup") + fmt.Printf("%d %d %v %s", os.Getuid(), os.Getgid(), errors.Join(err, noPrivileges()), cgroup) + return 3 case "hang": f, err := os.OpenFile("/data/hang", os.O_WRONLY, 0) if err != nil { diff --git a/contracts/agents-api/harness-onboarding.md b/contracts/agents-api/harness-onboarding.md index dd4ca62d4..8679f7cfd 100644 --- a/contracts/agents-api/harness-onboarding.md +++ b/contracts/agents-api/harness-onboarding.md @@ -289,7 +289,7 @@ An agent host runs the Harness outside the sandbox, in a per-Session view. The v ### Executables -Only mount flags grant execution. The closure, `Exec` overlays and the shim are read-only and are the only executable mounts; the sandbox's files and the home are noexec. `Launch` accepts only a `LocalExec` path as `Binary`. A dynamic binary, such as `node`, needs its ELF interpreter as an `Exec` overlay at its `PT_INTERP` path, and every library it loads in the closure, reached through `LD_LIBRARY_PATH`. Nothing loads from the sandbox's files. `viewloader.For` builds this from the binaries' ELF headers: the interpreter's host directory as the `lib` closure mount, the interpreter overlay, empty masks over `/etc/ld.so.preload` and `/etc/ld.so.cache`, and the `LD_LIBRARY_PATH` value. A layout it cannot present, such as a library outside the interpreter's directory, returns `ErrUnsupportedOperation`. +Only mount flags grant execution. The closure, `Exec` overlays and the shim are read-only and are the only executable mounts; the sandbox's files and the home are noexec. `Launch` and `Spawn` accept only a `LocalExec` path as `Binary` and otherwise return `ErrNotLocalExec`. A dynamic binary, such as `node`, needs its ELF interpreter as an `Exec` overlay at its `PT_INTERP` path, and every library it loads in the closure, reached through `LD_LIBRARY_PATH`. Nothing loads from the sandbox's files. `viewloader.For` builds this from the binaries' ELF headers: the interpreter's host directory as the `lib` closure mount, the interpreter overlay, empty masks over `/etc/ld.so.preload` and `/etc/ld.so.cache`, and the `LD_LIBRARY_PATH` value. A layout it cannot present, such as a library outside the interpreter's directory, returns `ErrUnsupportedOperation`. ### Shims @@ -323,6 +323,14 @@ With `ViewProxyNone`, `ViewSession.Proxy` is empty and the view has no generic p - When the Harness exits while other processes remain, the view sends them TERM unless Cancel already did, and ends once they exit or `KillTimeout` passes from the first TERM. - `Wait` closes the stdio ends, returns the context error when Cancel's TERM reached the running Harness and it then exited 0, and `ExitCode` reports the exit once `Done` closes. +### Spawn + +`ViewSession.Spawn` runs a `LocalExec` binary as another process in the live view while the Harness runs, such as a reader of the Harness's native history. Harness-side code that reads Harness-written data runs here, never on the agent host outside the view and never in a view of its own. `Spawn` takes `StartOptions` as `Launch` does and returns the same `clirunner.Process`. The process runs as the Harness does: as the same user, in the same namespaces, view cgroup, world and network, with no capabilities, `no_new_privs` and the same seccomp filter, in a process group of its own. + +- Cancel sends TERM to its process group and kills the group after `KillTimeout`. Once the process exits, what remains of its group is killed. +- The view's end ends it. A Cancel of the Harness reaches it, and when the Harness exits it is one of the processes that remain. +- `Spawn` returns `ErrNotLocalExec` when `Binary` is not a `LocalExec` path, `ErrNoLiveView` when the Session runs no view, and `ErrViewEnded` once the Harness has exited or the view has closed. + ### Qualify the view Run the adapter's Turns, cancellation and continuation in a view, then qualify each declared entry: diff --git a/contracts/agents-api/zh/harness-onboarding.md b/contracts/agents-api/zh/harness-onboarding.md index f9e9f1871..5da809f15 100644 --- a/contracts/agents-api/zh/harness-onboarding.md +++ b/contracts/agents-api/zh/harness-onboarding.md @@ -1,7 +1,7 @@ --- title: "将原生 Harness 添加到 OpenAgentCore" source: contracts/agents-api/harness-onboarding.md -source_hash: 54cdb8254d7f00fbcd78f8ea4a836556db90f12f21dd606a933adb0ead24f652 +source_hash: 1ee6da6ae23ad905c30595b899a8910a2e287670cea640ea5e4a68ae9e35989b --- **Harness** 是一种运行模型和工具循环的原生代理引擎(Codex、Claude Code、MiniMax Code)。**Harness 适配器**将 Runtime 的 Executor 和 Turn 契约转换到该引擎的 SDK 或协议。本文档定义 Runtime–Harness 协议:适配器接口及其生命周期义务、注册、Core 资格认定和验收。[Harness capabilities](harness-capabilities.md) 记录了当前每个 Harness 支持的功能。 @@ -291,7 +291,7 @@ agent host 在沙箱之外、在每个 Session 一个的视图中运行 Harness ### 可执行文件 {#executables} -只有挂载标志授予执行权限。closure、`Exec` overlay 和 shim 是只读的,也是仅有的可执行挂载;沙箱的文件和 home 都是 noexec。`Launch` 只接受 `LocalExec` 路径作为 `Binary`。动态二进制(例如 `node`)需要把它的 ELF 解释器作为 `Exec` overlay 放在其 `PT_INTERP` 路径上,并且它加载的每个库都要在 closure 中,通过 `LD_LIBRARY_PATH` 找到。任何内容都不从沙箱的文件加载。`viewloader.For` 根据二进制的 ELF header 构建这些内容:解释器所在的主机目录作为 `lib` closure 挂载、解释器 overlay、覆盖 `/etc/ld.so.preload` 和 `/etc/ld.so.cache` 的空 mask,以及 `LD_LIBRARY_PATH` 的值。它无法呈现的布局(例如位于解释器目录之外的库)返回 `ErrUnsupportedOperation`。 +只有挂载标志授予执行权限。closure、`Exec` overlay 和 shim 是只读的,也是仅有的可执行挂载;沙箱的文件和 home 都是 noexec。`Launch` 和 `Spawn` 只接受 `LocalExec` 路径作为 `Binary`,否则返回 `ErrNotLocalExec`。动态二进制(例如 `node`)需要把它的 ELF 解释器作为 `Exec` overlay 放在其 `PT_INTERP` 路径上,并且它加载的每个库都要在 closure 中,通过 `LD_LIBRARY_PATH` 找到。任何内容都不从沙箱的文件加载。`viewloader.For` 根据二进制的 ELF header 构建这些内容:解释器所在的主机目录作为 `lib` closure 挂载、解释器 overlay、覆盖 `/etc/ld.so.preload` 和 `/etc/ld.so.cache` 的空 mask,以及 `LD_LIBRARY_PATH` 的值。它无法呈现的布局(例如位于解释器目录之外的库)返回 `ErrUnsupportedOperation`。 ### Shim {#shims} @@ -325,6 +325,14 @@ agent host 根据声明推导进程 broker 的映射表:`/.oac/bin/` 在 - Harness 退出而仍有其他进程时,除非 Cancel 已发送过 TERM,视图会向它们发送 TERM,并在它们退出或自首次 TERM 起经过 `KillTimeout` 后结束。 - `Wait` 关闭 stdio 端;当 Cancel 的 TERM 到达运行中的 Harness 且 Harness 随后以 0 退出时,`Wait` 返回 context 错误;`Done` 关闭后,`ExitCode` 报告退出结果。 +### Spawn {#spawn} + +`ViewSession.Spawn` 在 Harness 运行期间,把一个 `LocalExec` 二进制作为另一个进程运行在活动视图中,例如读取 Harness 原生历史的程序。读取 Harness 所写数据的 Harness 侧代码在这里运行,从不在视图之外的 agent host 上运行,也从不获得自己的视图。`Spawn` 像 `Launch` 一样接收 `StartOptions`,并返回同样的 `clirunner.Process`。该进程的运行方式与 Harness 相同:同一用户,同一组命名空间、视图 cgroup、world 和网络,没有 capability,设置 `no_new_privs` 并使用同一 seccomp 过滤器,且位于自己的进程组中。 + +- Cancel 向它的进程组发送 TERM,并在 `KillTimeout` 后杀死该进程组。进程退出后,其进程组中剩余的进程会被杀死。 +- 视图结束时它也随之结束。Harness 的 Cancel 会到达它;Harness 退出时,它属于仍然存在的进程。 +- `Binary` 不是 `LocalExec` 路径时,`Spawn` 返回 `ErrNotLocalExec`;Session 没有运行中的视图时返回 `ErrNoLiveView`;Harness 已退出或视图已关闭后返回 `ErrViewEnded`。 + ### 认定视图资格 {#qualify-the-view} 在视图中运行适配器的 Turn、取消和续接,然后逐项认定每个声明条目: From 9a20c0f850186b514af3d5b985dc33606daf8f79 Mon Sep 17 00:00:00 2001 From: SaladDay <1203511142@qq.com> Date: Tue, 6 Oct 2026 18:51:41 +0000 Subject: [PATCH 2/5] Keep a blocked spawn from stalling the view and own spawns by ID The restricted thread now only forks. Reaping, the drain and the exit run on another goroutine, and signals and spawns no longer share a lock with a fork. Requests carry IDs, so a pending spawn never blocks a signal and Spawn honours its context. Spawned processes are known by ID until reaped, the group kill on exit is gone, the command travels over a pipe, and ErrViewEnded merges into ErrNoLiveView. --- .../daemon/internal/agent/clirunner/handle.go | 6 +- apps/daemon/internal/agent/harness.go | 17 +- .../daemon/internal/agenthost/launch_linux.go | 9 +- .../internal/agenthost/session_linux_test.go | 15 +- .../internal/agenthost/view_linux_test.go | 15 +- .../internal/sessionview/control_linux.go | 13 +- apps/daemon/internal/sessionview/doc.go | 2 +- .../internal/sessionview/launcher_linux.go | 168 ++++++++++-------- .../daemon/internal/sessionview/view_linux.go | 100 +++++++---- .../internal/sessionview/view_linux_test.go | 117 ++++++++++-- contracts/agents-api/harness-onboarding.md | 7 +- contracts/agents-api/zh/harness-onboarding.md | 9 +- 12 files changed, 313 insertions(+), 165 deletions(-) diff --git a/apps/daemon/internal/agent/clirunner/handle.go b/apps/daemon/internal/agent/clirunner/handle.go index c676efef1..e331ec3ca 100644 --- a/apps/daemon/internal/agent/clirunner/handle.go +++ b/apps/daemon/internal/agent/clirunner/handle.go @@ -12,11 +12,11 @@ import ( "time" ) -// Handle is a running process that clirunner did not start, such as a Harness in an agent-host Session view. Its end also ends every descendant. +// Handle is a running process that clirunner did not start, such as a Harness in an agent-host Session view. The implementation says which of the process's descendants Signal reaches and Wait waits for. type Handle interface { - // Signal delivers sig to the process and every descendant it still has. Once the process itself has exited it delivers nothing and returns an error that matches os.ErrProcessDone, even while its descendants are still ending. + // Signal delivers sig to the process and the descendants the implementation reaches. Once the process itself has exited it delivers nothing and returns an error that matches os.ErrProcessDone, even while its descendants are still ending. Signal(syscall.Signal) error - // Wait returns once the process and its descendants have ended. The code is -1 when a signal ended the process. An error means the exit is unknown. + // Wait returns once the process, and the descendants the implementation waits for, have ended. The code is -1 when a signal ended the process. An error means the exit is unknown. Wait() (int, error) // Close kills whatever still runs and releases the handle. It never closes the stdio ends in HandleOptions, which the Process owns. It is safe to call more than once and after Wait. Close() error diff --git a/apps/daemon/internal/agent/harness.go b/apps/daemon/internal/agent/harness.go index a6eb76ed2..d19e58405 100644 --- a/apps/daemon/internal/agent/harness.go +++ b/apps/daemon/internal/agent/harness.go @@ -154,10 +154,9 @@ var ErrViewHandoff = errors.New("agent: view request carries a connection outsid var ( // ErrNotLocalExec is a Launch or Spawn whose Binary is not a LocalExec path. ErrNotLocalExec = errors.New("agent: binary is not a LocalExec path") - // ErrNoLiveView is a Spawn while the Session has no running view. + // ErrNoLiveView is a Spawn while no view runs its Harness: none was + // launched yet, or its Harness has exited or its view has ended. ErrNoLiveView = errors.New("agent: the Session has no live view") - // ErrViewEnded is a Spawn after the view's Harness exited or its view closed. - ErrViewEnded = errors.New("agent: the view has ended") ) // View declares how the Harness runs in an agent-host Session view. View @@ -265,11 +264,13 @@ type ViewSession struct { // process runs as the Harness does: as the same user, in the same // namespaces, view cgroup, world and network, with no capabilities, // no_new_privs and the same seccomp filter, in a process group of its own. - // Cancel sends TERM to that group and kills it after KillTimeout, and the - // group is killed once the process exits. The view's end ends it too: a - // Cancel of the Harness reaches it, and when the Harness exits it is one - // of the processes that remain. Spawn returns ErrNotLocalExec, - // ErrNoLiveView or ErrViewEnded. + // Parent bounds the wait for the start. Cancel sends TERM to that group + // and kills it after KillTimeout; once the process has exited, Cancel + // delivers nothing and what it left runs on as other processes in the view + // do. The view's end ends them all: a Cancel of the Harness reaches them, + // and when the Harness exits they are among the processes that remain. A + // view that ends after Spawn returned shows in the process's Wait. Spawn + // returns ErrNotLocalExec or ErrNoLiveView. Spawn func(clirunner.StartOptions) (*clirunner.Process, error) } diff --git a/apps/daemon/internal/agenthost/launch_linux.go b/apps/daemon/internal/agenthost/launch_linux.go index 0b7cd2848..34508d74e 100644 --- a/apps/daemon/internal/agenthost/launch_linux.go +++ b/apps/daemon/internal/agenthost/launch_linux.go @@ -38,7 +38,7 @@ type runningView interface { Wait() (sessionview.Exit, error) Close() error Relay() *os.File - Spawn(path string, args, env []string, dir string, stdio [3]*os.File) (*sessionview.Spawned, error) + Spawn(ctx context.Context, path string, args, env []string, dir string, stdio [3]*os.File) (*sessionview.Spawned, error) } // viewWorld is the part of *worldfs.World the Session watches. @@ -116,11 +116,14 @@ func (s *session) spawn(opts clirunner.StartOptions) (*clirunner.Process, error) if v == nil { return nil, &Error{Kind: ErrLaunch, Op: "spawn", Err: agent.ErrNoLiveView} } + if opts.Parent == nil { + opts.Parent = context.Background() + } ends, err := newStdio(opts.NeedStdin) if err != nil { return nil, &Error{Kind: ErrLaunch, Op: "stdio", Err: err} } - p, err := v.Spawn(opts.Binary, append([]string{opts.Binary}, opts.Args...), opts.Env, opts.Dir, ends.child) + p, err := v.Spawn(opts.Parent, opts.Binary, append([]string{opts.Binary}, opts.Args...), opts.Env, opts.Dir, ends.child) ends.closeChild() var process *clirunner.Process if err == nil { @@ -132,7 +135,7 @@ func (s *session) spawn(opts clirunner.StartOptions) (*clirunner.Process, error) if err != nil { ends.closeParent() if errors.Is(err, sessionview.ErrExited) || errors.Is(err, sessionview.ErrClosed) || errors.Is(err, sessionview.ErrLauncher) { - err = fmt.Errorf("%w: %w", agent.ErrViewEnded, err) + err = fmt.Errorf("%w: %w", agent.ErrNoLiveView, err) } return nil, &Error{Kind: ErrLaunch, Op: "spawn", Err: err} } diff --git a/apps/daemon/internal/agenthost/session_linux_test.go b/apps/daemon/internal/agenthost/session_linux_test.go index 7cdf7fa38..f6030ca52 100644 --- a/apps/daemon/internal/agenthost/session_linux_test.go +++ b/apps/daemon/internal/agenthost/session_linux_test.go @@ -77,19 +77,6 @@ func TestViewEndReleasesTheSlotBeforeTheProcessEnds(t *testing.T) { } } -func TestSpawnNeedsARunningView(t *testing.T) { - s := newOwnerSession(t) - s.plan = &plan{view: agent.View{LocalExec: []string{"/.oac/tool/tool"}}} - opts := clirunner.StartOptions{Binary: "/.oac/tool/tool", Dir: "/", OwnProcessGroup: true} - if _, err := s.spawn(opts); !errors.Is(err, agent.ErrNoLiveView) { - t.Fatalf("Spawn without a view = %v, want ErrNoLiveView", err) - } - s.live = &liveView{view: &fakeView{exit: make(chan struct{})}} - if _, err := s.spawn(opts); !errors.Is(err, agent.ErrViewEnded) { - t.Fatalf("Spawn in a closed view = %v, want ErrViewEnded", err) - } -} - func TestFailureDuringTeardownCounts(t *testing.T) { s := newOwnerSession(t) // The relay revokes the attachment while Executor.Close waits. @@ -357,7 +344,7 @@ func (v *fakeView) Signal(syscall.Signal) error { return nil } func (v *fakeView) Relay() *os.File { return nil } -func (v *fakeView) Spawn(string, []string, []string, string, [3]*os.File) (*sessionview.Spawned, error) { +func (v *fakeView) Spawn(context.Context, string, []string, []string, string, [3]*os.File) (*sessionview.Spawned, error) { return nil, sessionview.ErrClosed } diff --git a/apps/daemon/internal/agenthost/view_linux_test.go b/apps/daemon/internal/agenthost/view_linux_test.go index f2e0a4df0..354387fde 100644 --- a/apps/daemon/internal/agenthost/view_linux_test.go +++ b/apps/daemon/internal/agenthost/view_linux_test.go @@ -90,6 +90,7 @@ func TestSessionRunsInAViewOverItsAttachment(t *testing.T) { t.Fatal(err) } copyExecutable(t, filepath.Join(closure, "harness")) + executors := make(chan *testExecutor, 1) register(reg, "test", &agent.View{ Closure: []agent.ViewMount{{Name: "harness", HostDir: closure}}, Masks: []agent.ViewMask{{Path: "/etc/ld.so.preload"}, {Path: "/etc/hostname"}, {Path: "/etc/apt", Dir: true}}, @@ -101,8 +102,13 @@ func TestSessionRunsInAViewOverItsAttachment(t *testing.T) { if err != nil { return nil, err } - return &testExecutor{session: s, dir: req.LocalEnvironment.WorkspaceRoot, - env: []string{harnessEnv + "=1", modelEnv + "=" + provider.BaseURL, caEnv + "=" + cfg.CADir}}, nil + e := &testExecutor{session: s, dir: req.LocalEnvironment.WorkspaceRoot, + env: []string{harnessEnv + "=1", modelEnv + "=" + provider.BaseURL, caEnv + "=" + cfg.CADir}} + select { + case executors <- e: + default: + } + return e, nil }, }) sb.auth.AddRuntime(cfg.Credential, cfg.RuntimeID) @@ -132,6 +138,11 @@ func TestSessionRunsInAViewOverItsAttachment(t *testing.T) { if r.Exit != "" { t.Errorf("Harness: %s; stderr %s", r.Exit, r.Stderr) } + // The Turn waited for its Harness, so no view runs. + e := <-executors + if _, err := e.session.Spawn(clirunner.StartOptions{Binary: harnessPath, Dir: e.dir, OwnProcessGroup: true}); !errors.Is(err, agent.ErrNoLiveView) { + t.Errorf("Spawn after the Harness exited = %v, want ErrNoLiveView", err) + } select { case ok := <-keyed: if !ok { diff --git a/apps/daemon/internal/sessionview/control_linux.go b/apps/daemon/internal/sessionview/control_linux.go index c4476268d..689636c0f 100644 --- a/apps/daemon/internal/sessionview/control_linux.go +++ b/apps/daemon/internal/sessionview/control_linux.go @@ -55,22 +55,23 @@ const ( msgProceed // daemon: the world serves and the network is set up; carries the mount targets msgStarted // launcher: the process runs msgFailed // launcher: construction failed - msgExited // launcher: the process, or the spawned process Pid, ended - msgSignal // daemon: signal the process, or the spawned process Pid + msgExited // launcher: the process, or the spawned process ID, ended + msgSignal // daemon: signal the process, or the spawned process Spawn msgSignaled // launcher: whether msgSignal reached it - msgSpawn // daemon: start Command as another process; carries its stdin, stdout and stderr + msgSpawn // daemon: start another process; carries a pipe with its command, then its stdin, stdout and stderr msgSpawned // launcher: the spawned process's Pid, or why none started ) type message struct { Kind msgKind + ID uint64 // pairs a reply with its request; a spawn's ID also names the process it started, and 0 names the process + Spawn uint64 // the ID of the spawned process msgSignal signals, or 0 Pid int Signal syscall.Signal Delivered bool Exit Exit Fail failure Targets map[string]string // each mountpoint's view path to the path the world presents it at - Command command } // failure carries a launcher *Error across the control socket. @@ -154,7 +155,7 @@ func (c *control) send(m message, fds ...int) error { // recv returns the next message and the files it carries. It returns io.EOF once the peer has closed its end. func (c *control) recv() (message, []*os.File, error) { buf := make([]byte, 64<<10) - oob := make([]byte, unix.CmsgSpace(3*4)) + oob := make([]byte, unix.CmsgSpace(4*4)) n, oobn, flags, _, err := c.conn.ReadMsgUnix(buf, oob) if err != nil { return message{}, nil, err @@ -177,7 +178,7 @@ func (c *control) recv() (message, []*os.File, error) { return m, files, nil } -// interrupt shuts the daemon's end down in both directions, so that a pending recv returns io.EOF and every later send fails. +// interrupt shuts the socket down in both directions, whoever else holds it, so that a pending recv at either end returns io.EOF and every later send fails. func (c *control) interrupt() { c.conn.CloseRead() c.conn.CloseWrite() diff --git a/apps/daemon/internal/sessionview/doc.go b/apps/daemon/internal/sessionview/doc.go index d68a121aa..97698b4e4 100644 --- a/apps/daemon/internal/sessionview/doc.go +++ b/apps/daemon/internal/sessionview/doc.go @@ -6,7 +6,7 @@ // // The daemon calls [Init] first thing in main. [Start] re-executes the daemon binary as the launcher, which becomes PID 1 of the view: it builds the view, starts the process, delivers signals to every process in the view, reaps orphans and exits with the process status. Once the process has exited, Signal reports [ErrExited] and delivers nothing. When the process exits while others remain, the launcher sends them TERM unless one already went to the view, and waits for them up to [Process].Grace from the first TERM. Its exit kills what remains and tears the view down. // -// While the process runs, [View.Spawn] has the launcher start another process in the view, with the stdio descriptors it passes over the control socket. The launcher starts it from the thread that carries the restrictions, as it started the process, so it runs as the same user, in the view's namespaces and cgroup, under the same restrictions and in a session of its own. A spawned process counts like any other process in the view: a signal to the view reaches it, and when the process exits it is among those the launcher sends TERM and waits for. [Spawned].Signal reaches only its process group. Once the launcher reaps a spawned process, it reports the exit and kills what remains of the process group; the view's end ends it in any case. +// While the process runs, [View.Spawn] has the launcher start another process in the view, with the command on a pipe and the stdio descriptors it passes over the control socket. The launcher starts it from the thread that carries the restrictions, as it started the process, so it runs as the same user, in the view's namespaces and cgroup, under the same restrictions and in a session of its own. That thread does nothing else once the process runs: reaping, signals, the drain and the exit go on while a spawn blocks. A spawned process counts like any other process in the view: a signal to the view reaches it, and when the process exits it is among those the launcher sends TERM and waits for. Each spawn has an ID, which [Spawned] uses rather than the pid: [Spawned].Signal reaches the process group until the launcher reaps the process, and reports [ErrExited] from then on. What the process leaves runs on as other processes in the view do, and the view's end ends it. // // A view that declares a [Shim] also runs the Session's process relay, the shim binary in relay mode (package processshim). The launcher starts it before the process, under the same restrictions and as the same user, with a listening socket at processshim.SocketPath on a read-only mount and its end of a socket pair whose other end is [View.Relay]. The relay is the only process that receives the descriptors a shim hands over; the broker outside the view holds only its end of the pair. The launcher's drain ignores the relay, which ends with the view. // diff --git a/apps/daemon/internal/sessionview/launcher_linux.go b/apps/daemon/internal/sessionview/launcher_linux.go index 5242a8c59..2ed56ce10 100644 --- a/apps/daemon/internal/sessionview/launcher_linux.go +++ b/apps/daemon/internal/sessionview/launcher_linux.go @@ -39,15 +39,15 @@ type launcher struct { proc int // the view's /proc relay int // the relay's pid, or 0 - // spawns carries each spawn to the restricted thread, which runs them until reaped closes once the process has been reaped. - spawns chan func() - reaped chan struct{} + spawns chan func() // to the restricted thread, which runs them one at a time - // mu orders signals and spawns against the exits. running holds from the process's start until it is reaped; termAt is when TERM first went to the view; spawned holds each spawned process until it is reaped. - mu sync.Mutex - running bool - termAt time.Time - spawned map[int]bool + // mu orders signals and spawns against the reaping. running holds from the process's start until it is reaped; termAt is when TERM first went to the view; spawned maps the pid of the process and of each spawned process to its ID until the pid is reaped; while forking, forkExits keeps how each pid reaped without an ID ended. + mu sync.Mutex + running bool + termAt time.Time + spawned map[int]uint64 + forking bool + forkExits map[int]unix.WaitStatus } func runLauncher() int { @@ -56,10 +56,14 @@ func runLauncher() int { fmt.Fprintf(os.Stderr, "sessionview launcher: %v\n", err) return 1 } - l := &launcher{ctl: ctl, proceed: make(chan struct{}), spawns: make(chan func()), reaped: make(chan struct{}), spawned: map[int]bool{}} - code, err := l.run() + l := &launcher{ctl: ctl, proceed: make(chan struct{}), spawns: make(chan func()), spawned: map[int]uint64{}, forkExits: map[int]unix.WaitStatus{}} + return l.exit(l.run()) +} + +// exit reports err, if any, and returns the code the launcher exits with. +func (l *launcher) exit(code int, err error) int { if err != nil { - _ = ctl.send(message{Kind: msgFailed, Fail: failureOf(err)}) + _ = l.ctl.send(message{Kind: msgFailed, Fail: failureOf(err)}) return 1 } return code @@ -108,7 +112,7 @@ func (l *launcher) run() (int, error) { return 0, err } l.mu.Lock() - l.running = true + l.running, l.spawned[pid] = true, 0 l.mu.Unlock() for _, fd := range []int{stdinFD, stdoutFD, stderrFD} { unix.Close(fd) @@ -117,22 +121,19 @@ func (l *launcher) run() (int, error) { return 0, &Error{Kind: ErrLauncher, Op: "report start", Err: err} } l.forwardSignals() - var code int go func() { - code, err = l.reap(pid) - close(l.reaped) + code, err := l.reap(pid) + if err == nil { + l.drain(spec.Grace) + } + code = l.exit(code, err) + // A spawn blocked before its exec holds a copy of the control socket, so the daemon learns of the exit from the shutdown. + l.ctl.interrupt() + os.Exit(code) }() - // Only this thread carries the restrictions a process inherits, so spawns start here. + // Only this thread carries the restrictions a process inherits, so every spawn starts here. A spawn that blocks delays only the spawns after it. for { - select { - case spawn := <-l.spawns: - spawn() - case <-l.reaped: - if err == nil { - l.drain(spec.Grace) - } - return code, err - } + (<-l.spawns)() } } @@ -160,45 +161,64 @@ func (l *launcher) serveControl(spec *launchSpec) { close(l.proceed) }) case msgSignal: - _ = l.ctl.send(message{Kind: msgSignaled, Delivered: l.signal(m.Pid, m.Signal)}) + _ = l.ctl.send(message{Kind: msgSignaled, ID: m.ID, Delivered: l.signal(m.Spawn, m.Signal)}) case msgSpawn: - spawn := func() { l.spawn(spec, m.Command, files) } - select { - case l.spawns <- spawn: - case <-l.reaped: - // The process has been reaped, so spawn starts nothing and needs no restricted thread. - spawn() - } + go func() { l.spawns <- func() { l.spawn(spec, m.ID, files) } }() continue } closeFiles(files) } } -// spawn starts c with files as its stdin, stdout and stderr, as the process's user and in a session of its own, and answers with its pid. Holding mu until the answer is sent keeps the report of its exit behind the answer. -func (l *launcher) spawn(spec *launchSpec, c command, files []*os.File) { +// spawn starts the command read from the first of files as the spawned process id, with the other three as its stdin, stdout and stderr, as the process's user and in a session of its own, and answers with its pid. It forks without holding mu, so a child may be reaped before it has an ID; forkExits keeps how it ended. +func (l *launcher) spawn(spec *launchSpec, id uint64, files []*os.File) { defer closeFiles(files) + var c command + err := gob.NewDecoder(files[0]).Decode(&c) + if err != nil { + err = &Error{Kind: ErrLauncher, Op: "read spawn", Err: err} + } l.mu.Lock() - defer l.mu.Unlock() - m := message{Kind: msgSpawned} - err := error(&Error{Kind: ErrExited, Op: "spawn"}) - if l.running { - m.Pid, err = startProcess(spec, c, []uintptr{files[0].Fd(), files[1].Fd(), files[2].Fd()}) + if err == nil && !l.running { + err = &Error{Kind: ErrExited, Op: "spawn"} } + l.forking = err == nil + l.mu.Unlock() + m := message{Kind: msgSpawned, ID: id} + if err == nil { + m.Pid, err = startProcess(spec, c, []uintptr{files[1].Fd(), files[2].Fd(), files[3].Fd()}) + } + l.mu.Lock() + defer l.mu.Unlock() + ws, reaped := l.forkExits[m.Pid] + l.forking = false + clear(l.forkExits) if err != nil { m.Fail = failureOf(err) - } else { - l.spawned[m.Pid] = true } _ = l.ctl.send(m) + switch { + case err != nil: + // An earlier process with the same pid may have been reaped while forking; only a pid that is gone is the child's. + case reaped && unix.Kill(m.Pid, 0) == unix.ESRCH: + exit, _ := exitOf(ws) + _ = l.ctl.send(message{Kind: msgExited, ID: id, Exit: exit}) + default: + l.spawned[m.Pid] = id + } } -// signal delivers sig and reports whether it did. With pid 0 it signals every process in the view while the process runs: as PID 1 of the view, the launcher reaches them all with kill(-1) and is itself spared. Once the process has been reaped, only the drain signals what remains. Otherwise it signals the process group of the spawned process pid until that is reaped. -func (l *launcher) signal(pid int, sig syscall.Signal) bool { +// signal delivers sig and reports whether it did. With id 0 it signals every process in the view while the process runs: as PID 1 of the view, the launcher reaches them all with kill(-1) and is itself spared. Once the process has been reaped, only the drain signals what remains. Otherwise it signals the process group of the spawned process id until that is reaped; until then its pid, and so its group, cannot be reused. +func (l *launcher) signal(id uint64, sig syscall.Signal) bool { l.mu.Lock() defer l.mu.Unlock() - if pid != 0 { - return l.spawned[pid] && unix.Kill(-pid, sig) == nil + if id != 0 { + for pid, sid := range l.spawned { + if sid == id { + return unix.Kill(-pid, sig) == nil + } + } + return false } if !l.running { return false @@ -234,16 +254,11 @@ func (l *launcher) drain(grace time.Duration) { reaped := make(chan struct{}) go func() { defer close(reaped) - // The last process other than the relay to exit is a child of the launcher by then, so its exit ends the Wait4. + // The last process other than the relay to exit is a child of the launcher by then, so its exit ends the wait. for l.othersRemain() { - var ws unix.WaitStatus - pid, err := unix.Wait4(-1, &ws, 0, nil) - if err != nil && err != unix.EINTR { + if _, _, err := l.reapOne(); err != nil { return } - if err == nil { - l.spawnExited(pid, ws) - } } }() timer := time.NewTimer(grace) @@ -277,38 +292,39 @@ func (l *launcher) othersRemain() bool { // reap collects every child, since orphans in the view reparent to PID 1, until the process exits. func (l *launcher) reap(pid int) (int, error) { for { - var ws unix.WaitStatus - wpid, err := unix.Wait4(-1, &ws, 0, nil) - if err == unix.EINTR { - continue - } + wpid, ws, err := l.reapOne() if err != nil { return 0, &Error{Kind: ErrLauncher, Op: "wait", Err: err} } - if wpid != pid { - l.spawnExited(wpid, ws) - continue + if wpid == pid { + _, code := exitOf(ws) + return code, nil } - l.mu.Lock() - l.running = false - l.mu.Unlock() - exit, code := exitOf(ws) - _ = l.ctl.send(message{Kind: msgExited, Exit: exit}) - return code, nil } } -// spawnExited reports the exit of the spawned process pid, which a wait collected, and kills what remains of its process group. Any other pid is one of the view's orphans. -func (l *launcher) spawnExited(pid int, ws unix.WaitStatus) { +// reapOne waits for a child to end and reaps one child, if any is left to reap, and reports the exit of a pid with an ID. It waits without mu and reaps under it, so a pid keeps its ID until it is reaped and signals never wait for an exit. +func (l *launcher) reapOne() (int, unix.WaitStatus, error) { + if err := unix.Waitid(unix.P_ALL, 0, nil, unix.WEXITED|unix.WNOWAIT, nil); err != nil && err != unix.EINTR { + return 0, 0, err + } l.mu.Lock() defer l.mu.Unlock() - if !l.spawned[pid] { - return - } - delete(l.spawned, pid) - _ = unix.Kill(-pid, unix.SIGKILL) - exit, _ := exitOf(ws) - _ = l.ctl.send(message{Kind: msgExited, Pid: pid, Exit: exit}) + var ws unix.WaitStatus + // A failed exec reaps its own child, so there may be none left to reap. + pid, err := unix.Wait4(-1, &ws, unix.WNOHANG, nil) + if err != nil || pid <= 0 { + return 0, 0, nil + } + if id, ok := l.spawned[pid]; ok { + delete(l.spawned, pid) + l.running = l.running && id != 0 + exit, _ := exitOf(ws) + _ = l.ctl.send(message{Kind: msgExited, ID: id, Exit: exit}) + } else if l.forking { + l.forkExits[pid] = ws + } + return pid, ws, nil } // exitOf describes how a process ended, with the code the launcher exits with for it. diff --git a/apps/daemon/internal/sessionview/view_linux.go b/apps/daemon/internal/sessionview/view_linux.go index 5f00d65d0..474ac87eb 100644 --- a/apps/daemon/internal/sessionview/view_linux.go +++ b/apps/daemon/internal/sessionview/view_linux.go @@ -55,7 +55,8 @@ type View struct { waitErr error cleanupErr error // set before done closes requestMu sync.Mutex - replies chan reply + lastID uint64 + requests map[uint64]chan reply // by ID; nil once the launcher stopped answering exited atomic.Bool closing atomic.Bool closeOnce sync.Once @@ -72,7 +73,7 @@ func Start(ctx context.Context, spec Spec) (*View, error) { if err := Probe(); err != nil { return nil, err } - v := &View{done: make(chan struct{}), replies: make(chan reply, 1)} + v := &View{done: make(chan struct{}), requests: map[uint64]chan reply{}} if err := v.launch(&spec); err != nil { return nil, v.abort(err) } @@ -401,7 +402,7 @@ func (v *View) watch() { defer close(v.done) var exit *Exit var failed error - spawned := map[int]*Spawned{} + spawned := map[uint64]*Spawned{} for { m, files, err := v.ctl.recv() closeFiles(files) @@ -413,28 +414,37 @@ func (v *View) watch() { } switch m.Kind { case msgExited: - if s := spawned[m.Pid]; s != nil { - delete(spawned, m.Pid) + if m.ID == 0 { + exit = &m.Exit + v.exited.Store(true) + } else if s := spawned[m.ID]; s != nil { + delete(spawned, m.ID) s.exit = m.Exit close(s.done) - break } - exit = &m.Exit - v.exited.Store(true) - case msgSignaled: - v.replies <- reply{message: m} - case msgSpawned: + case msgSignaled, msgSpawned: r := reply{message: m} - if m.Pid != 0 { + if m.Kind == msgSpawned && m.Pid != 0 { // Registered before the next message, which may report its exit. - r.spawned = &Spawned{v: v, pid: m.Pid, done: make(chan struct{})} - spawned[m.Pid] = r.spawned + r.spawned = &Spawned{v: v, id: m.ID, done: make(chan struct{})} + spawned[m.ID] = r.spawned } - v.replies <- r + v.requestMu.Lock() + if replies := v.requests[m.ID]; replies != nil { + delete(v.requests, m.ID) + replies <- r + } + v.requestMu.Unlock() case msgFailed: failed = m.Fail.err() } } + v.requestMu.Lock() + for _, replies := range v.requests { + close(replies) + } + v.requests = nil + v.requestMu.Unlock() werr, terr := v.teardown() for _, s := range spawned { s.err = ErrClosed @@ -472,40 +482,64 @@ func (v *View) Signal(sig syscall.Signal) error { if v.exited.Load() { return errSignalExited } - r, err := v.request(message{Kind: msgSignal, Signal: sig}) + r, err := v.request(context.Background(), message{Kind: msgSignal, Signal: sig}) if (err == ErrClosed && v.exited.Load()) || (err == nil && !r.Delivered) { return errSignalExited } return err } -// request sends m with the descriptors of files and returns the launcher's reply, or ErrClosed once the view has ended. The launcher answers requests in order, one at a time. -func (v *View) request(m message, files ...*os.File) (reply, error) { +// request sends m with the descriptors of files under a new ID and returns the launcher's reply to it, or ErrClosed once the launcher stopped answering. When ctx ends first, it returns ctx's error and closes the process a late spawn reply brings. +func (v *View) request(ctx context.Context, m message, files ...*os.File) (reply, error) { + replies := make(chan reply, 1) v.requestMu.Lock() - defer v.requestMu.Unlock() - select { - case <-v.done: + if v.requests == nil { + v.requestMu.Unlock() return reply{}, ErrClosed - default: } + v.lastID++ + m.ID = v.lastID + v.requests[m.ID] = replies + v.requestMu.Unlock() fds := make([]int, len(files)) for i, f := range files { fds[i] = int(f.Fd()) } if err := v.ctl.send(m, fds...); err != nil { + v.requestMu.Lock() + delete(v.requests, m.ID) + v.requestMu.Unlock() return reply{}, &Error{Kind: ErrLauncher, Op: "request", Err: err} } select { - case r := <-v.replies: + case r, ok := <-replies: + if !ok { + return reply{}, ErrClosed + } return r, nil - case <-v.done: - return reply{}, ErrClosed + case <-ctx.Done(): + go func() { + if r := <-replies; r.spawned != nil { + r.spawned.Close() + } + }() + return reply{}, ctx.Err() } } -// Spawn starts path with args, env and dir as another process in the view while the view's process runs, with stdio as its stdin, stdout and stderr; the caller keeps its files. The package documentation says how a spawned process runs and ends. Spawn returns ErrExited once the view's process has exited, ErrClosed once the view has ended, ErrLauncher when the launcher is lost and ErrExec when path did not start. -func (v *View) Spawn(path string, args, env []string, dir string, stdio [3]*os.File) (*Spawned, error) { - r, err := v.request(message{Kind: msgSpawn, Command: command{Path: path, Args: args, Env: env, Dir: dir}}, stdio[:]...) +// Spawn starts path with args, env and dir as another process in the view while the view's process runs, with stdio as its stdin, stdout and stderr; the caller keeps its files. The package documentation says how a spawned process runs and ends. ctx bounds the wait for the start: once it ends, Spawn returns its error, and a process that starts after all is killed. Spawn returns ErrExited once the view's process has exited, ErrClosed once the view has ended, ErrLauncher when the launcher is lost and ErrExec when path did not start. +func (v *View) Spawn(ctx context.Context, path string, args, env []string, dir string, stdio [3]*os.File) (*Spawned, error) { + cmdR, cmdW, err := os.Pipe() + if err != nil { + return nil, &Error{Kind: ErrLauncher, Op: "pipe", Err: err} + } + // The command travels over a pipe, as the spec does, so its size is exec's to bound. A write the launcher never reads fails once the launcher's end closes. + go func() { + _ = gob.NewEncoder(cmdW).Encode(command{Path: path, Args: args, Env: env, Dir: dir}) + cmdW.Close() + }() + r, err := v.request(ctx, message{Kind: msgSpawn}, cmdR, stdio[0], stdio[1], stdio[2]) + cmdR.Close() switch { case err != nil: return nil, err @@ -518,7 +552,7 @@ func (v *View) Spawn(path string, args, env []string, dir string, stdio [3]*os.F // Spawned is a process that Spawn started. It is a clirunner.Handle. type Spawned struct { v *View - pid int + id uint64 done chan struct{} // closed once it has been reaped or the view has ended exit Exit err error @@ -526,7 +560,7 @@ type Spawned struct { // Signal delivers sig to the process's group until the process has exited, and then returns ErrExited. func (s *Spawned) Signal(sig syscall.Signal) error { - r, err := s.v.request(message{Kind: msgSignal, Pid: s.pid, Signal: sig}) + r, err := s.v.request(context.Background(), message{Kind: msgSignal, Spawn: s.id, Signal: sig}) // The view's end has ended the process. if err == ErrClosed || (err == nil && !r.Delivered) { return errSignalExited @@ -534,7 +568,7 @@ func (s *Spawned) Signal(sig syscall.Signal) error { return err } -// Wait returns the process's exit code, or -1 when a signal ended it, once it has been reaped, which also kills what remains of its process group. It returns ErrClosed when the view ended first. +// Wait returns the process's exit code, or -1 when a signal ended it, once it has been reaped. What remains of its process group runs on as other processes in the view do. Wait returns ErrClosed when the view ended first. func (s *Spawned) Wait() (int, error) { <-s.done if s.exit.Signal != 0 { @@ -543,7 +577,7 @@ func (s *Spawned) Wait() (int, error) { return s.exit.Code, s.err } -// Close kills the process's group. The view's end kills it in any case. +// Close kills the process's group while the process runs. The view's end kills it in any case. func (s *Spawned) Close() error { _ = s.Signal(syscall.SIGKILL) return nil @@ -554,6 +588,8 @@ func (v *View) Close() error { v.closeOnce.Do(func() { v.closing.Store(true) _ = v.cmd.Process.Kill() + // A spawn blocked before its exec outlives the launcher with a copy of the control socket, until the teardown stops the world. + v.ctl.interrupt() <-v.done v.closePipes() }) diff --git a/apps/daemon/internal/sessionview/view_linux_test.go b/apps/daemon/internal/sessionview/view_linux_test.go index 3b305f7e1..c3c5f96f7 100644 --- a/apps/daemon/internal/sessionview/view_linux_test.go +++ b/apps/daemon/internal/sessionview/view_linux_test.go @@ -143,7 +143,7 @@ func TestViewSignalAndTeardown(t *testing.T) { } } -// TestViewDescendantsKeepTheGrace checks that a helper still cleaning up when the Harness exits gets TERM and finishes within the grace. +// TestViewDescendantsKeepTheGrace checks that helpers still cleaning up when their parent exits, the Harness or a spawned process, get TERM and finish within the grace. func TestViewDescendantsKeepTheGrace(t *testing.T) { requireView(t) f := newFixture(t) @@ -155,8 +155,25 @@ func TestViewDescendantsKeepTheGrace(t *testing.T) { t.Fatalf("Start: %v", err) } defer v.Close() - if line, err := bufio.NewReader(v.Stdout()).ReadString('\n'); err != nil || line != "ready\n" { - t.Fatalf("helper said %q, %v", line, err) + null, err := os.Open(os.DevNull) + if err != nil { + t.Fatal(err) + } + defer null.Close() + r, wr, err := os.Pipe() + if err != nil { + t.Fatal(err) + } + defer r.Close() + spawned, err := v.Spawn(context.Background(), "/.oac/harness/harness", []string{"harness"}, []string{helperEnv + "=cleanup", "OAC_VIEW_TOKEN=-spawned"}, "/data", [3]*os.File{null, wr, os.Stderr}) + wr.Close() + if err != nil { + t.Fatalf("Spawn: %v", err) + } + for _, stdout := range []io.Reader{v.Stdout(), r} { + if line, err := bufio.NewReader(stdout).ReadString('\n'); err != nil || line != "ready\n" { + t.Fatalf("helper said %q, %v", line, err) + } } started := time.Now() if err := v.Signal(syscall.SIGTERM); err != nil { @@ -178,12 +195,17 @@ func TestViewDescendantsKeepTheGrace(t *testing.T) { if elapsed := time.Since(started); elapsed >= spec.Process.Grace { t.Errorf("view ended after %v, want once the helper exited", elapsed) } - if got, err := os.ReadFile(filepath.Join(f.world, "data", "cleaned")); err != nil || string(got) != "done" { - t.Errorf("helper cleanup = %q, %v; want it finished", got, err) + for _, name := range []string{"cleaned", "cleaned-spawned"} { + if got, err := os.ReadFile(filepath.Join(f.world, "data", name)); err != nil || string(got) != "done" { + t.Errorf("helper cleanup %s = %q, %v; want it finished", name, got, err) + } + } + if code, err := spawned.Wait(); err != nil || code != 7 { + t.Errorf("spawned Wait = %d, %v; want exit code 7", code, err) } } -// TestSpawnRunsInTheView checks that a spawned process runs as the process's user, unprivileged and in the view's cgroup, that its output and exit come back, and that the view's end ends it. +// TestSpawnRunsInTheView checks that a spawned process runs as the process's user, unprivileged and in the view's cgroup, that its output and exit come back, that its handle reaches nothing once it has exited, that a command of any size either starts or fails alone, and that the view's end ends it. func TestSpawnRunsInTheView(t *testing.T) { requireView(t) f := newFixture(t) @@ -204,7 +226,7 @@ func TestSpawnRunsInTheView(t *testing.T) { token := fmt.Sprintf("oac-spawned-%d", time.Now().UnixNano()) spawn := func(mode string, stdout *os.File) *Spawned { t.Helper() - s, err := v.Spawn("/.oac/harness/harness", []string{"harness", token}, []string{helperEnv + "=" + mode}, "/data", [3]*os.File{null, stdout, stdout}) + s, err := v.Spawn(context.Background(), "/.oac/harness/harness", []string{"harness", token}, []string{helperEnv + "=" + mode}, "/data", [3]*os.File{null, stdout, stdout}) if err != nil { t.Fatalf("Spawn %s: %v", mode, err) } @@ -233,6 +255,25 @@ func TestSpawnRunsInTheView(t *testing.T) { if n := processesWith(t, token); n != 1 { t.Fatalf("%d spawned processes running, want 1", n) } + if err := report.Signal(syscall.SIGKILL); !errors.Is(err, ErrExited) { + t.Errorf("Signal after the spawned process exited = %v, want ErrExited", err) + } + if err := sleeper.Signal(0); err != nil { + t.Errorf("Signal to the running spawned process = %v", err) + } + // One argument above the control socket's packet size starts; one above exec's limit fails alone. + for size, want := range map[int]error{100 << 10: nil, 200 << 10: ErrExec} { + s, err := v.Spawn(context.Background(), "/.oac/harness/harness", []string{"harness", strings.Repeat("x", size)}, []string{helperEnv + "=noop"}, "/data", [3]*os.File{null, null, null}) + if err == nil { + _, err = s.Wait() + } + if !errors.Is(err, want) { + t.Errorf("Spawn with a %d byte argument = %v, want %v", size, err, want) + } + } + if err := v.Signal(0); err != nil || processesWith(t, token) != 1 { + t.Fatalf("after the spawns, the view's Signal = %v and the sleeper runs %d times; want both running", err, processesWith(t, token)) + } if err := v.Close(); err != nil { t.Fatalf("Close: %v", err) } @@ -244,6 +285,55 @@ func TestSpawnRunsInTheView(t *testing.T) { } } +// TestBlockedSpawnBlocksNothingElse checks that a spawn blocked on the world leaves the view's signals, its own context, the process's exit and the teardown free. +func TestBlockedSpawnBlocksNothingElse(t *testing.T) { + requireView(t) + f := newFixture(t) + mkdir(t, filepath.Join(f.world, "data", "stall")) + w := &stallWorld{loopbackWorld: loopbackWorld{dir: f.world}, name: "stall", stalled: make(chan struct{}), release: make(chan struct{})} + spec := f.spec(&w.loopbackWorld, "wait", "OAC_VIEW_TOKEN=oac-unused") + spec.World = w.serve + v, err := Start(context.Background(), spec) + if err != nil { + t.Fatalf("Start: %v", err) + } + defer v.Close() + if line, err := bufio.NewReader(v.Stdout()).ReadString('\n'); err != nil || line != "ready\n" { + t.Fatalf("harness said %q, %v", line, err) + } + null, err := os.Open(os.DevNull) + if err != nil { + t.Fatal(err) + } + defer null.Close() + ctx, cancel := context.WithCancel(context.Background()) + spawned := make(chan error, 1) + go func() { + _, err := v.Spawn(ctx, "/.oac/harness/harness", []string{"harness"}, []string{helperEnv + "=noop"}, "/data/stall", [3]*os.File{null, null, null}) + spawned <- err + }() + select { + case <-w.stalled: + case <-time.After(10 * time.Second): + t.Fatal("the spawn never looked up its directory in the world") + } + if err := v.Signal(syscall.SIGTERM); err != nil { + t.Fatalf("Signal while a spawn blocks: %v", err) + } + cancel() + select { + case err := <-spawned: + if !errors.Is(err, context.Canceled) { + t.Errorf("Spawn = %v, want context.Canceled", err) + } + case <-time.After(10 * time.Second): + t.Fatal("Spawn did not return after its context ended") + } + if exit, err := v.Wait(); err != nil || exit != (Exit{Code: 7}) { + t.Fatalf("Wait = %+v, %v; want exit code 7", exit, err) + } +} + // TestTeardownIsBounded checks that a world server that never ends the request the view's process is blocked on fails the teardown with ErrCleanup within the bound instead of hanging it, and that the view's cgroup stays for Recover after its processes end past the bound. func TestTeardownIsBounded(t *testing.T) { requireView(t) @@ -388,7 +478,7 @@ func TestRemoveCgroupKeepsItPastTheDeadline(t *testing.T) { func TestCancelledStartStopsTheWorld(t *testing.T) { requireView(t) f := newFixture(t) - w := &stallWorld{loopbackWorld: loopbackWorld{dir: f.world}, stalled: make(chan struct{}), release: make(chan struct{})} + w := &stallWorld{loopbackWorld: loopbackWorld{dir: f.world}, name: "proc", stalled: make(chan struct{}), release: make(chan struct{})} spec := f.spec(&w.loopbackWorld, "noop") spec.World = w.serve ctx, cancel := context.WithCancel(context.Background()) @@ -670,10 +760,11 @@ func (n *hangNode) Write(context.Context, gofs.FileHandle, []byte, int64) (uint3 return 0, syscall.EIO } -// stallWorld is a loopbackWorld that answers no lookup of proc until Stop. +// stallWorld is a loopbackWorld that answers no lookup of name until Stop. type stallWorld struct { loopbackWorld - stalled, release chan struct{} // stalled closes at the first lookup of proc + name string + stalled, release chan struct{} // stalled closes at the first lookup of name stallOnce sync.Once } @@ -683,7 +774,7 @@ func (w *stallWorld) serve(_ context.Context, dev *os.File, mount WorldMount) (W return nil, Presentation{}, err } root.(*gofs.LoopbackNode).RootData.NewNode = func(r *gofs.LoopbackRoot, _ *gofs.Inode, name string, _ *syscall.Stat_t) gofs.InodeEmbedder { - if name == "proc" { + if name == w.name { w.stallOnce.Do(func() { close(w.stalled) }) <-w.release } @@ -786,7 +877,7 @@ func runHelper(mode string) int { sigs := make(chan os.Signal, 1) signal.Notify(sigs, syscall.SIGTERM) child := exec.Command("/.oac/harness/harness") - child.Env = []string{helperEnv + "=slow-term"} + child.Env = []string{helperEnv + "=slow-term", "OAC_VIEW_TOKEN=" + os.Getenv("OAC_VIEW_TOKEN")} child.Stdout = os.Stdout if err := child.Start(); err != nil { fmt.Fprintln(os.Stderr, err) @@ -802,7 +893,7 @@ func runHelper(mode string) int { <-sigs signal.Reset(syscall.SIGTERM) time.Sleep(300 * time.Millisecond) - if err := os.WriteFile("/data/cleaned", []byte("done"), 0o644); err != nil { + if err := os.WriteFile("/data/cleaned"+os.Getenv("OAC_VIEW_TOKEN"), []byte("done"), 0o644); err != nil { fmt.Fprintln(os.Stderr, err) return 1 } diff --git a/contracts/agents-api/harness-onboarding.md b/contracts/agents-api/harness-onboarding.md index 8679f7cfd..940932126 100644 --- a/contracts/agents-api/harness-onboarding.md +++ b/contracts/agents-api/harness-onboarding.md @@ -327,9 +327,10 @@ With `ViewProxyNone`, `ViewSession.Proxy` is empty and the view has no generic p `ViewSession.Spawn` runs a `LocalExec` binary as another process in the live view while the Harness runs, such as a reader of the Harness's native history. Harness-side code that reads Harness-written data runs here, never on the agent host outside the view and never in a view of its own. `Spawn` takes `StartOptions` as `Launch` does and returns the same `clirunner.Process`. The process runs as the Harness does: as the same user, in the same namespaces, view cgroup, world and network, with no capabilities, `no_new_privs` and the same seccomp filter, in a process group of its own. -- Cancel sends TERM to its process group and kills the group after `KillTimeout`. Once the process exits, what remains of its group is killed. -- The view's end ends it. A Cancel of the Harness reaches it, and when the Harness exits it is one of the processes that remain. -- `Spawn` returns `ErrNotLocalExec` when `Binary` is not a `LocalExec` path, `ErrNoLiveView` when the Session runs no view, and `ErrViewEnded` once the Harness has exited or the view has closed. +- `Parent` bounds the wait for the start. Once it ends, `Spawn` returns its error and kills a process that starts after all. +- Cancel sends TERM to its process group and kills the group after `KillTimeout`. Once the process has exited, Cancel delivers nothing, and what it left runs on as the view's other processes do. +- The view's end ends them all. A Cancel of the Harness reaches them, and when the Harness exits they are among the processes that remain. A view that ends after `Spawn` returned shows in the process's `Wait`. +- `Spawn` returns `ErrNotLocalExec` when `Binary` is not a `LocalExec` path, and `ErrNoLiveView` when no view runs its Harness: none was launched yet, or its Harness has exited or its view has ended. ### Qualify the view diff --git a/contracts/agents-api/zh/harness-onboarding.md b/contracts/agents-api/zh/harness-onboarding.md index 5da809f15..bfbe34bde 100644 --- a/contracts/agents-api/zh/harness-onboarding.md +++ b/contracts/agents-api/zh/harness-onboarding.md @@ -1,7 +1,7 @@ --- title: "将原生 Harness 添加到 OpenAgentCore" source: contracts/agents-api/harness-onboarding.md -source_hash: 1ee6da6ae23ad905c30595b899a8910a2e287670cea640ea5e4a68ae9e35989b +source_hash: db98d69501c492ec945f9e4088cccb202ee9a5f0b99c041c33e1fd303cb0623c --- **Harness** 是一种运行模型和工具循环的原生代理引擎(Codex、Claude Code、MiniMax Code)。**Harness 适配器**将 Runtime 的 Executor 和 Turn 契约转换到该引擎的 SDK 或协议。本文档定义 Runtime–Harness 协议:适配器接口及其生命周期义务、注册、Core 资格认定和验收。[Harness capabilities](harness-capabilities.md) 记录了当前每个 Harness 支持的功能。 @@ -329,9 +329,10 @@ agent host 根据声明推导进程 broker 的映射表:`/.oac/bin/` 在 `ViewSession.Spawn` 在 Harness 运行期间,把一个 `LocalExec` 二进制作为另一个进程运行在活动视图中,例如读取 Harness 原生历史的程序。读取 Harness 所写数据的 Harness 侧代码在这里运行,从不在视图之外的 agent host 上运行,也从不获得自己的视图。`Spawn` 像 `Launch` 一样接收 `StartOptions`,并返回同样的 `clirunner.Process`。该进程的运行方式与 Harness 相同:同一用户,同一组命名空间、视图 cgroup、world 和网络,没有 capability,设置 `no_new_privs` 并使用同一 seccomp 过滤器,且位于自己的进程组中。 -- Cancel 向它的进程组发送 TERM,并在 `KillTimeout` 后杀死该进程组。进程退出后,其进程组中剩余的进程会被杀死。 -- 视图结束时它也随之结束。Harness 的 Cancel 会到达它;Harness 退出时,它属于仍然存在的进程。 -- `Binary` 不是 `LocalExec` 路径时,`Spawn` 返回 `ErrNotLocalExec`;Session 没有运行中的视图时返回 `ErrNoLiveView`;Harness 已退出或视图已关闭后返回 `ErrViewEnded`。 +- `Parent` 限定等待启动的时间。它结束后,`Spawn` 返回它的错误,并杀死此后仍然启动的进程。 +- Cancel 向它的进程组发送 TERM,并在 `KillTimeout` 后杀死该进程组。进程退出后,Cancel 不再投递任何信号,它遗留的进程像视图中的其他进程一样继续运行。 +- 视图结束时它们全部随之结束。Harness 的 Cancel 会到达它们;Harness 退出时,它们属于仍然存在的进程。`Spawn` 返回之后视图才结束的情况,体现在该进程的 `Wait` 中。 +- `Binary` 不是 `LocalExec` 路径时,`Spawn` 返回 `ErrNotLocalExec`;没有视图在运行其 Harness 时返回 `ErrNoLiveView`:尚未启动视图,或其 Harness 已退出,或其视图已结束。 ### 认定视图资格 {#qualify-the-view} From 3d1dba31c20152f97fd4d063fd35c1aa1e7cd35d Mon Sep 17 00:00:00 2001 From: SaladDay <1203511142@qq.com> Date: Tue, 6 Oct 2026 20:40:23 +0000 Subject: [PATCH 3/5] Start one spawn at a time and keep a child's pid until it is registered A view admits one spawn at a time, and a spawn waiting for its turn holds no descriptors; Spawn makes the stdio pipes. The restricted thread only forks. While it forks, the SIGCHLD-driven reaper reaps registered children only, so a child that exits before its registration keeps its pid. The launcher holds off garbage collection during a fork, which a world stall would otherwise turn into a frozen launcher. Control sends honour their context and a socket shutdown, replies and exits go through one writer, and only an ended view maps to ErrNoLiveView. --- apps/daemon/internal/agent/harness.go | 7 +- .../daemon/internal/agenthost/launch_linux.go | 26 +- .../internal/agenthost/session_linux_test.go | 28 +- .../internal/agenthost/view_linux_test.go | 3 + .../internal/sessionview/control_linux.go | 31 +- .../sessionview/control_linux_test.go | 72 +++ apps/daemon/internal/sessionview/doc.go | 2 +- .../internal/sessionview/launcher_linux.go | 235 +++++---- .../daemon/internal/sessionview/view_linux.go | 139 ++++-- .../internal/sessionview/view_linux_test.go | 471 ++++++++++++++---- contracts/agents-api/harness-onboarding.md | 4 +- contracts/agents-api/zh/harness-onboarding.md | 6 +- 12 files changed, 752 insertions(+), 272 deletions(-) create mode 100644 apps/daemon/internal/sessionview/control_linux_test.go diff --git a/apps/daemon/internal/agent/harness.go b/apps/daemon/internal/agent/harness.go index d19e58405..2f46dae28 100644 --- a/apps/daemon/internal/agent/harness.go +++ b/apps/daemon/internal/agent/harness.go @@ -264,13 +264,16 @@ type ViewSession struct { // process runs as the Harness does: as the same user, in the same // namespaces, view cgroup, world and network, with no capabilities, // no_new_privs and the same seccomp filter, in a process group of its own. - // Parent bounds the wait for the start. Cancel sends TERM to that group + // The view starts one Spawn at a time, and Parent bounds the wait for its + // turn and for the start; once Parent ends, Spawn returns its error and + // kills a process that starts after all. Cancel sends TERM to the group // and kills it after KillTimeout; once the process has exited, Cancel // delivers nothing and what it left runs on as other processes in the view // do. The view's end ends them all: a Cancel of the Harness reaches them, // and when the Harness exits they are among the processes that remain. A // view that ends after Spawn returned shows in the process's Wait. Spawn - // returns ErrNotLocalExec or ErrNoLiveView. + // returns ErrNotLocalExec, ErrNoLiveView, or the error that kept the + // process from starting. Spawn func(clirunner.StartOptions) (*clirunner.Process, error) } diff --git a/apps/daemon/internal/agenthost/launch_linux.go b/apps/daemon/internal/agenthost/launch_linux.go index 34508d74e..6df910e06 100644 --- a/apps/daemon/internal/agenthost/launch_linux.go +++ b/apps/daemon/internal/agenthost/launch_linux.go @@ -38,7 +38,7 @@ type runningView interface { Wait() (sessionview.Exit, error) Close() error Relay() *os.File - Spawn(ctx context.Context, path string, args, env []string, dir string, stdio [3]*os.File) (*sessionview.Spawned, error) + Spawn(ctx context.Context, path string, args, env []string, dir string, stdin bool) (*sessionview.Spawned, error) } // viewWorld is the part of *worldfs.World the Session watches. @@ -119,26 +119,20 @@ func (s *session) spawn(opts clirunner.StartOptions) (*clirunner.Process, error) if opts.Parent == nil { opts.Parent = context.Background() } - ends, err := newStdio(opts.NeedStdin) + p, err := v.Spawn(opts.Parent, opts.Binary, append([]string{opts.Binary}, opts.Args...), opts.Env, opts.Dir, opts.NeedStdin) if err != nil { - return nil, &Error{Kind: ErrLaunch, Op: "stdio", Err: err} - } - p, err := v.Spawn(opts.Parent, opts.Binary, append([]string{opts.Binary}, opts.Args...), opts.Env, opts.Dir, ends.child) - ends.closeChild() - var process *clirunner.Process - if err == nil { - if process, err = clirunner.FromHandle(p, clirunner.HandleOptions{Parent: opts.Parent, Stdin: ends.stdin(), - Stdout: ends.parent[1], Stderr: ends.parent[2], KillTimeout: opts.KillTimeout}); err != nil { - p.Close() - } - } - if err != nil { - ends.closeParent() - if errors.Is(err, sessionview.ErrExited) || errors.Is(err, sessionview.ErrClosed) || errors.Is(err, sessionview.ErrLauncher) { + // Only a view that has ended has no live view; any other failure keeps its own error. + if errors.Is(err, sessionview.ErrExited) || errors.Is(err, sessionview.ErrClosed) { err = fmt.Errorf("%w: %w", agent.ErrNoLiveView, err) } return nil, &Error{Kind: ErrLaunch, Op: "spawn", Err: err} } + var stdin io.WriteCloser + if p.Stdin != nil { + stdin = p.Stdin + } + // FromHandle fails only without stdout and stderr, which a spawned process always has. + process, _ := clirunner.FromHandle(p, clirunner.HandleOptions{Parent: opts.Parent, Stdin: stdin, Stdout: p.Stdout, Stderr: p.Stderr, KillTimeout: opts.KillTimeout}) return process, nil } diff --git a/apps/daemon/internal/agenthost/session_linux_test.go b/apps/daemon/internal/agenthost/session_linux_test.go index f6030ca52..b77848bb0 100644 --- a/apps/daemon/internal/agenthost/session_linux_test.go +++ b/apps/daemon/internal/agenthost/session_linux_test.go @@ -77,6 +77,23 @@ func TestViewEndReleasesTheSlotBeforeTheProcessEnds(t *testing.T) { } } +// TestSpawnKeepsItsErrors checks that only a view that has ended makes a Spawn fail with ErrNoLiveView, and that every other failure keeps its own error. +func TestSpawnKeepsItsErrors(t *testing.T) { + emfile := &sessionview.Error{Kind: sessionview.ErrLauncher, Op: "pipe", Err: syscall.EMFILE} + for _, c := range []struct { + err error + ended bool + }{{sessionview.ErrExited, true}, {sessionview.ErrClosed, true}, {emfile, false}, {sessionview.ErrExec, false}, {context.Canceled, false}} { + s := newOwnerSession(t) + s.plan = &plan{view: agent.View{LocalExec: []string{"/bin/true"}}} + s.live = &liveView{view: &fakeView{exit: make(chan struct{}), spawnErr: c.err}} + _, err := s.spawn(clirunner.StartOptions{Binary: "/bin/true", Dir: "/", OwnProcessGroup: true}) + if !errors.Is(err, c.err) || errors.Is(err, agent.ErrNoLiveView) != c.ended { + t.Errorf("Spawn failing with %v = %v; want that error, and ErrNoLiveView only for an ended view", c.err, err) + } + } +} + func TestFailureDuringTeardownCounts(t *testing.T) { s := newOwnerSession(t) // The relay revokes the attachment while Executor.Close waits. @@ -334,18 +351,19 @@ func (t *fakeTurn) AwaitSettlement(context.Context) (agent.TurnSettlement, error return agent.TurnSettlement{Reusable: t.settleErr == nil}, t.settleErr } -// fakeView is a view that ends when closed. +// fakeView is a view that ends when closed and whose spawns fail with spawnErr. type fakeView struct { - exit chan struct{} - once sync.Once + exit chan struct{} + once sync.Once + spawnErr error } func (v *fakeView) Signal(syscall.Signal) error { return nil } func (v *fakeView) Relay() *os.File { return nil } -func (v *fakeView) Spawn(context.Context, string, []string, []string, string, [3]*os.File) (*sessionview.Spawned, error) { - return nil, sessionview.ErrClosed +func (v *fakeView) Spawn(context.Context, string, []string, []string, string, bool) (*sessionview.Spawned, error) { + return nil, v.spawnErr } func (v *fakeView) Wait() (sessionview.Exit, error) { diff --git a/apps/daemon/internal/agenthost/view_linux_test.go b/apps/daemon/internal/agenthost/view_linux_test.go index 354387fde..04a88b1b0 100644 --- a/apps/daemon/internal/agenthost/view_linux_test.go +++ b/apps/daemon/internal/agenthost/view_linux_test.go @@ -143,6 +143,9 @@ func TestSessionRunsInAViewOverItsAttachment(t *testing.T) { if _, err := e.session.Spawn(clirunner.StartOptions{Binary: harnessPath, Dir: e.dir, OwnProcessGroup: true}); !errors.Is(err, agent.ErrNoLiveView) { t.Errorf("Spawn after the Harness exited = %v, want ErrNoLiveView", err) } + if _, err := e.session.Spawn(clirunner.StartOptions{Binary: "/bin/sh", Dir: e.dir, OwnProcessGroup: true}); !errors.Is(err, agent.ErrNotLocalExec) { + t.Errorf("Spawn of a binary outside LocalExec = %v, want ErrNotLocalExec", err) + } select { case ok := <-keyed: if !ok { diff --git a/apps/daemon/internal/sessionview/control_linux.go b/apps/daemon/internal/sessionview/control_linux.go index 689636c0f..9a77375eb 100644 --- a/apps/daemon/internal/sessionview/control_linux.go +++ b/apps/daemon/internal/sessionview/control_linux.go @@ -4,13 +4,13 @@ package sessionview import ( "bytes" + "context" "encoding/gob" "errors" "fmt" "io" "net" "os" - "sync" "syscall" "time" @@ -119,8 +119,8 @@ func (f failure) err() error { // control is one end of the launcher's SOCK_SEQPACKET control socket. Each packet holds one gob-encoded message. type control struct { - conn *net.UnixConn - mu sync.Mutex + conn *net.UnixConn + sending chan struct{} // held by the send in progress } func newControl(f *os.File) (*control, error) { @@ -134,10 +134,11 @@ func newControl(f *os.File) (*control, error) { c.Close() return nil, fmt.Errorf("control socket is a %T", c) } - return &control{conn: uc}, nil + return &control{conn: uc, sending: make(chan struct{}, 1)}, nil } -func (c *control) send(m message, fds ...int) error { +// send sends m with fds. It returns ctx's error when ctx ends before m is sent, whether it waits for another send or for room on the socket; a packet goes whole or not at all. Once the socket is shut down, a waiting send fails. +func (c *control) send(ctx context.Context, m message, fds ...int) error { var buf bytes.Buffer if err := gob.NewEncoder(&buf).Encode(m); err != nil { return err @@ -146,9 +147,25 @@ func (c *control) send(m message, fds ...int) error { if len(fds) > 0 { oob = unix.UnixRights(fds...) } - c.mu.Lock() - defer c.mu.Unlock() + select { + case c.sending <- struct{}{}: + case <-ctx.Done(): + return ctx.Err() + } + defer func() { <-c.sending }() + expired := make(chan struct{}) + stop := context.AfterFunc(ctx, func() { + c.conn.SetWriteDeadline(time.Unix(1, 0)) + close(expired) + }) _, _, err := c.conn.WriteMsgUnix(buf.Bytes(), oob, nil) + if !stop() { + <-expired + c.conn.SetWriteDeadline(time.Time{}) + if err != nil { + err = ctx.Err() + } + } return err } diff --git a/apps/daemon/internal/sessionview/control_linux_test.go b/apps/daemon/internal/sessionview/control_linux_test.go new file mode 100644 index 000000000..01c2ec337 --- /dev/null +++ b/apps/daemon/internal/sessionview/control_linux_test.go @@ -0,0 +1,72 @@ +//go:build linux + +package sessionview + +import ( + "context" + "errors" + "os" + "strings" + "testing" + "time" + + "golang.org/x/sys/unix" +) + +// TestSendIsBounded checks that a send waiting for room on the socket or for another send returns once its context ends, without disturbing the send in progress, and that a shutdown ends the send in progress. +func TestSendIsBounded(t *testing.T) { + pair, err := unix.Socketpair(unix.AF_UNIX, unix.SOCK_SEQPACKET|unix.SOCK_CLOEXEC, 0) + if err != nil { + t.Fatal(err) + } + peer := os.NewFile(uintptr(pair[1]), "peer") + defer peer.Close() + c, err := newControl(os.NewFile(uintptr(pair[0]), "control")) + if err != nil { + t.Fatal(err) + } + defer c.close() + big := message{Targets: map[string]string{"/": strings.Repeat("x", 32<<10)}} + bounded := func(d time.Duration) error { + ctx, cancel := context.WithTimeout(context.Background(), d) + defer cancel() + return c.send(ctx, big) + } + // The peer reads nothing, so the socket fills. + for { + start := time.Now() + err := bounded(20 * time.Millisecond) + if time.Since(start) > 2*time.Second { + t.Fatalf("send took %v with a 20ms bound", time.Since(start)) + } + if errors.Is(err, context.DeadlineExceeded) { + break + } + if err != nil { + t.Fatal(err) + } + } + blocked := make(chan error, 1) + go func() { blocked <- c.send(context.Background(), big) }() + for len(c.sending) == 0 { + time.Sleep(time.Millisecond) + } + start := time.Now() + if err := bounded(20 * time.Millisecond); !errors.Is(err, context.DeadlineExceeded) || time.Since(start) > 2*time.Second { + t.Errorf("send behind a blocked send = %v after %v, want context.DeadlineExceeded within 20ms", err, time.Since(start)) + } + select { + case err := <-blocked: + t.Fatalf("unbounded send returned %v with the socket full", err) + case <-time.After(50 * time.Millisecond): + } + c.interrupt() + select { + case err := <-blocked: + if err == nil { + t.Error("send after the shutdown succeeded") + } + case <-time.After(2 * time.Second): + t.Fatal("send still blocked 2s after the shutdown") + } +} diff --git a/apps/daemon/internal/sessionview/doc.go b/apps/daemon/internal/sessionview/doc.go index 97698b4e4..1e682da0a 100644 --- a/apps/daemon/internal/sessionview/doc.go +++ b/apps/daemon/internal/sessionview/doc.go @@ -6,7 +6,7 @@ // // The daemon calls [Init] first thing in main. [Start] re-executes the daemon binary as the launcher, which becomes PID 1 of the view: it builds the view, starts the process, delivers signals to every process in the view, reaps orphans and exits with the process status. Once the process has exited, Signal reports [ErrExited] and delivers nothing. When the process exits while others remain, the launcher sends them TERM unless one already went to the view, and waits for them up to [Process].Grace from the first TERM. Its exit kills what remains and tears the view down. // -// While the process runs, [View.Spawn] has the launcher start another process in the view, with the command on a pipe and the stdio descriptors it passes over the control socket. The launcher starts it from the thread that carries the restrictions, as it started the process, so it runs as the same user, in the view's namespaces and cgroup, under the same restrictions and in a session of its own. That thread does nothing else once the process runs: reaping, signals, the drain and the exit go on while a spawn blocks. A spawned process counts like any other process in the view: a signal to the view reaches it, and when the process exits it is among those the launcher sends TERM and waits for. Each spawn has an ID, which [Spawned] uses rather than the pid: [Spawned].Signal reaches the process group until the launcher reaps the process, and reports [ErrExited] from then on. What the process leaves runs on as other processes in the view do, and the view's end ends it. +// While the process runs, [View.Spawn] has the launcher start another process in the view, passing the command on a pipe and the stdio pipes that Spawn makes over the control socket. A view starts one spawn at a time, and a spawn waiting for its turn holds no descriptors. The launcher starts each from the thread that carries the restrictions, as it started the process, so it runs as the same user, in the view's namespaces and cgroup, under the same restrictions and in a session of its own. That thread does nothing else once the process runs, and no garbage collection runs while it forks, so a fork stuck on the world holds up no other thread: reaping, signals, the drain and the exit go on. The launcher reaps as SIGCHLD reports exits. While a spawn forks, it reaps only the processes it has registered, so the pid of a child that exits before its registration stays its own until the registration ties it to its spawn. A spawned process counts like any other process in the view: a signal to the view reaches it, and when the process exits it is among those the launcher sends TERM and waits for. Each spawn has an ID, which [Spawned] uses rather than the pid: [Spawned].Signal reaches the process group until the launcher reaps the process, and reports [ErrExited] from then on. What the process leaves runs on as other processes in the view do, and the view's end ends it. // // A view that declares a [Shim] also runs the Session's process relay, the shim binary in relay mode (package processshim). The launcher starts it before the process, under the same restrictions and as the same user, with a listening socket at processshim.SocketPath on a read-only mount and its end of a socket pair whose other end is [View.Relay]. The relay is the only process that receives the descriptors a shim hands over; the broker outside the view holds only its end of the pair. The launcher's drain ignores the relay, which ends with the view. // diff --git a/apps/daemon/internal/sessionview/launcher_linux.go b/apps/daemon/internal/sessionview/launcher_linux.go index 2ed56ce10..86e4f7902 100644 --- a/apps/daemon/internal/sessionview/launcher_linux.go +++ b/apps/daemon/internal/sessionview/launcher_linux.go @@ -3,11 +3,13 @@ package sessionview import ( + "context" "encoding/gob" "fmt" "os" "os/signal" "runtime" + "runtime/debug" "strconv" "strings" "sync" @@ -26,6 +28,8 @@ func Init() { } // Capability sets, no_new_privs and seccomp filters are per thread: the thread that sets them must be the one that forks the process. runtime.LockOSThread() + // A fork that blocks in the world holds this thread and its P, so another P runs everything else. A fixed count also keeps the runtime from stopping the world to change it. + runtime.GOMAXPROCS(max(2, runtime.GOMAXPROCS(0))) os.Exit(runLauncher()) } @@ -39,15 +43,18 @@ type launcher struct { proc int // the view's /proc relay int // the relay's pid, or 0 - spawns chan func() // to the restricted thread, which runs them one at a time + spawns chan func() // to the restricted thread; the view sends one spawn at a time + chld chan os.Signal // SIGCHLD, and a wake once a spawn has registered its child + ready chan struct{} // wakes the writer - // mu orders signals and spawns against the reaping. running holds from the process's start until it is reaped; termAt is when TERM first went to the view; spawned maps the pid of the process and of each spawned process to its ID until the pid is reaped; while forking, forkExits keeps how each pid reaped without an ID ended. - mu sync.Mutex - running bool - termAt time.Time - spawned map[int]uint64 - forking bool - forkExits map[int]unix.WaitStatus + // mu orders signals, spawns and messages against the reaping. running holds from the process's start until it is reaped, and code is how it exited; termAt is when TERM first went to the view. spawned maps the pid of the process and of each spawned process to its ID until the pid is reaped; forking holds while a spawn starts a child it has not registered yet. out holds the messages the writer sends next. + mu sync.Mutex + running bool + code int + termAt time.Time + spawned map[int]uint64 + forking bool + out []message } func runLauncher() int { @@ -56,46 +63,39 @@ func runLauncher() int { fmt.Fprintf(os.Stderr, "sessionview launcher: %v\n", err) return 1 } - l := &launcher{ctl: ctl, proceed: make(chan struct{}), spawns: make(chan func()), spawned: map[int]uint64{}, forkExits: map[int]unix.WaitStatus{}} - return l.exit(l.run()) + l := &launcher{ctl: ctl, proceed: make(chan struct{}), spawns: make(chan func(), 1), chld: make(chan os.Signal, 1), ready: make(chan struct{}, 1), spawned: map[int]uint64{}} + // run returns only when the build fails; once the process runs, the writer exits. + _ = l.ctl.send(context.Background(), message{Kind: msgFailed, Fail: failureOf(l.run())}) + return 1 } -// exit reports err, if any, and returns the code the launcher exits with. -func (l *launcher) exit(code int, err error) int { - if err != nil { - _ = l.ctl.send(message{Kind: msgFailed, Fail: failureOf(err)}) - return 1 - } - return code -} - -func (l *launcher) run() (int, error) { +func (l *launcher) run() error { spec, err := readSpec() if err != nil { - return 0, err + return err } go l.serveControl(spec) if err := unix.Mount("", "/", "", unix.MS_REC|unix.MS_PRIVATE, ""); err != nil { - return 0, mountError("make-rprivate", "/", err) + return mountError("make-rprivate", "/", err) } if err := loopbackUp(); err != nil { - return 0, &Error{Kind: ErrNetwork, Op: "loopback", Err: err} + return &Error{Kind: ErrNetwork, Op: "loopback", Err: err} } if err := l.mountWorld(spec.Staging); err != nil { - return 0, err + return err } <-l.proceed b := &builder{root: -1, proc: -1, listener: -1, targets: l.targets} defer b.close() if err := b.build(spec); err != nil { - return 0, err + return err } l.proc = b.proc if err := switchRoot(b.root); err != nil { - return 0, err + return err } if err := restrict(); err != nil { - return 0, err + return err } b.closeBuild() if b.listener >= 0 { @@ -104,12 +104,12 @@ func (l *launcher) run() (int, error) { b.closeListener() unix.Close(relayFD) if err != nil { - return 0, err + return err } } pid, err := startProcess(spec, spec.Command, []uintptr{stdinFD, stdoutFD, stderrFD}) if err != nil { - return 0, err + return err } l.mu.Lock() l.running, l.spawned[pid] = true, 0 @@ -117,21 +117,13 @@ func (l *launcher) run() (int, error) { for _, fd := range []int{stdinFD, stdoutFD, stderrFD} { unix.Close(fd) } - if err := l.ctl.send(message{Kind: msgStarted, Pid: pid}); err != nil { - return 0, &Error{Kind: ErrLauncher, Op: "report start", Err: err} + if err := l.ctl.send(context.Background(), message{Kind: msgStarted, Pid: pid}); err != nil { + return &Error{Kind: ErrLauncher, Op: "report start", Err: err} } l.forwardSignals() - go func() { - code, err := l.reap(pid) - if err == nil { - l.drain(spec.Grace) - } - code = l.exit(code, err) - // A spawn blocked before its exec holds a copy of the control socket, so the daemon learns of the exit from the shutdown. - l.ctl.interrupt() - os.Exit(code) - }() - // Only this thread carries the restrictions a process inherits, so every spawn starts here. A spawn that blocks delays only the spawns after it. + go l.write() + go l.reap(spec.Grace) + // Only this thread carries the restrictions a process inherits, so every spawn starts here. for { (<-l.spawns)() } @@ -161,16 +153,19 @@ func (l *launcher) serveControl(spec *launchSpec) { close(l.proceed) }) case msgSignal: - _ = l.ctl.send(message{Kind: msgSignaled, ID: m.ID, Delivered: l.signal(m.Spawn, m.Signal)}) + l.mu.Lock() + l.post(message{Kind: msgSignaled, ID: m.ID, Delivered: l.signal(m.Spawn, m.Signal)}) + l.mu.Unlock() case msgSpawn: - go func() { l.spawns <- func() { l.spawn(spec, m.ID, files) } }() + // The restricted thread takes each spawn before it forks, and the view sends the next only once the last has been answered, so this never waits. + l.spawns <- func() { l.spawn(spec, m.ID, files) } continue } closeFiles(files) } } -// spawn starts the command read from the first of files as the spawned process id, with the other three as its stdin, stdout and stderr, as the process's user and in a session of its own, and answers with its pid. It forks without holding mu, so a child may be reaped before it has an ID; forkExits keeps how it ended. +// spawn starts the command read from the first of files as the spawned process id, with the other three as its stdin, stdout and stderr, as the process's user and in a session of its own, and answers with its pid. It forks without holding mu; while it does, the reaper leaves unregistered children, so the child's pid stays its own until it is registered. func (l *launcher) spawn(spec *launchSpec, id uint64, files []*os.File) { defer closeFiles(files) var c command @@ -186,32 +181,57 @@ func (l *launcher) spawn(spec *launchSpec, id uint64, files []*os.File) { l.mu.Unlock() m := message{Kind: msgSpawned, ID: id} if err == nil { + // A fork that blocks in the world holds this thread and its P where no stop of the world reaches them, so no collection may run or be about to start until the fork returns. Turning collection off stops new ones; runtime.GC waits out one that started before. + gc := debug.SetGCPercent(-1) + runtime.GC() m.Pid, err = startProcess(spec, c, []uintptr{files[1].Fd(), files[2].Fd(), files[3].Fd()}) + debug.SetGCPercent(gc) } l.mu.Lock() - defer l.mu.Unlock() - ws, reaped := l.forkExits[m.Pid] - l.forking = false - clear(l.forkExits) if err != nil { m.Fail = failureOf(err) + } else { + l.spawned[m.Pid] = id } - _ = l.ctl.send(m) - switch { - case err != nil: - // An earlier process with the same pid may have been reaped while forking; only a pid that is gone is the child's. - case reaped && unix.Kill(m.Pid, 0) == unix.ESRCH: - exit, _ := exitOf(ws) - _ = l.ctl.send(message{Kind: msgExited, ID: id, Exit: exit}) + l.forking = false + l.post(m) + l.mu.Unlock() + // SIGCHLD does not queue: the child may have exited while the reaper left it. + select { + case l.chld <- unix.SIGCHLD: + default: + } +} + +// post queues m for the writer, which sends the queued messages in order; a message without a kind exits the launcher once those before it are sent. mu is held. +func (l *launcher) post(m message) { + l.out = append(l.out, m) + select { + case l.ready <- struct{}{}: default: - l.spawned[m.Pid] = id } } -// signal delivers sig and reports whether it did. With id 0 it signals every process in the view while the process runs: as PID 1 of the view, the launcher reaches them all with kill(-1) and is itself spared. Once the process has been reaped, only the drain signals what remains. Otherwise it signals the process group of the spawned process id until that is reaped; until then its pid, and so its group, cannot be reused. +// write sends what post queues, so that neither the reaper nor a signal waits on the socket. +func (l *launcher) write() { + for range l.ready { + l.mu.Lock() + out := l.out + l.out = nil + l.mu.Unlock() + for _, m := range out { + if m.Kind == 0 { + // A spawn blocked before its exec holds a copy of this end, and the exit waits for it, so the daemon learns of the exit from the shutdown. + l.ctl.interrupt() + os.Exit(l.code) + } + _ = l.ctl.send(context.Background(), m) + } + } +} + +// signal delivers sig and reports whether it did. mu is held. With id 0 it signals every process in the view while the process runs: as PID 1 of the view, the launcher reaches them all with kill(-1) and is itself spared. Once the process has been reaped, only the drain signals what remains. Otherwise it signals the process group of the spawned process id until that is reaped; until then its pid, and so its group, cannot be reused. func (l *launcher) signal(id uint64, sig syscall.Signal) bool { - l.mu.Lock() - defer l.mu.Unlock() if id != 0 { for pid, sid := range l.spawned { if sid == id { @@ -235,7 +255,9 @@ func (l *launcher) forwardSignals() { signal.Notify(sigs, unix.SIGHUP, unix.SIGINT, unix.SIGQUIT, unix.SIGTERM, unix.SIGUSR1, unix.SIGUSR2, unix.SIGWINCH) go func() { for s := range sigs { + l.mu.Lock() l.signal(0, s.(syscall.Signal)) + l.mu.Unlock() } }() } @@ -251,21 +273,16 @@ func (l *launcher) drain(grace time.Duration) { if grace <= 0 || (termAt.IsZero() && unix.Kill(-1, unix.SIGTERM) != nil) { return } - reaped := make(chan struct{}) - go func() { - defer close(reaped) - // The last process other than the relay to exit is a child of the launcher by then, so its exit ends the wait. - for l.othersRemain() { - if _, _, err := l.reapOne(); err != nil { - return - } - } - }() timer := time.NewTimer(grace) defer timer.Stop() - select { - case <-reaped: - case <-timer.C: + // The last process other than the relay to exit is a child of the launcher by then, so its SIGCHLD ends the wait. + for l.othersRemain() { + select { + case <-l.chld: + l.reapChildren() + case <-timer.C: + return + } } } @@ -289,42 +306,52 @@ func (l *launcher) othersRemain() bool { return false } -// reap collects every child, since orphans in the view reparent to PID 1, until the process exits. -func (l *launcher) reap(pid int) (int, error) { - for { - wpid, ws, err := l.reapOne() - if err != nil { - return 0, &Error{Kind: ErrLauncher, Op: "wait", Err: err} - } - if wpid == pid { - _, code := exitOf(ws) - return code, nil - } +// reap reaps children as SIGCHLD reports them, starting with those that exited before it began, until the process has exited and the drain has ended, and then has the writer exit the launcher. +func (l *launcher) reap(grace time.Duration) { + signal.Notify(l.chld, unix.SIGCHLD) + for l.reapChildren() { + <-l.chld } + l.drain(grace) + l.mu.Lock() + l.post(message{}) + l.mu.Unlock() } -// reapOne waits for a child to end and reaps one child, if any is left to reap, and reports the exit of a pid with an ID. It waits without mu and reaps under it, so a pid keeps its ID until it is reaped and signals never wait for an exit. -func (l *launcher) reapOne() (int, unix.WaitStatus, error) { - if err := unix.Waitid(unix.P_ALL, 0, nil, unix.WEXITED|unix.WNOWAIT, nil); err != nil && err != unix.EINTR { - return 0, 0, err - } +// reapChildren reaps the children that have exited and reports whether the process still runs. It reaps registered children at any time and other children, the view's orphans and the relay, only while no spawn forks: until its registration, a spawned child that has exited stays a zombie and keeps its pid. +func (l *launcher) reapChildren() bool { l.mu.Lock() defer l.mu.Unlock() var ws unix.WaitStatus - // A failed exec reaps its own child, so there may be none left to reap. - pid, err := unix.Wait4(-1, &ws, unix.WNOHANG, nil) - if err != nil || pid <= 0 { - return 0, 0, nil - } - if id, ok := l.spawned[pid]; ok { - delete(l.spawned, pid) - l.running = l.running && id != 0 - exit, _ := exitOf(ws) - _ = l.ctl.send(message{Kind: msgExited, ID: id, Exit: exit}) - } else if l.forking { - l.forkExits[pid] = ws - } - return pid, ws, nil + if l.forking { + for pid := range l.spawned { + if wpid, _ := unix.Wait4(pid, &ws, unix.WNOHANG, nil); wpid == pid { + l.reaped(pid, ws) + } + } + return l.running + } + for { + pid, _ := unix.Wait4(-1, &ws, unix.WNOHANG, nil) + if pid <= 0 { + return l.running + } + l.reaped(pid, ws) + } +} + +// reaped retires the ID of pid, if it has one, and reports its exit. mu is held. +func (l *launcher) reaped(pid int, ws unix.WaitStatus) { + id, ok := l.spawned[pid] + if !ok { + return + } + delete(l.spawned, pid) + exit, code := exitOf(ws) + if id == 0 { + l.running, l.code = false, code + } + l.post(message{Kind: msgExited, ID: id, Exit: exit}) } // exitOf describes how a process ended, with the code the launcher exits with for it. @@ -350,7 +377,7 @@ func (l *launcher) mountWorld(staging string) error { return &Error{Kind: ErrNetwork, Op: "open", Path: "/proc/self/ns/net", Err: err} } defer unix.Close(netns) - if err := l.ctl.send(message{Kind: msgMounted}, dev, netns); err != nil { + if err := l.ctl.send(context.Background(), message{Kind: msgMounted}, dev, netns); err != nil { return &Error{Kind: ErrLauncher, Op: "report mount", Err: err} } return nil diff --git a/apps/daemon/internal/sessionview/view_linux.go b/apps/daemon/internal/sessionview/view_linux.go index 474ac87eb..fb696b3fd 100644 --- a/apps/daemon/internal/sessionview/view_linux.go +++ b/apps/daemon/internal/sessionview/view_linux.go @@ -57,6 +57,7 @@ type View struct { requestMu sync.Mutex lastID uint64 requests map[uint64]chan reply // by ID; nil once the launcher stopped answering + spawning chan struct{} // held from a spawn's start until it has settled exited atomic.Bool closing atomic.Bool closeOnce sync.Once @@ -73,7 +74,7 @@ func Start(ctx context.Context, spec Spec) (*View, error) { if err := Probe(); err != nil { return nil, err } - v := &View{done: make(chan struct{}), requests: map[uint64]chan reply{}} + v := &View{done: make(chan struct{}), requests: map[uint64]chan reply{}, spawning: make(chan struct{}, 1)} if err := v.launch(&spec); err != nil { return nil, v.abort(err) } @@ -248,7 +249,7 @@ func (v *View) handshake(ctx context.Context, spec *Spec) error { return &Error{Kind: ErrNetwork, Op: "setup", Err: err} } } - if err := v.ctl.send(message{Kind: msgProceed, Targets: targets}); err != nil { + if err := v.ctl.send(context.Background(), message{Kind: msgProceed, Targets: targets}); err != nil { return v.lost("proceed", err) } m, files, err = v.ctl.recv() @@ -482,20 +483,20 @@ func (v *View) Signal(sig syscall.Signal) error { if v.exited.Load() { return errSignalExited } - r, err := v.request(context.Background(), message{Kind: msgSignal, Signal: sig}) + r, err := v.ask(message{Kind: msgSignal, Signal: sig}) if (err == ErrClosed && v.exited.Load()) || (err == nil && !r.Delivered) { return errSignalExited } return err } -// request sends m with the descriptors of files under a new ID and returns the launcher's reply to it, or ErrClosed once the launcher stopped answering. When ctx ends first, it returns ctx's error and closes the process a late spawn reply brings. -func (v *View) request(ctx context.Context, m message, files ...*os.File) (reply, error) { +// request sends m with the descriptors of files under a new ID and returns the channel its reply comes on, which closes without one once the launcher stopped answering. ctx bounds the send: when it ends first, request returns its error and m is not sent. It returns ErrClosed once the view has ended. +func (v *View) request(ctx context.Context, m message, files ...*os.File) (chan reply, error) { replies := make(chan reply, 1) v.requestMu.Lock() if v.requests == nil { v.requestMu.Unlock() - return reply{}, ErrClosed + return nil, ErrClosed } v.lastID++ m.ID = v.lastID @@ -505,52 +506,120 @@ func (v *View) request(ctx context.Context, m message, files ...*os.File) (reply for i, f := range files { fds[i] = int(f.Fd()) } - if err := v.ctl.send(m, fds...); err != nil { + if err := v.ctl.send(ctx, m, fds...); err != nil { v.requestMu.Lock() delete(v.requests, m.ID) v.requestMu.Unlock() - return reply{}, &Error{Kind: ErrLauncher, Op: "request", Err: err} - } - select { - case r, ok := <-replies: - if !ok { - return reply{}, ErrClosed + switch { + case err == ctx.Err(): + return nil, err + case errors.Is(err, syscall.EPIPE): + // The socket is shut down: the view has ended. + return nil, ErrClosed } - return r, nil - case <-ctx.Done(): - go func() { - if r := <-replies; r.spawned != nil { - r.spawned.Close() - } - }() - return reply{}, ctx.Err() + return nil, &Error{Kind: ErrLauncher, Op: "request", Err: err} } + return replies, nil } -// Spawn starts path with args, env and dir as another process in the view while the view's process runs, with stdio as its stdin, stdout and stderr; the caller keeps its files. The package documentation says how a spawned process runs and ends. ctx bounds the wait for the start: once it ends, Spawn returns its error, and a process that starts after all is killed. Spawn returns ErrExited once the view's process has exited, ErrClosed once the view has ended, ErrLauncher when the launcher is lost and ErrExec when path did not start. -func (v *View) Spawn(ctx context.Context, path string, args, env []string, dir string, stdio [3]*os.File) (*Spawned, error) { - cmdR, cmdW, err := os.Pipe() +// ask sends m and returns the launcher's reply to it, or ErrClosed once the launcher stopped answering. +func (v *View) ask(m message) (reply, error) { + replies, err := v.request(context.Background(), m) + if err != nil { + return reply{}, err + } + r, ok := <-replies + if !ok { + return reply{}, ErrClosed + } + return r, nil +} + +// Spawn starts path with args, env and dir as another process in the view while the view's process runs. Its stdout and stderr are pipes, and so is its stdin when stdin is set; otherwise its stdin is /dev/null. The package documentation says how a spawned process runs and ends. +// +// A view starts one spawn at a time: Spawn waits for the one before it to settle, holding nothing. ctx bounds that wait and the start; once it ends, Spawn returns its error, and a process that starts after all is killed before the next spawn begins. Spawn returns ErrExited once the view's process has exited, ErrClosed once the view has ended, ErrExec when path did not start, and ErrLauncher when a pipe could not be made or the launcher could not be reached. +func (v *View) Spawn(ctx context.Context, path string, args, env []string, dir string, stdin bool) (*Spawned, error) { + select { + case v.spawning <- struct{}{}: + case <-ctx.Done(): + return nil, ctx.Err() + } + if v.exited.Load() { + <-v.spawning + return nil, &Error{Kind: ErrExited, Op: "spawn"} + } + // files are the launcher's: the command's read end and the process's stdio. ends are the caller's. + files := make([]*os.File, 4) + var cmdW *os.File + var ends [3]*os.File + var err error + files[0], cmdW, err = os.Pipe() + if err == nil && stdin { + files[1], ends[0], err = os.Pipe() + } else if err == nil { + files[1], err = os.Open(os.DevNull) + } + for i := 1; i < 3 && err == nil; i++ { + ends[i], files[i+1], err = os.Pipe() + } if err != nil { + closeFiles(append(files, cmdW)) + closeFiles(ends[:]) + <-v.spawning return nil, &Error{Kind: ErrLauncher, Op: "pipe", Err: err} } - // The command travels over a pipe, as the spec does, so its size is exec's to bound. A write the launcher never reads fails once the launcher's end closes. + // The command travels over a pipe, as the spec does, so its size is exec's to bound. The write ends once the launcher has read it, or once no read end remains or the spawn is cancelled. go func() { _ = gob.NewEncoder(cmdW).Encode(command{Path: path, Args: args, Env: env, Dir: dir}) cmdW.Close() }() - r, err := v.request(ctx, message{Kind: msgSpawn}, cmdR, stdio[0], stdio[1], stdio[2]) - cmdR.Close() - switch { - case err != nil: - return nil, err - case r.spawned == nil: - return nil, r.Fail.err() + replies, err := v.request(ctx, message{Kind: msgSpawn}, files...) + closeFiles(files) + if err == nil { + select { + case r, ok := <-replies: + defer func() { <-v.spawning }() + return started(r, ok, ends) + case <-ctx.Done(): + err = ctx.Err() + cmdW.Close() + go func() { + defer func() { <-v.spawning }() + r, ok := <-replies + if s, err := started(r, ok, ends); err == nil { + s.Close() + <-s.done + closeFiles([]*os.File{s.Stdin, s.Stdout, s.Stderr}) + } + }() + return nil, err + } } - return r.spawned, nil + cmdW.Close() + closeFiles(ends[:]) + <-v.spawning + return nil, err +} + +// started returns the process a spawn's reply reports, with ends as its stdio, or why none started, closing ends. +func started(r reply, ok bool, ends [3]*os.File) (*Spawned, error) { + err := ErrClosed + switch { + case ok && r.spawned != nil: + r.spawned.Stdin, r.spawned.Stdout, r.spawned.Stderr = ends[0], ends[1], ends[2] + return r.spawned, nil + case ok: + err = r.Fail.err() + } + closeFiles(ends[:]) + return nil, err } // Spawned is a process that Spawn started. It is a clirunner.Handle. type Spawned struct { + // Stdin is the write end of the process's stdin pipe, or nil without one; Stdout and Stderr are the read ends of its stdout and stderr pipes. The caller owns them. + Stdin, Stdout, Stderr *os.File + v *View id uint64 done chan struct{} // closed once it has been reaped or the view has ended @@ -560,7 +629,7 @@ type Spawned struct { // Signal delivers sig to the process's group until the process has exited, and then returns ErrExited. func (s *Spawned) Signal(sig syscall.Signal) error { - r, err := s.v.request(context.Background(), message{Kind: msgSignal, Spawn: s.id, Signal: sig}) + r, err := s.v.ask(message{Kind: msgSignal, Spawn: s.id, Signal: sig}) // The view's end has ended the process. if err == ErrClosed || (err == nil && !r.Delivered) { return errSignalExited @@ -588,7 +657,7 @@ func (v *View) Close() error { v.closeOnce.Do(func() { v.closing.Store(true) _ = v.cmd.Process.Kill() - // A spawn blocked before its exec outlives the launcher with a copy of the control socket, until the teardown stops the world. + // A spawn blocked before its exec outlives the launcher with a copy of its end, until the teardown stops the world. v.ctl.interrupt() <-v.done v.closePipes() diff --git a/apps/daemon/internal/sessionview/view_linux_test.go b/apps/daemon/internal/sessionview/view_linux_test.go index c3c5f96f7..abe53231f 100644 --- a/apps/daemon/internal/sessionview/view_linux_test.go +++ b/apps/daemon/internal/sessionview/view_linux_test.go @@ -12,6 +12,7 @@ import ( "fmt" "io" "io/fs" + "maps" "net" "os" "os/exec" @@ -110,7 +111,7 @@ func TestViewSignalAndTeardown(t *testing.T) { if line, err := bufio.NewReader(v.Stdout()).ReadString('\n'); err != nil || line != "ready\n" { t.Fatalf("harness said %q, %v", line, err) } - if n := processesWith(t, token); n != 1 { + if n := len(pidsWith(t, token)); n != 1 { t.Fatalf("%d grandchildren before exit, want 1", n) } if staged, _ := os.ReadDir(f.staging); len(staged) != 1 { @@ -127,7 +128,7 @@ func TestViewSignalAndTeardown(t *testing.T) { if exit, err := v.Wait(); err != nil || exit != (Exit{Code: 7}) { t.Fatalf("Wait = %+v, %v; want exit code 7", exit, err) } - if n := processesWith(t, token); n != 0 { + if n := len(pidsWith(t, token)); n != 0 { t.Errorf("%d grandchildren survived the Harness", n) } select { @@ -155,22 +156,12 @@ func TestViewDescendantsKeepTheGrace(t *testing.T) { t.Fatalf("Start: %v", err) } defer v.Close() - null, err := os.Open(os.DevNull) - if err != nil { - t.Fatal(err) - } - defer null.Close() - r, wr, err := os.Pipe() - if err != nil { - t.Fatal(err) - } - defer r.Close() - spawned, err := v.Spawn(context.Background(), "/.oac/harness/harness", []string{"harness"}, []string{helperEnv + "=cleanup", "OAC_VIEW_TOKEN=-spawned"}, "/data", [3]*os.File{null, wr, os.Stderr}) - wr.Close() + spawned, err := v.Spawn(context.Background(), "/.oac/harness/harness", []string{"harness"}, []string{helperEnv + "=cleanup", "OAC_VIEW_TOKEN=-spawned"}, "/data", false) if err != nil { t.Fatalf("Spawn: %v", err) } - for _, stdout := range []io.Reader{v.Stdout(), r} { + defer closeStdio(spawned) + for _, stdout := range []io.Reader{v.Stdout(), spawned.Stdout} { if line, err := bufio.NewReader(stdout).ReadString('\n'); err != nil || line != "ready\n" { t.Fatalf("helper said %q, %v", line, err) } @@ -203,76 +194,86 @@ func TestViewDescendantsKeepTheGrace(t *testing.T) { if code, err := spawned.Wait(); err != nil || code != 7 { t.Errorf("spawned Wait = %d, %v; want exit code 7", code, err) } + // Once the spawned process has exited, its handle no longer reaches the group it led. + if err := spawned.Signal(syscall.SIGTERM); !errors.Is(err, ErrExited) { + t.Errorf("Signal after the spawned process exited = %v, want ErrExited", err) + } } -// TestSpawnRunsInTheView checks that a spawned process runs as the process's user, unprivileged and in the view's cgroup, that its output and exit come back, that its handle reaches nothing once it has exited, that a command of any size either starts or fails alone, and that the view's end ends it. +// TestSpawnRunsInTheView checks that a spawned process runs as the process does, with its stdio and exit coming back, that its handle reaches nothing once it has exited, that a spawn whose pipes or command fail fails alone, and that the view's end ends it. func TestSpawnRunsInTheView(t *testing.T) { requireView(t) f := newFixture(t) w := &loopbackWorld{dir: f.world} - v, err := Start(context.Background(), f.spec(w, "wait", "OAC_VIEW_TOKEN=oac-unused")) + v, err := Start(context.Background(), f.spec(w, "identity")) if err != nil { t.Fatalf("Start: %v", err) } defer v.Close() - if line, err := bufio.NewReader(v.Stdout()).ReadString('\n'); err != nil || line != "ready\n" { - t.Fatalf("harness said %q, %v", line, err) - } - null, err := os.Open(os.DevNull) + token := fmt.Sprintf("oac-spawned-%d", time.Now().UnixNano()) + sleeper, err := spawnHelper(context.Background(), v, "identity", "/data", token) if err != nil { - t.Fatal(err) + t.Fatalf("Spawn: %v", err) } - defer null.Close() - token := fmt.Sprintf("oac-spawned-%d", time.Now().UnixNano()) - spawn := func(mode string, stdout *os.File) *Spawned { - t.Helper() - s, err := v.Spawn(context.Background(), "/.oac/harness/harness", []string{"harness", token}, []string{helperEnv + "=" + mode}, "/data", [3]*os.File{null, stdout, stdout}) - if err != nil { - t.Fatalf("Spawn %s: %v", mode, err) + defer closeStdio(sleeper) + var ids [2]map[string]string + for i, stdout := range []io.Reader{v.Stdout(), sleeper.Stdout} { + if err := json.NewDecoder(stdout).Decode(&ids[i]); err != nil { + t.Fatal(err) } - return s } - r, wr, err := os.Pipe() - if err != nil { - t.Fatal(err) + if len(ids[0]) != 20 || !maps.Equal(ids[0], ids[1]) { + t.Errorf("spawned process runs as and in %v, the process as and in %v", ids[1], ids[0]) } - report := spawn("report", wr) - wr.Close() - out, err := io.ReadAll(r) - r.Close() + report, err := v.Spawn(context.Background(), "/.oac/harness/harness", []string{"harness"}, []string{helperEnv + "=report"}, "/data", true) if err != nil { - t.Fatal(err) + t.Fatalf("Spawn: %v", err) + } + report.Stdin.Write([]byte("ping")) + report.Stdin.Close() + out, _ := io.ReadAll(report.Stdout) + errOut, _ := io.ReadAll(report.Stderr) + closeStdio(report) + if string(out) != "ping /data hello from the world " || string(errOut) != "to stderr" { + t.Errorf("spawned process wrote %q and %q", out, errOut) } if code, err := report.Wait(); err != nil || code != 3 { t.Fatalf("spawned Wait = %d, %v; want exit code 3", code, err) } - cgroups := sessionviewtest.Cgroups(t, f.cgroups) - want := fmt.Sprintf("%d %d 0::", viewID, viewID) - if len(cgroups) != 1 || !strings.HasPrefix(string(out), want) || !strings.HasSuffix(string(out), "/"+filepath.Base(cgroups[0])+"\n") { - t.Fatalf("spawned process reported %q in cgroups %v, want %q and the view's cgroup", out, cgroups, want) - } - sleeper := spawn("sleep", null) - if n := processesWith(t, token); n != 1 { - t.Fatalf("%d spawned processes running, want 1", n) - } if err := report.Signal(syscall.SIGKILL); !errors.Is(err, ErrExited) { t.Errorf("Signal after the spawned process exited = %v, want ErrExited", err) } if err := sleeper.Signal(0); err != nil { t.Errorf("Signal to the running spawned process = %v", err) } + // A spawn that cannot make its pipes fails with the reason. + var limit unix.Rlimit + if err := unix.Prlimit(0, unix.RLIMIT_NOFILE, nil, &limit); err != nil { + t.Fatal(err) + } + if err := unix.Prlimit(0, unix.RLIMIT_NOFILE, &unix.Rlimit{Max: limit.Max}, nil); err != nil { + t.Fatal(err) + } + _, err = spawnHelper(context.Background(), v, "noop", "/data") + if err := unix.Prlimit(0, unix.RLIMIT_NOFILE, &limit, nil); err != nil { + t.Fatal(err) + } + if !errors.Is(err, ErrLauncher) || !errors.Is(err, syscall.EMFILE) { + t.Errorf("Spawn without descriptors = %v, want ErrLauncher with EMFILE", err) + } // One argument above the control socket's packet size starts; one above exec's limit fails alone. for size, want := range map[int]error{100 << 10: nil, 200 << 10: ErrExec} { - s, err := v.Spawn(context.Background(), "/.oac/harness/harness", []string{"harness", strings.Repeat("x", size)}, []string{helperEnv + "=noop"}, "/data", [3]*os.File{null, null, null}) + s, err := spawnHelper(context.Background(), v, "noop", "/data", strings.Repeat("x", size)) if err == nil { + closeStdio(s) _, err = s.Wait() } if !errors.Is(err, want) { t.Errorf("Spawn with a %d byte argument = %v, want %v", size, err, want) } } - if err := v.Signal(0); err != nil || processesWith(t, token) != 1 { - t.Fatalf("after the spawns, the view's Signal = %v and the sleeper runs %d times; want both running", err, processesWith(t, token)) + if err := v.Signal(0); err != nil || len(pidsWith(t, token)) != 1 { + t.Fatalf("after the spawns, the view's Signal = %v and the sleeper runs %d times; want both running", err, len(pidsWith(t, token))) } if err := v.Close(); err != nil { t.Fatalf("Close: %v", err) @@ -280,57 +281,174 @@ func TestSpawnRunsInTheView(t *testing.T) { if _, err := sleeper.Wait(); !errors.Is(err, ErrClosed) { t.Errorf("spawned Wait after the view ended = %v, want ErrClosed", err) } - if n := processesWith(t, token); n != 0 { + if n := len(pidsWith(t, token)); n != 0 { t.Errorf("%d spawned processes survived the view", n) } } -// TestBlockedSpawnBlocksNothingElse checks that a spawn blocked on the world leaves the view's signals, its own context, the process's exit and the teardown free. -func TestBlockedSpawnBlocksNothingElse(t *testing.T) { - requireView(t) - f := newFixture(t) - mkdir(t, filepath.Join(f.world, "data", "stall")) - w := &stallWorld{loopbackWorld: loopbackWorld{dir: f.world}, name: "stall", stalled: make(chan struct{}), release: make(chan struct{})} - spec := f.spec(&w.loopbackWorld, "wait", "OAC_VIEW_TOKEN=oac-unused") - spec.World = w.serve - v, err := Start(context.Background(), spec) - if err != nil { - t.Fatalf("Start: %v", err) - } - defer v.Close() - if line, err := bufio.NewReader(v.Stdout()).ReadString('\n'); err != nil || line != "ready\n" { - t.Fatalf("harness said %q, %v", line, err) - } - null, err := os.Open(os.DevNull) +// TestStalledSpawnBlocksNothingElse checks that while a spawn's child is stuck on the world before its exec, the spawns behind it wait holding no descriptors and return once their contexts end, and that the view's signals, the exits of its other processes, the stuck caller's context, the process's exit and the teardown all go on. +func TestStalledSpawnBlocksNothingElse(t *testing.T) { + v, w := startStalled(t) + sleeper, err := spawnHelper(context.Background(), v, "sleep", "/data") if err != nil { - t.Fatal(err) + t.Fatalf("Spawn: %v", err) } - defer null.Close() + defer closeStdio(sleeper) + launcherFDs := fdCount(t, v.cmd.Process.Pid) ctx, cancel := context.WithCancel(context.Background()) - spawned := make(chan error, 1) + stuck := spawnAsync(ctx, v, "sleep", "/data/stall") + await(t, w.stalled, "the spawn's lookup in the world") + fds := fdCount(t, os.Getpid()) + waitCtx, stopWaiting := context.WithCancel(context.Background()) + var waiting []<-chan spawnResult + for range 50 { + waiting = append(waiting, spawnAsync(waitCtx, v, "noop", "/data")) + } + eventually(t, "50 spawns waiting", func() bool { return inSpawn() == 51 }) + if n := fdCount(t, os.Getpid()); n > fds { + t.Errorf("%d descriptors with 50 spawns waiting, %d before", n, fds) + } + // The launcher holds the stuck spawn's descriptors and its fork's pipe, nothing for the spawns waiting. + if n := fdCount(t, v.cmd.Process.Pid); n > launcherFDs+6 { + t.Errorf("launcher holds %d descriptors with a spawn stuck and 50 waiting, %d before", n, launcherFDs) + } + stopWaiting() + for _, r := range waiting { + if r := await(t, r, "a waiting spawn's return"); !errors.Is(r.err, context.Canceled) { + t.Errorf("waiting Spawn = %v, want context.Canceled", r.err) + } + } + // Enough requests that the launcher's heap would call for a collection. + signaled := make(chan error, 1) go func() { - _, err := v.Spawn(ctx, "/.oac/harness/harness", []string{"harness"}, []string{helperEnv + "=noop"}, "/data/stall", [3]*os.File{null, null, null}) - spawned <- err + for range 1000 { + if err := v.Signal(0); err != nil { + signaled <- err + return + } + } + signaled <- nil }() - select { - case <-w.stalled: - case <-time.After(10 * time.Second): - t.Fatal("the spawn never looked up its directory in the world") + if err := await(t, signaled, "1000 signals"); err != nil { + t.Errorf("Signal while a spawn is stuck = %v", err) } - if err := v.Signal(syscall.SIGTERM); err != nil { - t.Fatalf("Signal while a spawn blocks: %v", err) + if err := sleeper.Signal(syscall.SIGKILL); err != nil { + t.Fatalf("Signal to the sleeper = %v", err) + } + if code := await(t, waitFor(sleeper), "the sleeper's exit"); code != -1 { + t.Errorf("sleeper Wait = %d, want -1", code) } cancel() - select { - case err := <-spawned: - if !errors.Is(err, context.Canceled) { - t.Errorf("Spawn = %v, want context.Canceled", err) + if r := await(t, stuck, "the stuck spawn's return"); !errors.Is(r.err, context.Canceled) { + t.Errorf("stuck Spawn = %v, want context.Canceled", r.err) + } + if err := v.Signal(syscall.SIGTERM); err != nil { + t.Fatalf("Signal: %v", err) + } + exited := make(chan error, 1) + go func() { + exit, err := v.Wait() + if exit != (Exit{Code: 7}) { + err = errors.Join(err, fmt.Errorf("exit %+v", exit)) } - case <-time.After(10 * time.Second): - t.Fatal("Spawn did not return after its context ended") + exited <- err + }() + if err := await(t, exited, "the view's end"); err != nil { + t.Errorf("Wait with a spawn still starting: %v, want exit code 7", err) } - if exit, err := v.Wait(); err != nil || exit != (Exit{Code: 7}) { - t.Fatalf("Wait = %+v, %v; want exit code 7", exit, err) +} + +// TestLateSpawnIsEnded checks that a process whose spawn was cancelled before it started is killed once it starts, before the next spawn begins. +func TestLateSpawnIsEnded(t *testing.T) { + v, w := startStalled(t) + token := fmt.Sprintf("oac-late-%d", time.Now().UnixNano()) + ctx, cancel := context.WithCancel(context.Background()) + late := spawnAsync(ctx, v, "sleep", "/data/stall", token) + await(t, w.stalled, "the spawn's lookup in the world") + cancel() + if r := await(t, late, "the cancelled spawn's return"); !errors.Is(r.err, context.Canceled) { + t.Fatalf("Spawn = %v, want context.Canceled", r.err) + } + next := spawnAsync(context.Background(), v, "noop", "/data") + eventually(t, "the next spawn waiting", func() bool { return inSpawn() == 1 }) + w.unstall() + r := await(t, next, "the next spawn") + if r.err != nil { + t.Fatalf("next Spawn: %v", r.err) + } + closeStdio(r.s) + if n := len(pidsWith(t, token)); n != 0 { + t.Errorf("the late process runs %d times once the next spawn started", n) + } + if code, err := r.s.Wait(); err != nil || code != 0 { + t.Errorf("next Wait = %d, %v", code, err) + } +} + +// TestSpawnExitsBeforeItsRegistration checks that a spawned child that dies before its exec, while orphans exit around it, reports its own exit to its own handle and to no other, that a handle whose process has exited reaches nothing, and that Close ends what remains. +func TestSpawnExitsBeforeItsRegistration(t *testing.T) { + v, w := startStalled(t) + launcher := v.cmd.Process.Pid + token := fmt.Sprintf("oac-bystander-%d", time.Now().UnixNano()) + bystander, err := spawnHelper(context.Background(), v, "sleep", "/data", token) + if err != nil { + t.Fatalf("Spawn: %v", err) + } + defer closeStdio(bystander) + parent, err := v.Spawn(context.Background(), "/.oac/harness/harness", []string{"harness"}, []string{helperEnv + "=wait", "OAC_VIEW_TOKEN=oac-orphan"}, "/data", false) + if err != nil { + t.Fatalf("Spawn: %v", err) + } + defer closeStdio(parent) + if line, err := bufio.NewReader(parent.Stdout).ReadString('\n'); err != nil || line != "ready\n" { + t.Fatalf("parent said %q, %v", line, err) + } + pending := spawnAsync(context.Background(), v, "noop", "/data/stall") + await(t, w.stalled, "the spawn's lookup in the world") + // The parent's group ends while the child forks; its orphan stays unreaped until the child is registered. + if err := parent.Signal(syscall.SIGKILL); err != nil { + t.Fatalf("Signal to the parent = %v", err) + } + if code := await(t, waitFor(parent), "the parent's exit"); code != -1 { + t.Errorf("parent Wait = %d, want -1", code) + } + if err := parent.Signal(syscall.SIGKILL); !errors.Is(err, ErrExited) { + t.Errorf("Signal after the parent exited = %v, want ErrExited", err) + } + eventually(t, "the orphan's exit", func() bool { return zombies(t, launcher) > 0 }) + // The child dies before its exec, once the world answers. + child := pidsWith(t, launcherArg0) + child = slices.DeleteFunc(child, func(pid int) bool { return pid == launcher }) + if len(child) != 1 { + t.Fatalf("children before their exec: %v, want 1", child) + } + if err := unix.Kill(child[0], unix.SIGKILL); err != nil { + t.Fatal(err) + } + w.unstall() + r := await(t, pending, "the spawn") + if r.err != nil { + t.Fatalf("Spawn = %v, want the child that died before its exec", r.err) + } + defer closeStdio(r.s) + if code := await(t, waitFor(r.s), "the child's exit"); code != -1 { + t.Errorf("child Wait = %d, want -1", code) + } + if err := r.s.Signal(0); !errors.Is(err, ErrExited) { + t.Errorf("Signal after the child exited = %v, want ErrExited", err) + } + if err := bystander.Signal(0); err != nil { + t.Errorf("Signal to the bystander = %v", err) + } + eventually(t, "the orphans reaped", func() bool { return zombies(t, launcher) == 0 }) + if err := v.Close(); err != nil { + t.Fatalf("Close: %v", err) + } + if _, err := bystander.Wait(); !errors.Is(err, ErrClosed) { + t.Errorf("bystander Wait after Close = %v, want ErrClosed", err) + } + if n := len(pidsWith(t, token)); n != 0 { + t.Errorf("%d bystanders survived the view", n) } } @@ -760,14 +878,16 @@ func (n *hangNode) Write(context.Context, gofs.FileHandle, []byte, int64) (uint3 return 0, syscall.EIO } -// stallWorld is a loopbackWorld that answers no lookup of name until Stop. +// stallWorld is a loopbackWorld that answers no lookup of name until unstall or Stop. type stallWorld struct { loopbackWorld - name string - stalled, release chan struct{} // stalled closes at the first lookup of name - stallOnce sync.Once + name string + stalled, release chan struct{} // stalled closes at the first lookup of name + stallOnce, releaseOnce sync.Once } +func (w *stallWorld) unstall() { w.releaseOnce.Do(func() { close(w.release) }) } + func (w *stallWorld) serve(_ context.Context, dev *os.File, mount WorldMount) (WorldServer, Presentation, error) { root, err := gofs.NewLoopbackRoot(w.dir) if err != nil { @@ -788,7 +908,7 @@ func (w *stallWorld) serve(_ context.Context, dev *os.File, mount WorldMount) (W } func (w *stallWorld) Stop() error { - close(w.release) + w.unstall() return w.loopbackWorld.Stop() } @@ -829,22 +949,165 @@ func serveBroker(t *testing.T) func(*os.File) error { } } -// processesWith counts processes whose command line contains token. -func processesWith(t *testing.T, token string) int { +// pidsWith lists the processes whose command line contains token. +func pidsWith(t *testing.T, token string) []int { t.Helper() cmdlines, err := filepath.Glob("/proc/[0-9]*/cmdline") if err != nil { t.Fatal(err) } - n := 0 + var pids []int for _, p := range cmdlines { if b, err := os.ReadFile(p); err == nil && bytes.Contains(b, []byte(token)) { + pid, _ := strconv.Atoi(filepath.Base(filepath.Dir(p))) + pids = append(pids, pid) + } + } + return pids +} + +// zombies counts the children of pid that have exited and are not reaped yet. +func zombies(t *testing.T, pid int) int { + t.Helper() + stats, err := filepath.Glob("/proc/[0-9]*/stat") + if err != nil { + t.Fatal(err) + } + n := 0 + for _, p := range stats { + b, err := os.ReadFile(p) + // The state and the parent's pid follow the command name, which may hold anything. + if f := strings.Fields(string(b[bytes.LastIndexByte(b, ')')+1:])); err == nil && len(f) > 1 && f[0] == "Z" && f[1] == strconv.Itoa(pid) { n++ } } return n } +func fdCount(t *testing.T, pid int) int { + t.Helper() + fds, err := os.ReadDir(fmt.Sprintf("/proc/%d/fd", pid)) + if err != nil { + t.Fatal(err) + } + return len(fds) +} + +// inSpawn counts the goroutines in View.Spawn. +func inSpawn() int { + buf := make([]byte, 1<<20) + return strings.Count(string(buf[:runtime.Stack(buf, true)]), "sessionview.(*View).Spawn(") +} + +// startStalled starts a view whose process waits for TERM and whose world answers no lookup of /data/stall until w.unstall. +func startStalled(t *testing.T) (*View, *stallWorld) { + t.Helper() + requireView(t) + f := newFixture(t) + mkdir(t, filepath.Join(f.world, "data", "stall")) + w := &stallWorld{loopbackWorld: loopbackWorld{dir: f.world}, name: "stall", stalled: make(chan struct{}), release: make(chan struct{})} + spec := f.spec(&w.loopbackWorld, "wait", "OAC_VIEW_TOKEN=oac-unused") + spec.World = w.serve + v, err := Start(context.Background(), spec) + if err != nil { + t.Fatalf("Start: %v", err) + } + t.Cleanup(func() { v.Close() }) + if line, err := bufio.NewReader(v.Stdout()).ReadString('\n'); err != nil || line != "ready\n" { + t.Fatalf("harness said %q, %v", line, err) + } + return v, w +} + +// spawnHelper spawns this binary as helper mode in dir, with args after its name and no stdin. +func spawnHelper(ctx context.Context, v *View, mode, dir string, args ...string) (*Spawned, error) { + return v.Spawn(ctx, "/.oac/harness/harness", append([]string{"harness"}, args...), []string{helperEnv + "=" + mode}, dir, false) +} + +type spawnResult struct { + s *Spawned + err error +} + +func spawnAsync(ctx context.Context, v *View, mode, dir string, args ...string) <-chan spawnResult { + ch := make(chan spawnResult, 1) + go func() { + s, err := spawnHelper(ctx, v, mode, dir, args...) + ch <- spawnResult{s, err} + }() + return ch +} + +// waitFor delivers s's exit code, or -2 when Wait fails. +func waitFor(s *Spawned) <-chan int { + ch := make(chan int, 1) + go func() { + code, err := s.Wait() + if err != nil { + code = -2 + } + ch <- code + }() + return ch +} + +func closeStdio(s *Spawned) { closeFiles([]*os.File{s.Stdin, s.Stdout, s.Stderr}) } + +// await returns what ch delivers, failing the test when nothing comes within 10 seconds. +func await[T any](t *testing.T, ch <-chan T, what string) T { + t.Helper() + select { + case v := <-ch: + return v + case <-time.After(10 * time.Second): + t.Fatalf("%s: nothing within 10s", what) + } + var zero T + return zero +} + +// eventually polls cond, failing the test when it does not hold within 10 seconds. +func eventually(t *testing.T, what string, cond func() bool) { + t.Helper() + for deadline := time.Now().Add(10 * time.Second); !cond(); time.Sleep(5 * time.Millisecond) { + if time.Now().After(deadline) { + t.Fatalf("%s not within 10s", what) + } + } +} + +// identity describes what this process runs as and in: its credentials and restrictions, its namespaces, its cgroup and its root. +func identity() (map[string]string, error) { + status, err := os.ReadFile("/proc/self/status") + if err != nil { + return nil, err + } + id := map[string]string{} + for _, line := range strings.Split(string(status), "\n") { + k, v, _ := strings.Cut(line, ":") + switch k { + case "Uid", "Gid", "Groups", "CapInh", "CapPrm", "CapEff", "CapBnd", "CapAmb", "NoNewPrivs", "Seccomp", "Seccomp_filters": + id[k] = strings.TrimSpace(v) + } + } + for _, ns := range []string{"mnt", "net", "pid", "ipc", "uts", "user", "cgroup"} { + if id["ns "+ns], err = os.Readlink("/proc/self/ns/" + ns); err != nil { + return nil, err + } + } + cgroup, err := os.ReadFile("/proc/self/cgroup") + if err != nil { + return nil, err + } + id["cgroup"] = string(cgroup) + var st unix.Stat_t + if err := unix.Stat("/", &st); err != nil { + return nil, err + } + id["root"] = fmt.Sprintf("%d:%d", st.Dev, st.Ino) + return id, nil +} + func runHelper(mode string) int { switch mode { case "probe": @@ -900,9 +1163,23 @@ func runHelper(mode string) int { return 0 case "noop": return 0 + case "identity": + sigs := make(chan os.Signal, 1) + signal.Notify(sigs, syscall.SIGTERM) + id, err := identity() + if err != nil { + fmt.Fprintln(os.Stderr, err) + return 1 + } + json.NewEncoder(os.Stdout).Encode(id) + <-sigs + return 7 case "report": - cgroup, err := os.ReadFile("/proc/self/cgroup") - fmt.Printf("%d %d %v %s", os.Getuid(), os.Getgid(), errors.Join(err, noPrivileges()), cgroup) + in, err := io.ReadAll(os.Stdin) + wd, werr := os.Getwd() + data, rerr := os.ReadFile("in.txt") + fmt.Printf("%s %s %s %v", in, wd, data, errors.Join(err, werr, rerr)) + fmt.Fprint(os.Stderr, "to stderr") return 3 case "hang": f, err := os.OpenFile("/data/hang", os.O_WRONLY, 0) diff --git a/contracts/agents-api/harness-onboarding.md b/contracts/agents-api/harness-onboarding.md index 940932126..8a510833e 100644 --- a/contracts/agents-api/harness-onboarding.md +++ b/contracts/agents-api/harness-onboarding.md @@ -327,10 +327,10 @@ With `ViewProxyNone`, `ViewSession.Proxy` is empty and the view has no generic p `ViewSession.Spawn` runs a `LocalExec` binary as another process in the live view while the Harness runs, such as a reader of the Harness's native history. Harness-side code that reads Harness-written data runs here, never on the agent host outside the view and never in a view of its own. `Spawn` takes `StartOptions` as `Launch` does and returns the same `clirunner.Process`. The process runs as the Harness does: as the same user, in the same namespaces, view cgroup, world and network, with no capabilities, `no_new_privs` and the same seccomp filter, in a process group of its own. -- `Parent` bounds the wait for the start. Once it ends, `Spawn` returns its error and kills a process that starts after all. +- The view starts one `Spawn` at a time. `Parent` bounds the wait for its turn and for the start. Once it ends, `Spawn` returns its error and kills a process that starts after all. - Cancel sends TERM to its process group and kills the group after `KillTimeout`. Once the process has exited, Cancel delivers nothing, and what it left runs on as the view's other processes do. - The view's end ends them all. A Cancel of the Harness reaches them, and when the Harness exits they are among the processes that remain. A view that ends after `Spawn` returned shows in the process's `Wait`. -- `Spawn` returns `ErrNotLocalExec` when `Binary` is not a `LocalExec` path, and `ErrNoLiveView` when no view runs its Harness: none was launched yet, or its Harness has exited or its view has ended. +- `Spawn` returns `ErrNotLocalExec` when `Binary` is not a `LocalExec` path, and `ErrNoLiveView` when no view runs its Harness: none was launched yet, or its Harness has exited or its view has ended. Any other failure, such as a binary that does not start or no descriptors left, keeps its own error. ### Qualify the view diff --git a/contracts/agents-api/zh/harness-onboarding.md b/contracts/agents-api/zh/harness-onboarding.md index bfbe34bde..66fa44adc 100644 --- a/contracts/agents-api/zh/harness-onboarding.md +++ b/contracts/agents-api/zh/harness-onboarding.md @@ -1,7 +1,7 @@ --- title: "将原生 Harness 添加到 OpenAgentCore" source: contracts/agents-api/harness-onboarding.md -source_hash: db98d69501c492ec945f9e4088cccb202ee9a5f0b99c041c33e1fd303cb0623c +source_hash: aa5ee32e9ae72b923addad7b049d586a32ec18319e6454c52cc21a96e0f69614 --- **Harness** 是一种运行模型和工具循环的原生代理引擎(Codex、Claude Code、MiniMax Code)。**Harness 适配器**将 Runtime 的 Executor 和 Turn 契约转换到该引擎的 SDK 或协议。本文档定义 Runtime–Harness 协议:适配器接口及其生命周期义务、注册、Core 资格认定和验收。[Harness capabilities](harness-capabilities.md) 记录了当前每个 Harness 支持的功能。 @@ -329,10 +329,10 @@ agent host 根据声明推导进程 broker 的映射表:`/.oac/bin/` 在 `ViewSession.Spawn` 在 Harness 运行期间,把一个 `LocalExec` 二进制作为另一个进程运行在活动视图中,例如读取 Harness 原生历史的程序。读取 Harness 所写数据的 Harness 侧代码在这里运行,从不在视图之外的 agent host 上运行,也从不获得自己的视图。`Spawn` 像 `Launch` 一样接收 `StartOptions`,并返回同样的 `clirunner.Process`。该进程的运行方式与 Harness 相同:同一用户,同一组命名空间、视图 cgroup、world 和网络,没有 capability,设置 `no_new_privs` 并使用同一 seccomp 过滤器,且位于自己的进程组中。 -- `Parent` 限定等待启动的时间。它结束后,`Spawn` 返回它的错误,并杀死此后仍然启动的进程。 +- 视图一次只启动一个 `Spawn`。`Parent` 限定等待轮次和等待启动的时间。它结束后,`Spawn` 返回它的错误,并杀死此后仍然启动的进程。 - Cancel 向它的进程组发送 TERM,并在 `KillTimeout` 后杀死该进程组。进程退出后,Cancel 不再投递任何信号,它遗留的进程像视图中的其他进程一样继续运行。 - 视图结束时它们全部随之结束。Harness 的 Cancel 会到达它们;Harness 退出时,它们属于仍然存在的进程。`Spawn` 返回之后视图才结束的情况,体现在该进程的 `Wait` 中。 -- `Binary` 不是 `LocalExec` 路径时,`Spawn` 返回 `ErrNotLocalExec`;没有视图在运行其 Harness 时返回 `ErrNoLiveView`:尚未启动视图,或其 Harness 已退出,或其视图已结束。 +- `Binary` 不是 `LocalExec` 路径时,`Spawn` 返回 `ErrNotLocalExec`;没有视图在运行其 Harness 时返回 `ErrNoLiveView`:尚未启动视图,或其 Harness 已退出,或其视图已结束。其他失败(例如二进制无法启动或描述符耗尽)保留各自的错误。 ### 认定视图资格 {#qualify-the-view} From eda83278a2a22f76a72c00ff8872b800271166e4 Mon Sep 17 00:00:00 2001 From: SaladDay <1203511142@qq.com> Date: Tue, 6 Oct 2026 21:11:42 +0000 Subject: [PATCH 4/5] Enter a spawn's directory on the launcher's thread The launcher's forking thread enters the command's directory itself, as the process's user and with signals blocked, before it forks, so the child inherits it and does nothing before its exec that waits on the world. A slow directory then blocks that thread in an ordinary syscall that holds no P, and garbage collection no longer needs to be paused around the fork. --- apps/daemon/internal/sessionview/doc.go | 2 +- .../internal/sessionview/launcher_linux.go | 65 ++++++++++++--- .../daemon/internal/sessionview/view_linux.go | 2 +- .../internal/sessionview/view_linux_test.go | 81 ++++++++++++------- 4 files changed, 108 insertions(+), 42 deletions(-) diff --git a/apps/daemon/internal/sessionview/doc.go b/apps/daemon/internal/sessionview/doc.go index 1e682da0a..8e1c63099 100644 --- a/apps/daemon/internal/sessionview/doc.go +++ b/apps/daemon/internal/sessionview/doc.go @@ -6,7 +6,7 @@ // // The daemon calls [Init] first thing in main. [Start] re-executes the daemon binary as the launcher, which becomes PID 1 of the view: it builds the view, starts the process, delivers signals to every process in the view, reaps orphans and exits with the process status. Once the process has exited, Signal reports [ErrExited] and delivers nothing. When the process exits while others remain, the launcher sends them TERM unless one already went to the view, and waits for them up to [Process].Grace from the first TERM. Its exit kills what remains and tears the view down. // -// While the process runs, [View.Spawn] has the launcher start another process in the view, passing the command on a pipe and the stdio pipes that Spawn makes over the control socket. A view starts one spawn at a time, and a spawn waiting for its turn holds no descriptors. The launcher starts each from the thread that carries the restrictions, as it started the process, so it runs as the same user, in the view's namespaces and cgroup, under the same restrictions and in a session of its own. That thread does nothing else once the process runs, and no garbage collection runs while it forks, so a fork stuck on the world holds up no other thread: reaping, signals, the drain and the exit go on. The launcher reaps as SIGCHLD reports exits. While a spawn forks, it reaps only the processes it has registered, so the pid of a child that exits before its registration stays its own until the registration ties it to its spawn. A spawned process counts like any other process in the view: a signal to the view reaches it, and when the process exits it is among those the launcher sends TERM and waits for. Each spawn has an ID, which [Spawned] uses rather than the pid: [Spawned].Signal reaches the process group until the launcher reaps the process, and reports [ErrExited] from then on. What the process leaves runs on as other processes in the view do, and the view's end ends it. +// While the process runs, [View.Spawn] has the launcher start another process in the view, passing the command on a pipe and the stdio pipes that Spawn makes over the control socket. A view starts one spawn at a time, and a spawn waiting for its turn holds no descriptors. The launcher starts each from the thread that carries the restrictions, as it started the process, so it runs as the same user, in the view's namespaces and cgroup, under the same restrictions and in a session of its own. That thread does nothing else once the process runs. It enters each command's directory itself, with the process's file system identity, before it forks, so the child inherits the directory and a directory stuck on the world holds up no other thread: reaping, signals, the drain and the exit go on. A command's path does not belong in the world: a child that waits on the world before its exec holds the fork, and with it the launcher, until the world answers. The launcher reaps as SIGCHLD reports exits. While a spawn forks, it reaps only the processes it has registered, so the pid of a child that exits before its registration stays its own until the registration ties it to its spawn. A spawned process counts like any other process in the view: a signal to the view reaches it, and when the process exits it is among those the launcher sends TERM and waits for. Each spawn has an ID, which [Spawned] uses rather than the pid: [Spawned].Signal reaches the process group until the launcher reaps the process, and reports [ErrExited] from then on. What the process leaves runs on as other processes in the view do, and the view's end ends it. // // A view that declares a [Shim] also runs the Session's process relay, the shim binary in relay mode (package processshim). The launcher starts it before the process, under the same restrictions and as the same user, with a listening socket at processshim.SocketPath on a read-only mount and its end of a socket pair whose other end is [View.Relay]. The relay is the only process that receives the descriptors a shim hands over; the broker outside the view holds only its end of the pair. The launcher's drain ignores the relay, which ends with the view. // diff --git a/apps/daemon/internal/sessionview/launcher_linux.go b/apps/daemon/internal/sessionview/launcher_linux.go index 86e4f7902..17f57dbad 100644 --- a/apps/daemon/internal/sessionview/launcher_linux.go +++ b/apps/daemon/internal/sessionview/launcher_linux.go @@ -9,7 +9,6 @@ import ( "os" "os/signal" "runtime" - "runtime/debug" "strconv" "strings" "sync" @@ -28,8 +27,6 @@ func Init() { } // Capability sets, no_new_privs and seccomp filters are per thread: the thread that sets them must be the one that forks the process. runtime.LockOSThread() - // A fork that blocks in the world holds this thread and its P, so another P runs everything else. A fixed count also keeps the runtime from stopping the world to change it. - runtime.GOMAXPROCS(max(2, runtime.GOMAXPROCS(0))) os.Exit(runLauncher()) } @@ -97,6 +94,9 @@ func (l *launcher) run() error { if err := restrict(); err != nil { return err } + if err := takeIdentity(spec); err != nil { + return err + } b.closeBuild() if b.listener >= 0 { // The launcher keeps neither the listener nor the broker connection, so the relay's end is theirs alone. @@ -107,6 +107,9 @@ func (l *launcher) run() error { return err } } + if err := chdir(spec.Command.Dir); err != nil { + return err + } pid, err := startProcess(spec, spec.Command, []uintptr{stdinFD, stdoutFD, stderrFD}) if err != nil { return err @@ -165,13 +168,15 @@ func (l *launcher) serveControl(spec *launchSpec) { } } -// spawn starts the command read from the first of files as the spawned process id, with the other three as its stdin, stdout and stderr, as the process's user and in a session of its own, and answers with its pid. It forks without holding mu; while it does, the reaper leaves unregistered children, so the child's pid stays its own until it is registered. +// spawn starts the command read from the first of files as the spawned process id, with the other three as its stdin, stdout and stderr, as the process's user and in a session of its own, and answers with its pid. It enters the command's directory before it forks, so a directory the world is slow to answer blocks this thread in an ordinary syscall. It forks without holding mu; while it does, the reaper leaves unregistered children, so the child's pid stays its own until it is registered. func (l *launcher) spawn(spec *launchSpec, id uint64, files []*os.File) { defer closeFiles(files) var c command err := gob.NewDecoder(files[0]).Decode(&c) if err != nil { err = &Error{Kind: ErrLauncher, Op: "read spawn", Err: err} + } else { + err = chdir(c.Dir) } l.mu.Lock() if err == nil && !l.running { @@ -181,11 +186,7 @@ func (l *launcher) spawn(spec *launchSpec, id uint64, files []*os.File) { l.mu.Unlock() m := message{Kind: msgSpawned, ID: id} if err == nil { - // A fork that blocks in the world holds this thread and its P where no stop of the world reaches them, so no collection may run or be about to start until the fork returns. Turning collection off stops new ones; runtime.GC waits out one that started before. - gc := debug.SetGCPercent(-1) - runtime.GC() m.Pid, err = startProcess(spec, c, []uintptr{files[1].Fd(), files[2].Fd(), files[3].Fd()}) - debug.SetGCPercent(gc) } l.mu.Lock() if err != nil { @@ -221,7 +222,7 @@ func (l *launcher) write() { l.mu.Unlock() for _, m := range out { if m.Kind == 0 { - // A spawn blocked before its exec holds a copy of this end, and the exit waits for it, so the daemon learns of the exit from the shutdown. + // A spawn blocked on the world keeps the launcher, and this end, from closing until the teardown stops the world, so the daemon learns of the exit from the shutdown. l.ctl.interrupt() os.Exit(l.code) } @@ -408,7 +409,6 @@ func startRelay(spec *launchSpec, listener int) (int, error) { } files[processshim.RelayBrokerFD], files[processshim.RelayListenerFD] = relayFD, uintptr(listener) pid, err := syscall.ForkExec(processshim.RelayPath, processshim.RelayArgs, &syscall.ProcAttr{ - Dir: "/", Env: []string{}, Files: files, Sys: &syscall.SysProcAttr{ @@ -422,10 +422,51 @@ func startRelay(spec *launchSpec, listener int) (int, error) { return pid, nil } -// startProcess starts c as the process's user, in a session of its own, with stdio as its standard descriptors. +// takeIdentity gives this thread a working directory of its own and the process's file system identity, with no capability but the two a fork needs to set the child's user, so that it enters a directory as the process would. The other threads keep theirs. +func takeIdentity(spec *launchSpec) error { + groups := make([]int, len(spec.Groups)) + for i, g := range spec.Groups { + groups[i] = int(g) + } + const setID = 1< fds { t.Errorf("%d descriptors with 50 spawns waiting, %d before", n, fds) } - // The launcher holds the stuck spawn's descriptors and its fork's pipe, nothing for the spawns waiting. - if n := fdCount(t, v.cmd.Process.Pid); n > launcherFDs+6 { + // The launcher holds the stuck spawn's descriptors, nothing for the spawns waiting. + if n := fdCount(t, v.cmd.Process.Pid); n > launcherFDs+4 { t.Errorf("launcher holds %d descriptors with a spawn stuck and 50 waiting, %d before", n, launcherFDs) } stopWaiting() @@ -385,7 +392,7 @@ func TestLateSpawnIsEnded(t *testing.T) { } } -// TestSpawnExitsBeforeItsRegistration checks that a spawned child that dies before its exec, while orphans exit around it, reports its own exit to its own handle and to no other, that a handle whose process has exited reaches nothing, and that Close ends what remains. +// TestSpawnExitsBeforeItsRegistration checks that a spawned child that dies before its exec, while orphans exit around it, reports its own exit to its own handle and to no other, that a handle whose process has exited reaches nothing, and that Close ends what remains. While the child is stuck, its fork may hold up the launcher, so the test acts on the processes directly. func TestSpawnExitsBeforeItsRegistration(t *testing.T) { v, w := startStalled(t) launcher := v.cmd.Process.Pid @@ -395,7 +402,8 @@ func TestSpawnExitsBeforeItsRegistration(t *testing.T) { t.Fatalf("Spawn: %v", err) } defer closeStdio(bystander) - parent, err := v.Spawn(context.Background(), "/.oac/harness/harness", []string{"harness"}, []string{helperEnv + "=wait", "OAC_VIEW_TOKEN=oac-orphan"}, "/data", false) + orphanToken := fmt.Sprintf("oac-orphan-%d", time.Now().UnixNano()) + parent, err := v.Spawn(context.Background(), "/.oac/harness/harness", []string{"harness"}, []string{helperEnv + "=wait", "OAC_VIEW_TOKEN=" + orphanToken}, "/data", false) if err != nil { t.Fatalf("Spawn: %v", err) } @@ -403,25 +411,29 @@ func TestSpawnExitsBeforeItsRegistration(t *testing.T) { if line, err := bufio.NewReader(parent.Stdout).ReadString('\n'); err != nil || line != "ready\n" { t.Fatalf("parent said %q, %v", line, err) } - pending := spawnAsync(context.Background(), v, "noop", "/data/stall") - await(t, w.stalled, "the spawn's lookup in the world") - // The parent's group ends while the child forks; its orphan stays unreaped until the child is registered. - if err := parent.Signal(syscall.SIGKILL); err != nil { - t.Fatalf("Signal to the parent = %v", err) - } - if code := await(t, waitFor(parent), "the parent's exit"); code != -1 { - t.Errorf("parent Wait = %d, want -1", code) + orphan := pidsWith(t, orphanToken) + if len(orphan) != 1 { + t.Fatalf("orphans: %v, want 1", orphan) } - if err := parent.Signal(syscall.SIGKILL); !errors.Is(err, ErrExited) { - t.Errorf("Signal after the parent exited = %v, want ErrExited", err) + group, _ := strconv.Atoi(stat(orphan[0])[2]) + // The child's exec stays on the world. + pending := make(chan spawnResult, 1) + go func() { + s, err := v.Spawn(context.Background(), "/data/stall", []string{"stall"}, nil, "/data", false) + pending <- spawnResult{s, err} + }() + await(t, w.stalled, "the exec's lookup in the world") + // The parent's group ends while the child forks; its orphan stays unreaped until the child is registered. + if err := unix.Kill(-group, unix.SIGKILL); err != nil { + t.Fatal(err) } - eventually(t, "the orphan's exit", func() bool { return zombies(t, launcher) > 0 }) - // The child dies before its exec, once the world answers. + eventually(t, "the orphan's exit", func() bool { return slices.Contains(zombies(t, launcher), orphan[0]) }) child := pidsWith(t, launcherArg0) child = slices.DeleteFunc(child, func(pid int) bool { return pid == launcher }) if len(child) != 1 { t.Fatalf("children before their exec: %v, want 1", child) } + // The child dies before its exec, once the world answers. if err := unix.Kill(child[0], unix.SIGKILL); err != nil { t.Fatal(err) } @@ -434,13 +446,18 @@ func TestSpawnExitsBeforeItsRegistration(t *testing.T) { if code := await(t, waitFor(r.s), "the child's exit"); code != -1 { t.Errorf("child Wait = %d, want -1", code) } - if err := r.s.Signal(0); !errors.Is(err, ErrExited) { - t.Errorf("Signal after the child exited = %v, want ErrExited", err) + if code := await(t, waitFor(parent), "the parent's exit"); code != -1 { + t.Errorf("parent Wait = %d, want -1", code) + } + for _, s := range []*Spawned{r.s, parent} { + if err := s.Signal(0); !errors.Is(err, ErrExited) { + t.Errorf("Signal after the process exited = %v, want ErrExited", err) + } } if err := bystander.Signal(0); err != nil { t.Errorf("Signal to the bystander = %v", err) } - eventually(t, "the orphans reaped", func() bool { return zombies(t, launcher) == 0 }) + eventually(t, "the orphans reaped", func() bool { return len(zombies(t, launcher)) == 0 }) if err := v.Close(); err != nil { t.Fatalf("Close: %v", err) } @@ -966,22 +983,30 @@ func pidsWith(t *testing.T, token string) []int { return pids } -// zombies counts the children of pid that have exited and are not reaped yet. -func zombies(t *testing.T, pid int) int { +// stat returns the fields of pid's stat after its command name, which may hold anything: the state, the parent's pid and the process group come first. It returns nil once pid is gone. +func stat(pid int) []string { + b, err := os.ReadFile(fmt.Sprintf("/proc/%d/stat", pid)) + if err != nil { + return nil + } + return strings.Fields(string(b[bytes.LastIndexByte(b, ')')+1:])) +} + +// zombies returns the children of pid that have exited and are not reaped yet. +func zombies(t *testing.T, pid int) []int { t.Helper() stats, err := filepath.Glob("/proc/[0-9]*/stat") if err != nil { t.Fatal(err) } - n := 0 + var z []int for _, p := range stats { - b, err := os.ReadFile(p) - // The state and the parent's pid follow the command name, which may hold anything. - if f := strings.Fields(string(b[bytes.LastIndexByte(b, ')')+1:])); err == nil && len(f) > 1 && f[0] == "Z" && f[1] == strconv.Itoa(pid) { - n++ + child, _ := strconv.Atoi(filepath.Base(filepath.Dir(p))) + if f := stat(child); len(f) > 1 && f[0] == "Z" && f[1] == strconv.Itoa(pid) { + z = append(z, child) } } - return n + return z } func fdCount(t *testing.T, pid int) int { From 6af05a379988f21550fd88ada9706a920834b00b Mon Sep 17 00:00:00 2001 From: SaladDay <1203511142@qq.com> Date: Tue, 6 Oct 2026 21:48:47 +0000 Subject: [PATCH 5/5] Keep the command out of the launcher's replies and let an ended context win a spawn A failed chdir or exec no longer carries the caller's directory or path back over the control socket, so no reply grows with what a caller passed; the daemon fills the path in from its own command. A control write that fails ends the launcher, as a failed read does, so no request waits for a reply that will not come. Spawn checks its context once the reply is in and sends a spawn whose context has ended through the late-process cleanup. --- .../internal/sessionview/control_linux.go | 15 +++++- .../internal/sessionview/launcher_linux.go | 10 ++-- .../daemon/internal/sessionview/view_linux.go | 48 +++++++++++-------- .../internal/sessionview/view_linux_test.go | 19 +++++++- 4 files changed, 67 insertions(+), 25 deletions(-) diff --git a/apps/daemon/internal/sessionview/control_linux.go b/apps/daemon/internal/sessionview/control_linux.go index 9a77375eb..2e28bf14f 100644 --- a/apps/daemon/internal/sessionview/control_linux.go +++ b/apps/daemon/internal/sessionview/control_linux.go @@ -74,7 +74,7 @@ type message struct { Targets map[string]string // each mountpoint's view path to the path the world presents it at } -// failure carries a launcher *Error across the control socket. +// failure carries a launcher *Error across the control socket. A failure to start a command leaves the command out, so that no message grows with what a caller passed: the daemon has the command, and startErr puts it back. type failure struct { Kind int Op string @@ -117,6 +117,19 @@ func (f failure) err() error { return e } +// startErr returns the error f reports while the launcher starts c, with the directory or path of c that a failed chdir or exec concerns. +func (f failure) startErr(c command) error { + if f.Path == "" { + switch f.Op { + case "chdir": + f.Path = c.Dir + case "exec": + f.Path = c.Path + } + } + return f.err() +} + // control is one end of the launcher's SOCK_SEQPACKET control socket. Each packet holds one gob-encoded message. type control struct { conn *net.UnixConn diff --git a/apps/daemon/internal/sessionview/launcher_linux.go b/apps/daemon/internal/sessionview/launcher_linux.go index 17f57dbad..440be1d62 100644 --- a/apps/daemon/internal/sessionview/launcher_linux.go +++ b/apps/daemon/internal/sessionview/launcher_linux.go @@ -226,7 +226,11 @@ func (l *launcher) write() { l.ctl.interrupt() os.Exit(l.code) } - _ = l.ctl.send(context.Background(), m) + // A send that fails ends the control channel, as a receive that fails does, so that no request waits for a reply that will not come. + if err := l.ctl.send(context.Background(), m); err != nil { + l.ctl.interrupt() + os.Exit(1) + } } } } @@ -459,7 +463,7 @@ func chdir(dir string) error { err := unix.Chdir(dir) _ = unix.PthreadSigmask(unix.SIG_SETMASK, &mask, nil) if err != nil { - return &Error{Kind: ErrExec, Op: "chdir", Path: dir, Err: err} + return &Error{Kind: ErrExec, Op: "chdir", Err: err} } return nil } @@ -475,7 +479,7 @@ func startProcess(spec *launchSpec, c command, stdio []uintptr) (int, error) { }, }) if err != nil { - return 0, &Error{Kind: ErrExec, Op: "exec", Path: c.Path, Err: err} + return 0, &Error{Kind: ErrExec, Op: "exec", Err: err} } return pid, nil } diff --git a/apps/daemon/internal/sessionview/view_linux.go b/apps/daemon/internal/sessionview/view_linux.go index bdcfeed62..9713f23dc 100644 --- a/apps/daemon/internal/sessionview/view_linux.go +++ b/apps/daemon/internal/sessionview/view_linux.go @@ -258,7 +258,7 @@ func (v *View) handshake(ctx context.Context, spec *Spec) error { case err != nil: return v.lost("start", err) case m.Kind == msgFailed: - return m.Fail.err() + return m.Fail.startErr(command{Path: spec.Process.Path, Dir: spec.Process.Dir}) case m.Kind != msgStarted: return &Error{Kind: ErrLauncher, Op: "start", Err: fmt.Errorf("unexpected message %d", m.Kind)} } @@ -569,31 +569,39 @@ func (v *View) Spawn(ctx context.Context, path string, args, env []string, dir s return nil, &Error{Kind: ErrLauncher, Op: "pipe", Err: err} } // The command travels over a pipe, as the spec does, so its size is exec's to bound. The write ends once the launcher has read it, or once no read end remains or the spawn is cancelled. + c := command{Path: path, Args: args, Env: env, Dir: dir} go func() { - _ = gob.NewEncoder(cmdW).Encode(command{Path: path, Args: args, Env: env, Dir: dir}) + _ = gob.NewEncoder(cmdW).Encode(c) cmdW.Close() }() replies, err := v.request(ctx, message{Kind: msgSpawn}, files...) closeFiles(files) if err == nil { + var r reply + ok := false select { - case r, ok := <-replies: - defer func() { <-v.spawning }() - return started(r, ok, ends) + case r, ok = <-replies: + replies = nil case <-ctx.Done(): - err = ctx.Err() - cmdW.Close() - go func() { - defer func() { <-v.spawning }() - r, ok := <-replies - if s, err := started(r, ok, ends); err == nil { - s.Close() - <-s.done - closeFiles([]*os.File{s.Stdin, s.Stdout, s.Stderr}) - } - }() - return nil, err } + // A context that has ended wins over a reply that came as well, and the process goes as a late one does. + if err = ctx.Err(); err == nil { + defer func() { <-v.spawning }() + return started(c, r, ok, ends) + } + cmdW.Close() + go func() { + defer func() { <-v.spawning }() + if replies != nil { // the reply is still to come + r, ok = <-replies + } + if s, err := started(c, r, ok, ends); err == nil { + s.Close() + <-s.done + closeFiles([]*os.File{s.Stdin, s.Stdout, s.Stderr}) + } + }() + return nil, err } cmdW.Close() closeFiles(ends[:]) @@ -601,15 +609,15 @@ func (v *View) Spawn(ctx context.Context, path string, args, env []string, dir s return nil, err } -// started returns the process a spawn's reply reports, with ends as its stdio, or why none started, closing ends. -func started(r reply, ok bool, ends [3]*os.File) (*Spawned, error) { +// started returns the process a spawn of c's reply reports, with ends as its stdio, or why none started, closing ends. +func started(c command, r reply, ok bool, ends [3]*os.File) (*Spawned, error) { err := ErrClosed switch { case ok && r.spawned != nil: r.spawned.Stdin, r.spawned.Stdout, r.spawned.Stderr = ends[0], ends[1], ends[2] return r.spawned, nil case ok: - err = r.Fail.err() + err = r.Fail.startErr(c) } closeFiles(ends[:]) return nil, err diff --git a/apps/daemon/internal/sessionview/view_linux_test.go b/apps/daemon/internal/sessionview/view_linux_test.go index 0aec89a32..dc5bd0c3e 100644 --- a/apps/daemon/internal/sessionview/view_linux_test.go +++ b/apps/daemon/internal/sessionview/view_linux_test.go @@ -200,7 +200,7 @@ func TestViewDescendantsKeepTheGrace(t *testing.T) { } } -// TestSpawnRunsInTheView checks that a spawned process runs as the process does, with its stdio and exit coming back, that its handle reaches nothing once it has exited, that a spawn whose pipes or command fail fails alone, and that the view's end ends it. +// TestSpawnRunsInTheView checks that a spawned process runs as the process does, with its stdio and exit coming back, that its handle reaches nothing once it has exited, that a spawn whose pipes, command or directory fail fails alone, that one whose context ended before it began returns that error and leaves no process, and that the view's end ends it. func TestSpawnRunsInTheView(t *testing.T) { requireView(t) f := newFixture(t) @@ -250,6 +250,23 @@ func TestSpawnRunsInTheView(t *testing.T) { if _, err := spawnHelper(context.Background(), v, "noop", "/.oac/harness/root-only"); !errors.Is(err, ErrExec) || !errors.Is(err, syscall.EACCES) { t.Errorf("Spawn in a directory only root may enter = %v, want ErrExec with EACCES", err) } + cancelled, cancel := context.WithCancel(context.Background()) + cancel() + late := fmt.Sprintf("oac-cancelled-%d", time.Now().UnixNano()) + for range 20 { + if _, err := spawnHelper(cancelled, v, "sleep", "/data", late); !errors.Is(err, context.Canceled) { + t.Fatalf("Spawn with a cancelled context = %v, want context.Canceled", err) + } + } + // A directory longer than a control packet fails alone. This spawn begins once what the cancelled ones started is reaped. + bounded, cancel := context.WithTimeout(context.Background(), 10*time.Second) + defer cancel() + if _, err := spawnHelper(bounded, v, "noop", "/"+strings.Repeat("x", 1<<20)); !errors.Is(err, ErrExec) || !errors.Is(err, syscall.ENAMETOOLONG) { + t.Errorf("Spawn in a 1 MiB directory = %.200v, want ErrExec with ENAMETOOLONG", err) + } + if n := len(pidsWith(t, late)); n != 0 { + t.Errorf("the cancelled spawns left %d processes", n) + } if err := sleeper.Signal(0); err != nil { t.Errorf("Signal to the running spawned process = %v", err) }