fix(agent): supervise Cursor background shells with a liveness guard

Cursor's stream-json reports a launched background shell as the
`isBackground` boolean at the root of its completed tool-call result, and
the adapter acted on none of it: a top-level execution that went quiet
after launching one kept holding its runtime slot until the execution
deadline (upstream #7833).

Bound that shape instead of condemning it. The structural observation now
arms a Cursor-private liveness guard: a rolling semantic-progress
deadline (10 minutes in production, overridable per backend so tests do
not have to wait a minute-scale timeout) while execution continues
normally. Only a supervised run that stops progressing and never emits
its terminal result is ended, and it fails closed with a dedicated,
content-free reason rather than a generic cancellation.

Precedence is decided when the cancellation happens, not guessed at
finalization: fire() declines to claim a run the execution deadline or
the caller had already ended, so `timeout` and `aborted` keep their
classifications, while a cancellation the guard provoked is never
rewritten into "execution cancelled". A valid terminal Cursor result
outranks a latched background observation and still completes.

Detection is structural only — no reading of the command string, no
substring of captured output, no assistant text — and malformed or
non-object result metadata leaves the call foreground, so it cannot arm
the guard or fail the rest of the tool parse. Raw result bytes are
forwarded unchanged. The guard owns one goroutine and one timer per
execution, started on the first background observation and waited for
before the reader hands over its last channel; the overwhelmingly common
foreground run never arms it and pays nothing.

Tests run cross-platform by re-executing the test binary as a fake
cursor-agent, so the pathological stream stays alive and only the daemon
can end it. Test A is RED on unpatched main, where the same run is
bounded solely by the 6s execution deadline with status "timeout".

Co-authored-by: multica-agent <github@multica.ai>
This commit is contained in:
worker-opencode
2026-09-05 00:50:29 +09:00
co-authored by multica-agent
parent fb5af91416
commit 15605057ed
3 changed files with 1033 additions and 1 deletions
+7
View File
@@ -28,6 +28,13 @@ func TestMain(m *testing.M) {
runFakeOpencodeStdinHelper()
os.Exit(0)
}
// Cursor fakes are dispatched the same way as Claude's: the behavior lives
// with the Cursor tests (cursor_liveness_test.go), and only this dispatcher is
// package-wide, because the package may have exactly one TestMain.
if mode := os.Getenv(cursorFakeModeEnv); mode != "" {
runFakeCursorStream(mode)
os.Exit(0)
}
switch mode := os.Getenv("CLAUDE_FAKE_MODE"); mode {
case "":
os.Exit(m.Run())
+277 -1
View File
@@ -20,6 +20,31 @@ import (
// format: events are newline-delimited JSON objects with a "type" field.
type cursorBackend struct {
cfg Config
// backgroundGuardTimeout overrides the semantic-progress deadline of the
// background-shell liveness guard for this backend. Zero — the value every
// production construction path leaves it at, including New — means
// defaultCursorBackgroundGuardTimeout. It is a per-backend field rather than
// a package-level knob precisely because this package's cursor tests run in
// parallel under -race: a shared variable shortened for one run would be
// read (and raced) by every other concurrent Cursor execution.
backgroundGuardTimeout time.Duration
}
// defaultCursorBackgroundGuardTimeout is how long a Cursor run may go without
// meaningful top-level progress after Cursor has reported a background shell,
// before the liveness guard ends the run. Long enough that a real dev server
// plus test loop keeps its slot, short enough that a run stalled in the #7833
// shape stops holding a runtime slot well before the daemon-wide watchdog.
const defaultCursorBackgroundGuardTimeout = 10 * time.Minute
// guardTimeout resolves the per-backend override; zero means the production
// default so no caller has to know about the field.
func (b *cursorBackend) guardTimeout() time.Duration {
if b.backgroundGuardTimeout > 0 {
return b.backgroundGuardTimeout
}
return defaultCursorBackgroundGuardTimeout
}
func (b *cursorBackend) Execute(ctx context.Context, prompt string, opts ExecOptions) (*Session, error) {
@@ -35,6 +60,12 @@ func (b *cursorBackend) Execute(ctx context.Context, prompt string, opts ExecOpt
timeout := opts.Timeout
runCtx, cancel := runContext(ctx, timeout)
// The guard is inert until the stream reports a background shell, so
// creating it costs nothing on the overwhelmingly common foreground run and
// owns no goroutine until then. The early-return paths below never reach the
// reader, so the guard is never armed there.
guard := newCursorBackgroundGuard(ctx, runCtx, cancel, b.guardTimeout())
args := buildCursorArgs(opts, b.cfg.Logger)
cmd, _, _ := b.cfg.commandAt(execName).execVia(runCtx, chooseCursorInvocation, lookedUp, args, b.cfg.Logger)
hideAgentWindow(cmd)
@@ -88,6 +119,10 @@ func (b *cursorBackend) Execute(ctx context.Context, prompt string, opts ExecOpt
defer cancel()
defer close(msgCh)
defer close(resCh)
// Disarm the guard, and wait for its goroutine, before this goroutine
// hands over its last channel: a completed run must not leave a timer
// watching a process that is already gone.
defer guard.Stop()
// Close stdout when the context is cancelled so scanner.Scan() unblocks.
// Closing stdin too releases a prompt write still blocked on a full pipe
@@ -159,6 +194,16 @@ func (b *cursorBackend) Execute(ctx context.Context, prompt string, opts ExecOpt
sessionID = sid
}
// Semantic progress: only events showing the top-level agent is still
// working refresh the background liveness deadline. Malformed JSON
// never reaches here, unknown event types are deliberately not
// believed as progress, and anything the child writes to stderr or a
// background server logs on its own is not part of this stream at
// all — noisy output must not keep a stalled run alive.
if cursorSemanticProgressEvent(evt.Type, evt.Subtype) {
guard.ObserveProgress()
}
switch evt.Type {
case "system":
if evt.Subtype == "init" {
@@ -215,6 +260,14 @@ func (b *cursorBackend) Execute(ctx context.Context, prompt string, opts ExecOpt
})
case "completed":
call := parseCursorToolCall(&evt)
if call.Background {
// A background shell is a lifecycle state that needs extra
// supervision, not a failure. Execution continues: Cursor may
// legitimately go on to test the server it just started, stop
// it, and emit its terminal result. What changes is that the
// run is now bounded by the liveness guard.
guard.ObserveBackground()
}
trySend(msgCh, Message{
Type: MessageToolResult,
Tool: call.Name,
@@ -264,6 +317,12 @@ func (b *cursorBackend) Execute(ctx context.Context, prompt string, opts ExecOpt
// Current Cursor Agent versions can emit the terminal result
// event but keep a worker process alive. Treat result as the
// protocol boundary so the daemon can report completion.
//
// A previously observed background shell does NOT outrank this: the
// terminal result is Cursor's authoritative completion boundary, so
// the guard is disarmed here and finalization takes the resultSeen
// branch.
guard.Stop()
cancel()
case "error":
@@ -346,7 +405,18 @@ func (b *cursorBackend) Execute(ctx context.Context, prompt string, opts ExecOpt
finalError = "cursor-agent returned an error result without details"
}
} else {
switch {
// Ordering of the remaining causes. The guard is examined first, but
// that is not #8042's rule returning through another door: it can only
// ever report a cancellation it actually caused, because fire()
// declines to claim a run the execution deadline or the caller had
// already ended. So a configured deadline still reports `timeout`,
// external cancellation still reports `aborted`, and a cancellation the
// guard itself provoked is never rewritten into the generic
// "execution cancelled".
switch guardCause := guard.Cause(); {
case guardCause != "":
finalStatus = "failed"
finalError = guardCause
case runCtx.Err() == context.DeadlineExceeded:
finalStatus = "timeout"
finalError = fmt.Sprintf("cursor-agent timed out after %s", timeout)
@@ -669,6 +739,11 @@ type cursorToolCall struct {
CallID string
Input map[string]any
Result string
// Background carries the one piece of lifecycle metadata this fix needs: the
// structural `isBackground` boolean at the root of a completed call's result,
// never a reading of the command string or of the result's text.
Background bool
}
// cursorToolCallKeySuffix is how Cursor names the per-tool payload: the tool is
@@ -714,10 +789,30 @@ func parseCursorToolCall(evt *cursorStreamEvent) cursorToolCall {
call.Input = payload.Args
if len(payload.Result) > 0 {
call.Result = string(payload.Result)
call.Background = cursorResultReportsBackground(payload.Result)
}
return call
}
// cursorResultReportsBackground reads the single lifecycle flag the adapter acts
// on: the `isBackground` boolean at the ROOT of a completed tool call's result.
//
// It is a structural read on purpose. A result that merely mentions the text
// "isBackground" (a shell's captured stdout, a nested field, a string where a
// boolean belongs) is not an observation of a background shell, and neither is
// metadata that will not decode: malformed or non-object result metadata leaves
// the call foreground, must not panic, and must not fail the rest of the tool
// parse. The raw result stays exactly as `parseCursorToolCall` captured it.
func cursorResultReportsBackground(raw json.RawMessage) bool {
var resultMeta struct {
IsBackground bool `json:"isBackground"`
}
if err := json.Unmarshal(raw, &resultMeta); err != nil {
return false
}
return resultMeta.IsBackground
}
// cursorToolPayloadKey picks the `<name>ToolCall` key of a tool_call envelope.
// Observed payloads carry exactly one; the sort keeps the choice deterministic
// if a future CLI ever emits more than one.
@@ -782,6 +877,187 @@ func (b *cursorBackend) accumulateResultUsage(usage map[string]TokenUsage, evt *
usage[model] = u
}
// ── Background liveness guard ──
// cursorBackgroundLivenessGuardError is the dedicated failure reason for a run
// the guard ended. Content-free by construction: naming the shape of the
// failure (background shell alive, progress stopped, no terminal result) is
// useful to an operator, while the command, its output, paths and PIDs are not.
const cursorBackgroundLivenessGuardError = "cursor-agent stopped producing semantic progress after launching a background shell and did not emit a terminal result"
// cursorSemanticProgressEvent reports whether a parsed top-level event shows the
// Cursor agent itself still working, which is what the background liveness guard
// measures. Deliberately excluded: `system` (startup/control), `error`, the
// terminal `result` (the guard is disarmed by then anyway), and every type or
// subtype this parser does not recognize — believing an unknown event as
// progress would let the exact protocol drift MUL-5231 and MUL-5434 fixed
// silently extend a stalled run forever.
func cursorSemanticProgressEvent(eventType, subtype string) bool {
switch eventType {
case "assistant", "tool_call", "tool_use", "tool_result", "text", "step_finish":
return true
case "thinking":
// Only the two recognized subtypes: a delta carries reasoning the agent
// is producing now, `completed` closes a block. An unknown subtype is
// counted as unhandled elsewhere and is not progress here.
return subtype == "delta" || subtype == "completed"
default:
return false
}
}
// cursorBackgroundGuard bounds the pathological run from #7833: Cursor launches
// a background shell, stops making meaningful top-level progress, and never
// emits its terminal result, so the task keeps holding a runtime slot until the
// daemon-wide watchdog.
//
// It holds the two pieces of state the design calls for — `active` (at least one
// background shell has been observed, so this run needs the extra supervision)
// and the rolling semantic-progress deadline (materialised as the timer, which
// IS last-progress-plus-timeout) — and owns exactly one goroutine, started on
// the first background observation and never restarted.
//
// The timer is touched only by that goroutine: observers hand it a coalesced
// reset signal instead of calling Reset themselves, so there is no racing
// timer.Reset and no drained-channel double-fire to reason about. Every method
// is safe for the reader goroutine to call at any point in the run.
type cursorBackgroundGuard struct {
parent context.Context // caller cancellation, which outranks the guard
runCtx context.Context // execution deadline, which outranks the guard
cancel context.CancelFunc // the managed process's own cancellation path
timeout time.Duration // semantic-progress deadline for this run
progress chan struct{} // buffered 1: resets coalesce, sends never block the reader
stop chan struct{}
done chan struct{}
// The terminal-result path and the deferred teardown both disarm the same
// guard, so closing `done` has to survive being called twice.
closeDone sync.Once
mu sync.Mutex
active bool // a background shell has been observed
watching bool // the single monitor goroutine has been started
stopped bool
cause string // non-empty once the guard itself ended the run
}
func newCursorBackgroundGuard(parent, runCtx context.Context, cancel context.CancelFunc, timeout time.Duration) *cursorBackgroundGuard {
return &cursorBackgroundGuard{
parent: parent,
runCtx: runCtx,
cancel: cancel,
timeout: timeout,
progress: make(chan struct{}, 1),
stop: make(chan struct{}),
done: make(chan struct{}),
}
}
// ObserveBackground records that Cursor structurally reported a completed
// background shell. It changes no task status and cancels nothing; it arms the
// deadline. Repeated background shells in one run are a no-op, which is what
// keeps the goroutine and timer count bounded at one per execution.
func (g *cursorBackgroundGuard) ObserveBackground() {
g.mu.Lock()
if g.active || g.stopped {
g.mu.Unlock()
return
}
g.active = true
started := g.watching
g.watching = true
g.mu.Unlock()
if !started {
go g.watch()
}
}
// ObserveProgress rolls the deadline forward. It is non-blocking by design: the
// stream reader must never wait on supervision, and a burst of events needs only
// one pending reset.
func (g *cursorBackgroundGuard) ObserveProgress() {
select {
case g.progress <- struct{}{}:
default:
}
}
// Stop disarms the guard and waits for its goroutine to exit. Idempotent, safe
// from any goroutine, and safe to call for a guard that was never armed.
func (g *cursorBackgroundGuard) Stop() {
g.mu.Lock()
watching := g.watching
g.stopped = true
g.mu.Unlock()
select {
case <-g.stop:
default:
close(g.stop)
}
if !watching {
// No goroutine will ever close it.
g.closeDone.Do(func() { close(g.done) })
}
<-g.done
}
func (g *cursorBackgroundGuard) watch() {
defer g.closeDone.Do(func() { close(g.done) })
timer := time.NewTimer(g.timeout)
defer timer.Stop()
for {
select {
case <-timer.C:
g.fire()
return
case <-g.progress:
// Stop-and-drain keeps a signal that raced the expiry from
// resurrecting an already-fired deadline.
if !timer.Stop() {
select {
case <-timer.C:
default:
}
}
timer.Reset(g.timeout)
case <-g.stop:
return
}
}
}
// fire ends the run, but only if the guard is genuinely the first cause.
//
// This is where the precedence rules are enforced rather than guessed at
// finalization time: if the execution deadline had passed, or the caller had
// cancelled, the run was already coming down for its own reason and the guard
// must not claim it — those two retain `timeout` and `aborted`. Conversely, when
// the guard does cancel, that cancellation shows up later as runCtx being
// cancelled, and must stay the dedicated background-liveness failure.
func (g *cursorBackgroundGuard) fire() {
g.mu.Lock()
defer g.mu.Unlock()
if g.stopped || g.runCtx.Err() != nil || g.parent.Err() != nil {
return
}
g.cause = cursorBackgroundLivenessGuardError
g.cancel()
}
// Cause returns the dedicated failure reason when the guard ended this run, and
// "" when it never fired or when a deadline or cancellation got there first.
func (g *cursorBackgroundGuard) Cause() string {
g.mu.Lock()
defer g.mu.Unlock()
return g.cause
}
// ── Cursor stream-json types ──
type cursorStreamEvent struct {
+749
View File
@@ -0,0 +1,749 @@
package agent
import (
"context"
"encoding/json"
"fmt"
"io"
"log/slog"
"os"
"strings"
"testing"
"time"
)
// Background-shell liveness regression for #7833 (PUCK-147).
//
// The shape being bounded is: Cursor launches a background shell, its top-level
// execution then stops making semantic progress, and its terminal `result` never
// arrives — so the task holds a runtime slot until the daemon-wide watchdog.
//
// A background shell is a lifecycle state that needs supervision, not a failure:
// these tests pin both halves of that distinction. The pathological run must end
// inside a bounded window, and a legitimate run that starts a server, keeps
// working, and reports a result must be left alone.
//
// These run on every platform, so the fake is the test binary re-executed as
// cursor-agent (dispatched from the package TestMain via CURSOR_FAKE_MODE)
// rather than the /bin/sh fixtures in cursor_execute_unix_test.go.
// cursorFakeModeEnv selects the fake cursor-agent behaviour when this test
// binary is re-executed as the child. Mirrors CLAUDE_FAKE_MODE.
const cursorFakeModeEnv = "CURSOR_FAKE_MODE"
const (
cursorFakeStall = "background_shell_stall"
cursorFakeWorkThenResult = "background_shell_work_then_result"
cursorFakePulses = "background_shell_progress_pulses"
cursorFakeResultWins = "background_shell_result_wins"
cursorFakeMalformedStall = "background_shell_malformed_stall"
cursorFakeUnverifiedActivity = "background_shell_unverified_activity"
)
// cursorBGSessionID is emitted by every fake below; the guard must fail closed
// without losing it.
const cursorBGSessionID = "sess-cursor-bg"
const (
cursorBGInit = `{"type":"system","subtype":"init","session_id":"` + cursorBGSessionID + `"}`
cursorBGShellStarted = `{"type":"tool_call","subtype":"started","call_id":"call-bg",` +
`"tool_call":{"shellToolCall":{"args":{"command":"python3 -m http.server 8000"},` +
`"toolCallId":"call-bg"}}}`
// cursorBGShellCompleted is the authoritative structural observation: the
// `result` object's root carries `"isBackground": true`.
cursorBGShellCompleted = `{"type":"tool_call","subtype":"completed","call_id":"call-bg",` +
`"tool_call":{"shellToolCall":{"args":{"command":"python3 -m http.server 8000"},` +
`"result":{"success":{"exitCode":0},"isBackground":true},"toolCallId":"call-bg"}}}`
// cursorBGMalformedCompleted carries the same text where the result is a
// JSON string rather than an object. It is not a lifecycle observation and
// must not arm the guard.
cursorBGMalformedCompleted = `{"type":"tool_call","subtype":"completed","call_id":"call-bg",` +
`"tool_call":{"shellToolCall":{"args":{"command":"python3 -m http.server 8000"},` +
`"result":"{\"isBackground\":true}","toolCallId":"call-bg"}}}`
cursorBGThinking = `{"type":"thinking","subtype":"delta","text":"the server is answering"}`
cursorBGReadStarted = `{"type":"tool_call","subtype":"started","call_id":"call-read",` +
`"tool_call":{"readToolCall":{"args":{"path":"server.log"},"toolCallId":"call-read"}}}`
cursorBGReadCompleted = `{"type":"tool_call","subtype":"completed","call_id":"call-read",` +
`"tool_call":{"readToolCall":{"args":{"path":"server.log"},` +
`"result":{"content":"GET / 200"},"toolCallId":"call-read"}}}`
cursorBGStepFinish = `{"type":"step_finish","model":"cursor",` +
`"part":{"tokens":{"input":120,"output":8,"cache":{"read":0}}}}`
cursorBGResult = `{"type":"result","subtype":"success","is_error":false,` +
`"result":"dev server checked and stopped","session_id":"` + cursorBGSessionID + `"}`
)
// cursorFakeEvent is one stream-json line, written after the given delay.
type cursorFakeEvent struct {
delay time.Duration
line string
}
// runFakeCursorStream is the fake cursor-agent, dispatched from the package
// TestMain before the testing package parses the CLI's argv.
//
// It honours the real CLI's stdin contract first (read to EOF, the prompt is
// delivered on stdin — see buildCursorArgs and the drainStdin note in
// cursor_execute_unix_test.go), because a fake that never reads it races EPIPE
// against the daemon's prompt write, and that write outranks most failures when
// the error is classified.
func runFakeCursorStream(mode string) {
var events []cursorFakeEvent
// emitUnverifiedActivity is the "loud but meaningless" stream: unknown event
// types, malformed JSON and stderr noise, none of which is semantic progress.
var unverified bool
switch mode {
case cursorFakeStall:
// The #7833 shape: background shell confirmed, then silence forever.
events = []cursorFakeEvent{
{line: cursorBGInit},
{line: cursorBGShellStarted},
{line: cursorBGShellCompleted},
}
case cursorFakeWorkThenResult:
// Legitimate workflow: start the server, inspect it, report normally.
events = []cursorFakeEvent{
{line: cursorBGInit},
{line: cursorBGShellStarted},
{line: cursorBGShellCompleted},
{delay: 150 * time.Millisecond, line: cursorBGThinking},
{delay: 150 * time.Millisecond, line: cursorBGReadStarted},
{delay: 150 * time.Millisecond, line: cursorBGReadCompleted},
{delay: 150 * time.Millisecond, line: cursorBGStepFinish},
{delay: 150 * time.Millisecond, line: cursorBGResult},
}
case cursorFakePulses:
// Progress keeps landing well inside each guard interval, so the total
// run spans several intervals without ever going quiet long enough.
events = []cursorFakeEvent{{line: cursorBGInit}, {line: cursorBGShellStarted}, {line: cursorBGShellCompleted}}
for i := 0; i < 5; i++ {
events = append(events,
cursorFakeEvent{delay: 500 * time.Millisecond, line: cursorBGThinking},
cursorFakeEvent{delay: 0, line: cursorBGStepFinish},
)
}
events = append(events, cursorFakeEvent{delay: 200 * time.Millisecond, line: cursorBGResult})
case cursorFakeResultWins:
// Terminal result arrives right after the background observation.
events = []cursorFakeEvent{
{line: cursorBGInit},
{line: cursorBGShellStarted},
{line: cursorBGShellCompleted},
{delay: 150 * time.Millisecond, line: cursorBGResult},
}
case cursorFakeMalformedStall:
// Same stall, but the lifecycle flag only exists as text inside a
// non-object result: there is nothing to supervise here.
events = []cursorFakeEvent{
{line: cursorBGInit},
{line: cursorBGShellStarted},
{line: cursorBGMalformedCompleted},
}
case cursorFakeUnverifiedActivity:
unverified = true
events = []cursorFakeEvent{
{line: cursorBGInit},
{line: cursorBGShellStarted},
{line: cursorBGShellCompleted},
}
default:
fmt.Fprintf(os.Stderr, "unknown CURSOR_FAKE_MODE: %q\n", mode)
os.Exit(2)
}
if _, err := io.Copy(io.Discard, os.Stdin); err != nil {
fmt.Fprintf(os.Stderr, "fake cursor-agent: read prompt: %v\n", err)
os.Exit(21)
}
if unverified {
go func() {
// Noise of every kind the guard is told not to believe: a background
// server's own log lines on stderr, an unknown top-level event type,
// malformed JSON, and the CLI echoing our prompt back.
for i := 0; i < 200; i++ {
fmt.Fprintf(os.Stderr, "GET / 200 OK - %d bytes\n", i)
fmt.Printf("%s\n", `{"type":"background_log","line":"still serving"}`)
fmt.Printf("%s\n", `{"type":"result","subtype":"success","result":"trunc`)
fmt.Printf("%s\n", `{"type":"user","message":{"role":"user","content":"still serving"}}`)
time.Sleep(50 * time.Millisecond)
}
}()
}
for _, evt := range events {
if evt.delay > 0 {
time.Sleep(evt.delay)
}
if _, err := fmt.Fprintln(os.Stdout, evt.line); err != nil {
os.Exit(22)
}
}
if mode != cursorFakeStall && mode != cursorFakeMalformedStall && mode != cursorFakeUnverifiedActivity {
// The fake completed normally: exit before the test framework can print
// PASS/ok into the JSON stream.
os.Exit(0)
}
// Stay alive with stdout open, so only the daemon can end this run. This is
// the state that held a runtime slot in #7833. A sleep, not a bare
// select{}: the runtime deadlock detector would crash the child instead.
time.Sleep(10 * time.Minute)
}
// cursorFakeRun is the observable outcome of one fake cursor-agent run.
type cursorFakeRun struct {
result Result
messages []Message
elapsed time.Duration
}
// runCursorFake executes the cursor backend against the re-executed fake.
//
// The guard duration is set on the backend instance rather than through a
// package-level knob: this package's cursor tests run in parallel under -race,
// and a shared variable would be read by every concurrent Cursor execution.
//
// Waiting for the message channel to close (not just for Result) matters: the
// reader closes it after it has disarmed the guard and waited for its watcher,
// so a guard goroutine that outlived the run would hang this instead of
// silently leaking.
func runCursorFake(t *testing.T, mode string, guard, execTimeout time.Duration) cursorFakeRun {
t.Helper()
self, err := os.Executable()
if err != nil {
t.Fatalf("os.Executable: %v", err)
}
backend, err := New("cursor", Config{
ExecutablePath: self,
Env: map[string]string{cursorFakeModeEnv: mode},
Logger: slog.Default(),
})
if err != nil {
t.Fatalf("New(cursor): %v", err)
}
cb, ok := backend.(*cursorBackend)
if !ok {
t.Fatalf("cursor backend is %T", backend)
}
cb.backgroundGuardTimeout = guard
ctx, cancel := context.WithTimeout(context.Background(), execTimeout+10*time.Second)
defer cancel()
start := time.Now()
session, err := cb.Execute(ctx, "start a dev server and check it", ExecOptions{Timeout: execTimeout})
if err != nil {
t.Fatalf("Execute: %v", err)
}
var messages []Message
msgsDone := make(chan struct{})
go func() {
defer close(msgsDone)
for msg := range session.Messages {
messages = append(messages, msg)
}
}()
var result Result
select {
case res, ok := <-session.Result:
if !ok {
t.Fatal("result channel closed without a value")
}
result = res
case <-time.After(execTimeout + 5*time.Second):
t.Fatal("cursor backend never reported a result")
}
elapsed := time.Since(start)
select {
case <-msgsDone:
case <-time.After(10 * time.Second):
t.Fatal("message channel never closed: the run's reader or its liveness watcher outlived the run")
}
return cursorFakeRun{result: result, messages: messages, elapsed: elapsed}
}
// assertGuardFailure pins the fail-closed outcome: failed, for the dedicated
// background-liveness reason, with no partial transcript promoted to output.
func (r cursorFakeRun) assertGuardFailure(t *testing.T) {
t.Helper()
if r.result.Status != "failed" {
t.Fatalf("status = %q, want failed; error=%q", r.result.Status, r.result.Error)
}
if !strings.Contains(r.result.Error, "background shell") ||
!strings.Contains(r.result.Error, "did not emit a terminal result") {
t.Errorf("error = %q, want the dedicated background-liveness reason", r.result.Error)
}
// The guard's own cancellation must not be rewritten into the generic one.
if strings.Contains(r.result.Error, "execution cancelled") {
t.Errorf("error = %q, must not be reported as generic cancellation", r.result.Error)
}
if r.result.Status == "timeout" || r.result.Status == "aborted" {
t.Errorf("status = %q, want the guard's own classification", r.result.Status)
}
if r.result.Output != "" {
t.Errorf("output = %q, want the partial transcript withheld", r.result.Output)
}
}
// TestCursorBackgroundShellStallFailsWithinGuardWindow is Test A: the #7833
// pathological run. It must be ended by the guard, long before the execution
// deadline, and must fail closed.
//
// RED on unpatched main: main ignores `isBackground` entirely, so the same fake
// stream holds the slot until ExecOptions.Timeout and reports
// status="timeout" — see the RED probe evidence in the PUCK-147 report.
func TestCursorBackgroundShellStallFailsWithinGuardWindow(t *testing.T) {
t.Parallel()
const (
guard = 500 * time.Millisecond
execTimeout = 12 * time.Second
)
run := runCursorFake(t, cursorFakeStall, guard, execTimeout)
// Well inside the 12s execution deadline: the guard, not the timeout, ended
// the run.
if run.elapsed > 6*time.Second {
t.Errorf("elapsed = %s, want the guard to bound the run well before the 12s deadline", run.elapsed)
}
run.assertGuardFailure(t)
if run.result.SessionID != cursorBGSessionID {
t.Errorf("session id = %q, want %q", run.result.SessionID, cursorBGSessionID)
}
// Detection must not cost the transcript anything: the raw tool result is
// still forwarded verbatim.
var sawRawResult bool
for _, msg := range run.messages {
if msg.Type == MessageToolResult && strings.Contains(msg.Output, `"isBackground":true`) {
sawRawResult = true
}
}
if !sawRawResult {
t.Errorf("background shell tool result was not forwarded unchanged: %+v", run.messages)
}
}
// TestCursorBackgroundShellAllowsLegitimateFollowOnWork is Test B: a background
// shell followed by real work and a terminal result must complete normally.
// #8042's immediate cancel() fails this — the run is killed at the first step.
func TestCursorBackgroundShellAllowsLegitimateFollowOnWork(t *testing.T) {
t.Parallel()
run := runCursorFake(t, cursorFakeWorkThenResult, 600*time.Millisecond, 12*time.Second)
if run.result.Status != "completed" {
t.Fatalf("status = %q, want completed; error=%q", run.result.Status, run.result.Error)
}
if run.result.Output != "dev server checked and stopped" {
t.Errorf("output = %q, want the Cursor terminal result preserved", run.result.Output)
}
if strings.Contains(run.result.Error, "background shell") {
t.Errorf("error = %q, want no background-liveness failure on a normal run", run.result.Error)
}
if run.result.SessionID != cursorBGSessionID {
t.Errorf("session id = %q, want %q", run.result.SessionID, cursorBGSessionID)
}
}
// TestCursorBackgroundShellProgressExtendsGuard is Test C: the guard measures
// inactivity, not wall-clock time since the shell started. Five progress pulses
// land 500ms apart inside a 1s deadline, so the run spans several guard
// intervals and still completes. A deadline that never moved would fire ~1s in.
func TestCursorBackgroundShellProgressExtendsGuard(t *testing.T) {
t.Parallel()
run := runCursorFake(t, cursorFakePulses, time.Second, 12*time.Second)
if run.elapsed < 2*time.Second {
t.Errorf("elapsed = %s, want the run to outlive at least one guard interval", run.elapsed)
}
if run.result.Status != "completed" {
t.Fatalf("status = %q, want completed; error=%q", run.result.Status, run.result.Error)
}
if run.result.Output != "dev server checked and stopped" {
t.Errorf("output = %q, want the Cursor terminal result preserved", run.result.Output)
}
}
// TestCursorBackgroundOutputTextCannotFakeLifecycleIsStructural is Test D: the
// observation comes from the result object's root boolean and nowhere else.
func TestCursorBackgroundOutputTextCannotFakeLifecycleIsStructural(t *testing.T) {
t.Parallel()
tests := []struct {
name string
event string
wantBackground bool
wantResult string
}{
{
name: "root boolean true is the observation",
event: `{"type":"tool_call","subtype":"completed","call_id":"c1",` +
`"tool_call":{"shellToolCall":{"args":{"command":"serve"},"result":{"success":{"exitCode":0},"isBackground":true},"toolCallId":"c1"}}}`,
wantBackground: true,
wantResult: `{"success":{"exitCode":0},"isBackground":true}`,
},
{
name: "explicit false is foreground",
event: `{"type":"tool_call","subtype":"completed","call_id":"c2",` +
`"tool_call":{"shellToolCall":{"args":{"command":"ls"},"result":{"success":{"exitCode":0},"isBackground":false},"toolCallId":"c2"}}}`,
wantResult: `{"success":{"exitCode":0},"isBackground":false}`,
},
{
name: "isBackground only inside captured stdout is not an observation",
event: `{"type":"tool_call","subtype":"completed","call_id":"c3",` +
`"tool_call":{"shellToolCall":{"args":{"command":"grep -r isBackground ."},` +
`"result":{"stdout":"{\"isBackground\":true}","exitCode":0},"toolCallId":"c3"}}}`,
wantResult: `{"stdout":"{\"isBackground\":true}","exitCode":0}`,
},
{
name: "a nested isBackground is not a root field",
event: `{"type":"tool_call","subtype":"completed","call_id":"c4",` +
`"tool_call":{"shellToolCall":{"args":{"command":"serve"},` +
`"result":{"meta":{"isBackground":true},"exitCode":0},"toolCallId":"c4"}}}`,
wantResult: `{"meta":{"isBackground":true},"exitCode":0}`,
},
{
name: "a string where the boolean belongs is not an observation",
event: `{"type":"tool_call","subtype":"completed","call_id":"c5",` +
`"tool_call":{"shellToolCall":{"args":{"command":"serve"},` +
`"result":{"isBackground":"true"},"toolCallId":"c5"}}}`,
wantResult: `{"isBackground":"true"}`,
},
{
name: "non-object result is not an observation",
event: `{"type":"tool_call","subtype":"completed","call_id":"c6",` +
`"tool_call":{"shellToolCall":{"args":{"command":"serve"},"result":[{"isBackground":true}],"toolCallId":"c6"}}}`,
wantResult: `[{"isBackground":true}]`,
},
{
name: "missing result stays foreground",
event: `{"type":"tool_call","subtype":"completed","call_id":"c7",` +
`"tool_call":{"shellToolCall":{"args":{"command":"serve"},"toolCallId":"c7"}}}`,
},
{
name: "null result stays foreground",
event: `{"type":"tool_call","subtype":"completed","call_id":"c8",` +
`"tool_call":{"shellToolCall":{"args":{"command":"serve"},"result":null,"toolCallId":"c8"}}}`,
wantResult: `null`,
},
}
for _, tt := range tests {
tt := tt
t.Run(tt.name, func(t *testing.T) {
t.Parallel()
var evt cursorStreamEvent
if err := json.Unmarshal([]byte(tt.event), &evt); err != nil {
t.Fatalf("unmarshal event: %v", err)
}
call := parseCursorToolCall(&evt)
if call.Background != tt.wantBackground {
t.Errorf("Background = %v, want %v", call.Background, tt.wantBackground)
}
if call.Result != tt.wantResult {
t.Errorf("Result = %q, want %q (raw result must be preserved verbatim)", call.Result, tt.wantResult)
}
if call.Name != "shell" || call.CallID == "" {
t.Errorf("shell identity lost: name=%q callID=%q", call.Name, call.CallID)
}
})
}
}
// TestCursorBackgroundShellMalformedResultDoesNotArmGuard is Test E: metadata
// that cannot be decoded must not panic, must keep the raw result, and must not
// put the run under the guard — so this stalled run is still ended by the
// execution deadline it always had, not by a background-liveness failure.
func TestCursorBackgroundShellMalformedResultDoesNotArmGuard(t *testing.T) {
t.Parallel()
const (
guard = 500 * time.Millisecond
execTimeout = 3 * time.Second
)
// A guard armed on this stream would fire ~0.5s in; the deadline is 3s, so
// the outcome below can only come from the guard having stayed disarmed.
run := runCursorFake(t, cursorFakeMalformedStall, guard, execTimeout)
if run.result.Status != "timeout" {
t.Errorf("status = %q, want timeout: the malformed result is not an observation, so only the execution deadline may end this run; error=%q",
run.result.Status, run.result.Error)
}
if strings.Contains(run.result.Error, "background shell") {
t.Errorf("error = %q, must not claim a background-liveness failure", run.result.Error)
}
if run.result.Output != "" {
t.Errorf("output = %q, want the partial transcript withheld", run.result.Output)
}
if run.result.SessionID != cursorBGSessionID {
t.Errorf("session id = %q, want %q", run.result.SessionID, cursorBGSessionID)
}
// Requirement 1: raw result metadata preserved, whole parse intact.
var sawRawResult bool
for _, msg := range run.messages {
if msg.Type == MessageToolResult && strings.Contains(msg.Output, "isBackground") {
sawRawResult = true
}
}
if !sawRawResult {
t.Errorf("malformed tool result was not forwarded unchanged: %+v", run.messages)
}
}
// TestCursorBackgroundShellTerminalResultBeatsObservation is Test F: once a
// valid terminal result arrives, Cursor's native protocol owns the outcome. A
// latched background observation must not override it during finalization.
func TestCursorBackgroundShellTerminalResultBeatsObservation(t *testing.T) {
t.Parallel()
// The guard is deliberately longer than the run: the result must win on its
// own authority, not because nothing had time to fire.
run := runCursorFake(t, cursorFakeResultWins, 30*time.Second, 12*time.Second)
if run.result.Status != "completed" {
t.Fatalf("status = %q, want completed; error=%q", run.result.Status, run.result.Error)
}
if run.result.Output != "dev server checked and stopped" {
t.Errorf("output = %q, want the Cursor terminal result preserved", run.result.Output)
}
if run.result.DurationMs <= 0 {
t.Errorf("duration = %d, want a normal completed run", run.result.DurationMs)
}
}
// TestCursorBackgroundShellNoiseDoesNotExtendGuard covers the other half of
// requirement 4: the deadline measures *semantic* progress. A run whose child
// keeps producing stderr output, unknown event types and malformed lines is
// still stalled, and must not be allowed to hold its slot indefinitely.
func TestCursorBackgroundShellNoiseDoesNotExtendGuard(t *testing.T) {
t.Parallel()
const (
guard = 800 * time.Millisecond
execTimeout = 12 * time.Second
)
run := runCursorFake(t, cursorFakeUnverifiedActivity, guard, execTimeout)
if run.elapsed > 6*time.Second {
t.Errorf("elapsed = %s, want noisy-but-meaningless output to leave the deadline untouched", run.elapsed)
}
run.assertGuardFailure(t)
}
// TestCursorSemanticProgressEvent pins which events may refresh the deadline.
// Everything unrecognized is excluded by construction: believing it would let
// the protocol drift MUL-5231 and MUL-5434 already fixed extend a stalled run
// forever.
func TestCursorSemanticProgressEvent(t *testing.T) {
t.Parallel()
progress := [][2]string{
{"assistant", ""},
{"thinking", "delta"},
{"thinking", "completed"},
{"tool_call", "started"},
{"tool_call", "completed"},
{"tool_use", ""},
{"tool_result", ""},
{"text", ""},
{"step_finish", ""},
}
for _, evt := range progress {
if !cursorSemanticProgressEvent(evt[0], evt[1]) {
t.Errorf("(%s,%s) must count as semantic progress", evt[0], evt[1])
}
}
notProgress := [][2]string{
{"system", "init"},
{"system", "error"},
{"error", ""},
{"result", "success"},
{"thinking", "progress"}, // the non-terminal noise MUL-5231/MUL-5434 exclude
{"thinking", ""},
{"background_log", ""}, // an unhandled type is not evidence of progress
{"user", ""}, // a CLI echoing our own prompt proves nothing
{"connection", ""},
{"", ""},
}
for _, evt := range notProgress {
if cursorSemanticProgressEvent(evt[0], evt[1]) {
t.Errorf("(%s,%s) must NOT count as semantic progress", evt[0], evt[1])
}
}
}
// TestCursorBackgroundGuardFiresOnceThroughCancellationPath exercises the guard
// directly: one observation, one cancellation, a dedicated cause, and a watcher
// that is gone once Stop returns.
func TestCursorBackgroundGuardFiresOnceThroughCancellationPath(t *testing.T) {
t.Parallel()
runCtx, cancel := context.WithCancel(context.Background())
defer cancel()
g := newCursorBackgroundGuard(context.Background(), runCtx, cancel, 40*time.Millisecond)
if got := g.Cause(); got != "" {
t.Fatalf("Cause = %q before any background observation, want empty", got)
}
for i := 0; i < 20; i++ {
g.ObserveBackground()
}
if !waitForGuardFire(g, 2*time.Second) {
t.Fatal("guard never fired for a stalled background run")
}
if got := g.Cause(); got != cursorBackgroundLivenessGuardError {
t.Errorf("Cause = %q, want the dedicated background-liveness reason", got)
}
if runCtx.Err() == nil {
t.Error("the guard must end the run through the existing cancellation path")
}
stopGuardWithin(t, g, 2*time.Second)
g.Stop() // idempotent: the result path and the deferred teardown both call it
}
// TestCursorBackgroundGuardProgressExtendsDeadline proves the deadline rolls
// forward on semantic progress and only expires once the stream goes quiet.
func TestCursorBackgroundGuardProgressExtendsDeadline(t *testing.T) {
t.Parallel()
runCtx, cancel := context.WithCancel(context.Background())
defer cancel()
const guard = 200 * time.Millisecond
g := newCursorBackgroundGuard(context.Background(), runCtx, cancel, guard)
g.ObserveBackground()
// Six pulses at a third of the interval each: 600ms of supervised time,
// three times the deadline.
for i := 0; i < 6; i++ {
time.Sleep(guard / 3)
g.ObserveProgress()
}
if got := g.Cause(); got != "" {
t.Fatalf("Cause = %q while progress kept landing, want empty", got)
}
if !waitForGuardFire(g, 2*time.Second) {
t.Fatal("guard never fired once the progress stopped")
}
stopGuardWithin(t, g, 2*time.Second)
}
// TestCursorBackgroundGuardYieldsToRealCauses covers precedence: an execution
// deadline or a caller cancellation that was already in effect is not the
// guard's to claim, and the guard must not cancel on top of it.
func TestCursorBackgroundGuardYieldsToRealCauses(t *testing.T) {
t.Parallel()
t.Run("execution deadline", func(t *testing.T) {
t.Parallel()
parent := context.Background()
runCtx, cancel := context.WithTimeout(parent, 20*time.Millisecond)
defer cancel()
time.Sleep(80 * time.Millisecond) // let the deadline be the real cause
g := newCursorBackgroundGuard(parent, runCtx, cancel, 20*time.Millisecond)
g.ObserveBackground()
time.Sleep(200 * time.Millisecond)
if got := g.Cause(); got != "" {
t.Errorf("Cause = %q, want the timeout to keep its own classification", got)
}
stopGuardWithin(t, g, 2*time.Second)
})
t.Run("external cancellation", func(t *testing.T) {
t.Parallel()
parent, cancelParent := context.WithCancel(context.Background())
runCtx, cancel := runContext(parent, 0)
cancelParent() // the caller ended this run, not the guard
time.Sleep(20 * time.Millisecond)
g := newCursorBackgroundGuard(parent, runCtx, cancel, 20*time.Millisecond)
g.ObserveBackground()
time.Sleep(200 * time.Millisecond)
if got := g.Cause(); got != "" {
t.Errorf("Cause = %q, want external cancellation to keep its own classification", got)
}
if runCtx.Err() == nil {
t.Error("runCtx should already be cancelled by the parent")
}
stopGuardWithin(t, g, 2*time.Second)
})
}
// TestCursorBackgroundGuardWithoutObservationOwnsNoTimer documents that the
// overwhelmingly common foreground run pays nothing: Stop on a guard that never
// armed must still return, and there is no watcher to leak.
func TestCursorBackgroundGuardWithoutObservationOwnsNoTimer(t *testing.T) {
t.Parallel()
runCtx, cancel := context.WithCancel(context.Background())
defer cancel()
g := newCursorBackgroundGuard(context.Background(), runCtx, cancel, 20*time.Millisecond)
g.ObserveProgress() // progress before any background shell is inert
time.Sleep(100 * time.Millisecond)
if got := g.Cause(); got != "" {
t.Errorf("Cause = %q, want empty: no background shell was ever observed", got)
}
if runCtx.Err() != nil {
t.Error("an unobserved guard must not cancel the run")
}
stopGuardWithin(t, g, 2*time.Second)
}
func waitForGuardFire(g *cursorBackgroundGuard, timeout time.Duration) bool {
deadline := time.Now().Add(timeout)
for time.Now().Before(deadline) {
if g.Cause() != "" {
return true
}
time.Sleep(5 * time.Millisecond)
}
return false
}
// stopGuardWithin asserts the guard's watcher goroutine is gone, because Stop
// only returns once the monitor has closed `done`.
func stopGuardWithin(t *testing.T, g *cursorBackgroundGuard, timeout time.Duration) {
t.Helper()
stopped := make(chan struct{})
go func() {
g.Stop()
close(stopped)
}()
select {
case <-stopped:
case <-time.After(timeout):
t.Fatal("guard.Stop did not return: the background-liveness watcher outlived the run")
}
}