fix(agent): task completion safety net — stop tasks sticking at Running
Implements fixes 1-3 of plans/2026-07-11-task-completion-safety-net.md. Confirmed live that 50/50 production sessions never reached a terminal status because the model almost never calls complete_task, even for trivial single-tool Q&A turns SOUL.md explicitly calls out as needing it. - Inline safety net (agent.go): a session that never called set_goal never framed itself as a structured task, so its first plain-text turn-end IS the task ending — auto-complete it there instead of leaving status stuck at its creation default forever. - Idle sweep (continue.go, new completion_nudges column): goal-bearing sessions that stall get one nudge, then auto-close with outcome=partial if the nudge goes unanswered, mirroring the pattern resumeSession already uses for a different stuck-session failure mode. Fix 4 (backfill of the 50 already-stuck live sessions) is deliberately separate — deferred until this is deployed and verified live, per the plan's implementation order. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
@@ -205,6 +205,16 @@ func (a *agent) chatWith(ctx context.Context, sessionID, message, systemInject s
|
|||||||
"[System: this task has been running long enough that only the most recent %d turns of its history are included above your context — earlier turns happened but aren't shown. If you need to know what was already tried or found, check search_knowledge/get_entity_knowledge (if you recorded it) rather than assuming it didn't happen.]",
|
"[System: this task has been running long enough that only the most recent %d turns of its history are included above your context — earlier turns happened but aren't shown. If you need to know what was already tried or found, check search_knowledge/get_entity_knowledge (if you recorded it) rather than assuming it didn't happen.]",
|
||||||
historyWindowSize)))
|
historyWindowSize)))
|
||||||
}
|
}
|
||||||
|
// sawSetGoal / sawCompleteTask track whether this session has EVER framed
|
||||||
|
// itself as a structured task (set_goal) or already reached a terminal
|
||||||
|
// state (complete_task) — across both replayed history and this turn's
|
||||||
|
// own tool calls (updated again below as they happen live). Used by the
|
||||||
|
// end-of-turn safety net (plans/2026-07-11-task-completion-safety-net.md,
|
||||||
|
// fix 1): most sessions are a single trivial Q&A exchange that answers in
|
||||||
|
// text and never calls either tool, leaving agent_sessions.status stuck
|
||||||
|
// at its creation-time default forever. If a session never framed itself
|
||||||
|
// as a task, its first plain-text turn-end IS the task ending.
|
||||||
|
var sawSetGoal, sawCompleteTask bool
|
||||||
var lastAssistantCalls []persistedCall
|
var lastAssistantCalls []persistedCall
|
||||||
for _, m := range history {
|
for _, m := range history {
|
||||||
text := extractText(m.Content)
|
text := extractText(m.Content)
|
||||||
@@ -216,6 +226,12 @@ func (a *agent) chatWith(ctx context.Context, sessionID, message, systemInject s
|
|||||||
messages = append(messages, assistantToolCallMessage(calls))
|
messages = append(messages, assistantToolCallMessage(calls))
|
||||||
for _, c := range calls {
|
for _, c := range calls {
|
||||||
messages = append(messages, openai.ToolMessage(c.resultText(), c.id))
|
messages = append(messages, openai.ToolMessage(c.resultText(), c.id))
|
||||||
|
switch c.name {
|
||||||
|
case "set_goal":
|
||||||
|
sawSetGoal = true
|
||||||
|
case "complete_task":
|
||||||
|
sawCompleteTask = true
|
||||||
|
}
|
||||||
}
|
}
|
||||||
lastAssistantCalls = calls
|
lastAssistantCalls = calls
|
||||||
}
|
}
|
||||||
@@ -358,6 +374,9 @@ func (a *agent) chatWith(ctx context.Context, sessionID, message, systemInject s
|
|||||||
|
|
||||||
if len(msg.ToolCalls) == 0 {
|
if len(msg.ToolCalls) == 0 {
|
||||||
emit(agentEvent{Type: "text", Data: msg.Content, SessionID: sessionID})
|
emit(agentEvent{Type: "text", Data: msg.Content, SessionID: sessionID})
|
||||||
|
if !sawSetGoal && !sawCompleteTask {
|
||||||
|
a.autoCompleteTrivialTask(ctx, sessionID, msg.Content)
|
||||||
|
}
|
||||||
emit(agentEvent{Type: "done", Data: map[string]any{
|
emit(agentEvent{Type: "done", Data: map[string]any{
|
||||||
"session_id": sessionID,
|
"session_id": sessionID,
|
||||||
"usage": acc.Usage,
|
"usage": acc.Usage,
|
||||||
@@ -377,6 +396,13 @@ func (a *agent) chatWith(ctx context.Context, sessionID, message, systemInject s
|
|||||||
args = map[string]any{}
|
args = map[string]any{}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
switch tc.Function.Name {
|
||||||
|
case "set_goal":
|
||||||
|
sawSetGoal = true
|
||||||
|
case "complete_task":
|
||||||
|
sawCompleteTask = true
|
||||||
|
}
|
||||||
|
|
||||||
emit(agentEvent{
|
emit(agentEvent{
|
||||||
Type: "tool_use",
|
Type: "tool_use",
|
||||||
Data: map[string]any{"name": tc.Function.Name, "args": args, "id": tc.ID},
|
Data: map[string]any{"name": tc.Function.Name, "args": args, "id": tc.ID},
|
||||||
|
|||||||
@@ -32,6 +32,71 @@ func extractExecutionIDs(toolResult string) []uuid.UUID {
|
|||||||
return out
|
return out
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// idleTaskThreshold is how long a goal-bearing session can sit non-terminal
|
||||||
|
// with no activity before the idle sweep nudges it, per
|
||||||
|
// plans/2026-07-11-task-completion-safety-net.md. Arbitrary starting point,
|
||||||
|
// not measured against real task durations — long enough that it won't fire
|
||||||
|
// mid-turn, short enough the board doesn't lie for hours.
|
||||||
|
const idleTaskThreshold = 15 * time.Minute
|
||||||
|
|
||||||
|
// runIdleSweepWorker is the safety net for case 2 of
|
||||||
|
// plans/2026-07-11-task-completion-safety-net.md: sessions that called
|
||||||
|
// set_goal (so the inline safety net in agent.go correctly left them alone,
|
||||||
|
// since they framed themselves as a real task) but then stalled without
|
||||||
|
// ever calling complete_task. Coarser than runContinuationWorker's 4s tick
|
||||||
|
// since "gone idle" is a much slower signal than "an execution just
|
||||||
|
// finished." Blocks until ctx is cancelled.
|
||||||
|
func (a *agent) runIdleSweepWorker(ctx context.Context) {
|
||||||
|
if a.store == nil {
|
||||||
|
slog.Warn("nomos: idle sweep worker disabled (no store)")
|
||||||
|
return
|
||||||
|
}
|
||||||
|
slog.Info("nomos: idle sweep worker started")
|
||||||
|
ticker := time.NewTicker(5 * time.Minute)
|
||||||
|
defer ticker.Stop()
|
||||||
|
for {
|
||||||
|
select {
|
||||||
|
case <-ctx.Done():
|
||||||
|
return
|
||||||
|
case <-ticker.C:
|
||||||
|
a.processIdleSweep(ctx)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// processIdleSweep nudges a stalled goal-bearing session once; if it's still
|
||||||
|
// non-terminal on the NEXT sweep (meaning the nudge itself went unanswered,
|
||||||
|
// not just that the model is still working), auto-closes it with a
|
||||||
|
// visible "auto-closed" outcome instead of leaving it stuck forever — same
|
||||||
|
// reasoning resumeSession already applies below for a different failure
|
||||||
|
// mode (a resume that produces no response at all).
|
||||||
|
func (a *agent) processIdleSweep(ctx context.Context) {
|
||||||
|
stale := a.store.staleGoalSessions(ctx, idleTaskThreshold, 5)
|
||||||
|
for _, s := range stale {
|
||||||
|
s := s
|
||||||
|
if s.CompletionNudges == 0 {
|
||||||
|
safego.Go("nomos:idle-nudge:"+s.ID, func() {
|
||||||
|
if err := a.store.bumpCompletionNudge(ctx, s.ID); err != nil {
|
||||||
|
slog.Error("nomos: idle nudge bump failed", "session", s.ID, "error", err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
note := fmt.Sprintf("[System: this task ('%s') has been idle for %s with no complete_task call. "+
|
||||||
|
"If the goal is done (or can't be completed), call complete_task now with the outcome and a "+
|
||||||
|
"one-line summary. If you're still genuinely working through the plan, ignore this and continue.]",
|
||||||
|
s.Goal, idleTaskThreshold)
|
||||||
|
a.resumeSession(ctx, s.ID, note)
|
||||||
|
})
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
safego.Go("nomos:idle-autoclose:"+s.ID, func() {
|
||||||
|
summary := fmt.Sprintf("Auto-closed after %s idle with no response to a completion nudge.", idleTaskThreshold)
|
||||||
|
if err := a.store.completeTask(ctx, s.ID, "partial", summary); err != nil {
|
||||||
|
slog.Error("nomos: idle auto-close failed", "session", s.ID, "error", err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// runContinuationWorker is the event loop that replaces the human typing
|
// runContinuationWorker is the event loop that replaces the human typing
|
||||||
// "continue". It polls for gated executions that (a) were initiated by a chat
|
// "continue". It polls for gated executions that (a) were initiated by a chat
|
||||||
// session and (b) have just finished, and — while that agent has an open assent
|
// session and (b) have just finished, and — while that agent has an open assent
|
||||||
|
|||||||
@@ -80,6 +80,11 @@ func main() {
|
|||||||
// from failures) without the operator ticking it forward each step.
|
// from failures) without the operator ticking it forward each step.
|
||||||
safego.Go("nomos:continuation-worker", func() { nAgent.runContinuationWorker(ctx) })
|
safego.Go("nomos:continuation-worker", func() { nAgent.runContinuationWorker(ctx) })
|
||||||
|
|
||||||
|
// Idle sweep for stalled goal-bearing tasks (fix 2+3 of
|
||||||
|
// plans/2026-07-11-task-completion-safety-net.md) — a coarser,
|
||||||
|
// slower-ticking counterpart to the continuation worker above.
|
||||||
|
safego.Go("nomos:idle-sweep-worker", func() { nAgent.runIdleSweepWorker(ctx) })
|
||||||
|
|
||||||
safego.Go("nomos:mcp-pool-sweeper", func() {
|
safego.Go("nomos:mcp-pool-sweeper", func() {
|
||||||
ticker := time.NewTicker(5 * time.Minute)
|
ticker := time.NewTicker(5 * time.Minute)
|
||||||
defer ticker.Stop()
|
defer ticker.Stop()
|
||||||
|
|||||||
@@ -508,6 +508,60 @@ func (s *store) completeTask(ctx context.Context, sessionID, outcome, summary st
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// staleGoalSession is a goal-bearing task that's gone idle without reaching
|
||||||
|
// a terminal state — the idle-sweep worker's work list (fix 2+3 of
|
||||||
|
// plans/2026-07-11-task-completion-safety-net.md).
|
||||||
|
type staleGoalSession struct {
|
||||||
|
ID string
|
||||||
|
Goal string
|
||||||
|
CompletionNudges int
|
||||||
|
}
|
||||||
|
|
||||||
|
// staleGoalSessions finds sessions that framed themselves as a real task
|
||||||
|
// (goal != '', so the inline safety net in agent.go intentionally left them
|
||||||
|
// alone) but have sat non-terminal past idleThreshold. completion_nudges
|
||||||
|
// tells the caller whether to nudge (0) or give up and auto-close (>=1) —
|
||||||
|
// see processIdleSweep in continue.go.
|
||||||
|
func (s *store) staleGoalSessions(ctx context.Context, idleThreshold time.Duration, limit int) []staleGoalSession {
|
||||||
|
if s == nil {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
rows, err := s.pool.Query(ctx, `
|
||||||
|
SELECT id, goal, completion_nudges
|
||||||
|
FROM agent_sessions
|
||||||
|
WHERE goal <> ''
|
||||||
|
AND status IN ('active', 'planning', 'executing')
|
||||||
|
AND last_active_at < now() - ($1 * interval '1 second')
|
||||||
|
ORDER BY last_active_at
|
||||||
|
LIMIT $2`, idleThreshold.Seconds(), limit)
|
||||||
|
if err != nil {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
defer rows.Close()
|
||||||
|
var out []staleGoalSession
|
||||||
|
for rows.Next() {
|
||||||
|
var s staleGoalSession
|
||||||
|
if err := rows.Scan(&s.ID, &s.Goal, &s.CompletionNudges); err == nil {
|
||||||
|
out = append(out, s)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
// bumpCompletionNudge records that the idle sweep nudged a stalled session,
|
||||||
|
// stamping last_active_at so it isn't picked up again until it's genuinely
|
||||||
|
// idle again (a fresh nudge shouldn't fire every tick while the model is
|
||||||
|
// mid-response to the previous one).
|
||||||
|
func (s *store) bumpCompletionNudge(ctx context.Context, sessionID string) error {
|
||||||
|
if s == nil {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
_, err := s.pool.Exec(ctx, `
|
||||||
|
UPDATE agent_sessions SET completion_nudges = completion_nudges + 1, last_active_at = now()
|
||||||
|
WHERE id = $1`, sessionID)
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
|
||||||
// planStep is a persisted plan step, as returned to the frontend for hydration
|
// planStep is a persisted plan step, as returned to the frontend for hydration
|
||||||
// (the panel otherwise only sees steps live via plan.proposed/plan.step.*).
|
// (the panel otherwise only sees steps live via plan.proposed/plan.step.*).
|
||||||
type planStep struct {
|
type planStep struct {
|
||||||
|
|||||||
@@ -266,3 +266,28 @@ func (a *agent) handleTaskTool(ctx context.Context, sessionID, name string, args
|
|||||||
return nil, false
|
return nil, false
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// autoCompleteTrivialTask is the case-1 fix from
|
||||||
|
// plans/2026-07-11-task-completion-safety-net.md: a session that never
|
||||||
|
// called set_goal never framed itself as a structured task, so a turn that
|
||||||
|
// ends with a plain-text answer and no further tool calls IS the task
|
||||||
|
// ending — but the model consistently skips complete_task for exactly this
|
||||||
|
// case (confirmed live: 43/50 production sessions were a single trivial
|
||||||
|
// Q&A exchange, none of which ever reached a terminal status). Rather than
|
||||||
|
// leave agent_sessions.status stuck at its creation-time default forever,
|
||||||
|
// close it out mechanically here: no judgment call needed, since SOUL.md
|
||||||
|
// already treats a one-shot answered question as done by definition.
|
||||||
|
func (a *agent) autoCompleteTrivialTask(ctx context.Context, sessionID, responseText string) {
|
||||||
|
summary := strings.TrimSpace(responseText)
|
||||||
|
summary = strings.SplitN(summary, "\n", 2)[0] // first line only — the board shows one line
|
||||||
|
const maxLen = 120
|
||||||
|
if len(summary) > maxLen {
|
||||||
|
summary = summary[:maxLen] + "…"
|
||||||
|
}
|
||||||
|
if summary == "" {
|
||||||
|
summary = "Answered without further action needed."
|
||||||
|
}
|
||||||
|
if err := a.store.completeTask(ctx, sessionID, "success", summary); err != nil {
|
||||||
|
slog.Error("nomos: auto-complete trivial task failed", "session", sessionID, "error", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
10
migrations/019_task_completion_nudges.up.sql
Normal file
10
migrations/019_task_completion_nudges.up.sql
Normal file
@@ -0,0 +1,10 @@
|
|||||||
|
-- 019_task_completion_nudges.up.sql
|
||||||
|
-- See plans/2026-07-11-task-completion-safety-net.md (fix 2+3): a
|
||||||
|
-- goal-bearing session (set_goal was called, so it's a real structured
|
||||||
|
-- task, not the trivial-Q&A case handled by the inline safety net) can
|
||||||
|
-- still stall without ever calling complete_task. completion_nudges tracks
|
||||||
|
-- how many times the idle sweep has already nudged a stalled session, so it
|
||||||
|
-- can tell "never nudged" (nudge it) from "nudged once already, still
|
||||||
|
-- stuck" (auto-close it) rather than nudging forever.
|
||||||
|
|
||||||
|
ALTER TABLE agent_sessions ADD COLUMN IF NOT EXISTS completion_nudges INT NOT NULL DEFAULT 0;
|
||||||
Reference in New Issue
Block a user