package main import ( "context" "errors" "fmt" "log/slog" "strings" ) // Task tools are nomos-LOCAL, not MCP tools. They are session-scoped, and the // shared MCP server (api:8090/mcp) has no session id — so these are handled // in-process by nomos, which knows the session/task and holds the store. // buildTools appends these to the model's tool list; the agent loop routes a // call whose name isTaskTool to handleTaskTool instead of the MCP client. // // Phase 3 ships complete_task; set_goal / propose_plan / update_plan_step / // ask_operator land in later phases through the same mechanism. func taskToolDefs() []toolDef { return []toolDef{ { Name: "set_goal", Description: "State the goal of this task in one sentence, as early as you " + "can. This is what the task is trying to achieve (e.g. 'Deploy TypeType " + "as an LXC on strong'); it heads the task on the board and the context " + "panel. Call it once you understand what the operator wants.", InputSchema: map[string]any{ "type": "object", "properties": map[string]any{ "goal": map[string]any{"type": "string", "description": "The task's goal, one sentence."}, }, "required": []string{"goal"}, }, }, { Name: "propose_plan", Description: "Propose the full ordered plan for this task. Call ONCE, before any " + "execution, with EVERY step end-to-end (not one step at a time). FIRST step: " + "research (prior knowledge, relations, blast radius). If your plan runs `run` " + "against any target, include a LAST step: write back " + "(update_entity_attributes + create_relationship + upsert_knowledge) — if you " + "omit it, one is auto-appended. After this call: STOP and wait for operator " + "approval (approval vocabulary: approved, yes, go, proceed, continue, ok, " + "go ahead). Once a step has started (running/done/...), this tool REFUSES " + "further calls — advance with update_plan_step + run instead. Re-propose only " + "if the operator explicitly asks you to revise the whole plan. complete_task " + "with outcome=success is REFUSED if you ran `run` but didn't call " + "update_entity_attributes/create_relationship — write back before completing.", InputSchema: map[string]any{ "type": "object", "properties": map[string]any{ "steps": map[string]any{ "type": "array", "description": "Ordered steps, first to last.", "items": map[string]any{ "type": "object", "properties": map[string]any{ "title": map[string]any{"type": "string", "description": "Short imperative step title (e.g. 'Create the LXC')."}, "detail": map[string]any{"type": "string", "description": "Optional one-line detail."}, "target_slug": map[string]any{"type": "string", "description": "Optional entity slug this step acts on (e.g. lxc:typetype)."}, }, "required": []string{"title"}, }, }, }, "required": []string{"steps"}, }, }, { Name: "update_plan_step", Description: "Advance a plan step as you work it. Set status to 'running' when " + "you start it (pass execution_id if the step queued a gated action, so " + "the board can auto-close it when that finishes), then 'done' / 'failed' " + "/ 'skipped' / 'blocked' when it resolves. Keeps the operator's progress " + "view honest.", InputSchema: map[string]any{ "type": "object", "properties": map[string]any{ "seq": map[string]any{"type": "integer", "description": "1-based step number from propose_plan."}, "status": map[string]any{"type": "string", "enum": []string{"running", "done", "failed", "skipped", "blocked"}, "description": "New status for the step."}, "execution_id": map[string]any{"type": "string", "description": "Optional execution UUID this step is running, so it auto-closes on completion."}, }, "required": []string{"seq", "status"}, }, }, { Name: "ask_operator", Description: "Ask the operator a question when you hit a real decision only " + "they can make — an ambiguous target, a trade-off, missing information, " + "or a destructive choice not already approved. This pins a structured " + "question card in the context panel (with your options and the entities " + "involved) and PAUSES the task until they answer; their answer resumes " + "you automatically. Do NOT use it for things you can determine yourself " + "with tools — only for genuine decisions.", InputSchema: map[string]any{ "type": "object", "properties": map[string]any{ "prompt": map[string]any{"type": "string", "description": "The question, stated plainly."}, "why": map[string]any{"type": "string", "description": "Why you're asking / what's at stake."}, "options": map[string]any{ "type": "array", "items": map[string]any{"type": "string"}, "description": "The choices, if it's a pick-one decision.", }, "context_entities": map[string]any{ "type": "array", "items": map[string]any{"type": "string"}, "description": "Entity slugs relevant to the decision (shown as chips).", }, }, "required": []string{"prompt"}, }, }, { Name: "complete_task", Description: "Mark the current task finished. Call this once the goal is " + "verified done — or when you've genuinely failed or only partially " + "succeeded. Sets the task's outcome and a one-line summary shown on the " + "task board. Record what you learned with upsert_knowledge BEFORE " + "completing, so future tasks on the same entities benefit.", InputSchema: map[string]any{ "type": "object", "properties": map[string]any{ "outcome": map[string]any{ "type": "string", "enum": []string{"success", "failure", "partial"}, "description": "Did the task achieve its goal?", }, "summary": map[string]any{ "type": "string", "description": "One line describing the result (shown on the task card).", }, }, "required": []string{"outcome", "summary"}, }, }, } } // toInt coerces a JSON tool-arg number (float64 after unmarshal) to int. func toInt(v any) int { switch n := v.(type) { case float64: return int(n) case int: return n default: return 0 } } // toStringSlice coerces a JSON tool-arg array to a non-empty []string. func toStringSlice(v any) []string { arr, ok := v.([]any) if !ok { return nil } out := make([]string, 0, len(arr)) for _, e := range arr { if s, ok := e.(string); ok && strings.TrimSpace(s) != "" { out = append(out, s) } } return out } // handleTaskTool executes a nomos-local task tool. Returns (result, true) if it // handled the call, or (nil, false) if name is not a local task tool (so the // caller forwards it to the MCP client). func (a *agent) handleTaskTool(ctx context.Context, sessionID, name string, args map[string]any) (any, bool) { switch name { case "set_goal": goal, _ := args["goal"].(string) if strings.TrimSpace(goal) == "" { return "error: set_goal needs a goal", true } if err := a.store.setGoal(ctx, sessionID, goal); err != nil { return fmt.Sprintf("error setting goal: %v", err), true } // P1: the plan window is NOT opened here. Opening it on set_goal // meant any config_mutation `run` auto-executed with zero operator // approval, before a plan was even proposed (let alone approved) — // a safety regression confirmed live in session d0d562e0. The // window is now opened only when the operator approves a plan // (chat-assent grant or explicit approval in agent.go), which is // what the SOUL.md "approve the plan, not each step" model actually // describes. set_goal records the goal + flips status to executing // and nothing more. return "Goal set: " + goal + ". NEXT: pre-plan with read-only tools (search_knowledge, get_entity, list_lxcs, get_relations), then propose_plan (mandatory — even read-only tasks need a one-step plan; the run handler refuses without one). After propose_plan: if all steps are read-only, execute immediately (no approval needed). If any step is config_mutation/destructive, stop and wait for operator approval.", true case "propose_plan": raw, _ := args["steps"].([]any) var steps []planStepInput for _, r := range raw { m, ok := r.(map[string]any) if !ok { continue } title, _ := m["title"].(string) if strings.TrimSpace(title) == "" { continue } detail, _ := m["detail"].(string) target, _ := m["target_slug"].(string) steps = append(steps, planStepInput{Title: title, Detail: detail, TargetSlug: target}) } if len(steps) == 0 { return "error: propose_plan needs at least one step with a title", true } // D.2: auto-append a writeback step if the agent didn't include one. // The agent consistently writes vague last steps ("record findings") // and then skips update_entity_attributes entirely (the #1 cause of // knowledge-graph drift). Appending an explicit writeback step makes // the seq-order enforcement (5.6) require it to be completed last, // and D.1's complete_task gate enforces the actual calls. Together // they close the loop structurally — neither relies on the agent // reading SOUL.md. hasWritebackStep := false for _, st := range steps { if strings.Contains(st.Title, "update_entity_attributes") || strings.Contains(st.Title, "create_relationship") || strings.Contains(st.Detail, "update_entity_attributes") || strings.Contains(st.Detail, "create_relationship") { hasWritebackStep = true break } } appendedNote := "" if !hasWritebackStep { steps = append(steps, planStepInput{ Title: "Write back: update_entity_attributes + create_relationship + upsert_knowledge", Detail: "Call update_entity_attributes for every entity you ran against (versions, states, counts, timestamps). Call create_relationship for any edge you discovered. Then upsert_knowledge about the affected entities (pass `about` as an array).", }) appendedNote = fmt.Sprintf(" (appended a writeback step — your plan didn't include one; step %d)", len(steps)) } persisted, err := a.store.proposePlan(ctx, sessionID, steps) if err != nil { if errors.Is(err, errPlanInFlight) { // The plan is already in flight — refuse the re-proposal. // The agent must advance the existing plan with // update_plan_step + run. This is the structural fix for // the "plan added twice" sidebar drift the operator // reported: instead of appending (which duplicated) or // wiping (which lost progress), we refuse and direct. return "Plan already in flight — refusing duplicate proposal. Steps exist and at least one has started (running/done/...). To advance: call update_plan_step(seq=K, status=\"running\") then run(...) for step K's target, then update_plan_step(seq=K, status=\"done\"). Do not call propose_plan again. Re-propose only if the operator explicitly asks you to revise the whole plan (the session is reopened on a follow-up — prior steps are marked `replaced` and a fresh generation is started), and say so in your reply before calling it.", true } return fmt.Sprintf("error proposing plan: %v", err), true } // The writeback step is now always present (D.2 auto-appends it if // the agent forgot), so the old advisory nudge is replaced by the // structural gate: D.1 refuses complete_task without the actual // update_entity_attributes/create_relationship calls. result := fmt.Sprintf("Plan set (%d steps)%s. If all steps are read-only, execute now — call update_plan_step(running) + run for each step, no approval needed. If any step is config_mutation/destructive, STOP and wait for operator approval (\"approved\", \"yes\", \"go\", \"proceed\", \"continue\", \"ok\", \"go ahead\"). Do not call propose_plan again.", len(persisted), appendedNote) return result, true case "update_plan_step": seq := toInt(args["seq"]) status, _ := args["status"].(string) execID, _ := args["execution_id"].(string) if seq <= 0 || status == "" { return "error: update_plan_step needs seq (>=1) and status", true } if err := a.store.updatePlanStep(ctx, sessionID, seq, status, execID); err != nil { return fmt.Sprintf("error updating step %d: %v", seq, err), true } return fmt.Sprintf("Step %d → %s. (Advance with update_plan_step + run; do not re-propose.)", seq, status), true case "ask_operator": prompt, _ := args["prompt"].(string) if strings.TrimSpace(prompt) == "" { return "error: ask_operator needs a prompt", true } qctx := map[string]any{} if why, _ := args["why"].(string); strings.TrimSpace(why) != "" { qctx["why"] = why } if opts := toStringSlice(args["options"]); len(opts) > 0 { qctx["options"] = opts } if ents := toStringSlice(args["context_entities"]); len(ents) > 0 { qctx["entities"] = ents } if _, err := a.store.askOperator(ctx, sessionID, prompt, qctx); err != nil { return fmt.Sprintf("error posting question: %v", err), true } return "Question posted to the operator; the task is paused until they answer. " + "Do not continue or call more tools — end your turn now and wait for their answer.", true case "complete_task": outcome, _ := args["outcome"].(string) summary, _ := args["summary"].(string) switch outcome { case "": outcome = "success" // no outcome given at all — assume success, the common case case "success", "failure", "partial": // valid, use as-is default: // The tool schema declares an enum, but a weaker model (or a // typo) can still send anything — an unrecognized value used to // persist as-is, silently, with only "failure" special-cased // (store.completeTask derives status='failed' from it; anything // else became status='done' regardless of what the value // actually said). Default to "partial" rather than silently // treating an unrecognized value as "success" — safer to // under-claim than over-claim a task's outcome. slog.Warn("nomos: complete_task got an unrecognized outcome, defaulting to partial", "session", sessionID, "outcome", outcome) outcome = "partial" } // D.1: refuse success when discovery ran but no writeback followed. // The prior advisory warning (below) was ignorable — the agent // saw it and ended the task anyway. This gate fires BEFORE // completeTask runs, so the session stays in 'executing' state // and the agent must call update_entity_attributes/create_relationship // then retry complete_task. Only blocks `success`; an explicit // `failure` or `partial` is allowed through (the agent is // acknowledging it didn't finish — no reason to force writeback). if outcome == "success" && a.store.hadDiscovery(ctx, sessionID) && !a.store.hadEntityWriteback(ctx, sessionID) { return "Refused: this session ran `run` against live targets (discovery) but did not call update_entity_attributes or create_relationship to persist what you learned. The knowledge graph will drift if you complete without writeback. Call update_entity_attributes for each entity you ran against (versions, states, counts, timestamps), and create_relationship for any edge you discovered, then call complete_task again. Outcome is held at 'executing' until you do.", true } if err := a.store.completeTask(ctx, sessionID, outcome, summary); err != nil { if errors.Is(err, errTaskAlreadyComplete) { return "Task is already complete. Do not call complete_task again. If the operator pointed out a UI/sidebar inconsistency, fix it with update_plan_step (reconcile step states) or summarize the panel in your reply — do not re-execute the work.", true } return fmt.Sprintf("error completing task: %v", err), true } result := fmt.Sprintf("Task marked %s: %s", outcome, summary) if !a.store.hadEntityWriteback(ctx, sessionID) { result += "\n\n⚠️ No entity attributes or relationships were updated in this session. Call update_entity_attributes and create_relationship to persist what you learned about entities before the next session starts from scratch." } return result, true default: return nil, false } } // autoCompleteTrivialTask is the case-1 fix from // plans/2026-07-11-task-completion-safety-net.md: a session that never // called set_goal never framed itself as a structured task, so a turn that // ends with a plain-text answer and no further tool calls IS the task // ending — but the model consistently skips complete_task for exactly this // case (confirmed live: 43/50 production sessions were a single trivial // Q&A exchange, none of which ever reached a terminal status). Rather than // leave agent_sessions.status stuck at its creation-time default forever, // close it out mechanically here: no judgment call needed, since SOUL.md // already treats a one-shot answered question as done by definition. func (a *agent) autoCompleteTrivialTask(ctx context.Context, sessionID, responseText string) { summary := strings.TrimSpace(responseText) summary = strings.SplitN(summary, "\n", 2)[0] // first line only — the board shows one line const maxLen = 120 if len(summary) > maxLen { summary = summary[:maxLen] + "…" } if summary == "" { summary = "Answered without further action needed." } if err := a.store.completeTask(ctx, sessionID, "success", summary); err != nil { slog.Error("nomos: auto-complete trivial task failed", "session", sessionID, "error", err) } }