package main import ( "context" "encoding/json" "fmt" "log/slog" "net/http" "os" "os/signal" "strconv" "strings" "sync" "syscall" "time" "github.com/dtoro/oikos/internal/safego" "github.com/dtoro/oikos/internal/secrets" "github.com/jackc/pgx/v5" ) func main() { if len(os.Args) < 2 { fmt.Fprintln(os.Stderr, "usage: nomos serve") os.Exit(1) } if os.Args[1] == "healthcheck" { runHealthcheck() return } mcpURL := os.Getenv("NOMOS_MCP_URL") if mcpURL == "" { mcpURL = "http://localhost:8090/mcp" } mcpToken := os.Getenv("OIKOS_MCP_BEARER_TOKEN") agentSlug := os.Getenv("NOMOS_AGENT_SLUG") if agentSlug == "" { agentSlug = "agent:nomos" } databaseURL := os.Getenv("DATABASE_URL") if databaseURL == "" { databaseURL = os.Getenv("OIKOS_DATABASE_URL") } sec := secrets.NewManagerFromConfig( os.Getenv("OIKOS_INFISICAL_SITE_URL"), os.Getenv("OIKOS_INFISICAL_CLIENT_ID"), os.Getenv("OIKOS_INFISICAL_CLIENT_SECRET"), os.Getenv("OIKOS_INFISICAL_PROJECT_ID"), os.Getenv("OIKOS_INFISICAL_ENV"), os.Getenv("OIKOS_SECRETS_DIR"), ) var openrouterAPIKey string var secretsResolved int if sec != nil { resCtx, resCancel := context.WithTimeout(context.Background(), 10*time.Second) if v := secrets.ResolveSecret(resCtx, sec, "mcp_bearer-token", ""); v != "" { mcpToken = v secretsResolved++ } openrouterAPIKey = secrets.ResolveSecret(resCtx, sec, "openrouter_api-key", os.Getenv("OPENROUTER_API_KEY")) if openrouterAPIKey != "" && openrouterAPIKey != os.Getenv("OPENROUTER_API_KEY") { secretsResolved++ } resCancel() if secretsResolved > 0 { slog.Info("nomos: secrets resolved from Infisical", "count", secretsResolved) } } switch os.Args[1] { case "serve": ctx, cancel := signal.NotifyContext(context.Background(), syscall.SIGTERM, syscall.SIGINT) defer cancel() // One MCP client PER SESSION, not one shared client for the whole // process — see mcpClientPool's doc comment. A dedicated client is // created lazily on each session's first tool call. clientPool := newMCPClientPool(mcpURL, mcpToken) // Prove connectivity at startup the same way the old single-client // constructor did, so a misconfigured/unreachable MCP endpoint still // fails fast on boot instead of only on the first real chat. Doesn't // reuse the pool (nothing to key it by yet) — just a throwaway probe. if probe, err := newMCPClient(mcpURL, mcpToken); err != nil { slog.Error("nomos: mcp connect", "url", mcpURL, "error", err) os.Exit(1) } else { probe.close() } st, err := newStore(ctx, databaseURL) if err != nil { slog.Error("nomos: db connect", "error", err) os.Exit(1) } if st != nil { defer st.close() } nAgent, err := newAgent(ctx, clientPool, st, agentSlug, openrouterAPIKey) if err != nil { slog.Error("nomos: agent init", "error", err) os.Exit(1) } // Event-driven auto-continuation: feed finished async executions back // into the agent so an approved plan runs to completion (and recovers // from failures) without the operator ticking it forward each step. safego.Go("nomos:continuation-worker", func() { nAgent.runContinuationWorker(ctx) }) // Idle sweep for stalled goal-bearing tasks (fix 2+3 of // plans/2026-07-11-task-completion-safety-net.md) — a coarser, // slower-ticking counterpart to the continuation worker above. safego.Go("nomos:idle-sweep-worker", func() { nAgent.runIdleSweepWorker(ctx) }) safego.Go("nomos:mcp-pool-sweeper", func() { ticker := time.NewTicker(5 * time.Minute) defer ticker.Stop() for { select { case <-ctx.Done(): return case <-ticker.C: clientPool.sweep() } } }) // Stale execution sweep: cancels non-terminal executions older than // 10 minutes (orphaned by MCP timeouts — see cleanupStaleExecutions). safego.Go("nomos:stale-execution-sweeper", func() { ticker := time.NewTicker(5 * time.Minute) defer ticker.Stop() for { select { case <-ctx.Done(): return case <-ticker.C: st.cleanupStaleExecutions(ctx, 10*time.Minute) } } }) mux := http.NewServeMux() mux.HandleFunc("/healthz", func(w http.ResponseWriter, r *http.Request) { w.WriteHeader(200) w.Write([]byte("ok")) }) mux.HandleFunc("/query", func(w http.ResponseWriter, r *http.Request) { handleQuery(w, r, clientPool, agentSlug, mcpURL) }) mux.HandleFunc("/chat", func(w http.ResponseWriter, r *http.Request) { handleChat(w, r, nAgent, st) }) mux.HandleFunc("/sessions", func(w http.ResponseWriter, r *http.Request) { handleSessionsList(w, r, st) }) mux.HandleFunc("/sessions/", func(w http.ResponseWriter, r *http.Request) { handleSessionDetail(w, r, st, nAgent) }) addr := os.Getenv("NOMOS_LISTEN") if addr == "" { addr = ":8092" } srv := &http.Server{Addr: addr, Handler: mux} safego.Go("nomos:http-server", func() { slog.Info("nomos: gateway listening", "addr", addr, "mcp", mcpURL, "db", databaseURL != "") if err := srv.ListenAndServe(); err != http.ErrServerClosed { slog.Error("nomos: serve", "error", err) } }) <-ctx.Done() slog.Info("nomos: shutting down") srv.Shutdown(context.Background()) clientPool.closeAll() default: fmt.Fprintf(os.Stderr, "unknown command: %s\n", os.Args[1]) os.Exit(1) } } func runHealthcheck() { addr := os.Getenv("NOMOS_LISTEN") if addr == "" { addr = ":8092" } host := addr if strings.HasPrefix(host, ":") { host = "127.0.0.1" + host } client := &http.Client{Timeout: 3 * time.Second} resp, err := client.Get("http://" + host + "/healthz") if err != nil { os.Exit(1) } defer resp.Body.Close() if resp.StatusCode != http.StatusOK { os.Exit(1) } } func sseEvent(w http.ResponseWriter, flusher http.Flusher, event agentEvent) { data, _ := json.Marshal(event) fmt.Fprintf(w, "data: %s\n\n", data) flusher.Flush() } func handleChat(w http.ResponseWriter, r *http.Request, a *agent, st *store) { if r.Method != http.MethodPost { http.Error(w, "method not allowed", 405) return } var req struct { SessionID string `json:"session_id"` Message string `json:"message"` } if err := json.NewDecoder(r.Body).Decode(&req); err != nil { http.Error(w, "bad request: "+err.Error(), 400) return } if req.Message == "" && req.SessionID == "" { http.Error(w, "message is required", 400) return } // Empty message with an existing session = reconnect/resume. This path is // defensive now — the frontend (post F2) recovers a dropped SSE via the // poller + terminal task.status clearing, and no longer POSTs empty // messages. If a client ever does, route into resumeSession so the agent // reports current state — but SKIP a terminal session (done/failed/ // abandoned): there's nothing to resume, and running a "report state" // turn there is just a spare turn the operator never asked for (P2.1). if req.Message == "" && req.SessionID != "" { if sess, err := st.getSession(context.Background(), req.SessionID); err == nil { switch sess.Status { case "done", "failed", "abandoned": slog.Info("nomos: reconnect skipped — session already terminal", "session", req.SessionID, "status", sess.Status) w.WriteHeader(202) return } } slog.Info("nomos: reconnect", "session", req.SessionID) safego.Go("nomos:reconnect:"+req.SessionID, func() { base := "[System: the operator's connection was re-established. The task may have progressed in the background.]" note := st.enrichResumeNote(context.Background(), req.SessionID, base) a.resumeSession(context.Background(), req.SessionID, note) }) // Return 202 so the frontend doesn't try to consume an SSE stream // from this POST — resumeSession writes to the DB directly and // the poller picks it up. w.WriteHeader(202) return } flusher, ok := w.(http.Flusher) if !ok { http.Error(w, "streaming not supported", 500) return } w.Header().Set("Content-Type", "text/event-stream") w.Header().Set("Cache-Control", "no-cache") w.Header().Set("Connection", "keep-alive") w.Header().Set("X-Accel-Buffering", "no") // disable proxy buffering w.WriteHeader(200) // All writes to w (events + the keepalive comment below) go through one // mutex: http.ResponseWriter is NOT safe for concurrent use, and the // keepalive ticker runs alongside the turn's event sink (plan 2026-08-03 // F3). Without this, interleaved writes corrupt the SSE stream. var writeMu sync.Mutex writeEvent := func(ev agentEvent) { writeMu.Lock() defer writeMu.Unlock() sseEvent(w, flusher, ev) } ctx := r.Context() sessionID := req.SessionID // pctx (persistence context) is deliberately context.Background(), not // ctx/r.Context(), for every DB write in this handler — ctx cancels the // instant the client disconnects (Stop button, tab close, network blip), // and a write made with an already-cancelled context fails. Before this // fix, the assistant message was only ever saved ONCE, at the very end, // using ctx — so a disconnect mid-turn silently lost the ENTIRE turn's // tool-call history from the persisted transcript, even though real work // (executions launched, knowledge written) had already happened // server-side. The agent's own work (a.chat below) still correctly stops // when ctx cancels — this only changes what happens to persistence. pctx := context.Background() if sessionID == "" { title := truncate(req.Message, 80) sess, err := st.createSession(pctx, title) if err != nil { slog.Error("nomos: create session", "error", err) sessionID = "ephemeral" } else { sessionID = sess.ID } } else { // P2 iteration: if the operator sends a follow-up on a session // that already reached a terminal state (done/failed), reopen it // so a new sub-task can be framed (set_goal → propose_plan → // execute). reopenSession marks the prior plan's steps as // `replaced` (proposePlan ignores those) and clears outcome/ // summary. Without this, propose_plan refuses the follow-up with // errPlanInFlight because the prior steps are all `done`. If the // session is still active, reopen is a no-op — the follow-up is // just a continuation of in-flight work. st.reopenSession(pctx, sessionID) st.touchSession(pctx, sessionID) } slog.Info("nomos: chat", "session", sessionID, "message", truncate(req.Message, 100)) userMsg, _ := json.Marshal(map[string]any{"role": "user", "text": req.Message}) st.saveMessage(pctx, sessionID, "user", userMsg) // If this task has a pending operator question, the incoming message IS the // answer — close it so the panel clears. No separate resume needed: this // chat turn is the resume, and the agent sees the question + answer in its // replayed history. if qid := st.openQuestionID(pctx, sessionID); qid != "" { st.answerQuestion(pctx, sessionID, qid, req.Message) } writeEvent(agentEvent{Type: "session", Data: sessionID, SessionID: sessionID}) // F1/F2 (plan 2026-08-03): serialize turns per session. The user message is // already persisted above, so it is never lost. Wait briefly for a finishing // background turn; if one is still running after that, QUEUE this message // (don't reject it) and tell the client so it shows a "queued" state. The // in-flight turn's release drains the queue (drainQueued) and runs it as a // real turn server-side. This never stacks concurrent turns — the gate still // guarantees one in-flight turn per session. const turnWait = 5 * time.Second if !a.gate.acquire(sessionID, turnWait) { a.queue.enqueue(sessionID, req.Message) slog.Info("nomos: turn already active, queued operator message", "session", sessionID) writeEvent(agentEvent{Type: "queued", Data: sessionID, SessionID: sessionID}) writeEvent(agentEvent{Type: "done", Data: map[string]any{ "session_id": sessionID, "queued": true, }, SessionID: sessionID}) return } defer func() { a.gate.release(sessionID) // Run any message that was queued while this turn held the gate. In a // goroutine so the HTTP response finishes without waiting on the next // turn; the queued turn has no SSE client of its own. safego.Go("nomos:drain:"+sessionID, func() { a.drainQueued(context.Background(), sessionID) }) }() // F3 (plan 2026-08-03): keep the SSE alive during long turns. A turn can // run for many minutes (provisioning chains, deep research); the model // often takes 20-40s between tool iterations, and with nothing flushed in // that gap a proxy/browser idle timeout silently closes the stream. The // client then sees streaming=false while the server keeps working — the // "I can't tell it's working" desync. An SSE comment line (":keepalive") is // ignored by EventSource but resets idle timers. keepDone := make(chan struct{}) go func() { t := time.NewTicker(12 * time.Second) defer t.Stop() for { select { case <-keepDone: return case <-t.C: writeMu.Lock() fmt.Fprintf(w, ":keepalive\n\n") flusher.Flush() writeMu.Unlock() } } }() // Defer the close (not a statement after runChatTurn) so the goroutine // exits even if runChatTurn panics — net/http recovers handler panics, so // a non-deferred close would be skipped and the ticker would keep writing // to a dead ResponseWriter forever. defer close(keepDone) a.runChatTurn(pctx, ctx, sessionID, req.Message, func(ev agentEvent) { writeEvent(ev) }) } func handleSessionsList(w http.ResponseWriter, r *http.Request, st *store) { if st == nil { w.Header().Set("Content-Type", "application/json") json.NewEncoder(w).Encode(map[string]any{"sessions": []any{}}) return } if r.Method == http.MethodOptions { return } // P2.8 (2026-07-20): filtering + pagination. The audit script in // .agents/skills/session-review/SKILL.md slices `.sessions[:10]` // client-side; "show me partial sessions touching lxc:rclone" // required fetching the full list and filtering in JS. Push the // filters into SQL so the audit becomes a single `curl | jq`. // Supported query params (all optional, composable): // ?outcome=partial|success|failure — exact match on outcome // ?status=active|done|failed|executing — exact match on status // ?entity_id= — exact match on entity_id // ?since= — last_active_at >= ... // ?blocker= — exact match on blocker // ?limit= — default 50, max 200 // ?cursor= — last_active_at < cursor (page back) q := r.URL.Query() limit := 50 if v := q.Get("limit"); v != "" { if n, err := strconv.Atoi(v); err == nil && n > 0 && n <= 200 { limit = n } } sessions, err := st.listSessionsFiltered(r.Context(), listFilter{ Outcome: q.Get("outcome"), Status: q.Get("status"), EntityID: q.Get("entity_id"), Blocker: q.Get("blocker"), Since: q.Get("since"), Cursor: q.Get("cursor"), Limit: limit, }) if err != nil { http.Error(w, err.Error(), 500) return } // Next-page cursor: the oldest last_active_at in this page. The next // request passes it as ?cursor=... to get the page before it. Empty // when the list is exhausted. var nextCursor string if len(sessions) > 0 { oldest := sessions[len(sessions)-1].LastActiveAt nextCursor = oldest.UTC().Format(time.RFC3339Nano) if len(sessions) < limit { nextCursor = "" // last page } } w.Header().Set("Content-Type", "application/json") json.NewEncoder(w).Encode(map[string]any{ "sessions": sessions, "next_cursor": nextCursor, "limit": limit, }) } func handleSessionDetail(w http.ResponseWriter, r *http.Request, st *store, a *agent) { if st == nil { http.Error(w, "not found", 404) return } rest := strings.TrimPrefix(r.URL.Path, "/sessions/") parts := strings.Split(rest, "/") id := parts[0] if id == "" { http.Error(w, "session id required", 400) return } // POST /sessions/{id}/questions/{qid}/answer — the operator answers a // pinned question from the context panel; resume the agent with the answer. if len(parts) == 4 && parts[1] == "questions" && parts[3] == "answer" { if r.Method != http.MethodPost { http.Error(w, "method not allowed", 405) return } handleAnswerQuestion(w, r, st, a, id, parts[2]) return } // POST /sessions/{id}/resume — the operator asks the agent to continue. if len(parts) == 2 && parts[1] == "resume" && r.Method == http.MethodPost { base := "[System: the operator wants you to continue. Pick up where you left off — execute the next step of the plan, diagnose and fix any failures, or report progress if everything is done.]" note := st.enrichResumeNote(context.Background(), id, base) safego.Go("nomos:resume-session", func() { a.resumeSession(context.Background(), id, note) }) w.WriteHeader(202) return } // GET /sessions/{id}/plan and /sessions/{id}/questions — REST hydration for // the context panel when it first opens a task; live events carry deltas // from there. // GET /sessions/{id}/tool_calls — flat view of every tool call in the // session, without the two-level message-shell nesting. The audit at // plans/2026-07-20-session-review-ten-sessions.md P2.10 had to write // Python to walk messages[].content.tool_calls[]; this endpoint makes // it a single `curl | jq`. if len(parts) == 2 && r.Method == http.MethodGet { switch parts[1] { case "plan": all := r.URL.Query().Has("all") && r.URL.Query().Get("all") != "0" && r.URL.Query().Get("all") != "false" steps, err := st.getPlanSteps(r.Context(), id, all) if err != nil { http.Error(w, err.Error(), 500) return } w.Header().Set("Content-Type", "application/json") json.NewEncoder(w).Encode(map[string]any{"steps": steps}) return case "questions": questions, err := st.getQuestions(r.Context(), id) if err != nil { http.Error(w, err.Error(), 500) return } w.Header().Set("Content-Type", "application/json") json.NewEncoder(w).Encode(map[string]any{"questions": questions}) return case "tool_calls": calls, err := st.getSessionToolCalls(r.Context(), id) if err != nil { http.Error(w, err.Error(), 500) return } w.Header().Set("Content-Type", "application/json") json.NewEncoder(w).Encode(map[string]any{"session_id": id, "tool_calls": calls}) return } } switch r.Method { case http.MethodDelete: if err := st.deleteSession(r.Context(), id); err != nil { http.Error(w, err.Error(), 500) return } w.WriteHeader(204) case http.MethodGet: // P2.7 (2026-07-20): return BOTH session metadata and messages // from GET /sessions/{id}. Previously this endpoint returned only // {session_id, messages} — the operator had to merge with the // /sessions list view to get title/goal/outcome. The eval harness // at cmd/nomos/eval/main.go:302-303 already carries a comment // about this leaky abstraction. The session field carries the // full metadata: title, goal, outcome, summary, blocker, // pending_approvals, message_count, tool_call_count, etc. The // messages field is unchanged. Clients that only read // `messages` keep working. sess, err := st.getSession(r.Context(), id) if err != nil { if err == pgx.ErrNoRows { http.Error(w, "session not found", 404) return } http.Error(w, err.Error(), 500) return } messages, err := st.getMessages(r.Context(), id) if err != nil { http.Error(w, err.Error(), 500) return } w.Header().Set("Content-Type", "application/json") json.NewEncoder(w).Encode(map[string]any{ "session_id": id, "session": sess, "messages": messages, }) default: http.Error(w, "method not allowed", 405) } } // handleAnswerQuestion records the operator's answer to a pinned question and // resumes the agent in the background with that answer injected. Returns 202 — // the agent's response lands via the normal message-polling path, not this POST. func handleAnswerQuestion(w http.ResponseWriter, r *http.Request, st *store, a *agent, sessionID, questionID string) { var req struct { Answer string `json:"answer"` } if err := json.NewDecoder(r.Body).Decode(&req); err != nil || strings.TrimSpace(req.Answer) == "" { http.Error(w, "answer is required", 400) return } prompt, _, _ := st.getQuestion(r.Context(), questionID) if err := st.answerQuestion(r.Context(), sessionID, questionID, req.Answer); err != nil { http.Error(w, err.Error(), 500) return } if a != nil { base := fmt.Sprintf("[System: the operator answered your question %q with: %q. "+ "Continue the task from here — do not re-ask.]", prompt, req.Answer) note := st.enrichResumeNote(context.Background(), sessionID, base) safego.Go("nomos:resume-session", func() { a.resumeSession(context.Background(), sessionID, note) }) } w.WriteHeader(202) } func handleQuery(w http.ResponseWriter, r *http.Request, pool *mcpClientPool, agentSlug, mcpURL string) { if r.Method != http.MethodPost { http.Error(w, "method not allowed", 405) return } var req struct { Query string `json:"query"` Tool string `json:"tool"` Args map[string]any `json:"args"` } if err := json.NewDecoder(r.Body).Decode(&req); err != nil { http.Error(w, "bad request: "+err.Error(), 400) return } // The structured /query endpoint is stateless/session-less — "query" is a // fixed pool key (not a real session id) so repeated calls reuse one // dedicated connection instead of paying a fresh MCP handshake every time, // while still never sharing a connection with an actual chat task. client, err := pool.get("query") if err != nil { http.Error(w, "mcp unavailable: "+err.Error(), 502) return } start := time.Now() if req.Tool != "" { result, err := client.callTool(req.Tool, req.Args) duration := time.Since(start).Milliseconds() if err != nil { slog.Error("nomos: query failed", "tool", req.Tool, "error", err) w.Header().Set("Content-Type", "application/json") json.NewEncoder(w).Encode(map[string]any{ "error": err.Error(), "elapsed_ms": duration, "agent_slug": agentSlug, }) return } w.Header().Set("Content-Type", "application/json") json.NewEncoder(w).Encode(map[string]any{ "result": result, "elapsed_ms": duration, "agent_slug": agentSlug, "mcp_url": mcpURL, }) return } if req.Query != "" { if strings.Contains(strings.ToLower(req.Query), "what can you do") || strings.Contains(strings.ToLower(req.Query), "help") { tools, err := client.listTools() duration := time.Since(start).Milliseconds() if err != nil { w.Header().Set("Content-Type", "application/json") json.NewEncoder(w).Encode(map[string]any{ "error": err.Error(), "elapsed_ms": duration, }) return } w.Header().Set("Content-Type", "application/json") json.NewEncoder(w).Encode(map[string]any{ "message": "natural language queries belong to /chat. Use structured /query with 'tool' for direct MCP calls.", "tools": tools, "elapsed_ms": duration, }) return } w.Header().Set("Content-Type", "application/json") json.NewEncoder(w).Encode(map[string]any{ "message": "natural language queries belong to /chat. Use structured /query with 'tool' for direct MCP calls.", "elapsed_ms": time.Since(start).Milliseconds(), }) return } http.Error(w, "either 'tool' or 'query' required", 400) } func truncate(s string, n int) string { if len(s) <= n { return s } return s[:n] + "..." }