E1: split monolithic files — cmd/nomos (main.go → server.go + mcp.go + workers.go),
internal/mcp/tools.go → entity_tools/ops_tools/knowledge_tools/analysis_tools,
internal/httpapi/impl.go → domain files (entities, events, signals, ontology,
fleet_health, client_context, client_lifecycle, entity_mutations, query_audit).
E2: migrate raw pool.Exec queries to sqlc (entities/relationships queries + generated).
E3: unify SSH — consolidate crypto/ssh dial into actuator/client.go (+client_test).
E4/E5: add tests — db/lifecycle, checkdefaults/build, ontology/preconditions, policy/risk.
710 lines
24 KiB
Go
710 lines
24 KiB
Go
package main
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"fmt"
|
|
"log/slog"
|
|
"net/http"
|
|
"os"
|
|
"os/signal"
|
|
"strconv"
|
|
"strings"
|
|
"sync"
|
|
"syscall"
|
|
"time"
|
|
|
|
"github.com/dtoro/oikos/internal/safego"
|
|
"github.com/dtoro/oikos/internal/secrets"
|
|
"github.com/jackc/pgx/v5"
|
|
)
|
|
|
|
func main() {
|
|
if len(os.Args) < 2 {
|
|
fmt.Fprintln(os.Stderr, "usage: nomos serve")
|
|
os.Exit(1)
|
|
}
|
|
// Fast-path the Docker healthcheck BEFORE any Infisical/secrets init. The
|
|
// nomos runtime image is distroless (no shell/wget), so the container
|
|
// probes itself via `nomos healthcheck`. Secrets resolution retries
|
|
// Infisical ~4x per key when it's down (~30s), which would blow the 5s
|
|
// healthcheck timeout — so this must run first and stay trivial.
|
|
if os.Args[1] == "healthcheck" {
|
|
runHealthcheck()
|
|
return
|
|
}
|
|
mcpURL := os.Getenv("NOMOS_MCP_URL")
|
|
if mcpURL == "" {
|
|
mcpURL = "http://localhost:8090/mcp"
|
|
}
|
|
// api's combinedAuth requires a bearer token on every request (no
|
|
// dev-open bypass — plans/2026-07-12-wails-desktop-app.md 0.4); this is
|
|
// the same shared secret api validates against (OIKOS_MCP_BEARER_TOKEN).
|
|
mcpToken := os.Getenv("OIKOS_MCP_BEARER_TOKEN")
|
|
|
|
agentSlug := os.Getenv("NOMOS_AGENT_SLUG")
|
|
if agentSlug == "" {
|
|
agentSlug = "agent:nomos"
|
|
}
|
|
|
|
databaseURL := os.Getenv("DATABASE_URL")
|
|
if databaseURL == "" {
|
|
databaseURL = os.Getenv("OIKOS_DATABASE_URL")
|
|
}
|
|
|
|
// Resolve secrets from Infisical, falling back to env vars.
|
|
// The MCP token and OpenRouter key are fetched once at startup and
|
|
// injected via os.Setenv so downstream code (newAgent) picks them up
|
|
// without signature changes.
|
|
sec := secrets.NewManagerFromConfig(
|
|
os.Getenv("OIKOS_INFISICAL_SITE_URL"),
|
|
os.Getenv("OIKOS_INFISICAL_CLIENT_ID"),
|
|
os.Getenv("OIKOS_INFISICAL_CLIENT_SECRET"),
|
|
os.Getenv("OIKOS_INFISICAL_PROJECT_ID"),
|
|
os.Getenv("OIKOS_INFISICAL_ENV"),
|
|
os.Getenv("OIKOS_SECRETS_DIR"),
|
|
)
|
|
var openrouterAPIKey string
|
|
var secretsResolved int
|
|
if sec != nil {
|
|
resCtx, resCancel := context.WithTimeout(context.Background(), 10*time.Second)
|
|
if v := secrets.ResolveSecret(resCtx, sec, "mcp_bearer-token", ""); v != "" {
|
|
mcpToken = v
|
|
secretsResolved++
|
|
}
|
|
openrouterAPIKey = secrets.ResolveSecret(resCtx, sec, "openrouter_api-key", os.Getenv("OPENROUTER_API_KEY"))
|
|
if openrouterAPIKey != "" && openrouterAPIKey != os.Getenv("OPENROUTER_API_KEY") {
|
|
secretsResolved++
|
|
}
|
|
resCancel()
|
|
if secretsResolved > 0 {
|
|
slog.Info("nomos: secrets resolved from Infisical", "count", secretsResolved)
|
|
}
|
|
}
|
|
|
|
switch os.Args[1] {
|
|
case "serve":
|
|
ctx, cancel := signal.NotifyContext(context.Background(), syscall.SIGTERM, syscall.SIGINT)
|
|
defer cancel()
|
|
|
|
// One MCP client PER SESSION, not one shared client for the whole
|
|
// process — see mcpClientPool's doc comment. A dedicated client is
|
|
// created lazily on each session's first tool call.
|
|
clientPool := newMCPClientPool(mcpURL, mcpToken)
|
|
// Prove connectivity at startup the same way the old single-client
|
|
// constructor did, so a misconfigured/unreachable MCP endpoint still
|
|
// fails fast on boot instead of only on the first real chat. Doesn't
|
|
// reuse the pool (nothing to key it by yet) — just a throwaway probe.
|
|
if probe, err := newMCPClient(mcpURL, mcpToken); err != nil {
|
|
slog.Error("nomos: mcp connect", "url", mcpURL, "error", err)
|
|
os.Exit(1)
|
|
} else {
|
|
probe.close()
|
|
}
|
|
|
|
st, err := newStore(ctx, databaseURL)
|
|
if err != nil {
|
|
slog.Error("nomos: db connect", "error", err)
|
|
os.Exit(1)
|
|
}
|
|
if st != nil {
|
|
defer st.close()
|
|
}
|
|
|
|
nAgent, err := newAgent(ctx, clientPool, st, agentSlug, openrouterAPIKey)
|
|
if err != nil {
|
|
slog.Error("nomos: agent init", "error", err)
|
|
os.Exit(1)
|
|
}
|
|
|
|
// Event-driven auto-continuation: feed finished async executions back
|
|
// into the agent so an approved plan runs to completion (and recovers
|
|
// from failures) without the operator ticking it forward each step.
|
|
safego.Go("nomos:continuation-worker", func() { nAgent.runContinuationWorker(ctx) })
|
|
|
|
// Idle sweep for stalled goal-bearing tasks (fix 2+3 of
|
|
// plans/2026-07-11-task-completion-safety-net.md) — a coarser,
|
|
// slower-ticking counterpart to the continuation worker above.
|
|
safego.Go("nomos:idle-sweep-worker", func() { nAgent.runIdleSweepWorker(ctx) })
|
|
|
|
safego.Go("nomos:mcp-pool-sweeper", func() {
|
|
ticker := time.NewTicker(5 * time.Minute)
|
|
defer ticker.Stop()
|
|
for {
|
|
select {
|
|
case <-ctx.Done():
|
|
return
|
|
case <-ticker.C:
|
|
clientPool.sweep()
|
|
}
|
|
}
|
|
})
|
|
|
|
// Stale execution sweep: cancels non-terminal executions older than
|
|
// 10 minutes (orphaned by MCP timeouts — see cleanupStaleExecutions).
|
|
safego.Go("nomos:stale-execution-sweeper", func() {
|
|
ticker := time.NewTicker(5 * time.Minute)
|
|
defer ticker.Stop()
|
|
for {
|
|
select {
|
|
case <-ctx.Done():
|
|
return
|
|
case <-ticker.C:
|
|
st.cleanupStaleExecutions(ctx, 10*time.Minute)
|
|
}
|
|
}
|
|
})
|
|
|
|
mux := http.NewServeMux()
|
|
mux.HandleFunc("/healthz", func(w http.ResponseWriter, r *http.Request) {
|
|
w.WriteHeader(200)
|
|
w.Write([]byte("ok"))
|
|
})
|
|
mux.HandleFunc("/query", func(w http.ResponseWriter, r *http.Request) {
|
|
handleQuery(w, r, clientPool, agentSlug, mcpURL)
|
|
})
|
|
mux.HandleFunc("/chat", func(w http.ResponseWriter, r *http.Request) {
|
|
handleChat(w, r, nAgent, st)
|
|
})
|
|
mux.HandleFunc("/sessions", func(w http.ResponseWriter, r *http.Request) {
|
|
handleSessionsList(w, r, st)
|
|
})
|
|
mux.HandleFunc("/sessions/", func(w http.ResponseWriter, r *http.Request) {
|
|
handleSessionDetail(w, r, st, nAgent)
|
|
})
|
|
|
|
addr := os.Getenv("NOMOS_LISTEN")
|
|
if addr == "" {
|
|
addr = ":8092"
|
|
}
|
|
|
|
srv := &http.Server{Addr: addr, Handler: mux}
|
|
safego.Go("nomos:http-server", func() {
|
|
slog.Info("nomos: gateway listening", "addr", addr, "mcp", mcpURL, "db", databaseURL != "")
|
|
if err := srv.ListenAndServe(); err != http.ErrServerClosed {
|
|
slog.Error("nomos: serve", "error", err)
|
|
}
|
|
})
|
|
|
|
<-ctx.Done()
|
|
slog.Info("nomos: shutting down")
|
|
srv.Shutdown(context.Background())
|
|
clientPool.closeAll()
|
|
|
|
default:
|
|
fmt.Fprintf(os.Stderr, "unknown command: %s\n", os.Args[1])
|
|
os.Exit(1)
|
|
}
|
|
}
|
|
|
|
// runHealthcheck self-probes NOMOS_LISTEN/healthz and exits 0 on HTTP 200,
|
|
// 1 otherwise. Used by the Docker healthcheck (the distroless runtime image
|
|
// has no wget/shell). Must stay fast — call it before any secrets init.
|
|
func runHealthcheck() {
|
|
addr := os.Getenv("NOMOS_LISTEN")
|
|
if addr == "" {
|
|
addr = ":8092"
|
|
}
|
|
host := addr
|
|
if strings.HasPrefix(host, ":") {
|
|
host = "127.0.0.1" + host
|
|
}
|
|
client := &http.Client{Timeout: 3 * time.Second}
|
|
resp, err := client.Get("http://" + host + "/healthz")
|
|
if err != nil {
|
|
os.Exit(1)
|
|
}
|
|
defer resp.Body.Close()
|
|
if resp.StatusCode != http.StatusOK {
|
|
os.Exit(1)
|
|
}
|
|
}
|
|
|
|
func sseEvent(w http.ResponseWriter, flusher http.Flusher, event agentEvent) {
|
|
data, _ := json.Marshal(event)
|
|
fmt.Fprintf(w, "data: %s\n\n", data)
|
|
flusher.Flush()
|
|
}
|
|
|
|
func handleChat(w http.ResponseWriter, r *http.Request, a *agent, st *store) {
|
|
if r.Method != http.MethodPost {
|
|
http.Error(w, "method not allowed", 405)
|
|
return
|
|
}
|
|
|
|
var req struct {
|
|
SessionID string `json:"session_id"`
|
|
Message string `json:"message"`
|
|
}
|
|
if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
|
|
http.Error(w, "bad request: "+err.Error(), 400)
|
|
return
|
|
}
|
|
if req.Message == "" && req.SessionID == "" {
|
|
http.Error(w, "message is required", 400)
|
|
return
|
|
}
|
|
|
|
// Empty message with an existing session = reconnect/resume. This path is
|
|
// defensive now — the frontend (post F2) recovers a dropped SSE via the
|
|
// poller + terminal task.status clearing, and no longer POSTs empty
|
|
// messages. If a client ever does, route into resumeSession so the agent
|
|
// reports current state — but SKIP a terminal session (done/failed/
|
|
// abandoned): there's nothing to resume, and running a "report state"
|
|
// turn there is just a spare turn the operator never asked for (P2.1).
|
|
if req.Message == "" && req.SessionID != "" {
|
|
if sess, err := st.getSession(context.Background(), req.SessionID); err == nil {
|
|
switch sess.Status {
|
|
case "done", "failed", "abandoned":
|
|
slog.Info("nomos: reconnect skipped — session already terminal", "session", req.SessionID, "status", sess.Status)
|
|
w.WriteHeader(202)
|
|
return
|
|
}
|
|
}
|
|
slog.Info("nomos: reconnect", "session", req.SessionID)
|
|
safego.Go("nomos:reconnect:"+req.SessionID, func() {
|
|
base := "[System: the operator's connection was re-established. The task may have progressed in the background.]"
|
|
note := st.enrichResumeNote(context.Background(), req.SessionID, base)
|
|
a.resumeSession(context.Background(), req.SessionID, note)
|
|
})
|
|
// Return 202 so the frontend doesn't try to consume an SSE stream
|
|
// from this POST — resumeSession writes to the DB directly and
|
|
// the poller picks it up.
|
|
w.WriteHeader(202)
|
|
return
|
|
}
|
|
|
|
flusher, ok := w.(http.Flusher)
|
|
if !ok {
|
|
http.Error(w, "streaming not supported", 500)
|
|
return
|
|
}
|
|
|
|
w.Header().Set("Content-Type", "text/event-stream")
|
|
w.Header().Set("Cache-Control", "no-cache")
|
|
w.Header().Set("Connection", "keep-alive")
|
|
w.Header().Set("X-Accel-Buffering", "no") // disable proxy buffering
|
|
w.WriteHeader(200)
|
|
|
|
// All writes to w (events + the keepalive comment below) go through one
|
|
// mutex: http.ResponseWriter is NOT safe for concurrent use, and the
|
|
// keepalive ticker runs alongside the turn's event sink (plan 2026-08-03
|
|
// F3). Without this, interleaved writes corrupt the SSE stream.
|
|
var writeMu sync.Mutex
|
|
writeEvent := func(ev agentEvent) {
|
|
writeMu.Lock()
|
|
defer writeMu.Unlock()
|
|
sseEvent(w, flusher, ev)
|
|
}
|
|
|
|
ctx := r.Context()
|
|
sessionID := req.SessionID
|
|
|
|
// pctx (persistence context) is deliberately context.Background(), not
|
|
// ctx/r.Context(), for every DB write in this handler — ctx cancels the
|
|
// instant the client disconnects (Stop button, tab close, network blip),
|
|
// and a write made with an already-cancelled context fails. Before this
|
|
// fix, the assistant message was only ever saved ONCE, at the very end,
|
|
// using ctx — so a disconnect mid-turn silently lost the ENTIRE turn's
|
|
// tool-call history from the persisted transcript, even though real work
|
|
// (executions launched, knowledge written) had already happened
|
|
// server-side. The agent's own work (a.chat below) still correctly stops
|
|
// when ctx cancels — this only changes what happens to persistence.
|
|
pctx := context.Background()
|
|
|
|
if sessionID == "" {
|
|
title := truncate(req.Message, 80)
|
|
sess, err := st.createSession(pctx, title)
|
|
if err != nil {
|
|
slog.Error("nomos: create session", "error", err)
|
|
sessionID = "ephemeral"
|
|
} else {
|
|
sessionID = sess.ID
|
|
}
|
|
} else {
|
|
// P2 iteration: if the operator sends a follow-up on a session
|
|
// that already reached a terminal state (done/failed), reopen it
|
|
// so a new sub-task can be framed (set_goal → propose_plan →
|
|
// execute). reopenSession marks the prior plan's steps as
|
|
// `replaced` (proposePlan ignores those) and clears outcome/
|
|
// summary. Without this, propose_plan refuses the follow-up with
|
|
// errPlanInFlight because the prior steps are all `done`. If the
|
|
// session is still active, reopen is a no-op — the follow-up is
|
|
// just a continuation of in-flight work.
|
|
st.reopenSession(pctx, sessionID)
|
|
st.touchSession(pctx, sessionID)
|
|
}
|
|
|
|
slog.Info("nomos: chat", "session", sessionID, "message", truncate(req.Message, 100))
|
|
|
|
userMsg, _ := json.Marshal(map[string]any{"role": "user", "text": req.Message})
|
|
st.saveMessage(pctx, sessionID, "user", userMsg)
|
|
|
|
// If this task has a pending operator question, the incoming message IS the
|
|
// answer — close it so the panel clears. No separate resume needed: this
|
|
// chat turn is the resume, and the agent sees the question + answer in its
|
|
// replayed history.
|
|
if qid := st.openQuestionID(pctx, sessionID); qid != "" {
|
|
st.answerQuestion(pctx, sessionID, qid, req.Message)
|
|
}
|
|
|
|
writeEvent(agentEvent{Type: "session", Data: sessionID, SessionID: sessionID})
|
|
|
|
// F1/F2 (plan 2026-08-03): serialize turns per session. The user message is
|
|
// already persisted above, so it is never lost. Wait briefly for a finishing
|
|
// background turn; if one is still running after that, QUEUE this message
|
|
// (don't reject it) and tell the client so it shows a "queued" state. The
|
|
// in-flight turn's release drains the queue (drainQueued) and runs it as a
|
|
// real turn server-side. This never stacks concurrent turns — the gate still
|
|
// guarantees one in-flight turn per session.
|
|
const turnWait = 5 * time.Second
|
|
if !a.gate.acquire(sessionID, turnWait) {
|
|
a.queue.enqueue(sessionID, req.Message)
|
|
slog.Info("nomos: turn already active, queued operator message", "session", sessionID)
|
|
writeEvent(agentEvent{Type: "queued", Data: sessionID, SessionID: sessionID})
|
|
writeEvent(agentEvent{Type: "done", Data: map[string]any{
|
|
"session_id": sessionID,
|
|
"queued": true,
|
|
}, SessionID: sessionID})
|
|
return
|
|
}
|
|
defer func() {
|
|
a.gate.release(sessionID)
|
|
// Run any message that was queued while this turn held the gate. In a
|
|
// goroutine so the HTTP response finishes without waiting on the next
|
|
// turn; the queued turn has no SSE client of its own.
|
|
safego.Go("nomos:drain:"+sessionID, func() { a.drainQueued(context.Background(), sessionID) })
|
|
}()
|
|
|
|
// F3 (plan 2026-08-03): keep the SSE alive during long turns. A turn can
|
|
// run for many minutes (provisioning chains, deep research); the model
|
|
// often takes 20-40s between tool iterations, and with nothing flushed in
|
|
// that gap a proxy/browser idle timeout silently closes the stream. The
|
|
// client then sees streaming=false while the server keeps working — the
|
|
// "I can't tell it's working" desync. An SSE comment line (":keepalive") is
|
|
// ignored by EventSource but resets idle timers.
|
|
keepDone := make(chan struct{})
|
|
go func() {
|
|
t := time.NewTicker(12 * time.Second)
|
|
defer t.Stop()
|
|
for {
|
|
select {
|
|
case <-keepDone:
|
|
return
|
|
case <-t.C:
|
|
writeMu.Lock()
|
|
fmt.Fprintf(w, ":keepalive\n\n")
|
|
flusher.Flush()
|
|
writeMu.Unlock()
|
|
}
|
|
}
|
|
}()
|
|
// Defer the close (not a statement after runChatTurn) so the goroutine
|
|
// exits even if runChatTurn panics — net/http recovers handler panics, so
|
|
// a non-deferred close would be skipped and the ticker would keep writing
|
|
// to a dead ResponseWriter forever.
|
|
defer close(keepDone)
|
|
a.runChatTurn(pctx, ctx, sessionID, req.Message, func(ev agentEvent) {
|
|
writeEvent(ev)
|
|
})
|
|
}
|
|
|
|
func handleSessionsList(w http.ResponseWriter, r *http.Request, st *store) {
|
|
if st == nil {
|
|
w.Header().Set("Content-Type", "application/json")
|
|
json.NewEncoder(w).Encode(map[string]any{"sessions": []any{}})
|
|
return
|
|
}
|
|
|
|
if r.Method == http.MethodOptions {
|
|
return
|
|
}
|
|
|
|
// P2.8 (2026-07-20): filtering + pagination. The audit script in
|
|
// .agents/skills/session-review/SKILL.md slices `.sessions[:10]`
|
|
// client-side; "show me partial sessions touching lxc:rclone"
|
|
// required fetching the full list and filtering in JS. Push the
|
|
// filters into SQL so the audit becomes a single `curl | jq`.
|
|
// Supported query params (all optional, composable):
|
|
// ?outcome=partial|success|failure — exact match on outcome
|
|
// ?status=active|done|failed|executing — exact match on status
|
|
// ?entity_id=<uuid> — exact match on entity_id
|
|
// ?since=<RFC3339 or duration> — last_active_at >= ...
|
|
// ?blocker=<reason> — exact match on blocker
|
|
// ?limit=<int> — default 50, max 200
|
|
// ?cursor=<iso timestamp> — last_active_at < cursor (page back)
|
|
q := r.URL.Query()
|
|
limit := 50
|
|
if v := q.Get("limit"); v != "" {
|
|
if n, err := strconv.Atoi(v); err == nil && n > 0 && n <= 200 {
|
|
limit = n
|
|
}
|
|
}
|
|
sessions, err := st.listSessionsFiltered(r.Context(), listFilter{
|
|
Outcome: q.Get("outcome"),
|
|
Status: q.Get("status"),
|
|
EntityID: q.Get("entity_id"),
|
|
Blocker: q.Get("blocker"),
|
|
Since: q.Get("since"),
|
|
Cursor: q.Get("cursor"),
|
|
Limit: limit,
|
|
})
|
|
if err != nil {
|
|
http.Error(w, err.Error(), 500)
|
|
return
|
|
}
|
|
// Next-page cursor: the oldest last_active_at in this page. The next
|
|
// request passes it as ?cursor=... to get the page before it. Empty
|
|
// when the list is exhausted.
|
|
var nextCursor string
|
|
if len(sessions) > 0 {
|
|
oldest := sessions[len(sessions)-1].LastActiveAt
|
|
nextCursor = oldest.UTC().Format(time.RFC3339Nano)
|
|
if len(sessions) < limit {
|
|
nextCursor = "" // last page
|
|
}
|
|
}
|
|
w.Header().Set("Content-Type", "application/json")
|
|
json.NewEncoder(w).Encode(map[string]any{
|
|
"sessions": sessions,
|
|
"next_cursor": nextCursor,
|
|
"limit": limit,
|
|
})
|
|
}
|
|
|
|
func handleSessionDetail(w http.ResponseWriter, r *http.Request, st *store, a *agent) {
|
|
if st == nil {
|
|
http.Error(w, "not found", 404)
|
|
return
|
|
}
|
|
|
|
rest := strings.TrimPrefix(r.URL.Path, "/sessions/")
|
|
parts := strings.Split(rest, "/")
|
|
id := parts[0]
|
|
if id == "" {
|
|
http.Error(w, "session id required", 400)
|
|
return
|
|
}
|
|
|
|
// POST /sessions/{id}/questions/{qid}/answer — the operator answers a
|
|
// pinned question from the context panel; resume the agent with the answer.
|
|
if len(parts) == 4 && parts[1] == "questions" && parts[3] == "answer" {
|
|
if r.Method != http.MethodPost {
|
|
http.Error(w, "method not allowed", 405)
|
|
return
|
|
}
|
|
handleAnswerQuestion(w, r, st, a, id, parts[2])
|
|
return
|
|
}
|
|
|
|
// POST /sessions/{id}/resume — the operator asks the agent to continue.
|
|
if len(parts) == 2 && parts[1] == "resume" && r.Method == http.MethodPost {
|
|
base := "[System: the operator wants you to continue. Pick up where you left off — execute the next step of the plan, diagnose and fix any failures, or report progress if everything is done.]"
|
|
note := st.enrichResumeNote(context.Background(), id, base)
|
|
safego.Go("nomos:resume-session", func() { a.resumeSession(context.Background(), id, note) })
|
|
w.WriteHeader(202)
|
|
return
|
|
}
|
|
|
|
// GET /sessions/{id}/plan and /sessions/{id}/questions — REST hydration for
|
|
// the context panel when it first opens a task; live events carry deltas
|
|
// from there.
|
|
// GET /sessions/{id}/tool_calls — flat view of every tool call in the
|
|
// session, without the two-level message-shell nesting. The audit at
|
|
// plans/2026-07-20-session-review-ten-sessions.md P2.10 had to write
|
|
// Python to walk messages[].content.tool_calls[]; this endpoint makes
|
|
// it a single `curl | jq`.
|
|
if len(parts) == 2 && r.Method == http.MethodGet {
|
|
switch parts[1] {
|
|
case "plan":
|
|
all := r.URL.Query().Has("all") && r.URL.Query().Get("all") != "0" && r.URL.Query().Get("all") != "false"
|
|
steps, err := st.getPlanSteps(r.Context(), id, all)
|
|
if err != nil {
|
|
http.Error(w, err.Error(), 500)
|
|
return
|
|
}
|
|
w.Header().Set("Content-Type", "application/json")
|
|
json.NewEncoder(w).Encode(map[string]any{"steps": steps})
|
|
return
|
|
case "questions":
|
|
questions, err := st.getQuestions(r.Context(), id)
|
|
if err != nil {
|
|
http.Error(w, err.Error(), 500)
|
|
return
|
|
}
|
|
w.Header().Set("Content-Type", "application/json")
|
|
json.NewEncoder(w).Encode(map[string]any{"questions": questions})
|
|
return
|
|
case "tool_calls":
|
|
calls, err := st.getSessionToolCalls(r.Context(), id)
|
|
if err != nil {
|
|
http.Error(w, err.Error(), 500)
|
|
return
|
|
}
|
|
w.Header().Set("Content-Type", "application/json")
|
|
json.NewEncoder(w).Encode(map[string]any{"session_id": id, "tool_calls": calls})
|
|
return
|
|
}
|
|
}
|
|
|
|
switch r.Method {
|
|
case http.MethodDelete:
|
|
if err := st.deleteSession(r.Context(), id); err != nil {
|
|
http.Error(w, err.Error(), 500)
|
|
return
|
|
}
|
|
w.WriteHeader(204)
|
|
|
|
case http.MethodGet:
|
|
// P2.7 (2026-07-20): return BOTH session metadata and messages
|
|
// from GET /sessions/{id}. Previously this endpoint returned only
|
|
// {session_id, messages} — the operator had to merge with the
|
|
// /sessions list view to get title/goal/outcome. The eval harness
|
|
// at cmd/nomos/eval/main.go:302-303 already carries a comment
|
|
// about this leaky abstraction. The session field carries the
|
|
// full metadata: title, goal, outcome, summary, blocker,
|
|
// pending_approvals, message_count, tool_call_count, etc. The
|
|
// messages field is unchanged. Clients that only read
|
|
// `messages` keep working.
|
|
sess, err := st.getSession(r.Context(), id)
|
|
if err != nil {
|
|
if err == pgx.ErrNoRows {
|
|
http.Error(w, "session not found", 404)
|
|
return
|
|
}
|
|
http.Error(w, err.Error(), 500)
|
|
return
|
|
}
|
|
messages, err := st.getMessages(r.Context(), id)
|
|
if err != nil {
|
|
http.Error(w, err.Error(), 500)
|
|
return
|
|
}
|
|
w.Header().Set("Content-Type", "application/json")
|
|
json.NewEncoder(w).Encode(map[string]any{
|
|
"session_id": id,
|
|
"session": sess,
|
|
"messages": messages,
|
|
})
|
|
|
|
default:
|
|
http.Error(w, "method not allowed", 405)
|
|
}
|
|
}
|
|
|
|
// handleAnswerQuestion records the operator's answer to a pinned question and
|
|
// resumes the agent in the background with that answer injected. Returns 202 —
|
|
// the agent's response lands via the normal message-polling path, not this POST.
|
|
func handleAnswerQuestion(w http.ResponseWriter, r *http.Request, st *store, a *agent, sessionID, questionID string) {
|
|
var req struct {
|
|
Answer string `json:"answer"`
|
|
}
|
|
if err := json.NewDecoder(r.Body).Decode(&req); err != nil || strings.TrimSpace(req.Answer) == "" {
|
|
http.Error(w, "answer is required", 400)
|
|
return
|
|
}
|
|
prompt, _, _ := st.getQuestion(r.Context(), questionID)
|
|
if err := st.answerQuestion(r.Context(), sessionID, questionID, req.Answer); err != nil {
|
|
http.Error(w, err.Error(), 500)
|
|
return
|
|
}
|
|
if a != nil {
|
|
base := fmt.Sprintf("[System: the operator answered your question %q with: %q. "+
|
|
"Continue the task from here — do not re-ask.]", prompt, req.Answer)
|
|
note := st.enrichResumeNote(context.Background(), sessionID, base)
|
|
safego.Go("nomos:resume-session", func() { a.resumeSession(context.Background(), sessionID, note) })
|
|
}
|
|
w.WriteHeader(202)
|
|
}
|
|
|
|
func handleQuery(w http.ResponseWriter, r *http.Request, pool *mcpClientPool, agentSlug, mcpURL string) {
|
|
if r.Method != http.MethodPost {
|
|
http.Error(w, "method not allowed", 405)
|
|
return
|
|
}
|
|
|
|
var req struct {
|
|
Query string `json:"query"`
|
|
Tool string `json:"tool"`
|
|
Args map[string]any `json:"args"`
|
|
}
|
|
if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
|
|
http.Error(w, "bad request: "+err.Error(), 400)
|
|
return
|
|
}
|
|
|
|
// The structured /query endpoint is stateless/session-less — "query" is a
|
|
// fixed pool key (not a real session id) so repeated calls reuse one
|
|
// dedicated connection instead of paying a fresh MCP handshake every time,
|
|
// while still never sharing a connection with an actual chat task.
|
|
client, err := pool.get("query")
|
|
if err != nil {
|
|
http.Error(w, "mcp unavailable: "+err.Error(), 502)
|
|
return
|
|
}
|
|
|
|
start := time.Now()
|
|
|
|
if req.Tool != "" {
|
|
result, err := client.callTool(req.Tool, req.Args)
|
|
duration := time.Since(start).Milliseconds()
|
|
if err != nil {
|
|
slog.Error("nomos: query failed", "tool", req.Tool, "error", err)
|
|
w.Header().Set("Content-Type", "application/json")
|
|
json.NewEncoder(w).Encode(map[string]any{
|
|
"error": err.Error(),
|
|
"elapsed_ms": duration,
|
|
"agent_slug": agentSlug,
|
|
})
|
|
return
|
|
}
|
|
w.Header().Set("Content-Type", "application/json")
|
|
json.NewEncoder(w).Encode(map[string]any{
|
|
"result": result,
|
|
"elapsed_ms": duration,
|
|
"agent_slug": agentSlug,
|
|
"mcp_url": mcpURL,
|
|
})
|
|
return
|
|
}
|
|
|
|
if req.Query != "" {
|
|
if strings.Contains(strings.ToLower(req.Query), "what can you do") ||
|
|
strings.Contains(strings.ToLower(req.Query), "help") {
|
|
|
|
tools, err := client.listTools()
|
|
duration := time.Since(start).Milliseconds()
|
|
if err != nil {
|
|
w.Header().Set("Content-Type", "application/json")
|
|
json.NewEncoder(w).Encode(map[string]any{
|
|
"error": err.Error(),
|
|
"elapsed_ms": duration,
|
|
})
|
|
return
|
|
}
|
|
w.Header().Set("Content-Type", "application/json")
|
|
json.NewEncoder(w).Encode(map[string]any{
|
|
"message": "natural language queries belong to /chat. Use structured /query with 'tool' for direct MCP calls.",
|
|
"tools": tools,
|
|
"elapsed_ms": duration,
|
|
})
|
|
return
|
|
}
|
|
|
|
w.Header().Set("Content-Type", "application/json")
|
|
json.NewEncoder(w).Encode(map[string]any{
|
|
"message": "natural language queries belong to /chat. Use structured /query with 'tool' for direct MCP calls.",
|
|
"elapsed_ms": time.Since(start).Milliseconds(),
|
|
})
|
|
return
|
|
}
|
|
|
|
http.Error(w, "either 'tool' or 'query' required", 400)
|
|
}
|
|
|
|
func truncate(s string, n int) string {
|
|
if len(s) <= n {
|
|
return s
|
|
}
|
|
return s[:n] + "..."
|
|
} |