feat(agent): plan-first gate, iterative follow-ups, reasoning persistence
P1 plan-first: run handler refuses without propose_plan (structural gate,
not SOUL.md prose). Plan window decoupled from set_goal — config_mutation
auto-run only on operator approval (assent window). Closes the approval-free
config_mutation hole confirmed in session d0d562e0.
P2 iteration: reopenSession flips terminal→executing, marks prior plan steps
replaced, clears outcome. proposePlan excludes replaced from in-flight check,
bumps generation. A follow-up on a completed session starts a new sub-task
with a fresh plan — no more errPlanInFlight dead end.
P3 reasoning: accumulate per-iteration text into the persisted row instead
of overwriting with the last text event. Reload shows intermediate thinking,
not just the final summary.
P4 read-only allowlist: add find, tree, locate, systemctl list-timers/
list-unit-files/show, timedatectl, hostnamectl, systemd-analyze, rclone
ls/lsl/md5sum/check/cryptcheck. Fixes the find misclassification from
d0d562e0.
P5 eval harness: new assertion kinds (proposes_plan, plan_before_run,
plan_generations), multi-turn followups, fetch /sessions/{id}/plan. Four
manifests under evals/.
P6 SOUL.md: strip degenerate-case carve-out, add ITERATE step, update
set_goal guidance.
VERSION 0.6.0 → 0.7.0
This commit is contained in:
@@ -125,8 +125,8 @@ func runConversation(ctx context.Context, gateway string, c conversation, timeou
|
||||
}
|
||||
|
||||
// Send followup if any.
|
||||
if c.Followup != "" {
|
||||
if _, err := sendChat(ctx, gateway, sid, c.Followup); err != nil {
|
||||
for _, fu := range c.followups() {
|
||||
if _, err := sendChat(ctx, gateway, sid, fu); err != nil {
|
||||
res.Assertions = []assertionResult{{Name: "send_followup", Passed: false, Detail: err.Error()}}
|
||||
res.Duration = time.Since(start)
|
||||
return res
|
||||
@@ -246,6 +246,19 @@ type transcript struct {
|
||||
ToolCalls []map[string]any `json:"tool_calls"`
|
||||
} `json:"content"`
|
||||
} `json:"messages"`
|
||||
// PlanSteps is fetched from /sessions/{id}/plan (P5 plan_generations
|
||||
// assertion). Each step carries a `generation` int; distinctGenerations
|
||||
// counts the unique values. nil when the endpoint returned no plan
|
||||
// (e.g. a pure-DB Q&A with no propose_plan call).
|
||||
PlanSteps []planStep `json:"steps"`
|
||||
}
|
||||
|
||||
// planStep is one step from /sessions/{id}/plan, carrying only the fields the
|
||||
// eval needs: the generation number (P2 iteration counter).
|
||||
type planStep struct {
|
||||
Generation int `json:"generation"`
|
||||
Status string `json:"status"`
|
||||
Title string `json:"title"`
|
||||
}
|
||||
|
||||
func (t transcript) toolCallCount() int {
|
||||
@@ -268,6 +281,17 @@ func (t transcript) toolNames() []string {
|
||||
return names
|
||||
}
|
||||
|
||||
// distinctGenerations counts unique plan generation values across all plan
|
||||
// steps. Used by the `plan_generations` assertion (P2 iteration). Returns 0
|
||||
// when there are no plan steps (no propose_plan was called).
|
||||
func (t transcript) distinctGenerations() int {
|
||||
seen := map[int]bool{}
|
||||
for _, s := range t.PlanSteps {
|
||||
seen[s.Generation] = true
|
||||
}
|
||||
return len(seen)
|
||||
}
|
||||
|
||||
type sessionState struct {
|
||||
ID string `json:"id"`
|
||||
Status string `json:"status"`
|
||||
@@ -277,7 +301,8 @@ type sessionState struct {
|
||||
|
||||
// fetchTranscript fetches the messages from /sessions/{id} (which returns
|
||||
// only session_id + messages) and the session metadata from /sessions
|
||||
// (which returns status/outcome/last_active_at for each session).
|
||||
// (which returns status/outcome/last_active_at for each session). P5 also
|
||||
// fetches /sessions/{id}/plan for the plan_generations assertion.
|
||||
func fetchTranscript(ctx context.Context, gateway, sid string) (transcript, sessionState, error) {
|
||||
var t transcript
|
||||
resp, err := http.Get(gateway + "/sessions/" + sid)
|
||||
@@ -292,6 +317,16 @@ func fetchTranscript(ctx context.Context, gateway, sid string) (transcript, sess
|
||||
if err := json.Unmarshal(b, &t); err != nil {
|
||||
return t, sessionState{}, err
|
||||
}
|
||||
// Fetch the plan (steps with generation numbers) for the
|
||||
// plan_generations assertion. A 404 or empty response is fine — a
|
||||
// pure-DB Q&A with no propose_plan has no plan.
|
||||
if planResp, perr := http.Get(gateway + "/sessions/" + sid + "/plan"); perr == nil {
|
||||
if planResp.StatusCode == 200 {
|
||||
pb, _ := io.ReadAll(planResp.Body)
|
||||
_ = json.Unmarshal(pb, &t) // fills t.PlanSteps via "steps" field
|
||||
}
|
||||
planResp.Body.Close()
|
||||
}
|
||||
// The detail endpoint doesn't return status/outcome — fetch from the
|
||||
// sessions list and find the matching id.
|
||||
s, err := fetchSessionMeta(ctx, gateway, sid)
|
||||
|
||||
@@ -11,10 +11,24 @@ import (
|
||||
type conversation struct {
|
||||
Name string `yaml:"name"`
|
||||
Prompt string `yaml:"prompt"`
|
||||
Followup string `yaml:"followup"`
|
||||
Followup string `yaml:"followup"` // backward compat: single followup
|
||||
Followups []string `yaml:"followups"` // P5: multi-turn followups
|
||||
Assertions []assertion `yaml:"assertions"`
|
||||
}
|
||||
|
||||
// followups returns the full list of follow-up messages, supporting both
|
||||
// the single `followup` field (backward compat) and the multi-turn
|
||||
// `followups` list.
|
||||
func (c conversation) followups() []string {
|
||||
if len(c.Followups) > 0 {
|
||||
return c.Followups
|
||||
}
|
||||
if c.Followup != "" {
|
||||
return []string{c.Followup}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// assertion is one check against the final transcript. The `kind` field
|
||||
// selects the scorer; the rest are scorer-specific parameters.
|
||||
//
|
||||
@@ -23,13 +37,15 @@ type conversation struct {
|
||||
// completes — session status reached done/failed (not stuck executing)
|
||||
// outcome_is — session outcome == value (success/failure/partial)
|
||||
// no_propose_plan — propose_plan was never called
|
||||
// proposes_plan — propose_plan called >= 1 time (plan-always model; P1)
|
||||
// proposes_plan_once — propose_plan was called exactly once
|
||||
// no_duplicate_proposal — propose_plan called at most once
|
||||
// plan_before_run — the first `run` call comes after the first `propose_plan` (P1 ordering gate)
|
||||
// plan_generations — the persisted plan has exactly `value` distinct generations (P2 iteration: 1 = single, 2 = one followup)
|
||||
// writes_back — update_entity_attributes or create_relationship was called
|
||||
// max_tool_calls — total tool calls <= value
|
||||
// max_run_calls — total `run` calls <= value
|
||||
// no_run — `run` was never called
|
||||
// no_rerun — `run` was NOT called after the followup turn (if any)
|
||||
// calls_tool — the named tool appears in the transcript
|
||||
// plan_step_count — the plan has exactly `value` steps
|
||||
// no_duplicate_complete — complete_task called at most once
|
||||
@@ -88,6 +104,14 @@ func scoreOne(a assertion, t transcript, s sessionState) (bool, string) {
|
||||
}
|
||||
return false, fmt.Sprintf("propose_plan called %d time(s)", n)
|
||||
|
||||
case "proposes_plan":
|
||||
// P1 plan-always: propose_plan called >= 1 time.
|
||||
n := countTool(tools, "propose_plan")
|
||||
if n >= 1 {
|
||||
return true, fmt.Sprintf("propose_plan called %d time(s)", n)
|
||||
}
|
||||
return false, "propose_plan never called (plan-always requires >= 1)"
|
||||
|
||||
case "proposes_plan_once":
|
||||
n := countTool(tools, "propose_plan")
|
||||
if n == 1 {
|
||||
@@ -102,6 +126,44 @@ func scoreOne(a assertion, t transcript, s sessionState) (bool, string) {
|
||||
}
|
||||
return false, fmt.Sprintf("propose_plan called %d time(s), want <= 1", n)
|
||||
|
||||
case "plan_before_run":
|
||||
// P1 ordering gate: the first `run` call's global index in the
|
||||
// transcript is strictly greater than the first `propose_plan`
|
||||
// index. Both indices are over the flat tool-call list (across all
|
||||
// messages, in order).
|
||||
planIdx, runIdx := -1, -1
|
||||
for i, name := range tools {
|
||||
if name == "propose_plan" && planIdx == -1 {
|
||||
planIdx = i
|
||||
}
|
||||
if name == "run" && runIdx == -1 {
|
||||
runIdx = i
|
||||
}
|
||||
}
|
||||
if runIdx == -1 {
|
||||
return true, "run never called (ordering trivially satisfied)"
|
||||
}
|
||||
if planIdx == -1 {
|
||||
return false, "run called but propose_plan never called"
|
||||
}
|
||||
if planIdx < runIdx {
|
||||
return true, fmt.Sprintf("propose_plan at index %d before run at index %d", planIdx, runIdx)
|
||||
}
|
||||
return false, fmt.Sprintf("run at index %d before propose_plan at index %d", runIdx, planIdx)
|
||||
|
||||
case "plan_generations":
|
||||
// P2 iteration: counts distinct `generation` values in
|
||||
// session_plan_steps. 1 = single sub-task, 2 = one follow-up
|
||||
// sub-task, etc. Requires the plan endpoint to return generation
|
||||
// values; the eval fetches /sessions/{id}/plan and passes it via
|
||||
// the transcript's PlanSteps field.
|
||||
want := toInt(a.Value)
|
||||
gens := t.distinctGenerations()
|
||||
if gens == want {
|
||||
return true, fmt.Sprintf("%d plan generation(s)", gens)
|
||||
}
|
||||
return false, fmt.Sprintf("%d plan generation(s), want %d", gens, want)
|
||||
|
||||
case "writes_back":
|
||||
n := countTool(tools, "update_entity_attributes") + countTool(tools, "create_relationship")
|
||||
if n > 0 {
|
||||
|
||||
Reference in New Issue
Block a user