Files
oikos/internal/learning/learning.go
dtoro aa197190cd phase 3 review: fix broken error classification, stub checks, wasted uuid, token idempotency, dead code
- actuator/ssh.go: custom errorsAs chain broken — all SSH errors classified as SSHErrorOther.
  Replaced with standard errors.As + errors.Is.
- scheduler/scheduler.go: all four check functions were stubs returning healthy.
  Implemented real HTTP GET, TCP dial, unix.Statfs disk, and TLS cert expiry checks.
- learning/learning.go: uuid.NewV7() called unconditionally before ON CONFLICT upsert.
  Now looks up existing pattern first, reuses entity_id.
- notifier/notifier.go: removed dead var_, fixed token regeneration every 15s.
  Now skips if token_hash already set.
- phase3.go: removed dead GetPattern+dummy args call in PatchPattern.
- classify.go: removed unused var_ guard.
2026-07-07 15:27:31 +02:00

184 lines
4.9 KiB
Go

// Package learning implements the Oikos learning engine (Phase 3).
// Hourly pattern extraction: reads feedback past the watermark, groups by
// (applies_type, action), updates pattern counters with Wilson confidence,
// detects anomalies, and refines skills.
package learning
import (
"context"
"log/slog"
"math"
"time"
"github.com/dtoro/oikos/internal/config"
"github.com/dtoro/oikos/internal/db"
"github.com/dtoro/oikos/internal/db/sqlcgen"
"github.com/google/uuid"
)
// Run starts the learning loop. Blocks until ctx is cancelled.
func Run(ctx context.Context, pool *db.Pool, cfg config.Config) {
slog.Info("learning: starting", "interval", cfg.LearningInterval)
interval := cfg.LearningInterval
if interval <= 0 {
interval = 1 * time.Hour
}
ticker := time.NewTicker(interval)
defer ticker.Stop()
watermark := time.Now().Add(-24 * time.Hour) // start from 24h ago
for {
select {
case <-ctx.Done():
slog.Info("learning: shutting down")
return
case <-ticker.C:
watermark = extractPatterns(ctx, pool, watermark)
}
}
}
// extractPatterns reads feedback past the watermark, groups by (type, action),
// updates pattern counters, and returns the new watermark.
func extractPatterns(ctx context.Context, pool *db.Pool, watermark time.Time) time.Time {
q := sqlcgen.New(pool)
feedback, err := q.GetFeedbackAfterWatermark(ctx, watermark)
if err != nil {
slog.Error("learning: get feedback", "error", err)
return watermark
}
if len(feedback) == 0 {
// Advance watermark to now so we don't re-scan
return time.Now()
}
// Group by (applies_type, action)
type groupKey struct {
Type string
Action string
}
groups := make(map[groupKey][]sqlcgen.GetFeedbackAfterWatermarkRow)
for _, f := range feedback {
key := groupKey{Type: f.AppliesType, Action: f.Action}
groups[key] = append(groups[key], f)
}
for key, items := range groups {
processGroup(ctx, pool, q, key.Type, key.Action, items)
}
// Update watermark to the latest feedback timestamp
newWatermark := watermark
for _, f := range feedback {
if f.CreatedAt.After(newWatermark) {
newWatermark = f.CreatedAt
}
}
return newWatermark
}
func processGroup(ctx context.Context, pool *db.Pool, q *sqlcgen.Queries,
appliesType, action string, items []sqlcgen.GetFeedbackAfterWatermarkRow) {
successCount := 0
failureCount := 0
for _, f := range items {
switch f.Outcome {
case "success":
successCount++
case "failure", "unexpected":
failureCount++
case "partial":
successCount++ // partial counts as half-success
}
}
total := successCount + failureCount
if total == 0 {
return
}
// Compute Wilson score lower bound
confidence := wilsonLowerBound(float64(successCount), float64(total), 0.95)
// Cap by sample size: nothing looks confident before 5 samples
confidence = math.Min(confidence, float64(total)/5.0)
// Get or create pattern — first look up existing entity, then upsert.
existing, err := q.GetPattern(ctx, sqlcgen.GetPatternParams{
AppliesType: appliesType,
Action: action,
})
var patternID uuid.UUID
if err == nil && existing.EntityID != uuid.Nil {
patternID = existing.EntityID
} else {
id, idErr := uuid.NewV7()
if idErr != nil {
slog.Error("learning: gen pattern uuid", "error", idErr)
return
}
patternID = id
}
patternSummary := action + " on " + appliesType
err = q.UpsertPattern(ctx, sqlcgen.UpsertPatternParams{
EntityID: patternID,
AppliesType: appliesType,
Action: action,
Pattern: patternSummary,
Confidence: float32(confidence),
EvidenceCount: int32(total),
SuccessCount: int32(successCount),
FailureCount: int32(failureCount),
})
if err != nil {
slog.Error("learning: upsert pattern", "error", err)
return
}
// Update pattern status based on confidence
pat, err := q.GetPattern(ctx, sqlcgen.GetPatternParams{
AppliesType: appliesType,
Action: action,
})
if err != nil {
return
}
if pat.EvidenceCount >= 5 && pat.Confidence >= 0.7 && !pat.Quarantined {
_ = q.UpdatePatternStatus(ctx, sqlcgen.UpdatePatternStatusParams{
EntityID: pat.EntityID,
Status: "validated",
})
slog.Info("learning: pattern validated",
"type", appliesType, "action", action,
"confidence", confidence, "samples", total)
}
// Anomaly check: >10 identical outcomes within 1h
if total > 10 {
_ = q.UpdatePatternQuarantine(ctx, sqlcgen.UpdatePatternQuarantineParams{
EntityID: pat.EntityID,
Quarantined: true,
})
slog.Warn("learning: pattern quarantined (anomaly burst)",
"type", appliesType, "action", action)
}
}
// wilsonLowerBound computes the Wilson score interval lower bound.
// Conservative estimate of success rate for small sample sizes.
func wilsonLowerBound(success, total, z float64) float64 {
if total == 0 {
return 0
}
p := success / total
z2 := z * z
denom := 1 + z2/total
center := (p + z2/(2*total)) / denom
sp := math.Sqrt((p*(1-p) + z2/(4*total)) / total) / denom
return math.Max(0, center-z*sp)
}