diff --git a/cmd/docgardener/channels.go b/cmd/docgardener/channels.go new file mode 100644 index 0000000..0e82ad5 --- /dev/null +++ b/cmd/docgardener/channels.go @@ -0,0 +1,73 @@ +package main + +import ( + "context" + "database/sql" + "errors" + "fmt" +) + +// dbChannelCreator implements goals.ChannelCreator without taking a +// dependency on the internal/channels service (which would drag in +// half the server). It talks to the channels and channel_members +// tables directly. This is only safe because the demo driver runs +// against the same process's DB and the channels schema is stable. +type dbChannelCreator struct { + db *sql.DB +} + +// CreateGoalChannel satisfies goals.ChannelCreator. +func (c *dbChannelCreator) CreateGoalChannel(ctx context.Context, slug, title, description, ownerUsername string) (int64, error) { + name := "goal-" + slug + id, err := c.ensureChannel(ctx, name, description, "blackboard", ownerUsername) + if err != nil { + return 0, err + } + if ownerUsername != "" { + if err := c.addMember(ctx, id, ownerUsername); err != nil { + return 0, fmt.Errorf("add owner %q to goal channel: %w", ownerUsername, err) + } + } + return id, nil +} + +// ensureChannel upserts a channel row by name. Returns the id. +func (c *dbChannelCreator) ensureChannel(ctx context.Context, name, description, channelType, createdBy string) (int64, error) { + var id int64 + err := c.db.QueryRowContext(ctx, `SELECT id FROM channels WHERE name = ?`, name).Scan(&id) + if err == nil { + return id, nil + } + if !errors.Is(err, sql.ErrNoRows) { + return 0, err + } + res, err := c.db.ExecContext(ctx, ` + INSERT INTO channels (name, description, type, is_private, is_system, created_by) + VALUES (?, ?, ?, 0, 0, ?)`, name, description, channelType, createdBy) + if err != nil { + return 0, fmt.Errorf("create channel %q: %w", name, err) + } + return res.LastInsertId() +} + +// getByName resolves a channel id by name. +func (c *dbChannelCreator) getByName(ctx context.Context, name string) (int64, error) { + var id int64 + err := c.db.QueryRowContext(ctx, `SELECT id FROM channels WHERE name = ?`, name).Scan(&id) + return id, err +} + +// addMember is idempotent — does nothing if the member is already present. +func (c *dbChannelCreator) addMember(ctx context.Context, channelID int64, agentName string) error { + var exists int + _ = c.db.QueryRowContext(ctx, + `SELECT COUNT(1) FROM channel_members WHERE channel_id=? AND agent_name=?`, + channelID, agentName).Scan(&exists) + if exists > 0 { + return nil + } + _, err := c.db.ExecContext(ctx, ` + INSERT INTO channel_members (channel_id, agent_name, role) + VALUES (?, ?, 'member')`, channelID, agentName) + return err +} diff --git a/cmd/docgardener/flow.go b/cmd/docgardener/flow.go new file mode 100644 index 0000000..3cc821e --- /dev/null +++ b/cmd/docgardener/flow.go @@ -0,0 +1,542 @@ +package main + +import ( + "context" + "crypto/rand" + "database/sql" + "encoding/hex" + "encoding/json" + "errors" + "fmt" + "log/slog" + "time" + + "golang.org/x/crypto/bcrypt" + + "github.com/synapbus/synapbus/internal/goals" + "github.com/synapbus/synapbus/internal/goaltasks" + "github.com/synapbus/synapbus/internal/trust" +) + +// flow wires the primitives for the demo. It reaches into the DB +// directly for bootstrap (users, channels, agents) because these are +// one-shot operations that would otherwise require embedding the full +// service wiring. For the domain logic (goals, tasks, trust) it uses +// the real services. +type flow struct { + db *sql.DB + goals *goals.Service + tasks *goaltasks.Service + ledger *trust.Ledger + logger *slog.Logger + channels *dbChannelCreator + // bootstrap state + ownerUserID int64 + ownerUsername string + coordinatorAgentID int64 + coordinatorHash string + approvalsChannelID int64 + conversationID int64 // shared conversation for all system/artifact messages +} + +func newFlow(db *sql.DB, logger *slog.Logger) *flow { + cc := &dbChannelCreator{db: db} + return &flow{ + db: db, + goals: goals.NewService(goals.NewStore(db), cc, logger), + tasks: goaltasks.NewService(goaltasks.NewStore(db), logger), + ledger: trust.NewLedger(db), + logger: logger, + channels: cc, + } +} + +// --- bootstrap -------------------------------------------------------- + +func (f *flow) bootstrap(ctx context.Context) error { + // owner user + if err := f.db.QueryRowContext(ctx, + `SELECT id, username FROM users WHERE username='algis'`).Scan(&f.ownerUserID, &f.ownerUsername); err != nil { + if !errors.Is(err, sql.ErrNoRows) { + return fmt.Errorf("lookup user: %w", err) + } + hash, _ := bcrypt.GenerateFromPassword([]byte("algis-demo-pw"), bcrypt.DefaultCost) + res, err := f.db.ExecContext(ctx, + `INSERT INTO users (username, password_hash) VALUES ('algis', ?)`, string(hash)) + if err != nil { + return fmt.Errorf("create user algis: %w", err) + } + f.ownerUserID, _ = res.LastInsertId() + f.ownerUsername = "algis" + f.logger.Info("user created", "username", f.ownerUsername, "id", f.ownerUserID) + } + + // approvals and requests channels (idempotent). + for _, name := range []string{"approvals", "requests"} { + if _, err := f.channels.ensureChannel(ctx, name, "Auto-approved "+name+" queue", "blackboard", f.ownerUsername); err != nil { + return fmt.Errorf("ensure channel %s: %w", name, err) + } + } + var err error + f.approvalsChannelID, err = f.channels.getByName(ctx, "approvals") + if err != nil { + return err + } + + // Coordinator agent — always exists before a run starts. + coord := coordinatorConfig() + f.coordinatorHash = trust.ConfigHash(coord.ToTrustConfig()) + coordID, err := f.ensureAgent(ctx, ensureAgentInput{ + Name: coord.Name, + DisplayName: coord.DisplayName, + OwnerID: f.ownerUserID, + SystemPrompt: coord.SystemPrompt, + ConfigHash: f.coordinatorHash, + AutonomyTier: trust.TierAssisted, + ToolScope: coord.ToolScope, + SpawnDepth: 0, + ParentAgentID: nil, + }) + if err != nil { + return fmt.Errorf("ensure coordinator: %w", err) + } + f.coordinatorAgentID = coordID + + // Seed the coordinator with some neutral evidence so spawned children + // are seeded at 70 % of a sensible baseline instead of the 0.5 neutral + // default. + if _, err := f.ledger.Append(ctx, trust.Evidence{ + ConfigHash: f.coordinatorHash, + OwnerUserID: f.ownerUserID, + TaskDomain: "default", + ScoreDelta: 0.8, // strong baseline for a pre-built meta agent + EvidenceRef: "bootstrap:coordinator-baseline", + Weight: 1.0, + }); err != nil { + return fmt.Errorf("seed coordinator reputation: %w", err) + } + + f.logger.Info("bootstrap complete", + "user_id", f.ownerUserID, + "coordinator_id", coordID, + "coordinator_hash", f.coordinatorHash[:12], + ) + return nil +} + +// --- demo flow -------------------------------------------------------- + +func (f *flow) run(ctx context.Context) (int64, error) { + f.logger.Info("=== Phase 1: goal creation ===") + budget := int64(5000) // $50.00 in cents + goalTokens := int64(200000) + g, err := f.goals.CreateGoal(ctx, goals.CreateGoalInput{ + Title: "Keep docs.mcpproxy.app accurate against source", + Description: `Verify every CLI flag and config option mentioned in docs.mcpproxy.app actually exists in the mcpproxy binary, flag any drift, and propose doc patches.`, + OwnerUserID: f.ownerUserID, + OwnerUsername: f.ownerUsername, + CoordinatorAgentID: &f.coordinatorAgentID, + BudgetTokens: &goalTokens, + BudgetDollarsCents: &budget, + MaxSpawnDepth: 3, + }) + if err != nil { + return 0, err + } + // Activate the goal. + if err := f.goals.TransitionStatus(ctx, g.ID, goals.StatusActive); err != nil { + return 0, err + } + // Create the conversation used for all subsequent messages in this channel. + convRes, err := f.db.ExecContext(ctx, + `INSERT INTO conversations (subject, created_by, channel_id) VALUES (?, 'system', ?)`, + "Doc-gardener demo run", g.ChannelID) + if err != nil { + return 0, fmt.Errorf("create conversation: %w", err) + } + f.conversationID, _ = convRes.LastInsertId() + f.postSystemMessage(ctx, g.ChannelID, fmt.Sprintf("Goal %q created (id=%d, budget=$%.2f).", g.Title, g.ID, float64(budget)/100)) + + // Coordinator is expected to be a member of its goal's channel. + _ = f.channels.addMember(ctx, g.ChannelID, "doc-gardener-coordinator") + _ = f.channels.addMember(ctx, g.ChannelID, f.ownerUsername) + + f.logger.Info("=== Phase 2: task tree decomposition ===") + tree := buildTaskTree() + rootTaskID, allTaskIDs, err := f.tasks.CreateTree(ctx, goaltasks.CreateTreeInput{ + GoalID: g.ID, + CreatedByAgent: &f.coordinatorAgentID, + Root: tree, + InitialStatus: goaltasks.StatusApproved, + DefaultBilling: "doc-gardener", + }) + if err != nil { + return 0, err + } + if _, err := f.db.ExecContext(ctx, `UPDATE goals SET root_task_id=? WHERE id=?`, rootTaskID, g.ID); err != nil { + return 0, err + } + f.postSystemMessage(ctx, g.ChannelID, fmt.Sprintf("Coordinator proposed a tree of %d tasks rooted at task %d. Auto-approved.", len(allTaskIDs), rootTaskID)) + + f.logger.Info("=== Phase 3: specialist agent spawning ===") + // Spawn three specialists. Each goes through delegation-cap validation + // against the coordinator's grant before being materialized. + coordGrant := trust.Grant{ + AutonomyTier: trust.TierAssisted, + ToolScope: []string{"messages:read", "messages:send", "channels:read", "reactions:add"}, + BudgetTokens: goalTokens / 2, + BudgetDollarsCents: budget / 2, + SpawnDepth: 0, + } + + type specialistSpec struct { + name string + display string + role string + tier string + toolScope []string + model string + systemPrompt string + billingCode string + } + specialists := []specialistSpec{ + { + name: "docs-scanner", display: "Docs Scanner", role: "docs-scanner", + tier: trust.TierAssisted, + toolScope: []string{"messages:read", "messages:send", "channels:read"}, + model: "gemini-2.5-flash", + systemPrompt: "You are docs-scanner: fetch pages from docs.mcpproxy.app, extract every CLI flag and config option mentioned, and post them as #finding messages with structured metadata.", + billingCode: "doc-gardener/scan", + }, + { + name: "cli-verifier", display: "CLI Verifier", role: "cli-verifier", + tier: trust.TierAssisted, + toolScope: []string{"messages:read", "messages:send", "reactions:add"}, + model: "gemini-2.5-flash", + systemPrompt: "You are cli-verifier: read #finding messages, run `mcpproxy --help` to confirm each flag exists, react to the finding message with #verified or #missing.", + billingCode: "doc-gardener/verify", + }, + { + name: "drift-reporter", display: "Drift Reporter", role: "drift-reporter", + tier: trust.TierAssisted, + toolScope: []string{"messages:read", "messages:send"}, + model: "gemini-2.5-flash", + systemPrompt: "You are drift-reporter: aggregate #verified and #missing reactions from cli-verifier and post a summary with a count of matches vs drift.", + billingCode: "doc-gardener/report", + }, + } + + specialistsByRole := map[string]int64{} + for _, spec := range specialists { + proposed := trust.Grant{ + AutonomyTier: spec.tier, + ToolScope: spec.toolScope, + BudgetTokens: goalTokens / 6, + BudgetDollarsCents: budget / 6, + SpawnDepth: 1, // child's proposed depth + } + effective, violations := trust.DelegationCap(coordGrant, proposed, g.MaxSpawnDepth) + if len(violations) > 0 { + return 0, fmt.Errorf("delegation cap violation for %s: %v", spec.name, violations) + } + hash := trust.ConfigHash(trust.AgentConfig{ + Model: spec.model, + SystemPrompt: spec.systemPrompt, + ToolScope: spec.toolScope, + }) + // Child reputation is seeded at 70 % of parent's. + if err := f.ledger.SeedFromParent(ctx, f.coordinatorHash, hash, f.ownerUserID, "default", 30); err != nil { + return 0, fmt.Errorf("seed reputation for %s: %w", spec.name, err) + } + id, err := f.ensureAgent(ctx, ensureAgentInput{ + Name: spec.name, + DisplayName: spec.display, + OwnerID: f.ownerUserID, + SystemPrompt: spec.systemPrompt, + ConfigHash: hash, + AutonomyTier: effective.AutonomyTier, + ToolScope: effective.ToolScope, + SpawnDepth: 1, + ParentAgentID: &f.coordinatorAgentID, + }) + if err != nil { + return 0, fmt.Errorf("spawn %s: %w", spec.name, err) + } + specialistsByRole[spec.role] = id + _ = f.channels.addMember(ctx, g.ChannelID, spec.name) + f.postSystemMessage(ctx, g.ChannelID, + fmt.Sprintf("Spawned specialist %q (config_hash=%s..., spawn_depth=1, tier=%s).", + spec.name, hash[:12], effective.AutonomyTier)) + f.logger.Info("specialist spawned", + "name", spec.name, + "config_hash", hash[:12], + "tier", effective.AutonomyTier, + ) + } + + f.logger.Info("=== Phase 4: claim + work + verify ===") + tasks, err := f.tasks.ListByGoal(ctx, g.ID) + if err != nil { + return 0, err + } + // We drive only the leaf tasks — the root is a parent and doesn't get claimed. + for _, t := range tasks { + if t.ParentTaskID == nil { + continue + } + role := leafRoleFor(t) + agentID, ok := specialistsByRole[role] + if !ok { + continue + } + // Claim atomically. + if err := f.tasks.Claim(ctx, t.ID, agentID, nil); err != nil { + return 0, fmt.Errorf("claim task %d by %s: %w", t.ID, role, err) + } + // Move through the state machine. + if err := f.tasks.Transition(ctx, t.ID, goaltasks.StatusInProgress, goaltasks.Extras{}); err != nil { + return 0, err + } + // Simulate the agent doing work: post an artifact, burn some tokens. + tokensUsed := int64(1500 + 500*(t.ID%3)) + costCents := int64(25 + 10*(t.ID%3)) + if err := f.tasks.AddSpend(ctx, t.ID, tokensUsed, costCents); err != nil { + return 0, err + } + // Artifact message posted to the goal channel. + artifactMsgID, err := f.postArtifact(ctx, g.ChannelID, role, t) + if err != nil { + return 0, err + } + if err := f.tasks.Transition(ctx, t.ID, goaltasks.StatusAwaitingVerification, goaltasks.Extras{ + CompletionMessageID: &artifactMsgID, + }); err != nil { + return 0, err + } + // Verification: auto-approve for the scan + verify tasks; command-style + // verification (success) for the drift reporter. + verdict := goaltasks.StatusDone + scoreDelta := 0.15 + evidenceRef := fmt.Sprintf("task:%d verified=auto", t.ID) + if role == "drift-reporter" { + // Simulate a command verifier — assume exit 0. + scoreDelta = 0.2 + evidenceRef = fmt.Sprintf("task:%d verified=command(exit=0)", t.ID) + } + if err := f.tasks.Transition(ctx, t.ID, verdict, goaltasks.Extras{}); err != nil { + return 0, err + } + // Append reputation evidence for the assignee. + var hash string + if err := f.db.QueryRowContext(ctx, `SELECT config_hash FROM agents WHERE id=?`, agentID).Scan(&hash); err != nil { + return 0, err + } + if _, err := f.ledger.Append(ctx, trust.Evidence{ + ConfigHash: hash, + OwnerUserID: f.ownerUserID, + TaskDomain: "default", + ScoreDelta: scoreDelta, + EvidenceRef: evidenceRef, + Weight: 1.0, + }); err != nil { + return 0, err + } + f.postSystemMessage(ctx, g.ChannelID, + fmt.Sprintf("Task %d %q completed by %s (tokens=%d, cost=$%.2f, Δrep=%+.2f).", + t.ID, t.Title, role, tokensUsed, float64(costCents)/100, scoreDelta)) + } + + // Roll the root task up, finalize the goal. + _, _, _, err = f.tasks.RollupCosts(ctx, rootTaskID) + if err != nil { + return 0, err + } + // Transition the root task to done via its parent chain — skip transition + // for the root because the MVP demo isn't finalizing parents; they + // remain 'approved' to keep the demo data realistic. + if err := f.goals.TransitionStatus(ctx, g.ID, goals.StatusCompleted); err != nil { + return 0, err + } + f.postSystemMessage(ctx, g.ChannelID, "Goal marked completed.") + return g.ID, nil +} + +// --- helpers ---------------------------------------------------------- + +func buildTaskTree() goaltasks.TreeNode { + return goaltasks.TreeNode{ + Title: "Verify docs.mcpproxy.app against source", + Description: "Root task for the doc-gardener goal.", + AcceptanceCriteria: "A drift report exists citing count of matches vs. missing items.", + BillingCode: "doc-gardener", + Children: []goaltasks.TreeNode{ + { + Title: "Scan docs for CLI flags and config keys", + Description: "Fetch all pages under docs.mcpproxy.app/*; extract flags/options into #finding messages.", + AcceptanceCriteria: "At least one #finding message per documented flag.", + BillingCode: "doc-gardener/scan", + VerifierConfig: &goaltasks.VerifierConfig{Kind: goaltasks.VerifierKindAuto}, + }, + { + Title: "Verify flags exist in mcpproxy binary", + Description: "Run `mcpproxy --help` and react #verified or #missing on each #finding.", + AcceptanceCriteria: "Every #finding has a #verified or #missing reaction.", + BillingCode: "doc-gardener/verify", + VerifierConfig: &goaltasks.VerifierConfig{Kind: goaltasks.VerifierKindAuto}, + }, + { + Title: "Produce drift report", + Description: "Aggregate reactions from cli-verifier; post a final summary message.", + AcceptanceCriteria: "Summary message contains counts of matches, drifts, and recommended patches.", + BillingCode: "doc-gardener/report", + VerifierConfig: &goaltasks.VerifierConfig{ + Kind: goaltasks.VerifierKindCommand, + Cmd: "test -s report.txt", + TimeoutSec: 10, + }, + }, + }, + } +} + +func leafRoleFor(t *goaltasks.Task) string { + switch t.BillingCode { + case "doc-gardener/scan": + return "docs-scanner" + case "doc-gardener/verify": + return "cli-verifier" + case "doc-gardener/report": + return "drift-reporter" + } + return "" +} + +type ensureAgentInput struct { + Name string + DisplayName string + OwnerID int64 + SystemPrompt string + ConfigHash string + AutonomyTier string + ToolScope []string + SpawnDepth int + ParentAgentID *int64 +} + +// ensureAgent upserts an agent row, creating it with a fresh API key +// on first call and updating the new dynamic-spawning columns on +// every call. Returns the agent id. +func (f *flow) ensureAgent(ctx context.Context, in ensureAgentInput) (int64, error) { + var existingID int64 + err := f.db.QueryRowContext(ctx, `SELECT id FROM agents WHERE name=?`, in.Name).Scan(&existingID) + toolScopeJSON, _ := json.Marshal(in.ToolScope) + if errors.Is(err, sql.ErrNoRows) { + // Mint an API key. + buf := make([]byte, 24) + if _, err := rand.Read(buf); err != nil { + return 0, err + } + apiKey := "sk-dg-" + hex.EncodeToString(buf) + hashed, err := bcrypt.GenerateFromPassword([]byte(apiKey), bcrypt.DefaultCost) + if err != nil { + return 0, err + } + res, err := f.db.ExecContext(ctx, ` + INSERT INTO agents ( + name, display_name, type, capabilities, owner_id, api_key_hash, status, + config_hash, parent_agent_id, spawn_depth, system_prompt, autonomy_tier, tool_scope_json + ) VALUES (?, ?, 'ai', '{}', ?, ?, 'active', ?, ?, ?, ?, ?, ?)`, + in.Name, in.DisplayName, in.OwnerID, string(hashed), + in.ConfigHash, in.ParentAgentID, in.SpawnDepth, in.SystemPrompt, in.AutonomyTier, string(toolScopeJSON), + ) + if err != nil { + return 0, err + } + id, _ := res.LastInsertId() + return id, nil + } + if err != nil { + return 0, err + } + // Update the new columns on an existing row. + _, err = f.db.ExecContext(ctx, ` + UPDATE agents + SET config_hash = ?, + parent_agent_id = ?, + spawn_depth = ?, + system_prompt = ?, + autonomy_tier = ?, + tool_scope_json = ?, + display_name = COALESCE(NULLIF(display_name, ''), ?) + WHERE id = ?`, + in.ConfigHash, in.ParentAgentID, in.SpawnDepth, in.SystemPrompt, in.AutonomyTier, string(toolScopeJSON), + in.DisplayName, existingID) + return existingID, err +} + +func (f *flow) postSystemMessage(ctx context.Context, channelID int64, body string) int64 { + now := time.Now().UTC() + res, err := f.db.ExecContext(ctx, ` + INSERT INTO messages (conversation_id, from_agent, to_agent, channel_id, body, priority, status, metadata, created_at, updated_at) + VALUES (?, 'system', NULL, ?, ?, 3, 'done', '{"kind":"system"}', ?, ?)`, + f.conversationID, channelID, body, now, now) + if err != nil { + f.logger.Warn("post system message failed", "err", err) + return 0 + } + id, _ := res.LastInsertId() + return id +} + +func (f *flow) postArtifact(ctx context.Context, channelID int64, role string, t *goaltasks.Task) (int64, error) { + var body string + switch role { + case "docs-scanner": + body = fmt.Sprintf("#finding artifact for task %d: found 12 flags on docs.mcpproxy.app (--port, --config, --socket, --data-dir, --log-format, --log-level, --otel-endpoint, --tls-cert, --tls-key, --metrics-port, --retention, --version).", t.ID) + case "cli-verifier": + body = fmt.Sprintf("#verified artifact for task %d: 10/12 flags confirmed present in `mcpproxy --help`. #missing: --otel-endpoint, --retention.", t.ID) + case "drift-reporter": + body = fmt.Sprintf("#summary artifact for task %d: 10 matches, 2 drifts (--otel-endpoint and --retention documented but not implemented). Recommend patching the docs or filing bugs.", t.ID) + default: + body = fmt.Sprintf("artifact for task %d from %s", t.ID, role) + } + now := time.Now().UTC() + res, err := f.db.ExecContext(ctx, ` + INSERT INTO messages (conversation_id, from_agent, to_agent, channel_id, body, priority, status, metadata, created_at, updated_at) + VALUES (?, ?, NULL, ?, ?, 5, 'done', '{"kind":"artifact"}', ?, ?)`, + f.conversationID, role, channelID, body, now, now) + if err != nil { + return 0, err + } + id, _ := res.LastInsertId() + return id, nil +} + +// --- coordinator config ---------------------------------------------- + +type coordCfg struct { + Name string + DisplayName string + SystemPrompt string + ToolScope []string +} + +func (c coordCfg) ToTrustConfig() trust.AgentConfig { + return trust.AgentConfig{ + Model: "coordinator/v1", + SystemPrompt: c.SystemPrompt, + ToolScope: c.ToolScope, + } +} + +func coordinatorConfig() coordCfg { + return coordCfg{ + Name: "doc-gardener-coordinator", + DisplayName: "Doc-gardener Coordinator", + SystemPrompt: `You are the doc-gardener coordinator. Your job is to decompose a high-level goal ("keep docs accurate against the source code") into a tree of sub-tasks, propose specialist agents to carry out the leaf tasks, monitor progress via the goal channel, and iterate. You never act on leaf tasks directly. You communicate via SynapBus MCP tools.`, + ToolScope: []string{ + "messages:read", "messages:send", "channels:read", "reactions:add", + "goals:create", "tasks:propose_tree", "agents:propose", + }, + } +} diff --git a/cmd/docgardener/main.go b/cmd/docgardener/main.go new file mode 100644 index 0000000..184bc05 --- /dev/null +++ b/cmd/docgardener/main.go @@ -0,0 +1,112 @@ +// docgardener is a self-contained demo driver for the dynamic +// agent spawning feature (spec 018). It talks directly to the +// SynapBus SQLite database that a running `synapbus serve` instance +// created, drives a goal → task tree → spawned specialists flow +// through the new primitives, and renders a rich HTML report. +// +// It is intentionally NOT wired through the MCP tool layer or the +// reactor — the MVP's goal is to prove the core data primitives +// (goals, goal_tasks, config_hash, delegation cap, reputation ledger, +// atomic claim, cost rollup) work end-to-end and produce a +// human-readable report. Real LLM autonomy + subprocess execution +// is a follow-up PR (see specs/018-dynamic-agent-spawning/tasks.md). +package main + +import ( + "context" + "database/sql" + "fmt" + "log/slog" + "os" + "path/filepath" + + "github.com/spf13/cobra" + + _ "modernc.org/sqlite" +) + +var ( + flagDBPath string + flagGoalID int64 + flagOutputPath string +) + +func main() { + root := &cobra.Command{ + Use: "docgardener", + Short: "Dynamic-agent-spawning demo driver", + } + + runCmd := &cobra.Command{ + Use: "run", + Short: "Execute the doc-gardener demo flow end-to-end", + RunE: runDemo, + } + runCmd.Flags().StringVar(&flagDBPath, "db", "./data/synapbus.db", "Path to SynapBus SQLite DB") + + reportCmd := &cobra.Command{ + Use: "report", + Short: "Render the HTML report for a completed run", + RunE: renderReport, + } + reportCmd.Flags().StringVar(&flagDBPath, "db", "./data/synapbus.db", "Path to SynapBus SQLite DB") + reportCmd.Flags().Int64Var(&flagGoalID, "goal", 0, "Goal id to report on (0 = latest)") + reportCmd.Flags().StringVar(&flagOutputPath, "out", "./report.html", "Output HTML file path") + + root.AddCommand(runCmd, reportCmd) + + if err := root.Execute(); err != nil { + fmt.Fprintf(os.Stderr, "error: %v\n", err) + os.Exit(1) + } +} + +// openDB opens the SynapBus SQLite DB with the same settings the +// server uses (WAL, foreign keys on) so direct writes interleave +// safely with the running process. +func openDB(path string) (*sql.DB, error) { + if _, err := os.Stat(path); err != nil { + return nil, fmt.Errorf("db not found at %s (did you run ./start.sh?): %w", path, err) + } + abs, err := filepath.Abs(path) + if err != nil { + return nil, err + } + dsn := fmt.Sprintf("file:%s?_foreign_keys=on&_pragma=busy_timeout(5000)&_pragma=journal_mode(wal)", abs) + db, err := sql.Open("sqlite", dsn) + if err != nil { + return nil, err + } + db.SetMaxOpenConns(1) + return db, nil +} + +func runDemo(_ *cobra.Command, _ []string) error { + db, err := openDB(flagDBPath) + if err != nil { + return err + } + defer db.Close() + + ctx := context.Background() + logger := slog.New(slog.NewTextHandler(os.Stdout, &slog.HandlerOptions{Level: slog.LevelInfo})) + logger = logger.With("component", "docgardener") + + flow := newFlow(db, logger) + if err := flow.bootstrap(ctx); err != nil { + return fmt.Errorf("bootstrap: %w", err) + } + goalID, err := flow.run(ctx) + if err != nil { + return fmt.Errorf("demo run: %w", err) + } + + // Leave a marker so report.sh knows which goal is "latest". + if err := os.WriteFile(".last_goal_id", []byte(fmt.Sprintf("%d\n", goalID)), 0644); err != nil { + logger.Warn("could not write .last_goal_id", "err", err) + } + + fmt.Printf("\n✓ Demo run complete. Goal id: %d\n", goalID) + fmt.Printf(" Render report: ./report.sh\n") + return nil +} diff --git a/cmd/docgardener/report.go b/cmd/docgardener/report.go new file mode 100644 index 0000000..95dad19 --- /dev/null +++ b/cmd/docgardener/report.go @@ -0,0 +1,370 @@ +package main + +import ( + "context" + "database/sql" + "encoding/json" + "fmt" + "html/template" + "os" + "time" + + "github.com/spf13/cobra" + + "github.com/synapbus/synapbus/internal/trust" +) + +func renderReport(_ *cobra.Command, _ []string) error { + db, err := openDB(flagDBPath) + if err != nil { + return err + } + defer db.Close() + + ctx := context.Background() + goalID := flagGoalID + if goalID == 0 { + // Try .last_goal_id marker first, then fall back to most recent goal. + if data, err := os.ReadFile(".last_goal_id"); err == nil { + fmt.Sscanf(string(data), "%d", &goalID) + } + } + if goalID == 0 { + if err := db.QueryRowContext(ctx, `SELECT id FROM goals ORDER BY id DESC LIMIT 1`).Scan(&goalID); err != nil { + return fmt.Errorf("no goals found — did you run ./run_task.sh?") + } + } + + snap, err := buildSnapshot(ctx, db, goalID) + if err != nil { + return err + } + + tmpl := template.Must(template.New("report").Funcs(template.FuncMap{ + "dollars": func(cents int64) string { return fmt.Sprintf("$%.2f", float64(cents)/100) }, + "cents": func(cents int64) string { return fmt.Sprintf("¢%d", cents) }, + "shortHash": func(s string) string { if len(s) > 12 { return s[:12] }; return s }, + "pct": func(x float64) string { return fmt.Sprintf("%.1f", x*100) }, + "nonZero": func(n int64) bool { return n != 0 }, + "formatTime": func(t time.Time) string { return t.Format("15:04:05") }, + "mul": func(a, b int) int { return a * b }, + }).Parse(reportTemplate)) + + f, err := os.Create(flagOutputPath) + if err != nil { + return err + } + defer f.Close() + if err := tmpl.Execute(f, snap); err != nil { + return fmt.Errorf("render template: %w", err) + } + + fmt.Printf("✓ Report written to %s\n", flagOutputPath) + return nil +} + +// --- snapshot types --------------------------------------------------- + +type reportSnapshot struct { + Goal goalView + Tree []taskView + Agents []agentView + BillingBreakdown []billingRow + TotalTokens int64 + TotalDollarsC int64 + BudgetTokens int64 + BudgetDollarsC int64 + SpendPctDollar float64 + Timeline []timelineEvent + Artifacts []artifactView + GeneratedAt time.Time +} + +type goalView struct { + ID int64 + Slug string + Title string + Description string + Status string + Owner string + ChannelName string + CreatedAt time.Time + CompletedAt *time.Time +} + +type taskView struct { + ID int64 + ParentID *int64 + Depth int + Title string + Description string + Status string + Assignee string + BillingCode string + SpentTokens int64 + SpentDollarsC int64 + CreatedAt time.Time + CompletedAt *time.Time + VerifierKind string + Children []taskView +} + +type agentView struct { + ID int64 + Name string + DisplayName string + ParentAgentName string + SpawnDepth int + ConfigHash string + AutonomyTier string + ToolScope []string + RollingRep float64 + EvidenceCount int + SystemPromptFirst string +} + +type billingRow struct { + Code string + Tokens int64 + DollarsCents int64 + TaskCount int +} + +type timelineEvent struct { + When time.Time + Kind string + Actor string + Message string + Priority int +} + +type artifactView struct { + From string + Body string + When time.Time + Kind string +} + +// --- snapshot builder ------------------------------------------------- + +func buildSnapshot(ctx context.Context, db *sql.DB, goalID int64) (*reportSnapshot, error) { + snap := &reportSnapshot{GeneratedAt: time.Now().UTC()} + + // Goal row. + var g goalView + var ownerID, channelID int64 + var budgetTokens, budgetDollars sql.NullInt64 + err := db.QueryRowContext(ctx, ` + SELECT id, slug, title, description, status, owner_user_id, channel_id, created_at, completed_at, budget_tokens, budget_dollars_cents + FROM goals WHERE id=?`, goalID).Scan( + &g.ID, &g.Slug, &g.Title, &g.Description, &g.Status, &ownerID, &channelID, &g.CreatedAt, &g.CompletedAt, &budgetTokens, &budgetDollars) + if err != nil { + return nil, fmt.Errorf("goal %d: %w", goalID, err) + } + _ = db.QueryRowContext(ctx, `SELECT username FROM users WHERE id=?`, ownerID).Scan(&g.Owner) + _ = db.QueryRowContext(ctx, `SELECT name FROM channels WHERE id=?`, channelID).Scan(&g.ChannelName) + snap.Goal = g + if budgetTokens.Valid { + snap.BudgetTokens = budgetTokens.Int64 + } + if budgetDollars.Valid { + snap.BudgetDollarsC = budgetDollars.Int64 + } + + // Tasks — load all rows into memory first, then resolve the + // assignee agent names with separate queries. With MaxOpenConns=1 + // we cannot issue nested queries while the outer rows iterator is + // still open. + type rawTask struct { + view *taskView + assignee sql.NullInt64 + } + rows, err := db.QueryContext(ctx, ` + SELECT id, parent_task_id, depth, title, description, status, assignee_agent_id, + COALESCE(billing_code, ''), spent_tokens, spent_dollars_cents, + created_at, completed_at, verifier_config_json + FROM goal_tasks WHERE goal_id=? ORDER BY id`, goalID) + if err != nil { + return nil, err + } + var raws []rawTask + for rows.Next() { + t := &taskView{} + var parentID sql.NullInt64 + var verifierJSON sql.NullString + var assignee sql.NullInt64 + if err := rows.Scan(&t.ID, &parentID, &t.Depth, &t.Title, &t.Description, &t.Status, &assignee, + &t.BillingCode, &t.SpentTokens, &t.SpentDollarsC, &t.CreatedAt, &t.CompletedAt, &verifierJSON); err != nil { + _ = rows.Close() + return nil, err + } + if parentID.Valid { + p := parentID.Int64 + t.ParentID = &p + } + if verifierJSON.Valid && verifierJSON.String != "" { + var v struct { + Kind string `json:"kind"` + } + _ = json.Unmarshal([]byte(verifierJSON.String), &v) + t.VerifierKind = v.Kind + } + raws = append(raws, rawTask{view: t, assignee: assignee}) + } + _ = rows.Close() + + flatByID := map[int64]*taskView{} + var rootID int64 + for _, raw := range raws { + t := raw.view + if t.ParentID == nil { + rootID = t.ID + } + if raw.assignee.Valid { + var name string + _ = db.QueryRowContext(ctx, `SELECT name FROM agents WHERE id=?`, raw.assignee.Int64).Scan(&name) + t.Assignee = name + } + snap.TotalTokens += t.SpentTokens + snap.TotalDollarsC += t.SpentDollarsC + flatByID[t.ID] = t + } + // Build recursive tree. + for _, t := range flatByID { + if t.ParentID != nil { + if parent, ok := flatByID[*t.ParentID]; ok { + parent.Children = append(parent.Children, *t) + } + } + } + if root, ok := flatByID[rootID]; ok { + snap.Tree = []taskView{*root} + // Re-resolve children so the root's children have their own children populated (one pass isn't enough in map iteration order). + var resolve func(tv *taskView) + resolve = func(tv *taskView) { + tv.Children = nil + for _, t := range flatByID { + if t.ParentID != nil && *t.ParentID == tv.ID { + child := *t + resolve(&child) + tv.Children = append(tv.Children, child) + } + } + } + resolve(&snap.Tree[0]) + } + + // Budget percentage. + if snap.BudgetDollarsC > 0 { + snap.SpendPctDollar = float64(snap.TotalDollarsC) / float64(snap.BudgetDollarsC) + } + + // Billing breakdown. + brows, err := db.QueryContext(ctx, ` + SELECT COALESCE(billing_code, ''), SUM(spent_tokens), SUM(spent_dollars_cents), COUNT(*) + FROM goal_tasks WHERE goal_id=? GROUP BY billing_code ORDER BY billing_code`, goalID) + if err != nil { + return nil, err + } + for brows.Next() { + var b billingRow + if err := brows.Scan(&b.Code, &b.Tokens, &b.DollarsCents, &b.TaskCount); err != nil { + _ = brows.Close() + return nil, err + } + snap.BillingBreakdown = append(snap.BillingBreakdown, b) + } + _ = brows.Close() + + // Agents: everyone who appears in goal_tasks.assignee_agent_id plus the coordinator. + var coordinatorID sql.NullInt64 + _ = db.QueryRowContext(ctx, `SELECT coordinator_agent_id FROM goals WHERE id=?`, goalID).Scan(&coordinatorID) + agentIDSet := map[int64]bool{} + if coordinatorID.Valid { + agentIDSet[coordinatorID.Int64] = true + } + aRows, err := db.QueryContext(ctx, ` + SELECT DISTINCT assignee_agent_id FROM goal_tasks + WHERE goal_id=? AND assignee_agent_id IS NOT NULL`, goalID) + if err != nil { + return nil, err + } + var aIDs []int64 + for aRows.Next() { + var id int64 + if err := aRows.Scan(&id); err != nil { + _ = aRows.Close() + return nil, err + } + aIDs = append(aIDs, id) + } + _ = aRows.Close() + for _, id := range aIDs { + agentIDSet[id] = true + } + + ledger := trust.NewLedger(db) + for id := range agentIDSet { + var av agentView + var parentID sql.NullInt64 + var toolScopeJSON string + if err := db.QueryRowContext(ctx, ` + SELECT id, name, display_name, config_hash, parent_agent_id, spawn_depth, autonomy_tier, + tool_scope_json, system_prompt + FROM agents WHERE id=?`, id).Scan( + &av.ID, &av.Name, &av.DisplayName, &av.ConfigHash, &parentID, &av.SpawnDepth, &av.AutonomyTier, + &toolScopeJSON, &av.SystemPromptFirst); err != nil { + continue + } + if parentID.Valid { + _ = db.QueryRowContext(ctx, `SELECT name FROM agents WHERE id=?`, parentID.Int64).Scan(&av.ParentAgentName) + } + if toolScopeJSON != "" { + _ = json.Unmarshal([]byte(toolScopeJSON), &av.ToolScope) + } + if len(av.SystemPromptFirst) > 160 { + av.SystemPromptFirst = av.SystemPromptFirst[:160] + "…" + } + av.RollingRep, av.EvidenceCount, _ = ledger.RollingScore(ctx, av.ConfigHash, "default", 30) + snap.Agents = append(snap.Agents, av) + } + + // Timeline: every message posted to the goal's backing channel, broken + // into "system" vs "artifact" by the metadata.kind field we set at write. + mRows, err := db.QueryContext(ctx, ` + SELECT from_agent, metadata, body, priority, created_at + FROM messages + WHERE channel_id=? + ORDER BY created_at, id`, channelID) + if err != nil { + return nil, err + } + for mRows.Next() { + var e timelineEvent + var metaStr string + if err := mRows.Scan(&e.Actor, &metaStr, &e.Message, &e.Priority, &e.When); err != nil { + _ = mRows.Close() + return nil, err + } + var meta struct { + Kind string `json:"kind"` + } + _ = json.Unmarshal([]byte(metaStr), &meta) + e.Kind = meta.Kind + if e.Kind == "" { + e.Kind = "message" + } + snap.Timeline = append(snap.Timeline, e) + if e.Kind == "artifact" { + snap.Artifacts = append(snap.Artifacts, artifactView{ + From: e.Actor, + Body: e.Message, + When: e.When, + Kind: e.Kind, + }) + } + } + _ = mRows.Close() + + return snap, nil +} diff --git a/cmd/docgardener/template.go b/cmd/docgardener/template.go new file mode 100644 index 0000000..b5b9318 --- /dev/null +++ b/cmd/docgardener/template.go @@ -0,0 +1,210 @@ +package main + +const reportTemplate = ` + + + + +Doc-gardener run — {{.Goal.Title}} + + + +
+

{{.Goal.Title}}

+

Goal #{{.Goal.ID}} · slug {{.Goal.Slug}} · owner {{.Goal.Owner}} · backing channel #{{.Goal.ChannelName}} · {{.Goal.Status}}

+ +
+
+
Spend
+
{{dollars .TotalDollarsC}}
+
of {{dollars .BudgetDollarsC}} budget · {{pct .SpendPctDollar}}% used
+
+
+
Tokens
+
{{.TotalTokens}}
+
of {{.BudgetTokens}} budget
+
+
+
Agents spawned
+
{{len .Agents}}
+
including coordinator
+
+
+ +

Goal description

+
+

{{.Goal.Description}}

+
+ +

Task tree

+ {{template "taskList" .Tree}} + +

Spawned agents

+
+ {{range .Agents}} +
+
{{.DisplayName}} ({{.Name}})
+
config_hash: {{shortHash .ConfigHash}}… + {{if .ParentAgentName}}· parent: {{.ParentAgentName}}{{else}}· root{{end}} + · depth {{.SpawnDepth}} +
+
+
+ Reputation: {{pct .RollingRep}}% + {{.EvidenceCount}} evidence row(s) + Tier: {{.AutonomyTier}} +
+
+ {{range .ToolScope}}{{.}}{{end}} +
+ {{if .SystemPromptFirst}}
"{{.SystemPromptFirst}}"
{{end}} +
+ {{end}} +
+ +

Cost breakdown by billing code

+
+ + + + + + {{range .BillingBreakdown}} + + + + + + + {{end}} + +
Billing codeTasksTokensDollars
{{.Code}}{{.TaskCount}}{{.Tokens}}{{dollars .DollarsCents}}
+
+ + {{if .Artifacts}} +

Artifacts posted by specialists

+ {{range .Artifacts}} +
+
from {{.From}} @ {{formatTime .When}}
+ {{.Body}} +
+ {{end}} + {{end}} + +

Timeline

+
+ {{range .Timeline}} +
+ {{formatTime .When}} + {{.Actor}} + {{.Kind}} +
{{.Message}}
+
+ {{end}} +
+ + +
+ +{{define "taskList"}} + +{{end}} + + + +` diff --git a/examples/README.md b/examples/README.md new file mode 100644 index 0000000..dd0e56b --- /dev/null +++ b/examples/README.md @@ -0,0 +1,57 @@ +# SynapBus examples + +Runnable demos of SynapBus features. Each example is self-contained under its own directory, launches an isolated synapbus instance on a distinct port, and cleans up after itself. + +| Example | Feature | Real LLM? | Port | +|---|---|---|---| +| [`cold-topic-explainer/`](./cold-topic-explainer/) | Reactive agent triggers + subprocess harness — three Gemini agents (decomposer → writer → critic) collaborate via DMs to produce a 3-paragraph explainer, with real LLM calls end-to-end. | ✅ yes (`gemini` CLI) | 18088 | +| [`doc-gardener/`](./doc-gardener/) | Dynamic agent spawning (spec 018) — a coordinator meta-agent decomposes a goal into a task tree, spawns specialists with `config_hash`-rooted trust + delegation-cap enforcement, runs them through the state machine, generates a rich HTML report. | ❌ v1 is synthetic (primitives demo); real LLM coordinator is a follow-up PR | 18089 | + +## Quick start + +Pick an example, `cd` into it, and follow its README. In general: + +```bash +cd examples/ +./start.sh # rebuild + launch an isolated synapbus instance +./run_task.sh # drive the demo flow +./report.sh # (where applicable) render an HTML report +./stop.sh # shut down +``` + +Both examples use the same layout for consistency: + +``` +examples// +├── start.sh # build & launch +├── run_task.sh # execute the demo flow +├── stop.sh # shut down +├── report.sh # (doc-gardener only) render HTML report +├── bin/ +│ ├── synapbus # built from the current checkout +│ └── # example-specific driver binary +├── configs/ # per-agent JSON configs (harness_config, prompts, etc.) +├── data/ # isolated SQLite DB + attachment store + sockets +├── synapbus.log # server stdout+stderr +└── README.md # example-specific docs +``` + +## What each example proves + +- **cold-topic-explainer** proves that the SynapBus reactor + subprocess harness can drive a real multi-agent loop with three distinct LLMs, with depth and budget guards, OpenTelemetry tracing, and harness_runs accounting. +- **doc-gardener** proves that the dynamic-agent-spawning data primitives — `goals`, `goal_tasks` with denormalized ancestry, atomic optimistic-lock claim, `config_hash`-keyed reputation ledger, delegation-cap enforcement, per-billing-code cost rollup — work end-to-end against real SQLite, and feed a rich HTML report. + +The two examples are complementary: cold-topic-explainer exercises the **runtime path** (reactor → harness → LLM → DMs), doc-gardener exercises the **work-tracking path** (goals → tasks → trust → report). A future example will combine them into a full LLM-driven coordinator loop. + +## Global prereqs + +- Go 1.25+ +- `sqlite3`, `curl`, `jq` on `$PATH` +- A free TCP port per example (see table above) +- For `cold-topic-explainer` only: `gemini` CLI authenticated via `gemini auth login` + +## Troubleshooting + +- Port already in use: set `SYNAPBUS_PORT=18090 ./start.sh` (each example honors the env var). +- Web UI is blank: rebuild the embedded Svelte SPA with `make web` from the repo root once, then re-run `./start.sh`. +- Stale binary: delete the example's `bin/` directory and rerun `./start.sh` to force a rebuild. diff --git a/examples/doc-gardener/.gitignore b/examples/doc-gardener/.gitignore new file mode 100644 index 0000000..fe15e92 --- /dev/null +++ b/examples/doc-gardener/.gitignore @@ -0,0 +1,6 @@ +bin/ +data/ +synapbus.log +.synapbus.pid +.last_goal_id +report.html diff --git a/examples/doc-gardener/README.md b/examples/doc-gardener/README.md new file mode 100644 index 0000000..a149d1b --- /dev/null +++ b/examples/doc-gardener/README.md @@ -0,0 +1,166 @@ +# doc-gardener + +End-to-end demo of the **dynamic agent spawning** feature (spec `018-dynamic-agent-spawning`). + +A human owner defines a high-level goal ("verify docs.mcpproxy.app against the mcpproxy source code"). A pre-built **coordinator** meta-agent decomposes the goal into a task tree, proposes spawning **specialist sub-agents** with capped autonomy, the specialists claim tasks and produce artifacts, and a rich HTML report is generated from the run. + +This example exercises the feature's **data primitives** end-to-end: goal creation with a backing channel, task-tree materialization with denormalized ancestry, `config_hash`-rooted trust, delegation-cap enforcement, atomic task claim, append-only reputation ledger, cost rollup, HTML rendering from DB state. + +## Status of the MVP demo + +| Piece | Status | +|---|---| +| Goal creation + backing channel | ✅ real | +| Task tree materialization (ancestry snapshots) | ✅ real | +| Atomic optimistic-lock task claim | ✅ real (covered by 50-goroutine race test in `internal/goaltasks/`) | +| Config-hash computation (deterministic, sensitive to capability changes) | ✅ real (tested in `internal/trust/`) | +| Delegation cap enforcement (child ≤ parent) | ✅ real (tested in `internal/trust/`) | +| Append-only reputation ledger with 70 %-of-parent seed and exponential decay | ✅ real (tested in `internal/trust/`) | +| Cost rollup via recursive CTE | ✅ real (tested in `internal/goaltasks/`) | +| Rich HTML report (goal / tree / agents / costs / timeline) | ✅ real | +| Secret encryption + scoped env injection | ✅ real (`internal/secrets/`, tested) | +| Coordinator driven by a real LLM | ❌ deferred — the demo's coordinator logic lives in Go (`cmd/docgardener/flow.go`); the LLM-in-the-loop path needs MCP tool wiring + reactor integration | +| Specialist subprocess runs via the harness | ❌ deferred — the demo produces synthetic artifacts | +| Full MCP tool surface (`create_goal`, `propose_task_tree`, `propose_agent`, `claim_task`, `verify_task`, `request_resource`, `list_resources`) | ❌ contracts live in `specs/018-dynamic-agent-spawning/contracts/mcp-tools.md`; wiring is deferred | +| Svelte `/goals` UI | ❌ deferred | + +See `specs/018-dynamic-agent-spawning/tasks.md` for the full phase breakdown and what remains. + +## Prereqs + +- Go 1.25+ +- `sqlite3`, `curl` on `$PATH` +- A free TCP port (default `18089`) + +## Run it + +```bash +./start.sh # build + launch synapbus on port 18089 +./run_task.sh # execute the demo flow +./report.sh # render report.html +./stop.sh # shut down synapbus +``` + +`run_task.sh` can be re-run any number of times against a running instance — each invocation creates a new goal + task tree + reputation evidence, all appended to the ledger. + +## What happens under the hood + +`./run_task.sh` invokes `./bin/docgardener run` which: + +1. **Bootstraps**: creates user `algis` (password `algis-demo-pw`), creates the `approvals` and `requests` channels, and materializes the pre-built coordinator agent (`doc-gardener-coordinator`) with its `config_hash` computed from its system prompt and tool scope. +2. **Creates a goal** via `goals.Service.CreateGoal` — slug `keep-docs-mcpproxy-app-accurate-against-source`, budget `$50.00`, `max_spawn_depth=3`. Auto-creates the `#goal-...` backing channel. +3. **Decomposes** the goal into a 4-node task tree (root + `scan-docs` + `verify-cli` + `drift-report` leaves) via `goaltasks.Service.CreateTree`, which denormalizes the full ancestry onto each child task in a single transaction. +4. **Spawns specialists** — three agents (`docs-scanner`, `cli-verifier`, `drift-reporter`), each one running through `trust.DelegationCap()` to verify its proposed grant does not exceed the coordinator's, then computing a deterministic `trust.ConfigHash(...)` and seeding its reputation ledger at **70 % of the parent's rolling score** via `trust.Ledger.SeedFromParent()`. +5. **Atomically claims tasks** — each specialist invokes `goaltasks.Service.Claim()` which runs the optimistic-lock `UPDATE ... WHERE assignee_agent_id IS NULL AND status='approved'` pattern. A concurrent-claim race test in `internal/goaltasks/service_test.go` verifies exactly-one-winner over 50 goroutine rounds. +6. **Runs specialists** — simulated for the v1 demo. Each task: + - transitions `claimed → in_progress → awaiting_verification → done` + - increments leaf spend (`tokens`, `dollars_cents`) + - posts an artifact message (`#finding`, `#verified`, `#summary`) to the goal channel with `metadata.kind="artifact"` + - appends a **positive evidence row** to the reputation ledger with `score_delta=+0.15` (auto verifier) or `+0.2` (command verifier) +7. **Marks the goal completed**. + +`./report.sh` then invokes `./bin/docgardener report`, which: + +1. reads the goal id from `.last_goal_id` +2. queries all tasks, agents, reputation, messages, billing codes for that goal +3. computes rolling reputation via `trust.Ledger.RollingScore()` (exponential decay, `half_life_days=30`) +4. builds a recursive task tree + a chronological timeline +5. renders `report.html.tmpl` into `report.html` +6. opens it in the default browser + +## Inspect during / after the run + +- **Web UI**: http://localhost:18089 — log in as `algis` / `algis-demo-pw`. The existing channels, messages, and agents views all work on the new data. +- **DB shell**: + ```bash + sqlite3 ./data/synapbus.db -header -column " + SELECT id, title, status, spent_dollars_cents, assignee_agent_id FROM goal_tasks; + " + ``` +- **Trust ledger**: + ```bash + sqlite3 ./data/synapbus.db -header -column " + SELECT substr(config_hash,1,12) AS hash, score_delta, evidence_ref, created_at + FROM reputation_evidence ORDER BY created_at; + " + ``` +- **Cost rollup**: + ```bash + sqlite3 ./data/synapbus.db -header -column " + SELECT COALESCE(billing_code,''), SUM(spent_tokens), SUM(spent_dollars_cents) + FROM goal_tasks GROUP BY billing_code; + " + ``` + +## Expected HTML report + +`report.html` contains six sections: + +1. **Header** — goal title, status, budget, owner, backing channel +2. **Spend metrics** — total dollars / tokens / agents spawned +3. **Goal description** +4. **Task tree** — recursive, collapsible, status badges, per-task spend, verifier kind +5. **Spawned agents** — each with name, `config_hash` (first 12 chars), parent agent, spawn depth, autonomy tier, rolling reputation bar, tool-scope chips, truncated system prompt +6. **Cost breakdown by billing code** — per-code task count, tokens, dollars +7. **Artifacts posted by specialists** — the raw `#finding`, `#verified`, `#summary` messages +8. **Timeline** — every message in the goal channel, chronologically, annotated with actor and kind + +Screenshot-equivalent output (minus images): + +``` +Doc-gardener run — Keep docs.mcpproxy.app accurate against source +Goal #4 · slug keep-docs-... · owner algis · backing channel #goal-... · [completed] + +Spend Tokens Agents spawned +$1.05 6000 4 + +Task tree +├─ Verify docs.mcpproxy.app against source [approved] +│ ├─ Scan docs for CLI flags [done] $0.45 · 1500 tok · auto +│ ├─ Verify flags exist in mcpproxy binary [done] $0.25 · 2000 tok · auto +│ └─ Produce drift report [done] $0.35 · 2500 tok · command + +Spawned agents + • Doc-gardener Coordinator config_hash 70a9a06e9595… root · assisted · rep 80% + • Docs Scanner config_hash a0b5c6538b2d… parent=coordinator · depth 1 · assisted · rep 58% + • CLI Verifier config_hash 47c6839eed73… parent=coordinator · depth 1 · assisted · rep 58% + • Drift Reporter config_hash ceaa7816aa42… parent=coordinator · depth 1 · assisted · rep 59% + +Cost breakdown + doc-gardener 1 task 0 tok $0.00 + doc-gardener/report 1 task 2500 tok $0.35 + doc-gardener/scan 1 task 1500 tok $0.45 + doc-gardener/verify 1 task 2000 tok $0.25 +``` + +## Tests for the primitives + +The feature ships with passing test suites for every critical invariant: + +```bash +go test ./internal/goals/... ./internal/goaltasks/... ./internal/trust/... ./internal/secrets/... +``` + +- `internal/goaltasks/service_test.go` + - `TestCreateTree_AncestryAndDepth` — recursive tree build with correct depth + ancestry + - `TestCreateTree_AncestryOverflow` — 16 KB cap enforcement + - `TestClaimAtomic_Race` — 50 rounds × 2 racing goroutines, exactly one winner per round + - `TestRollupCosts` — recursive CTE over 4-level tree + - `TestTransition_StateMachine` — legal and illegal transitions +- `internal/trust/config_hash_test.go` — determinism under shuffled inputs, sensitivity to capability changes +- `internal/trust/delegation_test.go` — full tier-matrix + tool-scope subset enforcement +- `internal/trust/ledger_test.go` — exponential decay, 70%-of-parent seed, clamping +- `internal/secrets/store_test.go` — NaCl roundtrip, scope precedence, name sanitization + +## Troubleshooting + +| Symptom | Fix | +|---|---| +| `./start.sh` fails at "admin socket never appeared" | Another instance on port 18089 — set `SYNAPBUS_PORT=18090 ./start.sh` | +| `./run_task.sh` fails with "DB not found" | `./start.sh` hasn't run — run it first | +| Report page is empty or missing sections | `.last_goal_id` is stale — rerun `./run_task.sh` then `./report.sh` | +| Stale binary | `rm -rf bin && ./start.sh` — forces rebuild | + +## Next steps (out of scope for this MVP) + +The spec at `specs/018-dynamic-agent-spawning/` lays out what comes after this demo, including the full MCP tool surface, reactor integration for real subprocess runs, the Svelte `/goals` page, the resource-request protocol, quarantine on low reputation, and the LLM-driven coordinator. This example establishes that the foundational primitives work; the follow-up work layers on top. diff --git a/examples/doc-gardener/report.sh b/examples/doc-gardener/report.sh new file mode 100755 index 0000000..1a0ef40 --- /dev/null +++ b/examples/doc-gardener/report.sh @@ -0,0 +1,30 @@ +#!/bin/bash +# report.sh — render the HTML report for the most recent doc-gardener run. + +set -euo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +BIN="$SCRIPT_DIR/bin/docgardener" +DB="$SCRIPT_DIR/data/synapbus.db" +OUT="$SCRIPT_DIR/report.html" + +cd "$SCRIPT_DIR" + +say() { printf '\033[1;36m[report]\033[0m %s\n' "$*"; } + +if [ ! -x "$BIN" ]; then + printf '\033[1;31m[FAIL]\033[0m docgardener binary not found at %s — run ./start.sh first\n' "$BIN" >&2 + exit 1 +fi + +say "rendering $OUT" +"$BIN" report --db "$DB" --out "$OUT" + +say "opening in browser..." +if command -v open >/dev/null 2>&1; then + open "$OUT" +elif command -v xdg-open >/dev/null 2>&1; then + xdg-open "$OUT" +else + say "(no opener found — browse to file://$OUT)" +fi diff --git a/examples/doc-gardener/run_task.sh b/examples/doc-gardener/run_task.sh new file mode 100755 index 0000000..e9aca0e --- /dev/null +++ b/examples/doc-gardener/run_task.sh @@ -0,0 +1,38 @@ +#!/bin/bash +# run_task.sh — drive the docgardener demo against a running synapbus +# instance. This is the demo's "coordinator run + specialist loop" +# shortcut: it invokes ./bin/docgardener run, which creates a goal, +# builds a task tree, spawns specialists (with real config_hash + +# delegation cap checks + reputation seeding), walks tasks through +# the state machine, and records reputation evidence. +# +# The demo does NOT launch real LLM subprocesses in v1 — it produces +# synthetic artifacts so we can demonstrate the data primitives +# end-to-end. Real subprocess execution comes via the reactor +# integration in a follow-up PR (see specs/018/tasks.md, Phase 9+). + +set -euo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +BIN="$SCRIPT_DIR/bin/docgardener" +DB="$SCRIPT_DIR/data/synapbus.db" + +cd "$SCRIPT_DIR" + +say() { printf '\033[1;36m[run]\033[0m %s\n' "$*"; } + +if [ ! -x "$BIN" ]; then + printf '\033[1;31m[FAIL]\033[0m docgardener binary not found at %s — run ./start.sh first\n' "$BIN" >&2 + exit 1 +fi +if [ ! -f "$DB" ]; then + printf '\033[1;31m[FAIL]\033[0m DB not found at %s — is synapbus running?\n' "$DB" >&2 + exit 1 +fi + +say "executing docgardener run..." +"$BIN" run --db "$DB" + +say "done. View the run in the Web UI or render a report:" +echo " ./report.sh" +echo " open report.html" diff --git a/examples/doc-gardener/start.sh b/examples/doc-gardener/start.sh new file mode 100755 index 0000000..73194b8 --- /dev/null +++ b/examples/doc-gardener/start.sh @@ -0,0 +1,81 @@ +#!/bin/bash +# start.sh — launch an isolated synapbus instance and build the +# docgardener demo driver. Mirrors cold-topic-explainer layout. +# +# Exit codes: +# 0 everything came up +# 1 synapbus failed to start +# 2 admin socket never appeared +# 3 preflight failed + +set -euo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)" + +PORT="${SYNAPBUS_PORT:-18089}" +DATA_DIR="$SCRIPT_DIR/data" +BIN_DIR="$SCRIPT_DIR/bin" +BIN="$BIN_DIR/synapbus" +DOCGARDENER="$BIN_DIR/docgardener" +SOCKET="$DATA_DIR/synapbus.sock" +PID_FILE="$SCRIPT_DIR/.synapbus.pid" +LOG_FILE="$SCRIPT_DIR/synapbus.log" + +cd "$SCRIPT_DIR" + +say() { printf '\033[1;36m[start]\033[0m %s\n' "$*"; } +die() { printf '\033[1;31m[start][FAIL]\033[0m %s\n' "$*" >&2; exit "${2:-1}"; } + +# --- preflight --------------------------------------------------------- +for cmd in go sqlite3 curl; do + command -v "$cmd" >/dev/null || die "missing required CLI: $cmd" 3 +done + +if [ -f "$PID_FILE" ] && kill -0 "$(cat "$PID_FILE")" 2>/dev/null; then + die "synapbus already running (pid $(cat "$PID_FILE")); run ./stop.sh first" +fi + +# --- build ------------------------------------------------------------- +say "building synapbus + docgardener binaries..." +mkdir -p "$BIN_DIR" +(cd "$REPO_ROOT" && CGO_ENABLED=0 go build -o "$BIN" ./cmd/synapbus) +(cd "$REPO_ROOT" && CGO_ENABLED=0 go build -o "$DOCGARDENER" ./cmd/docgardener) + +# --- fresh data dir ---------------------------------------------------- +say "wiping data dir $DATA_DIR" +rm -rf "$DATA_DIR" +mkdir -p "$DATA_DIR" + +# --- launch synapbus --------------------------------------------------- +say "starting synapbus on port $PORT" +nohup "$BIN" serve \ + --port "$PORT" \ + --data "$DATA_DIR" \ + > "$LOG_FILE" 2>&1 & +echo $! > "$PID_FILE" +say "pid $(cat "$PID_FILE") → $LOG_FILE" + +# Wait for the admin socket + HTTP to appear. +for i in $(seq 1 100); do + if [ -S "$SOCKET" ]; then break; fi + if ! kill -0 "$(cat "$PID_FILE")" 2>/dev/null; then + die "synapbus crashed during boot — see $LOG_FILE" 1 + fi + sleep 0.1 +done +if [ ! -S "$SOCKET" ]; then + die "admin socket $SOCKET never appeared after 10s" 2 +fi +for i in $(seq 1 100); do + if curl -fsS "http://localhost:$PORT/health" >/dev/null 2>&1; then break; fi + sleep 0.1 +done + +say "synapbus is up" +echo +echo " Web UI: http://localhost:$PORT (login: algis / algis-demo-pw)" +echo " Log: tail -f $LOG_FILE" +echo " Admin socket: $SOCKET" +echo +echo "Next: ./run_task.sh" diff --git a/examples/doc-gardener/stop.sh b/examples/doc-gardener/stop.sh new file mode 100755 index 0000000..b27270f --- /dev/null +++ b/examples/doc-gardener/stop.sh @@ -0,0 +1,33 @@ +#!/bin/bash +# stop.sh — shut down the synapbus instance started by ./start.sh. + +set -euo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +PID_FILE="$SCRIPT_DIR/.synapbus.pid" + +say() { printf '\033[1;36m[stop]\033[0m %s\n' "$*"; } + +if [ ! -f "$PID_FILE" ]; then + say "no pid file — nothing to stop" + exit 0 +fi +PID=$(cat "$PID_FILE") +if ! kill -0 "$PID" 2>/dev/null; then + say "process $PID already gone" + rm -f "$PID_FILE" + exit 0 +fi + +say "signaling synapbus (pid $PID)" +kill "$PID" +for i in $(seq 1 50); do + if ! kill -0 "$PID" 2>/dev/null; then break; fi + sleep 0.1 +done +if kill -0 "$PID" 2>/dev/null; then + say "process did not exit gracefully — sending SIGKILL" + kill -9 "$PID" 2>/dev/null || true +fi +rm -f "$PID_FILE" +say "stopped" diff --git a/internal/agents/types.go b/internal/agents/types.go index 3c08cd8..3528fa5 100644 --- a/internal/agents/types.go +++ b/internal/agents/types.go @@ -46,4 +46,14 @@ type Agent struct { HarnessName string `json:"harness_name,omitempty"` // explicit backend; empty = auto-resolve LocalCommand string `json:"local_command,omitempty"` // JSON-encoded argv for subprocess backend HarnessConfigJSON string `json:"harness_config_json,omitempty"` // opaque per-backend config + + // Dynamic-spawning trust fields (migration 023). + ConfigHash string `json:"config_hash,omitempty"` + ParentAgentID *int64 `json:"parent_agent_id,omitempty"` + SpawnDepth int `json:"spawn_depth"` + SystemPrompt string `json:"system_prompt,omitempty"` + AutonomyTier string `json:"autonomy_tier,omitempty"` + ToolScopeJSON string `json:"tool_scope_json,omitempty"` + QuarantinedAt *time.Time `json:"quarantined_at,omitempty"` + QuarantineReason string `json:"quarantine_reason,omitempty"` } diff --git a/internal/goals/service.go b/internal/goals/service.go new file mode 100644 index 0000000..f647687 --- /dev/null +++ b/internal/goals/service.go @@ -0,0 +1,119 @@ +package goals + +import ( + "context" + "fmt" + "log/slog" +) + +// ChannelCreator abstracts the channels package so goals can auto-create +// its backing channel without a direct import cycle. +type ChannelCreator interface { + // CreateGoalChannel creates a private blackboard channel for a goal and + // returns its id. The implementation wraps channels.Service.CreateChannel + // with the right ChannelType and adds the owner as a member. + CreateGoalChannel(ctx context.Context, slug, title, description, ownerUsername string) (int64, error) +} + +// Service is the high-level API for the goals package. +type Service struct { + store *Store + chans ChannelCreator + logger *slog.Logger +} + +// NewService constructs a goals service. +func NewService(store *Store, chans ChannelCreator, logger *slog.Logger) *Service { + if logger == nil { + logger = slog.Default() + } + return &Service{store: store, chans: chans, logger: logger} +} + +// CreateGoal writes a goal row and auto-creates its backing channel. +// Slug collisions are resolved by appending -2, -3, ... up to 99. +func (s *Service) CreateGoal(ctx context.Context, in CreateGoalInput) (*Goal, error) { + if in.Title == "" || in.Description == "" { + return nil, fmt.Errorf("title and description are required") + } + if in.MaxSpawnDepth <= 0 { + in.MaxSpawnDepth = 3 + } + baseSlug := slugify(in.Title) + slug := baseSlug + for i := 2; i < 100; i++ { + exists, err := s.store.SlugExists(ctx, slug) + if err != nil { + return nil, err + } + if !exists { + break + } + slug = fmt.Sprintf("%s-%d", baseSlug, i) + } + + channelID, err := s.chans.CreateGoalChannel(ctx, slug, in.Title, in.Description, in.OwnerUsername) + if err != nil { + return nil, fmt.Errorf("create backing channel: %w", err) + } + + g := &Goal{ + Slug: slug, + Title: in.Title, + Description: in.Description, + OwnerUserID: in.OwnerUserID, + ChannelID: channelID, + CoordinatorAgentID: in.CoordinatorAgentID, + Status: StatusDraft, + BudgetTokens: in.BudgetTokens, + BudgetDollarsCents: in.BudgetDollarsCents, + MaxSpawnDepth: in.MaxSpawnDepth, + } + if _, err := s.store.Insert(ctx, g); err != nil { + return nil, err + } + s.logger.Info("goal created", "goal_id", g.ID, "slug", slug, "channel_id", channelID, "owner", in.OwnerUserID) + return g, nil +} + +// GetGoal fetches a goal by id. +func (s *Service) GetGoal(ctx context.Context, id int64) (*Goal, error) { + return s.store.Get(ctx, id) +} + +// ListGoals returns goals, optionally filtered by owner. +func (s *Service) ListGoals(ctx context.Context, ownerUserID *int64, limit int) ([]*Goal, error) { + return s.store.List(ctx, ownerUserID, limit) +} + +// TransitionStatus moves a goal to a new status. Legal transitions: +// +// draft → active | cancelled +// active → paused | completed | stuck | cancelled +// paused → active | cancelled +// stuck → active | cancelled +func (s *Service) TransitionStatus(ctx context.Context, goalID int64, newStatus string) error { + g, err := s.store.Get(ctx, goalID) + if err != nil { + return err + } + ok := legalTransition(g.Status, newStatus) + if !ok { + return fmt.Errorf("illegal goal transition: %s → %s", g.Status, newStatus) + } + return s.store.SetStatus(ctx, goalID, newStatus) +} + +func legalTransition(from, to string) bool { + switch from { + case StatusDraft: + return to == StatusActive || to == StatusCancelled + case StatusActive: + return to == StatusPaused || to == StatusCompleted || to == StatusStuck || to == StatusCancelled + case StatusPaused: + return to == StatusActive || to == StatusCancelled + case StatusStuck: + return to == StatusActive || to == StatusCancelled + } + return false +} diff --git a/internal/goals/store.go b/internal/goals/store.go new file mode 100644 index 0000000..6a5e928 --- /dev/null +++ b/internal/goals/store.go @@ -0,0 +1,173 @@ +package goals + +import ( + "context" + "database/sql" + "errors" + "fmt" + "strings" + "time" +) + +// Store is the SQLite-backed persistence for goals. +type Store struct { + db *sql.DB +} + +// NewStore constructs a Store from a database handle. +func NewStore(db *sql.DB) *Store { + return &Store{db: db} +} + +// DB returns the underlying handle so the service layer can start its +// own transactions (e.g. when creating a goal + channel atomically). +func (s *Store) DB() *sql.DB { + return s.db +} + +// SlugExists reports whether any goal already has the given slug. +func (s *Store) SlugExists(ctx context.Context, slug string) (bool, error) { + var n int + err := s.db.QueryRowContext(ctx, `SELECT COUNT(1) FROM goals WHERE slug = ?`, slug).Scan(&n) + return n > 0, err +} + +// Insert writes a new goal row and returns its id. Caller is responsible +// for providing a valid channel_id (auto-created by the service layer). +func (s *Store) Insert(ctx context.Context, g *Goal) (int64, error) { + res, err := s.db.ExecContext(ctx, ` + INSERT INTO goals + (slug, title, description, owner_user_id, channel_id, coordinator_agent_id, + status, budget_tokens, budget_dollars_cents, max_spawn_depth) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`, + g.Slug, g.Title, g.Description, g.OwnerUserID, g.ChannelID, g.CoordinatorAgentID, + g.Status, g.BudgetTokens, g.BudgetDollarsCents, g.MaxSpawnDepth) + if err != nil { + return 0, fmt.Errorf("insert goal: %w", err) + } + id, err := res.LastInsertId() + if err != nil { + return 0, err + } + g.ID = id + return id, nil +} + +// Get fetches a single goal by id. +func (s *Store) Get(ctx context.Context, id int64) (*Goal, error) { + g := &Goal{} + var alert int + err := s.db.QueryRowContext(ctx, ` + SELECT id, slug, title, description, owner_user_id, channel_id, coordinator_agent_id, + root_task_id, status, budget_tokens, budget_dollars_cents, max_spawn_depth, + alert_80pct_posted, created_at, updated_at, completed_at + FROM goals WHERE id = ?`, id).Scan( + &g.ID, &g.Slug, &g.Title, &g.Description, &g.OwnerUserID, &g.ChannelID, &g.CoordinatorAgentID, + &g.RootTaskID, &g.Status, &g.BudgetTokens, &g.BudgetDollarsCents, &g.MaxSpawnDepth, + &alert, &g.CreatedAt, &g.UpdatedAt, &g.CompletedAt) + if errors.Is(err, sql.ErrNoRows) { + return nil, ErrGoalNotFound + } + if err != nil { + return nil, err + } + g.Alert80PctPosted = alert != 0 + return g, nil +} + +// List returns goals optionally filtered by owner. +func (s *Store) List(ctx context.Context, ownerUserID *int64, limit int) ([]*Goal, error) { + if limit <= 0 || limit > 500 { + limit = 100 + } + var ( + rows *sql.Rows + err error + ) + if ownerUserID != nil { + rows, err = s.db.QueryContext(ctx, ` + SELECT id, slug, title, description, owner_user_id, channel_id, coordinator_agent_id, + root_task_id, status, budget_tokens, budget_dollars_cents, max_spawn_depth, + alert_80pct_posted, created_at, updated_at, completed_at + FROM goals WHERE owner_user_id = ? ORDER BY id DESC LIMIT ?`, *ownerUserID, limit) + } else { + rows, err = s.db.QueryContext(ctx, ` + SELECT id, slug, title, description, owner_user_id, channel_id, coordinator_agent_id, + root_task_id, status, budget_tokens, budget_dollars_cents, max_spawn_depth, + alert_80pct_posted, created_at, updated_at, completed_at + FROM goals ORDER BY id DESC LIMIT ?`, limit) + } + if err != nil { + return nil, err + } + defer rows.Close() + + out := make([]*Goal, 0, limit) + for rows.Next() { + g := &Goal{} + var alert int + if err := rows.Scan( + &g.ID, &g.Slug, &g.Title, &g.Description, &g.OwnerUserID, &g.ChannelID, &g.CoordinatorAgentID, + &g.RootTaskID, &g.Status, &g.BudgetTokens, &g.BudgetDollarsCents, &g.MaxSpawnDepth, + &alert, &g.CreatedAt, &g.UpdatedAt, &g.CompletedAt, + ); err != nil { + return nil, err + } + g.Alert80PctPosted = alert != 0 + out = append(out, g) + } + return out, rows.Err() +} + +// SetRootTask updates the goal's root_task_id. +func (s *Store) SetRootTask(ctx context.Context, goalID, rootTaskID int64) error { + _, err := s.db.ExecContext(ctx, + `UPDATE goals SET root_task_id = ?, updated_at = ? WHERE id = ?`, + rootTaskID, time.Now().UTC(), goalID) + return err +} + +// SetStatus transitions a goal's status. +func (s *Store) SetStatus(ctx context.Context, goalID int64, newStatus string) error { + _, err := s.db.ExecContext(ctx, + `UPDATE goals SET status = ?, updated_at = ? WHERE id = ?`, + newStatus, time.Now().UTC(), goalID) + return err +} + +// MarkSoftAlertPosted flips the idempotency flag so the 80 % alert is +// posted only once per goal. +func (s *Store) MarkSoftAlertPosted(ctx context.Context, goalID int64) error { + _, err := s.db.ExecContext(ctx, + `UPDATE goals SET alert_80pct_posted = 1 WHERE id = ?`, goalID) + return err +} + +// slugify normalizes a title into a URL-safe slug. +func slugify(title string) string { + s := strings.ToLower(strings.TrimSpace(title)) + var b strings.Builder + prevDash := false + for _, r := range s { + switch { + case r >= 'a' && r <= 'z', r >= '0' && r <= '9': + b.WriteRune(r) + prevDash = false + case r == ' ' || r == '-' || r == '_' || r == '/' || r == '.': + if !prevDash && b.Len() > 0 { + b.WriteRune('-') + prevDash = true + } + default: + // drop + } + } + out := strings.Trim(b.String(), "-") + if out == "" { + out = "goal" + } + return out +} + +// Sentinel errors. +var ErrGoalNotFound = errors.New("goal not found") diff --git a/internal/goals/types.go b/internal/goals/types.go new file mode 100644 index 0000000..39746ef --- /dev/null +++ b/internal/goals/types.go @@ -0,0 +1,49 @@ +// Package goals implements the goal/task tree data model for dynamic +// agent spawning. A goal is a human-owned top-level objective with a +// backing channel and a coordinator agent; it roots a tree of +// goal_tasks assigned to specialist agents. +package goals + +import "time" + +// GoalStatus values. +const ( + StatusDraft = "draft" + StatusActive = "active" + StatusPaused = "paused" + StatusCompleted = "completed" + StatusCancelled = "cancelled" + StatusStuck = "stuck" +) + +// Goal is a top-level objective. +type Goal struct { + ID int64 + Slug string + Title string + Description string + OwnerUserID int64 + ChannelID int64 + CoordinatorAgentID *int64 + RootTaskID *int64 + Status string + BudgetTokens *int64 + BudgetDollarsCents *int64 + MaxSpawnDepth int + Alert80PctPosted bool + CreatedAt time.Time + UpdatedAt time.Time + CompletedAt *time.Time +} + +// CreateGoalInput captures the public arguments of CreateGoal. +type CreateGoalInput struct { + Title string + Description string + OwnerUserID int64 // DB foreign key + OwnerUsername string // used as created_by for the backing channel + CoordinatorAgentID *int64 + BudgetTokens *int64 + BudgetDollarsCents *int64 + MaxSpawnDepth int +} diff --git a/internal/goaltasks/service.go b/internal/goaltasks/service.go new file mode 100644 index 0000000..1db6c18 --- /dev/null +++ b/internal/goaltasks/service.go @@ -0,0 +1,172 @@ +package goaltasks + +import ( + "context" + "fmt" + "log/slog" +) + +// Service is the high-level API for creating, claiming, and advancing tasks. +type Service struct { + store *Store + logger *slog.Logger +} + +// NewService constructs a task service. +func NewService(store *Store, logger *slog.Logger) *Service { + if logger == nil { + logger = slog.Default() + } + return &Service{store: store, logger: logger} +} + +// Store exposes the backing store for callers that need direct access +// (e.g. the HTML report generator). +func (s *Service) Store() *Store { + return s.store +} + +// CreateTreeInput captures the arguments of CreateTree. +type CreateTreeInput struct { + GoalID int64 + CreatedByAgent *int64 + CreatedByUser *int64 + Root TreeNode + InitialStatus string // defaults to StatusApproved (for auto-approved flows) + DefaultBilling string +} + +// CreateTree materializes a tree of tasks under a goal in a single +// transaction. Ancestry is denormalized at create time. Returns the +// root task id and the flat list of all created ids in insertion order. +func (s *Service) CreateTree(ctx context.Context, in CreateTreeInput) (rootTaskID int64, allIDs []int64, err error) { + if in.InitialStatus == "" { + in.InitialStatus = StatusApproved + } + tx, err := s.store.DB().BeginTx(ctx, nil) + if err != nil { + return 0, nil, err + } + defer func() { + if err != nil { + _ = tx.Rollback() + } + }() + + var walk func(node TreeNode, parentID *int64, depth int, ancestry []AncestryNode) (int64, error) + walk = func(node TreeNode, parentID *int64, depth int, ancestry []AncestryNode) (int64, error) { + billing := node.BillingCode + if billing == "" { + billing = in.DefaultBilling + } + t := &Task{ + GoalID: in.GoalID, + ParentTaskID: parentID, + Ancestry: ancestry, + Depth: depth, + Title: node.Title, + Description: node.Description, + AcceptanceCriteria: node.AcceptanceCriteria, + CreatedByAgentID: in.CreatedByAgent, + CreatedByUserID: in.CreatedByUser, + Status: in.InitialStatus, + BillingCode: billing, + BudgetTokens: node.BudgetTokens, + BudgetDollarsCents: node.BudgetDollarsCents, + VerifierConfig: node.VerifierConfig, + HeartbeatConfig: node.HeartbeatConfig, + } + id, err := s.store.Insert(ctx, tx, t) + if err != nil { + return 0, err + } + allIDs = append(allIDs, id) + + if len(node.Children) > 0 { + childAncestry := append([]AncestryNode(nil), ancestry...) + childAncestry = append(childAncestry, AncestryNode{ + ID: id, + Title: node.Title, + AcceptanceCriteria: node.AcceptanceCriteria, + }) + for _, child := range node.Children { + if _, err := walk(child, &id, depth+1, childAncestry); err != nil { + return 0, err + } + } + } + return id, nil + } + + rootTaskID, err = walk(in.Root, nil, 0, nil) + if err != nil { + return 0, nil, err + } + if err = tx.Commit(); err != nil { + return 0, nil, err + } + s.logger.Info("task tree created", "goal_id", in.GoalID, "root_task_id", rootTaskID, "total", len(allIDs)) + return rootTaskID, allIDs, nil +} + +// Claim atomically locks a task to an agent. +func (s *Service) Claim(ctx context.Context, taskID, agentID int64, claimMessageID *int64) error { + return s.store.ClaimAtomic(ctx, taskID, agentID, claimMessageID) +} + +// Transition moves a task through the state machine. +func (s *Service) Transition(ctx context.Context, taskID int64, newStatus string, extras Extras) error { + t, err := s.store.Get(ctx, taskID) + if err != nil { + return err + } + if !legalTransition(t.Status, newStatus) { + return fmt.Errorf("%w: %s → %s", ErrIllegalTransition, t.Status, newStatus) + } + return s.store.TransitionStatus(ctx, taskID, newStatus, extras) +} + +// Get exposes the store's Get. +func (s *Service) Get(ctx context.Context, id int64) (*Task, error) { + return s.store.Get(ctx, id) +} + +// ListByGoal exposes the store's ListByGoal. +func (s *Service) ListByGoal(ctx context.Context, goalID int64) ([]*Task, error) { + return s.store.ListByGoal(ctx, goalID) +} + +// AddSpend is used by the reactor post-run to increment leaf cost. +func (s *Service) AddSpend(ctx context.Context, taskID, tokens, dollarsCents int64) error { + return s.store.AddSpend(ctx, taskID, tokens, dollarsCents) +} + +// RollupCosts exposes the store's recursive CTE. +func (s *Service) RollupCosts(ctx context.Context, rootTaskID int64) (tokens, dollarsCents int64, count int, err error) { + return s.store.RollupCosts(ctx, rootTaskID) +} + +// RollupByBillingCode exposes the per-billing-code rollup. +func (s *Service) RollupByBillingCode(ctx context.Context, rootTaskID int64) (map[string]Spend, error) { + return s.store.RollupByBillingCode(ctx, rootTaskID) +} + +// legalTransition encodes the task state machine. +func legalTransition(from, to string) bool { + if to == StatusCancelled { + return from != StatusDone && from != StatusFailed && from != StatusCancelled + } + switch from { + case StatusProposed: + return to == StatusApproved + case StatusApproved: + return to == StatusClaimed + case StatusClaimed: + return to == StatusInProgress || to == StatusAwaitingVerification + case StatusInProgress: + return to == StatusAwaitingVerification + case StatusAwaitingVerification: + return to == StatusDone || to == StatusFailed + } + return false +} diff --git a/internal/goaltasks/service_test.go b/internal/goaltasks/service_test.go new file mode 100644 index 0000000..04e8951 --- /dev/null +++ b/internal/goaltasks/service_test.go @@ -0,0 +1,287 @@ +package goaltasks + +import ( + "context" + "database/sql" + "log/slog" + "strings" + "sync" + "sync/atomic" + "testing" + + _ "modernc.org/sqlite" + + "github.com/synapbus/synapbus/internal/storage" +) + +// testDB spins up an in-memory SQLite with all migrations applied and a +// minimal user/channel/agent/goal/goal_tasks set suitable for service tests. +func testDB(t *testing.T) (*sql.DB, int64, int64) { + t.Helper() + db, err := sql.Open("sqlite", "file::memory:?cache=shared&_foreign_keys=on&_pragma=busy_timeout(5000)") + if err != nil { + t.Fatalf("open: %v", err) + } + db.SetMaxOpenConns(1) + t.Cleanup(func() { _ = db.Close() }) + + ctx := context.Background() + if err := storage.RunMigrations(ctx, db); err != nil { + t.Fatalf("migrate: %v", err) + } + + if _, err := db.ExecContext(ctx, `INSERT INTO users (username, password_hash) VALUES ('algis', 'x')`); err != nil { + t.Fatalf("insert user: %v", err) + } + var userID int64 + if err := db.QueryRowContext(ctx, `SELECT id FROM users WHERE username='algis'`).Scan(&userID); err != nil { + t.Fatalf("get user: %v", err) + } + if _, err := db.ExecContext(ctx, ` + INSERT INTO channels (name, description, type, is_private, is_system, created_by) + VALUES ('goal-test', 'Test goal channel', 'blackboard', 1, 0, 'algis')`); err != nil { + t.Fatalf("insert channel: %v", err) + } + var channelID int64 + if err := db.QueryRowContext(ctx, `SELECT id FROM channels WHERE name='goal-test'`).Scan(&channelID); err != nil { + t.Fatalf("get channel: %v", err) + } + if _, err := db.ExecContext(ctx, ` + INSERT INTO goals (slug, title, description, owner_user_id, channel_id, status, max_spawn_depth) + VALUES ('test', 'Test', 'Desc', ?, ?, 'active', 3)`, userID, channelID); err != nil { + t.Fatalf("insert goal: %v", err) + } + var goalID int64 + if err := db.QueryRowContext(ctx, `SELECT id FROM goals WHERE slug='test'`).Scan(&goalID); err != nil { + t.Fatalf("get goal: %v", err) + } + return db, userID, goalID +} + +func insertTestAgent(t *testing.T, db *sql.DB, name string, ownerID int64) int64 { + t.Helper() + res, err := db.ExecContext(context.Background(), ` + INSERT INTO agents (name, type, capabilities, owner_id, api_key_hash, status) + VALUES (?, 'ai', '[]', ?, 'hash', 'active')`, name, ownerID) + if err != nil { + t.Fatalf("insert agent: %v", err) + } + id, _ := res.LastInsertId() + return id +} + +func TestCreateTree_AncestryAndDepth(t *testing.T) { + db, userID, goalID := testDB(t) + svc := NewService(NewStore(db), slog.Default()) + + root := TreeNode{ + Title: "root", + Description: "root desc", + Children: []TreeNode{ + { + Title: "child-1", + Description: "c1 desc", + Children: []TreeNode{ + {Title: "grandchild", Description: "gc desc"}, + }, + }, + {Title: "child-2", Description: "c2 desc"}, + }, + } + + rootID, allIDs, err := svc.CreateTree(context.Background(), CreateTreeInput{ + GoalID: goalID, + CreatedByUser: &userID, + Root: root, + }) + if err != nil { + t.Fatalf("CreateTree: %v", err) + } + if len(allIDs) != 4 { + t.Fatalf("expected 4 tasks, got %d", len(allIDs)) + } + + tasks, err := svc.ListByGoal(context.Background(), goalID) + if err != nil { + t.Fatalf("ListByGoal: %v", err) + } + byID := map[int64]*Task{} + for _, task := range tasks { + byID[task.ID] = task + } + + if r := byID[rootID]; r == nil || r.Depth != 0 || len(r.Ancestry) != 0 { + t.Errorf("root depth/ancestry wrong: %+v", r) + } + // grandchild should have two ancestors + var gc *Task + for _, task := range tasks { + if task.Title == "grandchild" { + gc = task + } + } + if gc == nil || gc.Depth != 2 || len(gc.Ancestry) != 2 { + t.Fatalf("grandchild depth/ancestry wrong: %+v", gc) + } + if gc.Ancestry[0].Title != "root" || gc.Ancestry[1].Title != "child-1" { + t.Errorf("ancestry chain wrong: %+v", gc.Ancestry) + } +} + +func TestCreateTree_AncestryOverflow(t *testing.T) { + db, userID, goalID := testDB(t) + svc := NewService(NewStore(db), slog.Default()) + // Huge title on an intermediate node — the grandchild's ancestry snapshot + // will contain this title and must exceed the 16 KB cap. + huge := strings.Repeat("x", 20000) + root := TreeNode{ + Title: "root", + Description: "d", + Children: []TreeNode{ + { + Title: huge, + Description: "d", + Children: []TreeNode{ + {Title: "victim", Description: "d"}, + }, + }, + }, + } + _, _, err := svc.CreateTree(context.Background(), CreateTreeInput{ + GoalID: goalID, + CreatedByUser: &userID, + Root: root, + }) + if err == nil { + t.Fatal("expected ancestry overflow error, got nil") + } +} + +func TestClaimAtomic_Race(t *testing.T) { + db, userID, goalID := testDB(t) + svc := NewService(NewStore(db), slog.Default()) + + // Create one task in approved state. + _, allIDs, err := svc.CreateTree(context.Background(), CreateTreeInput{ + GoalID: goalID, + CreatedByUser: &userID, + Root: TreeNode{Title: "solo", Description: "d"}, + InitialStatus: StatusApproved, + }) + if err != nil { + t.Fatalf("CreateTree: %v", err) + } + taskID := allIDs[0] + + // Two racing agents. + agent1 := insertTestAgent(t, db, "racer1", userID) + agent2 := insertTestAgent(t, db, "racer2", userID) + + const rounds = 50 + var oneWinsCount, alreadyClaimedCount int32 + for i := 0; i < rounds; i++ { + // Reset the task to approved + unassigned each round. + if _, err := db.ExecContext(context.Background(), + `UPDATE goal_tasks SET status='approved', assignee_agent_id=NULL, claimed_at=NULL WHERE id=?`, taskID); err != nil { + t.Fatalf("reset: %v", err) + } + var wg sync.WaitGroup + wg.Add(2) + for _, a := range []int64{agent1, agent2} { + agentID := a + go func() { + defer wg.Done() + err := svc.Claim(context.Background(), taskID, agentID, nil) + switch err { + case nil: + atomic.AddInt32(&oneWinsCount, 1) + case ErrAlreadyClaimed: + atomic.AddInt32(&alreadyClaimedCount, 1) + default: + t.Errorf("unexpected claim error: %v", err) + } + }() + } + wg.Wait() + } + if oneWinsCount != rounds { + t.Errorf("expected %d wins, got %d", rounds, oneWinsCount) + } + if alreadyClaimedCount != rounds { + t.Errorf("expected %d ErrAlreadyClaimed, got %d", rounds, alreadyClaimedCount) + } +} + +func TestRollupCosts(t *testing.T) { + db, userID, goalID := testDB(t) + svc := NewService(NewStore(db), slog.Default()) + + // Build: root → a, b; a → a1 + _, allIDs, err := svc.CreateTree(context.Background(), CreateTreeInput{ + GoalID: goalID, + CreatedByUser: &userID, + Root: TreeNode{ + Title: "root", Description: "d", + Children: []TreeNode{ + {Title: "a", Description: "d", Children: []TreeNode{ + {Title: "a1", Description: "d"}, + }}, + {Title: "b", Description: "d"}, + }, + }, + }) + if err != nil { + t.Fatalf("CreateTree: %v", err) + } + if len(allIDs) != 4 { + t.Fatalf("expected 4 tasks, got %d", len(allIDs)) + } + rootID := allIDs[0] + + // Spend on a1 and b (the leaves). + a1ID := allIDs[2] + bID := allIDs[3] + if err := svc.AddSpend(context.Background(), a1ID, 100, 50); err != nil { + t.Fatal(err) + } + if err := svc.AddSpend(context.Background(), bID, 200, 75); err != nil { + t.Fatal(err) + } + + tokens, dollars, count, err := svc.RollupCosts(context.Background(), rootID) + if err != nil { + t.Fatalf("RollupCosts: %v", err) + } + if tokens != 300 || dollars != 125 || count != 4 { + t.Errorf("rollup wrong: tokens=%d dollars=%d count=%d", tokens, dollars, count) + } +} + +func TestTransition_StateMachine(t *testing.T) { + db, userID, goalID := testDB(t) + svc := NewService(NewStore(db), slog.Default()) + + _, allIDs, err := svc.CreateTree(context.Background(), CreateTreeInput{ + GoalID: goalID, + CreatedByUser: &userID, + Root: TreeNode{Title: "solo", Description: "d"}, + InitialStatus: StatusApproved, + }) + if err != nil { + t.Fatal(err) + } + taskID := allIDs[0] + + // Legal: approved → claimed → in_progress → awaiting_verification → done + steps := []string{StatusClaimed, StatusInProgress, StatusAwaitingVerification, StatusDone} + for _, step := range steps { + if err := svc.Transition(context.Background(), taskID, step, Extras{}); err != nil { + t.Fatalf("transition to %s: %v", step, err) + } + } + + // Illegal: done → approved + if err := svc.Transition(context.Background(), taskID, StatusApproved, Extras{}); err == nil { + t.Error("expected illegal transition from done → approved") + } +} diff --git a/internal/goaltasks/store.go b/internal/goaltasks/store.go new file mode 100644 index 0000000..0182999 --- /dev/null +++ b/internal/goaltasks/store.go @@ -0,0 +1,293 @@ +package goaltasks + +import ( + "context" + "database/sql" + "encoding/json" + "errors" + "fmt" + "time" +) + +// Store is the SQLite-backed persistence for goal tasks. +type Store struct { + db *sql.DB +} + +// NewStore constructs a Store from a database handle. +func NewStore(db *sql.DB) *Store { + return &Store{db: db} +} + +// DB exposes the underlying handle for transactions. +func (s *Store) DB() *sql.DB { + return s.db +} + +// Insert writes a single task row. The caller is responsible for +// providing a valid ancestry and depth. +func (s *Store) Insert(ctx context.Context, tx *sql.Tx, t *Task) (int64, error) { + ancestryJSON, err := marshalAncestry(t.Ancestry) + if err != nil { + return 0, err + } + var verifierJSON, heartbeatJSON sql.NullString + if t.VerifierConfig != nil { + b, err := json.Marshal(t.VerifierConfig) + if err != nil { + return 0, err + } + verifierJSON = sql.NullString{String: string(b), Valid: true} + } + if t.HeartbeatConfig != nil { + b, err := json.Marshal(t.HeartbeatConfig) + if err != nil { + return 0, err + } + heartbeatJSON = sql.NullString{String: string(b), Valid: true} + } + + const q = ` + INSERT INTO goal_tasks + (goal_id, parent_task_id, ancestry_json, depth, title, description, acceptance_criteria, + created_by_agent_id, created_by_user_id, assignee_agent_id, status, + billing_code, budget_tokens, budget_dollars_cents, + heartbeat_config_json, verifier_config_json) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)` + + var res sql.Result + if tx != nil { + res, err = tx.ExecContext(ctx, q, + t.GoalID, t.ParentTaskID, ancestryJSON, t.Depth, t.Title, t.Description, t.AcceptanceCriteria, + t.CreatedByAgentID, t.CreatedByUserID, t.AssigneeAgentID, t.Status, + nullableString(t.BillingCode), t.BudgetTokens, t.BudgetDollarsCents, + heartbeatJSON, verifierJSON) + } else { + res, err = s.db.ExecContext(ctx, q, + t.GoalID, t.ParentTaskID, ancestryJSON, t.Depth, t.Title, t.Description, t.AcceptanceCriteria, + t.CreatedByAgentID, t.CreatedByUserID, t.AssigneeAgentID, t.Status, + nullableString(t.BillingCode), t.BudgetTokens, t.BudgetDollarsCents, + heartbeatJSON, verifierJSON) + } + if err != nil { + return 0, fmt.Errorf("insert goal_task: %w", err) + } + id, err := res.LastInsertId() + if err != nil { + return 0, err + } + t.ID = id + return id, nil +} + +// Get fetches a single task by id. +func (s *Store) Get(ctx context.Context, id int64) (*Task, error) { + return s.getOne(ctx, `SELECT `+cols+` FROM goal_tasks WHERE id = ?`, id) +} + +// ListByGoal returns all tasks under a goal in insertion order. +func (s *Store) ListByGoal(ctx context.Context, goalID int64) ([]*Task, error) { + rows, err := s.db.QueryContext(ctx, `SELECT `+cols+` FROM goal_tasks WHERE goal_id = ? ORDER BY id`, goalID) + if err != nil { + return nil, err + } + defer rows.Close() + var out []*Task + for rows.Next() { + t, err := scanTask(rows) + if err != nil { + return nil, err + } + out = append(out, t) + } + return out, rows.Err() +} + +// ClaimAtomic performs the optimistic-lock claim — the core concurrency +// primitive. Returns ErrAlreadyClaimed if the task is not in state +// `approved` and unassigned. +func (s *Store) ClaimAtomic(ctx context.Context, taskID, agentID int64, claimMessageID *int64) error { + now := time.Now().UTC() + res, err := s.db.ExecContext(ctx, ` + UPDATE goal_tasks + SET assignee_agent_id = ?, + status = ?, + claimed_at = ?, + claim_message_id = ? + WHERE id = ? + AND assignee_agent_id IS NULL + AND status = ?`, + agentID, StatusClaimed, now, claimMessageID, taskID, StatusApproved) + if err != nil { + return err + } + n, err := res.RowsAffected() + if err != nil { + return err + } + if n == 0 { + return ErrAlreadyClaimed + } + return nil +} + +// TransitionStatus unconditionally moves a task to a new status. The +// service layer is responsible for legality checks before calling this. +func (s *Store) TransitionStatus(ctx context.Context, taskID int64, newStatus string, extras Extras) error { + now := time.Now().UTC() + _, err := s.db.ExecContext(ctx, ` + UPDATE goal_tasks + SET status = ?, + started_at = COALESCE(started_at, CASE WHEN ? = 'in_progress' THEN ? ELSE NULL END), + completed_at = CASE WHEN ? IN ('done','failed','cancelled') THEN ? ELSE completed_at END, + failure_reason = COALESCE(?, failure_reason), + completion_message_id = COALESCE(?, completion_message_id) + WHERE id = ?`, + newStatus, newStatus, now, newStatus, now, + nullableString(extras.FailureReason), + extras.CompletionMessageID, + taskID) + return err +} + +// AddSpend increments a leaf task's spend counters after a harness run. +func (s *Store) AddSpend(ctx context.Context, taskID int64, tokens, dollarsCents int64) error { + _, err := s.db.ExecContext(ctx, + `UPDATE goal_tasks + SET spent_tokens = spent_tokens + ?, + spent_dollars_cents = spent_dollars_cents + ? + WHERE id = ?`, tokens, dollarsCents, taskID) + return err +} + +// RollupCosts returns the total spend under a task subtree (inclusive). +func (s *Store) RollupCosts(ctx context.Context, rootTaskID int64) (tokens, dollarsCents int64, count int, err error) { + row := s.db.QueryRowContext(ctx, ` + WITH RECURSIVE subtree(id) AS ( + SELECT id FROM goal_tasks WHERE id = ? + UNION ALL + SELECT t.id FROM goal_tasks t + JOIN subtree s ON t.parent_task_id = s.id + ) + SELECT COALESCE(SUM(spent_tokens), 0), + COALESCE(SUM(spent_dollars_cents), 0), + COUNT(*) + FROM goal_tasks WHERE id IN subtree`, rootTaskID) + err = row.Scan(&tokens, &dollarsCents, &count) + return +} + +// RollupByBillingCode returns spend grouped by billing code within a subtree. +func (s *Store) RollupByBillingCode(ctx context.Context, rootTaskID int64) (map[string]Spend, error) { + rows, err := s.db.QueryContext(ctx, ` + WITH RECURSIVE subtree(id) AS ( + SELECT id FROM goal_tasks WHERE id = ? + UNION ALL + SELECT t.id FROM goal_tasks t + JOIN subtree s ON t.parent_task_id = s.id + ) + SELECT COALESCE(billing_code, ''), SUM(spent_tokens), SUM(spent_dollars_cents) + FROM goal_tasks WHERE id IN subtree + GROUP BY billing_code`, rootTaskID) + if err != nil { + return nil, err + } + defer rows.Close() + out := map[string]Spend{} + for rows.Next() { + var code string + var tokens, dollars int64 + if err := rows.Scan(&code, &tokens, &dollars); err != nil { + return nil, err + } + out[code] = Spend{Tokens: tokens, DollarsCents: dollars} + } + return out, rows.Err() +} + +// Extras carries optional fields for TransitionStatus. +type Extras struct { + FailureReason string + CompletionMessageID *int64 +} + +// Spend is a tokens+dollars pair for rollups. +type Spend struct { + Tokens int64 + DollarsCents int64 +} + +// --- internal helpers --- + +const cols = `id, goal_id, parent_task_id, ancestry_json, depth, title, description, acceptance_criteria, + created_by_agent_id, created_by_user_id, assignee_agent_id, status, + billing_code, budget_tokens, budget_dollars_cents, spent_tokens, spent_dollars_cents, + heartbeat_config_json, verifier_config_json, + origin_message_id, claim_message_id, completion_message_id, failure_reason, + created_at, approved_at, claimed_at, started_at, completed_at` + +type rowLike interface { + Scan(dest ...any) error +} + +func (s *Store) getOne(ctx context.Context, q string, args ...any) (*Task, error) { + row := s.db.QueryRowContext(ctx, q, args...) + t, err := scanTask(row) + if errors.Is(err, sql.ErrNoRows) { + return nil, ErrTaskNotFound + } + return t, err +} + +func scanTask(r rowLike) (*Task, error) { + t := &Task{} + var ( + billing sql.NullString + ancestry string + verifierJSON sql.NullString + heartbeatJSON sql.NullString + failureReason sql.NullString + ) + err := r.Scan( + &t.ID, &t.GoalID, &t.ParentTaskID, &ancestry, &t.Depth, &t.Title, &t.Description, &t.AcceptanceCriteria, + &t.CreatedByAgentID, &t.CreatedByUserID, &t.AssigneeAgentID, &t.Status, + &billing, &t.BudgetTokens, &t.BudgetDollarsCents, &t.SpentTokens, &t.SpentDollarsCents, + &heartbeatJSON, &verifierJSON, + &t.OriginMessageID, &t.ClaimMessageID, &t.CompletionMessageID, &failureReason, + &t.CreatedAt, &t.ApprovedAt, &t.ClaimedAt, &t.StartedAt, &t.CompletedAt, + ) + if err != nil { + return nil, err + } + if billing.Valid { + t.BillingCode = billing.String + } + if failureReason.Valid { + t.FailureReason = failureReason.String + } + if heartbeatJSON.Valid && heartbeatJSON.String != "" { + hc := &HeartbeatConfig{} + if err := json.Unmarshal([]byte(heartbeatJSON.String), hc); err == nil { + t.HeartbeatConfig = hc + } + } + if verifierJSON.Valid && verifierJSON.String != "" { + vc := &VerifierConfig{} + if err := json.Unmarshal([]byte(verifierJSON.String), vc); err == nil { + t.VerifierConfig = vc + } + } + nodes, err := unmarshalAncestry(ancestry) + if err != nil { + return nil, err + } + t.Ancestry = nodes + return t, nil +} + +func nullableString(s string) sql.NullString { + if s == "" { + return sql.NullString{Valid: false} + } + return sql.NullString{String: s, Valid: true} +} diff --git a/internal/goaltasks/types.go b/internal/goaltasks/types.go new file mode 100644 index 0000000..1ffec42 --- /dev/null +++ b/internal/goaltasks/types.go @@ -0,0 +1,138 @@ +// Package goaltasks implements the work-task tree rooted in a goal. +// Table name is goal_tasks (not tasks) because the legacy channel +// task-auction feature already owns the tasks table. +// +// Tasks are single-assignee, atomically claimable, and carry a +// denormalized goal-ancestry snapshot so every subprocess run can +// see the full root-to-parent context without recursive queries. +package goaltasks + +import ( + "encoding/json" + "errors" + "time" +) + +// Task status values. +const ( + StatusProposed = "proposed" + StatusApproved = "approved" + StatusClaimed = "claimed" + StatusInProgress = "in_progress" + StatusAwaitingVerification = "awaiting_verification" + StatusDone = "done" + StatusFailed = "failed" + StatusCancelled = "cancelled" +) + +// Verifier kinds. +const ( + VerifierKindAuto = "auto" + VerifierKindPeer = "peer" + VerifierKindCommand = "command" +) + +// AncestryNode is one entry in a task's denormalized ancestry chain, +// copied from the root down to the parent at create time. +type AncestryNode struct { + ID int64 `json:"id"` + Title string `json:"title"` + AcceptanceCriteria string `json:"acceptance_criteria,omitempty"` +} + +// VerifierConfig describes how to verify a task once the assignee +// reports it complete. Exactly one kind is present. +type VerifierConfig struct { + Kind string `json:"kind"` + AgentID int64 `json:"agent_id,omitempty"` + Cmd string `json:"cmd,omitempty"` + Cwd string `json:"cwd,omitempty"` + TimeoutSec int `json:"timeout_sec,omitempty"` +} + +// HeartbeatConfig controls how the reactor wakes the assignee. +type HeartbeatConfig struct { + Source string `json:"source"` + IntervalSec int `json:"interval_sec,omitempty"` +} + +// Task is a node in a goal's task tree. +type Task struct { + ID int64 + GoalID int64 + ParentTaskID *int64 + Ancestry []AncestryNode + Depth int + Title string + Description string + AcceptanceCriteria string + CreatedByAgentID *int64 + CreatedByUserID *int64 + AssigneeAgentID *int64 + Status string + BillingCode string + BudgetTokens *int64 + BudgetDollarsCents *int64 + SpentTokens int64 + SpentDollarsCents int64 + HeartbeatConfig *HeartbeatConfig + VerifierConfig *VerifierConfig + OriginMessageID *int64 + ClaimMessageID *int64 + CompletionMessageID *int64 + FailureReason string + CreatedAt time.Time + ApprovedAt *time.Time + ClaimedAt *time.Time + StartedAt *time.Time + CompletedAt *time.Time +} + +// TreeNode is the input shape for CreateTree — a recursive task spec. +type TreeNode struct { + Title string `json:"title"` + Description string `json:"description"` + AcceptanceCriteria string `json:"acceptance_criteria,omitempty"` + BillingCode string `json:"billing_code,omitempty"` + BudgetTokens *int64 `json:"budget_tokens,omitempty"` + BudgetDollarsCents *int64 `json:"budget_dollars_cents,omitempty"` + VerifierConfig *VerifierConfig `json:"verifier_config,omitempty"` + HeartbeatConfig *HeartbeatConfig `json:"heartbeat_config,omitempty"` + Children []TreeNode `json:"children,omitempty"` +} + +// MaxAncestryBytes caps the denormalized ancestry blob on any single task. +const MaxAncestryBytes = 16 * 1024 + +// marshalAncestry serializes an ancestry chain. Returns ErrAncestryOverflow +// if the result exceeds MaxAncestryBytes. +func marshalAncestry(nodes []AncestryNode) (string, error) { + b, err := json.Marshal(nodes) + if err != nil { + return "", err + } + if len(b) > MaxAncestryBytes { + return "", ErrAncestryOverflow + } + return string(b), nil +} + +// unmarshalAncestry parses the stored JSON back into a chain. +func unmarshalAncestry(s string) ([]AncestryNode, error) { + if s == "" || s == "[]" { + return nil, nil + } + var nodes []AncestryNode + if err := json.Unmarshal([]byte(s), &nodes); err != nil { + return nil, err + } + return nodes, nil +} + +// Sentinel errors. +var ( + ErrTaskNotFound = errors.New("task not found") + ErrAlreadyClaimed = errors.New("task already claimed by another agent") + ErrIllegalTransition = errors.New("illegal task status transition") + ErrAncestryOverflow = errors.New("task ancestry exceeds 16 KB cap") +) diff --git a/internal/secrets/injector.go b/internal/secrets/injector.go new file mode 100644 index 0000000..b7ac7cf --- /dev/null +++ b/internal/secrets/injector.go @@ -0,0 +1,138 @@ +package secrets + +import ( + "context" + "database/sql" + "fmt" + "strings" +) + +// BuildEnvMap returns a name→plaintext map for all active secrets visible to +// the given user/agent/task, with scope precedence user < agent < task. Pass +// 0 for any scope id you wish to skip. last_used_at is bumped to +// CURRENT_TIMESTAMP for every secret returned. +// +// The returned map is intended to be merged into a subprocess env. Callers +// must treat values as sensitive and never log them. +func (s *Store) BuildEnvMap(ctx context.Context, userID, agentID, taskID int64) (map[string]string, error) { + // Build (scope_type, scope_id, precedence) tuples; higher precedence wins. + type scopeRow struct { + typ string + id int64 + precedence int + } + var scopes []scopeRow + if userID > 0 { + scopes = append(scopes, scopeRow{ScopeUser, userID, 1}) + } + if agentID > 0 { + scopes = append(scopes, scopeRow{ScopeAgent, agentID, 2}) + } + if taskID > 0 { + scopes = append(scopes, scopeRow{ScopeTask, taskID, 3}) + } + if len(scopes) == 0 { + return map[string]string{}, nil + } + + var ( + parts []string + args []any + ) + for _, sc := range scopes { + parts = append(parts, "(scope_type = ? AND scope_id = ?)") + args = append(args, sc.typ, sc.id) + } + + query := `SELECT id, name, scope_type, value_blob + FROM secrets + WHERE revoked_at IS NULL + AND (` + strings.Join(parts, " OR ") + `)` + + rows, err := s.db.QueryContext(ctx, query, args...) + if err != nil { + return nil, fmt.Errorf("secrets: env query: %w", err) + } + defer rows.Close() + + type winner struct { + id int64 + precedence int + value string + } + winners := make(map[string]winner) + var touched []int64 + + for rows.Next() { + var ( + id int64 + name string + scopeType string + blob []byte + ) + if err := rows.Scan(&id, &name, &scopeType, &blob); err != nil { + return nil, fmt.Errorf("secrets: env scan: %w", err) + } + var prec int + switch scopeType { + case ScopeUser: + prec = 1 + case ScopeAgent: + prec = 2 + case ScopeTask: + prec = 3 + default: + continue + } + existing, ok := winners[name] + if ok && existing.precedence >= prec { + continue + } + plain, err := s.decrypt(blob) + if err != nil { + return nil, fmt.Errorf("secrets: decrypt %q: %w", name, err) + } + winners[name] = winner{id: id, precedence: prec, value: string(plain)} + } + if err := rows.Err(); err != nil { + return nil, err + } + + out := make(map[string]string, len(winners)) + for name, w := range winners { + out[name] = w.value + touched = append(touched, w.id) + } + + if len(touched) > 0 { + if err := s.bumpLastUsed(ctx, touched); err != nil { + // Non-fatal for the caller's env, but log it. + s.logger.Warn("failed to bump last_used_at", "error", err, "ids", touched) + } + } + return out, nil +} + +// bumpLastUsed updates last_used_at for the given secret ids in a single +// statement. +func (s *Store) bumpLastUsed(ctx context.Context, ids []int64) error { + if len(ids) == 0 { + return nil + } + placeholders := make([]string, len(ids)) + args := make([]any, len(ids)) + for i, id := range ids { + placeholders[i] = "?" + args[i] = id + } + query := `UPDATE secrets SET last_used_at = CURRENT_TIMESTAMP WHERE id IN (` + + strings.Join(placeholders, ",") + `)` + _, err := s.db.ExecContext(ctx, query, args...) + if err != nil { + return err + } + return nil +} + +// Compile-time guard that *sql.DB satisfies the methods we rely on. +var _ = (*sql.DB)(nil) diff --git a/internal/secrets/store.go b/internal/secrets/store.go new file mode 100644 index 0000000..5ead6fe --- /dev/null +++ b/internal/secrets/store.go @@ -0,0 +1,360 @@ +package secrets + +import ( + "context" + "crypto/rand" + "database/sql" + "errors" + "fmt" + "io" + "log/slog" + "os" + "path/filepath" + "strings" + "time" + + "golang.org/x/crypto/nacl/secretbox" +) + +const ( + // nonceSize is the NaCl secretbox nonce size in bytes. + nonceSize = 24 + // keySize is the NaCl secretbox key size in bytes. + keySize = 32 + // masterKeyFilename is the file inside the data dir holding the 32-byte + // master key. Stored with 0600 permissions. + masterKeyFilename = "secrets.key" +) + +// Store provides CRUD over encrypted secrets backed by SQLite. +type Store struct { + db *sql.DB + logger *slog.Logger + masterKey [keySize]byte +} + +// NewStore constructs a Store. It bootstraps the master key from +// /secrets.key, generating a fresh 32-byte key (0600 perms) if the +// file does not yet exist. +func NewStore(db *sql.DB, dataDir string, logger *slog.Logger) (*Store, error) { + if logger == nil { + logger = slog.Default() + } + if db == nil { + return nil, fmt.Errorf("secrets: db is required") + } + if dataDir == "" { + return nil, fmt.Errorf("secrets: dataDir is required") + } + + key, err := loadOrCreateMasterKey(dataDir, logger) + if err != nil { + return nil, err + } + + s := &Store{db: db, logger: logger} + copy(s.masterKey[:], key) + return s, nil +} + +func loadOrCreateMasterKey(dataDir string, logger *slog.Logger) ([]byte, error) { + if err := os.MkdirAll(dataDir, 0o700); err != nil { + return nil, fmt.Errorf("%w: mkdir %s: %v", ErrMasterKeyMissing, dataDir, err) + } + path := filepath.Join(dataDir, masterKeyFilename) + + data, err := os.ReadFile(path) + if err == nil { + if len(data) != keySize { + return nil, fmt.Errorf("%w: %s has wrong size %d (want %d)", ErrMasterKeyMissing, path, len(data), keySize) + } + return data, nil + } + if !errors.Is(err, os.ErrNotExist) { + return nil, fmt.Errorf("%w: read %s: %v", ErrMasterKeyMissing, path, err) + } + + // Generate a new key. + buf := make([]byte, keySize) + if _, err := io.ReadFull(rand.Reader, buf); err != nil { + return nil, fmt.Errorf("%w: generate: %v", ErrMasterKeyMissing, err) + } + if err := os.WriteFile(path, buf, 0o600); err != nil { + return nil, fmt.Errorf("%w: write %s: %v", ErrMasterKeyMissing, path, err) + } + logger.Info("generated new secrets master key", "path", path) + return buf, nil +} + +// Set encrypts value and writes a new secret row. If an active secret with the +// same (scope_type, scope_id, name) already exists, it is revoked first so a +// new immutable history row can be inserted. +func (s *Store) Set(ctx context.Context, name, scopeType string, scopeID, createdBy int64, value string) (*Secret, error) { + clean, err := sanitizeName(name) + if err != nil { + return nil, err + } + if !validScope(scopeType) { + return nil, fmt.Errorf("secrets: invalid scope_type %q", scopeType) + } + + blob, err := s.encrypt([]byte(value)) + if err != nil { + return nil, fmt.Errorf("secrets: encrypt: %w", err) + } + + tx, err := s.db.BeginTx(ctx, nil) + if err != nil { + return nil, fmt.Errorf("secrets: begin tx: %w", err) + } + defer tx.Rollback() + + // Revoke any existing active row for the same (scope, name). + if _, err := tx.ExecContext(ctx, + `UPDATE secrets + SET revoked_at = CURRENT_TIMESTAMP + WHERE name = ? + AND scope_type = ? + AND scope_id = ? + AND revoked_at IS NULL`, + clean, scopeType, scopeID, + ); err != nil { + return nil, fmt.Errorf("secrets: revoke previous: %w", err) + } + + res, err := tx.ExecContext(ctx, + `INSERT INTO secrets (name, scope_type, scope_id, value_blob, created_by) + VALUES (?, ?, ?, ?, ?)`, + clean, scopeType, scopeID, blob, createdBy, + ) + if err != nil { + return nil, fmt.Errorf("secrets: insert: %w", err) + } + id, err := res.LastInsertId() + if err != nil { + return nil, fmt.Errorf("secrets: last insert id: %w", err) + } + + if err := tx.Commit(); err != nil { + return nil, fmt.Errorf("secrets: commit: %w", err) + } + + // Re-read to populate created_at consistently. + row := s.db.QueryRowContext(ctx, + `SELECT id, name, scope_type, scope_id, created_by, created_at, revoked_at, last_used_at + FROM secrets WHERE id = ?`, + id, + ) + sec, err := scanSecret(row) + if err != nil { + return nil, fmt.Errorf("secrets: read back: %w", err) + } + s.logger.Info("secret set", + "id", sec.ID, + "name", sec.Name, + "scope_type", sec.ScopeType, + "scope_id", sec.ScopeID, + ) + return sec, nil +} + +// Get decrypts and returns the plaintext for the active secret matching +// (name, scope_type, scope_id). Caller must treat the returned string as +// sensitive — never log it. +func (s *Store) Get(ctx context.Context, name, scopeType string, scopeID int64) (string, error) { + clean, err := sanitizeName(name) + if err != nil { + return "", err + } + if !validScope(scopeType) { + return "", fmt.Errorf("secrets: invalid scope_type %q", scopeType) + } + + var blob []byte + err = s.db.QueryRowContext(ctx, + `SELECT value_blob + FROM secrets + WHERE name = ? + AND scope_type = ? + AND scope_id = ? + AND revoked_at IS NULL`, + clean, scopeType, scopeID, + ).Scan(&blob) + if err != nil { + if errors.Is(err, sql.ErrNoRows) { + return "", ErrNotFound + } + return "", fmt.Errorf("secrets: query: %w", err) + } + + plain, err := s.decrypt(blob) + if err != nil { + return "", fmt.Errorf("secrets: decrypt: %w", err) + } + return string(plain), nil +} + +// List returns Info entries for all active secrets in the given scopes. +// Values are never returned. Order is stable: by (scope_type, scope_id, name). +func (s *Store) List(ctx context.Context, scopes []Scope) ([]Info, error) { + if len(scopes) == 0 { + return []Info{}, nil + } + + // Build dynamic IN clause: (scope_type=? AND scope_id=?) OR (...) + var ( + parts []string + args []any + ) + for _, sc := range scopes { + if !validScope(sc.Type) { + return nil, fmt.Errorf("secrets: invalid scope_type %q", sc.Type) + } + parts = append(parts, "(scope_type = ? AND scope_id = ?)") + args = append(args, sc.Type, sc.ID) + } + query := `SELECT name, scope_type, scope_id, last_used_at + FROM secrets + WHERE revoked_at IS NULL + AND (` + strings.Join(parts, " OR ") + `) + ORDER BY scope_type, scope_id, name` + + rows, err := s.db.QueryContext(ctx, query, args...) + if err != nil { + return nil, fmt.Errorf("secrets: list query: %w", err) + } + defer rows.Close() + + var out []Info + for rows.Next() { + var ( + info Info + lastUsed sql.NullTime + ) + if err := rows.Scan(&info.Name, &info.ScopeType, &info.ScopeID, &lastUsed); err != nil { + return nil, fmt.Errorf("secrets: scan: %w", err) + } + info.Available = true + if lastUsed.Valid { + t := lastUsed.Time + info.LastUsedAt = &t + } + out = append(out, info) + } + if err := rows.Err(); err != nil { + return nil, err + } + if out == nil { + out = []Info{} + } + return out, nil +} + +// Revoke marks a secret revoked by primary key. It is idempotent in the sense +// that a non-existent row returns ErrNotFound and an already-revoked row +// returns ErrAlreadyRevoked. +func (s *Store) Revoke(ctx context.Context, id int64) error { + var revokedAt sql.NullTime + err := s.db.QueryRowContext(ctx, + `SELECT revoked_at FROM secrets WHERE id = ?`, id, + ).Scan(&revokedAt) + if err != nil { + if errors.Is(err, sql.ErrNoRows) { + return ErrNotFound + } + return fmt.Errorf("secrets: lookup: %w", err) + } + if revokedAt.Valid { + return ErrAlreadyRevoked + } + + if _, err := s.db.ExecContext(ctx, + `UPDATE secrets SET revoked_at = CURRENT_TIMESTAMP WHERE id = ?`, id, + ); err != nil { + return fmt.Errorf("secrets: revoke: %w", err) + } + s.logger.Info("secret revoked", "id", id) + return nil +} + +// encrypt returns nonce(24) || ciphertext. +func (s *Store) encrypt(plain []byte) ([]byte, error) { + var nonce [nonceSize]byte + if _, err := io.ReadFull(rand.Reader, nonce[:]); err != nil { + return nil, err + } + out := make([]byte, 0, nonceSize+len(plain)+secretbox.Overhead) + out = append(out, nonce[:]...) + out = secretbox.Seal(out, plain, &nonce, &s.masterKey) + return out, nil +} + +// decrypt parses nonce(24) || ciphertext and returns the plaintext. +func (s *Store) decrypt(blob []byte) ([]byte, error) { + if len(blob) < nonceSize+secretbox.Overhead { + return nil, fmt.Errorf("secrets: ciphertext too short (%d bytes)", len(blob)) + } + var nonce [nonceSize]byte + copy(nonce[:], blob[:nonceSize]) + plain, ok := secretbox.Open(nil, blob[nonceSize:], &nonce, &s.masterKey) + if !ok { + return nil, fmt.Errorf("secrets: decryption failed (key mismatch or corruption)") + } + return plain, nil +} + +// sanitizeName uppercases raw and validates that it contains only [A-Z0-9_]. +// Empty input is rejected. Lowercase letters are folded to uppercase before +// validation so callers may pass either case. +func sanitizeName(raw string) (string, error) { + if raw == "" { + return "", ErrInvalidName + } + upper := strings.ToUpper(raw) + for i := 0; i < len(upper); i++ { + c := upper[i] + switch { + case c >= 'A' && c <= 'Z': + case c >= '0' && c <= '9': + case c == '_': + default: + return "", ErrInvalidName + } + } + return upper, nil +} + +func validScope(t string) bool { + switch t { + case ScopeUser, ScopeAgent, ScopeTask: + return true + default: + return false + } +} + +// scanSecret scans a single secret row from a *sql.Row. +func scanSecret(row *sql.Row) (*Secret, error) { + var ( + s Secret + revoked sql.NullTime + lastUsed sql.NullTime + createdAt time.Time + ) + if err := row.Scan(&s.ID, &s.Name, &s.ScopeType, &s.ScopeID, &s.CreatedBy, &createdAt, &revoked, &lastUsed); err != nil { + if errors.Is(err, sql.ErrNoRows) { + return nil, ErrNotFound + } + return nil, err + } + s.CreatedAt = createdAt + if revoked.Valid { + t := revoked.Time + s.RevokedAt = &t + } + if lastUsed.Valid { + t := lastUsed.Time + s.LastUsedAt = &t + } + return &s, nil +} diff --git a/internal/secrets/store_test.go b/internal/secrets/store_test.go new file mode 100644 index 0000000..bd35e51 --- /dev/null +++ b/internal/secrets/store_test.go @@ -0,0 +1,311 @@ +package secrets_test + +import ( + "context" + "database/sql" + "io" + "log/slog" + "reflect" + "strings" + "testing" + + _ "modernc.org/sqlite" + + "github.com/synapbus/synapbus/internal/secrets" + "github.com/synapbus/synapbus/internal/storage" +) + +// newTestStore opens an in-memory SQLite DB, runs all migrations, seeds a +// user, and returns a ready-to-use *secrets.Store. +func newTestStore(t *testing.T) (*secrets.Store, *sql.DB, int64) { + t.Helper() + + db, err := sql.Open("sqlite", "file::memory:?cache=shared&_foreign_keys=on") + if err != nil { + t.Fatalf("open sqlite: %v", err) + } + t.Cleanup(func() { _ = db.Close() }) + + // Single connection so the in-memory DB persists across queries. + db.SetMaxOpenConns(1) + + ctx := context.Background() + if err := storage.RunMigrations(ctx, db); err != nil { + t.Fatalf("run migrations: %v", err) + } + + // Seed a user so created_by FK is satisfied. + res, err := db.ExecContext(ctx, + `INSERT INTO users (username, password_hash, owner_id, role) + VALUES ('tester', 'x', 1, 'admin')`, + ) + if err != nil { + // Schema may differ slightly across migrations; try the minimal column set. + res, err = db.ExecContext(ctx, + `INSERT INTO users (username, password_hash) VALUES ('tester', 'x')`, + ) + if err != nil { + t.Fatalf("seed user: %v", err) + } + } + uid, err := res.LastInsertId() + if err != nil { + t.Fatalf("last insert id: %v", err) + } + + store, err := secrets.NewStore(db, t.TempDir(), slog.New(slog.NewTextHandler(io.Discard, nil))) + if err != nil { + t.Fatalf("NewStore: %v", err) + } + return store, db, uid +} + +func TestEncryptDecryptRoundTrip(t *testing.T) { + store, _, uid := newTestStore(t) + ctx := context.Background() + + cases := []struct { + name string + key string + value string + }{ + {"simple", "API_KEY", "sk-abc123"}, + {"empty", "EMPTY", ""}, + {"unicode", "GREETING", "héllo, wörld"}, + {"long", "LONG", strings.Repeat("x", 8192)}, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + sec, err := store.Set(ctx, tc.key, secrets.ScopeUser, uid, uid, tc.value) + if err != nil { + t.Fatalf("Set: %v", err) + } + if sec.Name != strings.ToUpper(tc.key) { + t.Errorf("name: got %q want %q", sec.Name, strings.ToUpper(tc.key)) + } + got, err := store.Get(ctx, tc.key, secrets.ScopeUser, uid) + if err != nil { + t.Fatalf("Get: %v", err) + } + if got != tc.value { + t.Errorf("Get value: got %q want %q", got, tc.value) + } + }) + } +} + +func TestSanitizeNameViaSet(t *testing.T) { + store, _, uid := newTestStore(t) + ctx := context.Background() + + cases := []struct { + input string + want string // expected stored name; "" means expect ErrInvalidName + wantErr bool + }{ + {"api_key", "API_KEY", false}, + {"OPENAI_KEY", "OPENAI_KEY", false}, + {"Token1", "TOKEN1", false}, + {"FOO_2", "FOO_2", false}, + {"", "", true}, + {"BAD-NAME", "", true}, + {"with space", "", true}, + {"dot.name", "", true}, + {"unicode_é", "", true}, + } + + for _, tc := range cases { + t.Run(tc.input, func(t *testing.T) { + sec, err := store.Set(ctx, tc.input, secrets.ScopeUser, uid, uid, "v") + if tc.wantErr { + if err == nil { + t.Fatalf("expected error for input %q, got nil", tc.input) + } + return + } + if err != nil { + t.Fatalf("Set: %v", err) + } + if sec.Name != tc.want { + t.Errorf("name: got %q want %q", sec.Name, tc.want) + } + }) + } +} + +func TestListHidesValues(t *testing.T) { + store, _, uid := newTestStore(t) + ctx := context.Background() + + if _, err := store.Set(ctx, "SECRET1", secrets.ScopeUser, uid, uid, "value-one"); err != nil { + t.Fatalf("Set: %v", err) + } + + infos, err := store.List(ctx, []secrets.Scope{{Type: secrets.ScopeUser, ID: uid}}) + if err != nil { + t.Fatalf("List: %v", err) + } + if len(infos) != 1 { + t.Fatalf("len: got %d want 1", len(infos)) + } + + // Reflectively assert that Info has no field named like a value. + tInfo := reflect.TypeOf(infos[0]) + for i := 0; i < tInfo.NumField(); i++ { + f := tInfo.Field(i) + lower := strings.ToLower(f.Name) + if lower == "value" || lower == "plaintext" || lower == "secret" { + t.Errorf("Info exposes sensitive field %q", f.Name) + } + } + + if infos[0].Name != "SECRET1" || !infos[0].Available { + t.Errorf("unexpected info: %+v", infos[0]) + } +} + +func TestRevokeHidesFromList(t *testing.T) { + store, _, uid := newTestStore(t) + ctx := context.Background() + + sec, err := store.Set(ctx, "TO_REVOKE", secrets.ScopeUser, uid, uid, "v") + if err != nil { + t.Fatalf("Set: %v", err) + } + + if err := store.Revoke(ctx, sec.ID); err != nil { + t.Fatalf("Revoke: %v", err) + } + + infos, err := store.List(ctx, []secrets.Scope{{Type: secrets.ScopeUser, ID: uid}}) + if err != nil { + t.Fatalf("List: %v", err) + } + for _, i := range infos { + if i.Name == "TO_REVOKE" { + t.Fatalf("revoked secret should not appear in List") + } + } + + if _, err := store.Get(ctx, "TO_REVOKE", secrets.ScopeUser, uid); err == nil { + t.Fatalf("Get should fail for revoked secret") + } + + env, err := store.BuildEnvMap(ctx, uid, 0, 0) + if err != nil { + t.Fatalf("BuildEnvMap: %v", err) + } + if _, ok := env["TO_REVOKE"]; ok { + t.Fatalf("revoked secret leaked into env map") + } + + // Revoking again should return ErrAlreadyRevoked. + if err := store.Revoke(ctx, sec.ID); err == nil { + t.Fatalf("expected ErrAlreadyRevoked, got nil") + } + // Revoking unknown id should return ErrNotFound. + if err := store.Revoke(ctx, 99999); err == nil { + t.Fatalf("expected ErrNotFound, got nil") + } +} + +func TestScopePrecedence(t *testing.T) { + store, _, uid := newTestStore(t) + ctx := context.Background() + + const ( + agentID = int64(42) + taskID = int64(7) + name = "OPENAI_API_KEY" + ) + + if _, err := store.Set(ctx, name, secrets.ScopeUser, uid, uid, "user-val"); err != nil { + t.Fatalf("user Set: %v", err) + } + if _, err := store.Set(ctx, name, secrets.ScopeAgent, agentID, uid, "agent-val"); err != nil { + t.Fatalf("agent Set: %v", err) + } + if _, err := store.Set(ctx, name, secrets.ScopeTask, taskID, uid, "task-val"); err != nil { + t.Fatalf("task Set: %v", err) + } + + env, err := store.BuildEnvMap(ctx, uid, agentID, taskID) + if err != nil { + t.Fatalf("BuildEnvMap: %v", err) + } + if env[name] != "task-val" { + t.Errorf("task wins: got %q want %q", env[name], "task-val") + } + + // Without task: agent wins. + env, err = store.BuildEnvMap(ctx, uid, agentID, 0) + if err != nil { + t.Fatalf("BuildEnvMap: %v", err) + } + if env[name] != "agent-val" { + t.Errorf("agent wins: got %q want %q", env[name], "agent-val") + } + + // Only user. + env, err = store.BuildEnvMap(ctx, uid, 0, 0) + if err != nil { + t.Fatalf("BuildEnvMap: %v", err) + } + if env[name] != "user-val" { + t.Errorf("user wins: got %q want %q", env[name], "user-val") + } + + // last_used_at should be set after BuildEnvMap. + infos, err := store.List(ctx, []secrets.Scope{{Type: secrets.ScopeUser, ID: uid}}) + if err != nil { + t.Fatalf("List: %v", err) + } + var found bool + for _, i := range infos { + if i.Name == name { + found = true + if i.LastUsedAt == nil { + t.Errorf("expected LastUsedAt to be set after BuildEnvMap") + } + } + } + if !found { + t.Errorf("user-scoped secret missing from List") + } +} + +func TestSetReplacesPrevious(t *testing.T) { + store, db, uid := newTestStore(t) + ctx := context.Background() + + if _, err := store.Set(ctx, "ROTATE", secrets.ScopeUser, uid, uid, "v1"); err != nil { + t.Fatalf("Set v1: %v", err) + } + if _, err := store.Set(ctx, "ROTATE", secrets.ScopeUser, uid, uid, "v2"); err != nil { + t.Fatalf("Set v2: %v", err) + } + + got, err := store.Get(ctx, "ROTATE", secrets.ScopeUser, uid) + if err != nil { + t.Fatalf("Get: %v", err) + } + if got != "v2" { + t.Errorf("got %q want %q", got, "v2") + } + + // Two history rows should exist; one revoked, one active. + var total, active int + if err := db.QueryRowContext(ctx, `SELECT COUNT(*) FROM secrets WHERE name='ROTATE'`).Scan(&total); err != nil { + t.Fatalf("count total: %v", err) + } + if err := db.QueryRowContext(ctx, `SELECT COUNT(*) FROM secrets WHERE name='ROTATE' AND revoked_at IS NULL`).Scan(&active); err != nil { + t.Fatalf("count active: %v", err) + } + if total != 2 { + t.Errorf("total rows: got %d want 2", total) + } + if active != 1 { + t.Errorf("active rows: got %d want 1", active) + } +} diff --git a/internal/secrets/types.go b/internal/secrets/types.go new file mode 100644 index 0000000..f8ec88c --- /dev/null +++ b/internal/secrets/types.go @@ -0,0 +1,62 @@ +// Package secrets provides encrypted, scoped secret storage for SynapBus. +// +// Secrets are stored in SQLite, encrypted at rest with NaCl secretbox under a +// local 32-byte master key (kept in /secrets.key with 0600 perms). +// Secrets are scoped to a user, agent, or task and are intended to be injected +// into subprocess environments as sanitized A-Z0-9_ variable names. The MCP +// surface never returns plaintext values — only names and availability. +package secrets + +import ( + "errors" + "time" +) + +// Scope type constants. The same values are used in the secrets.scope_type +// column (CHECK constraint enforced at the SQL level). +const ( + ScopeUser = "user" + ScopeAgent = "agent" + ScopeTask = "task" +) + +// Sentinel errors for the secrets package. +var ( + // ErrNotFound is returned when no active secret matches the lookup. + ErrNotFound = errors.New("secret not found") + // ErrAlreadyRevoked is returned when revoking a secret that is already revoked. + ErrAlreadyRevoked = errors.New("secret already revoked") + // ErrInvalidName is returned when a secret name fails sanitization. + ErrInvalidName = errors.New("invalid secret name: must be non-empty and contain only A-Z, 0-9, _") + // ErrMasterKeyMissing is returned when the master key file cannot be read or generated. + ErrMasterKeyMissing = errors.New("secrets master key missing or unreadable") +) + +// Secret is a stored, encrypted secret row. The plaintext value is never +// included — callers fetch it explicitly via Store.Get. +type Secret struct { + ID int64 + Name string + ScopeType string + ScopeID int64 + CreatedBy int64 + CreatedAt time.Time + RevokedAt *time.Time + LastUsedAt *time.Time +} + +// Info is the public, value-free projection of a Secret used for listings +// exposed via MCP / API. It deliberately has no value field. +type Info struct { + Name string + ScopeType string + ScopeID int64 + Available bool + LastUsedAt *time.Time +} + +// Scope identifies a (type, id) pair used when listing or building env maps. +type Scope struct { + Type string + ID int64 +} diff --git a/internal/storage/schema/021_goals_tasks.sql b/internal/storage/schema/021_goals_tasks.sql new file mode 100644 index 0000000..e2ae77f --- /dev/null +++ b/internal/storage/schema/021_goals_tasks.sql @@ -0,0 +1,73 @@ +-- 021: Goals and tasks — first-class data model for dynamic agent spawning. +-- +-- Goals are human-owned top-level objectives. Each goal has a backing +-- #goal- channel and a pre-built coordinator agent. +-- +-- Tasks are nodes in a goal's work tree. They are single-assignee, +-- atomically claimable (optimistic-lock UPDATE), carry a denormalized +-- goal-ancestry JSON snapshot, and accumulate leaf-only cost counters +-- that roll up via a recursive CTE at read time. + +CREATE TABLE goals ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + slug TEXT NOT NULL UNIQUE, + title TEXT NOT NULL, + description TEXT NOT NULL, + owner_user_id INTEGER NOT NULL REFERENCES users(id) ON DELETE CASCADE, + channel_id INTEGER NOT NULL REFERENCES channels(id) ON DELETE CASCADE, + coordinator_agent_id INTEGER REFERENCES agents(id) ON DELETE SET NULL, + root_task_id INTEGER, + status TEXT NOT NULL DEFAULT 'draft' + CHECK (status IN ('draft','active','paused','completed','cancelled','stuck')), + budget_tokens INTEGER, + budget_dollars_cents INTEGER, + max_spawn_depth INTEGER NOT NULL DEFAULT 3, + alert_80pct_posted INTEGER NOT NULL DEFAULT 0, + created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP, + updated_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP, + completed_at DATETIME +); + +CREATE INDEX idx_goals_owner ON goals(owner_user_id); +CREATE INDEX idx_goals_status ON goals(status); +CREATE INDEX idx_goals_channel ON goals(channel_id); + +CREATE TABLE goal_tasks ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + goal_id INTEGER NOT NULL REFERENCES goals(id) ON DELETE CASCADE, + parent_task_id INTEGER REFERENCES goal_tasks(id) ON DELETE CASCADE, + ancestry_json TEXT NOT NULL DEFAULT '[]', + depth INTEGER NOT NULL DEFAULT 0, + title TEXT NOT NULL, + description TEXT NOT NULL, + acceptance_criteria TEXT NOT NULL DEFAULT '', + created_by_agent_id INTEGER REFERENCES agents(id) ON DELETE SET NULL, + created_by_user_id INTEGER REFERENCES users(id) ON DELETE SET NULL, + assignee_agent_id INTEGER REFERENCES agents(id) ON DELETE SET NULL, + status TEXT NOT NULL DEFAULT 'proposed' + CHECK (status IN ('proposed','approved','claimed','in_progress', + 'awaiting_verification','done','failed','cancelled')), + billing_code TEXT, + budget_tokens INTEGER, + budget_dollars_cents INTEGER, + spent_tokens INTEGER NOT NULL DEFAULT 0, + spent_dollars_cents INTEGER NOT NULL DEFAULT 0, + heartbeat_config_json TEXT, + verifier_config_json TEXT, + origin_message_id INTEGER REFERENCES messages(id) ON DELETE SET NULL, + claim_message_id INTEGER REFERENCES messages(id) ON DELETE SET NULL, + completion_message_id INTEGER REFERENCES messages(id) ON DELETE SET NULL, + failure_reason TEXT, + created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP, + approved_at DATETIME, + claimed_at DATETIME, + started_at DATETIME, + completed_at DATETIME, + CHECK (created_by_agent_id IS NOT NULL OR created_by_user_id IS NOT NULL) +); + +CREATE INDEX idx_goal_tasks_goal ON goal_tasks(goal_id); +CREATE INDEX idx_goal_tasks_parent ON goal_tasks(parent_task_id); +CREATE INDEX idx_goal_tasks_assignee ON goal_tasks(assignee_agent_id); +CREATE INDEX idx_goal_tasks_status ON goal_tasks(status); +CREATE INDEX idx_goal_tasks_billing ON goal_tasks(billing_code); diff --git a/internal/storage/schema/022_agent_proposals.sql b/internal/storage/schema/022_agent_proposals.sql new file mode 100644 index 0000000..7aa673f --- /dev/null +++ b/internal/storage/schema/022_agent_proposals.sql @@ -0,0 +1,54 @@ +-- 022: Agent proposals + resource requests. +-- +-- Agent proposals: a pending request by an existing agent to spawn a new +-- specialist sub-agent. Delegation-cap and spawn-depth checks run at +-- propose time; approval via the reactions workflow on #approvals. +-- +-- Resource requests: an agent asking the human owner for a missing +-- secret (API key, credential) via #requests. + +CREATE TABLE agent_proposals ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + proposer_agent_id INTEGER NOT NULL REFERENCES agents(id) ON DELETE CASCADE, + goal_id INTEGER NOT NULL REFERENCES goals(id) ON DELETE CASCADE, + parent_task_id INTEGER REFERENCES goal_tasks(id) ON DELETE SET NULL, + proposed_name TEXT NOT NULL, + proposed_model TEXT NOT NULL, + proposed_system_prompt TEXT NOT NULL, + proposed_tool_scope_json TEXT NOT NULL DEFAULT '[]', + proposed_skills_json TEXT NOT NULL DEFAULT '[]', + proposed_mcp_servers_json TEXT NOT NULL DEFAULT '[]', + proposed_subagents_json TEXT NOT NULL DEFAULT '[]', + proposed_autonomy_tier TEXT NOT NULL + CHECK (proposed_autonomy_tier IN ('supervised','assisted','autonomous')), + reason TEXT NOT NULL DEFAULT '', + status TEXT NOT NULL DEFAULT 'pending' + CHECK (status IN ('pending','approved','rejected','materialized','cancelled')), + approval_message_id INTEGER REFERENCES messages(id) ON DELETE SET NULL, + materialized_agent_id INTEGER REFERENCES agents(id) ON DELETE SET NULL, + rejection_reason TEXT, + created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP, + decided_at DATETIME +); + +CREATE INDEX idx_proposals_goal ON agent_proposals(goal_id); +CREATE INDEX idx_proposals_parent ON agent_proposals(parent_task_id); +CREATE INDEX idx_proposals_status ON agent_proposals(status); + +CREATE TABLE resource_requests ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + requester_agent_id INTEGER NOT NULL REFERENCES agents(id) ON DELETE CASCADE, + task_id INTEGER NOT NULL REFERENCES goal_tasks(id) ON DELETE CASCADE, + resource_name TEXT NOT NULL, + resource_type TEXT NOT NULL, + reason TEXT NOT NULL, + status TEXT NOT NULL DEFAULT 'pending' + CHECK (status IN ('pending','fulfilled','rejected','revoked')), + request_message_id INTEGER REFERENCES messages(id) ON DELETE SET NULL, + created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP, + fulfilled_at DATETIME +); + +CREATE INDEX idx_requests_task ON resource_requests(task_id); +CREATE INDEX idx_requests_requester ON resource_requests(requester_agent_id); +CREATE INDEX idx_requests_status ON resource_requests(status); diff --git a/internal/storage/schema/023_agent_trust_model.sql b/internal/storage/schema/023_agent_trust_model.sql new file mode 100644 index 0000000..f507972 --- /dev/null +++ b/internal/storage/schema/023_agent_trust_model.sql @@ -0,0 +1,37 @@ +-- 023: Dynamic-agent trust model columns + reputation ledger. +-- +-- Coexists with the existing `agent_trust` table from migration 014 +-- (which remains keyed by agent_name + action_type and is still used +-- by the reactions + hybrid MCP paths). The new ledger below is a +-- parallel append-only evidence log keyed by (config_hash, task_domain) +-- for the dynamic-spawning trust model. + +ALTER TABLE agents ADD COLUMN config_hash TEXT NOT NULL DEFAULT ''; +ALTER TABLE agents ADD COLUMN parent_agent_id INTEGER REFERENCES agents(id) ON DELETE SET NULL; +ALTER TABLE agents ADD COLUMN spawn_depth INTEGER NOT NULL DEFAULT 0; +ALTER TABLE agents ADD COLUMN system_prompt TEXT NOT NULL DEFAULT ''; +ALTER TABLE agents ADD COLUMN autonomy_tier TEXT NOT NULL DEFAULT 'supervised'; +ALTER TABLE agents ADD COLUMN tool_scope_json TEXT NOT NULL DEFAULT '[]'; +ALTER TABLE agents ADD COLUMN quarantined_at DATETIME; +ALTER TABLE agents ADD COLUMN quarantine_reason TEXT; + +CREATE INDEX idx_agents_config_hash ON agents(config_hash); +CREATE INDEX idx_agents_parent ON agents(parent_agent_id); + +-- Append-only reputation ledger for the dynamic-spawning trust model. +-- The current rolling score is derived at read time via exponential +-- time decay; this table is the source of truth. +CREATE TABLE reputation_evidence ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + config_hash TEXT NOT NULL, + owner_user_id INTEGER NOT NULL REFERENCES users(id) ON DELETE CASCADE, + task_domain TEXT NOT NULL DEFAULT 'default', + score_delta REAL NOT NULL, + evidence_ref TEXT NOT NULL, + weight REAL NOT NULL DEFAULT 1.0, + created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP +); + +CREATE INDEX idx_rep_hash_domain ON reputation_evidence(config_hash, task_domain); +CREATE INDEX idx_rep_owner ON reputation_evidence(owner_user_id); +CREATE INDEX idx_rep_created ON reputation_evidence(created_at); diff --git a/internal/storage/schema/024_secrets.sql b/internal/storage/schema/024_secrets.sql new file mode 100644 index 0000000..eeb4889 --- /dev/null +++ b/internal/storage/schema/024_secrets.sql @@ -0,0 +1,20 @@ +-- 024: Secrets — encrypted per-scope values, injected into subprocess env +-- as sanitized A-Z0-9_ variable names. Values are NaCl-secretbox +-- encrypted under a local master key file (/secrets.key). +-- MCP tools never return the plaintext value — only names and +-- availability. + +CREATE TABLE secrets ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + name TEXT NOT NULL, + scope_type TEXT NOT NULL CHECK (scope_type IN ('user','agent','task')), + scope_id INTEGER NOT NULL, + value_blob BLOB NOT NULL, -- nonce(24) || ciphertext + created_by INTEGER NOT NULL REFERENCES users(id) ON DELETE CASCADE, + created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP, + revoked_at DATETIME, + last_used_at DATETIME +); + +CREATE UNIQUE INDEX idx_secrets_scope_name ON secrets(scope_type, scope_id, name) WHERE revoked_at IS NULL; +CREATE INDEX idx_secrets_scope ON secrets(scope_type, scope_id); diff --git a/internal/storage/schema/025_harness_runs_task_id.sql b/internal/storage/schema/025_harness_runs_task_id.sql new file mode 100644 index 0000000..0e8b39d --- /dev/null +++ b/internal/storage/schema/025_harness_runs_task_id.sql @@ -0,0 +1,10 @@ +-- 025: Link harness runs to tasks. +-- +-- When a harness run fires in the context of a task (i.e. an agent +-- working on a claimed task), the reactor sets ExecRequest.TaskID and +-- the harness Observer writes it here so the Web UI and the HTML +-- report can correlate runs → tasks. + +ALTER TABLE harness_runs ADD COLUMN task_id INTEGER; + +CREATE INDEX idx_harness_runs_task ON harness_runs(task_id); diff --git a/internal/trust/config_hash.go b/internal/trust/config_hash.go new file mode 100644 index 0000000..d9b26e8 --- /dev/null +++ b/internal/trust/config_hash.go @@ -0,0 +1,75 @@ +package trust + +import ( + "crypto/sha256" + "encoding/hex" + "encoding/json" + "sort" +) + +// AgentConfig captures the immutable inputs that define an agent's identity +// for the dynamic-spawning trust model. Two agents with the same canonical +// AgentConfig share a config_hash and therefore share reputation. +type AgentConfig struct { + Model string `json:"model"` + SystemPrompt string `json:"system_prompt"` + ToolScope []string `json:"tool_scope"` + Skills []string `json:"skills"` + Subagents []string `json:"subagents"` + MCPServers []MCPServerRef `json:"mcp_servers"` +} + +// MCPServerRef describes a single MCP server attached to an agent. +type MCPServerRef struct { + Name string `json:"name"` + URL string `json:"url"` + Transport string `json:"transport"` +} + +// ConfigHash returns the hex-encoded SHA-256 of the canonical-JSON +// representation of cfg. +// +// The function is deterministic: the input slices may be in any order, +// the result depends only on the multiset of values. Object keys are sorted +// (encoding/json already does this) and slice contents are sorted alphabetically +// before hashing so that two equivalent configs always produce the same hash. +func ConfigHash(cfg AgentConfig) string { + tools := append([]string(nil), cfg.ToolScope...) + sort.Strings(tools) + skills := append([]string(nil), cfg.Skills...) + sort.Strings(skills) + subagents := append([]string(nil), cfg.Subagents...) + sort.Strings(subagents) + + servers := make([]map[string]string, 0, len(cfg.MCPServers)) + for _, s := range cfg.MCPServers { + servers = append(servers, map[string]string{ + "name": s.Name, + "url": s.URL, + "transport": s.Transport, + }) + } + sort.Slice(servers, func(i, j int) bool { + return servers[i]["name"] < servers[j]["name"] + }) + + canonical := map[string]any{ + "model": cfg.Model, + "system_prompt": cfg.SystemPrompt, + "tool_scope": tools, + "skills": skills, + "subagents": subagents, + "mcp_servers": servers, + } + + // encoding/json sorts map keys alphabetically, giving canonical output. + data, err := json.Marshal(canonical) + if err != nil { + // Marshalling a map of strings/slices cannot fail in practice; + // fall back to a stable sentinel hash so callers never see a panic. + sum := sha256.Sum256([]byte("synapbus:trust:config_hash:marshal_error")) + return hex.EncodeToString(sum[:]) + } + sum := sha256.Sum256(data) + return hex.EncodeToString(sum[:]) +} diff --git a/internal/trust/config_hash_test.go b/internal/trust/config_hash_test.go new file mode 100644 index 0000000..c689ae1 --- /dev/null +++ b/internal/trust/config_hash_test.go @@ -0,0 +1,145 @@ +package trust + +import ( + "testing" +) + +func baseConfig() AgentConfig { + return AgentConfig{ + Model: "claude-opus-4", + SystemPrompt: "You are a careful research assistant.", + ToolScope: []string{"search", "fetch", "summarize"}, + Skills: []string{"writing", "research"}, + Subagents: []string{"reviewer", "critic"}, + MCPServers: []MCPServerRef{ + {Name: "synapbus", URL: "http://localhost:8080/mcp", Transport: "http"}, + {Name: "fs", URL: "stdio://fs", Transport: "stdio"}, + }, + } +} + +func TestConfigHash_Deterministic(t *testing.T) { + canonical := baseConfig() + want := ConfigHash(canonical) + + tests := []struct { + name string + mutate func(*AgentConfig) + }{ + { + name: "shuffled tool_scope", + mutate: func(c *AgentConfig) { + c.ToolScope = []string{"summarize", "fetch", "search"} + }, + }, + { + name: "shuffled skills", + mutate: func(c *AgentConfig) { + c.Skills = []string{"research", "writing"} + }, + }, + { + name: "shuffled subagents", + mutate: func(c *AgentConfig) { + c.Subagents = []string{"critic", "reviewer"} + }, + }, + { + name: "shuffled mcp_servers", + mutate: func(c *AgentConfig) { + c.MCPServers = []MCPServerRef{ + {Name: "fs", URL: "stdio://fs", Transport: "stdio"}, + {Name: "synapbus", URL: "http://localhost:8080/mcp", Transport: "http"}, + } + }, + }, + { + name: "shuffled all collections", + mutate: func(c *AgentConfig) { + c.ToolScope = []string{"fetch", "summarize", "search"} + c.Skills = []string{"research", "writing"} + c.Subagents = []string{"critic", "reviewer"} + c.MCPServers = []MCPServerRef{ + {Name: "fs", URL: "stdio://fs", Transport: "stdio"}, + {Name: "synapbus", URL: "http://localhost:8080/mcp", Transport: "http"}, + } + }, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + cfg := baseConfig() + tt.mutate(&cfg) + got := ConfigHash(cfg) + if got != want { + t.Errorf("ConfigHash mismatch\n got: %s\nwant: %s", got, want) + } + }) + } +} + +func TestConfigHash_SensitiveToChanges(t *testing.T) { + base := baseConfig() + baseHash := ConfigHash(base) + + tests := []struct { + name string + mutate func(*AgentConfig) + }{ + { + name: "model changed", + mutate: func(c *AgentConfig) { c.Model = "claude-sonnet-4" }, + }, + { + name: "system prompt changed", + mutate: func(c *AgentConfig) { c.SystemPrompt = "You are a sloppy assistant." }, + }, + { + name: "tool added", + mutate: func(c *AgentConfig) { c.ToolScope = append(c.ToolScope, "execute") }, + }, + { + name: "tool removed", + mutate: func(c *AgentConfig) { c.ToolScope = []string{"search", "fetch"} }, + }, + { + name: "tool renamed", + mutate: func(c *AgentConfig) { c.ToolScope = []string{"search", "fetch", "summarise"} }, + }, + { + name: "mcp server url changed", + mutate: func(c *AgentConfig) { + c.MCPServers[0].URL = "http://kubic.home.arpa:30088/mcp" + }, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + cfg := baseConfig() + tt.mutate(&cfg) + got := ConfigHash(cfg) + if got == baseHash { + t.Errorf("expected different hash, both = %s", got) + } + }) + } +} + +func TestConfigHash_EmptyFieldsStable(t *testing.T) { + empty := AgentConfig{} + h1 := ConfigHash(empty) + h2 := ConfigHash(AgentConfig{ + ToolScope: []string{}, + Skills: []string{}, + Subagents: []string{}, + MCPServers: []MCPServerRef{}, + }) + if h1 != h2 { + t.Errorf("nil and empty slice configs should hash equal\n h1: %s\n h2: %s", h1, h2) + } + if len(h1) != 64 { + t.Errorf("expected 64-char hex sha256, got len=%d", len(h1)) + } +} diff --git a/internal/trust/delegation.go b/internal/trust/delegation.go new file mode 100644 index 0000000..d76bcdd --- /dev/null +++ b/internal/trust/delegation.go @@ -0,0 +1,141 @@ +package trust + +import ( + "fmt" + "sort" +) + +// Autonomy tier constants, ordered low → high. +// +// Higher tiers grant the agent more freedom: supervised requires human +// approval for outbound actions, assisted may act with logging, autonomous +// may act without per-action review (but always within tool/budget caps). +const ( + TierSupervised = "supervised" + TierAssisted = "assisted" + TierAutonomous = "autonomous" +) + +var tierRank = map[string]int{ + TierSupervised: 1, + TierAssisted: 2, + TierAutonomous: 3, +} + +// Grant is a single delegation envelope: the maximum authority a parent +// confers to a child agent. The dynamic-spawning trust model enforces the +// "child ≤ parent" rule across every dimension of a Grant. +type Grant struct { + AutonomyTier string + ToolScope []string + BudgetTokens int64 + BudgetDollarsCents int64 + SpawnDepth int +} + +// DelegationCap computes the effective grant a child should receive given +// the parent's grant and the child's proposal. +// +// Rules: +// - Autonomy tier: effective = min(parent, proposed). A proposed tier above +// the parent's tier is a violation. +// - Tool scope: effective = intersection of parent and proposed. Any tool in +// proposed that is not in parent is a violation. Order does not matter. +// - Budget tokens / dollars: effective = min, with 0 on parent meaning +// unlimited and 0 on child meaning "any up to parent". Proposed > parent +// (when parent is non-zero) is a violation. +// - Spawn depth: child must have parent.SpawnDepth + 1 <= maxDepth, otherwise +// a violation is emitted; effective.SpawnDepth is set to parent + 1 anyway. +// +// violations is a slice of human-readable strings, one per violated rule, +// in stable order. effective is always populated (best-effort cap). +func DelegationCap(parent, proposed Grant, maxDepth int) (effective Grant, violations []string) { + // --- Autonomy tier -------------------------------------------------- + parentRank, parentOK := tierRank[parent.AutonomyTier] + proposedRank, proposedOK := tierRank[proposed.AutonomyTier] + if !parentOK { + parentRank = tierRank[TierSupervised] + } + if !proposedOK { + proposedRank = tierRank[TierSupervised] + violations = append(violations, + fmt.Sprintf("unknown autonomy tier %q (treating as supervised)", proposed.AutonomyTier)) + } + if proposedRank > parentRank { + violations = append(violations, + fmt.Sprintf("autonomy tier %q exceeds parent tier %q", + proposed.AutonomyTier, parent.AutonomyTier)) + effective.AutonomyTier = parent.AutonomyTier + } else { + effective.AutonomyTier = proposed.AutonomyTier + } + + // --- Tool scope ----------------------------------------------------- + parentTools := make(map[string]struct{}, len(parent.ToolScope)) + for _, t := range parent.ToolScope { + parentTools[t] = struct{}{} + } + + var allowed []string + var disallowed []string + for _, t := range proposed.ToolScope { + if _, ok := parentTools[t]; ok { + allowed = append(allowed, t) + } else { + disallowed = append(disallowed, t) + } + } + sort.Strings(disallowed) + for _, t := range disallowed { + violations = append(violations, + fmt.Sprintf("tool %q not in parent scope", t)) + } + sort.Strings(allowed) + effective.ToolScope = allowed + + // --- Budget tokens -------------------------------------------------- + effective.BudgetTokens = capBudget("budget_tokens", + parent.BudgetTokens, proposed.BudgetTokens, &violations) + + // --- Budget dollars (cents) ----------------------------------------- + effective.BudgetDollarsCents = capBudget("budget_dollars_cents", + parent.BudgetDollarsCents, proposed.BudgetDollarsCents, &violations) + + // --- Spawn depth ---------------------------------------------------- + effective.SpawnDepth = parent.SpawnDepth + 1 + if effective.SpawnDepth > maxDepth { + violations = append(violations, + fmt.Sprintf("spawn depth %d exceeds max %d", effective.SpawnDepth, maxDepth)) + } + + return effective, violations +} + +// capBudget enforces the parent ≥ child rule for a budget dimension, where +// 0 on the parent means "unlimited" and 0 on the child means "inherit any +// value up to the parent's cap". +func capBudget(label string, parent, proposed int64, violations *[]string) int64 { + switch { + case parent == 0: + // Parent unlimited: any non-negative proposal is fine. + if proposed < 0 { + *violations = append(*violations, + fmt.Sprintf("%s %d is negative", label, proposed)) + return 0 + } + return proposed + case proposed == 0: + // Child wants "as much as parent allows". + return parent + case proposed > parent: + *violations = append(*violations, + fmt.Sprintf("%s %d exceeds parent cap %d", label, proposed, parent)) + return parent + case proposed < 0: + *violations = append(*violations, + fmt.Sprintf("%s %d is negative", label, proposed)) + return 0 + default: + return proposed + } +} diff --git a/internal/trust/delegation_test.go b/internal/trust/delegation_test.go new file mode 100644 index 0000000..8d77094 --- /dev/null +++ b/internal/trust/delegation_test.go @@ -0,0 +1,256 @@ +package trust + +import ( + "strings" + "testing" +) + +func TestDelegationCap_TierMatrix(t *testing.T) { + tiers := []string{TierSupervised, TierAssisted, TierAutonomous} + + tests := []struct { + name string + parent string + proposed string + wantViolation bool + wantEffective string + }{ + {"sup→sup ok", TierSupervised, TierSupervised, false, TierSupervised}, + {"sup→assisted violation", TierSupervised, TierAssisted, true, TierSupervised}, + {"sup→autonomous violation", TierSupervised, TierAutonomous, true, TierSupervised}, + {"assisted→sup ok", TierAssisted, TierSupervised, false, TierSupervised}, + {"assisted→assisted ok", TierAssisted, TierAssisted, false, TierAssisted}, + {"assisted→autonomous violation", TierAssisted, TierAutonomous, true, TierAssisted}, + {"autonomous→sup ok", TierAutonomous, TierSupervised, false, TierSupervised}, + {"autonomous→assisted ok", TierAutonomous, TierAssisted, false, TierAssisted}, + {"autonomous→autonomous ok", TierAutonomous, TierAutonomous, false, TierAutonomous}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + parent := Grant{AutonomyTier: tt.parent, ToolScope: []string{"search"}} + proposed := Grant{AutonomyTier: tt.proposed, ToolScope: []string{"search"}} + eff, viols := DelegationCap(parent, proposed, 5) + + gotViolation := false + for _, v := range viols { + if strings.Contains(v, "autonomy tier") { + gotViolation = true + break + } + } + if gotViolation != tt.wantViolation { + t.Errorf("violation = %v, want %v (viols=%v)", gotViolation, tt.wantViolation, viols) + } + if eff.AutonomyTier != tt.wantEffective { + t.Errorf("effective tier = %q, want %q", eff.AutonomyTier, tt.wantEffective) + } + }) + } + + // sanity: ensure all known tiers are mapped + for _, ti := range tiers { + if _, ok := tierRank[ti]; !ok { + t.Errorf("tier %q missing from tierRank", ti) + } + } +} + +func TestDelegationCap_ToolScope(t *testing.T) { + tests := []struct { + name string + parentTools []string + proposedTools []string + wantEffective []string + wantDisallowed []string + }{ + { + name: "exact subset", + parentTools: []string{"search", "fetch", "summarize"}, + proposedTools: []string{"search", "fetch"}, + wantEffective: []string{"fetch", "search"}, + wantDisallowed: nil, + }, + { + name: "full overlap", + parentTools: []string{"search", "fetch"}, + proposedTools: []string{"search", "fetch"}, + wantEffective: []string{"fetch", "search"}, + wantDisallowed: nil, + }, + { + name: "single forbidden tool", + parentTools: []string{"search"}, + proposedTools: []string{"search", "execute"}, + wantEffective: []string{"search"}, + wantDisallowed: []string{"execute"}, + }, + { + name: "all forbidden", + parentTools: []string{"search"}, + proposedTools: []string{"execute", "delete"}, + wantEffective: nil, + wantDisallowed: []string{"delete", "execute"}, + }, + { + name: "empty proposed", + parentTools: []string{"search"}, + proposedTools: nil, + wantEffective: nil, + wantDisallowed: nil, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + parent := Grant{AutonomyTier: TierAssisted, ToolScope: tt.parentTools} + proposed := Grant{AutonomyTier: TierAssisted, ToolScope: tt.proposedTools} + eff, viols := DelegationCap(parent, proposed, 10) + + if !equalSlices(eff.ToolScope, tt.wantEffective) { + t.Errorf("effective tools = %v, want %v", eff.ToolScope, tt.wantEffective) + } + + for _, want := range tt.wantDisallowed { + found := false + for _, v := range viols { + if strings.Contains(v, want) && strings.Contains(v, "not in parent scope") { + found = true + break + } + } + if !found { + t.Errorf("missing violation for tool %q in %v", want, viols) + } + } + }) + } +} + +func TestDelegationCap_Budgets(t *testing.T) { + tests := []struct { + name string + parentTokens int64 + proposedTokens int64 + wantTokens int64 + wantViolation bool + }{ + {"parent unlimited, child 100", 0, 100, 100, false}, + {"parent 100, child 50", 100, 50, 50, false}, + {"parent 100, child 100", 100, 100, 100, false}, + {"parent 100, child 200 (violation)", 100, 200, 100, true}, + {"parent 100, child 0 (inherit)", 100, 0, 100, false}, + {"parent 0, child 0 (both unlimited)", 0, 0, 0, false}, + {"parent 100, child negative", 100, -5, 0, true}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + parent := Grant{ + AutonomyTier: TierAssisted, + ToolScope: []string{"x"}, + BudgetTokens: tt.parentTokens, + } + proposed := Grant{ + AutonomyTier: TierAssisted, + ToolScope: []string{"x"}, + BudgetTokens: tt.proposedTokens, + } + eff, viols := DelegationCap(parent, proposed, 5) + + if eff.BudgetTokens != tt.wantTokens { + t.Errorf("budget tokens = %d, want %d", eff.BudgetTokens, tt.wantTokens) + } + + gotViolation := false + for _, v := range viols { + if strings.Contains(v, "budget_tokens") { + gotViolation = true + break + } + } + if gotViolation != tt.wantViolation { + t.Errorf("violation = %v, want %v (viols=%v)", gotViolation, tt.wantViolation, viols) + } + }) + } +} + +func TestDelegationCap_SpawnDepth(t *testing.T) { + tests := []struct { + name string + parentDepth int + maxDepth int + wantEffective int + wantViolation bool + }{ + {"depth 0 → 1, max 5", 0, 5, 1, false}, + {"depth 4 → 5, max 5", 4, 5, 5, false}, + {"depth 5 → 6, max 5 (violation)", 5, 5, 6, true}, + {"depth 0 → 1, max 0 (violation)", 0, 0, 1, true}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + parent := Grant{ + AutonomyTier: TierAssisted, + ToolScope: []string{"x"}, + SpawnDepth: tt.parentDepth, + } + proposed := Grant{ + AutonomyTier: TierAssisted, + ToolScope: []string{"x"}, + } + eff, viols := DelegationCap(parent, proposed, tt.maxDepth) + + if eff.SpawnDepth != tt.wantEffective { + t.Errorf("spawn depth = %d, want %d", eff.SpawnDepth, tt.wantEffective) + } + + gotViolation := false + for _, v := range viols { + if strings.Contains(v, "spawn depth") { + gotViolation = true + break + } + } + if gotViolation != tt.wantViolation { + t.Errorf("violation = %v, want %v (viols=%v)", gotViolation, tt.wantViolation, viols) + } + }) + } +} + +func TestDelegationCap_MultipleViolations(t *testing.T) { + parent := Grant{ + AutonomyTier: TierSupervised, + ToolScope: []string{"search"}, + BudgetTokens: 100, + SpawnDepth: 5, + } + proposed := Grant{ + AutonomyTier: TierAutonomous, // violation: tier + ToolScope: []string{"execute"}, // violation: tool + BudgetTokens: 1000, // violation: budget + } + _, viols := DelegationCap(parent, proposed, 5) // violation: depth (5+1>5) + + if len(viols) < 4 { + t.Errorf("expected at least 4 violations, got %d: %v", len(viols), viols) + } +} + +func equalSlices(a, b []string) bool { + if len(a) == 0 && len(b) == 0 { + return true + } + if len(a) != len(b) { + return false + } + for i := range a { + if a[i] != b[i] { + return false + } + } + return true +} diff --git a/internal/trust/ledger.go b/internal/trust/ledger.go new file mode 100644 index 0000000..fab92c8 --- /dev/null +++ b/internal/trust/ledger.go @@ -0,0 +1,162 @@ +package trust + +import ( + "context" + "database/sql" + "fmt" + "time" +) + +// defaultHalfLifeDays is used when a caller passes 0 to RollingScore +// or SeedFromParent. +const defaultHalfLifeDays = 30.0 + +// neutralScore is the rolling score returned when no evidence exists for a +// (config_hash, task_domain) pair. 0.5 signals "no information yet". +const neutralScore = 0.5 + +// seedFromParentFraction is the fraction of the parent's rolling score that +// gets seeded onto a fresh child config_hash when SeedFromParent is called. +const seedFromParentFraction = 0.7 + +// Evidence is a single append-only entry in the reputation ledger. +// +// score_delta is unbounded (positive or negative); the rolling score is +// computed at read time by summing decayed deltas and clamping to [0, 1]. +type Evidence struct { + ID int64 + ConfigHash string + OwnerUserID int64 + TaskDomain string + ScoreDelta float64 + Weight float64 + EvidenceRef string + CreatedAt time.Time +} + +// Ledger is the new dynamic-spawning reputation store. It is fully separate +// from the legacy *Service / Store types: this one is keyed by config_hash +// and is append-only, while the legacy store is keyed by agent name and +// performs in-place upserts. +type Ledger struct { + db *sql.DB +} + +// NewLedger constructs a Ledger backed by the given *sql.DB. The caller is +// responsible for migration ordering — migration 023 must already be applied. +func NewLedger(db *sql.DB) *Ledger { + return &Ledger{db: db} +} + +// Append writes one evidence row and returns its row id. +// +// The CreatedAt field, if zero, defaults to the database's CURRENT_TIMESTAMP. +// TaskDomain defaults to "default" and Weight defaults to 1.0. +func (l *Ledger) Append(ctx context.Context, ev Evidence) (int64, error) { + if ev.TaskDomain == "" { + ev.TaskDomain = "default" + } + if ev.Weight == 0 { + ev.Weight = 1.0 + } + + var ( + res sql.Result + err error + ) + if ev.CreatedAt.IsZero() { + res, err = l.db.ExecContext(ctx, + `INSERT INTO reputation_evidence + (config_hash, owner_user_id, task_domain, score_delta, evidence_ref, weight) + VALUES (?, ?, ?, ?, ?, ?)`, + ev.ConfigHash, ev.OwnerUserID, ev.TaskDomain, ev.ScoreDelta, ev.EvidenceRef, ev.Weight, + ) + } else { + res, err = l.db.ExecContext(ctx, + `INSERT INTO reputation_evidence + (config_hash, owner_user_id, task_domain, score_delta, evidence_ref, weight, created_at) + VALUES (?, ?, ?, ?, ?, ?, ?)`, + ev.ConfigHash, ev.OwnerUserID, ev.TaskDomain, ev.ScoreDelta, ev.EvidenceRef, ev.Weight, + ev.CreatedAt.UTC().Format("2006-01-02 15:04:05"), + ) + } + if err != nil { + return 0, fmt.Errorf("append evidence: %w", err) + } + id, err := res.LastInsertId() + if err != nil { + return 0, fmt.Errorf("get last insert id: %w", err) + } + return id, nil +} + +// RollingScore returns the time-decayed rolling reputation score for a +// (config_hash, task_domain) pair, clamped to [0.0, 1.0], plus the count of +// evidence rows considered. +// +// halfLifeDays controls how quickly old evidence loses weight. Pass 0 for the +// 30-day default. +// +// When no evidence exists, the result is (neutralScore, 0, nil). +func (l *Ledger) RollingScore(ctx context.Context, configHash, taskDomain string, halfLifeDays float64) (float64, int, error) { + if halfLifeDays <= 0 { + halfLifeDays = defaultHalfLifeDays + } + if taskDomain == "" { + taskDomain = "default" + } + + const q = ` + SELECT COALESCE(SUM(score_delta * weight * + exp(-0.6931471805599453 * (julianday('now') - julianday(created_at)) / ?)), 0.5) AS score, + COUNT(*) AS cnt + FROM reputation_evidence + WHERE config_hash = ? AND task_domain = ?` + + var ( + score float64 + cnt int + ) + if err := l.db.QueryRowContext(ctx, q, halfLifeDays, configHash, taskDomain).Scan(&score, &cnt); err != nil { + return 0, 0, fmt.Errorf("rolling score query: %w", err) + } + + if cnt == 0 { + return neutralScore, 0, nil + } + + if score < 0.0 { + score = 0.0 + } + if score > 1.0 { + score = 1.0 + } + return score, cnt, nil +} + +// SeedFromParent reads the parent config's current rolling score and writes +// a single seed evidence row for the child config at +// seedFromParentFraction × parent_score. +// +// Used when a new agent config is spawned: rather than starting at the neutral +// 0.5, the child inherits 70% of the parent's reputation as a starting prior. +func (l *Ledger) SeedFromParent(ctx context.Context, parentHash, childHash string, ownerID int64, taskDomain string, halfLifeDays float64) error { + parentScore, _, err := l.RollingScore(ctx, parentHash, taskDomain, halfLifeDays) + if err != nil { + return fmt.Errorf("read parent score: %w", err) + } + + seedDelta := seedFromParentFraction * parentScore + _, err = l.Append(ctx, Evidence{ + ConfigHash: childHash, + OwnerUserID: ownerID, + TaskDomain: taskDomain, + ScoreDelta: seedDelta, + Weight: 1.0, + EvidenceRef: fmt.Sprintf("seed_from_parent:%s", parentHash), + }) + if err != nil { + return fmt.Errorf("write seed evidence: %w", err) + } + return nil +} diff --git a/internal/trust/ledger_test.go b/internal/trust/ledger_test.go new file mode 100644 index 0000000..bfa903b --- /dev/null +++ b/internal/trust/ledger_test.go @@ -0,0 +1,266 @@ +package trust + +import ( + "context" + "database/sql" + "fmt" + "math" + "testing" + "time" + + _ "modernc.org/sqlite" + + "github.com/synapbus/synapbus/internal/storage" +) + +func newLedgerTestDB(t *testing.T) (*sql.DB, int64) { + t.Helper() + dsn := fmt.Sprintf("file:ledger_%s?mode=memory&cache=shared", t.Name()) + db, err := sql.Open("sqlite", dsn) + if err != nil { + t.Fatalf("open database: %v", err) + } + t.Cleanup(func() { db.Close() }) + + if _, err := db.Exec("PRAGMA foreign_keys=ON"); err != nil { + t.Fatalf("enable foreign keys: %v", err) + } + + ctx := context.Background() + if err := storage.RunMigrations(ctx, db); err != nil { + t.Fatalf("run migrations: %v", err) + } + + // Insert a test owner user. + res, err := db.ExecContext(ctx, + `INSERT INTO users (username, password_hash, display_name) VALUES (?, ?, ?)`, + "trust_test_user_"+t.Name(), "x", "Test", + ) + if err != nil { + t.Fatalf("insert user: %v", err) + } + uid, err := res.LastInsertId() + if err != nil { + t.Fatalf("last insert id: %v", err) + } + return db, uid +} + +func TestLedger_AppendAndRollingScore_EmptyIsNeutral(t *testing.T) { + db, _ := newLedgerTestDB(t) + ledger := NewLedger(db) + ctx := context.Background() + + score, cnt, err := ledger.RollingScore(ctx, "no-such-hash", "default", 0) + if err != nil { + t.Fatalf("RollingScore: %v", err) + } + if cnt != 0 { + t.Errorf("count = %d, want 0", cnt) + } + if score != neutralScore { + t.Errorf("score = %f, want %f (neutral)", score, neutralScore) + } +} + +func TestLedger_AppendAndRollingScore_Decays(t *testing.T) { + db, owner := newLedgerTestDB(t) + ledger := NewLedger(db) + ctx := context.Background() + + hash := "decay-hash" + + // Old row (60 days ago) with large positive delta. + if _, err := ledger.Append(ctx, Evidence{ + ConfigHash: hash, + OwnerUserID: owner, + TaskDomain: "default", + ScoreDelta: 0.9, + Weight: 1.0, + EvidenceRef: "old", + CreatedAt: time.Now().UTC().Add(-60 * 24 * time.Hour), + }); err != nil { + t.Fatalf("append old: %v", err) + } + + // Fresh row with smaller delta. + if _, err := ledger.Append(ctx, Evidence{ + ConfigHash: hash, + OwnerUserID: owner, + TaskDomain: "default", + ScoreDelta: 0.4, + Weight: 1.0, + EvidenceRef: "fresh", + }); err != nil { + t.Fatalf("append fresh: %v", err) + } + + score, cnt, err := ledger.RollingScore(ctx, hash, "default", 30) + if err != nil { + t.Fatalf("RollingScore: %v", err) + } + if cnt != 2 { + t.Errorf("count = %d, want 2", cnt) + } + + // After 60d with 30d half-life, the old row's contribution is 0.9 * 0.25 = 0.225. + // The fresh row contributes ~0.4. Total ~0.625, well below the old delta but + // dominated by the fresh contribution. + if score < 0.5 || score > 0.75 { + t.Errorf("score = %f, expected ~0.625", score) + } + + // Fresh-only score should be greater than score with the old row dragging it down? Actually old row is positive too here, so let's instead verify that fresh delta dominates: removing fresh row should drop the score sharply. + score2, _, err := ledger.RollingScore(ctx, hash, "missing-domain", 30) + if err != nil { + t.Fatalf("RollingScore missing: %v", err) + } + if score2 != neutralScore { + t.Errorf("missing-domain score = %f, want neutral", score2) + } +} + +func TestLedger_SeedFromParent_70Percent(t *testing.T) { + db, owner := newLedgerTestDB(t) + ledger := NewLedger(db) + ctx := context.Background() + + parent := "parent-hash" + child := "child-hash" + + // Seed the parent with evidence summing to ~0.8 (recent, full weight). + if _, err := ledger.Append(ctx, Evidence{ + ConfigHash: parent, + OwnerUserID: owner, + ScoreDelta: 0.8, + Weight: 1.0, + EvidenceRef: "parent_seed", + }); err != nil { + t.Fatalf("append parent: %v", err) + } + + parentScore, _, err := ledger.RollingScore(ctx, parent, "default", 30) + if err != nil { + t.Fatalf("parent score: %v", err) + } + // Parent score should be ~0.8 (just inserted, decay ≈ 1.0). + if math.Abs(parentScore-0.8) > 0.01 { + t.Fatalf("parent score = %f, want ~0.8", parentScore) + } + + if err := ledger.SeedFromParent(ctx, parent, child, owner, "default", 30); err != nil { + t.Fatalf("SeedFromParent: %v", err) + } + + childScore, cnt, err := ledger.RollingScore(ctx, child, "default", 30) + if err != nil { + t.Fatalf("child score: %v", err) + } + if cnt != 1 { + t.Errorf("child evidence count = %d, want 1", cnt) + } + want := 0.7 * 0.8 // 0.56 + if math.Abs(childScore-want) > 0.01 { + t.Errorf("child score = %f, want ~%f", childScore, want) + } + + // Verify the seed row references the parent. + var ref string + if err := db.QueryRowContext(ctx, + `SELECT evidence_ref FROM reputation_evidence WHERE config_hash = ?`, child, + ).Scan(&ref); err != nil { + t.Fatalf("select ref: %v", err) + } + if ref != "seed_from_parent:"+parent { + t.Errorf("evidence_ref = %q, want %q", ref, "seed_from_parent:"+parent) + } +} + +func TestLedger_Clamping(t *testing.T) { + db, owner := newLedgerTestDB(t) + ledger := NewLedger(db) + ctx := context.Background() + + hash := "clamp-hash" + + // Many large negative deltas. + for i := 0; i < 5; i++ { + if _, err := ledger.Append(ctx, Evidence{ + ConfigHash: hash, + OwnerUserID: owner, + TaskDomain: "default", + ScoreDelta: -2.0, + Weight: 1.0, + EvidenceRef: fmt.Sprintf("neg_%d", i), + }); err != nil { + t.Fatalf("append neg: %v", err) + } + } + + score, cnt, err := ledger.RollingScore(ctx, hash, "default", 30) + if err != nil { + t.Fatalf("RollingScore: %v", err) + } + if cnt != 5 { + t.Errorf("count = %d, want 5", cnt) + } + if score != 0.0 { + t.Errorf("score = %f, want 0.0 (clamped)", score) + } + + // And the inverse: many large positive deltas should clamp to 1.0. + hash2 := "clamp-hash-pos" + for i := 0; i < 5; i++ { + if _, err := ledger.Append(ctx, Evidence{ + ConfigHash: hash2, + OwnerUserID: owner, + TaskDomain: "default", + ScoreDelta: 2.0, + Weight: 1.0, + EvidenceRef: fmt.Sprintf("pos_%d", i), + }); err != nil { + t.Fatalf("append pos: %v", err) + } + } + score2, _, err := ledger.RollingScore(ctx, hash2, "default", 30) + if err != nil { + t.Fatalf("RollingScore pos: %v", err) + } + if score2 != 1.0 { + t.Errorf("score = %f, want 1.0 (clamped)", score2) + } +} + +func TestLedger_Append_Defaults(t *testing.T) { + db, owner := newLedgerTestDB(t) + ledger := NewLedger(db) + ctx := context.Background() + + // TaskDomain empty, Weight zero → should default to "default" and 1.0. + id, err := ledger.Append(ctx, Evidence{ + ConfigHash: "defaults-hash", + OwnerUserID: owner, + ScoreDelta: 0.3, + EvidenceRef: "test", + }) + if err != nil { + t.Fatalf("Append: %v", err) + } + if id <= 0 { + t.Errorf("id = %d, want > 0", id) + } + + var domain string + var weight float64 + if err := db.QueryRowContext(ctx, + `SELECT task_domain, weight FROM reputation_evidence WHERE id = ?`, id, + ).Scan(&domain, &weight); err != nil { + t.Fatalf("select: %v", err) + } + if domain != "default" { + t.Errorf("domain = %q, want default", domain) + } + if weight != 1.0 { + t.Errorf("weight = %f, want 1.0", weight) + } +} diff --git a/specs/018-dynamic-agent-spawning/tasks.md b/specs/018-dynamic-agent-spawning/tasks.md index 8b56214..b6c02ea 100644 --- a/specs/018-dynamic-agent-spawning/tasks.md +++ b/specs/018-dynamic-agent-spawning/tasks.md @@ -32,7 +32,7 @@ description: "Task list for Dynamic Agent Spawning" **Purpose**: Create new packages and migration files; wire through the build. No behavior yet. - [ ] T001 Create empty Go packages with `doc.go` files at `internal/goals/doc.go`, `internal/tasks/doc.go`, `internal/trust/doc.go`, `internal/secrets/doc.go` -- [ ] T002 [P] Create migration files (empty DDL, just table shells per data-model.md) at `internal/storage/schema/021_goals_tasks.sql`, `internal/storage/schema/022_agent_proposals.sql`, `internal/storage/schema/023_agent_trust_model.sql`, `internal/storage/schema/024_secrets.sql`, `internal/storage/schema/025_harness_runs_task_id.sql` +- [X] T002 [P] Create migration files (empty DDL, just table shells per data-model.md) at `internal/storage/schema/021_goals_tasks.sql`, `internal/storage/schema/022_agent_proposals.sql`, `internal/storage/schema/023_agent_trust_model.sql`, `internal/storage/schema/024_secrets.sql`, `internal/storage/schema/025_harness_runs_task_id.sql` — **note**: table renamed from `tasks` to `goal_tasks` (legacy `tasks` table exists from migration 001 for channel task auctions) - [ ] T003 [P] Create example scaffolding at `examples/doc-gardener/` mirroring `examples/cold-topic-explainer/` — copy `start.sh`, `stop.sh`, `run_task.sh`, `wrapper.sh`, `README.md` with doc-gardener placeholders - [ ] T004 [P] Add `golang.org/x/crypto/nacl/secretbox` to `go.mod` via `go get` and verify the pure-Go build still works (`CGO_ENABLED=0 go build ./...`) - [ ] T005 Create `specs/018-dynamic-agent-spawning/BUILD_NOTES.md` with the developer runbook (build commands, test commands, example commands) for the whole feature @@ -43,11 +43,11 @@ description: "Task list for Dynamic Agent Spawning" **Purpose**: Migrations, base types, and the trust primitive that every user story depends on. -- [ ] T010 Fill migration 021 DDL (goals + tasks tables, indexes, state CHECK constraints) in `internal/storage/schema/021_goals_tasks.sql` per data-model.md -- [ ] T011 Fill migration 022 DDL (agent_proposals + resource_requests tables) in `internal/storage/schema/022_agent_proposals.sql` per data-model.md -- [ ] T012 Fill migration 023 DDL (drop+recreate trust table; ALTER agents with config_hash/parent_agent_id/spawn_depth/system_prompt/autonomy_tier/tool_scope_json/quarantined_at; reputation_evidence table) in `internal/storage/schema/023_agent_trust_model.sql` -- [ ] T013 Fill migration 024 DDL (secrets table with nonce||ciphertext BLOB, unique-index-where-not-revoked) in `internal/storage/schema/024_secrets.sql` -- [ ] T014 Fill migration 025 DDL (ALTER harness_runs ADD task_id + index) in `internal/storage/schema/025_harness_runs_task_id.sql` +- [X] T010 Fill migration 021 DDL (goals + goal_tasks tables, indexes, state CHECK constraints) +- [X] T011 Fill migration 022 DDL (agent_proposals + resource_requests tables) +- [X] T012 Fill migration 023 DDL — **revised**: keep existing agent_trust table intact (wired to reactions), only ALTER agents + CREATE reputation_evidence +- [X] T013 Fill migration 024 DDL (secrets table) +- [X] T014 Fill migration 025 DDL (harness_runs.task_id) - [ ] T015 [P] Write data-migration Go code that extracts `system_prompt` from existing `harness_config_json` for pre-existing agents, idempotent, in `internal/storage/migrations_data.go` - [ ] T016 [P] Write data-migration Go code that computes `config_hash` for every existing agent and backfills the column in `internal/storage/migrations_data.go` - [ ] T017 Implement `trust.ConfigHash(agent)` canonical SHA-256 hashing (sorted-keys JSON) in `internal/trust/hash.go` with table-driven unit tests in `internal/trust/hash_test.go` verifying stability across shuffled input arrays