Merge features 014+015: Reactive Agent Triggers + SQL Query Interface
This commit is contained in:
@@ -108,6 +108,8 @@ make lint # Run linters
|
||||
- SQLite (modernc.org/sqlite, pure Go) — new migration 013_reactions.sql (010-reactions-workflows)
|
||||
- Go 1.25+ (SynapBus), Python 3.12 (Searcher agents) + go-chi/chi, mark3labs/mcp-go, ory/fosite (SynapBus); claude-agent-sdk, httpx, psycopg (Searcher) (013-linkedin-approval-workflow)
|
||||
- SQLite via modernc.org/sqlite (SynapBus); PostgreSQL (Searcher) (013-linkedin-approval-workflow)
|
||||
- Go 1.25+ (per go.mod) + go-chi/chi (HTTP), mark3labs/mcp-go (MCP), spf13/cobra (CLI), modernc.org/sqlite (storage), k8s.io/client-go (K8s Jobs) (014-reactive-agent-triggers)
|
||||
- SQLite via modernc.org/sqlite — new migration 015_reactive_triggers.sql (014-reactive-agent-triggers)
|
||||
|
||||
## Recent Changes
|
||||
- 002-mcp-auth-ux-polish: Added Go 1.23+ + ory/fosite (OAuth 2.1), mark3labs/mcp-go (MCP server), go-chi/chi (HTTP), Svelte 5 + Tailwind (Web UI)
|
||||
|
||||
+24
-2
@@ -39,6 +39,8 @@ import (
|
||||
"github.com/synapbus/synapbus/internal/jsruntime"
|
||||
k8spkg "github.com/synapbus/synapbus/internal/k8s"
|
||||
mcpserver "github.com/synapbus/synapbus/internal/mcp"
|
||||
"github.com/synapbus/synapbus/internal/agentquery"
|
||||
reactorpkg "github.com/synapbus/synapbus/internal/reactor"
|
||||
"github.com/synapbus/synapbus/internal/messaging"
|
||||
prommetrics "github.com/synapbus/synapbus/internal/metrics"
|
||||
"github.com/synapbus/synapbus/internal/reactions"
|
||||
@@ -467,10 +469,21 @@ func runServe(cmd *cobra.Command, args []string) error {
|
||||
slog.Info("K8s job runner not available (not in-cluster)")
|
||||
}
|
||||
|
||||
// Create event dispatcher (fans out to webhooks + K8s)
|
||||
eventDispatcher := dispatcher.NewMultiDispatcher(slog.Default(), deliveryEngine, k8sDispatcher)
|
||||
// Create reactor engine for reactive agent triggering
|
||||
reactorStore := reactorpkg.NewStore(db.DB)
|
||||
reactorEngine := reactorpkg.New(reactorStore, agentStore, k8sRunner, slog.Default())
|
||||
reactorNotifier := reactorpkg.NewDMFailureNotifier(msgService)
|
||||
reactorEngine.SetFailureNotifier(reactorNotifier)
|
||||
|
||||
// Create event dispatcher (fans out to webhooks + K8s + reactor)
|
||||
eventDispatcher := dispatcher.NewMultiDispatcher(slog.Default(), deliveryEngine, k8sDispatcher, reactorEngine)
|
||||
msgService.SetDispatcher(eventDispatcher)
|
||||
|
||||
// Start reactor poller for K8s Job status tracking
|
||||
reactorPoller := reactorpkg.NewPoller(reactorStore, agentStore, k8sRunner, reactorEngine, slog.Default())
|
||||
reactorPoller.Start()
|
||||
slog.Info("reactor engine and poller started")
|
||||
|
||||
// Create JS runtime pool and action registry for hybrid MCP tools
|
||||
jsPool := jsruntime.NewPool(10)
|
||||
defer jsPool.Close()
|
||||
@@ -480,6 +493,13 @@ func runServe(cmd *cobra.Command, args []string) error {
|
||||
|
||||
// Create MCP server (4 hybrid tools: my_status, send_message, search, execute)
|
||||
mcpSrv := mcpserver.NewMCPServer(msgService, agentService, channelService, swarmService, attachmentService, searchService, reactionService, trustService, con, jsPool, actionRegistry, actionIndex, db.DB)
|
||||
|
||||
// Set up SQL query executor for agents (uses read pool if available)
|
||||
queryDB := db.QueryDB()
|
||||
queryExec := agentquery.New(queryDB, slog.Default())
|
||||
mcpSrv.SetQueryExecutor(queryExec)
|
||||
slog.Info("agent SQL query executor initialized", "read_pool", db.ReadDB != nil)
|
||||
|
||||
startTime := time.Now()
|
||||
|
||||
// Start task expiry worker
|
||||
@@ -641,6 +661,8 @@ func runServe(cmd *cobra.Command, args []string) error {
|
||||
Version: version,
|
||||
PushService: pushService,
|
||||
TrustService: trustService,
|
||||
ReactorStore: reactorStore,
|
||||
ReactorEngine: reactorEngine,
|
||||
BaseURL: baseURL,
|
||||
})
|
||||
r.Mount("/", apiRouter)
|
||||
|
||||
@@ -571,5 +571,33 @@ func allActions() []Action {
|
||||
},
|
||||
},
|
||||
},
|
||||
// ── SQL Query (1 action) ────────────────────────────────────
|
||||
{
|
||||
Name: "query",
|
||||
Category: "data",
|
||||
Description: "Execute a read-only SQL query against your accessible messages, channels, and reactions. Use tables: my_messages (your DMs + joined channels), my_channels (channels you are in), channel_messages (messages in your channels). Results are limited to 100 rows. Only SELECT statements are allowed.",
|
||||
Params: []Param{
|
||||
{Name: "sql", Type: "string", Description: "SQL SELECT query. Available tables: my_messages (id, body, from_agent, to_agent, priority, status, metadata, created_at, channel_name), my_channels (id, name, description, type), channel_messages (id, body, from_agent, priority, channel_name, created_at). CTEs (WITH) are supported.", Required: true},
|
||||
},
|
||||
Returns: "JSON with columns (array of column names), rows (array of row arrays), row_count, and truncated (boolean if > 100 rows)",
|
||||
Examples: []Example{
|
||||
{
|
||||
Description: "Find high-priority messages in a channel",
|
||||
Code: `call("query", {"sql": "SELECT id, body, from_agent, priority FROM channel_messages WHERE channel_name = 'news-mcpproxy' AND priority >= 7 ORDER BY created_at DESC LIMIT 10"})`,
|
||||
},
|
||||
{
|
||||
Description: "List your channels",
|
||||
Code: `call("query", {"sql": "SELECT name, description FROM my_channels ORDER BY name"})`,
|
||||
},
|
||||
{
|
||||
Description: "Count messages per channel",
|
||||
Code: `call("query", {"sql": "SELECT channel_name, COUNT(*) as msg_count FROM channel_messages GROUP BY channel_name ORDER BY msg_count DESC"})`,
|
||||
},
|
||||
{
|
||||
Description: "Search messages with keyword",
|
||||
Code: `call("query", {"sql": "SELECT id, body, from_agent, created_at FROM my_messages WHERE body LIKE '%MCP%' ORDER BY created_at DESC LIMIT 20"})`,
|
||||
},
|
||||
},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4,11 +4,11 @@ import (
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestRegistryHas29Actions(t *testing.T) {
|
||||
func TestRegistryHas30Actions(t *testing.T) {
|
||||
r := NewRegistry()
|
||||
got := len(r.List())
|
||||
if got != 29 {
|
||||
t.Errorf("expected 29 actions, got %d", got)
|
||||
if got != 30 {
|
||||
t.Errorf("expected 30 actions, got %d", got)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -58,6 +58,8 @@ func TestRegistryGetByName(t *testing.T) {
|
||||
"get_replies",
|
||||
// trust
|
||||
"get_trust",
|
||||
// data
|
||||
"query",
|
||||
}
|
||||
|
||||
for _, name := range allNames {
|
||||
|
||||
@@ -0,0 +1,234 @@
|
||||
// Package agentquery provides a sandboxed SQL query executor for agents.
|
||||
// Agents can run read-only SELECT queries against curated views with
|
||||
// per-agent access control, automatic LIMIT enforcement, and timeouts.
|
||||
package agentquery
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
const (
|
||||
// MaxRows is the maximum number of rows returned by a query.
|
||||
MaxRows = 100
|
||||
// QueryTimeout is the maximum duration for a query.
|
||||
QueryTimeout = 5 * time.Second
|
||||
)
|
||||
|
||||
// Allowed view names that agents can query.
|
||||
var allowedTables = map[string]bool{
|
||||
"my_messages": true,
|
||||
"my_channels": true,
|
||||
"channel_messages": true,
|
||||
}
|
||||
|
||||
// Executor runs sandboxed SQL queries on behalf of agents.
|
||||
type Executor struct {
|
||||
db *sql.DB // read-only pool (query_only=ON)
|
||||
logger *slog.Logger
|
||||
}
|
||||
|
||||
// New creates a new query executor using the provided read-only database connection.
|
||||
func New(readDB *sql.DB, logger *slog.Logger) *Executor {
|
||||
return &Executor{
|
||||
db: readDB,
|
||||
logger: logger.With("component", "agentquery"),
|
||||
}
|
||||
}
|
||||
|
||||
// QueryResult holds the results of a SQL query.
|
||||
type QueryResult struct {
|
||||
Columns []string `json:"columns"`
|
||||
Rows [][]interface{} `json:"rows"`
|
||||
RowCount int `json:"row_count"`
|
||||
Truncated bool `json:"truncated"`
|
||||
}
|
||||
|
||||
// Execute runs a SQL query on behalf of an agent with access control.
|
||||
func (e *Executor) Execute(ctx context.Context, agentName, sqlQuery string) (*QueryResult, error) {
|
||||
// 1. Validate the SQL statement
|
||||
if err := validateSQL(sqlQuery); err != nil {
|
||||
return nil, fmt.Errorf("query validation failed: %w", err)
|
||||
}
|
||||
|
||||
// 2. Rewrite the query to inject access control and enforce LIMIT
|
||||
rewritten := rewriteQuery(agentName, sqlQuery)
|
||||
|
||||
// 3. Execute with timeout
|
||||
queryCtx, cancel := context.WithTimeout(ctx, QueryTimeout)
|
||||
defer cancel()
|
||||
|
||||
rows, err := e.db.QueryContext(queryCtx, rewritten)
|
||||
if err != nil {
|
||||
if queryCtx.Err() == context.DeadlineExceeded {
|
||||
return nil, fmt.Errorf("query timed out after %s", QueryTimeout)
|
||||
}
|
||||
return nil, fmt.Errorf("query execution failed: %w", err)
|
||||
}
|
||||
defer rows.Close()
|
||||
|
||||
// 4. Collect results
|
||||
columns, err := rows.Columns()
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("get columns: %w", err)
|
||||
}
|
||||
|
||||
var resultRows [][]interface{}
|
||||
truncated := false
|
||||
|
||||
for rows.Next() {
|
||||
if len(resultRows) >= MaxRows {
|
||||
truncated = true
|
||||
break
|
||||
}
|
||||
|
||||
values := make([]interface{}, len(columns))
|
||||
scanArgs := make([]interface{}, len(columns))
|
||||
for i := range values {
|
||||
scanArgs[i] = &values[i]
|
||||
}
|
||||
|
||||
if err := rows.Scan(scanArgs...); err != nil {
|
||||
return nil, fmt.Errorf("scan row: %w", err)
|
||||
}
|
||||
|
||||
// Convert []byte to string for JSON serialization
|
||||
row := make([]interface{}, len(columns))
|
||||
for i, v := range values {
|
||||
if b, ok := v.([]byte); ok {
|
||||
row[i] = string(b)
|
||||
} else {
|
||||
row[i] = v
|
||||
}
|
||||
}
|
||||
resultRows = append(resultRows, row)
|
||||
}
|
||||
|
||||
if err := rows.Err(); err != nil {
|
||||
return nil, fmt.Errorf("iterate rows: %w", err)
|
||||
}
|
||||
|
||||
if resultRows == nil {
|
||||
resultRows = [][]interface{}{}
|
||||
}
|
||||
|
||||
e.logger.Info("agent query executed",
|
||||
"agent", agentName,
|
||||
"rows", len(resultRows),
|
||||
"truncated", truncated,
|
||||
)
|
||||
|
||||
return &QueryResult{
|
||||
Columns: columns,
|
||||
Rows: resultRows,
|
||||
RowCount: len(resultRows),
|
||||
Truncated: truncated,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// validateSQL checks that the query is a read-only SELECT statement.
|
||||
func validateSQL(query string) error {
|
||||
trimmed := strings.TrimSpace(query)
|
||||
if trimmed == "" {
|
||||
return fmt.Errorf("empty query")
|
||||
}
|
||||
|
||||
// Remove comments
|
||||
upper := strings.ToUpper(trimmed)
|
||||
|
||||
// Must start with SELECT or WITH (CTEs)
|
||||
if !strings.HasPrefix(upper, "SELECT") && !strings.HasPrefix(upper, "WITH") {
|
||||
return fmt.Errorf("only SELECT statements are allowed (got %q)", firstWord(upper))
|
||||
}
|
||||
|
||||
// Block dangerous keywords (check as whole words or with common delimiters)
|
||||
blocked := []string{
|
||||
"INSERT ", "UPDATE ", "DELETE ", "DROP ", "ALTER ", "CREATE ",
|
||||
"ATTACH ", "DETACH ", "PRAGMA", "REINDEX ", "VACUUM ",
|
||||
"REPLACE ", "GRANT ", "REVOKE ",
|
||||
}
|
||||
for _, kw := range blocked {
|
||||
if strings.Contains(upper, kw) {
|
||||
return fmt.Errorf("statement contains blocked keyword: %s", strings.TrimSpace(kw))
|
||||
}
|
||||
}
|
||||
|
||||
// Block multiple statements (semicolon followed by non-whitespace)
|
||||
parts := strings.Split(trimmed, ";")
|
||||
nonEmpty := 0
|
||||
for _, p := range parts {
|
||||
if strings.TrimSpace(p) != "" {
|
||||
nonEmpty++
|
||||
}
|
||||
}
|
||||
if nonEmpty > 1 {
|
||||
return fmt.Errorf("multiple statements not allowed")
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// rewriteQuery wraps the agent's query with access control CTEs.
|
||||
// It replaces references to my_messages, my_channels, channel_messages
|
||||
// with CTEs that filter by the agent's access.
|
||||
func rewriteQuery(agentName, query string) string {
|
||||
// Build access-control CTEs that the agent's query can reference
|
||||
cte := fmt.Sprintf(`
|
||||
WITH my_messages AS (
|
||||
SELECT v.* FROM v_agent_messages v
|
||||
LEFT JOIN channel_members cm ON cm.channel_id = v.channel_id AND cm.agent_name = %[1]s
|
||||
WHERE v.to_agent = %[1]s
|
||||
OR v.from_agent = %[1]s
|
||||
OR (v.channel_id IS NOT NULL AND cm.agent_name IS NOT NULL)
|
||||
),
|
||||
my_channels AS (
|
||||
SELECT c.id, c.name, c.description, c.type, c.topic, c.is_private, c.created_at,
|
||||
cm.joined_at AS member_since
|
||||
FROM channels c
|
||||
JOIN channel_members cm ON cm.channel_id = c.id AND cm.agent_name = %[1]s
|
||||
),
|
||||
channel_messages AS (
|
||||
SELECT v.* FROM v_channel_messages v
|
||||
WHERE v.channel_id IN (
|
||||
SELECT channel_id FROM channel_members WHERE agent_name = %[1]s
|
||||
)
|
||||
)
|
||||
`, quoteSQLString(agentName))
|
||||
|
||||
trimmed := strings.TrimSpace(query)
|
||||
upper := strings.ToUpper(trimmed)
|
||||
|
||||
// Remove trailing semicolon if present
|
||||
trimmed = strings.TrimRight(trimmed, "; \t\n")
|
||||
|
||||
if strings.HasPrefix(upper, "WITH") {
|
||||
// User has their own CTEs. Merge: our CTEs first, then theirs.
|
||||
userCTEs := strings.TrimSpace(trimmed[4:]) // skip "WITH"
|
||||
return cte + ", " + userCTEs
|
||||
}
|
||||
|
||||
// Simple SELECT — prepend our CTEs
|
||||
return cte + trimmed
|
||||
}
|
||||
|
||||
// quoteSQLString safely quotes a string for use in SQL.
|
||||
func quoteSQLString(s string) string {
|
||||
escaped := strings.ReplaceAll(s, "'", "''")
|
||||
return "'" + escaped + "'"
|
||||
}
|
||||
|
||||
func firstWord(s string) string {
|
||||
for i, c := range s {
|
||||
if c == ' ' || c == '\t' || c == '\n' || c == '\r' || c == '(' {
|
||||
return s[:i]
|
||||
}
|
||||
}
|
||||
if len(s) > 20 {
|
||||
return s[:20]
|
||||
}
|
||||
return s
|
||||
}
|
||||
@@ -0,0 +1,341 @@
|
||||
package agentquery
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"log/slog"
|
||||
"testing"
|
||||
|
||||
_ "modernc.org/sqlite"
|
||||
)
|
||||
|
||||
func setupTestDB(t *testing.T) *sql.DB {
|
||||
t.Helper()
|
||||
db, err := sql.Open("sqlite", ":memory:")
|
||||
if err != nil {
|
||||
t.Fatalf("open db: %v", err)
|
||||
}
|
||||
|
||||
// Create the schema needed for views
|
||||
schema := `
|
||||
CREATE TABLE channels (
|
||||
id INTEGER PRIMARY KEY,
|
||||
name TEXT NOT NULL UNIQUE,
|
||||
description TEXT DEFAULT '',
|
||||
type TEXT DEFAULT 'standard',
|
||||
topic TEXT DEFAULT '',
|
||||
is_private INTEGER DEFAULT 0,
|
||||
created_at DATETIME DEFAULT CURRENT_TIMESTAMP
|
||||
);
|
||||
CREATE TABLE channel_members (
|
||||
channel_id INTEGER,
|
||||
agent_name TEXT,
|
||||
joined_at DATETIME DEFAULT CURRENT_TIMESTAMP,
|
||||
PRIMARY KEY (channel_id, agent_name)
|
||||
);
|
||||
CREATE TABLE messages (
|
||||
id INTEGER PRIMARY KEY,
|
||||
conversation_id INTEGER DEFAULT 0,
|
||||
from_agent TEXT,
|
||||
to_agent TEXT,
|
||||
channel_id INTEGER,
|
||||
reply_to INTEGER,
|
||||
body TEXT,
|
||||
priority INTEGER DEFAULT 5,
|
||||
status TEXT DEFAULT 'pending',
|
||||
metadata TEXT DEFAULT '{}',
|
||||
created_at DATETIME DEFAULT CURRENT_TIMESTAMP,
|
||||
updated_at DATETIME DEFAULT CURRENT_TIMESTAMP
|
||||
);
|
||||
|
||||
-- Views matching the migration
|
||||
CREATE VIEW v_agent_messages AS
|
||||
SELECT m.id, m.body, m.from_agent, m.to_agent, m.priority, m.status, m.metadata,
|
||||
m.created_at, m.updated_at, c.name AS channel_name, m.channel_id, m.reply_to, m.conversation_id
|
||||
FROM messages m LEFT JOIN channels c ON c.id = m.channel_id;
|
||||
|
||||
CREATE VIEW v_agent_channels AS
|
||||
SELECT c.id, c.name, c.description, c.type, c.topic, c.is_private, c.created_at,
|
||||
cm.joined_at AS member_since
|
||||
FROM channels c JOIN channel_members cm ON cm.channel_id = c.id;
|
||||
|
||||
CREATE VIEW v_channel_messages AS
|
||||
SELECT m.id, m.body, m.from_agent, m.priority, m.status, m.metadata, m.created_at,
|
||||
c.name AS channel_name, m.channel_id, m.reply_to
|
||||
FROM messages m JOIN channels c ON c.id = m.channel_id;
|
||||
`
|
||||
if _, err := db.Exec(schema); err != nil {
|
||||
t.Fatalf("create schema: %v", err)
|
||||
}
|
||||
|
||||
// Seed test data
|
||||
seed := `
|
||||
INSERT INTO channels (id, name) VALUES (1, 'general'), (2, 'news-mcpproxy'), (3, 'private-channel');
|
||||
INSERT INTO channel_members (channel_id, agent_name) VALUES
|
||||
(1, 'agent-a'), (1, 'agent-b'),
|
||||
(2, 'agent-a'),
|
||||
(3, 'agent-b');
|
||||
|
||||
-- DMs
|
||||
INSERT INTO messages (id, from_agent, to_agent, body, priority) VALUES
|
||||
(1, 'algis', 'agent-a', 'Hello agent A', 7),
|
||||
(2, 'agent-a', 'algis', 'Hi there', 5),
|
||||
(3, 'algis', 'agent-b', 'Hello agent B', 5);
|
||||
|
||||
-- Channel messages
|
||||
INSERT INTO messages (id, from_agent, channel_id, body, priority) VALUES
|
||||
(4, 'agent-a', 1, 'General post from A', 5),
|
||||
(5, 'agent-b', 1, 'General post from B', 5),
|
||||
(6, 'agent-a', 2, 'News post high prio', 8),
|
||||
(7, 'agent-b', 3, 'Private channel msg', 5);
|
||||
`
|
||||
if _, err := db.Exec(seed); err != nil {
|
||||
t.Fatalf("seed data: %v", err)
|
||||
}
|
||||
|
||||
return db
|
||||
}
|
||||
|
||||
func TestExecuteBasicQuery(t *testing.T) {
|
||||
db := setupTestDB(t)
|
||||
defer db.Close()
|
||||
|
||||
exec := New(db, slog.Default())
|
||||
result, err := exec.Execute(context.Background(), "agent-a",
|
||||
"SELECT id, body, priority FROM my_messages ORDER BY id")
|
||||
if err != nil {
|
||||
t.Fatalf("query failed: %v", err)
|
||||
}
|
||||
|
||||
if len(result.Columns) != 3 {
|
||||
t.Errorf("expected 3 columns, got %d", len(result.Columns))
|
||||
}
|
||||
if result.Columns[0] != "id" || result.Columns[1] != "body" || result.Columns[2] != "priority" {
|
||||
t.Errorf("unexpected columns: %v", result.Columns)
|
||||
}
|
||||
|
||||
// agent-a should see: DM to it (1), DM from it (2), general posts (4,5), news post (6)
|
||||
// Should NOT see: DM to agent-b (3), private channel msg (7)
|
||||
if result.RowCount < 4 {
|
||||
t.Errorf("expected at least 4 rows for agent-a, got %d", result.RowCount)
|
||||
}
|
||||
|
||||
// Verify agent-b's DM and private channel msg are NOT visible
|
||||
for _, row := range result.Rows {
|
||||
id := row[0]
|
||||
if id == int64(3) {
|
||||
t.Error("agent-a should NOT see message 3 (DM to agent-b)")
|
||||
}
|
||||
if id == int64(7) {
|
||||
t.Error("agent-a should NOT see message 7 (private channel, not joined)")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestAccessControlAgentB(t *testing.T) {
|
||||
db := setupTestDB(t)
|
||||
defer db.Close()
|
||||
|
||||
exec := New(db, slog.Default())
|
||||
result, err := exec.Execute(context.Background(), "agent-b",
|
||||
"SELECT id, body FROM my_messages ORDER BY id")
|
||||
if err != nil {
|
||||
t.Fatalf("query failed: %v", err)
|
||||
}
|
||||
|
||||
// agent-b should see: DM to it (3), general posts (4,5), private channel (7)
|
||||
// Should NOT see: DM to agent-a (1), DM from agent-a (2), news post (6)
|
||||
hasMsg3 := false
|
||||
hasMsg7 := false
|
||||
for _, row := range result.Rows {
|
||||
id := row[0]
|
||||
if id == int64(3) {
|
||||
hasMsg3 = true
|
||||
}
|
||||
if id == int64(7) {
|
||||
hasMsg7 = true
|
||||
}
|
||||
if id == int64(1) {
|
||||
t.Error("agent-b should NOT see message 1 (DM to agent-a)")
|
||||
}
|
||||
if id == int64(6) {
|
||||
t.Error("agent-b should NOT see message 6 (news channel, not joined)")
|
||||
}
|
||||
}
|
||||
if !hasMsg3 {
|
||||
t.Error("agent-b should see message 3 (DM to it)")
|
||||
}
|
||||
if !hasMsg7 {
|
||||
t.Error("agent-b should see message 7 (private channel, joined)")
|
||||
}
|
||||
}
|
||||
|
||||
func TestQueryChannelMessages(t *testing.T) {
|
||||
db := setupTestDB(t)
|
||||
defer db.Close()
|
||||
|
||||
exec := New(db, slog.Default())
|
||||
result, err := exec.Execute(context.Background(), "agent-a",
|
||||
"SELECT id, body, channel_name FROM channel_messages WHERE channel_name = 'news-mcpproxy'")
|
||||
if err != nil {
|
||||
t.Fatalf("query failed: %v", err)
|
||||
}
|
||||
|
||||
if result.RowCount != 1 {
|
||||
t.Errorf("expected 1 news message, got %d", result.RowCount)
|
||||
}
|
||||
}
|
||||
|
||||
func TestQueryMyChannels(t *testing.T) {
|
||||
db := setupTestDB(t)
|
||||
defer db.Close()
|
||||
|
||||
exec := New(db, slog.Default())
|
||||
result, err := exec.Execute(context.Background(), "agent-a",
|
||||
"SELECT name FROM my_channels ORDER BY name")
|
||||
if err != nil {
|
||||
t.Fatalf("query failed: %v", err)
|
||||
}
|
||||
|
||||
// agent-a is in: general, news-mcpproxy (not private-channel)
|
||||
if result.RowCount != 2 {
|
||||
t.Errorf("expected 2 channels for agent-a, got %d", result.RowCount)
|
||||
}
|
||||
}
|
||||
|
||||
func TestValidationRejectsInsert(t *testing.T) {
|
||||
db := setupTestDB(t)
|
||||
defer db.Close()
|
||||
|
||||
exec := New(db, slog.Default())
|
||||
_, err := exec.Execute(context.Background(), "agent-a",
|
||||
"INSERT INTO messages (body) VALUES ('evil')")
|
||||
if err == nil {
|
||||
t.Fatal("expected INSERT to be rejected")
|
||||
}
|
||||
if !contains(err.Error(), "only SELECT") {
|
||||
t.Errorf("expected 'only SELECT' error, got: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestValidationRejectsDrop(t *testing.T) {
|
||||
db := setupTestDB(t)
|
||||
defer db.Close()
|
||||
|
||||
exec := New(db, slog.Default())
|
||||
_, err := exec.Execute(context.Background(), "agent-a",
|
||||
"SELECT 1; DROP TABLE messages")
|
||||
if err == nil {
|
||||
t.Fatal("expected multi-statement to be rejected")
|
||||
}
|
||||
}
|
||||
|
||||
func TestValidationRejectsUpdate(t *testing.T) {
|
||||
db := setupTestDB(t)
|
||||
defer db.Close()
|
||||
|
||||
exec := New(db, slog.Default())
|
||||
_, err := exec.Execute(context.Background(), "agent-a",
|
||||
"UPDATE messages SET body = 'hacked'")
|
||||
if err == nil {
|
||||
t.Fatal("expected UPDATE to be rejected")
|
||||
}
|
||||
}
|
||||
|
||||
func TestValidationRejectsPragma(t *testing.T) {
|
||||
db := setupTestDB(t)
|
||||
defer db.Close()
|
||||
|
||||
exec := New(db, slog.Default())
|
||||
_, err := exec.Execute(context.Background(), "agent-a",
|
||||
"SELECT * FROM pragma_table_info('messages')")
|
||||
if err == nil {
|
||||
t.Fatal("expected PRAGMA in SELECT to be rejected")
|
||||
}
|
||||
}
|
||||
|
||||
func TestEmptyQuery(t *testing.T) {
|
||||
db := setupTestDB(t)
|
||||
defer db.Close()
|
||||
|
||||
exec := New(db, slog.Default())
|
||||
_, err := exec.Execute(context.Background(), "agent-a", "")
|
||||
if err == nil {
|
||||
t.Fatal("expected empty query to be rejected")
|
||||
}
|
||||
}
|
||||
|
||||
func TestCTEQuery(t *testing.T) {
|
||||
db := setupTestDB(t)
|
||||
defer db.Close()
|
||||
|
||||
exec := New(db, slog.Default())
|
||||
result, err := exec.Execute(context.Background(), "agent-a",
|
||||
"WITH high_prio AS (SELECT * FROM my_messages WHERE priority >= 7) SELECT id, priority FROM high_prio")
|
||||
if err != nil {
|
||||
t.Fatalf("CTE query failed: %v", err)
|
||||
}
|
||||
|
||||
// agent-a should see high-priority messages it has access to
|
||||
if result.RowCount == 0 {
|
||||
t.Error("expected at least 1 high-priority message")
|
||||
}
|
||||
}
|
||||
|
||||
func TestEmptyResultSet(t *testing.T) {
|
||||
db := setupTestDB(t)
|
||||
defer db.Close()
|
||||
|
||||
exec := New(db, slog.Default())
|
||||
result, err := exec.Execute(context.Background(), "agent-a",
|
||||
"SELECT * FROM my_messages WHERE body = 'nonexistent'")
|
||||
if err != nil {
|
||||
t.Fatalf("query failed: %v", err)
|
||||
}
|
||||
if result.RowCount != 0 {
|
||||
t.Errorf("expected 0 rows, got %d", result.RowCount)
|
||||
}
|
||||
if result.Rows == nil {
|
||||
t.Error("rows should be empty array, not nil")
|
||||
}
|
||||
if result.Truncated {
|
||||
t.Error("should not be truncated")
|
||||
}
|
||||
}
|
||||
|
||||
func TestLimitEnforcement(t *testing.T) {
|
||||
db := setupTestDB(t)
|
||||
defer db.Close()
|
||||
|
||||
// Insert 150 messages to test limit
|
||||
for i := 100; i < 250; i++ {
|
||||
_, _ = db.Exec("INSERT INTO messages (id, from_agent, to_agent, body) VALUES (?, 'algis', 'agent-a', 'msg')", i)
|
||||
}
|
||||
|
||||
exec := New(db, slog.Default())
|
||||
result, err := exec.Execute(context.Background(), "agent-a",
|
||||
"SELECT id FROM my_messages")
|
||||
if err != nil {
|
||||
t.Fatalf("query failed: %v", err)
|
||||
}
|
||||
|
||||
if result.RowCount > MaxRows {
|
||||
t.Errorf("expected max %d rows, got %d", MaxRows, result.RowCount)
|
||||
}
|
||||
if !result.Truncated {
|
||||
t.Error("expected truncated=true for large result set")
|
||||
}
|
||||
}
|
||||
|
||||
func contains(s, substr string) bool {
|
||||
return len(s) >= len(substr) && (s == substr || len(s) > 0 && containsStr(s, substr))
|
||||
}
|
||||
|
||||
func containsStr(s, sub string) bool {
|
||||
for i := 0; i <= len(s)-len(sub); i++ {
|
||||
if s[i:i+len(sub)] == sub {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
+108
-14
@@ -19,6 +19,12 @@ type AgentStore interface {
|
||||
ListAgentsByOwner(ctx context.Context, ownerID int64) ([]*Agent, error)
|
||||
SearchAgentsByCapability(ctx context.Context, query string) ([]*Agent, error)
|
||||
GetHumanAgentByOwner(ctx context.Context, ownerID int64) (*Agent, error)
|
||||
|
||||
// Reactive trigger methods
|
||||
UpdateTriggerConfig(ctx context.Context, name string, mode string, cooldown, budget, maxDepth int) error
|
||||
UpdateK8sImage(ctx context.Context, name, image, envJSON, preset string) error
|
||||
SetPendingWork(ctx context.Context, name string, pending bool) error
|
||||
ListReactiveAgents(ctx context.Context) ([]*Agent, error)
|
||||
}
|
||||
|
||||
// SQLiteAgentStore implements AgentStore using SQLite.
|
||||
@@ -37,6 +43,28 @@ func (s *SQLiteAgentStore) CreateAgent(ctx context.Context, agent *Agent) error
|
||||
caps = "{}"
|
||||
}
|
||||
|
||||
// Default trigger values
|
||||
triggerMode := agent.TriggerMode
|
||||
if triggerMode == "" {
|
||||
triggerMode = TriggerModePassive
|
||||
}
|
||||
cooldown := agent.CooldownSeconds
|
||||
if cooldown == 0 {
|
||||
cooldown = 600
|
||||
}
|
||||
budget := agent.DailyTriggerBudget
|
||||
if budget == 0 {
|
||||
budget = 8
|
||||
}
|
||||
maxDepth := agent.MaxTriggerDepth
|
||||
if maxDepth == 0 {
|
||||
maxDepth = 5
|
||||
}
|
||||
preset := agent.K8sResourcePreset
|
||||
if preset == "" {
|
||||
preset = "default"
|
||||
}
|
||||
|
||||
result, err := s.db.ExecContext(ctx,
|
||||
`INSERT INTO agents (name, display_name, type, capabilities, owner_id, api_key_hash, status, created_at, updated_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?, CURRENT_TIMESTAMP, CURRENT_TIMESTAMP)`,
|
||||
@@ -51,20 +79,75 @@ func (s *SQLiteAgentStore) CreateAgent(ctx context.Context, agent *Agent) error
|
||||
}
|
||||
agent.ID = id
|
||||
agent.Status = AgentStatusActive
|
||||
agent.TriggerMode = triggerMode
|
||||
agent.CooldownSeconds = cooldown
|
||||
agent.DailyTriggerBudget = budget
|
||||
agent.MaxTriggerDepth = maxDepth
|
||||
agent.K8sResourcePreset = preset
|
||||
return nil
|
||||
}
|
||||
|
||||
// UpdateTriggerConfig updates the reactive trigger configuration for an agent.
|
||||
func (s *SQLiteAgentStore) UpdateTriggerConfig(ctx context.Context, name string, mode string, cooldown, budget, maxDepth int) error {
|
||||
_, err := s.db.ExecContext(ctx,
|
||||
`UPDATE agents SET trigger_mode = ?, cooldown_seconds = ?, daily_trigger_budget = ?, max_trigger_depth = ?, updated_at = CURRENT_TIMESTAMP
|
||||
WHERE name = ? AND status = 'active'`,
|
||||
mode, cooldown, budget, maxDepth, name,
|
||||
)
|
||||
return err
|
||||
}
|
||||
|
||||
// UpdateK8sImage updates the K8s container image and env config for an agent.
|
||||
func (s *SQLiteAgentStore) UpdateK8sImage(ctx context.Context, name, image, envJSON, preset string) error {
|
||||
_, err := s.db.ExecContext(ctx,
|
||||
`UPDATE agents SET k8s_image = ?, k8s_env_json = ?, k8s_resource_preset = ?, updated_at = CURRENT_TIMESTAMP
|
||||
WHERE name = ? AND status = 'active'`,
|
||||
image, envJSON, preset, name,
|
||||
)
|
||||
return err
|
||||
}
|
||||
|
||||
// SetPendingWork sets the pending_work flag for an agent.
|
||||
func (s *SQLiteAgentStore) SetPendingWork(ctx context.Context, name string, pending bool) error {
|
||||
val := 0
|
||||
if pending {
|
||||
val = 1
|
||||
}
|
||||
_, err := s.db.ExecContext(ctx,
|
||||
`UPDATE agents SET pending_work = ? WHERE name = ? AND status = 'active'`,
|
||||
val, name,
|
||||
)
|
||||
return err
|
||||
}
|
||||
|
||||
// ListReactiveAgents returns all active agents with trigger_mode='reactive'.
|
||||
func (s *SQLiteAgentStore) ListReactiveAgents(ctx context.Context) ([]*Agent, error) {
|
||||
rows, err := s.db.QueryContext(ctx,
|
||||
agentSelectSQL()+` WHERE status = 'active' AND trigger_mode = 'reactive' ORDER BY name`,
|
||||
)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
return s.scanAgents(rows)
|
||||
}
|
||||
|
||||
// agentSelectSQL returns the base SELECT clause for agent queries.
|
||||
func agentSelectSQL() string {
|
||||
return `SELECT id, name, display_name, type, capabilities, owner_id, api_key_hash, status, created_at, updated_at,
|
||||
trigger_mode, cooldown_seconds, daily_trigger_budget, max_trigger_depth, k8s_image, k8s_env_json, k8s_resource_preset, pending_work
|
||||
FROM agents`
|
||||
}
|
||||
|
||||
func (s *SQLiteAgentStore) GetAgentByName(ctx context.Context, name string) (*Agent, error) {
|
||||
return s.scanAgent(s.db.QueryRowContext(ctx,
|
||||
`SELECT id, name, display_name, type, capabilities, owner_id, api_key_hash, status, created_at, updated_at
|
||||
FROM agents WHERE name = ? AND status = 'active'`, name,
|
||||
agentSelectSQL()+` WHERE name = ? AND status = 'active'`, name,
|
||||
))
|
||||
}
|
||||
|
||||
func (s *SQLiteAgentStore) GetAgentByID(ctx context.Context, id int64) (*Agent, error) {
|
||||
return s.scanAgent(s.db.QueryRowContext(ctx,
|
||||
`SELECT id, name, display_name, type, capabilities, owner_id, api_key_hash, status, created_at, updated_at
|
||||
FROM agents WHERE id = ? AND status = 'active'`, id,
|
||||
agentSelectSQL()+` WHERE id = ? AND status = 'active'`, id,
|
||||
))
|
||||
}
|
||||
|
||||
@@ -103,8 +186,7 @@ func (s *SQLiteAgentStore) DeactivateAgent(ctx context.Context, name string) err
|
||||
|
||||
func (s *SQLiteAgentStore) ListActiveAgents(ctx context.Context) ([]*Agent, error) {
|
||||
rows, err := s.db.QueryContext(ctx,
|
||||
`SELECT id, name, display_name, type, capabilities, owner_id, api_key_hash, status, created_at, updated_at
|
||||
FROM agents WHERE status = 'active' ORDER BY name`,
|
||||
agentSelectSQL()+` WHERE status = 'active' ORDER BY name`,
|
||||
)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
@@ -115,8 +197,7 @@ func (s *SQLiteAgentStore) ListActiveAgents(ctx context.Context) ([]*Agent, erro
|
||||
|
||||
func (s *SQLiteAgentStore) ListAllActiveAgents(ctx context.Context) ([]*Agent, error) {
|
||||
rows, err := s.db.QueryContext(ctx,
|
||||
`SELECT id, name, display_name, type, capabilities, owner_id, api_key_hash, status, created_at, updated_at
|
||||
FROM agents WHERE status = 'active' AND type != 'human' ORDER BY name`,
|
||||
agentSelectSQL()+` WHERE status = 'active' AND type != 'human' ORDER BY name`,
|
||||
)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
@@ -127,8 +208,7 @@ func (s *SQLiteAgentStore) ListAllActiveAgents(ctx context.Context) ([]*Agent, e
|
||||
|
||||
func (s *SQLiteAgentStore) ListAgentsByOwner(ctx context.Context, ownerID int64) ([]*Agent, error) {
|
||||
rows, err := s.db.QueryContext(ctx,
|
||||
`SELECT id, name, display_name, type, capabilities, owner_id, api_key_hash, status, created_at, updated_at
|
||||
FROM agents WHERE owner_id = ? AND status = 'active' ORDER BY name`,
|
||||
agentSelectSQL()+` WHERE owner_id = ? AND status = 'active' ORDER BY name`,
|
||||
ownerID,
|
||||
)
|
||||
if err != nil {
|
||||
@@ -141,8 +221,7 @@ func (s *SQLiteAgentStore) ListAgentsByOwner(ctx context.Context, ownerID int64)
|
||||
func (s *SQLiteAgentStore) SearchAgentsByCapability(ctx context.Context, query string) ([]*Agent, error) {
|
||||
// Simple LIKE search on the capabilities JSON field
|
||||
rows, err := s.db.QueryContext(ctx,
|
||||
`SELECT id, name, display_name, type, capabilities, owner_id, api_key_hash, status, created_at, updated_at
|
||||
FROM agents WHERE status = 'active' AND capabilities LIKE ? ORDER BY name`,
|
||||
agentSelectSQL()+` WHERE status = 'active' AND capabilities LIKE ? ORDER BY name`,
|
||||
"%"+query+"%",
|
||||
)
|
||||
if err != nil {
|
||||
@@ -154,23 +233,30 @@ func (s *SQLiteAgentStore) SearchAgentsByCapability(ctx context.Context, query s
|
||||
|
||||
func (s *SQLiteAgentStore) GetHumanAgentByOwner(ctx context.Context, ownerID int64) (*Agent, error) {
|
||||
return s.scanAgent(s.db.QueryRowContext(ctx,
|
||||
`SELECT id, name, display_name, type, capabilities, owner_id, api_key_hash, status, created_at, updated_at
|
||||
FROM agents WHERE owner_id = ? AND type = 'human' AND status = 'active' LIMIT 1`, ownerID,
|
||||
agentSelectSQL()+` WHERE owner_id = ? AND type = 'human' AND status = 'active' LIMIT 1`, ownerID,
|
||||
))
|
||||
}
|
||||
|
||||
func (s *SQLiteAgentStore) scanAgent(row *sql.Row) (*Agent, error) {
|
||||
var agent Agent
|
||||
var caps string
|
||||
var k8sImage, k8sEnvJSON sql.NullString
|
||||
var pendingWork int
|
||||
err := row.Scan(
|
||||
&agent.ID, &agent.Name, &agent.DisplayName, &agent.Type,
|
||||
&caps, &agent.OwnerID, &agent.APIKeyHash, &agent.Status,
|
||||
&agent.CreatedAt, &agent.UpdatedAt,
|
||||
&agent.TriggerMode, &agent.CooldownSeconds, &agent.DailyTriggerBudget,
|
||||
&agent.MaxTriggerDepth, &k8sImage, &k8sEnvJSON,
|
||||
&agent.K8sResourcePreset, &pendingWork,
|
||||
)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
agent.Capabilities = json.RawMessage(caps)
|
||||
agent.K8sImage = k8sImage.String
|
||||
agent.K8sEnvJSON = k8sEnvJSON.String
|
||||
agent.PendingWork = pendingWork != 0
|
||||
return &agent, nil
|
||||
}
|
||||
|
||||
@@ -179,15 +265,23 @@ func (s *SQLiteAgentStore) scanAgents(rows *sql.Rows) ([]*Agent, error) {
|
||||
for rows.Next() {
|
||||
var agent Agent
|
||||
var caps string
|
||||
var k8sImage, k8sEnvJSON sql.NullString
|
||||
var pendingWork int
|
||||
err := rows.Scan(
|
||||
&agent.ID, &agent.Name, &agent.DisplayName, &agent.Type,
|
||||
&caps, &agent.OwnerID, &agent.APIKeyHash, &agent.Status,
|
||||
&agent.CreatedAt, &agent.UpdatedAt,
|
||||
&agent.TriggerMode, &agent.CooldownSeconds, &agent.DailyTriggerBudget,
|
||||
&agent.MaxTriggerDepth, &k8sImage, &k8sEnvJSON,
|
||||
&agent.K8sResourcePreset, &pendingWork,
|
||||
)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
agent.Capabilities = json.RawMessage(caps)
|
||||
agent.K8sImage = k8sImage.String
|
||||
agent.K8sEnvJSON = k8sEnvJSON.String
|
||||
agent.PendingWork = pendingWork != 0
|
||||
agents = append(agents, &agent)
|
||||
}
|
||||
if agents == nil {
|
||||
|
||||
@@ -12,6 +12,13 @@ const (
|
||||
AgentStatusInactive = "inactive"
|
||||
)
|
||||
|
||||
// Trigger mode constants.
|
||||
const (
|
||||
TriggerModePassive = "passive"
|
||||
TriggerModeReactive = "reactive"
|
||||
TriggerModeDisabled = "disabled"
|
||||
)
|
||||
|
||||
// Agent represents a registered entity that can send/receive messages.
|
||||
type Agent struct {
|
||||
ID int64 `json:"id"`
|
||||
@@ -24,4 +31,14 @@ type Agent struct {
|
||||
Status string `json:"status"`
|
||||
CreatedAt time.Time `json:"created_at"`
|
||||
UpdatedAt time.Time `json:"updated_at"`
|
||||
|
||||
// Reactive trigger fields
|
||||
TriggerMode string `json:"trigger_mode"`
|
||||
CooldownSeconds int `json:"cooldown_seconds"`
|
||||
DailyTriggerBudget int `json:"daily_trigger_budget"`
|
||||
MaxTriggerDepth int `json:"max_trigger_depth"`
|
||||
K8sImage string `json:"k8s_image,omitempty"`
|
||||
K8sEnvJSON string `json:"k8s_env_json,omitempty"`
|
||||
K8sResourcePreset string `json:"k8s_resource_preset"`
|
||||
PendingWork bool `json:"pending_work"`
|
||||
}
|
||||
|
||||
@@ -12,6 +12,7 @@ import (
|
||||
"github.com/synapbus/synapbus/internal/channels"
|
||||
"github.com/synapbus/synapbus/internal/k8s"
|
||||
"github.com/synapbus/synapbus/internal/messaging"
|
||||
"github.com/synapbus/synapbus/internal/reactor"
|
||||
"github.com/synapbus/synapbus/internal/push"
|
||||
"github.com/synapbus/synapbus/internal/reactions"
|
||||
"github.com/synapbus/synapbus/internal/trace"
|
||||
@@ -37,6 +38,8 @@ type RouterConfig struct {
|
||||
ReactionService *reactions.Service
|
||||
PushService *push.Service
|
||||
TrustService *trust.Service
|
||||
ReactorStore *reactor.Store
|
||||
ReactorEngine *reactor.Reactor
|
||||
SSEHub *SSEHub
|
||||
Broadcaster *SSEBroadcaster
|
||||
SessionMiddleware func(http.Handler) http.Handler
|
||||
@@ -238,6 +241,19 @@ func NewRouterWithConfig(cfg RouterConfig) chi.Router {
|
||||
}
|
||||
}
|
||||
|
||||
// Reactive Runs
|
||||
if cfg.ReactorStore != nil && cfg.ReactorEngine != nil && cfg.AgentService != nil {
|
||||
runsHandler := NewRunsHandler(cfg.ReactorStore, cfg.ReactorEngine, agents.NewSQLiteAgentStore(cfg.DB))
|
||||
r.Group(func(r chi.Router) {
|
||||
r.Use(authMiddleware)
|
||||
|
||||
r.Get("/api/runs", runsHandler.ListRuns)
|
||||
r.Get("/api/runs/{id}", runsHandler.GetRun)
|
||||
r.Post("/api/runs/{id}/retry", runsHandler.RetryRun)
|
||||
r.Get("/api/agents/reactive", runsHandler.ReactiveAgents)
|
||||
})
|
||||
}
|
||||
|
||||
// Trust Scores
|
||||
if cfg.TrustService != nil {
|
||||
trustHandler := NewTrustHandler(cfg.TrustService)
|
||||
|
||||
@@ -0,0 +1,165 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"strconv"
|
||||
"time"
|
||||
|
||||
"github.com/go-chi/chi/v5"
|
||||
|
||||
"github.com/synapbus/synapbus/internal/agents"
|
||||
"github.com/synapbus/synapbus/internal/reactor"
|
||||
)
|
||||
|
||||
// RunsHandler handles REST API requests for reactive runs.
|
||||
type RunsHandler struct {
|
||||
store *reactor.Store
|
||||
reactor *reactor.Reactor
|
||||
agentStore agents.AgentStore
|
||||
}
|
||||
|
||||
// NewRunsHandler creates a new runs handler.
|
||||
func NewRunsHandler(store *reactor.Store, r *reactor.Reactor, agentStore agents.AgentStore) *RunsHandler {
|
||||
return &RunsHandler{
|
||||
store: store,
|
||||
reactor: r,
|
||||
agentStore: agentStore,
|
||||
}
|
||||
}
|
||||
|
||||
// ListRuns returns reactive runs with optional filters.
|
||||
func (h *RunsHandler) ListRuns(w http.ResponseWriter, r *http.Request) {
|
||||
agentName := r.URL.Query().Get("agent")
|
||||
status := r.URL.Query().Get("status")
|
||||
limit := 50
|
||||
offset := 0
|
||||
|
||||
if l := r.URL.Query().Get("limit"); l != "" {
|
||||
if v, err := strconv.Atoi(l); err == nil && v > 0 && v <= 200 {
|
||||
limit = v
|
||||
}
|
||||
}
|
||||
if o := r.URL.Query().Get("offset"); o != "" {
|
||||
if v, err := strconv.Atoi(o); err == nil && v >= 0 {
|
||||
offset = v
|
||||
}
|
||||
}
|
||||
|
||||
runs, total, err := h.store.ListRuns(r.Context(), agentName, status, limit, offset)
|
||||
if err != nil {
|
||||
writeJSON(w, http.StatusInternalServerError, errorBody("internal_error", err.Error()))
|
||||
return
|
||||
}
|
||||
|
||||
writeJSON(w, http.StatusOK, map[string]any{
|
||||
"runs": runs,
|
||||
"total": total,
|
||||
})
|
||||
}
|
||||
|
||||
// GetRun returns a single run by ID.
|
||||
func (h *RunsHandler) GetRun(w http.ResponseWriter, r *http.Request) {
|
||||
idStr := chi.URLParam(r, "id")
|
||||
id, err := strconv.ParseInt(idStr, 10, 64)
|
||||
if err != nil {
|
||||
writeJSON(w, http.StatusBadRequest, errorBody("bad_request", "invalid run ID"))
|
||||
return
|
||||
}
|
||||
|
||||
run, err := h.store.GetRunByID(r.Context(), id)
|
||||
if err != nil {
|
||||
writeJSON(w, http.StatusNotFound, errorBody("not_found", "run not found"))
|
||||
return
|
||||
}
|
||||
|
||||
writeJSON(w, http.StatusOK, run)
|
||||
}
|
||||
|
||||
// RetryRun retries a failed run.
|
||||
func (h *RunsHandler) RetryRun(w http.ResponseWriter, r *http.Request) {
|
||||
idStr := chi.URLParam(r, "id")
|
||||
id, err := strconv.ParseInt(idStr, 10, 64)
|
||||
if err != nil {
|
||||
writeJSON(w, http.StatusBadRequest, errorBody("bad_request", "invalid run ID"))
|
||||
return
|
||||
}
|
||||
|
||||
newRun, err := h.reactor.RetryRun(r.Context(), id)
|
||||
if err != nil {
|
||||
writeJSON(w, http.StatusBadRequest, errorBody("retry_failed", err.Error()))
|
||||
return
|
||||
}
|
||||
|
||||
writeJSON(w, http.StatusOK, map[string]any{
|
||||
"new_run_id": newRun.ID,
|
||||
"status": newRun.Status,
|
||||
})
|
||||
}
|
||||
|
||||
// ReactiveAgents returns agents with reactive trigger config and current status.
|
||||
func (h *RunsHandler) ReactiveAgents(w http.ResponseWriter, r *http.Request) {
|
||||
agentsList, err := h.agentStore.ListReactiveAgents(r.Context())
|
||||
if err != nil {
|
||||
writeJSON(w, http.StatusInternalServerError, errorBody("internal_error", err.Error()))
|
||||
return
|
||||
}
|
||||
|
||||
type agentStatus struct {
|
||||
Name string `json:"name"`
|
||||
TriggerMode string `json:"trigger_mode"`
|
||||
CooldownSeconds int `json:"cooldown_seconds"`
|
||||
DailyTriggerBudget int `json:"daily_trigger_budget"`
|
||||
MaxTriggerDepth int `json:"max_trigger_depth"`
|
||||
K8sImage string `json:"k8s_image"`
|
||||
PendingWork bool `json:"pending_work"`
|
||||
State string `json:"state"`
|
||||
TodayRuns int `json:"today_runs"`
|
||||
CooldownUntil *string `json:"cooldown_until"`
|
||||
}
|
||||
|
||||
result := make([]agentStatus, 0, len(agentsList))
|
||||
for _, a := range agentsList {
|
||||
as := agentStatus{
|
||||
Name: a.Name,
|
||||
TriggerMode: a.TriggerMode,
|
||||
CooldownSeconds: a.CooldownSeconds,
|
||||
DailyTriggerBudget: a.DailyTriggerBudget,
|
||||
MaxTriggerDepth: a.MaxTriggerDepth,
|
||||
K8sImage: a.K8sImage,
|
||||
PendingWork: a.PendingWork,
|
||||
}
|
||||
|
||||
// Compute state
|
||||
todayCount, _ := h.store.CountTodayRuns(r.Context(), a.Name)
|
||||
as.TodayRuns = todayCount
|
||||
|
||||
running, _ := h.store.IsAgentRunning(r.Context(), a.Name)
|
||||
if running {
|
||||
as.State = "running"
|
||||
} else if a.PendingWork {
|
||||
as.State = "queued"
|
||||
} else if todayCount >= a.DailyTriggerBudget {
|
||||
as.State = "budget_exhausted"
|
||||
} else {
|
||||
lastRun, _ := h.store.GetLastRunTime(r.Context(), a.Name)
|
||||
if lastRun != nil {
|
||||
cooldownEnd := lastRun.Add(time.Duration(a.CooldownSeconds) * time.Second)
|
||||
if time.Now().Before(cooldownEnd) {
|
||||
as.State = "cooldown"
|
||||
t := cooldownEnd.UTC().Format(time.RFC3339)
|
||||
as.CooldownUntil = &t
|
||||
} else {
|
||||
as.State = "idle"
|
||||
}
|
||||
} else {
|
||||
as.State = "idle"
|
||||
}
|
||||
}
|
||||
|
||||
result = append(result, as)
|
||||
}
|
||||
|
||||
writeJSON(w, http.StatusOK, map[string]any{
|
||||
"agents": result,
|
||||
})
|
||||
}
|
||||
+54
-3
@@ -84,6 +84,11 @@ func (r *K8sJobRunner) IsAvailable() bool {
|
||||
return true
|
||||
}
|
||||
|
||||
// GetClientset returns the kubernetes clientset for direct API access (used by reactor poller).
|
||||
func (r *K8sJobRunner) GetClientset() kubernetes.Interface {
|
||||
return r.clientset
|
||||
}
|
||||
|
||||
func (r *K8sJobRunner) GetNamespace() string {
|
||||
return r.namespace
|
||||
}
|
||||
@@ -145,14 +150,18 @@ func (r *K8sJobRunner) CreateJob(ctx context.Context, handler *K8sHandler, msg *
|
||||
RestartPolicy: corev1.RestartPolicyNever,
|
||||
Containers: []corev1.Container{
|
||||
{
|
||||
Name: "handler",
|
||||
Image: handler.Image,
|
||||
Env: envVars,
|
||||
Name: "handler",
|
||||
Image: handler.Image,
|
||||
ImagePullPolicy: corev1.PullIfNotPresent,
|
||||
Args: handler.Args,
|
||||
Env: envVars,
|
||||
VolumeMounts: buildVolumeMounts(handler.VolumeMounts),
|
||||
Resources: corev1.ResourceRequirements{
|
||||
Limits: resourceLimits,
|
||||
},
|
||||
},
|
||||
},
|
||||
Volumes: buildVolumes(handler.Volumes),
|
||||
},
|
||||
},
|
||||
},
|
||||
@@ -228,6 +237,48 @@ func sanitizeJobName(name string) string {
|
||||
return name
|
||||
}
|
||||
|
||||
// buildVolumeMounts converts our VolumeMount type to K8s VolumeMounts.
|
||||
func buildVolumeMounts(mounts []VolumeMount) []corev1.VolumeMount {
|
||||
if len(mounts) == 0 {
|
||||
return nil
|
||||
}
|
||||
var result []corev1.VolumeMount
|
||||
for _, m := range mounts {
|
||||
result = append(result, corev1.VolumeMount{
|
||||
Name: m.Name,
|
||||
MountPath: m.MountPath,
|
||||
ReadOnly: m.ReadOnly,
|
||||
})
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
// buildVolumes converts our Volume type to K8s Volumes.
|
||||
func buildVolumes(volumes []Volume) []corev1.Volume {
|
||||
if len(volumes) == 0 {
|
||||
return nil
|
||||
}
|
||||
var result []corev1.Volume
|
||||
for _, v := range volumes {
|
||||
vol := corev1.Volume{Name: v.Name}
|
||||
if v.HostPath != "" {
|
||||
hostPathType := corev1.HostPathDirectory
|
||||
vol.VolumeSource = corev1.VolumeSource{
|
||||
HostPath: &corev1.HostPathVolumeSource{
|
||||
Path: v.HostPath,
|
||||
Type: &hostPathType,
|
||||
},
|
||||
}
|
||||
} else if v.EmptyDir {
|
||||
vol.VolumeSource = corev1.VolumeSource{
|
||||
EmptyDir: &corev1.EmptyDirVolumeSource{},
|
||||
}
|
||||
}
|
||||
result = append(result, vol)
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
// truncateBody truncates the message body to maxLen bytes.
|
||||
func truncateBody(body string, maxLen int) string {
|
||||
if len(body) <= maxLen {
|
||||
|
||||
@@ -22,6 +22,25 @@ type K8sHandler struct {
|
||||
Status string `json:"status"`
|
||||
CreatedAt time.Time `json:"created_at"`
|
||||
UpdatedAt time.Time `json:"updated_at"`
|
||||
|
||||
// Extended fields for reactive triggers (not persisted in k8s_handlers table)
|
||||
Args []string `json:"-"`
|
||||
VolumeMounts []VolumeMount `json:"-"`
|
||||
Volumes []Volume `json:"-"`
|
||||
}
|
||||
|
||||
// VolumeMount defines a mount point in the container.
|
||||
type VolumeMount struct {
|
||||
Name string
|
||||
MountPath string
|
||||
ReadOnly bool
|
||||
}
|
||||
|
||||
// Volume defines a volume source for the pod.
|
||||
type Volume struct {
|
||||
Name string
|
||||
HostPath string // If set, uses hostPath volume
|
||||
EmptyDir bool // If true, uses emptyDir volume
|
||||
}
|
||||
|
||||
// K8sJobRun represents a single Kubernetes job execution.
|
||||
|
||||
@@ -12,6 +12,7 @@ import (
|
||||
"time"
|
||||
|
||||
"github.com/synapbus/synapbus/internal/agents"
|
||||
"github.com/synapbus/synapbus/internal/agentquery"
|
||||
"github.com/synapbus/synapbus/internal/attachments"
|
||||
"github.com/synapbus/synapbus/internal/channels"
|
||||
"github.com/synapbus/synapbus/internal/messaging"
|
||||
@@ -31,6 +32,7 @@ type ServiceBridge struct {
|
||||
searchService *search.Service
|
||||
reactionService *reactions.Service
|
||||
trustService *trust.Service
|
||||
queryExecutor *agentquery.Executor
|
||||
agentName string
|
||||
}
|
||||
|
||||
@@ -130,6 +132,10 @@ func (b *ServiceBridge) Call(ctx context.Context, actionName string, args map[st
|
||||
case "get_trust":
|
||||
return b.callGetTrust(ctx, args)
|
||||
|
||||
// --- SQL Query ---
|
||||
case "query":
|
||||
return b.callQuery(ctx, args)
|
||||
|
||||
// --- DM send (also accessible via bridge for execute tool) ---
|
||||
case "send_message":
|
||||
return b.callSendMessage(ctx, args)
|
||||
@@ -1215,6 +1221,29 @@ func (b *ServiceBridge) callGetTrust(ctx context.Context, args map[string]any) (
|
||||
}, nil
|
||||
}
|
||||
|
||||
// SetQueryExecutor sets the SQL query executor for the bridge.
|
||||
func (b *ServiceBridge) SetQueryExecutor(exec *agentquery.Executor) {
|
||||
b.queryExecutor = exec
|
||||
}
|
||||
|
||||
func (b *ServiceBridge) callQuery(ctx context.Context, args map[string]any) (any, error) {
|
||||
if b.queryExecutor == nil {
|
||||
return nil, fmt.Errorf("SQL query not available")
|
||||
}
|
||||
|
||||
sqlStr := getString(args, "sql", "")
|
||||
if sqlStr == "" {
|
||||
return nil, fmt.Errorf("sql parameter is required")
|
||||
}
|
||||
|
||||
result, err := b.queryExecutor.Execute(ctx, b.agentName, sqlStr)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// --- Helpers ---
|
||||
|
||||
// resolveChannelID resolves a channel ID from either channel_id or channel_name in args.
|
||||
|
||||
+22
-12
@@ -12,6 +12,7 @@ import (
|
||||
"github.com/mark3labs/mcp-go/server"
|
||||
|
||||
"github.com/synapbus/synapbus/internal/actions"
|
||||
"github.com/synapbus/synapbus/internal/agentquery"
|
||||
"github.com/synapbus/synapbus/internal/agents"
|
||||
"github.com/synapbus/synapbus/internal/attachments"
|
||||
"github.com/synapbus/synapbus/internal/channels"
|
||||
@@ -26,12 +27,13 @@ import (
|
||||
|
||||
// MCPServer wraps the mcp-go server with SynapBus services.
|
||||
type MCPServer struct {
|
||||
mcpServer *server.MCPServer
|
||||
httpServer *server.StreamableHTTPServer
|
||||
connMgr *ConnectionManager
|
||||
agentService *agents.AgentService
|
||||
logger *slog.Logger
|
||||
console *console.Printer
|
||||
mcpServer *server.MCPServer
|
||||
httpServer *server.StreamableHTTPServer
|
||||
connMgr *ConnectionManager
|
||||
agentService *agents.AgentService
|
||||
hybridRegistrar *HybridToolRegistrar
|
||||
logger *slog.Logger
|
||||
console *console.Printer
|
||||
}
|
||||
|
||||
// NewMCPServer creates and configures a new MCP server with 4 hybrid tools registered.
|
||||
@@ -187,18 +189,26 @@ func NewMCPServer(
|
||||
)
|
||||
|
||||
s := &MCPServer{
|
||||
mcpServer: mcpSrv,
|
||||
httpServer: httpServer,
|
||||
connMgr: connMgr,
|
||||
agentService: agentService,
|
||||
logger: logger,
|
||||
console: consolePrinter,
|
||||
mcpServer: mcpSrv,
|
||||
httpServer: httpServer,
|
||||
connMgr: connMgr,
|
||||
agentService: agentService,
|
||||
hybridRegistrar: hybridRegistrar,
|
||||
logger: logger,
|
||||
console: consolePrinter,
|
||||
}
|
||||
|
||||
logger.Info("MCP server initialized (4 hybrid tools, 4 prompts, streamable HTTP transport)")
|
||||
return s
|
||||
}
|
||||
|
||||
// SetQueryExecutor sets the SQL query executor for agent queries via the execute tool.
|
||||
func (s *MCPServer) SetQueryExecutor(exec *agentquery.Executor) {
|
||||
if s.hybridRegistrar != nil {
|
||||
s.hybridRegistrar.SetQueryExecutor(exec)
|
||||
}
|
||||
}
|
||||
|
||||
// Handler returns the HTTP handler for mounting on a router.
|
||||
func (s *MCPServer) Handler() http.Handler {
|
||||
return s.httpServer
|
||||
|
||||
@@ -17,6 +17,7 @@ import (
|
||||
"github.com/synapbus/synapbus/internal/attachments"
|
||||
"github.com/synapbus/synapbus/internal/channels"
|
||||
"github.com/synapbus/synapbus/internal/jsruntime"
|
||||
"github.com/synapbus/synapbus/internal/agentquery"
|
||||
"github.com/synapbus/synapbus/internal/messaging"
|
||||
"github.com/synapbus/synapbus/internal/reactions"
|
||||
"github.com/synapbus/synapbus/internal/search"
|
||||
@@ -37,9 +38,15 @@ type HybridToolRegistrar struct {
|
||||
actionRegistry *actions.Registry
|
||||
actionIndex *actions.Index
|
||||
db *sql.DB
|
||||
queryExecutor *agentquery.Executor
|
||||
logger *slog.Logger
|
||||
}
|
||||
|
||||
// SetQueryExecutor sets the SQL query executor for all agent bridges.
|
||||
func (h *HybridToolRegistrar) SetQueryExecutor(exec *agentquery.Executor) {
|
||||
h.queryExecutor = exec
|
||||
}
|
||||
|
||||
// NewHybridToolRegistrar creates a new hybrid tool registrar.
|
||||
func NewHybridToolRegistrar(
|
||||
msgService *messaging.MessagingService,
|
||||
@@ -495,6 +502,9 @@ func (h *HybridToolRegistrar) handleExecute(ctx context.Context, req mcplib.Call
|
||||
h.trustService,
|
||||
agentName,
|
||||
)
|
||||
if h.queryExecutor != nil {
|
||||
bridge.SetQueryExecutor(h.queryExecutor)
|
||||
}
|
||||
|
||||
result, err := h.jsPool.Execute(ctx, code, bridge, jsruntime.ExecuteOptions{
|
||||
Timeout: timeout,
|
||||
|
||||
@@ -42,4 +42,46 @@ var (
|
||||
Name: "active_connections",
|
||||
Help: "Number of active connections",
|
||||
})
|
||||
|
||||
// Reactive agent triggering metrics
|
||||
ReactiveTriggersTotal = promauto.NewCounterVec(
|
||||
prometheus.CounterOpts{
|
||||
Namespace: "synapbus",
|
||||
Subsystem: "reactor",
|
||||
Name: "triggers_total",
|
||||
Help: "Total reactive trigger evaluations by agent and outcome",
|
||||
},
|
||||
[]string{"agent", "status"},
|
||||
)
|
||||
|
||||
ReactiveRunDuration = promauto.NewHistogramVec(
|
||||
prometheus.HistogramOpts{
|
||||
Namespace: "synapbus",
|
||||
Subsystem: "reactor",
|
||||
Name: "run_duration_seconds",
|
||||
Help: "Duration of reactive agent runs in seconds",
|
||||
Buckets: []float64{10, 30, 60, 120, 300, 600, 1200, 1800, 3600},
|
||||
},
|
||||
[]string{"agent"},
|
||||
)
|
||||
|
||||
ReactiveAgentState = promauto.NewGaugeVec(
|
||||
prometheus.GaugeOpts{
|
||||
Namespace: "synapbus",
|
||||
Subsystem: "reactor",
|
||||
Name: "agent_running",
|
||||
Help: "Whether a reactive agent is currently running (1) or idle (0)",
|
||||
},
|
||||
[]string{"agent"},
|
||||
)
|
||||
|
||||
ReactiveBudgetUsed = promauto.NewGaugeVec(
|
||||
prometheus.GaugeOpts{
|
||||
Namespace: "synapbus",
|
||||
Subsystem: "reactor",
|
||||
Name: "budget_used_today",
|
||||
Help: "Number of reactive runs used today per agent",
|
||||
},
|
||||
[]string{"agent"},
|
||||
)
|
||||
)
|
||||
|
||||
@@ -28,6 +28,14 @@ You are **{{.AgentName}}**, an autonomous agent connected to SynapBus.
|
||||
|
||||
Use ` + "`call(\"search\", {\"query\": \"workflow\"})`" + ` to discover all available tools.
|
||||
|
||||
### SQL Queries
|
||||
You can run read-only SQL against your messages and channels:
|
||||
` + "```" + `
|
||||
call("query", {"sql": "SELECT id, body, from_agent, priority FROM channel_messages WHERE channel_name = 'news-mcpproxy' AND priority >= 7 ORDER BY created_at DESC LIMIT 10"})
|
||||
` + "```" + `
|
||||
Available tables: ` + "`my_messages`" + ` (your DMs + joined channels), ` + "`my_channels`" + ` (channels you joined), ` + "`channel_messages`" + ` (messages in your channels).
|
||||
Results capped at 100 rows. CTEs (WITH) supported. Only SELECT allowed.
|
||||
|
||||
### Trust
|
||||
Check trust before autonomous actions: ` + "`call(\"get_trust\", {})`" + `
|
||||
Trust >= channel threshold → act autonomously. Otherwise post as "proposed" and wait for approval.
|
||||
|
||||
@@ -0,0 +1,52 @@
|
||||
package reactor
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
|
||||
"github.com/synapbus/synapbus/internal/messaging"
|
||||
)
|
||||
|
||||
// DMFailureNotifier sends system DMs to agent owners on job failure.
|
||||
type DMFailureNotifier struct {
|
||||
msgService *messaging.MessagingService
|
||||
}
|
||||
|
||||
// NewDMFailureNotifier creates a new failure notifier.
|
||||
func NewDMFailureNotifier(msgService *messaging.MessagingService) *DMFailureNotifier {
|
||||
return &DMFailureNotifier{msgService: msgService}
|
||||
}
|
||||
|
||||
// NotifyFailure sends a system DM to the agent's owner with error details.
|
||||
func (n *DMFailureNotifier) NotifyFailure(ctx context.Context, ownerAgentName, agentName, triggerFrom, triggerEvent string, durationMs int64, errorSummary string) error {
|
||||
durationStr := "< 1s"
|
||||
if durationMs > 0 {
|
||||
secs := durationMs / 1000
|
||||
if secs >= 60 {
|
||||
durationStr = fmt.Sprintf("%dm%ds", secs/60, secs%60)
|
||||
} else {
|
||||
durationStr = fmt.Sprintf("%ds", secs)
|
||||
}
|
||||
}
|
||||
|
||||
body := fmt.Sprintf(
|
||||
"⚠️ **Reactive run failed** for **%s**\n\n"+
|
||||
"**Trigger**: %s from %s\n"+
|
||||
"**Duration**: %s\n"+
|
||||
"**Error**: %s\n\n"+
|
||||
"View details in Agent Runs page.",
|
||||
agentName, triggerEvent, triggerFrom, durationStr, truncateError(errorSummary, 500),
|
||||
)
|
||||
|
||||
_, err := n.msgService.SendMessage(ctx, "system", ownerAgentName, body, messaging.SendOptions{
|
||||
Priority: 7,
|
||||
})
|
||||
return err
|
||||
}
|
||||
|
||||
func truncateError(s string, maxLen int) string {
|
||||
if len(s) <= maxLen {
|
||||
return s
|
||||
}
|
||||
return s[:maxLen] + "..."
|
||||
}
|
||||
@@ -0,0 +1,224 @@
|
||||
package reactor
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/synapbus/synapbus/internal/agents"
|
||||
"github.com/synapbus/synapbus/internal/dispatcher"
|
||||
k8spkg "github.com/synapbus/synapbus/internal/k8s"
|
||||
"github.com/synapbus/synapbus/internal/metrics"
|
||||
|
||||
batchv1 "k8s.io/api/batch/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/client-go/kubernetes"
|
||||
)
|
||||
|
||||
// Poller watches active reactive runs and updates their status from K8s.
|
||||
type Poller struct {
|
||||
store *Store
|
||||
agentStore agents.AgentStore
|
||||
clientset kubernetes.Interface
|
||||
runner k8spkg.JobRunner
|
||||
reactor *Reactor
|
||||
interval time.Duration
|
||||
logger *slog.Logger
|
||||
stopCh chan struct{}
|
||||
}
|
||||
|
||||
// NewPoller creates a new job status poller.
|
||||
func NewPoller(store *Store, agentStore agents.AgentStore, runner k8spkg.JobRunner, reactor *Reactor, logger *slog.Logger) *Poller {
|
||||
// Extract clientset from runner if it's the real K8s runner
|
||||
var clientset kubernetes.Interface
|
||||
if kr, ok := runner.(*k8spkg.K8sJobRunner); ok {
|
||||
clientset = kr.GetClientset()
|
||||
}
|
||||
|
||||
return &Poller{
|
||||
store: store,
|
||||
agentStore: agentStore,
|
||||
clientset: clientset,
|
||||
runner: runner,
|
||||
reactor: reactor,
|
||||
interval: 15 * time.Second,
|
||||
logger: logger.With("component", "reactor-poller"),
|
||||
stopCh: make(chan struct{}),
|
||||
}
|
||||
}
|
||||
|
||||
// Start begins the polling loop in a background goroutine.
|
||||
func (p *Poller) Start() {
|
||||
if !p.runner.IsAvailable() || p.clientset == nil {
|
||||
p.logger.Info("K8s not available, reactor poller disabled")
|
||||
return
|
||||
}
|
||||
go p.pollLoop()
|
||||
p.logger.Info("reactor poller started", "interval", p.interval)
|
||||
}
|
||||
|
||||
// Stop signals the poller to stop.
|
||||
func (p *Poller) Stop() {
|
||||
close(p.stopCh)
|
||||
}
|
||||
|
||||
func (p *Poller) pollLoop() {
|
||||
ticker := time.NewTicker(p.interval)
|
||||
defer ticker.Stop()
|
||||
|
||||
for {
|
||||
select {
|
||||
case <-p.stopCh:
|
||||
return
|
||||
case <-ticker.C:
|
||||
p.pollActiveRuns()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (p *Poller) pollActiveRuns() {
|
||||
ctx := context.Background()
|
||||
|
||||
runs, err := p.store.GetActiveRuns(ctx)
|
||||
if err != nil {
|
||||
p.logger.Error("failed to get active runs", "error", err)
|
||||
return
|
||||
}
|
||||
|
||||
for _, run := range runs {
|
||||
if run.K8sJobName == "" || run.K8sNamespace == "" {
|
||||
continue
|
||||
}
|
||||
p.checkJob(ctx, run)
|
||||
}
|
||||
}
|
||||
|
||||
func (p *Poller) checkJob(ctx context.Context, run *ReactiveRun) {
|
||||
ns := run.K8sNamespace
|
||||
jobName := run.K8sJobName
|
||||
|
||||
job, err := p.clientset.BatchV1().Jobs(ns).Get(ctx, jobName, metav1.GetOptions{})
|
||||
if err != nil {
|
||||
p.logger.Warn("failed to get K8s Job status", "job", jobName, "namespace", ns, "error", err)
|
||||
return
|
||||
}
|
||||
|
||||
// Check job conditions
|
||||
for _, cond := range job.Status.Conditions {
|
||||
switch cond.Type {
|
||||
case batchv1.JobComplete:
|
||||
if cond.Status == "True" {
|
||||
p.handleJobComplete(ctx, run, true, "")
|
||||
return
|
||||
}
|
||||
case batchv1.JobFailed:
|
||||
if cond.Status == "True" {
|
||||
reason := cond.Reason
|
||||
if cond.Message != "" {
|
||||
reason = reason + ": " + cond.Message
|
||||
}
|
||||
p.handleJobComplete(ctx, run, false, reason)
|
||||
return
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Check if active deadline exceeded
|
||||
if job.Status.Failed > 0 {
|
||||
p.handleJobComplete(ctx, run, false, "job failed (pod failure)")
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
func (p *Poller) handleJobComplete(ctx context.Context, run *ReactiveRun, success bool, failureReason string) {
|
||||
now := time.Now().UTC()
|
||||
|
||||
// Update metrics
|
||||
metrics.ReactiveAgentState.WithLabelValues(run.AgentName).Set(0)
|
||||
if run.StartedAt != nil {
|
||||
duration := now.Sub(*run.StartedAt).Seconds()
|
||||
metrics.ReactiveRunDuration.WithLabelValues(run.AgentName).Observe(duration)
|
||||
}
|
||||
todayCount, _ := p.store.CountTodayRuns(ctx, run.AgentName)
|
||||
metrics.ReactiveBudgetUsed.WithLabelValues(run.AgentName).Set(float64(todayCount))
|
||||
|
||||
if success {
|
||||
metrics.ReactiveTriggersTotal.WithLabelValues(run.AgentName, StatusSucceeded).Inc()
|
||||
_ = p.store.CompleteRun(ctx, run.ID, StatusSucceeded, "", now)
|
||||
p.logger.Info("reactive run succeeded",
|
||||
"agent", run.AgentName,
|
||||
"job", run.K8sJobName,
|
||||
"run_id", run.ID,
|
||||
)
|
||||
} else {
|
||||
// Retrieve logs
|
||||
errorLog := failureReason
|
||||
logs, err := p.runner.GetJobLogs(ctx, run.K8sNamespace, run.K8sJobName)
|
||||
if err == nil && logs != "" {
|
||||
// Keep last 100 lines
|
||||
lines := strings.Split(logs, "\n")
|
||||
if len(lines) > 100 {
|
||||
lines = lines[len(lines)-100:]
|
||||
}
|
||||
errorLog = strings.Join(lines, "\n")
|
||||
}
|
||||
|
||||
metrics.ReactiveTriggersTotal.WithLabelValues(run.AgentName, StatusFailed).Inc()
|
||||
_ = p.store.CompleteRun(ctx, run.ID, StatusFailed, errorLog, now)
|
||||
|
||||
p.logger.Warn("reactive run failed",
|
||||
"agent", run.AgentName,
|
||||
"job", run.K8sJobName,
|
||||
"run_id", run.ID,
|
||||
"reason", failureReason,
|
||||
)
|
||||
|
||||
// Send failure notification
|
||||
var durationMs int64
|
||||
if run.StartedAt != nil {
|
||||
durationMs = now.Sub(*run.StartedAt).Milliseconds()
|
||||
}
|
||||
agent, err := p.agentStore.GetAgentByName(ctx, run.AgentName)
|
||||
if err == nil && agent != nil {
|
||||
event := dispatcher.MessageEvent{
|
||||
EventType: run.TriggerEvent,
|
||||
FromAgent: run.TriggerFrom,
|
||||
}
|
||||
p.reactor.notifyFailure(ctx, agent, event, durationMs, fmt.Sprintf("Job %s failed: %s", run.K8sJobName, failureReason))
|
||||
}
|
||||
}
|
||||
|
||||
// Check for pending_work — launch coalesced run if needed
|
||||
p.checkPendingWork(ctx, run.AgentName)
|
||||
}
|
||||
|
||||
func (p *Poller) checkPendingWork(ctx context.Context, agentName string) {
|
||||
agent, err := p.agentStore.GetAgentByName(ctx, agentName)
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
|
||||
if !agent.PendingWork {
|
||||
return
|
||||
}
|
||||
|
||||
// Clear pending_work first
|
||||
_ = p.agentStore.SetPendingWork(ctx, agentName, false)
|
||||
|
||||
p.logger.Info("pending_work found, launching coalesced run", "agent", agentName)
|
||||
|
||||
// Create a synthetic event (coalesced — agent will pick up all pending messages via claim_messages)
|
||||
event := dispatcher.MessageEvent{
|
||||
EventType: "message.received",
|
||||
FromAgent: "system",
|
||||
ToAgent: agentName,
|
||||
Body: "Coalesced trigger: process all pending messages.",
|
||||
MentionedAgents: nil,
|
||||
Depth: 0,
|
||||
}
|
||||
|
||||
// Evaluate the trigger (it will check cooldown/budget again)
|
||||
_ = p.reactor.evaluateTrigger(ctx, agentName, event)
|
||||
}
|
||||
@@ -0,0 +1,359 @@
|
||||
// Package reactor provides the reactive agent triggering engine.
|
||||
// When a DM or @mention targets an agent with trigger_mode='reactive',
|
||||
// the reactor evaluates rate limits and creates a K8s Job to run the agent.
|
||||
package reactor
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/synapbus/synapbus/internal/agents"
|
||||
"github.com/synapbus/synapbus/internal/dispatcher"
|
||||
k8spkg "github.com/synapbus/synapbus/internal/k8s"
|
||||
"github.com/synapbus/synapbus/internal/metrics"
|
||||
)
|
||||
|
||||
// Reactor is the reactive agent triggering engine.
|
||||
type Reactor struct {
|
||||
store *Store
|
||||
agentStore agents.AgentStore
|
||||
runner k8spkg.JobRunner
|
||||
notifier FailureNotifier
|
||||
logger *slog.Logger
|
||||
}
|
||||
|
||||
// FailureNotifier sends system DMs on job failure.
|
||||
type FailureNotifier interface {
|
||||
NotifyFailure(ctx context.Context, ownerAgentName, agentName, triggerFrom, triggerEvent string, durationMs int64, errorSummary string) error
|
||||
}
|
||||
|
||||
// New creates a new Reactor.
|
||||
func New(store *Store, agentStore agents.AgentStore, runner k8spkg.JobRunner, logger *slog.Logger) *Reactor {
|
||||
return &Reactor{
|
||||
store: store,
|
||||
agentStore: agentStore,
|
||||
runner: runner,
|
||||
logger: logger.With("component", "reactor"),
|
||||
}
|
||||
}
|
||||
|
||||
// SetFailureNotifier sets the notifier for sending failure DMs.
|
||||
func (r *Reactor) SetFailureNotifier(n FailureNotifier) {
|
||||
r.notifier = n
|
||||
}
|
||||
|
||||
// Dispatch implements dispatcher.EventDispatcher. Called by MultiDispatcher
|
||||
// when a message event occurs.
|
||||
func (r *Reactor) Dispatch(ctx context.Context, event dispatcher.MessageEvent) error {
|
||||
switch event.EventType {
|
||||
case "message.received":
|
||||
// DM to an agent
|
||||
return r.evaluateTrigger(ctx, event.ToAgent, event)
|
||||
case "message.mentioned":
|
||||
// @mentions in channel messages
|
||||
for _, mentioned := range event.MentionedAgents {
|
||||
// Self-mention filter: agent can't trigger itself
|
||||
if mentioned == event.FromAgent {
|
||||
continue
|
||||
}
|
||||
if err := r.evaluateTrigger(ctx, mentioned, event); err != nil {
|
||||
r.logger.ErrorContext(ctx, "reactor trigger eval failed",
|
||||
"agent", mentioned,
|
||||
"error", err,
|
||||
)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
default:
|
||||
return nil // Ignore other event types
|
||||
}
|
||||
}
|
||||
|
||||
// evaluateTrigger runs the decision chain for a single agent.
|
||||
func (r *Reactor) evaluateTrigger(ctx context.Context, agentName string, event dispatcher.MessageEvent) error {
|
||||
// 1. Get agent config
|
||||
agent, err := r.agentStore.GetAgentByName(ctx, agentName)
|
||||
if err != nil {
|
||||
return nil // Agent doesn't exist, skip silently
|
||||
}
|
||||
|
||||
// 2. Check trigger mode
|
||||
if agent.TriggerMode != agents.TriggerModeReactive {
|
||||
return nil // Not reactive, skip
|
||||
}
|
||||
|
||||
// 3. Check K8s image configured
|
||||
if agent.K8sImage == "" {
|
||||
r.logger.Warn("reactive agent has no k8s_image configured", "agent", agentName)
|
||||
r.recordSkippedRun(ctx, agentName, event, StatusFailed, "no k8s_image configured")
|
||||
return nil
|
||||
}
|
||||
|
||||
// 4. Check K8s runner available
|
||||
if !r.runner.IsAvailable() {
|
||||
r.logger.Warn("K8s runner not available for reactive trigger", "agent", agentName)
|
||||
r.recordSkippedRun(ctx, agentName, event, StatusFailed, "K8s runner not available")
|
||||
return nil
|
||||
}
|
||||
|
||||
// 5. Extract depth from event metadata
|
||||
depth := event.Depth
|
||||
|
||||
// 6. Check trigger depth
|
||||
if depth >= agent.MaxTriggerDepth {
|
||||
r.logger.Info("trigger depth exceeded", "agent", agentName, "depth", depth, "max", agent.MaxTriggerDepth)
|
||||
r.recordSkippedRun(ctx, agentName, event, StatusDepthExceeded, "")
|
||||
return nil
|
||||
}
|
||||
|
||||
// 7. Check daily budget
|
||||
todayCount, err := r.store.CountTodayRuns(ctx, agentName)
|
||||
if err != nil {
|
||||
return fmt.Errorf("count today runs: %w", err)
|
||||
}
|
||||
if todayCount >= agent.DailyTriggerBudget {
|
||||
r.logger.Info("daily trigger budget exhausted", "agent", agentName, "count", todayCount, "budget", agent.DailyTriggerBudget)
|
||||
r.recordSkippedRun(ctx, agentName, event, StatusBudgetExhausted, "")
|
||||
return nil
|
||||
}
|
||||
|
||||
// 8. Check cooldown
|
||||
lastRun, err := r.store.GetLastRunTime(ctx, agentName)
|
||||
if err != nil {
|
||||
return fmt.Errorf("get last run time: %w", err)
|
||||
}
|
||||
if lastRun != nil {
|
||||
elapsed := time.Since(*lastRun)
|
||||
if elapsed < time.Duration(agent.CooldownSeconds)*time.Second {
|
||||
r.logger.Info("agent on cooldown", "agent", agentName, "elapsed", elapsed, "cooldown", agent.CooldownSeconds)
|
||||
// Set pending_work so we retry after cooldown
|
||||
_ = r.agentStore.SetPendingWork(ctx, agentName, true)
|
||||
r.recordSkippedRun(ctx, agentName, event, StatusCooldownSkipped, "")
|
||||
return nil
|
||||
}
|
||||
}
|
||||
|
||||
// 9. Check if agent is currently running
|
||||
running, err := r.store.IsAgentRunning(ctx, agentName)
|
||||
if err != nil {
|
||||
return fmt.Errorf("check agent running: %w", err)
|
||||
}
|
||||
if running {
|
||||
r.logger.Info("agent already running, setting pending_work", "agent", agentName)
|
||||
_ = r.agentStore.SetPendingWork(ctx, agentName, true)
|
||||
r.recordSkippedRun(ctx, agentName, event, StatusQueued, "")
|
||||
return nil
|
||||
}
|
||||
|
||||
// 10. All checks pass — create K8s Job
|
||||
return r.createJob(ctx, agent, event, depth)
|
||||
}
|
||||
|
||||
// createJob creates a K8s Job for the reactive trigger.
|
||||
func (r *Reactor) createJob(ctx context.Context, agent *agents.Agent, event dispatcher.MessageEvent, depth int) error {
|
||||
// Build handler from agent config
|
||||
handler := r.buildHandler(agent)
|
||||
|
||||
body := event.Body
|
||||
if len(body) > 4096 {
|
||||
body = body[:4096] + " [truncated]"
|
||||
}
|
||||
|
||||
msg := &k8spkg.JobMessage{
|
||||
MessageID: event.MessageID,
|
||||
FromAgent: event.FromAgent,
|
||||
Body: body,
|
||||
Event: event.EventType,
|
||||
Channel: event.Channel,
|
||||
Timestamp: time.Now().UTC().Format(time.RFC3339),
|
||||
}
|
||||
|
||||
// Add trigger depth env var to handler
|
||||
handler.Env["SYNAPBUS_TRIGGER_DEPTH"] = fmt.Sprintf("%d", depth)
|
||||
|
||||
// Create K8s Job FIRST (before DB insert to avoid stuck runs on SQLITE_BUSY)
|
||||
jobName, err := r.runner.CreateJob(ctx, handler, msg)
|
||||
if err != nil {
|
||||
errMsg := fmt.Sprintf("K8s Job creation failed: %s", err.Error())
|
||||
r.recordSkippedRun(ctx, agent.Name, event, StatusFailed, errMsg)
|
||||
r.notifyFailure(ctx, agent, event, 0, errMsg)
|
||||
return fmt.Errorf("create K8s job: %w", err)
|
||||
}
|
||||
|
||||
ns := handler.Namespace
|
||||
if ns == "" {
|
||||
ns = r.runner.GetNamespace()
|
||||
}
|
||||
|
||||
// Insert run record with job name already set (single atomic write)
|
||||
now := time.Now().UTC()
|
||||
run := &ReactiveRun{
|
||||
AgentName: agent.Name,
|
||||
TriggerMessageID: &event.MessageID,
|
||||
TriggerEvent: event.EventType,
|
||||
TriggerDepth: depth,
|
||||
TriggerFrom: event.FromAgent,
|
||||
Status: StatusRunning,
|
||||
K8sJobName: jobName,
|
||||
K8sNamespace: ns,
|
||||
StartedAt: &now,
|
||||
}
|
||||
|
||||
runID, err := r.store.InsertRun(ctx, run)
|
||||
if err != nil {
|
||||
r.logger.Error("failed to record reactive run (job already created)",
|
||||
"agent", agent.Name, "job", jobName, "error", err)
|
||||
runID = 0
|
||||
}
|
||||
|
||||
// Clear pending_work since we're launching
|
||||
_ = r.agentStore.SetPendingWork(ctx, agent.Name, false)
|
||||
|
||||
metrics.ReactiveTriggersTotal.WithLabelValues(agent.Name, StatusRunning).Inc()
|
||||
metrics.ReactiveAgentState.WithLabelValues(agent.Name).Set(1)
|
||||
|
||||
r.logger.Info("reactive K8s Job created",
|
||||
"agent", agent.Name,
|
||||
"job", jobName,
|
||||
"trigger_from", event.FromAgent,
|
||||
"trigger_event", event.EventType,
|
||||
"depth", depth,
|
||||
"run_id", runID,
|
||||
)
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// buildHandler constructs a K8sHandler from agent config.
|
||||
func (r *Reactor) buildHandler(agent *agents.Agent) *k8spkg.K8sHandler {
|
||||
env := map[string]string{}
|
||||
|
||||
// Parse k8s_env_json
|
||||
if agent.K8sEnvJSON != "" {
|
||||
var envMap map[string]json.RawMessage
|
||||
if err := json.Unmarshal([]byte(agent.K8sEnvJSON), &envMap); err == nil {
|
||||
for k, v := range envMap {
|
||||
// Plain string values
|
||||
var str string
|
||||
if err := json.Unmarshal(v, &str); err == nil {
|
||||
env[k] = str
|
||||
continue
|
||||
}
|
||||
// Secret refs are handled at K8s level; for now pass as-is
|
||||
// (the K8s runner would need extension for secretKeyRef)
|
||||
env[k] = strings.Trim(string(v), "\"")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Resource presets — default matches CronJob config (agent SDK needs ~1-2Gi)
|
||||
memory := "2Gi"
|
||||
cpu := "500m"
|
||||
if agent.K8sResourcePreset == "small" {
|
||||
memory = "512Mi"
|
||||
cpu = "100m"
|
||||
}
|
||||
|
||||
timeout := 3600 // 1 hour (matches CronJob config)
|
||||
|
||||
handler := &k8spkg.K8sHandler{
|
||||
AgentName: agent.Name,
|
||||
Image: agent.K8sImage,
|
||||
Events: []string{"message.received", "message.mentioned"},
|
||||
Namespace: "", // Use runner's namespace
|
||||
ResourcesMemory: memory,
|
||||
ResourcesCPU: cpu,
|
||||
Env: env,
|
||||
TimeoutSeconds: timeout,
|
||||
Status: "active",
|
||||
Args: []string{"--max-turns", "50", "--model", "claude-sonnet-4-6"},
|
||||
VolumeMounts: []k8spkg.VolumeMount{
|
||||
{Name: "claude-config", MountPath: "/app/.claude", ReadOnly: false},
|
||||
{Name: "workspace", MountPath: "/app/workspace", ReadOnly: false},
|
||||
},
|
||||
Volumes: []k8spkg.Volume{
|
||||
{Name: "claude-config", HostPath: "/home/user/.claude"},
|
||||
{Name: "workspace", EmptyDir: true},
|
||||
},
|
||||
}
|
||||
|
||||
// Override args for social-commenter (uses opus, more turns)
|
||||
if agent.Name == "social-commenter" {
|
||||
handler.Args = []string{"--max-turns", "80", "--model", "claude-opus-4-6"}
|
||||
}
|
||||
|
||||
return handler
|
||||
}
|
||||
|
||||
// RetryRun retries a failed run.
|
||||
func (r *Reactor) RetryRun(ctx context.Context, runID int64) (*ReactiveRun, error) {
|
||||
run, err := r.store.GetRunByID(ctx, runID)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("get run: %w", err)
|
||||
}
|
||||
if run.Status != StatusFailed {
|
||||
return nil, fmt.Errorf("can only retry failed runs, current status: %s", run.Status)
|
||||
}
|
||||
|
||||
agent, err := r.agentStore.GetAgentByName(ctx, run.AgentName)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("get agent: %w", err)
|
||||
}
|
||||
|
||||
// Create a synthetic event for the retry
|
||||
event := dispatcher.MessageEvent{
|
||||
EventType: run.TriggerEvent,
|
||||
MessageID: 0,
|
||||
FromAgent: run.TriggerFrom,
|
||||
ToAgent: run.AgentName,
|
||||
Body: "",
|
||||
Depth: run.TriggerDepth,
|
||||
}
|
||||
if run.TriggerMessageID != nil {
|
||||
event.MessageID = *run.TriggerMessageID
|
||||
}
|
||||
|
||||
if err := r.createJob(ctx, agent, event, run.TriggerDepth); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
// Return the newly created run
|
||||
runs, _, err := r.store.ListRuns(ctx, run.AgentName, StatusRunning, 1, 0)
|
||||
if err != nil || len(runs) == 0 {
|
||||
return nil, fmt.Errorf("retry succeeded but couldn't find new run")
|
||||
}
|
||||
return runs[0], nil
|
||||
}
|
||||
|
||||
func (r *Reactor) recordSkippedRun(ctx context.Context, agentName string, event dispatcher.MessageEvent, status, errorLog string) {
|
||||
metrics.ReactiveTriggersTotal.WithLabelValues(agentName, status).Inc()
|
||||
run := &ReactiveRun{
|
||||
AgentName: agentName,
|
||||
TriggerEvent: event.EventType,
|
||||
TriggerDepth: event.Depth,
|
||||
TriggerFrom: event.FromAgent,
|
||||
Status: status,
|
||||
ErrorLog: errorLog,
|
||||
}
|
||||
if event.MessageID > 0 {
|
||||
run.TriggerMessageID = &event.MessageID
|
||||
}
|
||||
_, _ = r.store.InsertRun(ctx, run)
|
||||
}
|
||||
|
||||
func (r *Reactor) notifyFailure(ctx context.Context, agent *agents.Agent, event dispatcher.MessageEvent, durationMs int64, errorSummary string) {
|
||||
if r.notifier == nil {
|
||||
return
|
||||
}
|
||||
// Find the owner's human agent name
|
||||
ownerAgent, err := r.agentStore.GetHumanAgentByOwner(ctx, agent.OwnerID)
|
||||
if err != nil || ownerAgent == nil {
|
||||
r.logger.Warn("could not find owner agent for failure notification", "agent", agent.Name)
|
||||
return
|
||||
}
|
||||
_ = r.notifier.NotifyFailure(ctx, ownerAgent.Name, agent.Name, event.FromAgent, event.EventType, durationMs, errorSummary)
|
||||
}
|
||||
@@ -0,0 +1,430 @@
|
||||
package reactor
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"encoding/json"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"fmt"
|
||||
"log/slog"
|
||||
|
||||
"github.com/synapbus/synapbus/internal/agents"
|
||||
"github.com/synapbus/synapbus/internal/dispatcher"
|
||||
k8spkg "github.com/synapbus/synapbus/internal/k8s"
|
||||
|
||||
_ "modernc.org/sqlite"
|
||||
)
|
||||
|
||||
// setupTestDB creates an in-memory SQLite database with schema for testing.
|
||||
func setupTestDB(t *testing.T) *sql.DB {
|
||||
t.Helper()
|
||||
db, err := sql.Open("sqlite", ":memory:")
|
||||
if err != nil {
|
||||
t.Fatalf("open db: %v", err)
|
||||
}
|
||||
|
||||
// Create minimal schema
|
||||
schema := `
|
||||
CREATE TABLE agents (
|
||||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
name TEXT NOT NULL UNIQUE,
|
||||
display_name TEXT NOT NULL DEFAULT '',
|
||||
type TEXT NOT NULL DEFAULT 'ai',
|
||||
capabilities TEXT NOT NULL DEFAULT '{}',
|
||||
owner_id INTEGER NOT NULL DEFAULT 1,
|
||||
api_key_hash TEXT NOT NULL DEFAULT '',
|
||||
status TEXT NOT NULL DEFAULT 'active',
|
||||
created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
updated_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
trigger_mode TEXT NOT NULL DEFAULT 'passive',
|
||||
cooldown_seconds INTEGER NOT NULL DEFAULT 600,
|
||||
daily_trigger_budget INTEGER NOT NULL DEFAULT 8,
|
||||
max_trigger_depth INTEGER NOT NULL DEFAULT 5,
|
||||
k8s_image TEXT,
|
||||
k8s_env_json TEXT,
|
||||
k8s_resource_preset TEXT NOT NULL DEFAULT 'default',
|
||||
pending_work INTEGER NOT NULL DEFAULT 0
|
||||
);
|
||||
CREATE TABLE reactive_runs (
|
||||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
agent_name TEXT NOT NULL,
|
||||
trigger_message_id INTEGER,
|
||||
trigger_event TEXT NOT NULL,
|
||||
trigger_depth INTEGER NOT NULL DEFAULT 0,
|
||||
trigger_from TEXT,
|
||||
status TEXT NOT NULL DEFAULT 'queued',
|
||||
k8s_job_name TEXT,
|
||||
k8s_namespace TEXT,
|
||||
started_at DATETIME,
|
||||
completed_at DATETIME,
|
||||
duration_ms INTEGER,
|
||||
error_log TEXT,
|
||||
token_cost_json TEXT,
|
||||
created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP
|
||||
);
|
||||
`
|
||||
if _, err := db.Exec(schema); err != nil {
|
||||
t.Fatalf("create schema: %v", err)
|
||||
}
|
||||
|
||||
return db
|
||||
}
|
||||
|
||||
func insertTestAgent(t *testing.T, db *sql.DB, name, triggerMode, image string, cooldown, budget, maxDepth int) {
|
||||
t.Helper()
|
||||
_, err := db.Exec(
|
||||
`INSERT INTO agents (name, display_name, type, owner_id, trigger_mode, cooldown_seconds, daily_trigger_budget, max_trigger_depth, k8s_image, k8s_resource_preset)
|
||||
VALUES (?, ?, 'ai', 1, ?, ?, ?, ?, ?, 'default')`,
|
||||
name, name, triggerMode, cooldown, budget, maxDepth, image,
|
||||
)
|
||||
if err != nil {
|
||||
t.Fatalf("insert agent: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestReactorPassiveAgentSkipped(t *testing.T) {
|
||||
db := setupTestDB(t)
|
||||
defer db.Close()
|
||||
|
||||
insertTestAgent(t, db, "passive-agent", "passive", "image:latest", 600, 8, 5)
|
||||
|
||||
store := NewStore(db)
|
||||
agentStore := agents.NewSQLiteAgentStore(db)
|
||||
runner := k8spkg.NewNoopRunner()
|
||||
logger := slog.Default()
|
||||
|
||||
reactor := New(store, agentStore, runner, logger)
|
||||
|
||||
event := dispatcher.MessageEvent{
|
||||
EventType: "message.received",
|
||||
MessageID: 1,
|
||||
FromAgent: "algis",
|
||||
ToAgent: "passive-agent",
|
||||
Body: "hello",
|
||||
}
|
||||
|
||||
err := reactor.Dispatch(context.Background(), event)
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
|
||||
// No runs should be created for passive agents
|
||||
runs, total, err := store.ListRuns(context.Background(), "passive-agent", "", 10, 0)
|
||||
if err != nil {
|
||||
t.Fatalf("list runs: %v", err)
|
||||
}
|
||||
if total != 0 || len(runs) != 0 {
|
||||
t.Errorf("expected 0 runs for passive agent, got %d", total)
|
||||
}
|
||||
}
|
||||
|
||||
func TestReactorNoK8sImage(t *testing.T) {
|
||||
db := setupTestDB(t)
|
||||
defer db.Close()
|
||||
|
||||
insertTestAgent(t, db, "no-image-agent", "reactive", "", 600, 8, 5)
|
||||
|
||||
store := NewStore(db)
|
||||
agentStore := agents.NewSQLiteAgentStore(db)
|
||||
runner := k8spkg.NewNoopRunner()
|
||||
logger := slog.Default()
|
||||
|
||||
reactor := New(store, agentStore, runner, logger)
|
||||
|
||||
event := dispatcher.MessageEvent{
|
||||
EventType: "message.received",
|
||||
MessageID: 1,
|
||||
FromAgent: "algis",
|
||||
ToAgent: "no-image-agent",
|
||||
Body: "hello",
|
||||
}
|
||||
|
||||
_ = reactor.Dispatch(context.Background(), event)
|
||||
|
||||
runs, _, _ := store.ListRuns(context.Background(), "no-image-agent", StatusFailed, 10, 0)
|
||||
if len(runs) != 1 {
|
||||
t.Fatalf("expected 1 failed run for agent with no image, got %d", len(runs))
|
||||
}
|
||||
if runs[0].ErrorLog != "no k8s_image configured" {
|
||||
t.Errorf("expected 'no k8s_image configured' error, got: %s", runs[0].ErrorLog)
|
||||
}
|
||||
}
|
||||
|
||||
func TestReactorDepthExceeded(t *testing.T) {
|
||||
db := setupTestDB(t)
|
||||
defer db.Close()
|
||||
|
||||
insertTestAgent(t, db, "deep-agent", "reactive", "image:latest", 600, 8, 3)
|
||||
|
||||
store := NewStore(db)
|
||||
agentStore := agents.NewSQLiteAgentStore(db)
|
||||
runner := &fakeRunner{available: true}
|
||||
logger := slog.Default()
|
||||
|
||||
reactor := New(store, agentStore, runner, logger)
|
||||
|
||||
event := dispatcher.MessageEvent{
|
||||
EventType: "message.received",
|
||||
MessageID: 1,
|
||||
FromAgent: "other-agent",
|
||||
ToAgent: "deep-agent",
|
||||
Body: "hello from depth 3",
|
||||
Depth: 3, // equals max depth
|
||||
}
|
||||
|
||||
_ = reactor.Dispatch(context.Background(), event)
|
||||
|
||||
runs, _, _ := store.ListRuns(context.Background(), "deep-agent", StatusDepthExceeded, 10, 0)
|
||||
if len(runs) != 1 {
|
||||
t.Fatalf("expected 1 depth_exceeded run, got %d", len(runs))
|
||||
}
|
||||
}
|
||||
|
||||
func TestReactorBudgetExhausted(t *testing.T) {
|
||||
db := setupTestDB(t)
|
||||
defer db.Close()
|
||||
|
||||
insertTestAgent(t, db, "budget-agent", "reactive", "image:latest", 0, 2, 5)
|
||||
|
||||
store := NewStore(db)
|
||||
agentStore := agents.NewSQLiteAgentStore(db)
|
||||
runner := &fakeRunner{available: true}
|
||||
logger := slog.Default()
|
||||
|
||||
reactor := New(store, agentStore, runner, logger)
|
||||
|
||||
// Record 2 existing runs today
|
||||
for i := 0; i < 2; i++ {
|
||||
_, _ = store.InsertRun(context.Background(), &ReactiveRun{
|
||||
AgentName: "budget-agent",
|
||||
TriggerEvent: "message.received",
|
||||
Status: StatusSucceeded,
|
||||
})
|
||||
}
|
||||
|
||||
event := dispatcher.MessageEvent{
|
||||
EventType: "message.received",
|
||||
MessageID: 10,
|
||||
FromAgent: "algis",
|
||||
ToAgent: "budget-agent",
|
||||
Body: "one more",
|
||||
}
|
||||
|
||||
_ = reactor.Dispatch(context.Background(), event)
|
||||
|
||||
runs, _, _ := store.ListRuns(context.Background(), "budget-agent", StatusBudgetExhausted, 10, 0)
|
||||
if len(runs) != 1 {
|
||||
t.Fatalf("expected 1 budget_exhausted run, got %d", len(runs))
|
||||
}
|
||||
}
|
||||
|
||||
func TestReactorCooldownSkipped(t *testing.T) {
|
||||
db := setupTestDB(t)
|
||||
defer db.Close()
|
||||
|
||||
insertTestAgent(t, db, "cool-agent", "reactive", "image:latest", 600, 8, 5)
|
||||
|
||||
store := NewStore(db)
|
||||
agentStore := agents.NewSQLiteAgentStore(db)
|
||||
runner := &fakeRunner{available: true}
|
||||
logger := slog.Default()
|
||||
|
||||
reactor := New(store, agentStore, runner, logger)
|
||||
|
||||
// Record a recent run
|
||||
now := time.Now().UTC()
|
||||
_, _ = store.InsertRun(context.Background(), &ReactiveRun{
|
||||
AgentName: "cool-agent",
|
||||
TriggerEvent: "message.received",
|
||||
Status: StatusSucceeded,
|
||||
})
|
||||
// Hack: the above uses CURRENT_TIMESTAMP which is "now", so cooldown should be active
|
||||
|
||||
event := dispatcher.MessageEvent{
|
||||
EventType: "message.received",
|
||||
MessageID: 10,
|
||||
FromAgent: "algis",
|
||||
ToAgent: "cool-agent",
|
||||
Body: "too soon",
|
||||
}
|
||||
|
||||
_ = reactor.Dispatch(context.Background(), event)
|
||||
_ = now // avoid unused
|
||||
|
||||
runs, _, _ := store.ListRuns(context.Background(), "cool-agent", StatusCooldownSkipped, 10, 0)
|
||||
if len(runs) != 1 {
|
||||
t.Fatalf("expected 1 cooldown_skipped run, got %d", len(runs))
|
||||
}
|
||||
|
||||
// Check pending_work was set
|
||||
agent, _ := agentStore.GetAgentByName(context.Background(), "cool-agent")
|
||||
if !agent.PendingWork {
|
||||
t.Error("expected pending_work to be set after cooldown skip")
|
||||
}
|
||||
}
|
||||
|
||||
func TestReactorSequentialExecution(t *testing.T) {
|
||||
db := setupTestDB(t)
|
||||
defer db.Close()
|
||||
|
||||
insertTestAgent(t, db, "busy-agent", "reactive", "image:latest", 0, 8, 5)
|
||||
|
||||
store := NewStore(db)
|
||||
agentStore := agents.NewSQLiteAgentStore(db)
|
||||
runner := &fakeRunner{available: true}
|
||||
logger := slog.Default()
|
||||
|
||||
reactor := New(store, agentStore, runner, logger)
|
||||
|
||||
// First trigger — should succeed
|
||||
event1 := dispatcher.MessageEvent{
|
||||
EventType: "message.received",
|
||||
MessageID: 1,
|
||||
FromAgent: "algis",
|
||||
ToAgent: "busy-agent",
|
||||
Body: "first",
|
||||
}
|
||||
_ = reactor.Dispatch(context.Background(), event1)
|
||||
|
||||
// Second trigger — agent is running, should queue
|
||||
event2 := dispatcher.MessageEvent{
|
||||
EventType: "message.received",
|
||||
MessageID: 2,
|
||||
FromAgent: "algis",
|
||||
ToAgent: "busy-agent",
|
||||
Body: "second",
|
||||
}
|
||||
_ = reactor.Dispatch(context.Background(), event2)
|
||||
|
||||
// Check: one running, one queued
|
||||
running, _, _ := store.ListRuns(context.Background(), "busy-agent", StatusRunning, 10, 0)
|
||||
queued, _, _ := store.ListRuns(context.Background(), "busy-agent", StatusQueued, 10, 0)
|
||||
|
||||
if len(running) != 1 {
|
||||
t.Errorf("expected 1 running, got %d", len(running))
|
||||
}
|
||||
if len(queued) != 1 {
|
||||
t.Errorf("expected 1 queued, got %d", len(queued))
|
||||
}
|
||||
|
||||
// Check pending_work is set
|
||||
agent, _ := agentStore.GetAgentByName(context.Background(), "busy-agent")
|
||||
if !agent.PendingWork {
|
||||
t.Error("expected pending_work to be set")
|
||||
}
|
||||
}
|
||||
|
||||
func TestReactorSelfMentionIgnored(t *testing.T) {
|
||||
db := setupTestDB(t)
|
||||
defer db.Close()
|
||||
|
||||
insertTestAgent(t, db, "self-agent", "reactive", "image:latest", 0, 8, 5)
|
||||
|
||||
store := NewStore(db)
|
||||
agentStore := agents.NewSQLiteAgentStore(db)
|
||||
runner := &fakeRunner{available: true}
|
||||
logger := slog.Default()
|
||||
|
||||
reactor := New(store, agentStore, runner, logger)
|
||||
|
||||
// Agent mentions itself
|
||||
event := dispatcher.MessageEvent{
|
||||
EventType: "message.mentioned",
|
||||
MessageID: 1,
|
||||
FromAgent: "self-agent",
|
||||
Body: "hey @self-agent",
|
||||
MentionedAgents: []string{"self-agent"},
|
||||
}
|
||||
|
||||
_ = reactor.Dispatch(context.Background(), event)
|
||||
|
||||
runs, total, _ := store.ListRuns(context.Background(), "self-agent", "", 10, 0)
|
||||
if total != 0 || len(runs) != 0 {
|
||||
t.Errorf("expected 0 runs for self-mention, got %d", total)
|
||||
}
|
||||
}
|
||||
|
||||
func TestReactorSuccessfulTrigger(t *testing.T) {
|
||||
db := setupTestDB(t)
|
||||
defer db.Close()
|
||||
|
||||
envJSON, _ := json.Marshal(map[string]string{
|
||||
"AGENT_GIT_REPO": "Dumbris/test-agent",
|
||||
})
|
||||
_, _ = db.Exec(
|
||||
`INSERT INTO agents (name, display_name, type, owner_id, trigger_mode, cooldown_seconds, daily_trigger_budget, max_trigger_depth, k8s_image, k8s_env_json, k8s_resource_preset)
|
||||
VALUES (?, ?, 'ai', 1, 'reactive', 0, 8, 5, 'image:latest', ?, 'default')`,
|
||||
"test-agent", "Test Agent", string(envJSON),
|
||||
)
|
||||
|
||||
store := NewStore(db)
|
||||
agentStore := agents.NewSQLiteAgentStore(db)
|
||||
runner := &fakeRunner{available: true}
|
||||
logger := slog.Default()
|
||||
|
||||
reactor := New(store, agentStore, runner, logger)
|
||||
|
||||
event := dispatcher.MessageEvent{
|
||||
EventType: "message.received",
|
||||
MessageID: 42,
|
||||
FromAgent: "algis",
|
||||
ToAgent: "test-agent",
|
||||
Body: "research this topic",
|
||||
}
|
||||
|
||||
err := reactor.Dispatch(context.Background(), event)
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
|
||||
// Verify job was created
|
||||
if runner.lastJobName == "" {
|
||||
t.Fatal("expected K8s Job to be created")
|
||||
}
|
||||
|
||||
// Verify run record
|
||||
runs, _, _ := store.ListRuns(context.Background(), "test-agent", StatusRunning, 10, 0)
|
||||
if len(runs) != 1 {
|
||||
t.Fatalf("expected 1 running run, got %d", len(runs))
|
||||
}
|
||||
run := runs[0]
|
||||
if run.TriggerFrom != "algis" {
|
||||
t.Errorf("expected trigger_from=algis, got %s", run.TriggerFrom)
|
||||
}
|
||||
if run.TriggerEvent != "message.received" {
|
||||
t.Errorf("expected trigger_event=message.received, got %s", run.TriggerEvent)
|
||||
}
|
||||
|
||||
// Verify env vars passed to job
|
||||
if runner.lastEnv["SYNAPBUS_TRIGGER_DEPTH"] != "0" {
|
||||
t.Errorf("expected SYNAPBUS_TRIGGER_DEPTH=0, got %s", runner.lastEnv["SYNAPBUS_TRIGGER_DEPTH"])
|
||||
}
|
||||
if runner.lastEnv["AGENT_GIT_REPO"] != "Dumbris/test-agent" {
|
||||
t.Errorf("expected AGENT_GIT_REPO from k8s_env_json, got %s", runner.lastEnv["AGENT_GIT_REPO"])
|
||||
}
|
||||
}
|
||||
|
||||
// fakeRunner is a test double for k8spkg.JobRunner.
|
||||
type fakeRunner struct {
|
||||
available bool
|
||||
lastJobName string
|
||||
lastEnv map[string]string
|
||||
callCount int
|
||||
}
|
||||
|
||||
func (f *fakeRunner) IsAvailable() bool { return f.available }
|
||||
func (f *fakeRunner) GetNamespace() string { return "test-ns" }
|
||||
func (f *fakeRunner) GetJobLogs(_ context.Context, _, _ string) (string, error) {
|
||||
return "test logs", nil
|
||||
}
|
||||
func (f *fakeRunner) CreateJob(_ context.Context, handler *k8spkg.K8sHandler, msg *k8spkg.JobMessage) (string, error) {
|
||||
f.callCount++
|
||||
f.lastJobName = fmt.Sprintf("synapbus-%s-%d", handler.AgentName, msg.MessageID)
|
||||
f.lastEnv = make(map[string]string)
|
||||
for k, v := range handler.Env {
|
||||
f.lastEnv[k] = v
|
||||
}
|
||||
return f.lastJobName, nil
|
||||
}
|
||||
@@ -0,0 +1,298 @@
|
||||
package reactor
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"fmt"
|
||||
"time"
|
||||
)
|
||||
|
||||
// RunStatus constants for reactive_runs.
|
||||
const (
|
||||
StatusQueued = "queued"
|
||||
StatusRunning = "running"
|
||||
StatusSucceeded = "succeeded"
|
||||
StatusFailed = "failed"
|
||||
StatusCooldownSkipped = "cooldown_skipped"
|
||||
StatusBudgetExhausted = "budget_exhausted"
|
||||
StatusDepthExceeded = "depth_exceeded"
|
||||
)
|
||||
|
||||
// ReactiveRun represents a single trigger evaluation and its outcome.
|
||||
type ReactiveRun struct {
|
||||
ID int64 `json:"id"`
|
||||
AgentName string `json:"agent_name"`
|
||||
TriggerMessageID *int64 `json:"trigger_message_id,omitempty"`
|
||||
TriggerEvent string `json:"trigger_event"`
|
||||
TriggerDepth int `json:"trigger_depth"`
|
||||
TriggerFrom string `json:"trigger_from,omitempty"`
|
||||
Status string `json:"status"`
|
||||
K8sJobName string `json:"k8s_job_name,omitempty"`
|
||||
K8sNamespace string `json:"k8s_namespace,omitempty"`
|
||||
StartedAt *time.Time `json:"started_at,omitempty"`
|
||||
CompletedAt *time.Time `json:"completed_at,omitempty"`
|
||||
DurationMs *int64 `json:"duration_ms,omitempty"`
|
||||
ErrorLog string `json:"error_log,omitempty"`
|
||||
TokenCostJSON string `json:"token_cost_json,omitempty"`
|
||||
CreatedAt time.Time `json:"created_at"`
|
||||
}
|
||||
|
||||
// Store handles SQLite persistence for reactive runs.
|
||||
type Store struct {
|
||||
db *sql.DB
|
||||
}
|
||||
|
||||
// NewStore creates a new reactor store.
|
||||
func NewStore(db *sql.DB) *Store {
|
||||
return &Store{db: db}
|
||||
}
|
||||
|
||||
// InsertRun creates a new reactive_runs record.
|
||||
func (s *Store) InsertRun(ctx context.Context, run *ReactiveRun) (int64, error) {
|
||||
now := time.Now().UTC()
|
||||
run.CreatedAt = now
|
||||
nowStr := now.Format(time.RFC3339)
|
||||
var startedAtStr *string
|
||||
if run.StartedAt != nil {
|
||||
s := run.StartedAt.UTC().Format(time.RFC3339)
|
||||
startedAtStr = &s
|
||||
}
|
||||
result, err := s.db.ExecContext(ctx,
|
||||
`INSERT INTO reactive_runs (agent_name, trigger_message_id, trigger_event, trigger_depth, trigger_from, status, k8s_job_name, k8s_namespace, started_at, error_log, created_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`,
|
||||
run.AgentName, run.TriggerMessageID, run.TriggerEvent, run.TriggerDepth,
|
||||
run.TriggerFrom, run.Status, run.K8sJobName, run.K8sNamespace, startedAtStr, run.ErrorLog, nowStr,
|
||||
)
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("insert reactive run: %w", err)
|
||||
}
|
||||
id, err := result.LastInsertId()
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
run.ID = id
|
||||
return id, nil
|
||||
}
|
||||
|
||||
// UpdateRunStatus updates a run's status and optional fields.
|
||||
func (s *Store) UpdateRunStatus(ctx context.Context, id int64, status string, jobName, namespace string, startedAt *time.Time) error {
|
||||
var startedAtStr *string
|
||||
if startedAt != nil {
|
||||
str := startedAt.UTC().Format(time.RFC3339)
|
||||
startedAtStr = &str
|
||||
}
|
||||
_, err := s.db.ExecContext(ctx,
|
||||
`UPDATE reactive_runs SET status = ?, k8s_job_name = ?, k8s_namespace = ?, started_at = ? WHERE id = ?`,
|
||||
status, jobName, namespace, startedAtStr, id,
|
||||
)
|
||||
return err
|
||||
}
|
||||
|
||||
// CompleteRun marks a run as completed (succeeded or failed).
|
||||
func (s *Store) CompleteRun(ctx context.Context, id int64, status, errorLog string, completedAt time.Time) error {
|
||||
completedStr := completedAt.UTC().Format(time.RFC3339)
|
||||
_, err := s.db.ExecContext(ctx,
|
||||
`UPDATE reactive_runs SET status = ?, error_log = ?, completed_at = ?,
|
||||
duration_ms = CAST((julianday(?) - julianday(started_at)) * 86400000 AS INTEGER)
|
||||
WHERE id = ?`,
|
||||
status, errorLog, completedStr, completedStr, id,
|
||||
)
|
||||
return err
|
||||
}
|
||||
|
||||
// GetRunByID returns a single run.
|
||||
func (s *Store) GetRunByID(ctx context.Context, id int64) (*ReactiveRun, error) {
|
||||
return s.scanRun(s.db.QueryRowContext(ctx, runSelectSQL()+` WHERE id = ?`, id))
|
||||
}
|
||||
|
||||
// ListRuns returns recent runs with optional filters.
|
||||
func (s *Store) ListRuns(ctx context.Context, agentName, status string, limit, offset int) ([]*ReactiveRun, int, error) {
|
||||
where := "WHERE 1=1"
|
||||
args := []any{}
|
||||
|
||||
if agentName != "" {
|
||||
where += " AND agent_name = ?"
|
||||
args = append(args, agentName)
|
||||
}
|
||||
if status != "" {
|
||||
where += " AND status = ?"
|
||||
args = append(args, status)
|
||||
}
|
||||
|
||||
// Count total
|
||||
var total int
|
||||
countArgs := make([]any, len(args))
|
||||
copy(countArgs, args)
|
||||
err := s.db.QueryRowContext(ctx, "SELECT COUNT(*) FROM reactive_runs "+where, countArgs...).Scan(&total)
|
||||
if err != nil {
|
||||
return nil, 0, err
|
||||
}
|
||||
|
||||
// Query with pagination
|
||||
query := runSelectSQL() + " " + where + " ORDER BY created_at DESC LIMIT ? OFFSET ?"
|
||||
args = append(args, limit, offset)
|
||||
rows, err := s.db.QueryContext(ctx, query, args...)
|
||||
if err != nil {
|
||||
return nil, 0, err
|
||||
}
|
||||
defer rows.Close()
|
||||
|
||||
runs, err := s.scanRuns(rows)
|
||||
return runs, total, err
|
||||
}
|
||||
|
||||
// GetActiveRuns returns runs with status 'running' (for polling).
|
||||
func (s *Store) GetActiveRuns(ctx context.Context) ([]*ReactiveRun, error) {
|
||||
rows, err := s.db.QueryContext(ctx, runSelectSQL()+` WHERE status = 'running'`)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
return s.scanRuns(rows)
|
||||
}
|
||||
|
||||
// CountTodayRuns counts runs that count against the daily budget for an agent.
|
||||
func (s *Store) CountTodayRuns(ctx context.Context, agentName string) (int, error) {
|
||||
// Compute start of today in UTC as RFC3339
|
||||
now := time.Now().UTC()
|
||||
startOfDay := time.Date(now.Year(), now.Month(), now.Day(), 0, 0, 0, 0, time.UTC)
|
||||
startStr := startOfDay.Format(time.RFC3339)
|
||||
|
||||
var count int
|
||||
err := s.db.QueryRowContext(ctx,
|
||||
`SELECT COUNT(*) FROM reactive_runs
|
||||
WHERE agent_name = ? AND status IN ('running', 'succeeded', 'failed')
|
||||
AND created_at >= ?`,
|
||||
agentName, startStr,
|
||||
).Scan(&count)
|
||||
return count, err
|
||||
}
|
||||
|
||||
// GetLastRunTime returns the created_at of the most recent countable run.
|
||||
func (s *Store) GetLastRunTime(ctx context.Context, agentName string) (*time.Time, error) {
|
||||
var t sql.NullString
|
||||
err := s.db.QueryRowContext(ctx,
|
||||
`SELECT MAX(created_at) FROM reactive_runs
|
||||
WHERE agent_name = ? AND status IN ('running', 'succeeded', 'failed')`,
|
||||
agentName,
|
||||
).Scan(&t)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if !t.Valid || t.String == "" {
|
||||
return nil, nil
|
||||
}
|
||||
parsed, err := parseTime(t.String)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return &parsed, nil
|
||||
}
|
||||
|
||||
// parseTime tries multiple time formats used by SQLite / Go driver.
|
||||
func parseTime(s string) (time.Time, error) {
|
||||
formats := []string{
|
||||
time.RFC3339,
|
||||
time.RFC3339Nano,
|
||||
"2006-01-02T15:04:05Z",
|
||||
"2006-01-02 15:04:05+00:00",
|
||||
"2006-01-02 15:04:05",
|
||||
"2006-01-02T15:04:05.999999999Z07:00",
|
||||
}
|
||||
for _, f := range formats {
|
||||
if t, err := time.Parse(f, s); err == nil {
|
||||
return t, nil
|
||||
}
|
||||
}
|
||||
return time.Time{}, fmt.Errorf("cannot parse time %q", s)
|
||||
}
|
||||
|
||||
// IsAgentRunning checks if the agent has an active (running) reactive run.
|
||||
func (s *Store) IsAgentRunning(ctx context.Context, agentName string) (bool, error) {
|
||||
var count int
|
||||
err := s.db.QueryRowContext(ctx,
|
||||
`SELECT COUNT(*) FROM reactive_runs WHERE agent_name = ? AND status = 'running'`,
|
||||
agentName,
|
||||
).Scan(&count)
|
||||
return count > 0, err
|
||||
}
|
||||
|
||||
func runSelectSQL() string {
|
||||
return `SELECT id, agent_name, trigger_message_id, trigger_event, trigger_depth, trigger_from,
|
||||
status, k8s_job_name, k8s_namespace, started_at, completed_at, duration_ms, error_log, token_cost_json, created_at
|
||||
FROM reactive_runs`
|
||||
}
|
||||
|
||||
func scanRunFields(r *ReactiveRun, msgID *sql.NullInt64, triggerFrom, jobName, namespace, errorLog, tokenCost *sql.NullString, startedAt, completedAt *sql.NullString, durationMs *sql.NullInt64, createdAt *string) {
|
||||
if msgID.Valid {
|
||||
r.TriggerMessageID = &msgID.Int64
|
||||
}
|
||||
r.TriggerFrom = triggerFrom.String
|
||||
r.K8sJobName = jobName.String
|
||||
r.K8sNamespace = namespace.String
|
||||
if startedAt.Valid && startedAt.String != "" {
|
||||
if t, err := parseTime(startedAt.String); err == nil {
|
||||
r.StartedAt = &t
|
||||
}
|
||||
}
|
||||
if completedAt.Valid && completedAt.String != "" {
|
||||
if t, err := parseTime(completedAt.String); err == nil {
|
||||
r.CompletedAt = &t
|
||||
}
|
||||
}
|
||||
if durationMs.Valid {
|
||||
r.DurationMs = &durationMs.Int64
|
||||
}
|
||||
r.ErrorLog = errorLog.String
|
||||
r.TokenCostJSON = tokenCost.String
|
||||
if *createdAt != "" {
|
||||
if t, err := parseTime(*createdAt); err == nil {
|
||||
r.CreatedAt = t
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (s *Store) scanRun(row *sql.Row) (*ReactiveRun, error) {
|
||||
var r ReactiveRun
|
||||
var msgID sql.NullInt64
|
||||
var triggerFrom, jobName, namespace, errorLog, tokenCost sql.NullString
|
||||
var startedAt, completedAt sql.NullString
|
||||
var durationMs sql.NullInt64
|
||||
var createdAt string
|
||||
|
||||
err := row.Scan(
|
||||
&r.ID, &r.AgentName, &msgID, &r.TriggerEvent, &r.TriggerDepth, &triggerFrom,
|
||||
&r.Status, &jobName, &namespace, &startedAt, &completedAt, &durationMs, &errorLog, &tokenCost, &createdAt,
|
||||
)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
scanRunFields(&r, &msgID, &triggerFrom, &jobName, &namespace, &errorLog, &tokenCost, &startedAt, &completedAt, &durationMs, &createdAt)
|
||||
return &r, nil
|
||||
}
|
||||
|
||||
func (s *Store) scanRuns(rows *sql.Rows) ([]*ReactiveRun, error) {
|
||||
var runs []*ReactiveRun
|
||||
for rows.Next() {
|
||||
var r ReactiveRun
|
||||
var msgID sql.NullInt64
|
||||
var triggerFrom, jobName, namespace, errorLog, tokenCost sql.NullString
|
||||
var startedAt, completedAt sql.NullString
|
||||
var durationMs sql.NullInt64
|
||||
var createdAt string
|
||||
|
||||
err := rows.Scan(
|
||||
&r.ID, &r.AgentName, &msgID, &r.TriggerEvent, &r.TriggerDepth, &triggerFrom,
|
||||
&r.Status, &jobName, &namespace, &startedAt, &completedAt, &durationMs, &errorLog, &tokenCost, &createdAt,
|
||||
)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
scanRunFields(&r, &msgID, &triggerFrom, &jobName, &namespace, &errorLog, &tokenCost, &startedAt, &completedAt, &durationMs, &createdAt)
|
||||
runs = append(runs, &r)
|
||||
}
|
||||
if runs == nil {
|
||||
runs = []*ReactiveRun{}
|
||||
}
|
||||
return runs, rows.Err()
|
||||
}
|
||||
@@ -0,0 +1,35 @@
|
||||
-- 013: Reactive agent triggering
|
||||
-- Extends agents with trigger configuration, adds reactive_runs tracking table.
|
||||
|
||||
-- Extend agents table with reactive trigger configuration
|
||||
ALTER TABLE agents ADD COLUMN trigger_mode TEXT NOT NULL DEFAULT 'passive';
|
||||
ALTER TABLE agents ADD COLUMN cooldown_seconds INTEGER NOT NULL DEFAULT 600;
|
||||
ALTER TABLE agents ADD COLUMN daily_trigger_budget INTEGER NOT NULL DEFAULT 8;
|
||||
ALTER TABLE agents ADD COLUMN max_trigger_depth INTEGER NOT NULL DEFAULT 5;
|
||||
ALTER TABLE agents ADD COLUMN k8s_image TEXT;
|
||||
ALTER TABLE agents ADD COLUMN k8s_env_json TEXT;
|
||||
ALTER TABLE agents ADD COLUMN k8s_resource_preset TEXT NOT NULL DEFAULT 'default';
|
||||
ALTER TABLE agents ADD COLUMN pending_work INTEGER NOT NULL DEFAULT 0;
|
||||
|
||||
-- Reactive trigger runs: tracks every trigger evaluation and K8s job lifecycle
|
||||
CREATE TABLE reactive_runs (
|
||||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
agent_name TEXT NOT NULL REFERENCES agents(name),
|
||||
trigger_message_id INTEGER,
|
||||
trigger_event TEXT NOT NULL,
|
||||
trigger_depth INTEGER NOT NULL DEFAULT 0,
|
||||
trigger_from TEXT,
|
||||
status TEXT NOT NULL DEFAULT 'queued',
|
||||
k8s_job_name TEXT,
|
||||
k8s_namespace TEXT,
|
||||
started_at DATETIME,
|
||||
completed_at DATETIME,
|
||||
duration_ms INTEGER,
|
||||
error_log TEXT,
|
||||
token_cost_json TEXT,
|
||||
created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP
|
||||
);
|
||||
|
||||
CREATE INDEX idx_reactive_runs_agent_created ON reactive_runs(agent_name, created_at);
|
||||
CREATE INDEX idx_reactive_runs_status ON reactive_runs(status);
|
||||
CREATE INDEX idx_reactive_runs_agent_status ON reactive_runs(agent_name, status);
|
||||
@@ -0,0 +1,58 @@
|
||||
-- 016: Agent SQL query views
|
||||
-- These views are used by the 'query' action to give agents read access
|
||||
-- to messages they can see. The views expose a stable schema that agents
|
||||
-- can query via SQL. Access control is enforced at the Go layer by
|
||||
-- rewriting queries to filter by agent name.
|
||||
|
||||
-- Note: SQLite views cannot be parameterized. The Go query executor
|
||||
-- wraps agent queries in a CTE that filters by the authenticated agent's
|
||||
-- access (own DMs + joined channels). These views provide the base schema.
|
||||
|
||||
-- my_messages: All messages accessible to the calling agent
|
||||
CREATE VIEW IF NOT EXISTS v_agent_messages AS
|
||||
SELECT
|
||||
m.id,
|
||||
m.body,
|
||||
m.from_agent,
|
||||
m.to_agent,
|
||||
m.priority,
|
||||
m.status,
|
||||
m.metadata,
|
||||
m.created_at,
|
||||
m.updated_at,
|
||||
c.name AS channel_name,
|
||||
m.channel_id,
|
||||
m.reply_to,
|
||||
m.conversation_id
|
||||
FROM messages m
|
||||
LEFT JOIN channels c ON c.id = m.channel_id;
|
||||
|
||||
-- my_channels: Channels the calling agent has joined
|
||||
CREATE VIEW IF NOT EXISTS v_agent_channels AS
|
||||
SELECT
|
||||
c.id,
|
||||
c.name,
|
||||
c.description,
|
||||
c.type,
|
||||
c.topic,
|
||||
c.is_private,
|
||||
c.created_at,
|
||||
cm.joined_at AS member_since
|
||||
FROM channels c
|
||||
JOIN channel_members cm ON cm.channel_id = c.id;
|
||||
|
||||
-- channel_messages: Messages in channels (filtered by membership at Go layer)
|
||||
CREATE VIEW IF NOT EXISTS v_channel_messages AS
|
||||
SELECT
|
||||
m.id,
|
||||
m.body,
|
||||
m.from_agent,
|
||||
m.priority,
|
||||
m.status,
|
||||
m.metadata,
|
||||
m.created_at,
|
||||
c.name AS channel_name,
|
||||
m.channel_id,
|
||||
m.reply_to
|
||||
FROM messages m
|
||||
JOIN channels c ON c.id = m.channel_id;
|
||||
+97
-23
@@ -12,17 +12,22 @@ import (
|
||||
_ "modernc.org/sqlite"
|
||||
)
|
||||
|
||||
// DB wraps a *sql.DB with SynapBus-specific configuration.
|
||||
// DB wraps a write-only *sql.DB and an optional read-only *sql.DB
|
||||
// for split connection pool architecture. The write pool has MaxOpenConns=1
|
||||
// to serialize writes and eliminate SQLITE_BUSY errors. The read pool has
|
||||
// MaxOpenConns=8 and query_only=ON for safe concurrent reads.
|
||||
type DB struct {
|
||||
*sql.DB
|
||||
*sql.DB // Write pool (MaxOpenConns=1)
|
||||
ReadDB *sql.DB // Read pool (MaxOpenConns=8, query_only=ON) — nil for :memory: DBs
|
||||
}
|
||||
|
||||
// New opens a SQLite database with WAL mode, busy_timeout, and foreign keys enabled.
|
||||
// If dataDir is empty or ":memory:", an in-memory database is used.
|
||||
// New opens a SQLite database with WAL mode, split read/write pools, and foreign keys.
|
||||
// If dataDir is empty or ":memory:", an in-memory database is used (single pool, no split).
|
||||
func New(ctx context.Context, dataDir string) (*DB, error) {
|
||||
var dsn string
|
||||
isMemory := dataDir == "" || dataDir == ":memory:"
|
||||
|
||||
if dataDir == "" || dataDir == ":memory:" {
|
||||
if isMemory {
|
||||
dsn = ":memory:"
|
||||
} else {
|
||||
if err := os.MkdirAll(dataDir, 0o755); err != nil {
|
||||
@@ -31,16 +36,76 @@ func New(ctx context.Context, dataDir string) (*DB, error) {
|
||||
dsn = filepath.Join(dataDir, "synapbus.db")
|
||||
}
|
||||
|
||||
db, err := sql.Open("sqlite", dsn)
|
||||
// Open WRITE pool (single connection, serializes all writes)
|
||||
writeDB, err := openPool(ctx, dsn, poolConfig{
|
||||
maxOpen: 1,
|
||||
maxIdle: 1,
|
||||
queryOnly: false,
|
||||
label: "write",
|
||||
})
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("open database: %w", err)
|
||||
return nil, fmt.Errorf("open write pool: %w", err)
|
||||
}
|
||||
|
||||
// Configure SQLite pragmas
|
||||
result := &DB{DB: writeDB}
|
||||
|
||||
// For file-based databases, open a separate READ pool
|
||||
if !isMemory {
|
||||
readDB, err := openPool(ctx, dsn, poolConfig{
|
||||
maxOpen: 8,
|
||||
maxIdle: 4,
|
||||
queryOnly: true,
|
||||
label: "read",
|
||||
})
|
||||
if err != nil {
|
||||
writeDB.Close()
|
||||
return nil, fmt.Errorf("open read pool: %w", err)
|
||||
}
|
||||
result.ReadDB = readDB
|
||||
}
|
||||
|
||||
// Verify settings on write pool
|
||||
var journalMode string
|
||||
if err := writeDB.QueryRowContext(ctx, "PRAGMA journal_mode").Scan(&journalMode); err != nil {
|
||||
result.Close()
|
||||
return nil, fmt.Errorf("verify journal_mode: %w", err)
|
||||
}
|
||||
|
||||
slog.Info("database opened",
|
||||
"dsn", dsn,
|
||||
"journal_mode", journalMode,
|
||||
"write_pool", "MaxOpenConns=1",
|
||||
"read_pool_enabled", result.ReadDB != nil,
|
||||
)
|
||||
|
||||
return result, nil
|
||||
}
|
||||
|
||||
type poolConfig struct {
|
||||
maxOpen int
|
||||
maxIdle int
|
||||
queryOnly bool
|
||||
label string
|
||||
}
|
||||
|
||||
func openPool(ctx context.Context, dsn string, cfg poolConfig) (*sql.DB, error) {
|
||||
db, err := sql.Open("sqlite", dsn)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("open %s pool: %w", cfg.label, err)
|
||||
}
|
||||
|
||||
db.SetMaxOpenConns(cfg.maxOpen)
|
||||
db.SetMaxIdleConns(cfg.maxIdle)
|
||||
|
||||
pragmas := []string{
|
||||
"PRAGMA journal_mode=WAL",
|
||||
"PRAGMA busy_timeout=5000",
|
||||
"PRAGMA busy_timeout=15000",
|
||||
"PRAGMA foreign_keys=ON",
|
||||
"PRAGMA synchronous=NORMAL",
|
||||
"PRAGMA wal_autocheckpoint=1000",
|
||||
}
|
||||
if cfg.queryOnly {
|
||||
pragmas = append(pragmas, "PRAGMA query_only=ON")
|
||||
}
|
||||
|
||||
for _, pragma := range pragmas {
|
||||
@@ -50,22 +115,31 @@ func New(ctx context.Context, dataDir string) (*DB, error) {
|
||||
}
|
||||
}
|
||||
|
||||
// Verify settings
|
||||
var journalMode string
|
||||
if err := db.QueryRowContext(ctx, "PRAGMA journal_mode").Scan(&journalMode); err != nil {
|
||||
db.Close()
|
||||
return nil, fmt.Errorf("verify journal_mode: %w", err)
|
||||
return db, nil
|
||||
}
|
||||
|
||||
// QueryDB returns the read pool if available, otherwise falls back to the write pool.
|
||||
// Use this for all SELECT queries to avoid blocking writers.
|
||||
func (db *DB) QueryDB() *sql.DB {
|
||||
if db.ReadDB != nil {
|
||||
return db.ReadDB
|
||||
}
|
||||
|
||||
slog.Info("database opened",
|
||||
"dsn", dsn,
|
||||
"journal_mode", journalMode,
|
||||
)
|
||||
|
||||
return &DB{DB: db}, nil
|
||||
return db.DB
|
||||
}
|
||||
|
||||
// Close closes the database connection.
|
||||
// Close closes both the write and read database connections.
|
||||
func (db *DB) Close() error {
|
||||
return db.DB.Close()
|
||||
var errs []error
|
||||
if db.ReadDB != nil {
|
||||
if err := db.ReadDB.Close(); err != nil {
|
||||
errs = append(errs, fmt.Errorf("close read pool: %w", err))
|
||||
}
|
||||
}
|
||||
if err := db.DB.Close(); err != nil {
|
||||
errs = append(errs, fmt.Errorf("close write pool: %w", err))
|
||||
}
|
||||
if len(errs) > 0 {
|
||||
return errs[0]
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -68,11 +68,11 @@ func TestNew(t *testing.T) {
|
||||
if err != nil {
|
||||
t.Fatalf("failed to query busy_timeout: %v", err)
|
||||
}
|
||||
if timeout != 5000 {
|
||||
t.Errorf("busy_timeout = %d, want 5000", timeout)
|
||||
if timeout != 15000 {
|
||||
t.Errorf("busy_timeout = %d, want 15000", timeout)
|
||||
}
|
||||
|
||||
// Verify database is usable
|
||||
// Verify database is usable via write pool
|
||||
_, err = db.Exec("CREATE TABLE test (id INTEGER PRIMARY KEY)")
|
||||
if err != nil {
|
||||
t.Fatalf("failed to create test table: %v", err)
|
||||
@@ -80,3 +80,77 @@ func TestNew(t *testing.T) {
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestSplitPools(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
dir := t.TempDir()
|
||||
|
||||
db, err := New(ctx, dir)
|
||||
if err != nil {
|
||||
t.Fatalf("New() error: %v", err)
|
||||
}
|
||||
defer db.Close()
|
||||
|
||||
// Run migrations to create tables
|
||||
if err := RunMigrations(ctx, db.DB); err != nil {
|
||||
t.Fatalf("migrations: %v", err)
|
||||
}
|
||||
|
||||
// Verify read pool exists for file-based DB
|
||||
if db.ReadDB == nil {
|
||||
t.Fatal("expected ReadDB to be non-nil for file-based database")
|
||||
}
|
||||
|
||||
// Verify QueryDB returns read pool
|
||||
if db.QueryDB() != db.ReadDB {
|
||||
t.Error("QueryDB() should return ReadDB when available")
|
||||
}
|
||||
|
||||
// Create user first (FK requirement)
|
||||
_, err = db.Exec("INSERT INTO users (id, username, password_hash, display_name) VALUES (1, 'testuser', 'hash', 'Test')")
|
||||
if err != nil {
|
||||
t.Fatalf("create user: %v", err)
|
||||
}
|
||||
|
||||
// Verify write pool can write
|
||||
_, err = db.Exec("INSERT INTO agents (name, display_name, type, capabilities, owner_id, api_key_hash, status) VALUES ('test-agent', 'Test', 'ai', '{}', 1, 'hash', 'active')")
|
||||
if err != nil {
|
||||
t.Fatalf("write pool should allow writes: %v", err)
|
||||
}
|
||||
|
||||
// Verify read pool can read
|
||||
var name string
|
||||
err = db.ReadDB.QueryRow("SELECT name FROM agents WHERE name = 'test-agent'").Scan(&name)
|
||||
if err != nil {
|
||||
t.Fatalf("read pool should allow reads: %v", err)
|
||||
}
|
||||
if name != "test-agent" {
|
||||
t.Errorf("expected 'test-agent', got %q", name)
|
||||
}
|
||||
|
||||
// Verify read pool rejects writes
|
||||
_, err = db.ReadDB.Exec("INSERT INTO agents (name, display_name, type, capabilities, owner_id, api_key_hash, status) VALUES ('bad', 'Bad', 'ai', '{}', 1, 'hash', 'active')")
|
||||
if err == nil {
|
||||
t.Fatal("read pool should reject writes (query_only=ON)")
|
||||
}
|
||||
}
|
||||
|
||||
func TestInMemoryNoSplitPool(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
|
||||
db, err := New(ctx, ":memory:")
|
||||
if err != nil {
|
||||
t.Fatalf("New() error: %v", err)
|
||||
}
|
||||
defer db.Close()
|
||||
|
||||
// In-memory DB should NOT have a separate read pool
|
||||
if db.ReadDB != nil {
|
||||
t.Error("in-memory DB should not have a separate ReadDB")
|
||||
}
|
||||
|
||||
// QueryDB should fall back to write pool
|
||||
if db.QueryDB() != db.DB {
|
||||
t.Error("QueryDB() should return write pool for in-memory DB")
|
||||
}
|
||||
}
|
||||
|
||||
@@ -6,6 +6,9 @@ import (
|
||||
"sort"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
|
||||
"github.com/prometheus/client_golang/prometheus"
|
||||
"github.com/prometheus/common/expfmt"
|
||||
)
|
||||
|
||||
// Metrics provides Prometheus-compatible metrics for SynapBus.
|
||||
@@ -93,6 +96,17 @@ func (m *Metrics) WritePrometheus(w io.Writer) {
|
||||
fmt.Fprintf(w, "# HELP synapbus_active_agents Number of currently active agents.\n")
|
||||
fmt.Fprintf(w, "# TYPE synapbus_active_agents gauge\n")
|
||||
fmt.Fprintf(w, "synapbus_active_agents %d\n", m.activeAgents.Load())
|
||||
fmt.Fprintf(w, "\n")
|
||||
|
||||
// Append metrics from the standard Prometheus registry (reactor metrics, etc.)
|
||||
mfs, _ := prometheus.DefaultGatherer.Gather()
|
||||
enc := expfmt.NewEncoder(w, expfmt.NewFormat(expfmt.TypeTextPlain))
|
||||
for _, mf := range mfs {
|
||||
// Only include our custom metrics, skip Go runtime metrics
|
||||
if name := mf.GetName(); len(name) > 8 && name[:8] == "synapbus" {
|
||||
_ = enc.Encode(mf)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// NullMetrics is a no-op metrics implementation for when metrics are disabled.
|
||||
|
||||
Vendored
+11
-11
@@ -11,30 +11,30 @@
|
||||
<link rel="preconnect" href="https://fonts.googleapis.com">
|
||||
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
|
||||
<link href="https://fonts.googleapis.com/css2?family=DM+Sans:wght@400;500;600;700&family=Instrument+Sans:wght@400;500;600;700&family=JetBrains+Mono:wght@400;500&display=swap" rel="stylesheet">
|
||||
<link href="/_app/immutable/entry/start.DB7lMQq7.js" rel="modulepreload">
|
||||
<link href="/_app/immutable/chunks/CWvhbc19.js" rel="modulepreload">
|
||||
<link href="/_app/immutable/entry/start.DZCPQR8C.js" rel="modulepreload">
|
||||
<link href="/_app/immutable/chunks/C-zifBrA.js" rel="modulepreload">
|
||||
<link href="/_app/immutable/chunks/BjgrqnN-.js" rel="modulepreload">
|
||||
<link href="/_app/immutable/chunks/DUR6aSWt.js" rel="modulepreload">
|
||||
<link href="/_app/immutable/chunks/CgiIxzzM.js" rel="modulepreload">
|
||||
<link href="/_app/immutable/chunks/CNG0tgnP.js" rel="modulepreload">
|
||||
<link href="/_app/immutable/chunks/anqNGRhz.js" rel="modulepreload">
|
||||
<link href="/_app/immutable/chunks/BWyv3yYM.js" rel="modulepreload">
|
||||
<link href="/_app/immutable/entry/app.B90d0BZx.js" rel="modulepreload">
|
||||
<link href="/_app/immutable/chunks/BK7DUW2U.js" rel="modulepreload">
|
||||
<link href="/_app/immutable/chunks/CslSvznw.js" rel="modulepreload">
|
||||
<link href="/_app/immutable/chunks/C_dJMdcr.js" rel="modulepreload">
|
||||
<link href="/_app/immutable/chunks/Du3f5uIc.js" rel="modulepreload">
|
||||
<link href="/_app/immutable/chunks/B3RSY5nb.js" rel="modulepreload">
|
||||
<link href="/_app/immutable/entry/app.BsYzd6w8.js" rel="modulepreload">
|
||||
|
||||
</head>
|
||||
<body data-sveltekit-preload-data="hover">
|
||||
<div style="display: contents">
|
||||
<script>
|
||||
{
|
||||
__sveltekit_1vssdz = {
|
||||
__sveltekit_3d3whq = {
|
||||
base: ""
|
||||
};
|
||||
|
||||
const element = document.currentScript.parentElement;
|
||||
|
||||
Promise.all([
|
||||
import("/_app/immutable/entry/start.DB7lMQq7.js"),
|
||||
import("/_app/immutable/entry/app.B90d0BZx.js")
|
||||
import("/_app/immutable/entry/start.DZCPQR8C.js"),
|
||||
import("/_app/immutable/entry/app.BsYzd6w8.js")
|
||||
]).then(([kit, app]) => {
|
||||
kit.start(app, element);
|
||||
});
|
||||
|
||||
@@ -0,0 +1,37 @@
|
||||
# Specification Quality Checklist: Reactive Agent Triggering System
|
||||
|
||||
**Purpose**: Validate specification completeness and quality before proceeding to planning
|
||||
**Created**: 2026-03-25
|
||||
**Feature**: [spec.md](../spec.md)
|
||||
|
||||
## Content Quality
|
||||
|
||||
- [x] No implementation details (languages, frameworks, APIs)
|
||||
- [x] Focused on user value and business needs
|
||||
- [x] Written for non-technical stakeholders
|
||||
- [x] All mandatory sections completed
|
||||
|
||||
## Requirement Completeness
|
||||
|
||||
- [x] No [NEEDS CLARIFICATION] markers remain
|
||||
- [x] Requirements are testable and unambiguous
|
||||
- [x] Success criteria are measurable
|
||||
- [x] Success criteria are technology-agnostic (no implementation details)
|
||||
- [x] All acceptance scenarios are defined
|
||||
- [x] Edge cases are identified
|
||||
- [x] Scope is clearly bounded
|
||||
- [x] Dependencies and assumptions identified
|
||||
|
||||
## Feature Readiness
|
||||
|
||||
- [x] All functional requirements have clear acceptance criteria
|
||||
- [x] User scenarios cover primary flows
|
||||
- [x] Feature meets measurable outcomes defined in Success Criteria
|
||||
- [x] No implementation details leak into specification
|
||||
|
||||
## Notes
|
||||
|
||||
- All items pass validation. Spec references K8s Jobs and env vars as these are domain terms (the deployment target), not implementation choices.
|
||||
- Assumptions section documents all design decisions from brainstorming including rate limit defaults, trigger events scope, and coalescing behavior.
|
||||
- 10 user stories covering P1 (core trigger + rate limiting), P2 (visibility + admin), P3 (future-proofing).
|
||||
- 20 functional requirements, 9 success criteria, 7 edge cases.
|
||||
@@ -0,0 +1,108 @@
|
||||
# CLI Command Contracts: Reactive Agent Triggering
|
||||
|
||||
## Agent Trigger Configuration
|
||||
|
||||
### synapbus agent set-triggers
|
||||
|
||||
Configure reactive trigger settings for an agent.
|
||||
|
||||
```bash
|
||||
synapbus agent set-triggers <agent-name> [flags]
|
||||
```
|
||||
|
||||
**Flags**:
|
||||
| Flag | Type | Default | Description |
|
||||
|------|------|---------|-------------|
|
||||
| `--mode` | string | - | Trigger mode: `passive`, `reactive`, `disabled` |
|
||||
| `--cooldown` | int | 600 | Cooldown seconds between runs |
|
||||
| `--daily-budget` | int | 8 | Max runs per UTC day |
|
||||
| `--max-depth` | int | 5 | Max cascade depth |
|
||||
|
||||
**Example**:
|
||||
```bash
|
||||
synapbus agent set-triggers research-mcpproxy \
|
||||
--mode reactive --cooldown 600 --daily-budget 8 --max-depth 5
|
||||
```
|
||||
|
||||
**Output**:
|
||||
```
|
||||
Updated trigger config for research-mcpproxy:
|
||||
mode: reactive
|
||||
cooldown: 600s
|
||||
daily budget: 8
|
||||
max depth: 5
|
||||
```
|
||||
|
||||
### synapbus agent set-image
|
||||
|
||||
Set the K8s container image and env vars for reactive runs.
|
||||
|
||||
```bash
|
||||
synapbus agent set-image <agent-name> [flags]
|
||||
```
|
||||
|
||||
**Flags**:
|
||||
| Flag | Type | Description |
|
||||
|------|------|-------------|
|
||||
| `--image` | string | Container image (required) |
|
||||
| `--env` | string[] | Plain env var: KEY=VALUE (repeatable) |
|
||||
| `--secret-env` | string[] | Secret ref: KEY=secret-name:key-name (repeatable) |
|
||||
| `--resource-preset` | string | `default` or `large` |
|
||||
|
||||
**Example**:
|
||||
```bash
|
||||
synapbus agent set-image research-mcpproxy \
|
||||
--image localhost:32000/universal-agent:latest \
|
||||
--env AGENT_GIT_REPO=Dumbris/agent-research-mcpproxy \
|
||||
--secret-env SYNAPBUS_API_KEY=synapbus-agent-keys:RESEARCH_MCPPROXY_API_KEY \
|
||||
--resource-preset default
|
||||
```
|
||||
|
||||
## Run Management
|
||||
|
||||
### synapbus runs list
|
||||
|
||||
List recent reactive runs.
|
||||
|
||||
```bash
|
||||
synapbus runs list [flags]
|
||||
```
|
||||
|
||||
**Flags**:
|
||||
| Flag | Type | Default | Description |
|
||||
|------|------|---------|-------------|
|
||||
| `--agent` | string | - | Filter by agent name |
|
||||
| `--status` | string | - | Filter by status |
|
||||
| `--limit` | int | 20 | Max results |
|
||||
|
||||
**Output**:
|
||||
```
|
||||
ID AGENT STATUS TRIGGER DURATION CREATED
|
||||
1 research-mcpproxy succeeded DM from algis 3m42s 2026-03-25 10:00
|
||||
2 social-commenter failed @mention in #news 0m45s 2026-03-25 10:15
|
||||
3 research-synapbus running DM from algis - 2026-03-25 10:30
|
||||
```
|
||||
|
||||
### synapbus runs logs
|
||||
|
||||
View error logs for a specific run.
|
||||
|
||||
```bash
|
||||
synapbus runs logs <run-id>
|
||||
```
|
||||
|
||||
**Output**: Last 100 lines of pod logs for the run.
|
||||
|
||||
### synapbus runs retry
|
||||
|
||||
Retry a failed run.
|
||||
|
||||
```bash
|
||||
synapbus runs retry <run-id>
|
||||
```
|
||||
|
||||
**Output**:
|
||||
```
|
||||
Retrying run 2 for social-commenter...
|
||||
New run ID: 4, status: running
|
||||
```
|
||||
@@ -0,0 +1,131 @@
|
||||
# MCP Tool Contracts: Reactive Agent Triggering
|
||||
|
||||
**Note**: These are owner-only admin tools, not agent-callable tools (Constitution Principle IV).
|
||||
|
||||
## configure_triggers
|
||||
|
||||
Configure reactive trigger settings for an agent.
|
||||
|
||||
**Parameters**:
|
||||
```json
|
||||
{
|
||||
"agent_name": "research-mcpproxy",
|
||||
"trigger_mode": "reactive",
|
||||
"cooldown_seconds": 600,
|
||||
"daily_trigger_budget": 8,
|
||||
"max_trigger_depth": 5
|
||||
}
|
||||
```
|
||||
|
||||
All fields except `agent_name` are optional — only provided fields are updated.
|
||||
|
||||
**Returns**:
|
||||
```json
|
||||
{
|
||||
"status": "ok",
|
||||
"agent_name": "research-mcpproxy",
|
||||
"trigger_mode": "reactive",
|
||||
"cooldown_seconds": 600,
|
||||
"daily_trigger_budget": 8,
|
||||
"max_trigger_depth": 5
|
||||
}
|
||||
```
|
||||
|
||||
## set_agent_image
|
||||
|
||||
Set the K8s container image and environment variables for reactive runs.
|
||||
|
||||
**Parameters**:
|
||||
```json
|
||||
{
|
||||
"agent_name": "research-mcpproxy",
|
||||
"k8s_image": "localhost:32000/universal-agent:latest",
|
||||
"k8s_env_json": {
|
||||
"AGENT_GIT_REPO": "Dumbris/agent-research-mcpproxy",
|
||||
"SYNAPBUS_API_KEY": {"secretRef": "synapbus-agent-keys", "key": "RESEARCH_MCPPROXY_API_KEY"}
|
||||
},
|
||||
"k8s_resource_preset": "default"
|
||||
}
|
||||
```
|
||||
|
||||
**Returns**:
|
||||
```json
|
||||
{
|
||||
"status": "ok",
|
||||
"agent_name": "research-mcpproxy",
|
||||
"k8s_image": "localhost:32000/universal-agent:latest"
|
||||
}
|
||||
```
|
||||
|
||||
## list_runs
|
||||
|
||||
List recent reactive runs for an agent.
|
||||
|
||||
**Parameters**:
|
||||
```json
|
||||
{
|
||||
"agent_name": "research-mcpproxy",
|
||||
"status": "failed",
|
||||
"limit": 20
|
||||
}
|
||||
```
|
||||
|
||||
All fields optional. Without `agent_name`, lists all agents' runs.
|
||||
|
||||
**Returns**:
|
||||
```json
|
||||
{
|
||||
"runs": [
|
||||
{
|
||||
"id": 1,
|
||||
"agent_name": "research-mcpproxy",
|
||||
"trigger_event": "message.received",
|
||||
"trigger_from": "algis",
|
||||
"status": "failed",
|
||||
"duration_ms": 45000,
|
||||
"error_log": "Exit code 1 — OOMKilled...",
|
||||
"created_at": "2026-03-25T10:00:00Z"
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
## get_run_logs
|
||||
|
||||
Get full error log for a specific run.
|
||||
|
||||
**Parameters**:
|
||||
```json
|
||||
{
|
||||
"run_id": 1
|
||||
}
|
||||
```
|
||||
|
||||
**Returns**:
|
||||
```json
|
||||
{
|
||||
"run_id": 1,
|
||||
"agent_name": "research-mcpproxy",
|
||||
"status": "failed",
|
||||
"error_log": "... last 100 lines of pod logs ..."
|
||||
}
|
||||
```
|
||||
|
||||
## retry_run
|
||||
|
||||
Retry a failed run.
|
||||
|
||||
**Parameters**:
|
||||
```json
|
||||
{
|
||||
"run_id": 1
|
||||
}
|
||||
```
|
||||
|
||||
**Returns**:
|
||||
```json
|
||||
{
|
||||
"new_run_id": 43,
|
||||
"status": "running"
|
||||
}
|
||||
```
|
||||
@@ -0,0 +1,93 @@
|
||||
# REST API Contracts: Reactive Agent Triggering
|
||||
|
||||
**Note**: REST API is for the embedded Web UI only (Constitution Principle II). Agents use MCP tools.
|
||||
|
||||
## Endpoints
|
||||
|
||||
### GET /api/runs
|
||||
|
||||
List reactive runs with optional filters.
|
||||
|
||||
**Query Parameters**:
|
||||
| Param | Type | Required | Description |
|
||||
|-------|------|----------|-------------|
|
||||
| `agent` | string | No | Filter by agent name |
|
||||
| `status` | string | No | Filter by status (comma-separated) |
|
||||
| `limit` | int | No | Max results (default: 50, max: 200) |
|
||||
| `offset` | int | No | Pagination offset |
|
||||
|
||||
**Response** (200):
|
||||
```json
|
||||
{
|
||||
"runs": [
|
||||
{
|
||||
"id": 1,
|
||||
"agent_name": "research-mcpproxy",
|
||||
"trigger_message_id": 12345,
|
||||
"trigger_event": "message.received",
|
||||
"trigger_depth": 0,
|
||||
"trigger_from": "algis",
|
||||
"status": "succeeded",
|
||||
"k8s_job_name": "reactive-research-mcpproxy-1",
|
||||
"started_at": "2026-03-25T10:00:00Z",
|
||||
"completed_at": "2026-03-25T10:03:42Z",
|
||||
"duration_ms": 222000,
|
||||
"error_log": null,
|
||||
"created_at": "2026-03-25T10:00:00Z"
|
||||
}
|
||||
],
|
||||
"total": 42
|
||||
}
|
||||
```
|
||||
|
||||
### GET /api/runs/:id
|
||||
|
||||
Get a single run with full details including error log.
|
||||
|
||||
**Response** (200): Single run object (same as above).
|
||||
|
||||
### POST /api/runs/:id/retry
|
||||
|
||||
Retry a failed run. Creates a new trigger evaluation for the same agent.
|
||||
|
||||
**Response** (200):
|
||||
```json
|
||||
{
|
||||
"new_run_id": 43,
|
||||
"status": "running"
|
||||
}
|
||||
```
|
||||
|
||||
**Response** (429 — rate limited):
|
||||
```json
|
||||
{
|
||||
"error": "cooldown_active",
|
||||
"cooldown_remaining_seconds": 342
|
||||
}
|
||||
```
|
||||
|
||||
### GET /api/agents/reactive
|
||||
|
||||
List agents with reactive trigger configuration and current status.
|
||||
|
||||
**Response** (200):
|
||||
```json
|
||||
{
|
||||
"agents": [
|
||||
{
|
||||
"name": "research-mcpproxy",
|
||||
"trigger_mode": "reactive",
|
||||
"cooldown_seconds": 600,
|
||||
"daily_trigger_budget": 8,
|
||||
"max_trigger_depth": 5,
|
||||
"k8s_image": "localhost:32000/universal-agent:latest",
|
||||
"pending_work": false,
|
||||
"state": "idle",
|
||||
"today_runs": 3,
|
||||
"cooldown_until": null
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
**`state` values**: `idle`, `running`, `queued` (pending_work set), `cooldown`, `budget_exhausted`
|
||||
@@ -0,0 +1,127 @@
|
||||
# Data Model: Reactive Agent Triggering System
|
||||
|
||||
**Feature**: 014-reactive-agent-triggers
|
||||
**Date**: 2026-03-25
|
||||
|
||||
## Entity Changes
|
||||
|
||||
### Agent (extended)
|
||||
|
||||
Existing `agents` table gains new columns for reactive trigger configuration.
|
||||
|
||||
| Field | Type | Default | Description |
|
||||
|-------|------|---------|-------------|
|
||||
| `trigger_mode` | TEXT | `'passive'` | `passive` (polls only), `reactive` (auto-triggered), `disabled` (no triggers) |
|
||||
| `cooldown_seconds` | INTEGER | `600` | Minimum seconds between reactive runs |
|
||||
| `daily_trigger_budget` | INTEGER | `8` | Max reactive runs per UTC calendar day |
|
||||
| `max_trigger_depth` | INTEGER | `5` | Max agent-to-agent cascade depth |
|
||||
| `k8s_image` | TEXT | `NULL` | Container image for reactive K8s Jobs |
|
||||
| `k8s_env_json` | TEXT | `NULL` | JSON object of env vars (plain + secret refs) |
|
||||
| `k8s_resource_preset` | TEXT | `'default'` | Resource limits: `default` (256Mi/100m) or `large` (2Gi/1CPU) |
|
||||
| `pending_work` | BOOLEAN | `0` | True if triggers arrived while agent was busy |
|
||||
|
||||
### Reactive Run (new)
|
||||
|
||||
New `reactive_runs` table tracking every trigger evaluation.
|
||||
|
||||
| Field | Type | Nullable | Description |
|
||||
|-------|------|----------|-------------|
|
||||
| `id` | INTEGER PK | No | Auto-increment |
|
||||
| `agent_name` | TEXT FK | No | References `agents(name)` |
|
||||
| `trigger_message_id` | INTEGER FK | Yes | References `messages(id)` — the message that caused the trigger |
|
||||
| `trigger_event` | TEXT | No | `message.received` or `message.mentioned` |
|
||||
| `trigger_depth` | INTEGER | No | Depth in the cascade chain (0 = human-initiated) |
|
||||
| `trigger_from` | TEXT | Yes | Agent/user who sent the trigger message |
|
||||
| `status` | TEXT | No | `queued`, `running`, `succeeded`, `failed`, `cooldown_skipped`, `budget_exhausted`, `depth_exceeded` |
|
||||
| `k8s_job_name` | TEXT | Yes | K8s Job name (set when job is created) |
|
||||
| `k8s_namespace` | TEXT | Yes | K8s namespace |
|
||||
| `started_at` | DATETIME | Yes | When the K8s Job was created |
|
||||
| `completed_at` | DATETIME | Yes | When the K8s Job finished |
|
||||
| `duration_ms` | INTEGER | Yes | Computed: completed_at - started_at |
|
||||
| `error_log` | TEXT | Yes | Last 100 lines of pod logs on failure |
|
||||
| `token_cost_json` | TEXT | Yes | Optional: `{"input": N, "output": N}` |
|
||||
| `created_at` | DATETIME | No | When the trigger evaluation happened |
|
||||
|
||||
**Indexes**:
|
||||
- `idx_reactive_runs_agent_created` on `(agent_name, created_at)` — for budget counting and cooldown checks
|
||||
- `idx_reactive_runs_status` on `(status)` — for poller to find active runs
|
||||
- `idx_reactive_runs_agent_status` on `(agent_name, status)` — for coalescing checks
|
||||
|
||||
### State Transitions
|
||||
|
||||
```
|
||||
Trigger evaluation:
|
||||
→ [all checks pass, agent idle] → status: 'running'
|
||||
→ [all checks pass, agent busy] → status: 'queued' (sets pending_work)
|
||||
→ [cooldown not elapsed] → status: 'cooldown_skipped'
|
||||
→ [daily budget exhausted] → status: 'budget_exhausted'
|
||||
→ [depth exceeded] → status: 'depth_exceeded'
|
||||
→ [no k8s_image configured] → status: 'failed'
|
||||
→ [K8s cluster unreachable] → status: 'failed'
|
||||
|
||||
Job completion (via poller):
|
||||
→ [exit code 0] → status: 'succeeded'
|
||||
→ [exit code != 0 / OOM / timeout] → status: 'failed'
|
||||
→ [pending_work set] → new run launched (back to evaluation)
|
||||
```
|
||||
|
||||
### Message Metadata (extended)
|
||||
|
||||
Messages sent by triggered agents carry `trigger_depth` in their metadata JSON field. When the dispatcher evaluates mentions from such a message, it reads the depth and increments it for the next trigger evaluation.
|
||||
|
||||
| Metadata Key | Type | Description |
|
||||
|-------------|------|-------------|
|
||||
| `trigger_depth` | INTEGER | Current depth in cascade chain |
|
||||
|
||||
## Migration: 015_reactive_triggers.sql
|
||||
|
||||
```sql
|
||||
-- Extend agents table with reactive trigger configuration
|
||||
ALTER TABLE agents ADD COLUMN trigger_mode TEXT NOT NULL DEFAULT 'passive';
|
||||
ALTER TABLE agents ADD COLUMN cooldown_seconds INTEGER NOT NULL DEFAULT 600;
|
||||
ALTER TABLE agents ADD COLUMN daily_trigger_budget INTEGER NOT NULL DEFAULT 8;
|
||||
ALTER TABLE agents ADD COLUMN max_trigger_depth INTEGER NOT NULL DEFAULT 5;
|
||||
ALTER TABLE agents ADD COLUMN k8s_image TEXT;
|
||||
ALTER TABLE agents ADD COLUMN k8s_env_json TEXT;
|
||||
ALTER TABLE agents ADD COLUMN k8s_resource_preset TEXT NOT NULL DEFAULT 'default';
|
||||
ALTER TABLE agents ADD COLUMN pending_work INTEGER NOT NULL DEFAULT 0;
|
||||
|
||||
-- New table: reactive trigger runs
|
||||
CREATE TABLE reactive_runs (
|
||||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
agent_name TEXT NOT NULL REFERENCES agents(name),
|
||||
trigger_message_id INTEGER,
|
||||
trigger_event TEXT NOT NULL,
|
||||
trigger_depth INTEGER NOT NULL DEFAULT 0,
|
||||
trigger_from TEXT,
|
||||
status TEXT NOT NULL DEFAULT 'queued',
|
||||
k8s_job_name TEXT,
|
||||
k8s_namespace TEXT,
|
||||
started_at DATETIME,
|
||||
completed_at DATETIME,
|
||||
duration_ms INTEGER,
|
||||
error_log TEXT,
|
||||
token_cost_json TEXT,
|
||||
created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP
|
||||
);
|
||||
|
||||
CREATE INDEX idx_reactive_runs_agent_created ON reactive_runs(agent_name, created_at);
|
||||
CREATE INDEX idx_reactive_runs_status ON reactive_runs(status);
|
||||
CREATE INDEX idx_reactive_runs_agent_status ON reactive_runs(agent_name, status);
|
||||
```
|
||||
|
||||
## k8s_env_json Format
|
||||
|
||||
```json
|
||||
{
|
||||
"AGENT_GIT_REPO": "Dumbris/agent-research-mcpproxy",
|
||||
"AGENT_PROMPT": "You are research-mcpproxy...",
|
||||
"MCPPROXY_URL": "http://kubic.home.arpa:30080",
|
||||
"SYNAPBUS_API_KEY": {
|
||||
"secretRef": "synapbus-agent-keys",
|
||||
"key": "RESEARCH_MCPPROXY_API_KEY"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
Plain string values become `env[].value`. Objects with `secretRef` become `env[].valueFrom.secretKeyRef`.
|
||||
@@ -0,0 +1,106 @@
|
||||
# Implementation Plan: Reactive Agent Triggering System
|
||||
|
||||
**Branch**: `014-reactive-agent-triggers` | **Date**: 2026-03-25 | **Spec**: [spec.md](spec.md)
|
||||
**Input**: Feature specification from `/specs/014-reactive-agent-triggers/spec.md`
|
||||
|
||||
## Summary
|
||||
|
||||
Add a reactor engine to SynapBus that automatically triggers K8s Jobs when agents receive DMs or @mentions. The reactor enforces per-agent rate limits (cooldown, daily budget, trigger depth), ensures sequential execution with coalescing, and provides visibility through a Web UI Agent Runs panel, failure DM notifications, and admin CLI commands.
|
||||
|
||||
## Technical Context
|
||||
|
||||
**Language/Version**: Go 1.25+ (per go.mod)
|
||||
**Primary Dependencies**: go-chi/chi (HTTP), mark3labs/mcp-go (MCP), spf13/cobra (CLI), modernc.org/sqlite (storage), k8s.io/client-go (K8s Jobs)
|
||||
**Storage**: SQLite via modernc.org/sqlite — new migration 015_reactive_triggers.sql
|
||||
**Testing**: `go test ./...` (table-driven tests, Go standard)
|
||||
**Target Platform**: linux/amd64 (kubic deployment), darwin/arm64 (development)
|
||||
**Project Type**: Web service (single binary) with embedded Svelte 5 SPA
|
||||
**Performance Goals**: Trigger evaluation < 100ms, K8s Job creation < 5s after evaluation, Job status polling every 15s
|
||||
**Constraints**: Zero CGO, single binary, single `--data` directory, all state in SQLite
|
||||
**Scale/Scope**: ~4 reactive agents, ~8 triggers/day each, single K8s cluster
|
||||
|
||||
## Constitution Check
|
||||
|
||||
*GATE: Must pass before Phase 0 research. Re-check after Phase 1 design.*
|
||||
|
||||
| Principle | Status | Notes |
|
||||
|-----------|--------|-------|
|
||||
| I. Local-First, Single Binary | PASS | Reactor lives inside SynapBus binary. K8s client is optional (no-op when not in-cluster). |
|
||||
| II. MCP-Native | PASS | Admin tools exposed via MCP. Agents interact through existing MCP tools only. |
|
||||
| III. Pure Go, Zero CGO | PASS | k8s.io/client-go is pure Go. No new CGO deps. |
|
||||
| IV. Multi-Tenant with Ownership | PASS | Trigger config scoped to agent's owner. Run visibility restricted to owner. |
|
||||
| V. Embedded OAuth 2.1 | N/A | No auth changes needed. |
|
||||
| VI. Semantic-Ready Storage | N/A | No vector search changes. |
|
||||
| VII. Swarm Intelligence | N/A | Not affected. |
|
||||
| VIII. Observable by Default | PASS | Every trigger evaluation recorded in reactive_runs. Failed runs send DM + show in Web UI. |
|
||||
| IX. Progressive Complexity | PASS | Reactive triggers are opt-in per agent (trigger_mode='reactive'). Default is 'passive' — no behavior change for existing agents. |
|
||||
| X. Web UI First-Class | PASS | New Agent Runs page with real-time status, filtering, expandable logs. |
|
||||
|
||||
**GATE RESULT: PASS** — No violations.
|
||||
|
||||
## Project Structure
|
||||
|
||||
### Documentation (this feature)
|
||||
|
||||
```text
|
||||
specs/014-reactive-agent-triggers/
|
||||
├── plan.md # This file
|
||||
├── research.md # Phase 0: research findings
|
||||
├── data-model.md # Phase 1: schema design
|
||||
├── quickstart.md # Phase 1: developer onboarding
|
||||
├── contracts/ # Phase 1: API contracts
|
||||
│ ├── rest-api.md # REST endpoints for Web UI
|
||||
│ ├── mcp-tools.md # MCP admin tools
|
||||
│ └── cli-commands.md # Admin CLI commands
|
||||
└── tasks.md # Phase 2: implementation tasks
|
||||
```
|
||||
|
||||
### Source Code (repository root)
|
||||
|
||||
```text
|
||||
internal/
|
||||
├── reactor/ # NEW: reactive trigger engine
|
||||
│ ├── reactor.go # Core decision logic
|
||||
│ ├── store.go # SQLite persistence for reactive_runs
|
||||
│ ├── poller.go # K8s Job status polling goroutine
|
||||
│ └── reactor_test.go # Unit tests
|
||||
├── agents/ # MODIFIED: add trigger fields to agent model
|
||||
│ ├── model.go # Add trigger_mode, cooldown, budget, etc.
|
||||
│ └── store.go # Add trigger config CRUD
|
||||
├── k8s/ # MODIFIED: extend job creation with trigger env vars
|
||||
│ └── runner.go # Add SYNAPBUS_MESSAGE_* env vars
|
||||
├── dispatcher/ # MODIFIED: add reactor as dispatch target
|
||||
│ └── dispatcher.go # Wire reactor into MultiDispatcher
|
||||
├── webhooks/ # MODIFIED: add trigger block to payloads
|
||||
│ └── delivery.go # Enrich payload with depth/run_id
|
||||
├── messaging/ # MODIFIED: propagate trigger depth on agent messages
|
||||
│ └── service.go # Track depth in message metadata
|
||||
├── mcp/ # MODIFIED: add admin MCP tools
|
||||
│ └── bridge.go # Register configure_triggers, list_runs, etc.
|
||||
├── api/ # MODIFIED: add REST endpoints for Web UI
|
||||
│ └── runs.go # NEW: /api/runs endpoints
|
||||
└── web/ # MODIFIED: embed updated SPA
|
||||
└── dist/ # Rebuilt after Svelte changes
|
||||
|
||||
web/ # Svelte source
|
||||
└── src/
|
||||
├── routes/
|
||||
│ └── runs/ # NEW: Agent Runs page
|
||||
│ └── +page.svelte
|
||||
└── lib/
|
||||
└── components/
|
||||
└── RunCard.svelte # NEW: run row component
|
||||
|
||||
schema/
|
||||
└── 015_reactive_triggers.sql # NEW: migration
|
||||
|
||||
cmd/synapbus/
|
||||
└── runs.go # NEW: CLI commands for runs
|
||||
└── agent_triggers.go # NEW: CLI commands for trigger config
|
||||
```
|
||||
|
||||
**Structure Decision**: Follows existing SynapBus layout. New `internal/reactor/` package for core logic. All other changes extend existing packages.
|
||||
|
||||
## Complexity Tracking
|
||||
|
||||
> No violations — section not needed.
|
||||
@@ -0,0 +1,83 @@
|
||||
# Quickstart: Reactive Agent Triggering
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- Go 1.25+ installed
|
||||
- Access to K8s cluster (MicroK8s on kubic) for integration tests
|
||||
- SynapBus built and running locally or on kubic
|
||||
|
||||
## Development Setup
|
||||
|
||||
```bash
|
||||
# Build SynapBus
|
||||
make build
|
||||
|
||||
# Run with hot reload
|
||||
make dev
|
||||
|
||||
# Run tests
|
||||
make test
|
||||
```
|
||||
|
||||
## Key Files to Modify
|
||||
|
||||
### New Files
|
||||
- `internal/reactor/reactor.go` — Core reactor engine
|
||||
- `internal/reactor/store.go` — SQLite persistence
|
||||
- `internal/reactor/poller.go` — K8s Job status poller
|
||||
- `internal/reactor/reactor_test.go` — Unit tests
|
||||
- `internal/api/runs.go` — REST API for Web UI
|
||||
- `schema/015_reactive_triggers.sql` — Migration
|
||||
- `cmd/synapbus/runs.go` — CLI commands
|
||||
- `cmd/synapbus/agent_triggers.go` — CLI trigger config commands
|
||||
- `web/src/routes/runs/+page.svelte` — Web UI Agent Runs page
|
||||
|
||||
### Modified Files
|
||||
- `internal/agents/model.go` — Add trigger fields
|
||||
- `internal/agents/store.go` — Add trigger config CRUD
|
||||
- `internal/k8s/runner.go` — Add trigger env vars to Job creation
|
||||
- `internal/dispatcher/dispatcher.go` — Wire reactor into fan-out
|
||||
- `internal/webhooks/delivery.go` — Add trigger block to payloads
|
||||
- `internal/mcp/bridge.go` — Register admin MCP tools
|
||||
- `cmd/synapbus/main.go` — Register CLI commands
|
||||
|
||||
## Testing Approach
|
||||
|
||||
### Unit Tests (no K8s required)
|
||||
```bash
|
||||
go test ./internal/reactor/... -v
|
||||
```
|
||||
|
||||
Test the reactor decision logic with mock K8s runner:
|
||||
- Cooldown enforcement
|
||||
- Budget counting
|
||||
- Depth checking
|
||||
- Sequential execution / coalescing
|
||||
- Self-mention filtering
|
||||
|
||||
### Integration Tests (requires K8s)
|
||||
```bash
|
||||
go test ./internal/reactor/... -tags=integration -v
|
||||
```
|
||||
|
||||
Test actual K8s Job creation and polling on kubic.
|
||||
|
||||
## Configuring an Agent
|
||||
|
||||
```bash
|
||||
# 1. Set trigger mode and rate limits
|
||||
./synapbus agent set-triggers research-mcpproxy \
|
||||
--mode reactive --cooldown 600 --daily-budget 8
|
||||
|
||||
# 2. Set K8s image and env vars
|
||||
./synapbus agent set-image research-mcpproxy \
|
||||
--image localhost:32000/universal-agent:latest \
|
||||
--env AGENT_GIT_REPO=Dumbris/agent-research-mcpproxy \
|
||||
--secret-env SYNAPBUS_API_KEY=synapbus-agent-keys:RESEARCH_MCPPROXY_API_KEY
|
||||
|
||||
# 3. Send a DM to test
|
||||
# (via Web UI or MCP client)
|
||||
|
||||
# 4. Check run status
|
||||
./synapbus runs list --agent research-mcpproxy
|
||||
```
|
||||
@@ -0,0 +1,76 @@
|
||||
# Research: Reactive Agent Triggering System
|
||||
|
||||
**Feature**: 014-reactive-agent-triggers
|
||||
**Date**: 2026-03-25
|
||||
|
||||
## R1: K8s Job Status Polling vs Callbacks
|
||||
|
||||
**Decision**: Polling via background goroutine every 15 seconds.
|
||||
|
||||
**Rationale**: SynapBus's K8s runner already uses in-cluster client-go. Polling is simpler than setting up K8s watch streams or webhooks back to SynapBus. With ~4 agents and max 8 runs/day each, polling is trivially cheap. The poller queries active runs from SQLite, then checks each K8s Job status via client-go.
|
||||
|
||||
**Alternatives considered**:
|
||||
- K8s Watch API: More responsive but requires long-lived connections, reconnect logic, and is overkill for <10 concurrent jobs.
|
||||
- K8s Job completion callbacks (via init containers or sidecars): Complex, adds container dependencies, violates single-binary principle.
|
||||
- Argo Events sensor: External dependency, violates Principle I.
|
||||
|
||||
## R2: Pending Work Flag Storage
|
||||
|
||||
**Decision**: Boolean `pending_work` column on the `agents` table.
|
||||
|
||||
**Rationale**: Simplest approach. The flag is set to true when a trigger arrives while the agent is busy, and cleared when the coalesced run launches. No need for a separate queue table since the agent's `claim_messages` workflow handles message ordering.
|
||||
|
||||
**Alternatives considered**:
|
||||
- Separate queue table tracking individual trigger messages: Unnecessary complexity — the agent processes all pending messages anyway via `claim_messages`.
|
||||
- In-memory flag: Lost on restart. SQLite is authoritative.
|
||||
- Field on the latest reactive_run record: Complicates queries; cleaner as agent field.
|
||||
|
||||
## R3: Trigger Depth Propagation
|
||||
|
||||
**Decision**: Depth is tracked at two levels: (1) K8s env var `SYNAPBUS_TRIGGER_DEPTH` for the agent to know its depth, (2) stored on each message sent by a triggered agent as metadata, so the reactor can read it when evaluating the next hop.
|
||||
|
||||
**Rationale**: When agent A is triggered at depth N and sends a message mentioning agent B, the message needs to carry depth N+1. The reactor reads this from message metadata when evaluating agent B's trigger. This aligns with the existing `X-SynapBus-Depth` header pattern used for webhooks.
|
||||
|
||||
**Alternatives considered**:
|
||||
- Global depth counter per conversation chain: Complex, requires conversation tracking.
|
||||
- Only counting via webhook headers: Doesn't work for MCP-originated messages.
|
||||
|
||||
## R4: Cooldown Timer — From Start or From Completion
|
||||
|
||||
**Decision**: Cooldown starts from the most recent run's `created_at` timestamp (i.e., when the job was launched, not when it completed).
|
||||
|
||||
**Rationale**: Simpler and more predictable. If an agent runs for 30 minutes, the cooldown is already partially elapsed by completion time. Starting from launch prevents rapid re-triggering even if the previous run was fast.
|
||||
|
||||
**Alternatives considered**:
|
||||
- From completion time: Could lead to very long effective cooldowns for long-running jobs. A 10-minute cooldown + 30-minute run = 40 minutes between runs.
|
||||
- Configurable (start vs completion): Over-engineering for current needs.
|
||||
|
||||
## R5: CronJob vs Reactive Job Overlap Detection
|
||||
|
||||
**Decision**: The reactor checks for any running K8s Job with the agent's label (`synapbus-agent=<name>`), regardless of whether it's a CronJob-spawned or reactor-spawned job. If any is running, `pending_work` is set.
|
||||
|
||||
**Rationale**: The sequential execution constraint applies to all runs, not just reactive ones. Using K8s label selectors is clean and already supported by client-go.
|
||||
|
||||
**Alternatives considered**:
|
||||
- Only tracking reactive runs in SQLite: Misses CronJob runs, could cause concurrent execution.
|
||||
- Requiring agents to report "busy" status via MCP: Adds agent-side complexity, unreliable if agent crashes.
|
||||
|
||||
## R6: Self-Mention Detection
|
||||
|
||||
**Decision**: When extracting mentions from a message, filter out the sender's own agent name. The reactor never triggers an agent based on its own message.
|
||||
|
||||
**Rationale**: Prevents trivial infinite loops where an agent mentions itself in its response.
|
||||
|
||||
**Alternatives considered**:
|
||||
- Relying on depth limit to catch self-loops: Too permissive — wastes budget on preventable triggers.
|
||||
- No self-mention filtering: Dangerous with reactive agents.
|
||||
|
||||
## R7: Web UI Polling vs SSE for Agent Runs
|
||||
|
||||
**Decision**: The Agent Runs page uses polling (every 10 seconds) to refresh run statuses, same as other SynapBus Web UI pages.
|
||||
|
||||
**Rationale**: Consistent with existing Web UI patterns. SSE is already used for message notifications but adding a new SSE channel for run status adds complexity. Polling at 10s intervals is adequate for runs that take minutes.
|
||||
|
||||
**Alternatives considered**:
|
||||
- SSE push: More responsive but adds server-side event infrastructure for a page that's not time-critical.
|
||||
- WebSocket: Overkill, not used elsewhere in SynapBus.
|
||||
@@ -0,0 +1,244 @@
|
||||
# Feature Specification: Reactive Agent Triggering System
|
||||
|
||||
**Feature Branch**: `014-reactive-agent-triggers`
|
||||
**Created**: 2026-03-25
|
||||
**Status**: Draft
|
||||
**Input**: Brainstormed and approved design from conversation — reactive agent triggering via DM/@mention with K8s Job orchestration.
|
||||
|
||||
## Assumptions
|
||||
|
||||
- Reactive triggers fire only on `message.received` (DM) and `message.mentioned` (@mention) events — not on `workflow.state_changed` or `channel.message` (deferred to a future feature)
|
||||
- All agents are hybrid: they have existing K8s CronJob schedules and can additionally be triggered reactively by SynapBus
|
||||
- The reactor engine lives inside SynapBus as `internal/reactor/` — no external coordinator service
|
||||
- K8s is the only trigger mechanism for v1; webhook-based triggers are deferred (infrastructure exists but is not wired to the reactor)
|
||||
- Agent K8s image and env config are stored on the agent registry record — the existing `k8s_handlers` table is for the legacy webhook-style K8s integration and remains unchanged
|
||||
- The universal agent template (`searcher/agents/universal/run_agent.py`) reads `SYNAPBUS_MESSAGE_ID`, `SYNAPBUS_MESSAGE_BODY`, `SYNAPBUS_FROM_AGENT`, `SYNAPBUS_EVENT` env vars and prepends trigger context to the agent prompt
|
||||
- Coalescing: when an agent is busy and new triggers arrive, a `pending_work` flag is set. On job completion, if the flag is set, a new run launches. The agent's `my_status` / `claim_messages` workflow handles processing all pending messages — SynapBus does not queue individual messages
|
||||
- Per-agent configurable rate limits with defaults: cooldown = 600 seconds, daily budget = 8 runs, max trigger depth = 5
|
||||
- Trigger depth is propagated via `SYNAPBUS_TRIGGER_DEPTH` env var and incremented on each agent-to-agent hop; if an agent sends a message via MCP that triggers another agent, the depth increases
|
||||
- Failed reactive jobs send a system DM to the agent's owner when a reactive job fails, including agent name, trigger context, duration, and error summary
|
||||
- The Web UI Agent Runs panel is a new page showing recent reactive triggers with status, trigger context, duration, and expandable error logs
|
||||
- Token cost tracking is optional — agents may report it back but it is not required for v1
|
||||
- Migration number: 015_reactive_triggers.sql (next after existing migrations)
|
||||
- Admin CLI commands use the existing `synapbus` cobra command tree
|
||||
- `k8s_env_json` stores both plain env vars and secret references (format: `{"AGENT_GIT_REPO": "value", "SYNAPBUS_API_KEY": {"secretRef": "secret-name", "key": "key-name"}}`)
|
||||
- SYNAPBUS_MESSAGE_BODY is truncated to 4KB when passed as an env var
|
||||
|
||||
## User Scenarios & Testing *(mandatory)*
|
||||
|
||||
### User Story 1 - Reactive Agent Trigger via DM (Priority: P1)
|
||||
|
||||
A human owner sends a DM to an agent (e.g., "research-mcpproxy") via SynapBus. SynapBus detects the agent has `trigger_mode='reactive'`, passes all rate-limit checks, and automatically launches a K8s Job running the agent's container image. The agent processes the DM as its first priority.
|
||||
|
||||
**Why this priority**: This is the core value proposition — agents respond to messages in near-real-time instead of waiting for the next cron cycle.
|
||||
|
||||
**Independent Test**: Send a DM to a reactive agent, verify a K8s Job is created with the correct env vars, and the agent responds to the message.
|
||||
|
||||
**Acceptance Scenarios**:
|
||||
|
||||
1. **Given** agent "research-mcpproxy" with `trigger_mode='reactive'` and no active runs, **When** a human sends it a DM, **Then** SynapBus creates a K8s Job within 5 seconds with `SYNAPBUS_MESSAGE_ID`, `SYNAPBUS_MESSAGE_BODY`, `SYNAPBUS_FROM_AGENT`, `SYNAPBUS_EVENT=message.received` env vars.
|
||||
2. **Given** agent "research-mcpproxy" with `trigger_mode='passive'`, **When** a human sends it a DM, **Then** no reactive trigger fires; the agent picks up the message on its next scheduled run.
|
||||
3. **Given** agent "research-mcpproxy" with `trigger_mode='reactive'`, **When** a DM is sent, **Then** a `reactive_runs` record is created with `status='running'` and `trigger_event='message.received'`.
|
||||
|
||||
---
|
||||
|
||||
### User Story 2 - Reactive Agent Trigger via @Mention (Priority: P1)
|
||||
|
||||
A human or agent @mentions another agent in a channel message (e.g., "@social-commenter check this thread"). SynapBus detects the mention, checks if the mentioned agent is reactive, and triggers it.
|
||||
|
||||
**Why this priority**: @mentions are the primary way to request agent attention in channel conversations — equally important as DMs.
|
||||
|
||||
**Independent Test**: Post a channel message mentioning a reactive agent, verify a K8s Job is created.
|
||||
|
||||
**Acceptance Scenarios**:
|
||||
|
||||
1. **Given** agent "social-commenter" with `trigger_mode='reactive'`, **When** a message containing "@social-commenter" is posted in a channel, **Then** SynapBus triggers the agent with `SYNAPBUS_EVENT=message.mentioned`.
|
||||
2. **Given** a message mentioning multiple reactive agents, **When** the message is sent, **Then** each mentioned agent is evaluated independently for triggering (subject to their own cooldown/budget).
|
||||
3. **Given** agent "social-commenter" already running, **When** a new @mention arrives, **Then** the `pending_work` flag is set and no additional job is created until the current one completes.
|
||||
|
||||
---
|
||||
|
||||
### User Story 3 - Rate Limiting: Cooldown (Priority: P1)
|
||||
|
||||
To control costs, each agent has a configurable cooldown period. After a reactive run starts, no new reactive run can be triggered for that agent until the cooldown elapses.
|
||||
|
||||
**Why this priority**: Without cooldown, a burst of messages could trigger many expensive runs in rapid succession.
|
||||
|
||||
**Independent Test**: Trigger an agent, then immediately send another DM. Verify the second trigger is recorded as `cooldown_skipped`.
|
||||
|
||||
**Acceptance Scenarios**:
|
||||
|
||||
1. **Given** agent with `cooldown_seconds=600` and a run that started 3 minutes ago, **When** a new DM arrives, **Then** the trigger is recorded as `cooldown_skipped` and no K8s Job is created.
|
||||
2. **Given** agent with `cooldown_seconds=600` and last run started 11 minutes ago, **When** a new DM arrives, **Then** the agent is triggered normally.
|
||||
3. **Given** a trigger that was `cooldown_skipped`, **When** the cooldown elapses, **Then** if `pending_work` is set, a new run launches automatically.
|
||||
|
||||
---
|
||||
|
||||
### User Story 4 - Rate Limiting: Daily Budget (Priority: P1)
|
||||
|
||||
Each agent has a configurable daily limit on the number of reactive runs. Once exhausted, no more reactive triggers fire until the next day.
|
||||
|
||||
**Why this priority**: Hard cap on daily spend per agent prevents runaway costs.
|
||||
|
||||
**Independent Test**: Configure an agent with daily budget of 2, trigger it twice successfully, then send a third DM. Verify the third is recorded as `budget_exhausted`.
|
||||
|
||||
**Acceptance Scenarios**:
|
||||
|
||||
1. **Given** agent with `daily_trigger_budget=8` and 7 runs today, **When** a new DM arrives, **Then** the agent is triggered (8th run).
|
||||
2. **Given** agent with `daily_trigger_budget=8` and 8 runs today, **When** a new DM arrives, **Then** the trigger is recorded as `budget_exhausted` and no job is created.
|
||||
3. **Given** budget-exhausted agent, **When** a new calendar day begins (UTC), **Then** the budget resets and new triggers can fire.
|
||||
|
||||
---
|
||||
|
||||
### User Story 5 - Rate Limiting: Trigger Depth (Priority: P1)
|
||||
|
||||
When agents trigger other agents (agent A's response mentions @agent-B), the depth counter increments. If depth exceeds the agent's `max_trigger_depth`, the cascade stops.
|
||||
|
||||
**Why this priority**: Prevents infinite agent-to-agent loops which could be extremely costly.
|
||||
|
||||
**Independent Test**: Set max_trigger_depth=2 on an agent, simulate a depth-3 trigger chain, verify the third hop is blocked.
|
||||
|
||||
**Acceptance Scenarios**:
|
||||
|
||||
1. **Given** agent with `max_trigger_depth=5` and an incoming trigger at depth 4, **When** evaluated, **Then** the trigger fires (depth 4 < max 5).
|
||||
2. **Given** agent with `max_trigger_depth=5` and an incoming trigger at depth 5, **When** evaluated, **Then** the trigger is blocked and recorded as `depth_exceeded`.
|
||||
3. **Given** a human-initiated DM (depth 0), **When** the triggered agent sends a message mentioning another agent, **Then** the second agent receives the trigger with depth 1.
|
||||
|
||||
---
|
||||
|
||||
### User Story 6 - Sequential Execution with Coalescing (Priority: P1)
|
||||
|
||||
Only one reactive K8s Job runs per agent at a time. If new triggers arrive while the agent is busy, they are coalesced — a single follow-up run launches when the current one completes, and the agent processes all accumulated messages.
|
||||
|
||||
**Why this priority**: Prevents concurrent modification of agent workspaces and saves tokens by avoiding redundant startups.
|
||||
|
||||
**Independent Test**: Trigger an agent, send 3 more DMs while it's running. Verify only one follow-up run launches after the first completes.
|
||||
|
||||
**Acceptance Scenarios**:
|
||||
|
||||
1. **Given** agent currently running a reactive job, **When** a new DM arrives, **Then** `pending_work` flag is set to true, no new job is created, and the trigger is recorded as `queued`.
|
||||
2. **Given** agent finishes a run and `pending_work` is true, **When** the poller detects job completion, **Then** `pending_work` is cleared and a new coalesced run is launched (subject to cooldown/budget checks).
|
||||
3. **Given** agent finishes a run and `pending_work` is false, **When** the poller detects job completion, **Then** no follow-up run launches.
|
||||
4. **Given** 5 DMs arrive while agent is busy, **When** the follow-up run launches, **Then** only one K8s Job is created (not 5), and the agent uses `claim_messages` to process all pending messages.
|
||||
|
||||
---
|
||||
|
||||
### User Story 7 - Job Failure Notification (Priority: P2)
|
||||
|
||||
When a reactive K8s Job fails (exit code != 0, OOMKilled, timeout), SynapBus detects the failure, retrieves pod logs, records the error, and sends a system DM to the agent's human owner.
|
||||
|
||||
**Why this priority**: Visibility into failures is essential for debugging but not strictly required for the trigger mechanism to function.
|
||||
|
||||
**Independent Test**: Configure an agent with an image that exits with error, trigger it, verify owner receives a system DM with error details.
|
||||
|
||||
**Acceptance Scenarios**:
|
||||
|
||||
1. **Given** a reactive job that fails with exit code 1, **When** the poller detects failure, **Then** the `reactive_runs` record is updated with `status='failed'`, `error_log` containing the last 100 lines of pod logs, and `completed_at` timestamp.
|
||||
2. **Given** a failed reactive run, **When** the failure is recorded, **Then** a system DM is sent to the agent's owner with agent name, trigger reason, duration, and error summary.
|
||||
3. **Given** a reactive job that exceeds its timeout, **When** the pod is killed, **Then** the run is recorded as `failed` with error indicating timeout.
|
||||
|
||||
---
|
||||
|
||||
### User Story 8 - Web UI Agent Runs Panel (Priority: P2)
|
||||
|
||||
The SynapBus Web UI includes an "Agent Runs" page showing recent reactive triggers, their status, and details. Owners can filter by agent and status, view error logs, click through to the original trigger message, and retry failed runs.
|
||||
|
||||
**Why this priority**: Complements DM notifications with a historical, browsable view — important for day-to-day management but not blocking core functionality.
|
||||
|
||||
**Independent Test**: Trigger several agents (some succeed, some fail), navigate to Agent Runs page, verify all runs are listed with correct status and details.
|
||||
|
||||
**Acceptance Scenarios**:
|
||||
|
||||
1. **Given** several reactive runs have occurred, **When** the owner navigates to the Agent Runs page, **Then** runs are listed in reverse chronological order showing: status badge, agent name, trigger reason, duration, and timestamp.
|
||||
2. **Given** a failed run, **When** the owner clicks to expand it, **Then** the error log and a "Retry" button are shown.
|
||||
3. **Given** the owner clicks "Retry" on a failed run, **When** the retry fires, **Then** a new reactive run is created for the same agent (subject to cooldown/budget checks).
|
||||
4. **Given** multiple reactive agents, **When** the owner views the page, **Then** agent summary cards at the top show: name, today's budget usage (e.g., "3/8 runs"), cooldown status, and current state (idle/running/queued).
|
||||
5. **Given** a run with a trigger message, **When** the owner clicks the message link, **Then** they are navigated to the message in the Web UI.
|
||||
|
||||
---
|
||||
|
||||
### User Story 9 - Admin CLI for Trigger Configuration (Priority: P2)
|
||||
|
||||
System administrators can configure reactive triggers per agent via the CLI: set trigger mode, cooldown, daily budget, max depth, K8s image, and environment variables.
|
||||
|
||||
**Why this priority**: Required for initial setup and ongoing management, but can be done via direct DB manipulation as a workaround.
|
||||
|
||||
**Independent Test**: Use CLI to configure an agent as reactive, then verify the agent triggers on DM.
|
||||
|
||||
**Acceptance Scenarios**:
|
||||
|
||||
1. **Given** agent "research-mcpproxy", **When** admin runs `synapbus agent set-triggers ... --mode reactive --cooldown 600 --daily-budget 8 --max-depth 5`, **Then** the agent's trigger configuration is updated in the registry.
|
||||
2. **Given** agent with no image configured, **When** admin runs `synapbus agent set-image ... --image <image> --env KEY=VALUE`, **Then** the image and env config are stored on the agent record.
|
||||
3. **Given** admin wants to view recent runs, **When** they run `synapbus runs list --agent <name>`, **Then** recent runs are displayed with status, duration, and trigger reason.
|
||||
4. **Given** a failed run, **When** admin runs `synapbus runs logs <run-id>`, **Then** the error log for that run is displayed.
|
||||
|
||||
---
|
||||
|
||||
### User Story 10 - Webhook Payload Enrichment (Priority: P3)
|
||||
|
||||
Webhook payloads for `message.received` and `message.mentioned` events include a `trigger` block with depth and run context, enabling future webhook-based agent triggers.
|
||||
|
||||
**Why this priority**: Future-proofing for webhook-based triggers. No immediate user need but prepares the infrastructure.
|
||||
|
||||
**Independent Test**: Register a webhook, send a message that triggers it, verify the payload includes the `trigger` block.
|
||||
|
||||
**Acceptance Scenarios**:
|
||||
|
||||
1. **Given** an agent with a registered webhook for `message.received`, **When** a DM is sent, **Then** the webhook payload includes a `trigger` object with `depth` and `triggered_by_run_id` fields.
|
||||
|
||||
---
|
||||
|
||||
### Edge Cases
|
||||
|
||||
- What happens when the K8s cluster is unreachable? The reactor records the run as `failed` with a connection error and sends a failure DM to the owner.
|
||||
- What happens when a reactive agent's K8s image is not configured? The reactor skips the trigger and logs a warning. The trigger is recorded as `failed` with reason "no k8s_image configured".
|
||||
- What happens when two DMs arrive simultaneously for the same agent? The reactor processes them sequentially (database-level locking on the agent). The first creates a job; the second sets `pending_work`.
|
||||
- What happens when a scheduled CronJob and a reactive trigger overlap? The reactor checks for any running K8s Job for that agent (both scheduled and reactive). If one is running, it sets `pending_work` and waits.
|
||||
- What happens when the daily budget resets while a coalesced run is pending? The pending run uses the new day's budget.
|
||||
- What happens when an agent is mentioned in its own message (self-mention)? Self-mentions are ignored — an agent cannot trigger itself.
|
||||
- What happens when the message body exceeds 4KB? It is truncated to 4KB in the `SYNAPBUS_MESSAGE_BODY` env var with a `[truncated]` suffix.
|
||||
|
||||
## Requirements *(mandatory)*
|
||||
|
||||
### Functional Requirements
|
||||
|
||||
- **FR-001**: System MUST detect DMs and @mentions to agents with `trigger_mode='reactive'` and initiate a reactive trigger evaluation.
|
||||
- **FR-002**: System MUST enforce per-agent cooldown periods between reactive runs, rejecting triggers during cooldown.
|
||||
- **FR-003**: System MUST enforce per-agent daily run budgets, rejecting triggers when the budget is exhausted.
|
||||
- **FR-004**: System MUST track and enforce trigger depth limits to prevent infinite agent-to-agent cascades.
|
||||
- **FR-005**: System MUST ensure only one reactive K8s Job runs per agent at any time (sequential execution).
|
||||
- **FR-006**: System MUST coalesce pending triggers — when a new trigger arrives while an agent is busy, a `pending_work` flag is set and a single follow-up run launches after the current job completes.
|
||||
- **FR-007**: System MUST pass trigger context to K8s Jobs via environment variables: `SYNAPBUS_MESSAGE_ID`, `SYNAPBUS_MESSAGE_BODY`, `SYNAPBUS_FROM_AGENT`, `SYNAPBUS_EVENT`, `SYNAPBUS_TRIGGER_DEPTH`.
|
||||
- **FR-008**: System MUST record every trigger evaluation (successful or not) in the `reactive_runs` table with appropriate status.
|
||||
- **FR-009**: System MUST poll K8s Job status and update `reactive_runs` records when jobs complete (succeed or fail).
|
||||
- **FR-010**: System MUST retrieve and store the last 100 lines of pod logs for failed reactive runs.
|
||||
- **FR-011**: System MUST send a system DM to the agent's owner when a reactive job fails, including agent name, trigger context, duration, and error summary.
|
||||
- **FR-012**: System MUST provide a Web UI page listing reactive runs with filtering by agent and status.
|
||||
- **FR-013**: System MUST display agent summary cards in the Web UI showing budget usage, cooldown status, and current state.
|
||||
- **FR-014**: System MUST allow retrying failed runs from the Web UI (subject to rate limits).
|
||||
- **FR-015**: System MUST provide CLI commands for configuring agent trigger settings (mode, cooldown, budget, depth, image, env vars).
|
||||
- **FR-016**: System MUST provide CLI commands for listing and inspecting reactive runs.
|
||||
- **FR-017**: System MUST ignore self-mentions (an agent cannot trigger itself).
|
||||
- **FR-018**: System MUST truncate `SYNAPBUS_MESSAGE_BODY` to 4KB when passed as an env var.
|
||||
- **FR-019**: System MUST include trigger context (`depth`, `triggered_by_run_id`) in webhook payloads for `message.received` and `message.mentioned` events.
|
||||
- **FR-020**: System MUST support configurable rate limits per agent (cooldown, daily budget, max depth) with system-wide defaults.
|
||||
|
||||
### Key Entities
|
||||
|
||||
- **Agent (extended)**: Gains `trigger_mode` (passive/reactive/disabled), `cooldown_seconds`, `daily_trigger_budget`, `max_trigger_depth`, `k8s_image`, `k8s_env_json`, `k8s_resource_preset` fields.
|
||||
- **Reactive Run**: A record of a trigger evaluation and its outcome. Tracks agent, trigger message, event type, depth, status (queued/running/succeeded/failed/cooldown_skipped/budget_exhausted/depth_exceeded), K8s job metadata, timing, error logs, and optional token cost.
|
||||
- **Pending Work Flag**: A per-agent boolean indicating that new triggers arrived while the agent was busy. Stored on the agent record or in a dedicated field on the latest running reactive_run.
|
||||
|
||||
## Success Criteria *(mandatory)*
|
||||
|
||||
### Measurable Outcomes
|
||||
|
||||
- **SC-001**: Reactive agents respond to DMs and @mentions within 30 seconds of message delivery (time from message sent to K8s Job created).
|
||||
- **SC-002**: No more than one reactive K8s Job runs per agent at any time — verified by checking job state and run records.
|
||||
- **SC-003**: Cooldown enforcement prevents back-to-back triggers — an agent triggered at time T cannot be triggered again before T + cooldown_seconds.
|
||||
- **SC-004**: Daily budget enforcement caps reactive runs — after N runs in a calendar day (UTC), all further triggers are recorded as `budget_exhausted`.
|
||||
- **SC-005**: Trigger depth enforcement prevents cascades beyond the configured limit — a trigger chain deeper than max_trigger_depth is blocked.
|
||||
- **SC-006**: Agent owners receive failure notifications within 60 seconds of job failure detection.
|
||||
- **SC-007**: The Web UI Agent Runs page accurately reflects all reactive runs with correct status, timing, and trigger context.
|
||||
- **SC-008**: Admin CLI commands successfully configure trigger settings and display run history.
|
||||
- **SC-009**: Coalesced runs process all pending messages in a single session — verified by checking that the agent handles all queued work.
|
||||
@@ -0,0 +1,109 @@
|
||||
# Feature Specification: SQL Query Interface + Split Connection Pools
|
||||
|
||||
**Feature Branch**: `015-sql-query-split-pools`
|
||||
**Created**: 2026-03-26
|
||||
**Status**: Draft
|
||||
**Input**: Architecture research from reactive agent triggering session
|
||||
|
||||
## Assumptions
|
||||
|
||||
- SQL query interface is exposed as a `query` action via the existing `execute` MCP tool, not a new top-level MCP tool
|
||||
- Queries are read-only (enforced via `PRAGMA query_only=ON` on a dedicated connection)
|
||||
- Agents query curated SQL views (not raw tables) that bake in per-agent access control
|
||||
- Views: `my_messages`, `my_channels`, `channel_messages` — parameterized by the authenticated agent's name
|
||||
- Results are automatically limited to 100 rows; agent can specify lower limit
|
||||
- Query timeout: 5 seconds max
|
||||
- Only SELECT statements allowed (validated before execution); WITH (CTEs) permitted
|
||||
- Split connection pools: writeDB (MaxOpenConns=1) for all INSERT/UPDATE/DELETE, readDB (MaxOpenConns=8) for all SELECT
|
||||
- Both pools share the same SQLite file with WAL mode
|
||||
- The read pool uses `PRAGMA query_only=ON` for safety
|
||||
- No schema changes needed — this is a runtime architecture change
|
||||
- Agent SQL queries use the read pool
|
||||
|
||||
## User Scenarios & Testing *(mandatory)*
|
||||
|
||||
### User Story 1 - Agent Queries Messages via SQL (Priority: P1)
|
||||
|
||||
An agent connected via MCP uses the `execute` tool to run a SQL query against its accessible messages. For example: "Show me all messages in #news-mcpproxy from the last 3 days with priority >= 7".
|
||||
|
||||
**Why this priority**: Removes the expressiveness ceiling — agents can compose arbitrary queries instead of being limited to fixed API endpoints.
|
||||
|
||||
**Independent Test**: Agent calls `execute` with `call('query', {sql: "SELECT * FROM my_messages WHERE channel_name = 'news-mcpproxy' AND priority >= 7 ORDER BY created_at DESC LIMIT 5"})` and gets results.
|
||||
|
||||
**Acceptance Scenarios**:
|
||||
|
||||
1. **Given** an authenticated agent, **When** it calls `query` with a valid SELECT, **Then** it receives JSON results with column names and rows.
|
||||
2. **Given** an agent, **When** it runs a query referencing `my_messages`, **Then** it only sees messages it has access to (own DMs + joined channels).
|
||||
3. **Given** an agent, **When** it runs `INSERT INTO messages ...`, **Then** the query is rejected with "only SELECT statements allowed".
|
||||
4. **Given** an agent, **When** it runs a query without LIMIT, **Then** results are automatically capped at 100 rows.
|
||||
5. **Given** an agent, **When** it runs a slow query (> 5s), **Then** the query is cancelled and an error is returned.
|
||||
|
||||
---
|
||||
|
||||
### User Story 2 - Split Read/Write Connection Pools (Priority: P1)
|
||||
|
||||
SynapBus uses separate connection pools for reads and writes to eliminate SQLITE_BUSY errors under concurrent agent load.
|
||||
|
||||
**Why this priority**: Directly fixes the SQLITE_BUSY errors observed during reactive agent runs.
|
||||
|
||||
**Independent Test**: Run concurrent read and write operations; verify no SQLITE_BUSY errors and writes serialize correctly.
|
||||
|
||||
**Acceptance Scenarios**:
|
||||
|
||||
1. **Given** concurrent agents sending messages, **When** writes happen simultaneously, **Then** they serialize through the single-writer pool without SQLITE_BUSY.
|
||||
2. **Given** a write in progress, **When** a read query arrives, **Then** the read executes immediately on the read pool (WAL mode).
|
||||
3. **Given** the read pool, **When** any write operation is attempted, **Then** it fails (query_only=ON enforcement).
|
||||
|
||||
---
|
||||
|
||||
### User Story 3 - Agent Queries Channel Messages (Priority: P2)
|
||||
|
||||
An agent queries messages from a specific channel with rich filtering — date ranges, keywords, reactions, workflow states.
|
||||
|
||||
**Why this priority**: Enables the social-commenter to query #opportunities channel structured data via SQL.
|
||||
|
||||
**Acceptance Scenarios**:
|
||||
|
||||
1. **Given** an agent that has joined #opportunities, **When** it queries `SELECT * FROM channel_messages WHERE channel_name = 'opportunities' AND created_at > datetime('now', '-3 days')`, **Then** it sees messages from that channel.
|
||||
2. **Given** an agent that has NOT joined a private channel, **When** it queries that channel's messages, **Then** no results are returned.
|
||||
|
||||
---
|
||||
|
||||
### Edge Cases
|
||||
|
||||
- Query with syntax error returns a clear error message, not a crash
|
||||
- Query referencing non-existent view returns "no such table" error
|
||||
- Empty result set returns empty array, not null
|
||||
- Very large result (>100 rows) is truncated with a warning
|
||||
- Concurrent SQL queries from multiple agents don't interfere
|
||||
|
||||
## Requirements *(mandatory)*
|
||||
|
||||
### Functional Requirements
|
||||
|
||||
- **FR-001**: System MUST provide a `query` action callable via the `execute` MCP tool that accepts a SQL string and returns results as JSON.
|
||||
- **FR-002**: System MUST enforce read-only execution — no INSERT, UPDATE, DELETE, DROP, ALTER, or PRAGMA statements allowed.
|
||||
- **FR-003**: System MUST expose curated views (`my_messages`, `my_channels`, `channel_messages`) that enforce per-agent access control.
|
||||
- **FR-004**: System MUST automatically limit query results to 100 rows (or fewer if agent specifies).
|
||||
- **FR-005**: System MUST cancel queries that exceed 5 seconds.
|
||||
- **FR-006**: System MUST use a separate read-only connection pool (MaxOpenConns=8) for all SELECT operations.
|
||||
- **FR-007**: System MUST use a single-writer connection pool (MaxOpenConns=1) for all write operations.
|
||||
- **FR-008**: System MUST configure `PRAGMA query_only=ON` on the read pool connections.
|
||||
- **FR-009**: System MUST validate SQL statements before execution — only SELECT and WITH (CTE) prefixes allowed.
|
||||
- **FR-010**: System MUST return query results as `{columns: [...], rows: [[...], ...], row_count: N, truncated: bool}`.
|
||||
|
||||
### Key Entities
|
||||
|
||||
- **Read Pool**: SQLite connection pool with MaxOpenConns=8, query_only=ON, for all SELECT operations including agent SQL queries.
|
||||
- **Write Pool**: SQLite connection pool with MaxOpenConns=1, for all INSERT/UPDATE/DELETE operations.
|
||||
- **Agent Views**: SQL views parameterized by agent name that enforce access control.
|
||||
|
||||
## Success Criteria *(mandatory)*
|
||||
|
||||
### Measurable Outcomes
|
||||
|
||||
- **SC-001**: Agents can execute arbitrary SELECT queries against curated views and receive structured JSON results within 5 seconds.
|
||||
- **SC-002**: No SQLITE_BUSY errors under concurrent 4-agent workload (verified by running all 4 reactive agents simultaneously).
|
||||
- **SC-003**: Write operations on the read pool are rejected at the SQLite engine level.
|
||||
- **SC-004**: Query results are limited to 100 rows maximum.
|
||||
- **SC-005**: All 28+ existing test packages continue to pass with the split pool architecture.
|
||||
@@ -299,4 +299,20 @@ export const onboarding = {
|
||||
}
|
||||
};
|
||||
|
||||
// Reactive Runs
|
||||
export const runs = {
|
||||
list: (params?: { agent?: string; status?: string; limit?: number; offset?: number }) => {
|
||||
const qs = new URLSearchParams();
|
||||
if (params?.agent) qs.set('agent', params.agent);
|
||||
if (params?.status) qs.set('status', params.status);
|
||||
if (params?.limit) qs.set('limit', String(params.limit));
|
||||
if (params?.offset) qs.set('offset', String(params.offset));
|
||||
const q = qs.toString();
|
||||
return request<{ runs: any[]; total: number }>('GET', `/api/runs${q ? '?' + q : ''}`);
|
||||
},
|
||||
get: (id: number) => request<any>('GET', `/api/runs/${id}`),
|
||||
retry: (id: number) => request<any>('POST', `/api/runs/${id}/retry`),
|
||||
reactiveAgents: () => request<{ agents: any[] }>('GET', '/api/agents/reactive')
|
||||
};
|
||||
|
||||
export { ApiError };
|
||||
|
||||
@@ -67,6 +67,7 @@
|
||||
|
||||
const adminLinks = [
|
||||
{ href: '/agents', label: 'Agents' },
|
||||
{ href: '/runs', label: 'Agent Runs' },
|
||||
{ href: '/skills', label: 'Skills' },
|
||||
{ href: '/settings', label: 'Settings' }
|
||||
];
|
||||
|
||||
@@ -0,0 +1,444 @@
|
||||
<script lang="ts">
|
||||
import { runs } from '$lib/api/client';
|
||||
import { onMount } from 'svelte';
|
||||
|
||||
let runsList = $state<any[]>([]);
|
||||
let total = $state(0);
|
||||
let reactiveAgents = $state<any[]>([]);
|
||||
let filterAgent = $state('');
|
||||
let filterStatus = $state('');
|
||||
let loading = $state(true);
|
||||
let expandedRun = $state<number | null>(null);
|
||||
|
||||
onMount(() => {
|
||||
loadData();
|
||||
const interval = setInterval(loadData, 10000);
|
||||
return () => clearInterval(interval);
|
||||
});
|
||||
|
||||
async function loadData() {
|
||||
try {
|
||||
const [runsRes, agentsRes] = await Promise.all([
|
||||
runs.list({ agent: filterAgent || undefined, status: filterStatus || undefined, limit: 50 }),
|
||||
runs.reactiveAgents()
|
||||
]);
|
||||
runsList = runsRes.runs ?? [];
|
||||
total = runsRes.total;
|
||||
reactiveAgents = agentsRes.agents ?? [];
|
||||
} catch {
|
||||
// handled
|
||||
}
|
||||
loading = false;
|
||||
}
|
||||
|
||||
function statusColor(status: string): string {
|
||||
switch (status) {
|
||||
case 'succeeded': return 'var(--color-success, #22c55e)';
|
||||
case 'running': return 'var(--color-warning, #eab308)';
|
||||
case 'failed': return 'var(--color-error, #ef4444)';
|
||||
case 'queued': return 'var(--color-info, #3b82f6)';
|
||||
case 'cooldown_skipped': return '#94a3b8';
|
||||
case 'budget_exhausted': return '#f97316';
|
||||
case 'depth_exceeded': return '#a855f7';
|
||||
default: return '#6b7280';
|
||||
}
|
||||
}
|
||||
|
||||
function formatDuration(ms: number | null): string {
|
||||
if (!ms) return '-';
|
||||
const secs = Math.floor(ms / 1000);
|
||||
if (secs < 60) return `${secs}s`;
|
||||
return `${Math.floor(secs / 60)}m${secs % 60}s`;
|
||||
}
|
||||
|
||||
function formatTime(t: string): string {
|
||||
if (!t) return '-';
|
||||
const d = new Date(t);
|
||||
return d.toLocaleTimeString([], { hour: '2-digit', minute: '2-digit' }) + ' ' + d.toLocaleDateString([], { month: 'short', day: 'numeric' });
|
||||
}
|
||||
|
||||
function triggerLabel(run: any): string {
|
||||
if (run.trigger_event === 'message.received') return `DM from ${run.trigger_from || 'unknown'}`;
|
||||
if (run.trigger_event === 'message.mentioned') return `@mention by ${run.trigger_from || 'unknown'}`;
|
||||
return run.trigger_event;
|
||||
}
|
||||
|
||||
function stateColor(state: string): string {
|
||||
switch (state) {
|
||||
case 'idle': return '#22c55e';
|
||||
case 'running': return '#eab308';
|
||||
case 'queued': return '#3b82f6';
|
||||
case 'cooldown': return '#94a3b8';
|
||||
case 'budget_exhausted': return '#f97316';
|
||||
default: return '#6b7280';
|
||||
}
|
||||
}
|
||||
|
||||
async function retryRun(id: number) {
|
||||
try {
|
||||
await runs.retry(id);
|
||||
await loadData();
|
||||
} catch (e: any) {
|
||||
alert(e.message || 'Retry failed');
|
||||
}
|
||||
}
|
||||
|
||||
function toggleExpand(id: number) {
|
||||
expandedRun = expandedRun === id ? null : id;
|
||||
}
|
||||
</script>
|
||||
|
||||
<svelte:head>
|
||||
<title>Agent Runs - SynapBus</title>
|
||||
</svelte:head>
|
||||
|
||||
<div class="page-container">
|
||||
<h1>Agent Runs</h1>
|
||||
|
||||
<!-- Agent summary cards -->
|
||||
{#if reactiveAgents.length > 0}
|
||||
<div class="agent-cards">
|
||||
{#each reactiveAgents as agent}
|
||||
<div class="agent-card">
|
||||
<div class="agent-card-header">
|
||||
<span class="agent-name">{agent.name}</span>
|
||||
<span class="state-badge" style="background:{stateColor(agent.state)}">{agent.state}</span>
|
||||
</div>
|
||||
<div class="agent-card-stats">
|
||||
<span class="stat">
|
||||
<span class="stat-value">{agent.today_runs}/{agent.daily_trigger_budget}</span>
|
||||
<span class="stat-label">runs today</span>
|
||||
</span>
|
||||
<span class="stat">
|
||||
<span class="stat-value">{agent.cooldown_seconds}s</span>
|
||||
<span class="stat-label">cooldown</span>
|
||||
</span>
|
||||
</div>
|
||||
</div>
|
||||
{/each}
|
||||
</div>
|
||||
{/if}
|
||||
|
||||
<!-- Filters -->
|
||||
<div class="filters">
|
||||
<select bind:value={filterAgent} onchange={loadData}>
|
||||
<option value="">All agents</option>
|
||||
{#each reactiveAgents as agent}
|
||||
<option value={agent.name}>{agent.name}</option>
|
||||
{/each}
|
||||
</select>
|
||||
<select bind:value={filterStatus} onchange={loadData}>
|
||||
<option value="">All statuses</option>
|
||||
<option value="running">Running</option>
|
||||
<option value="succeeded">Succeeded</option>
|
||||
<option value="failed">Failed</option>
|
||||
<option value="queued">Queued</option>
|
||||
<option value="cooldown_skipped">Cooldown Skipped</option>
|
||||
<option value="budget_exhausted">Budget Exhausted</option>
|
||||
<option value="depth_exceeded">Depth Exceeded</option>
|
||||
</select>
|
||||
<span class="total-count">{total} runs</span>
|
||||
</div>
|
||||
|
||||
<!-- Runs list -->
|
||||
{#if loading}
|
||||
<div class="loading">Loading...</div>
|
||||
{:else if runsList.length === 0}
|
||||
<div class="empty">No reactive runs found.</div>
|
||||
{:else}
|
||||
<div class="runs-list">
|
||||
{#each runsList as run}
|
||||
<div class="run-row" class:expanded={expandedRun === run.id}>
|
||||
<button class="run-row-main" onclick={() => toggleExpand(run.id)}>
|
||||
<span class="status-dot" style="background:{statusColor(run.status)}"></span>
|
||||
<span class="run-agent">{run.agent_name}</span>
|
||||
<span class="run-trigger">{triggerLabel(run)}</span>
|
||||
<span class="run-status">{run.status}</span>
|
||||
<span class="run-duration">{formatDuration(run.duration_ms)}</span>
|
||||
<span class="run-time">{formatTime(run.created_at)}</span>
|
||||
<span class="expand-arrow">{expandedRun === run.id ? '▼' : '▶'}</span>
|
||||
</button>
|
||||
|
||||
{#if expandedRun === run.id}
|
||||
<div class="run-details">
|
||||
<div class="detail-row">
|
||||
<span class="detail-label">Run ID</span>
|
||||
<span class="detail-value">{run.id}</span>
|
||||
</div>
|
||||
<div class="detail-row">
|
||||
<span class="detail-label">K8s Job</span>
|
||||
<span class="detail-value">{run.k8s_job_name || '-'}</span>
|
||||
</div>
|
||||
<div class="detail-row">
|
||||
<span class="detail-label">Depth</span>
|
||||
<span class="detail-value">{run.trigger_depth}</span>
|
||||
</div>
|
||||
{#if run.trigger_message_id}
|
||||
<div class="detail-row">
|
||||
<span class="detail-label">Trigger Message</span>
|
||||
<a href="/dm/{run.trigger_from}?msg={run.trigger_message_id}" class="detail-link">
|
||||
Message #{run.trigger_message_id}
|
||||
</a>
|
||||
</div>
|
||||
{/if}
|
||||
{#if run.error_log}
|
||||
<div class="error-log">
|
||||
<div class="error-log-header">Error Log</div>
|
||||
<pre>{run.error_log}</pre>
|
||||
</div>
|
||||
{/if}
|
||||
{#if run.status === 'failed'}
|
||||
<button class="retry-btn" onclick={() => retryRun(run.id)}>
|
||||
Retry
|
||||
</button>
|
||||
{/if}
|
||||
</div>
|
||||
{/if}
|
||||
</div>
|
||||
{/each}
|
||||
</div>
|
||||
{/if}
|
||||
</div>
|
||||
|
||||
<style>
|
||||
.page-container {
|
||||
max-width: 1000px;
|
||||
margin: 0 auto;
|
||||
padding: 1.5rem;
|
||||
}
|
||||
|
||||
h1 {
|
||||
font-size: 1.5rem;
|
||||
font-weight: 600;
|
||||
margin-bottom: 1rem;
|
||||
}
|
||||
|
||||
.agent-cards {
|
||||
display: flex;
|
||||
gap: 0.75rem;
|
||||
flex-wrap: wrap;
|
||||
margin-bottom: 1rem;
|
||||
}
|
||||
|
||||
.agent-card {
|
||||
background: var(--color-surface, #1e293b);
|
||||
border: 1px solid var(--color-border, #334155);
|
||||
border-radius: 0.5rem;
|
||||
padding: 0.75rem 1rem;
|
||||
min-width: 180px;
|
||||
flex: 1;
|
||||
}
|
||||
|
||||
.agent-card-header {
|
||||
display: flex;
|
||||
justify-content: space-between;
|
||||
align-items: center;
|
||||
margin-bottom: 0.5rem;
|
||||
}
|
||||
|
||||
.agent-name {
|
||||
font-weight: 600;
|
||||
font-size: 0.875rem;
|
||||
}
|
||||
|
||||
.state-badge {
|
||||
font-size: 0.7rem;
|
||||
padding: 0.125rem 0.5rem;
|
||||
border-radius: 9999px;
|
||||
color: white;
|
||||
font-weight: 500;
|
||||
}
|
||||
|
||||
.agent-card-stats {
|
||||
display: flex;
|
||||
gap: 1rem;
|
||||
}
|
||||
|
||||
.stat {
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
}
|
||||
|
||||
.stat-value {
|
||||
font-size: 0.875rem;
|
||||
font-weight: 600;
|
||||
}
|
||||
|
||||
.stat-label {
|
||||
font-size: 0.7rem;
|
||||
color: var(--color-text-muted, #94a3b8);
|
||||
}
|
||||
|
||||
.filters {
|
||||
display: flex;
|
||||
gap: 0.5rem;
|
||||
align-items: center;
|
||||
margin-bottom: 1rem;
|
||||
}
|
||||
|
||||
.filters select {
|
||||
background: var(--color-surface, #1e293b);
|
||||
border: 1px solid var(--color-border, #334155);
|
||||
color: var(--color-text, #e2e8f0);
|
||||
padding: 0.375rem 0.75rem;
|
||||
border-radius: 0.375rem;
|
||||
font-size: 0.875rem;
|
||||
}
|
||||
|
||||
.total-count {
|
||||
margin-left: auto;
|
||||
font-size: 0.8rem;
|
||||
color: var(--color-text-muted, #94a3b8);
|
||||
}
|
||||
|
||||
.loading, .empty {
|
||||
text-align: center;
|
||||
padding: 3rem;
|
||||
color: var(--color-text-muted, #94a3b8);
|
||||
}
|
||||
|
||||
.runs-list {
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 2px;
|
||||
}
|
||||
|
||||
.run-row {
|
||||
background: var(--color-surface, #1e293b);
|
||||
border: 1px solid var(--color-border, #334155);
|
||||
border-radius: 0.375rem;
|
||||
overflow: hidden;
|
||||
}
|
||||
|
||||
.run-row.expanded {
|
||||
border-color: var(--color-primary, #3b82f6);
|
||||
}
|
||||
|
||||
.run-row-main {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
gap: 0.75rem;
|
||||
padding: 0.625rem 0.75rem;
|
||||
width: 100%;
|
||||
background: none;
|
||||
border: none;
|
||||
color: inherit;
|
||||
cursor: pointer;
|
||||
font-size: 0.8125rem;
|
||||
text-align: left;
|
||||
}
|
||||
|
||||
.run-row-main:hover {
|
||||
background: var(--color-surface-hover, #334155);
|
||||
}
|
||||
|
||||
.status-dot {
|
||||
width: 8px;
|
||||
height: 8px;
|
||||
border-radius: 50%;
|
||||
flex-shrink: 0;
|
||||
}
|
||||
|
||||
.run-agent {
|
||||
font-weight: 600;
|
||||
min-width: 140px;
|
||||
}
|
||||
|
||||
.run-trigger {
|
||||
flex: 1;
|
||||
color: var(--color-text-muted, #94a3b8);
|
||||
overflow: hidden;
|
||||
text-overflow: ellipsis;
|
||||
white-space: nowrap;
|
||||
}
|
||||
|
||||
.run-status {
|
||||
min-width: 100px;
|
||||
font-size: 0.75rem;
|
||||
}
|
||||
|
||||
.run-duration {
|
||||
min-width: 60px;
|
||||
text-align: right;
|
||||
font-variant-numeric: tabular-nums;
|
||||
}
|
||||
|
||||
.run-time {
|
||||
min-width: 100px;
|
||||
text-align: right;
|
||||
color: var(--color-text-muted, #94a3b8);
|
||||
font-size: 0.75rem;
|
||||
}
|
||||
|
||||
.expand-arrow {
|
||||
font-size: 0.625rem;
|
||||
color: var(--color-text-muted, #94a3b8);
|
||||
}
|
||||
|
||||
.run-details {
|
||||
padding: 0.75rem 1rem;
|
||||
border-top: 1px solid var(--color-border, #334155);
|
||||
background: var(--color-surface-alt, #0f172a);
|
||||
}
|
||||
|
||||
.detail-row {
|
||||
display: flex;
|
||||
gap: 1rem;
|
||||
padding: 0.25rem 0;
|
||||
font-size: 0.8125rem;
|
||||
}
|
||||
|
||||
.detail-label {
|
||||
color: var(--color-text-muted, #94a3b8);
|
||||
min-width: 120px;
|
||||
}
|
||||
|
||||
.detail-link {
|
||||
color: var(--color-primary, #3b82f6);
|
||||
text-decoration: none;
|
||||
}
|
||||
|
||||
.detail-link:hover {
|
||||
text-decoration: underline;
|
||||
}
|
||||
|
||||
.error-log {
|
||||
margin-top: 0.5rem;
|
||||
}
|
||||
|
||||
.error-log-header {
|
||||
font-weight: 600;
|
||||
font-size: 0.8125rem;
|
||||
margin-bottom: 0.25rem;
|
||||
color: var(--color-error, #ef4444);
|
||||
}
|
||||
|
||||
.error-log pre {
|
||||
background: #0a0a0a;
|
||||
color: #e2e8f0;
|
||||
padding: 0.75rem;
|
||||
border-radius: 0.375rem;
|
||||
font-size: 0.75rem;
|
||||
overflow-x: auto;
|
||||
max-height: 300px;
|
||||
overflow-y: auto;
|
||||
white-space: pre-wrap;
|
||||
word-break: break-word;
|
||||
}
|
||||
|
||||
.retry-btn {
|
||||
margin-top: 0.5rem;
|
||||
padding: 0.375rem 1rem;
|
||||
background: var(--color-primary, #3b82f6);
|
||||
color: white;
|
||||
border: none;
|
||||
border-radius: 0.375rem;
|
||||
cursor: pointer;
|
||||
font-size: 0.8125rem;
|
||||
font-weight: 500;
|
||||
}
|
||||
|
||||
.retry-btn:hover {
|
||||
opacity: 0.9;
|
||||
}
|
||||
</style>
|
||||
Reference in New Issue
Block a user