Files
synapbus/benchmark/agents.py
T
Algis DumbrisandClaude Opus 4.6 02b8548eac feat(017): MuSiQue benchmark harness — marketplace stub, mixed-tier agents, Pareto scoring, HTML report
MVP implementation of the MuSiQue multi-agent benchmark (spec 017):

- benchmark/setup.py: downloads musique_v1.0.zip from the canonical Google
  Drive source (mirrors upstream download_data.sh). Idempotent.
- benchmark/curate.py: deterministic selection of 3 4-hop questions from
  the dev set sharing a US pivot entity; writes benchmark/trio.jsonl.
- benchmark/marketplace.py: in-process 016-marketplace stub with
  post_auction / bid / award / mark_done / query_reputation and a
  domain-scoped reputation ledger. Designed for mechanical swap to real
  SynapBus MCP tools.
- benchmark/agents.py: HaikuAgent + SonnetAgent, using the official
  anthropic SDK (no Claude Agent SDK, no subprocesses). Models pinned
  to claude-haiku-4-5-20251001 and claude-sonnet-4-6.
- benchmark/baseline.py: single Sonnet call with all 20 distractors
  plus chain-of-thought.
- benchmark/score.py: SQuAD-style normalized F1 + Pareto verdict
  (strictly northwest = PASS).
- benchmark/run.py: main entry. --mode single-shot, --question, --dry-run.
- benchmark/report.py: self-contained HTML with inline SVG scatter plot.
- benchmark/trio.jsonl: curated reproducible trio (all three converge on
  "Treaty of Paris" US territory cession).

Verified with benchmark/run.py --dry-run end-to-end; all 8 files
py_compile clean. Real-token execution is deferred to the user's main
session.

Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
2026-04-11 15:12:19 +03:00

257 lines
7.5 KiB
Python

#!/usr/bin/env python3
"""
Agent pool for the MuSiQue benchmark.
Two agents:
- haiku-agent (claude-haiku-4-5-20251001)
- sonnet-agent (claude-sonnet-4-6)
Each agent exposes:
- name, model, skill_card
- bid(task) -> {estimated_tokens, confidence, approach}
- execute(task, paragraphs) -> {answer, actual_tokens}
Design notes:
- We use the official ``anthropic`` Python SDK directly (NOT the
Claude Agent SDK). Simpler, no subprocesses, reliable token accounting.
- ``bid()`` is pure Python — it is a cheap heuristic so the marketplace
has something to pick from. Real 016 agents would emit a structured
reply. For MVP, heuristic bids are sufficient to exercise the auction
primitive.
- ``execute()`` is the only thing that actually burns tokens.
- ``--dry-run`` in run.py never calls execute(); it uses stub responses.
"""
from __future__ import annotations
import os
from dataclasses import dataclass
from typing import Any
try:
import anthropic # type: ignore
except ImportError: # pragma: no cover
anthropic = None # type: ignore
HAIKU_MODEL = "claude-haiku-4-5-20251001"
SONNET_MODEL = "claude-sonnet-4-6"
HAIKU_SKILL_CARD = """\
# haiku-agent
A fast, cheap agent best for single-hop fact lookups and short
extractive answers. Accepts multi-paragraph context but may miss
subtle bridging entities on 4-hop questions. Very low cost per call.
Domains: factual-lookup, extraction, summarization
"""
SONNET_SKILL_CARD = """\
# sonnet-agent
A deliberate mid-tier agent well-suited to multi-hop reasoning with
explicit chain-of-thought. Handles 4-hop MuSiQue questions with
decomposition when the context fits in one prompt. Higher cost per call
than Haiku but meaningfully better F1 on bridging questions.
Domains: multi-hop-qa, decomposition, reasoning
"""
SYSTEM_PROMPT = """\
You are a careful question-answering agent working on a MuSiQue
multi-hop benchmark. You are given a question and a set of numbered
paragraphs. Only a few of the paragraphs are relevant; the rest are
distractors.
Think step by step and cite the paragraphs you used. Then output a
final line starting with exactly:
ANSWER: <your short final answer>
Your final answer must be a short entity or phrase — not a sentence.
"""
@dataclass
class BidResult:
estimated_tokens: int
confidence: float
approach: str
def to_dict(self) -> dict[str, Any]:
return {
"estimated_tokens": self.estimated_tokens,
"confidence": self.confidence,
"approach": self.approach,
}
@dataclass
class ExecuteResult:
answer: str
actual_tokens: int
raw_text: str = ""
def to_dict(self) -> dict[str, Any]:
return {
"answer": self.answer,
"actual_tokens": self.actual_tokens,
}
class Agent:
name: str
model: str
skill_card: str
def __init__(self, name: str, model: str, skill_card: str) -> None:
self.name = name
self.model = model
self.skill_card = skill_card
# ---- bidding -----------------------------------------------------------
def bid(self, task: dict[str, Any]) -> BidResult:
raise NotImplementedError
# ---- execution ---------------------------------------------------------
def execute(
self,
task: dict[str, Any],
paragraphs: list[str],
*,
dry_run: bool = False,
max_budget_tokens: int = 100_000,
) -> ExecuteResult:
question = task["question"]
prompt = self._build_prompt(question, paragraphs)
if dry_run:
stub = (
"Thinking step by step... [dry-run stub]\n"
f"ANSWER: [stub answer from {self.name}]"
)
# Rough estimate: 1 token ~= 4 characters.
est = max(256, len(prompt) // 4 + 64)
return ExecuteResult(
answer=self._extract_answer(stub),
actual_tokens=est,
raw_text=stub,
)
if anthropic is None:
raise RuntimeError(
"anthropic SDK not installed — pip install anthropic"
)
api_key = os.environ.get("ANTHROPIC_API_KEY")
if not api_key:
raise RuntimeError(
"ANTHROPIC_API_KEY not set. Use --dry-run to stub it out."
)
client = anthropic.Anthropic(api_key=api_key)
# Cap max_tokens to min(1024, budget/2) so the worst case is tame.
max_tokens = min(1024, max(128, max_budget_tokens // 2))
msg = client.messages.create(
model=self.model,
max_tokens=max_tokens,
system=SYSTEM_PROMPT,
messages=[{"role": "user", "content": prompt}],
)
text_parts: list[str] = []
for block in msg.content:
t = getattr(block, "text", None)
if t:
text_parts.append(t)
text = "\n".join(text_parts).strip()
usage = getattr(msg, "usage", None)
actual = 0
if usage is not None:
actual = (
getattr(usage, "input_tokens", 0)
+ getattr(usage, "output_tokens", 0)
)
return ExecuteResult(
answer=self._extract_answer(text),
actual_tokens=int(actual),
raw_text=text,
)
# ---- helpers -----------------------------------------------------------
def _build_prompt(
self, question: str, paragraphs: list[str]
) -> str:
body = ["Paragraphs:"]
for i, p in enumerate(paragraphs, start=1):
body.append(f"[{i}] {p}")
body.append("")
body.append(f"Question: {question}")
body.append("")
body.append("Think step by step, then output your final ANSWER: line.")
return "\n".join(body)
def _extract_answer(self, text: str) -> str:
if not text:
return ""
for line in reversed(text.splitlines()):
line = line.strip()
if line.upper().startswith("ANSWER:"):
return line.split(":", 1)[1].strip()
# Fallback: last non-empty line.
for line in reversed(text.splitlines()):
line = line.strip()
if line:
return line
return ""
class HaikuAgent(Agent):
def __init__(self) -> None:
super().__init__(
name="haiku-agent",
model=HAIKU_MODEL,
skill_card=HAIKU_SKILL_CARD,
)
def bid(self, task: dict[str, Any]) -> BidResult:
# Cheap, low confidence on multi-hop bridging.
return BidResult(
estimated_tokens=4_000,
confidence=0.45,
approach=(
"Extract candidate entities from the paragraphs and "
"answer directly; may miss 4-hop bridges."
),
)
class SonnetAgent(Agent):
def __init__(self) -> None:
super().__init__(
name="sonnet-agent",
model=SONNET_MODEL,
skill_card=SONNET_SKILL_CARD,
)
def bid(self, task: dict[str, Any]) -> BidResult:
# More expensive, higher confidence on multi-hop.
return BidResult(
estimated_tokens=12_000,
confidence=0.80,
approach=(
"Decompose the question into sub-questions, resolve each "
"sub-answer against the paragraphs, then compose the final "
"bridged answer."
),
)
def default_pool() -> list[Agent]:
return [HaikuAgent(), SonnetAgent()]