MVP implementation of the MuSiQue multi-agent benchmark (spec 017): - benchmark/setup.py: downloads musique_v1.0.zip from the canonical Google Drive source (mirrors upstream download_data.sh). Idempotent. - benchmark/curate.py: deterministic selection of 3 4-hop questions from the dev set sharing a US pivot entity; writes benchmark/trio.jsonl. - benchmark/marketplace.py: in-process 016-marketplace stub with post_auction / bid / award / mark_done / query_reputation and a domain-scoped reputation ledger. Designed for mechanical swap to real SynapBus MCP tools. - benchmark/agents.py: HaikuAgent + SonnetAgent, using the official anthropic SDK (no Claude Agent SDK, no subprocesses). Models pinned to claude-haiku-4-5-20251001 and claude-sonnet-4-6. - benchmark/baseline.py: single Sonnet call with all 20 distractors plus chain-of-thought. - benchmark/score.py: SQuAD-style normalized F1 + Pareto verdict (strictly northwest = PASS). - benchmark/run.py: main entry. --mode single-shot, --question, --dry-run. - benchmark/report.py: self-contained HTML with inline SVG scatter plot. - benchmark/trio.jsonl: curated reproducible trio (all three converge on "Treaty of Paris" US territory cession). Verified with benchmark/run.py --dry-run end-to-end; all 8 files py_compile clean. Real-token execution is deferred to the user's main session. Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
257 lines
7.5 KiB
Python
257 lines
7.5 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
Agent pool for the MuSiQue benchmark.
|
|
|
|
Two agents:
|
|
- haiku-agent (claude-haiku-4-5-20251001)
|
|
- sonnet-agent (claude-sonnet-4-6)
|
|
|
|
Each agent exposes:
|
|
- name, model, skill_card
|
|
- bid(task) -> {estimated_tokens, confidence, approach}
|
|
- execute(task, paragraphs) -> {answer, actual_tokens}
|
|
|
|
Design notes:
|
|
- We use the official ``anthropic`` Python SDK directly (NOT the
|
|
Claude Agent SDK). Simpler, no subprocesses, reliable token accounting.
|
|
- ``bid()`` is pure Python — it is a cheap heuristic so the marketplace
|
|
has something to pick from. Real 016 agents would emit a structured
|
|
reply. For MVP, heuristic bids are sufficient to exercise the auction
|
|
primitive.
|
|
- ``execute()`` is the only thing that actually burns tokens.
|
|
- ``--dry-run`` in run.py never calls execute(); it uses stub responses.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import os
|
|
from dataclasses import dataclass
|
|
from typing import Any
|
|
|
|
try:
|
|
import anthropic # type: ignore
|
|
except ImportError: # pragma: no cover
|
|
anthropic = None # type: ignore
|
|
|
|
|
|
HAIKU_MODEL = "claude-haiku-4-5-20251001"
|
|
SONNET_MODEL = "claude-sonnet-4-6"
|
|
|
|
|
|
HAIKU_SKILL_CARD = """\
|
|
# haiku-agent
|
|
|
|
A fast, cheap agent best for single-hop fact lookups and short
|
|
extractive answers. Accepts multi-paragraph context but may miss
|
|
subtle bridging entities on 4-hop questions. Very low cost per call.
|
|
|
|
Domains: factual-lookup, extraction, summarization
|
|
"""
|
|
|
|
SONNET_SKILL_CARD = """\
|
|
# sonnet-agent
|
|
|
|
A deliberate mid-tier agent well-suited to multi-hop reasoning with
|
|
explicit chain-of-thought. Handles 4-hop MuSiQue questions with
|
|
decomposition when the context fits in one prompt. Higher cost per call
|
|
than Haiku but meaningfully better F1 on bridging questions.
|
|
|
|
Domains: multi-hop-qa, decomposition, reasoning
|
|
"""
|
|
|
|
|
|
SYSTEM_PROMPT = """\
|
|
You are a careful question-answering agent working on a MuSiQue
|
|
multi-hop benchmark. You are given a question and a set of numbered
|
|
paragraphs. Only a few of the paragraphs are relevant; the rest are
|
|
distractors.
|
|
|
|
Think step by step and cite the paragraphs you used. Then output a
|
|
final line starting with exactly:
|
|
|
|
ANSWER: <your short final answer>
|
|
|
|
Your final answer must be a short entity or phrase — not a sentence.
|
|
"""
|
|
|
|
|
|
@dataclass
|
|
class BidResult:
|
|
estimated_tokens: int
|
|
confidence: float
|
|
approach: str
|
|
|
|
def to_dict(self) -> dict[str, Any]:
|
|
return {
|
|
"estimated_tokens": self.estimated_tokens,
|
|
"confidence": self.confidence,
|
|
"approach": self.approach,
|
|
}
|
|
|
|
|
|
@dataclass
|
|
class ExecuteResult:
|
|
answer: str
|
|
actual_tokens: int
|
|
raw_text: str = ""
|
|
|
|
def to_dict(self) -> dict[str, Any]:
|
|
return {
|
|
"answer": self.answer,
|
|
"actual_tokens": self.actual_tokens,
|
|
}
|
|
|
|
|
|
class Agent:
|
|
name: str
|
|
model: str
|
|
skill_card: str
|
|
|
|
def __init__(self, name: str, model: str, skill_card: str) -> None:
|
|
self.name = name
|
|
self.model = model
|
|
self.skill_card = skill_card
|
|
|
|
# ---- bidding -----------------------------------------------------------
|
|
|
|
def bid(self, task: dict[str, Any]) -> BidResult:
|
|
raise NotImplementedError
|
|
|
|
# ---- execution ---------------------------------------------------------
|
|
|
|
def execute(
|
|
self,
|
|
task: dict[str, Any],
|
|
paragraphs: list[str],
|
|
*,
|
|
dry_run: bool = False,
|
|
max_budget_tokens: int = 100_000,
|
|
) -> ExecuteResult:
|
|
question = task["question"]
|
|
prompt = self._build_prompt(question, paragraphs)
|
|
|
|
if dry_run:
|
|
stub = (
|
|
"Thinking step by step... [dry-run stub]\n"
|
|
f"ANSWER: [stub answer from {self.name}]"
|
|
)
|
|
# Rough estimate: 1 token ~= 4 characters.
|
|
est = max(256, len(prompt) // 4 + 64)
|
|
return ExecuteResult(
|
|
answer=self._extract_answer(stub),
|
|
actual_tokens=est,
|
|
raw_text=stub,
|
|
)
|
|
|
|
if anthropic is None:
|
|
raise RuntimeError(
|
|
"anthropic SDK not installed — pip install anthropic"
|
|
)
|
|
api_key = os.environ.get("ANTHROPIC_API_KEY")
|
|
if not api_key:
|
|
raise RuntimeError(
|
|
"ANTHROPIC_API_KEY not set. Use --dry-run to stub it out."
|
|
)
|
|
|
|
client = anthropic.Anthropic(api_key=api_key)
|
|
# Cap max_tokens to min(1024, budget/2) so the worst case is tame.
|
|
max_tokens = min(1024, max(128, max_budget_tokens // 2))
|
|
msg = client.messages.create(
|
|
model=self.model,
|
|
max_tokens=max_tokens,
|
|
system=SYSTEM_PROMPT,
|
|
messages=[{"role": "user", "content": prompt}],
|
|
)
|
|
text_parts: list[str] = []
|
|
for block in msg.content:
|
|
t = getattr(block, "text", None)
|
|
if t:
|
|
text_parts.append(t)
|
|
text = "\n".join(text_parts).strip()
|
|
|
|
usage = getattr(msg, "usage", None)
|
|
actual = 0
|
|
if usage is not None:
|
|
actual = (
|
|
getattr(usage, "input_tokens", 0)
|
|
+ getattr(usage, "output_tokens", 0)
|
|
)
|
|
return ExecuteResult(
|
|
answer=self._extract_answer(text),
|
|
actual_tokens=int(actual),
|
|
raw_text=text,
|
|
)
|
|
|
|
# ---- helpers -----------------------------------------------------------
|
|
|
|
def _build_prompt(
|
|
self, question: str, paragraphs: list[str]
|
|
) -> str:
|
|
body = ["Paragraphs:"]
|
|
for i, p in enumerate(paragraphs, start=1):
|
|
body.append(f"[{i}] {p}")
|
|
body.append("")
|
|
body.append(f"Question: {question}")
|
|
body.append("")
|
|
body.append("Think step by step, then output your final ANSWER: line.")
|
|
return "\n".join(body)
|
|
|
|
def _extract_answer(self, text: str) -> str:
|
|
if not text:
|
|
return ""
|
|
for line in reversed(text.splitlines()):
|
|
line = line.strip()
|
|
if line.upper().startswith("ANSWER:"):
|
|
return line.split(":", 1)[1].strip()
|
|
# Fallback: last non-empty line.
|
|
for line in reversed(text.splitlines()):
|
|
line = line.strip()
|
|
if line:
|
|
return line
|
|
return ""
|
|
|
|
|
|
class HaikuAgent(Agent):
|
|
def __init__(self) -> None:
|
|
super().__init__(
|
|
name="haiku-agent",
|
|
model=HAIKU_MODEL,
|
|
skill_card=HAIKU_SKILL_CARD,
|
|
)
|
|
|
|
def bid(self, task: dict[str, Any]) -> BidResult:
|
|
# Cheap, low confidence on multi-hop bridging.
|
|
return BidResult(
|
|
estimated_tokens=4_000,
|
|
confidence=0.45,
|
|
approach=(
|
|
"Extract candidate entities from the paragraphs and "
|
|
"answer directly; may miss 4-hop bridges."
|
|
),
|
|
)
|
|
|
|
|
|
class SonnetAgent(Agent):
|
|
def __init__(self) -> None:
|
|
super().__init__(
|
|
name="sonnet-agent",
|
|
model=SONNET_MODEL,
|
|
skill_card=SONNET_SKILL_CARD,
|
|
)
|
|
|
|
def bid(self, task: dict[str, Any]) -> BidResult:
|
|
# More expensive, higher confidence on multi-hop.
|
|
return BidResult(
|
|
estimated_tokens=12_000,
|
|
confidence=0.80,
|
|
approach=(
|
|
"Decompose the question into sub-questions, resolve each "
|
|
"sub-answer against the paragraphs, then compose the final "
|
|
"bridged answer."
|
|
),
|
|
)
|
|
|
|
|
|
def default_pool() -> list[Agent]:
|
|
return [HaikuAgent(), SonnetAgent()]
|