merge: 017-musique-benchmark MVP (Python harness + trio + Pareto report)
This commit is contained in:
@@ -41,6 +41,12 @@ Thumbs.db
|
||||
__pycache__/
|
||||
*.pyc
|
||||
|
||||
# Benchmark (017) — large datasets, per-run outputs, local venvs
|
||||
benchmark/data/
|
||||
benchmark/results/
|
||||
.venv-bench/
|
||||
.venv/
|
||||
|
||||
# Debug
|
||||
__debug_bin*
|
||||
.claude/worktrees/
|
||||
|
||||
@@ -0,0 +1,256 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Agent pool for the MuSiQue benchmark.
|
||||
|
||||
Two agents:
|
||||
- haiku-agent (claude-haiku-4-5-20251001)
|
||||
- sonnet-agent (claude-sonnet-4-6)
|
||||
|
||||
Each agent exposes:
|
||||
- name, model, skill_card
|
||||
- bid(task) -> {estimated_tokens, confidence, approach}
|
||||
- execute(task, paragraphs) -> {answer, actual_tokens}
|
||||
|
||||
Design notes:
|
||||
- We use the official ``anthropic`` Python SDK directly (NOT the
|
||||
Claude Agent SDK). Simpler, no subprocesses, reliable token accounting.
|
||||
- ``bid()`` is pure Python — it is a cheap heuristic so the marketplace
|
||||
has something to pick from. Real 016 agents would emit a structured
|
||||
reply. For MVP, heuristic bids are sufficient to exercise the auction
|
||||
primitive.
|
||||
- ``execute()`` is the only thing that actually burns tokens.
|
||||
- ``--dry-run`` in run.py never calls execute(); it uses stub responses.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from dataclasses import dataclass
|
||||
from typing import Any
|
||||
|
||||
try:
|
||||
import anthropic # type: ignore
|
||||
except ImportError: # pragma: no cover
|
||||
anthropic = None # type: ignore
|
||||
|
||||
|
||||
HAIKU_MODEL = "claude-haiku-4-5-20251001"
|
||||
SONNET_MODEL = "claude-sonnet-4-6"
|
||||
|
||||
|
||||
HAIKU_SKILL_CARD = """\
|
||||
# haiku-agent
|
||||
|
||||
A fast, cheap agent best for single-hop fact lookups and short
|
||||
extractive answers. Accepts multi-paragraph context but may miss
|
||||
subtle bridging entities on 4-hop questions. Very low cost per call.
|
||||
|
||||
Domains: factual-lookup, extraction, summarization
|
||||
"""
|
||||
|
||||
SONNET_SKILL_CARD = """\
|
||||
# sonnet-agent
|
||||
|
||||
A deliberate mid-tier agent well-suited to multi-hop reasoning with
|
||||
explicit chain-of-thought. Handles 4-hop MuSiQue questions with
|
||||
decomposition when the context fits in one prompt. Higher cost per call
|
||||
than Haiku but meaningfully better F1 on bridging questions.
|
||||
|
||||
Domains: multi-hop-qa, decomposition, reasoning
|
||||
"""
|
||||
|
||||
|
||||
SYSTEM_PROMPT = """\
|
||||
You are a careful question-answering agent working on a MuSiQue
|
||||
multi-hop benchmark. You are given a question and a set of numbered
|
||||
paragraphs. Only a few of the paragraphs are relevant; the rest are
|
||||
distractors.
|
||||
|
||||
Think step by step and cite the paragraphs you used. Then output a
|
||||
final line starting with exactly:
|
||||
|
||||
ANSWER: <your short final answer>
|
||||
|
||||
Your final answer must be a short entity or phrase — not a sentence.
|
||||
"""
|
||||
|
||||
|
||||
@dataclass
|
||||
class BidResult:
|
||||
estimated_tokens: int
|
||||
confidence: float
|
||||
approach: str
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"estimated_tokens": self.estimated_tokens,
|
||||
"confidence": self.confidence,
|
||||
"approach": self.approach,
|
||||
}
|
||||
|
||||
|
||||
@dataclass
|
||||
class ExecuteResult:
|
||||
answer: str
|
||||
actual_tokens: int
|
||||
raw_text: str = ""
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"answer": self.answer,
|
||||
"actual_tokens": self.actual_tokens,
|
||||
}
|
||||
|
||||
|
||||
class Agent:
|
||||
name: str
|
||||
model: str
|
||||
skill_card: str
|
||||
|
||||
def __init__(self, name: str, model: str, skill_card: str) -> None:
|
||||
self.name = name
|
||||
self.model = model
|
||||
self.skill_card = skill_card
|
||||
|
||||
# ---- bidding -----------------------------------------------------------
|
||||
|
||||
def bid(self, task: dict[str, Any]) -> BidResult:
|
||||
raise NotImplementedError
|
||||
|
||||
# ---- execution ---------------------------------------------------------
|
||||
|
||||
def execute(
|
||||
self,
|
||||
task: dict[str, Any],
|
||||
paragraphs: list[str],
|
||||
*,
|
||||
dry_run: bool = False,
|
||||
max_budget_tokens: int = 100_000,
|
||||
) -> ExecuteResult:
|
||||
question = task["question"]
|
||||
prompt = self._build_prompt(question, paragraphs)
|
||||
|
||||
if dry_run:
|
||||
stub = (
|
||||
"Thinking step by step... [dry-run stub]\n"
|
||||
f"ANSWER: [stub answer from {self.name}]"
|
||||
)
|
||||
# Rough estimate: 1 token ~= 4 characters.
|
||||
est = max(256, len(prompt) // 4 + 64)
|
||||
return ExecuteResult(
|
||||
answer=self._extract_answer(stub),
|
||||
actual_tokens=est,
|
||||
raw_text=stub,
|
||||
)
|
||||
|
||||
if anthropic is None:
|
||||
raise RuntimeError(
|
||||
"anthropic SDK not installed — pip install anthropic"
|
||||
)
|
||||
api_key = os.environ.get("ANTHROPIC_API_KEY")
|
||||
if not api_key:
|
||||
raise RuntimeError(
|
||||
"ANTHROPIC_API_KEY not set. Use --dry-run to stub it out."
|
||||
)
|
||||
|
||||
client = anthropic.Anthropic(api_key=api_key)
|
||||
# Cap max_tokens to min(1024, budget/2) so the worst case is tame.
|
||||
max_tokens = min(1024, max(128, max_budget_tokens // 2))
|
||||
msg = client.messages.create(
|
||||
model=self.model,
|
||||
max_tokens=max_tokens,
|
||||
system=SYSTEM_PROMPT,
|
||||
messages=[{"role": "user", "content": prompt}],
|
||||
)
|
||||
text_parts: list[str] = []
|
||||
for block in msg.content:
|
||||
t = getattr(block, "text", None)
|
||||
if t:
|
||||
text_parts.append(t)
|
||||
text = "\n".join(text_parts).strip()
|
||||
|
||||
usage = getattr(msg, "usage", None)
|
||||
actual = 0
|
||||
if usage is not None:
|
||||
actual = (
|
||||
getattr(usage, "input_tokens", 0)
|
||||
+ getattr(usage, "output_tokens", 0)
|
||||
)
|
||||
return ExecuteResult(
|
||||
answer=self._extract_answer(text),
|
||||
actual_tokens=int(actual),
|
||||
raw_text=text,
|
||||
)
|
||||
|
||||
# ---- helpers -----------------------------------------------------------
|
||||
|
||||
def _build_prompt(
|
||||
self, question: str, paragraphs: list[str]
|
||||
) -> str:
|
||||
body = ["Paragraphs:"]
|
||||
for i, p in enumerate(paragraphs, start=1):
|
||||
body.append(f"[{i}] {p}")
|
||||
body.append("")
|
||||
body.append(f"Question: {question}")
|
||||
body.append("")
|
||||
body.append("Think step by step, then output your final ANSWER: line.")
|
||||
return "\n".join(body)
|
||||
|
||||
def _extract_answer(self, text: str) -> str:
|
||||
if not text:
|
||||
return ""
|
||||
for line in reversed(text.splitlines()):
|
||||
line = line.strip()
|
||||
if line.upper().startswith("ANSWER:"):
|
||||
return line.split(":", 1)[1].strip()
|
||||
# Fallback: last non-empty line.
|
||||
for line in reversed(text.splitlines()):
|
||||
line = line.strip()
|
||||
if line:
|
||||
return line
|
||||
return ""
|
||||
|
||||
|
||||
class HaikuAgent(Agent):
|
||||
def __init__(self) -> None:
|
||||
super().__init__(
|
||||
name="haiku-agent",
|
||||
model=HAIKU_MODEL,
|
||||
skill_card=HAIKU_SKILL_CARD,
|
||||
)
|
||||
|
||||
def bid(self, task: dict[str, Any]) -> BidResult:
|
||||
# Cheap, low confidence on multi-hop bridging.
|
||||
return BidResult(
|
||||
estimated_tokens=4_000,
|
||||
confidence=0.45,
|
||||
approach=(
|
||||
"Extract candidate entities from the paragraphs and "
|
||||
"answer directly; may miss 4-hop bridges."
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
class SonnetAgent(Agent):
|
||||
def __init__(self) -> None:
|
||||
super().__init__(
|
||||
name="sonnet-agent",
|
||||
model=SONNET_MODEL,
|
||||
skill_card=SONNET_SKILL_CARD,
|
||||
)
|
||||
|
||||
def bid(self, task: dict[str, Any]) -> BidResult:
|
||||
# More expensive, higher confidence on multi-hop.
|
||||
return BidResult(
|
||||
estimated_tokens=12_000,
|
||||
confidence=0.80,
|
||||
approach=(
|
||||
"Decompose the question into sub-questions, resolve each "
|
||||
"sub-answer against the paragraphs, then compose the final "
|
||||
"bridged answer."
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def default_pool() -> list[Agent]:
|
||||
return [HaikuAgent(), SonnetAgent()]
|
||||
@@ -0,0 +1,116 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Single-agent baseline: one Anthropic API call to claude-sonnet-4-6 with
|
||||
the question and all 20 distractor paragraphs plus chain-of-thought
|
||||
instructions. No decomposition, no marketplace, no tools.
|
||||
|
||||
Returns {"answer": str, "tokens": int, "raw_text": str}.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from typing import Any
|
||||
|
||||
try:
|
||||
import anthropic # type: ignore
|
||||
except ImportError: # pragma: no cover
|
||||
anthropic = None # type: ignore
|
||||
|
||||
|
||||
BASELINE_MODEL = "claude-sonnet-4-6"
|
||||
|
||||
BASELINE_SYSTEM = """\
|
||||
You are a careful multi-hop QA system. Given a question and a set of
|
||||
numbered paragraphs (some irrelevant distractors), think step by step
|
||||
and answer.
|
||||
|
||||
Output your reasoning first, then on a final line:
|
||||
|
||||
ANSWER: <short final answer>
|
||||
"""
|
||||
|
||||
|
||||
def _build_prompt(question: str, paragraphs: list[str]) -> str:
|
||||
parts = ["Paragraphs:"]
|
||||
for i, p in enumerate(paragraphs, start=1):
|
||||
parts.append(f"[{i}] {p}")
|
||||
parts.append("")
|
||||
parts.append(f"Question: {question}")
|
||||
parts.append("")
|
||||
parts.append(
|
||||
"Work through the reasoning step by step, then give your "
|
||||
"final ANSWER: line."
|
||||
)
|
||||
return "\n".join(parts)
|
||||
|
||||
|
||||
def _extract_answer(text: str) -> str:
|
||||
if not text:
|
||||
return ""
|
||||
for line in reversed(text.splitlines()):
|
||||
line = line.strip()
|
||||
if line.upper().startswith("ANSWER:"):
|
||||
return line.split(":", 1)[1].strip()
|
||||
for line in reversed(text.splitlines()):
|
||||
line = line.strip()
|
||||
if line:
|
||||
return line
|
||||
return ""
|
||||
|
||||
|
||||
def run_baseline(
|
||||
question: str,
|
||||
paragraphs: list[str],
|
||||
*,
|
||||
dry_run: bool = False,
|
||||
max_output_tokens: int = 1024,
|
||||
) -> dict[str, Any]:
|
||||
prompt = _build_prompt(question, paragraphs)
|
||||
|
||||
if dry_run:
|
||||
stub = (
|
||||
"Step 1: scanning paragraphs... [dry-run stub]\n"
|
||||
"Step 2: picking the most likely entity...\n"
|
||||
"ANSWER: [stub baseline answer]"
|
||||
)
|
||||
est = max(512, len(prompt) // 4 + 128)
|
||||
return {
|
||||
"answer": _extract_answer(stub),
|
||||
"tokens": est,
|
||||
"raw_text": stub,
|
||||
"model": BASELINE_MODEL,
|
||||
}
|
||||
|
||||
if anthropic is None:
|
||||
raise RuntimeError("anthropic SDK not installed")
|
||||
api_key = os.environ.get("ANTHROPIC_API_KEY")
|
||||
if not api_key:
|
||||
raise RuntimeError("ANTHROPIC_API_KEY not set")
|
||||
|
||||
client = anthropic.Anthropic(api_key=api_key)
|
||||
msg = client.messages.create(
|
||||
model=BASELINE_MODEL,
|
||||
max_tokens=max_output_tokens,
|
||||
system=BASELINE_SYSTEM,
|
||||
messages=[{"role": "user", "content": prompt}],
|
||||
)
|
||||
text_parts: list[str] = []
|
||||
for block in msg.content:
|
||||
t = getattr(block, "text", None)
|
||||
if t:
|
||||
text_parts.append(t)
|
||||
text = "\n".join(text_parts).strip()
|
||||
usage = getattr(msg, "usage", None)
|
||||
tokens = 0
|
||||
if usage is not None:
|
||||
tokens = (
|
||||
getattr(usage, "input_tokens", 0)
|
||||
+ getattr(usage, "output_tokens", 0)
|
||||
)
|
||||
return {
|
||||
"answer": _extract_answer(text),
|
||||
"tokens": int(tokens),
|
||||
"raw_text": text,
|
||||
"model": BASELINE_MODEL,
|
||||
}
|
||||
@@ -0,0 +1,169 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Curate a deterministic trio of MuSiQue 4-hop questions that share a
|
||||
pivot entity. For MVP we pivot on the United States.
|
||||
|
||||
Input: benchmark/data/musique_ans_v1.0_dev.jsonl
|
||||
Output: benchmark/trio.jsonl
|
||||
|
||||
Each output record:
|
||||
{
|
||||
"id": str,
|
||||
"question": str,
|
||||
"answer": str,
|
||||
"decomposition": [{"question": str, "answer": str}, ...],
|
||||
"paragraphs": [str, ...] # up to 20 distractor snippets
|
||||
}
|
||||
|
||||
MuSiQue dev records typically look like::
|
||||
|
||||
{
|
||||
"id": "4hop1__...",
|
||||
"question": "...",
|
||||
"question_decomposition": [
|
||||
{"id": N, "question": "...", "answer": "...",
|
||||
"paragraph_support_idx": int},
|
||||
...
|
||||
],
|
||||
"answer": "...",
|
||||
"answer_aliases": [...],
|
||||
"paragraphs": [
|
||||
{"idx": int, "title": "...", "paragraph_text": "...",
|
||||
"is_supporting": bool},
|
||||
...
|
||||
]
|
||||
}
|
||||
|
||||
The curation rule is deterministic (fixed input ordering; first 3 matches).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
DATA_FILE = Path(__file__).resolve().parent / "data" / "musique_ans_v1.0_dev.jsonl"
|
||||
OUT_FILE = Path(__file__).resolve().parent / "trio.jsonl"
|
||||
|
||||
PIVOT_TOKENS = ("united states", "u.s.", " us ", "america", "american")
|
||||
N_QUESTIONS = 3
|
||||
MAX_PARAGRAPHS = 20
|
||||
|
||||
|
||||
def _normalized(s: str) -> str:
|
||||
return f" {s.lower()} "
|
||||
|
||||
|
||||
def _mentions_pivot(record: dict) -> bool:
|
||||
blob_parts = [record.get("question", ""), record.get("answer", "")]
|
||||
for sub in record.get("question_decomposition", []) or []:
|
||||
blob_parts.append(sub.get("question", ""))
|
||||
blob_parts.append(sub.get("answer", ""))
|
||||
blob = _normalized(" ".join(str(x) for x in blob_parts if x))
|
||||
return any(tok in blob for tok in PIVOT_TOKENS)
|
||||
|
||||
|
||||
def _is_4hop(record: dict) -> bool:
|
||||
rid = record.get("id", "")
|
||||
if isinstance(rid, str) and rid.startswith("4hop"):
|
||||
return True
|
||||
# Fall back: count decomposition hops.
|
||||
decomp = record.get("question_decomposition") or []
|
||||
return len(decomp) == 4
|
||||
|
||||
|
||||
def _trim_paragraphs(record: dict, limit: int) -> list[str]:
|
||||
out: list[str] = []
|
||||
for p in record.get("paragraphs", []) or []:
|
||||
title = (p.get("title") or "").strip()
|
||||
text = (p.get("paragraph_text") or "").strip()
|
||||
if not text:
|
||||
continue
|
||||
snippet = f"[{title}] {text}" if title else text
|
||||
out.append(snippet)
|
||||
if len(out) >= limit:
|
||||
break
|
||||
return out
|
||||
|
||||
|
||||
def _simplify_decomp(record: dict) -> list[dict]:
|
||||
out = []
|
||||
for sub in record.get("question_decomposition", []) or []:
|
||||
out.append(
|
||||
{
|
||||
"question": sub.get("question", ""),
|
||||
"answer": sub.get("answer", ""),
|
||||
}
|
||||
)
|
||||
return out
|
||||
|
||||
|
||||
def curate() -> int:
|
||||
if not DATA_FILE.exists():
|
||||
print(
|
||||
f"[curate] ERROR: {DATA_FILE} not found. Run setup.py first.",
|
||||
file=sys.stderr,
|
||||
)
|
||||
return 2
|
||||
|
||||
selected: list[dict] = []
|
||||
total_scanned = 0
|
||||
total_4hop = 0
|
||||
total_pivot = 0
|
||||
|
||||
with open(DATA_FILE, "r", encoding="utf-8") as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
total_scanned += 1
|
||||
try:
|
||||
rec = json.loads(line)
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
if not _is_4hop(rec):
|
||||
continue
|
||||
total_4hop += 1
|
||||
if not _mentions_pivot(rec):
|
||||
continue
|
||||
total_pivot += 1
|
||||
|
||||
trio_record = {
|
||||
"id": rec.get("id", f"q{len(selected)+1}"),
|
||||
"question": rec.get("question", ""),
|
||||
"answer": rec.get("answer", ""),
|
||||
"answer_aliases": rec.get("answer_aliases", []),
|
||||
"decomposition": _simplify_decomp(rec),
|
||||
"paragraphs": _trim_paragraphs(rec, MAX_PARAGRAPHS),
|
||||
}
|
||||
selected.append(trio_record)
|
||||
if len(selected) >= N_QUESTIONS:
|
||||
break
|
||||
|
||||
print(
|
||||
f"[curate] scanned={total_scanned} 4hop={total_4hop} "
|
||||
f"pivot-matches={total_pivot} kept={len(selected)}"
|
||||
)
|
||||
|
||||
if len(selected) < N_QUESTIONS:
|
||||
print(
|
||||
f"[curate] ERROR: wanted {N_QUESTIONS} questions, "
|
||||
f"found {len(selected)}",
|
||||
file=sys.stderr,
|
||||
)
|
||||
return 3
|
||||
|
||||
OUT_FILE.parent.mkdir(parents=True, exist_ok=True)
|
||||
with open(OUT_FILE, "w", encoding="utf-8") as f:
|
||||
for i, rec in enumerate(selected, start=1):
|
||||
# Attach a stable short id q1/q2/q3 in addition to MuSiQue's id.
|
||||
rec["short_id"] = f"q{i}"
|
||||
f.write(json.dumps(rec, ensure_ascii=False) + "\n")
|
||||
|
||||
print(f"[curate] wrote {OUT_FILE} ({len(selected)} records)")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(curate())
|
||||
@@ -0,0 +1,192 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
In-process stub of the 016-agent-marketplace primitives.
|
||||
|
||||
*** IMPORTANT ***
|
||||
This module is an in-process stand-in for the SynapBus-hosted 016
|
||||
marketplace. The follow-up deliverable after this MVP is to replace the
|
||||
bodies of these functions with calls to the real SynapBus MCP tools
|
||||
(``post_auction``, ``bid``, ``award``, ``mark_done``, and
|
||||
``query_reputation``) once 016 lands. The public API here deliberately
|
||||
mirrors those tool names so the swap is mechanical.
|
||||
|
||||
Scope for MVP:
|
||||
- In-memory auctions, bids, awards, and done records
|
||||
- Domain-scoped reputation ledger stored in a dict
|
||||
- No persistence, no concurrency — single process, single thread
|
||||
- No schema enforcement beyond a couple of shape checks
|
||||
|
||||
The harness (``run.py``) holds a single ``Marketplace`` instance.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import itertools
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any
|
||||
|
||||
|
||||
@dataclass
|
||||
class Auction:
|
||||
auction_id: str
|
||||
task: dict[str, Any]
|
||||
domain: str
|
||||
max_budget_tokens: int
|
||||
posted_at: float
|
||||
bids: list[dict[str, Any]] = field(default_factory=list)
|
||||
awarded_to: str | None = None
|
||||
result: dict[str, Any] | None = None
|
||||
|
||||
|
||||
@dataclass
|
||||
class ReputationEntry:
|
||||
agent: str
|
||||
domain: str
|
||||
runs: int = 0
|
||||
correct: int = 0
|
||||
tokens_spent: int = 0
|
||||
|
||||
def score(self) -> float:
|
||||
if self.runs == 0:
|
||||
return 0.5 # prior
|
||||
quality = self.correct / self.runs
|
||||
avg_tokens = self.tokens_spent / self.runs
|
||||
# Arbitrary: quality dominates, tokens slightly penalize.
|
||||
return quality - min(avg_tokens / 200_000.0, 0.3)
|
||||
|
||||
|
||||
class Marketplace:
|
||||
"""In-process marketplace stub."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self._auctions: dict[str, Auction] = {}
|
||||
self._reputation: dict[tuple[str, str], ReputationEntry] = {}
|
||||
self._counter = itertools.count(1)
|
||||
|
||||
# ---- auction lifecycle -------------------------------------------------
|
||||
|
||||
def post_auction(
|
||||
self,
|
||||
task: dict[str, Any],
|
||||
domain: str,
|
||||
max_budget_tokens: int,
|
||||
) -> str:
|
||||
auction_id = f"auction-{next(self._counter)}"
|
||||
self._auctions[auction_id] = Auction(
|
||||
auction_id=auction_id,
|
||||
task=dict(task),
|
||||
domain=domain,
|
||||
max_budget_tokens=max_budget_tokens,
|
||||
posted_at=time.time(),
|
||||
)
|
||||
return auction_id
|
||||
|
||||
def bid(
|
||||
self,
|
||||
auction_id: str,
|
||||
agent: str,
|
||||
estimated_tokens: int,
|
||||
confidence: float,
|
||||
approach: str,
|
||||
) -> None:
|
||||
auction = self._auctions[auction_id]
|
||||
if auction.awarded_to is not None:
|
||||
raise RuntimeError(f"auction {auction_id} already awarded")
|
||||
auction.bids.append(
|
||||
{
|
||||
"agent": agent,
|
||||
"estimated_tokens": int(estimated_tokens),
|
||||
"confidence": float(confidence),
|
||||
"approach": approach,
|
||||
"submitted_at": time.time(),
|
||||
}
|
||||
)
|
||||
|
||||
def list_bids(self, auction_id: str) -> list[dict[str, Any]]:
|
||||
return list(self._auctions[auction_id].bids)
|
||||
|
||||
def score_bid(self, auction_id: str, bid: dict[str, Any]) -> float:
|
||||
"""
|
||||
Lower is better (we're minimizing tokens per unit confidence),
|
||||
but we add a reputation adjustment that rewards agents with a
|
||||
track record in this domain.
|
||||
"""
|
||||
auction = self._auctions[auction_id]
|
||||
rep = self._reputation.get((bid["agent"], auction.domain))
|
||||
rep_score = rep.score() if rep else 0.5
|
||||
conf = max(bid["confidence"], 1e-3)
|
||||
# Cost per confidence, lightly discounted by reputation.
|
||||
raw = bid["estimated_tokens"] / conf
|
||||
return raw * (1.15 - 0.3 * rep_score)
|
||||
|
||||
def award(self, auction_id: str) -> dict[str, Any]:
|
||||
auction = self._auctions[auction_id]
|
||||
if not auction.bids:
|
||||
raise RuntimeError(f"auction {auction_id} has no bids")
|
||||
if auction.awarded_to is not None:
|
||||
raise RuntimeError(f"auction {auction_id} already awarded")
|
||||
best = min(
|
||||
auction.bids,
|
||||
key=lambda b: self.score_bid(auction_id, b),
|
||||
)
|
||||
auction.awarded_to = best["agent"]
|
||||
return best
|
||||
|
||||
def mark_done(
|
||||
self,
|
||||
auction_id: str,
|
||||
answer: str,
|
||||
actual_tokens: int,
|
||||
correct: bool,
|
||||
) -> None:
|
||||
auction = self._auctions[auction_id]
|
||||
if auction.awarded_to is None:
|
||||
raise RuntimeError(f"auction {auction_id} not awarded yet")
|
||||
auction.result = {
|
||||
"answer": answer,
|
||||
"actual_tokens": int(actual_tokens),
|
||||
"correct": bool(correct),
|
||||
}
|
||||
key = (auction.awarded_to, auction.domain)
|
||||
entry = self._reputation.get(key) or ReputationEntry(
|
||||
agent=auction.awarded_to, domain=auction.domain
|
||||
)
|
||||
entry.runs += 1
|
||||
entry.tokens_spent += int(actual_tokens)
|
||||
if correct:
|
||||
entry.correct += 1
|
||||
self._reputation[key] = entry
|
||||
|
||||
# ---- reputation --------------------------------------------------------
|
||||
|
||||
def query_reputation(
|
||||
self, agent: str, domain: str
|
||||
) -> dict[str, Any]:
|
||||
rep = self._reputation.get((agent, domain))
|
||||
if rep is None:
|
||||
return {
|
||||
"agent": agent,
|
||||
"domain": domain,
|
||||
"runs": 0,
|
||||
"correct": 0,
|
||||
"tokens_spent": 0,
|
||||
"score": 0.5,
|
||||
}
|
||||
return {
|
||||
"agent": rep.agent,
|
||||
"domain": rep.domain,
|
||||
"runs": rep.runs,
|
||||
"correct": rep.correct,
|
||||
"tokens_spent": rep.tokens_spent,
|
||||
"score": rep.score(),
|
||||
}
|
||||
|
||||
def all_reputation(self) -> list[dict[str, Any]]:
|
||||
return [
|
||||
self.query_reputation(rep.agent, rep.domain)
|
||||
for rep in self._reputation.values()
|
||||
]
|
||||
|
||||
def auction(self, auction_id: str) -> Auction:
|
||||
return self._auctions[auction_id]
|
||||
@@ -0,0 +1,324 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Self-contained HTML report generator.
|
||||
|
||||
Renders a single HTML file with inline styles and an inline SVG scatter
|
||||
plot. No external assets, no CDN calls, nothing to fetch. Safe to open
|
||||
directly in a browser.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import html
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
def _esc(s: Any) -> str:
|
||||
return html.escape(str(s if s is not None else ""))
|
||||
|
||||
|
||||
def _scatter_svg(
|
||||
market_tokens: int,
|
||||
market_f1: float,
|
||||
baseline_tokens: int,
|
||||
baseline_f1: float,
|
||||
*,
|
||||
width: int = 520,
|
||||
height: int = 320,
|
||||
) -> str:
|
||||
pad_l, pad_r, pad_t, pad_b = 70, 30, 30, 50
|
||||
plot_w = width - pad_l - pad_r
|
||||
plot_h = height - pad_t - pad_b
|
||||
|
||||
max_tokens = max(market_tokens, baseline_tokens, 1)
|
||||
# Give a little headroom so points aren't on the axis.
|
||||
max_tokens_axis = max_tokens * 1.15
|
||||
min_tokens_axis = 0
|
||||
|
||||
def sx(tokens: float) -> float:
|
||||
frac = (tokens - min_tokens_axis) / max(
|
||||
max_tokens_axis - min_tokens_axis, 1
|
||||
)
|
||||
return pad_l + frac * plot_w
|
||||
|
||||
def sy(f1: float) -> float:
|
||||
# y=0 at top of plot, y=1 at bottom -> invert
|
||||
return pad_t + (1.0 - max(0.0, min(1.0, f1))) * plot_h
|
||||
|
||||
axis_color = "#555"
|
||||
grid_color = "#eee"
|
||||
market_color = "#2563eb"
|
||||
baseline_color = "#dc2626"
|
||||
|
||||
parts: list[str] = []
|
||||
parts.append(
|
||||
f'<svg xmlns="http://www.w3.org/2000/svg" width="{width}" '
|
||||
f'height="{height}" viewBox="0 0 {width} {height}" '
|
||||
f'role="img" aria-label="Pareto scatter: tokens vs F1">'
|
||||
)
|
||||
parts.append(
|
||||
f'<rect x="0" y="0" width="{width}" height="{height}" '
|
||||
f'fill="white"/>'
|
||||
)
|
||||
# Gridlines at F1 = 0, 0.25, 0.5, 0.75, 1.0
|
||||
for f in (0.0, 0.25, 0.5, 0.75, 1.0):
|
||||
y = sy(f)
|
||||
parts.append(
|
||||
f'<line x1="{pad_l}" y1="{y:.1f}" x2="{width-pad_r}" '
|
||||
f'y2="{y:.1f}" stroke="{grid_color}" stroke-width="1"/>'
|
||||
)
|
||||
parts.append(
|
||||
f'<text x="{pad_l-8}" y="{y+4:.1f}" font-family="sans-serif" '
|
||||
f'font-size="11" fill="{axis_color}" text-anchor="end">'
|
||||
f'{f:.2f}</text>'
|
||||
)
|
||||
# X-axis ticks
|
||||
for frac in (0.0, 0.25, 0.5, 0.75, 1.0):
|
||||
t_val = frac * max_tokens_axis
|
||||
x = sx(t_val)
|
||||
parts.append(
|
||||
f'<line x1="{x:.1f}" y1="{height-pad_b}" x2="{x:.1f}" '
|
||||
f'y2="{height-pad_b+4}" stroke="{axis_color}"/>'
|
||||
)
|
||||
parts.append(
|
||||
f'<text x="{x:.1f}" y="{height-pad_b+18}" '
|
||||
f'font-family="sans-serif" font-size="11" fill="{axis_color}" '
|
||||
f'text-anchor="middle">{int(t_val)}</text>'
|
||||
)
|
||||
# Axis lines
|
||||
parts.append(
|
||||
f'<line x1="{pad_l}" y1="{pad_t}" x2="{pad_l}" '
|
||||
f'y2="{height-pad_b}" stroke="{axis_color}"/>'
|
||||
)
|
||||
parts.append(
|
||||
f'<line x1="{pad_l}" y1="{height-pad_b}" x2="{width-pad_r}" '
|
||||
f'y2="{height-pad_b}" stroke="{axis_color}"/>'
|
||||
)
|
||||
# Axis labels
|
||||
parts.append(
|
||||
f'<text x="{width/2:.1f}" y="{height-10}" '
|
||||
f'font-family="sans-serif" font-size="12" fill="{axis_color}" '
|
||||
f'text-anchor="middle">tokens</text>'
|
||||
)
|
||||
parts.append(
|
||||
f'<text x="15" y="{height/2:.1f}" font-family="sans-serif" '
|
||||
f'font-size="12" fill="{axis_color}" text-anchor="middle" '
|
||||
f'transform="rotate(-90 15 {height/2:.1f})">F1</text>'
|
||||
)
|
||||
# Baseline point
|
||||
bx, by = sx(baseline_tokens), sy(baseline_f1)
|
||||
parts.append(
|
||||
f'<circle cx="{bx:.1f}" cy="{by:.1f}" r="7" '
|
||||
f'fill="{baseline_color}"/>'
|
||||
)
|
||||
parts.append(
|
||||
f'<text x="{bx+10:.1f}" y="{by+4:.1f}" font-family="sans-serif" '
|
||||
f'font-size="11" fill="{baseline_color}">baseline</text>'
|
||||
)
|
||||
# Market point
|
||||
mx, my = sx(market_tokens), sy(market_f1)
|
||||
parts.append(
|
||||
f'<circle cx="{mx:.1f}" cy="{my:.1f}" r="7" '
|
||||
f'fill="{market_color}"/>'
|
||||
)
|
||||
parts.append(
|
||||
f'<text x="{mx+10:.1f}" y="{my+4:.1f}" font-family="sans-serif" '
|
||||
f'font-size="11" fill="{market_color}">marketplace</text>'
|
||||
)
|
||||
|
||||
parts.append("</svg>")
|
||||
return "".join(parts)
|
||||
|
||||
|
||||
CSS = """\
|
||||
body { font-family: -apple-system, system-ui, sans-serif;
|
||||
max-width: 960px; margin: 2rem auto; padding: 0 1rem;
|
||||
color: #1f2937; line-height: 1.55; }
|
||||
h1, h2, h3 { color: #111827; }
|
||||
h1 { border-bottom: 2px solid #2563eb; padding-bottom: .4rem; }
|
||||
.verdict-pass { display: inline-block; background: #dcfce7;
|
||||
color: #166534; padding: .3rem .8rem; border-radius: 6px;
|
||||
font-weight: 600; }
|
||||
.verdict-fail { display: inline-block; background: #fee2e2;
|
||||
color: #991b1b; padding: .3rem .8rem; border-radius: 6px;
|
||||
font-weight: 600; }
|
||||
table { border-collapse: collapse; margin: .8rem 0; width: 100%; }
|
||||
th, td { border: 1px solid #e5e7eb; padding: .4rem .6rem;
|
||||
text-align: left; vertical-align: top; }
|
||||
th { background: #f9fafb; }
|
||||
pre, code { background: #f3f4f6; border-radius: 4px;
|
||||
padding: .1rem .4rem; font-size: .9rem; }
|
||||
pre { padding: .8rem; white-space: pre-wrap; word-break: break-word; }
|
||||
.card { border: 1px solid #e5e7eb; border-radius: 8px;
|
||||
padding: 1rem 1.2rem; margin: 1rem 0; background: #fff; }
|
||||
.kv { display: grid; grid-template-columns: 180px 1fr; gap: .3rem .8rem; }
|
||||
.small { color: #6b7280; font-size: .88rem; }
|
||||
"""
|
||||
|
||||
|
||||
def render_report(data: dict[str, Any], out_path: Path) -> None:
|
||||
verdict = data.get("pareto", {})
|
||||
is_pass = verdict.get("verdict") == "PASS"
|
||||
verdict_html = (
|
||||
'<span class="verdict-pass">PASS — strictly northwest</span>'
|
||||
if is_pass
|
||||
else '<span class="verdict-fail">FAIL — not dominating baseline</span>'
|
||||
)
|
||||
|
||||
bids_rows: list[str] = []
|
||||
for b in data.get("bids", []):
|
||||
conf_str = "{:.2f}".format(b.get("confidence", 0) or 0)
|
||||
bids_rows.append(
|
||||
f"<tr><td>{_esc(b.get('agent'))}</td>"
|
||||
f"<td>{_esc(b.get('estimated_tokens'))}</td>"
|
||||
f"<td>{_esc(conf_str)}</td>"
|
||||
f"<td>{_esc(b.get('approach'))}</td></tr>"
|
||||
)
|
||||
bids_table = "\n".join(bids_rows) or (
|
||||
"<tr><td colspan=4>no bids</td></tr>"
|
||||
)
|
||||
|
||||
decomp_rows: list[str] = []
|
||||
for i, sub in enumerate(data.get("decomposition", []) or [], start=1):
|
||||
decomp_rows.append(
|
||||
f"<tr><td>{i}</td><td>{_esc(sub.get('question'))}</td>"
|
||||
f"<td>{_esc(sub.get('answer'))}</td></tr>"
|
||||
)
|
||||
decomp_table = "\n".join(decomp_rows) or (
|
||||
"<tr><td colspan=3>(none)</td></tr>"
|
||||
)
|
||||
|
||||
rep_rows: list[str] = []
|
||||
for rep in data.get("reputation", []) or []:
|
||||
score_str = "{:.3f}".format(rep.get("score", 0) or 0)
|
||||
rep_rows.append(
|
||||
f"<tr><td>{_esc(rep.get('agent'))}</td>"
|
||||
f"<td>{_esc(rep.get('domain'))}</td>"
|
||||
f"<td>{_esc(rep.get('runs'))}</td>"
|
||||
f"<td>{_esc(rep.get('correct'))}</td>"
|
||||
f"<td>{_esc(rep.get('tokens_spent'))}</td>"
|
||||
f"<td>{_esc(score_str)}</td></tr>"
|
||||
)
|
||||
rep_table = "\n".join(rep_rows) or (
|
||||
"<tr><td colspan=6>(empty)</td></tr>"
|
||||
)
|
||||
|
||||
svg = _scatter_svg(
|
||||
market_tokens=int(data.get("market", {}).get("tokens", 0)),
|
||||
market_f1=float(data.get("market", {}).get("f1", 0.0)),
|
||||
baseline_tokens=int(data.get("baseline", {}).get("tokens", 0)),
|
||||
baseline_f1=float(data.get("baseline", {}).get("f1", 0.0)),
|
||||
)
|
||||
|
||||
market = data.get("market", {})
|
||||
baseline = data.get("baseline", {})
|
||||
|
||||
market_f1_str = "{:.3f}".format(verdict.get("market_f1", 0) or 0)
|
||||
baseline_f1_str = "{:.3f}".format(verdict.get("baseline_f1", 0) or 0)
|
||||
f1_delta_str = "{:.3f}".format(verdict.get("f1_delta", 0) or 0)
|
||||
market_run_f1_str = "{:.3f}".format(market.get("f1", 0) or 0)
|
||||
baseline_run_f1_str = "{:.3f}".format(baseline.get("f1", 0) or 0)
|
||||
|
||||
html_doc = f"""<!DOCTYPE html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="utf-8"/>
|
||||
<title>MuSiQue MAS Benchmark — {_esc(data.get('question_id', ''))}</title>
|
||||
<style>{CSS}</style>
|
||||
</head>
|
||||
<body>
|
||||
<h1>MuSiQue MAS Benchmark Report</h1>
|
||||
<p class="small">
|
||||
Mode: <code>{_esc(data.get('mode', ''))}</code>
|
||||
· Question: <code>{_esc(data.get('question_id', ''))}</code>
|
||||
· Dry-run: <code>{_esc(data.get('dry_run', False))}</code>
|
||||
</p>
|
||||
|
||||
<div class="card">
|
||||
<h2>Verdict</h2>
|
||||
<p>{verdict_html}</p>
|
||||
<div class="kv">
|
||||
<div>Market tokens</div><div>{_esc(verdict.get('market_tokens'))}</div>
|
||||
<div>Market F1</div><div>{_esc(market_f1_str)}</div>
|
||||
<div>Baseline tokens</div><div>{_esc(verdict.get('baseline_tokens'))}</div>
|
||||
<div>Baseline F1</div><div>{_esc(baseline_f1_str)}</div>
|
||||
<div>Tokens delta</div><div>{_esc(verdict.get('tokens_delta'))}</div>
|
||||
<div>F1 delta</div><div>{_esc(f1_delta_str)}</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="card">
|
||||
<h2>Pareto plot</h2>
|
||||
{svg}
|
||||
<p class="small">
|
||||
Lower-right = expensive and wrong. Upper-left = cheap and correct.
|
||||
Marketplace must sit strictly northwest of baseline to pass.
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div class="card">
|
||||
<h2>Question</h2>
|
||||
<p><strong>{_esc(data.get('question', ''))}</strong></p>
|
||||
<p>Gold answer: <code>{_esc(data.get('gold_answer', ''))}</code></p>
|
||||
|
||||
<h3>Gold decomposition</h3>
|
||||
<table>
|
||||
<tr><th>#</th><th>Sub-question</th><th>Sub-answer</th></tr>
|
||||
{decomp_table}
|
||||
</table>
|
||||
</div>
|
||||
|
||||
<div class="card">
|
||||
<h2>Auction</h2>
|
||||
<p>Domain: <code>{_esc(data.get('domain', ''))}</code>
|
||||
· Budget: <code>{_esc(data.get('max_budget_tokens', ''))}</code>
|
||||
· Awarded to: <code>{_esc(data.get('awarded_to', ''))}</code></p>
|
||||
<h3>Bids received</h3>
|
||||
<table>
|
||||
<tr><th>Agent</th><th>Est. tokens</th><th>Confidence</th><th>Approach</th></tr>
|
||||
{bids_table}
|
||||
</table>
|
||||
</div>
|
||||
|
||||
<div class="card">
|
||||
<h2>Marketplace run</h2>
|
||||
<div class="kv">
|
||||
<div>Winning agent</div><div>{_esc(market.get('agent'))}</div>
|
||||
<div>Model</div><div>{_esc(market.get('model'))}</div>
|
||||
<div>Tokens</div><div>{_esc(market.get('tokens'))}</div>
|
||||
<div>F1</div><div>{_esc(market_run_f1_str)}</div>
|
||||
<div>Answer</div><div><code>{_esc(market.get('answer'))}</code></div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="card">
|
||||
<h2>Single-agent baseline</h2>
|
||||
<div class="kv">
|
||||
<div>Model</div><div>{_esc(baseline.get('model'))}</div>
|
||||
<div>Tokens</div><div>{_esc(baseline.get('tokens'))}</div>
|
||||
<div>F1</div><div>{_esc(baseline_run_f1_str)}</div>
|
||||
<div>Answer</div><div><code>{_esc(baseline.get('answer'))}</code></div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="card">
|
||||
<h2>Reputation ledger (post-run)</h2>
|
||||
<table>
|
||||
<tr><th>Agent</th><th>Domain</th><th>Runs</th><th>Correct</th>
|
||||
<th>Tokens</th><th>Score</th></tr>
|
||||
{rep_table}
|
||||
</table>
|
||||
</div>
|
||||
|
||||
<p class="small">
|
||||
Generated by <code>benchmark/report.py</code>.
|
||||
Marketplace primitives are currently stubbed in-process — see
|
||||
<code>benchmark/marketplace.py</code> for the migration plan to the
|
||||
real 016 SynapBus MCP tools.
|
||||
</p>
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
out_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
out_path.write_text(html_doc, encoding="utf-8")
|
||||
@@ -0,0 +1,2 @@
|
||||
anthropic>=0.40.0
|
||||
requests>=2.31.0
|
||||
@@ -0,0 +1,277 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Main entry point for the MuSiQue MAS benchmark.
|
||||
|
||||
Usage::
|
||||
|
||||
python benchmark/run.py --mode single-shot --question q1
|
||||
python benchmark/run.py --mode single-shot --question q1 --dry-run
|
||||
|
||||
Flow (single-shot):
|
||||
1. Load trio.jsonl, find the requested question (by short_id).
|
||||
2. Marketplace run:
|
||||
a. post_auction(task, domain, max_budget)
|
||||
b. each agent in the pool submits a bid
|
||||
c. marketplace awards best bid
|
||||
d. winner executes (Anthropic call or dry-run stub)
|
||||
e. marketplace.mark_done records reputation
|
||||
3. Baseline run: one Sonnet call with all distractors.
|
||||
4. Score both, compute Pareto verdict.
|
||||
5. Write results/latest.json and results/latest.html.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
# Allow running as ``python benchmark/run.py`` from the repo root.
|
||||
_HERE = Path(__file__).resolve().parent
|
||||
if str(_HERE) not in sys.path:
|
||||
sys.path.insert(0, str(_HERE))
|
||||
|
||||
from agents import default_pool # noqa: E402
|
||||
from baseline import run_baseline, BASELINE_MODEL # noqa: E402
|
||||
from marketplace import Marketplace # noqa: E402
|
||||
from report import render_report # noqa: E402
|
||||
from score import best_f1_against_aliases, pareto_verdict # noqa: E402
|
||||
|
||||
|
||||
TRIO_FILE = _HERE / "trio.jsonl"
|
||||
RESULTS_DIR = _HERE / "results"
|
||||
DEFAULT_DOMAIN = "multi-hop-qa"
|
||||
DEFAULT_BUDGET = 50_000
|
||||
|
||||
|
||||
def _load_trio() -> list[dict[str, Any]]:
|
||||
if not TRIO_FILE.exists():
|
||||
raise SystemExit(
|
||||
f"[run] trio.jsonl not found at {TRIO_FILE}. "
|
||||
"Run curate.py first."
|
||||
)
|
||||
out: list[dict[str, Any]] = []
|
||||
with open(TRIO_FILE, "r", encoding="utf-8") as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
out.append(json.loads(line))
|
||||
return out
|
||||
|
||||
|
||||
def _pick_question(
|
||||
trio: list[dict[str, Any]], want: str
|
||||
) -> dict[str, Any]:
|
||||
for rec in trio:
|
||||
if rec.get("short_id") == want or rec.get("id") == want:
|
||||
return rec
|
||||
raise SystemExit(
|
||||
f"[run] question {want!r} not found. Available: "
|
||||
+ ", ".join(r.get("short_id", r.get("id", "?")) for r in trio)
|
||||
)
|
||||
|
||||
|
||||
def single_shot(
|
||||
question: str,
|
||||
*,
|
||||
dry_run: bool,
|
||||
verbose: bool = True,
|
||||
) -> dict[str, Any]:
|
||||
trio = _load_trio()
|
||||
rec = _pick_question(trio, question)
|
||||
|
||||
task = {
|
||||
"question": rec["question"],
|
||||
"short_id": rec.get("short_id"),
|
||||
}
|
||||
paragraphs = rec.get("paragraphs", []) or []
|
||||
gold_answer = rec.get("answer", "")
|
||||
aliases = rec.get("answer_aliases", []) or []
|
||||
|
||||
market = Marketplace()
|
||||
pool = default_pool()
|
||||
|
||||
if verbose:
|
||||
print(f"[run] question {rec.get('short_id')}: {rec['question']!r}")
|
||||
print(
|
||||
f"[run] agents: "
|
||||
+ ", ".join(f"{a.name}({a.model})" for a in pool)
|
||||
)
|
||||
print(f"[run] paragraphs: {len(paragraphs)}")
|
||||
|
||||
# --- Marketplace path ------------------------------------------------
|
||||
auction_id = market.post_auction(
|
||||
task=task,
|
||||
domain=DEFAULT_DOMAIN,
|
||||
max_budget_tokens=DEFAULT_BUDGET,
|
||||
)
|
||||
if verbose:
|
||||
print(f"[run] posted auction {auction_id}")
|
||||
|
||||
for agent in pool:
|
||||
bid = agent.bid(task)
|
||||
market.bid(
|
||||
auction_id=auction_id,
|
||||
agent=agent.name,
|
||||
estimated_tokens=bid.estimated_tokens,
|
||||
confidence=bid.confidence,
|
||||
approach=bid.approach,
|
||||
)
|
||||
if verbose:
|
||||
print(
|
||||
f"[run] bid {agent.name}: "
|
||||
f"est={bid.estimated_tokens} conf={bid.confidence:.2f}"
|
||||
)
|
||||
|
||||
winning_bid = market.award(auction_id)
|
||||
winner_name = winning_bid["agent"]
|
||||
winner = next(a for a in pool if a.name == winner_name)
|
||||
if verbose:
|
||||
print(f"[run] awarded to {winner_name}")
|
||||
|
||||
start = time.time()
|
||||
result = winner.execute(
|
||||
task=task,
|
||||
paragraphs=paragraphs,
|
||||
dry_run=dry_run,
|
||||
max_budget_tokens=DEFAULT_BUDGET,
|
||||
)
|
||||
market_wall = time.time() - start
|
||||
|
||||
market_f1 = best_f1_against_aliases(
|
||||
result.answer, gold_answer, aliases
|
||||
)
|
||||
market.mark_done(
|
||||
auction_id=auction_id,
|
||||
answer=result.answer,
|
||||
actual_tokens=result.actual_tokens,
|
||||
correct=market_f1 >= 0.5,
|
||||
)
|
||||
if verbose:
|
||||
print(
|
||||
f"[run] market answer: {result.answer!r} "
|
||||
f"(tokens={result.actual_tokens}, f1={market_f1:.3f})"
|
||||
)
|
||||
|
||||
# --- Baseline path ---------------------------------------------------
|
||||
start = time.time()
|
||||
baseline = run_baseline(
|
||||
question=rec["question"],
|
||||
paragraphs=paragraphs,
|
||||
dry_run=dry_run,
|
||||
)
|
||||
baseline_wall = time.time() - start
|
||||
baseline_f1 = best_f1_against_aliases(
|
||||
baseline["answer"], gold_answer, aliases
|
||||
)
|
||||
if verbose:
|
||||
print(
|
||||
f"[run] baseline answer: {baseline['answer']!r} "
|
||||
f"(tokens={baseline['tokens']}, f1={baseline_f1:.3f})"
|
||||
)
|
||||
|
||||
verdict = pareto_verdict(
|
||||
market_tokens=result.actual_tokens,
|
||||
market_f1=market_f1,
|
||||
baseline_tokens=baseline["tokens"],
|
||||
baseline_f1=baseline_f1,
|
||||
)
|
||||
if verbose:
|
||||
print(f"[run] PARETO VERDICT: {verdict['verdict']}")
|
||||
|
||||
return {
|
||||
"mode": "single-shot",
|
||||
"dry_run": dry_run,
|
||||
"question_id": rec.get("short_id"),
|
||||
"musique_id": rec.get("id"),
|
||||
"question": rec["question"],
|
||||
"gold_answer": gold_answer,
|
||||
"decomposition": rec.get("decomposition", []),
|
||||
"domain": DEFAULT_DOMAIN,
|
||||
"max_budget_tokens": DEFAULT_BUDGET,
|
||||
"awarded_to": winner_name,
|
||||
"bids": market.list_bids(auction_id),
|
||||
"market": {
|
||||
"agent": winner_name,
|
||||
"model": winner.model,
|
||||
"tokens": result.actual_tokens,
|
||||
"answer": result.answer,
|
||||
"f1": market_f1,
|
||||
"wall_seconds": market_wall,
|
||||
"raw_text": result.raw_text,
|
||||
},
|
||||
"baseline": {
|
||||
"model": baseline.get("model", BASELINE_MODEL),
|
||||
"tokens": baseline["tokens"],
|
||||
"answer": baseline["answer"],
|
||||
"f1": baseline_f1,
|
||||
"wall_seconds": baseline_wall,
|
||||
"raw_text": baseline.get("raw_text", ""),
|
||||
},
|
||||
"pareto": verdict,
|
||||
"reputation": market.all_reputation(),
|
||||
}
|
||||
|
||||
|
||||
def _write_outputs(result: dict[str, Any]) -> tuple[Path, Path]:
|
||||
RESULTS_DIR.mkdir(parents=True, exist_ok=True)
|
||||
json_path = RESULTS_DIR / "latest.json"
|
||||
html_path = RESULTS_DIR / "latest.html"
|
||||
# Trim raw_text from json to keep it small and readable.
|
||||
trimmed = dict(result)
|
||||
for key in ("market", "baseline"):
|
||||
section = dict(trimmed.get(key, {}))
|
||||
if "raw_text" in section:
|
||||
section["raw_text"] = (section["raw_text"] or "")[:2000]
|
||||
trimmed[key] = section
|
||||
json_path.write_text(
|
||||
json.dumps(trimmed, indent=2, ensure_ascii=False), encoding="utf-8"
|
||||
)
|
||||
render_report(result, html_path)
|
||||
return json_path, html_path
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="MuSiQue multi-agent benchmark harness"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--mode",
|
||||
choices=["single-shot"],
|
||||
default="single-shot",
|
||||
help="Run mode (only single-shot is implemented in MVP)",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--question",
|
||||
default="q1",
|
||||
help="Question short_id from trio.jsonl (q1/q2/q3)",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--dry-run",
|
||||
action="store_true",
|
||||
help="Skip real Anthropic API calls; use stub responses",
|
||||
)
|
||||
args = parser.parse_args(argv)
|
||||
|
||||
if args.mode != "single-shot":
|
||||
print(f"[run] mode {args.mode} not implemented in MVP", file=sys.stderr)
|
||||
return 2
|
||||
|
||||
result = single_shot(
|
||||
question=args.question,
|
||||
dry_run=args.dry_run,
|
||||
verbose=True,
|
||||
)
|
||||
json_path, html_path = _write_outputs(result)
|
||||
print(f"[run] wrote {json_path}")
|
||||
print(f"[run] wrote {html_path}")
|
||||
print(f"[run] verdict: {result['pareto']['verdict']}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,86 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Scoring utilities for the MuSiQue benchmark.
|
||||
|
||||
- Normalized exact-match F1 (SQuAD-style): lowercase, strip articles,
|
||||
strip punctuation, collapse whitespace.
|
||||
- Pareto verdict: the marketplace point is strictly northwest of the
|
||||
baseline iff it uses fewer tokens AND has F1 >= baseline, with at
|
||||
least one of those strict.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import string
|
||||
from collections import Counter
|
||||
from typing import Any
|
||||
|
||||
_ARTICLE_RE = re.compile(r"\b(a|an|the)\b", re.IGNORECASE)
|
||||
|
||||
|
||||
def normalize(text: str) -> str:
|
||||
if text is None:
|
||||
return ""
|
||||
text = text.lower()
|
||||
text = _ARTICLE_RE.sub(" ", text)
|
||||
text = "".join(ch for ch in text if ch not in string.punctuation)
|
||||
text = " ".join(text.split())
|
||||
return text
|
||||
|
||||
|
||||
def f1(prediction: str, gold: str) -> float:
|
||||
pred_tokens = normalize(prediction).split()
|
||||
gold_tokens = normalize(gold).split()
|
||||
if not pred_tokens and not gold_tokens:
|
||||
return 1.0
|
||||
if not pred_tokens or not gold_tokens:
|
||||
return 0.0
|
||||
common = Counter(pred_tokens) & Counter(gold_tokens)
|
||||
overlap = sum(common.values())
|
||||
if overlap == 0:
|
||||
return 0.0
|
||||
precision = overlap / len(pred_tokens)
|
||||
recall = overlap / len(gold_tokens)
|
||||
return 2 * precision * recall / (precision + recall)
|
||||
|
||||
|
||||
def exact_match(prediction: str, gold: str) -> bool:
|
||||
return normalize(prediction) == normalize(gold)
|
||||
|
||||
|
||||
def best_f1_against_aliases(
|
||||
prediction: str, gold: str, aliases: list[str] | None = None
|
||||
) -> float:
|
||||
candidates = [gold] + list(aliases or [])
|
||||
return max(f1(prediction, c) for c in candidates if c is not None)
|
||||
|
||||
|
||||
def pareto_verdict(
|
||||
market_tokens: int,
|
||||
market_f1: float,
|
||||
baseline_tokens: int,
|
||||
baseline_f1: float,
|
||||
) -> dict[str, Any]:
|
||||
"""
|
||||
Strictly northwest of baseline: fewer tokens AND higher-or-equal F1,
|
||||
with at least one strict inequality.
|
||||
"""
|
||||
tokens_better = market_tokens < baseline_tokens
|
||||
quality_atleast = market_f1 >= baseline_f1
|
||||
quality_better = market_f1 > baseline_f1
|
||||
|
||||
strictly_nw = (
|
||||
(tokens_better and quality_atleast)
|
||||
or (quality_better and market_tokens <= baseline_tokens)
|
||||
)
|
||||
return {
|
||||
"verdict": "PASS" if strictly_nw else "FAIL",
|
||||
"strictly_northwest": strictly_nw,
|
||||
"market_tokens": int(market_tokens),
|
||||
"market_f1": float(market_f1),
|
||||
"baseline_tokens": int(baseline_tokens),
|
||||
"baseline_f1": float(baseline_f1),
|
||||
"tokens_delta": int(market_tokens - baseline_tokens),
|
||||
"f1_delta": float(market_f1 - baseline_f1),
|
||||
}
|
||||
@@ -0,0 +1,198 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
MuSiQue dataset downloader.
|
||||
|
||||
Downloads ``musique_v1.0.zip`` from the canonical source used by the
|
||||
upstream project (https://github.com/StonyBrookNLP/musique). The zip is
|
||||
hosted on Google Drive (file id ``1tGdADlNjWFaHLeZZGShh2IRcpO6Lv24h``);
|
||||
this mirrors the behavior of the project's ``download_data.sh`` which
|
||||
uses ``gdown`` under the hood.
|
||||
|
||||
Idempotent — skips download if the target dev-set jsonl already exists.
|
||||
Run: ``python benchmark/setup.py``
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import zipfile
|
||||
from pathlib import Path
|
||||
|
||||
import requests
|
||||
|
||||
GDRIVE_FILE_ID = "1tGdADlNjWFaHLeZZGShh2IRcpO6Lv24h"
|
||||
GDRIVE_URL = "https://docs.google.com/uc?export=download"
|
||||
|
||||
DATA_DIR = Path(__file__).resolve().parent / "data"
|
||||
ZIP_PATH = DATA_DIR / "musique_v1.0.zip"
|
||||
TARGET_FILE = DATA_DIR / "musique_ans_v1.0_dev.jsonl"
|
||||
|
||||
|
||||
def _write_stream(resp: requests.Response, dest: Path) -> int:
|
||||
total = int(resp.headers.get("Content-Length", 0))
|
||||
downloaded = 0
|
||||
dest.parent.mkdir(parents=True, exist_ok=True)
|
||||
with open(dest, "wb") as f:
|
||||
for chunk in resp.iter_content(chunk_size=1024 * 1024):
|
||||
if not chunk:
|
||||
continue
|
||||
f.write(chunk)
|
||||
downloaded += len(chunk)
|
||||
if total:
|
||||
pct = 100.0 * downloaded / total
|
||||
print(
|
||||
f"\r downloading: {downloaded/1e6:6.1f} MB "
|
||||
f"/ {total/1e6:6.1f} MB ({pct:5.1f}%)",
|
||||
end="",
|
||||
file=sys.stderr,
|
||||
)
|
||||
print("", file=sys.stderr)
|
||||
return downloaded
|
||||
|
||||
|
||||
def _download_gdrive(file_id: str, dest: Path) -> bool:
|
||||
"""
|
||||
Download a large file from Google Drive, handling the virus-scan
|
||||
confirmation page that Drive injects for anything over ~100 MB.
|
||||
"""
|
||||
session = requests.Session()
|
||||
try:
|
||||
resp = session.get(
|
||||
GDRIVE_URL,
|
||||
params={"id": file_id, "export": "download"},
|
||||
stream=True,
|
||||
timeout=60,
|
||||
)
|
||||
except requests.RequestException as exc:
|
||||
print(f" -> request failed: {exc}", file=sys.stderr)
|
||||
return False
|
||||
|
||||
# Case 1: Drive returns the file directly (small file or cached).
|
||||
ctype = resp.headers.get("Content-Type", "")
|
||||
if "text/html" not in ctype.lower():
|
||||
_write_stream(resp, dest)
|
||||
return dest.exists() and dest.stat().st_size > 0
|
||||
|
||||
# Case 2: HTML confirmation page. Extract the confirm token and/or
|
||||
# the form action URL.
|
||||
html = resp.text
|
||||
# Newer Drive flow: a <form ...> with all the params we need.
|
||||
form_match = re.search(
|
||||
r'<form[^>]*id="download-form"[^>]*action="([^"]+)"', html
|
||||
)
|
||||
if form_match:
|
||||
action = form_match.group(1).replace("&", "&")
|
||||
params = dict(
|
||||
re.findall(
|
||||
r'name="([^"]+)"[^>]*value="([^"]+)"', html
|
||||
)
|
||||
)
|
||||
try:
|
||||
resp2 = session.get(action, params=params, stream=True, timeout=120)
|
||||
if resp2.status_code == 200:
|
||||
_write_stream(resp2, dest)
|
||||
return dest.exists() and dest.stat().st_size > 0
|
||||
except requests.RequestException as exc:
|
||||
print(f" -> form post failed: {exc}", file=sys.stderr)
|
||||
return False
|
||||
|
||||
# Older flow: confirm cookie token.
|
||||
token = None
|
||||
for k, v in session.cookies.items():
|
||||
if k.startswith("download_warning"):
|
||||
token = v
|
||||
break
|
||||
if token is None:
|
||||
m = re.search(r'confirm=([0-9A-Za-z_-]+)', html)
|
||||
if m:
|
||||
token = m.group(1)
|
||||
if token:
|
||||
try:
|
||||
resp3 = session.get(
|
||||
GDRIVE_URL,
|
||||
params={
|
||||
"id": file_id,
|
||||
"export": "download",
|
||||
"confirm": token,
|
||||
},
|
||||
stream=True,
|
||||
timeout=120,
|
||||
)
|
||||
if resp3.status_code == 200:
|
||||
_write_stream(resp3, dest)
|
||||
return dest.exists() and dest.stat().st_size > 0
|
||||
except requests.RequestException as exc:
|
||||
print(f" -> confirm fetch failed: {exc}", file=sys.stderr)
|
||||
return False
|
||||
|
||||
print(" -> could not navigate Google Drive download flow", file=sys.stderr)
|
||||
return False
|
||||
|
||||
|
||||
def _extract(zip_path: Path, out_dir: Path) -> None:
|
||||
"""Extract the dev set jsonl from the zip."""
|
||||
wanted_suffixes = (
|
||||
"musique_ans_v1.0_dev.jsonl",
|
||||
"musique_ans_v1.0_train.jsonl",
|
||||
)
|
||||
with zipfile.ZipFile(zip_path) as zf:
|
||||
members = zf.namelist()
|
||||
extracted_any = False
|
||||
for m in members:
|
||||
base = os.path.basename(m)
|
||||
if base in wanted_suffixes:
|
||||
with zf.open(m) as src, open(out_dir / base, "wb") as dst:
|
||||
dst.write(src.read())
|
||||
print(f" extracted: {base}")
|
||||
extracted_any = True
|
||||
if not extracted_any:
|
||||
# Fall back: extract everything so a human can inspect.
|
||||
zf.extractall(out_dir)
|
||||
print(
|
||||
" could not find canonical filenames; extracted all",
|
||||
file=sys.stderr,
|
||||
)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
DATA_DIR.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
if TARGET_FILE.exists():
|
||||
size = TARGET_FILE.stat().st_size
|
||||
print(f"[setup] already present: {TARGET_FILE} ({size/1e6:.1f} MB)")
|
||||
return 0
|
||||
|
||||
print(f"[setup] downloading Google Drive file id {GDRIVE_FILE_ID}")
|
||||
ok = _download_gdrive(GDRIVE_FILE_ID, ZIP_PATH)
|
||||
|
||||
if not ok:
|
||||
print(
|
||||
"[setup] ERROR: failed to download MuSiQue. Please download "
|
||||
"manually from "
|
||||
f"https://drive.google.com/file/d/{GDRIVE_FILE_ID}/view "
|
||||
f"and place the zip at {ZIP_PATH}",
|
||||
file=sys.stderr,
|
||||
)
|
||||
return 2
|
||||
|
||||
print(f"[setup] extracting {ZIP_PATH}")
|
||||
_extract(ZIP_PATH, DATA_DIR)
|
||||
|
||||
if not TARGET_FILE.exists():
|
||||
print(
|
||||
f"[setup] WARNING: {TARGET_FILE.name} not found after extract. "
|
||||
f"Listing {DATA_DIR}:",
|
||||
file=sys.stderr,
|
||||
)
|
||||
for p in sorted(DATA_DIR.iterdir()):
|
||||
print(f" - {p.name}", file=sys.stderr)
|
||||
return 3
|
||||
|
||||
print(f"[setup] ready: {TARGET_FILE}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
File diff suppressed because one or more lines are too long
@@ -0,0 +1,160 @@
|
||||
# MAS Benchmark — Design Document
|
||||
|
||||
**Date**: 2026-04-11
|
||||
**Status**: Approved (autonomous mode)
|
||||
**Author**: Claude Opus 4.6 (1M context) via brainstorming skill
|
||||
**Next**: speckit specification at `specs/017-musique-benchmark/spec.md`
|
||||
**Related**: `specs/016-agent-marketplace/spec.md`
|
||||
|
||||
## Problem
|
||||
|
||||
The current "Fermi piano tuners in Chicago" example in `multiagent_systems_report.html` is dated and rare as a profession. It needs to be replaced with a modern task that:
|
||||
|
||||
- Exercises the same MAS features (dynamic decomposition, dedup, uncertainty aggregation, cost accounting, orphaned-spawn recovery)
|
||||
- Runs on local data only — no `WebSearch` tool required
|
||||
- Serves both as a readable narrative example and a runnable integration test
|
||||
- Measures the Pareto frontier of quality vs token cost — never quality alone
|
||||
- Fits a tight dev-loop token budget (≤ 500k tokens per full learning run)
|
||||
|
||||
## Decisions (from brainstorming)
|
||||
|
||||
1. **Purpose**: dual-use — narrative example in the report AND integration test for `016-agent-marketplace`.
|
||||
2. **Dataset**: **MuSiQue-Ans 4-hop** as primary (~100 MB, gold decomposition DAGs, anti-shortcut filtered). **FRAMES** as future cross-eval runner-up.
|
||||
3. **Scale**: N=3 curated questions. Deliberately cherry-picked to share a bridge entity so sub-agents naturally re-lookup the same Wikipedia paragraphs (dedup metric becomes observable at N=3).
|
||||
4. **Agent pool**: **mixed-tier** — Haiku + Sonnet + Opus. Each agent publishes its own capability manifest with per-domain cost profile. The auction has to learn when paying for Opus is worth it and when Haiku suffices.
|
||||
5. **Run modes — tiered**:
|
||||
- **Single-shot** (CI smoke test): run the 3 questions once. ~100k tokens.
|
||||
- **Learning tier**: run the same 3 questions for 5 epochs. Reputation and skill cards persist; scratchpad resets per epoch. Measure tokens-per-correct-answer declining across epochs. ~500k tokens.
|
||||
6. **Execution strategy**: harness talks to the spec-016 MCP tool surface. Runs against real SynapBus once 016 is implemented.
|
||||
7. **Scoring is Pareto**: report both quality (F1, decomposition F1) AND cost (total tokens, tokens-per-correct). A passing marketplace strictly dominates a single-agent baseline.
|
||||
|
||||
## Architecture
|
||||
|
||||
### Components
|
||||
|
||||
1. **Curated trio file** (`benchmark/trio.jsonl`) — three MuSiQue-Ans questions with shared pivot entity, gold answers, gold decomposition DAGs. Selected by deterministic rule from the dev set and checked into the repo for reproducibility.
|
||||
|
||||
2. **Benchmark harness** (Python) — orchestrates a run:
|
||||
- Reads trio.jsonl and initial skill-card configuration
|
||||
- Seeds the marketplace (posts skill-card wiki articles for each agent, creates the `#bench-auction` auction channel)
|
||||
- For each question, posts an auction, waits for bids, awards via MCP, polls for completion
|
||||
- Collects metrics (per-question tokens, F1, cache-hit rate, decomposition F1, wall time)
|
||||
- Runs single-shot or 5-epoch learning tier per CLI flag
|
||||
|
||||
3. **Agent runner** (Python) — spawns N agents, each a Claude Agent SDK session with:
|
||||
- A system prompt built from the agent's skill card
|
||||
- SynapBus MCP tools configured
|
||||
- Distinct model tier (Haiku / Sonnet / Opus)
|
||||
- Token counter hook for real-time budget enforcement
|
||||
|
||||
4. **Baseline runner** — a single Claude call that receives the question and all 20 distractor paragraphs in one shot, using chain-of-thought, no decomposition, no marketplace. Produces reference `(tokens, F1)` for Pareto comparison.
|
||||
|
||||
5. **Scoring module** — computes metrics per run, writes `results/{run_id}.json`, generates Pareto plot data.
|
||||
|
||||
6. **Report generator** — renders a rich HTML with the narrative, Pareto chart, per-question trace, and learning curve.
|
||||
|
||||
### Data flow (single question)
|
||||
|
||||
```
|
||||
trio.jsonl → harness.post_auction(question, max_budget, domains)
|
||||
↓
|
||||
SynapBus auction channel (reactive trigger)
|
||||
↓
|
||||
┌────────────┬────────────┬──────────────┐
|
||||
↓ ↓ ↓ ↓
|
||||
Haiku agent Sonnet agent Opus agent (poller)
|
||||
↓ ↓ ↓
|
||||
bid() bid() bid()
|
||||
└────────────┴────────────┘
|
||||
↓
|
||||
harness.award(best bid)
|
||||
↓
|
||||
winner.claim → execute
|
||||
↓
|
||||
(reads distractor paragraphs via MCP)
|
||||
↓
|
||||
shared scratchpad
|
||||
(dedup: same entity → cache hit)
|
||||
↓
|
||||
winner.mark_done(answer, tokens)
|
||||
↓
|
||||
reputation ledger update (per-domain tuple)
|
||||
↓
|
||||
harness.score(answer vs gold)
|
||||
```
|
||||
|
||||
### Scoring (Pareto)
|
||||
|
||||
Three metrics plotted together, one point per run configuration:
|
||||
|
||||
- **Quality**: final-answer exact-match F1 (0.0 / 0.33 / 0.67 / 1.0 at N=3)
|
||||
- **Cost**: total tokens consumed (orchestrator + all sub-agents across all 3 questions)
|
||||
- **Efficiency**: tokens-per-correct-answer = total_tokens / max(F1 × 3, 1)
|
||||
|
||||
Four configurations plotted on the Pareto chart:
|
||||
|
||||
| # | Config | Expected quality | Expected cost |
|
||||
|---|---|---|---|
|
||||
| 1 | Single-agent baseline (Opus, all distractors in context) | high (~2/3) | high (~30k) |
|
||||
| 2 | Single-agent baseline (Sonnet, same) | medium (~2/3) | medium (~15k) |
|
||||
| 3 | Naive marketplace (no reputation, no reflection, no dedup) | medium (~2/3) | medium-high (~40k) |
|
||||
| 4 | Full 016 marketplace (reputation + dedup + mixed-tier routing) | ≥ baseline | should be **strictly less** than baseline |
|
||||
|
||||
The marketplace passes only if it lands **strictly northwest** of Sonnet baseline on the Pareto plot.
|
||||
|
||||
### Learning tier
|
||||
|
||||
5 epochs of the same 3 questions. What persists between epochs:
|
||||
|
||||
- Reputation ledger entries (accumulate)
|
||||
- Capability manifest revisions (reflection loop proposes diffs — auto-approved for benchmark)
|
||||
- Per-agent skill-card example-tasks list (grows monotonically)
|
||||
|
||||
What resets between epochs:
|
||||
|
||||
- Shared scratchpad (within-task coordination, not long-term memory)
|
||||
- Auction channel contents (each epoch creates fresh auctions)
|
||||
|
||||
**Expected learning curve**: tokens-per-correct-answer should drop monotonically from epoch 1 (all agents uncalibrated, ε-greedy bootstrap dominates) to epoch 5 (reputation converged, routing stable). If it doesn't — the marketplace has a bug.
|
||||
|
||||
### Failure injection
|
||||
|
||||
For orphaned-spawn recovery: one epoch runs with a 10% random sub-agent failure rate (agents randomly return "timeout" instead of bid). Measure accuracy degradation. Target: ≤ 5 percentage-point drop.
|
||||
|
||||
## Realistic MVP scope
|
||||
|
||||
Given execution constraints, the MVP for **today's autonomous run** scopes down:
|
||||
|
||||
- **Questions**: N=1 instead of N=3 (save 3× tokens on the actual run; the trio.jsonl file still contains all 3 for future runs)
|
||||
- **Agents**: 2 (Haiku + Sonnet) instead of 3 (Haiku + Sonnet + Opus). Mixed-tier proved on 2 tiers.
|
||||
- **Epochs**: 1 single-shot run, no learning tier. Design doc describes the 5-epoch protocol for future runs.
|
||||
- **Reflection loop**: skipped. Full 016 spec has it; MVP implementation focuses on US1 + US2 + US3 (auction + manifests + reputation).
|
||||
|
||||
**Still measured and reported**:
|
||||
- Dynamic decomposition on one real 4-hop MuSiQue question
|
||||
- Auction → bid → award → claim → done full lifecycle
|
||||
- Per-model cost differentials (Haiku vs Sonnet on same task)
|
||||
- Pareto comparison against single-agent baseline
|
||||
- Reputation ledger write-through
|
||||
|
||||
**Documented-but-deferred**:
|
||||
- Reflection loop + skill-card diff proposals (US4 of spec 016)
|
||||
- Tombstoning on failure rate (FR-020a/b)
|
||||
- Full 5-epoch learning tier
|
||||
- 3-question curated trio dedup measurement
|
||||
- FRAMES cross-eval
|
||||
|
||||
## Acceptance criteria for autonomous run
|
||||
|
||||
1. `016-agent-marketplace` MVP compiles, passes its own Go tests, and exposes the required MCP tools.
|
||||
2. Benchmark harness downloads MuSiQue, curates trio.jsonl, runs 1 question end-to-end against local SynapBus with the 016 implementation.
|
||||
3. Real token counts and real F1 recorded.
|
||||
4. Pareto plot generated comparing full marketplace vs Sonnet baseline.
|
||||
5. HTML report renders with live numbers, not placeholders.
|
||||
6. `autonomous_summary.md` written documenting what shipped, what passed, what deferred.
|
||||
|
||||
## Honest caveats
|
||||
|
||||
- N=1 cannot support statistical claims. The benchmark's purpose at this scale is **mechanism verification**, not efficacy proof.
|
||||
- Claude Agent SDK integration is a known pain point — may need fallback to direct Anthropic SDK if MCP wiring fails.
|
||||
- Single-epoch run cannot show the learning curve. Design doc + spec describe the full protocol for future scaling.
|
||||
@@ -0,0 +1,36 @@
|
||||
# Specification Quality Checklist: MuSiQue MAS Benchmark
|
||||
|
||||
**Purpose**: Validate specification completeness and quality before proceeding to planning
|
||||
**Created**: 2026-04-11
|
||||
**Feature**: [spec.md](../spec.md)
|
||||
|
||||
## Content Quality
|
||||
|
||||
- [x] No implementation details (languages, frameworks, APIs)
|
||||
- [x] Focused on user value and business needs
|
||||
- [x] Written for non-technical stakeholders
|
||||
- [x] All mandatory sections completed
|
||||
|
||||
## Requirement Completeness
|
||||
|
||||
- [x] No [NEEDS CLARIFICATION] markers remain
|
||||
- [x] Requirements are testable and unambiguous
|
||||
- [x] Success criteria are measurable
|
||||
- [x] Success criteria are technology-agnostic
|
||||
- [x] All acceptance scenarios are defined
|
||||
- [x] Edge cases are identified
|
||||
- [x] Scope is clearly bounded
|
||||
- [x] Dependencies and assumptions identified
|
||||
|
||||
## Feature Readiness
|
||||
|
||||
- [x] All functional requirements have clear acceptance criteria
|
||||
- [x] User scenarios cover primary flows
|
||||
- [x] Feature meets measurable outcomes defined in Success Criteria
|
||||
- [x] No implementation details leak into specification
|
||||
|
||||
## Notes
|
||||
|
||||
- Specification intentionally allows Claude Agent SDK with direct Anthropic SDK fallback — this is an operational concern rather than an architectural one.
|
||||
- FR-021 through FR-023 (learning tier) are marked deferred for the autonomous MVP run but remain in scope for the full spec.
|
||||
- The benchmark is paired with `016-agent-marketplace` and will not function without it (or a stub).
|
||||
@@ -0,0 +1,169 @@
|
||||
# Feature Specification: MuSiQue Multi-Agent Benchmark Harness
|
||||
|
||||
**Feature Branch**: `017-musique-benchmark`
|
||||
**Created**: 2026-04-11
|
||||
**Status**: Draft
|
||||
**Input**: MuSiQue-based MAS benchmark harness that integration-tests the 016-agent-marketplace. N=3 curated trio sharing a pivot entity, mixed-tier agent pool (Haiku + Sonnet + Opus), Pareto scoring vs single-agent baseline, optional 5-epoch learning tier, Python harness using Claude Agent SDK.
|
||||
|
||||
## User Scenarios & Testing *(mandatory)*
|
||||
|
||||
### User Story 1 — Single-shot Pareto verification (Priority: P1)
|
||||
|
||||
A developer wants to verify that the 016 agent marketplace actually delivers better quality-per-token than a single-agent baseline on a real multi-hop reasoning task. They run the benchmark harness in single-shot mode. The harness posts one MuSiQue 4-hop question to the marketplace, mixed-tier agents bid, one wins, executes the task using distractor paragraphs from the local MuSiQue corpus, and reports the final answer. The harness also runs the same question through a single-agent baseline (one Claude call with all distractors in context). It computes and plots both runs on a Pareto chart of quality vs tokens. The marketplace passes only if it lands strictly northwest of the baseline.
|
||||
|
||||
**Why this priority**: This is the irreducible proof-of-work for the marketplace. Without it, the 016 spec is unvalidated. With it, we have evidence that the auction + mixed-tier routing primitives actually produce a Pareto improvement on a real task.
|
||||
|
||||
**Independent Test**: Start a fresh SynapBus instance with the 016 MVP. Run `python benchmark/run.py --mode single-shot --question q1`. The script completes within a token budget, produces a results JSON with real token counts and F1 scores for both the marketplace and the baseline, and exits zero.
|
||||
|
||||
**Acceptance Scenarios**:
|
||||
|
||||
1. **Given** a fresh SynapBus instance with the 016 marketplace primitives loaded, **When** the harness posts the MuSiQue question as an auction, **Then** at least two agents submit structured bids as threaded replies within a configurable wait window.
|
||||
2. **Given** bids have been submitted, **When** the harness awards the best bid via the `awarded` reaction, **Then** the winning agent receives a claim and begins executing against the MuSiQue distractor paragraphs.
|
||||
3. **Given** the winning agent completes the task, **When** the harness compares the answer to the gold answer, **Then** an F1 score is recorded and the token count for the entire run is aggregated.
|
||||
4. **Given** both the marketplace run and the baseline run have completed, **When** the harness generates the Pareto report, **Then** both data points are shown with their exact token counts and F1 scores, and the marketplace run is either strictly northwest of the baseline or an explicit warning is raised.
|
||||
|
||||
---
|
||||
|
||||
### User Story 2 — Curated trio with observable dedup (Priority: P2)
|
||||
|
||||
A developer wants to measure the shared-scratchpad duplicate-work detection primitive. The trio file contains three carefully selected MuSiQue questions sharing a pivot bridge entity (e.g., all involve the United States as an intermediate hop). When the harness runs the trio in sequence with a persistent shared scratchpad across questions, the second and third questions should hit cached entity lookups from the first, producing an observable cache-hit rate above 0%.
|
||||
|
||||
**Why this priority**: Dedup is one of the five MAS features the benchmark is supposed to exercise. Without the curated trio, dedup metrics are 0 at N=3 and the primitive is silent. This is P2 because US1 delivers single-question value first; the trio extends it.
|
||||
|
||||
**Independent Test**: Run `python benchmark/run.py --mode trio --scratchpad persistent`. The harness runs all three questions in sequence. The scratchpad stats show at least 5 cache hits on entity lookups across the three questions, and the aggregate token count is lower than 3× the single-question token count due to cache re-use.
|
||||
|
||||
**Acceptance Scenarios**:
|
||||
|
||||
1. **Given** the curated trio.jsonl exists with three questions sharing a pivot entity, **When** the harness runs the trio with persistent scratchpad, **Then** cache-hit count at the end of the run is greater than zero.
|
||||
2. **Given** the same trio is run with a non-persistent scratchpad, **When** the run completes, **Then** total tokens are strictly greater than the persistent-scratchpad run.
|
||||
|
||||
---
|
||||
|
||||
### User Story 3 — Learning tier with reputation convergence (Priority: P3)
|
||||
|
||||
A developer wants to verify that reputation and reflection actually cause improvement over time. The learning tier runs the same trio for 5 epochs. Reputation and capability manifests persist; scratchpad resets per epoch. Tokens-per-correct-answer should decline monotonically across epochs as the marketplace learns which model tier is best for which sub-task.
|
||||
|
||||
**Why this priority**: This is the full marketplace proof-of-learning. It is P3 because US1 and US2 are load-bearing; the learning tier depends on both being stable first, and it is the most token-expensive mode (~500k tokens per full run).
|
||||
|
||||
**Independent Test**: Run `python benchmark/run.py --mode learning --epochs 5`. The harness completes all 5 epochs, writes a results JSON containing per-epoch tokens and F1, and plots a learning curve showing the tokens-per-correct-answer trend.
|
||||
|
||||
**Acceptance Scenarios**:
|
||||
|
||||
1. **Given** the learning tier has run 5 epochs on the trio, **When** the per-epoch metrics are plotted, **Then** the tokens-per-correct-answer for epoch 5 is less than epoch 1.
|
||||
2. **Given** between-epoch state, **When** epoch N starts, **Then** reputation entries from epoch N-1 are present and influence bid scoring, while the scratchpad is empty.
|
||||
|
||||
---
|
||||
|
||||
### User Story 4 — Rich HTML report (Priority: P1)
|
||||
|
||||
The benchmark run produces a rich, self-contained HTML report that shows the task, the decomposition, the bids, the award, the execution trace, the gold answer, the marketplace answer, the baseline answer, the Pareto plot, and all token counts. Non-technical readers can open the file and understand what happened.
|
||||
|
||||
**Why this priority**: The benchmark has to serve as a compelling illustration for the `multiagent_systems_report.html`. Without a rendered report, the run is just a JSON file nobody reads. It is P1 because without it the benchmark fails its dual-use requirement.
|
||||
|
||||
**Independent Test**: After a successful run, `results/latest.html` exists, opens in a browser without errors, and visibly contains the question text, the Pareto plot (as SVG or inline data), and the per-run token counts.
|
||||
|
||||
**Acceptance Scenarios**:
|
||||
|
||||
1. **Given** a completed benchmark run, **When** the report generator runs, **Then** `results/latest.html` is written and contains the question, decomposition, answers, tokens, and Pareto data.
|
||||
2. **Given** the HTML report is opened in a browser, **When** a reader scrolls through it, **Then** they can tell whether the marketplace passed without needing to open any JSON files.
|
||||
|
||||
---
|
||||
|
||||
### Edge Cases
|
||||
|
||||
- **MuSiQue download failure**: the setup script retries once, then fails with a clear error pointing at the canonical URL.
|
||||
- **No agents bid within window**: the harness declares auction-timeout, records it as a test failure, and exits non-zero.
|
||||
- **Agent returns malformed bid**: the bid is rejected by schema validation, the agent is not awarded the task, other bidders still compete.
|
||||
- **Winner exceeds budget**: the harness hard-stops execution at 100% of declared budget, records the auto-fail, baseline wins the Pareto comparison.
|
||||
- **Scratchpad cache conflict**: two sub-agents write different values for the same entity key — last-writer-wins with a logged warning; the answer reflects the final state.
|
||||
- **Single-agent baseline also fails**: this is a valid outcome and means the question is genuinely beyond frontier model capability; the report marks the question as such and the benchmark still reports its tokens.
|
||||
- **Learning tier plateau**: if epoch 5 is not strictly better than epoch 1, the report raises a warning and the test fails.
|
||||
- **Model API rate limit**: the harness retries with exponential backoff, up to 3 attempts per call.
|
||||
- **Claude Agent SDK unavailable**: falls back to direct Anthropic SDK calls with manual tool-use formatting.
|
||||
|
||||
## Requirements *(mandatory)*
|
||||
|
||||
### Functional Requirements
|
||||
|
||||
**Harness Core**
|
||||
|
||||
- **FR-001**: The harness MUST accept a `--mode` flag with values `single-shot`, `trio`, or `learning`.
|
||||
- **FR-002**: The harness MUST read a configuration file defining agent pool composition (model tiers, initial skill cards).
|
||||
- **FR-003**: The harness MUST talk to SynapBus via the existing MCP protocol using the 016 marketplace tools.
|
||||
- **FR-004**: The harness MUST record per-run metrics in a structured JSON file under `results/`.
|
||||
|
||||
**Dataset Preparation**
|
||||
|
||||
- **FR-005**: A setup script MUST download the MuSiQue-Ans dev set from the canonical GitHub release.
|
||||
- **FR-006**: A curation script MUST select three questions from the 4-hop subset that share a pivot bridge entity and write them to `benchmark/trio.jsonl` with gold answers and decomposition DAGs.
|
||||
- **FR-007**: The curation rule MUST be deterministic with a fixed seed so the trio is reproducible.
|
||||
|
||||
**Agent Runner**
|
||||
|
||||
- **FR-008**: Agents MUST run as isolated Python processes using the Claude Agent SDK where available; when the SDK is unavailable, direct Anthropic SDK calls MUST be used as fallback.
|
||||
- **FR-009**: Each agent MUST publish a capability manifest to SynapBus wiki at startup.
|
||||
- **FR-010**: Each agent MUST poll the auction channel and submit bids when it sees a task in a declared domain.
|
||||
- **FR-011**: Each agent MUST respect the max_budget_tokens of its awarded tasks and hard-stop at 100% of budget.
|
||||
|
||||
**Baseline**
|
||||
|
||||
- **FR-012**: The harness MUST run a single-agent baseline for every question, passing all 20 distractor paragraphs plus the question in one prompt with chain-of-thought instructions.
|
||||
- **FR-013**: The baseline MUST record its token count and final F1 so the Pareto comparison is apples-to-apples.
|
||||
|
||||
**Scoring**
|
||||
|
||||
- **FR-014**: For each question, the harness MUST compute exact-match F1 against the gold answer string after normalization (lowercasing, punctuation removal, article stripping).
|
||||
- **FR-015**: The harness MUST compute decomposition-F1 by comparing the marketplace's actual sub-questions with the gold DAG.
|
||||
- **FR-016**: The harness MUST record total tokens consumed per run, separated by orchestrator, sub-agents, and baseline.
|
||||
- **FR-017**: The harness MUST compute tokens-per-correct-answer as `total_tokens / max(correct_count, 1)`.
|
||||
- **FR-018**: The harness MUST report a Pareto verdict: `PASS` if the marketplace is strictly northwest of the baseline, `FAIL` otherwise.
|
||||
|
||||
**Report Generation**
|
||||
|
||||
- **FR-019**: After every run, the harness MUST generate a self-contained HTML report at `results/{run_id}.html` showing the question, decomposition, bids, award, answer, gold, tokens, Pareto plot, and verdict.
|
||||
- **FR-020**: The report MUST be readable without requiring any external server or CDN.
|
||||
|
||||
**Learning Tier (P3, deferred for MVP)**
|
||||
|
||||
- **FR-021**: In learning mode, the harness MUST run the trio for N epochs with reputation and skill-card state persisting across epochs.
|
||||
- **FR-022**: In learning mode, the scratchpad MUST reset between epochs.
|
||||
- **FR-023**: The learning report MUST plot tokens-per-correct-answer across epochs.
|
||||
|
||||
### Key Entities
|
||||
|
||||
- **Question record**: One entry in trio.jsonl. Contains `id`, `question`, `gold_answer`, `gold_decomposition` (list of sub-questions), `distractor_paragraphs` (20 Wikipedia snippets), `pivot_entity`.
|
||||
- **Agent config**: One agent in the pool. Contains `name`, `model_id`, `system_prompt_template`, `skill_card_markdown`, `domains`, `initial_confidence_per_domain`, `initial_cost_per_domain`.
|
||||
- **Run result**: One benchmark run output. Contains `run_id`, `timestamp`, `mode`, `per_question_results`, `baseline_results`, `pareto_verdict`, `total_tokens`, `wall_time_seconds`.
|
||||
- **Scratchpad entry**: One cached `(entity, attribute, value)` tuple with a `cache_hit_count` counter.
|
||||
|
||||
## Success Criteria *(mandatory)*
|
||||
|
||||
### Measurable Outcomes
|
||||
|
||||
- **SC-001**: On a fresh SynapBus instance with 016 MVP loaded, `python benchmark/run.py --mode single-shot --question q1` completes successfully within 10 minutes wall-time and 150k total tokens.
|
||||
- **SC-002**: The Pareto verdict on the single-shot run is `PASS` — the marketplace lands strictly northwest of the Sonnet-baseline point.
|
||||
- **SC-003**: The HTML report is generated, is self-contained (no external asset fetches), and visibly shows the question, both answers, tokens, and Pareto plot.
|
||||
- **SC-004**: The benchmark can be re-run deterministically on the same machine and produce the same Pareto verdict (token counts may vary ±5% due to sampling, but verdict must be stable).
|
||||
- **SC-005**: The benchmark identifies and reports per-agent, per-domain reputation entries after the run, writable back to SynapBus.
|
||||
- **SC-006**: In trio mode, the scratchpad cache hit count is greater than zero — the dedup primitive is observably exercised.
|
||||
- **SC-007** (deferred): In learning mode, tokens-per-correct-answer declines from epoch 1 to epoch 5 by at least 15%.
|
||||
|
||||
## Assumptions
|
||||
|
||||
- The 016-agent-marketplace MVP is implemented before this benchmark runs, or this benchmark provides a mock marketplace stub for early validation.
|
||||
- MuSiQue-Ans is available at its canonical GitHub release URL and is downloadable without authentication.
|
||||
- Anthropic API access is available via `ANTHROPIC_API_KEY` environment variable.
|
||||
- Claude Agent SDK may or may not be available in the target environment; the harness degrades gracefully to direct Anthropic SDK if not.
|
||||
- Tokens reported by the Anthropic SDK are trusted as ground truth.
|
||||
- F1 computed via string normalization is sufficient for MVP; semantic similarity matching is future work.
|
||||
- The MuSiQue license (CC BY 4.0) permits redistribution of the curated trio file as a derivative work.
|
||||
|
||||
## Out of Scope
|
||||
|
||||
- Training any model or fine-tuning.
|
||||
- Non-English questions (MuSiQue is English-only).
|
||||
- Multi-modal questions (images, audio).
|
||||
- FRAMES cross-eval — scheduled as a follow-up once MuSiQue pipeline is stable.
|
||||
- Adversarial or byzantine agent behavior; agents in the pool are assumed cooperative.
|
||||
- Replacing the MuSiQue corpus with a live Wikipedia API.
|
||||
- Multi-run statistical power analysis; N=3 is mechanism verification only.
|
||||
Reference in New Issue
Block a user