MVP implementation of the MuSiQue multi-agent benchmark (spec 017): - benchmark/setup.py: downloads musique_v1.0.zip from the canonical Google Drive source (mirrors upstream download_data.sh). Idempotent. - benchmark/curate.py: deterministic selection of 3 4-hop questions from the dev set sharing a US pivot entity; writes benchmark/trio.jsonl. - benchmark/marketplace.py: in-process 016-marketplace stub with post_auction / bid / award / mark_done / query_reputation and a domain-scoped reputation ledger. Designed for mechanical swap to real SynapBus MCP tools. - benchmark/agents.py: HaikuAgent + SonnetAgent, using the official anthropic SDK (no Claude Agent SDK, no subprocesses). Models pinned to claude-haiku-4-5-20251001 and claude-sonnet-4-6. - benchmark/baseline.py: single Sonnet call with all 20 distractors plus chain-of-thought. - benchmark/score.py: SQuAD-style normalized F1 + Pareto verdict (strictly northwest = PASS). - benchmark/run.py: main entry. --mode single-shot, --question, --dry-run. - benchmark/report.py: self-contained HTML with inline SVG scatter plot. - benchmark/trio.jsonl: curated reproducible trio (all three converge on "Treaty of Paris" US territory cession). Verified with benchmark/run.py --dry-run end-to-end; all 8 files py_compile clean. Real-token execution is deferred to the user's main session. Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
87 lines
2.5 KiB
Python
87 lines
2.5 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
Scoring utilities for the MuSiQue benchmark.
|
|
|
|
- Normalized exact-match F1 (SQuAD-style): lowercase, strip articles,
|
|
strip punctuation, collapse whitespace.
|
|
- Pareto verdict: the marketplace point is strictly northwest of the
|
|
baseline iff it uses fewer tokens AND has F1 >= baseline, with at
|
|
least one of those strict.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
import string
|
|
from collections import Counter
|
|
from typing import Any
|
|
|
|
_ARTICLE_RE = re.compile(r"\b(a|an|the)\b", re.IGNORECASE)
|
|
|
|
|
|
def normalize(text: str) -> str:
|
|
if text is None:
|
|
return ""
|
|
text = text.lower()
|
|
text = _ARTICLE_RE.sub(" ", text)
|
|
text = "".join(ch for ch in text if ch not in string.punctuation)
|
|
text = " ".join(text.split())
|
|
return text
|
|
|
|
|
|
def f1(prediction: str, gold: str) -> float:
|
|
pred_tokens = normalize(prediction).split()
|
|
gold_tokens = normalize(gold).split()
|
|
if not pred_tokens and not gold_tokens:
|
|
return 1.0
|
|
if not pred_tokens or not gold_tokens:
|
|
return 0.0
|
|
common = Counter(pred_tokens) & Counter(gold_tokens)
|
|
overlap = sum(common.values())
|
|
if overlap == 0:
|
|
return 0.0
|
|
precision = overlap / len(pred_tokens)
|
|
recall = overlap / len(gold_tokens)
|
|
return 2 * precision * recall / (precision + recall)
|
|
|
|
|
|
def exact_match(prediction: str, gold: str) -> bool:
|
|
return normalize(prediction) == normalize(gold)
|
|
|
|
|
|
def best_f1_against_aliases(
|
|
prediction: str, gold: str, aliases: list[str] | None = None
|
|
) -> float:
|
|
candidates = [gold] + list(aliases or [])
|
|
return max(f1(prediction, c) for c in candidates if c is not None)
|
|
|
|
|
|
def pareto_verdict(
|
|
market_tokens: int,
|
|
market_f1: float,
|
|
baseline_tokens: int,
|
|
baseline_f1: float,
|
|
) -> dict[str, Any]:
|
|
"""
|
|
Strictly northwest of baseline: fewer tokens AND higher-or-equal F1,
|
|
with at least one strict inequality.
|
|
"""
|
|
tokens_better = market_tokens < baseline_tokens
|
|
quality_atleast = market_f1 >= baseline_f1
|
|
quality_better = market_f1 > baseline_f1
|
|
|
|
strictly_nw = (
|
|
(tokens_better and quality_atleast)
|
|
or (quality_better and market_tokens <= baseline_tokens)
|
|
)
|
|
return {
|
|
"verdict": "PASS" if strictly_nw else "FAIL",
|
|
"strictly_northwest": strictly_nw,
|
|
"market_tokens": int(market_tokens),
|
|
"market_f1": float(market_f1),
|
|
"baseline_tokens": int(baseline_tokens),
|
|
"baseline_f1": float(baseline_f1),
|
|
"tokens_delta": int(market_tokens - baseline_tokens),
|
|
"f1_delta": float(market_f1 - baseline_f1),
|
|
}
|