Files
synapbus/benchmark/score.py
T
Algis DumbrisandClaude Opus 4.6 02b8548eac feat(017): MuSiQue benchmark harness — marketplace stub, mixed-tier agents, Pareto scoring, HTML report
MVP implementation of the MuSiQue multi-agent benchmark (spec 017):

- benchmark/setup.py: downloads musique_v1.0.zip from the canonical Google
  Drive source (mirrors upstream download_data.sh). Idempotent.
- benchmark/curate.py: deterministic selection of 3 4-hop questions from
  the dev set sharing a US pivot entity; writes benchmark/trio.jsonl.
- benchmark/marketplace.py: in-process 016-marketplace stub with
  post_auction / bid / award / mark_done / query_reputation and a
  domain-scoped reputation ledger. Designed for mechanical swap to real
  SynapBus MCP tools.
- benchmark/agents.py: HaikuAgent + SonnetAgent, using the official
  anthropic SDK (no Claude Agent SDK, no subprocesses). Models pinned
  to claude-haiku-4-5-20251001 and claude-sonnet-4-6.
- benchmark/baseline.py: single Sonnet call with all 20 distractors
  plus chain-of-thought.
- benchmark/score.py: SQuAD-style normalized F1 + Pareto verdict
  (strictly northwest = PASS).
- benchmark/run.py: main entry. --mode single-shot, --question, --dry-run.
- benchmark/report.py: self-contained HTML with inline SVG scatter plot.
- benchmark/trio.jsonl: curated reproducible trio (all three converge on
  "Treaty of Paris" US territory cession).

Verified with benchmark/run.py --dry-run end-to-end; all 8 files
py_compile clean. Real-token execution is deferred to the user's main
session.

Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
2026-04-11 15:12:19 +03:00

87 lines
2.5 KiB
Python

#!/usr/bin/env python3
"""
Scoring utilities for the MuSiQue benchmark.
- Normalized exact-match F1 (SQuAD-style): lowercase, strip articles,
strip punctuation, collapse whitespace.
- Pareto verdict: the marketplace point is strictly northwest of the
baseline iff it uses fewer tokens AND has F1 >= baseline, with at
least one of those strict.
"""
from __future__ import annotations
import re
import string
from collections import Counter
from typing import Any
_ARTICLE_RE = re.compile(r"\b(a|an|the)\b", re.IGNORECASE)
def normalize(text: str) -> str:
if text is None:
return ""
text = text.lower()
text = _ARTICLE_RE.sub(" ", text)
text = "".join(ch for ch in text if ch not in string.punctuation)
text = " ".join(text.split())
return text
def f1(prediction: str, gold: str) -> float:
pred_tokens = normalize(prediction).split()
gold_tokens = normalize(gold).split()
if not pred_tokens and not gold_tokens:
return 1.0
if not pred_tokens or not gold_tokens:
return 0.0
common = Counter(pred_tokens) & Counter(gold_tokens)
overlap = sum(common.values())
if overlap == 0:
return 0.0
precision = overlap / len(pred_tokens)
recall = overlap / len(gold_tokens)
return 2 * precision * recall / (precision + recall)
def exact_match(prediction: str, gold: str) -> bool:
return normalize(prediction) == normalize(gold)
def best_f1_against_aliases(
prediction: str, gold: str, aliases: list[str] | None = None
) -> float:
candidates = [gold] + list(aliases or [])
return max(f1(prediction, c) for c in candidates if c is not None)
def pareto_verdict(
market_tokens: int,
market_f1: float,
baseline_tokens: int,
baseline_f1: float,
) -> dict[str, Any]:
"""
Strictly northwest of baseline: fewer tokens AND higher-or-equal F1,
with at least one strict inequality.
"""
tokens_better = market_tokens < baseline_tokens
quality_atleast = market_f1 >= baseline_f1
quality_better = market_f1 > baseline_f1
strictly_nw = (
(tokens_better and quality_atleast)
or (quality_better and market_tokens <= baseline_tokens)
)
return {
"verdict": "PASS" if strictly_nw else "FAIL",
"strictly_northwest": strictly_nw,
"market_tokens": int(market_tokens),
"market_f1": float(market_f1),
"baseline_tokens": int(baseline_tokens),
"baseline_f1": float(baseline_f1),
"tokens_delta": int(market_tokens - baseline_tokens),
"f1_delta": float(market_f1 - baseline_f1),
}