Files
Algis DumbrisandClaude Opus 4.6 0e25fbcccb feat: sdk_backend + autonomous run integration
Adds benchmark/sdk_backend.py that routes model calls through either
the anthropic SDK (preferred, requires ANTHROPIC_API_KEY) or the
claude-agent-sdk as a Claude Code session fallback. agents.py and
baseline.py now go through this unified backend instead of calling
anthropic directly.

Ran benchmark/run.py --mode single-shot --question q1 end-to-end
with real Claude API calls via claude-agent-sdk. Real numbers:
- Marketplace (Haiku 4.5): 3314 tokens, F1 1.000 (exact match)
- Baseline (Sonnet 4.6): 697 tokens, F1 0.857 (penalized for "1783")
- Pareto verdict: FAIL (not strictly NW; marketplace wins quality,
  loses cost — informative failure per spec design).

Added autonomous_report.html (rich narrative with Pareto chart)
and autonomous_summary.md. All 34 Go packages still green.

Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
2026-04-11 15:29:56 +03:00

95 lines
2.4 KiB
Python

#!/usr/bin/env python3
"""
Single-agent baseline: one Anthropic API call to claude-sonnet-4-6 with
the question and all 20 distractor paragraphs plus chain-of-thought
instructions. No decomposition, no marketplace, no tools.
Returns {"answer": str, "tokens": int, "raw_text": str}.
"""
from __future__ import annotations
from typing import Any
from sdk_backend import call_model
BASELINE_MODEL = "claude-sonnet-4-6"
BASELINE_SYSTEM = """\
You are a careful multi-hop QA system. Given a question and a set of
numbered paragraphs (some irrelevant distractors), think step by step
and answer.
Output your reasoning first, then on a final line:
ANSWER: <short final answer>
"""
def _build_prompt(question: str, paragraphs: list[str]) -> str:
parts = ["Paragraphs:"]
for i, p in enumerate(paragraphs, start=1):
parts.append(f"[{i}] {p}")
parts.append("")
parts.append(f"Question: {question}")
parts.append("")
parts.append(
"Work through the reasoning step by step, then give your "
"final ANSWER: line."
)
return "\n".join(parts)
def _extract_answer(text: str) -> str:
if not text:
return ""
for line in reversed(text.splitlines()):
line = line.strip()
if line.upper().startswith("ANSWER:"):
return line.split(":", 1)[1].strip()
for line in reversed(text.splitlines()):
line = line.strip()
if line:
return line
return ""
def run_baseline(
question: str,
paragraphs: list[str],
*,
dry_run: bool = False,
max_output_tokens: int = 1024,
) -> dict[str, Any]:
prompt = _build_prompt(question, paragraphs)
if dry_run:
stub = (
"Step 1: scanning paragraphs... [dry-run stub]\n"
"Step 2: picking the most likely entity...\n"
"ANSWER: [stub baseline answer]"
)
est = max(512, len(prompt) // 4 + 128)
return {
"answer": _extract_answer(stub),
"tokens": est,
"raw_text": stub,
"model": BASELINE_MODEL,
}
result = call_model(
model=BASELINE_MODEL,
system=BASELINE_SYSTEM,
user=prompt,
max_tokens=max_output_tokens,
)
text = result["text"]
tokens = int(result["total_tokens"])
return {
"answer": _extract_answer(text),
"tokens": tokens,
"raw_text": text,
"model": BASELINE_MODEL,
}