#!/usr/bin/env python3 """ MuSiQue dataset downloader. Downloads ``musique_v1.0.zip`` from the canonical source used by the upstream project (https://github.com/StonyBrookNLP/musique). The zip is hosted on Google Drive (file id ``1tGdADlNjWFaHLeZZGShh2IRcpO6Lv24h``); this mirrors the behavior of the project's ``download_data.sh`` which uses ``gdown`` under the hood. Idempotent — skips download if the target dev-set jsonl already exists. Run: ``python benchmark/setup.py`` """ from __future__ import annotations import os import re import sys import zipfile from pathlib import Path import requests GDRIVE_FILE_ID = "1tGdADlNjWFaHLeZZGShh2IRcpO6Lv24h" GDRIVE_URL = "https://docs.google.com/uc?export=download" DATA_DIR = Path(__file__).resolve().parent / "data" ZIP_PATH = DATA_DIR / "musique_v1.0.zip" TARGET_FILE = DATA_DIR / "musique_ans_v1.0_dev.jsonl" def _write_stream(resp: requests.Response, dest: Path) -> int: total = int(resp.headers.get("Content-Length", 0)) downloaded = 0 dest.parent.mkdir(parents=True, exist_ok=True) with open(dest, "wb") as f: for chunk in resp.iter_content(chunk_size=1024 * 1024): if not chunk: continue f.write(chunk) downloaded += len(chunk) if total: pct = 100.0 * downloaded / total print( f"\r downloading: {downloaded/1e6:6.1f} MB " f"/ {total/1e6:6.1f} MB ({pct:5.1f}%)", end="", file=sys.stderr, ) print("", file=sys.stderr) return downloaded def _download_gdrive(file_id: str, dest: Path) -> bool: """ Download a large file from Google Drive, handling the virus-scan confirmation page that Drive injects for anything over ~100 MB. """ session = requests.Session() try: resp = session.get( GDRIVE_URL, params={"id": file_id, "export": "download"}, stream=True, timeout=60, ) except requests.RequestException as exc: print(f" -> request failed: {exc}", file=sys.stderr) return False # Case 1: Drive returns the file directly (small file or cached). ctype = resp.headers.get("Content-Type", "") if "text/html" not in ctype.lower(): _write_stream(resp, dest) return dest.exists() and dest.stat().st_size > 0 # Case 2: HTML confirmation page. Extract the confirm token and/or # the form action URL. html = resp.text # Newer Drive flow: a