#!/usr/bin/env python3 # -*- coding: utf-8 -*- """Build the prepared 英汉词典 resource for LexGo (#40). This is tooling, not product runtime: the Go server only imports the prepared ZIP. Two pinned sources are combined: * ECDICT StarDict (MIT) -> Chinese glosses and a fallback phonetic notation * CMUdict (BSD-2) -> ARPAbet phonemes transcribed to IPA The subset is limited to lowercase single-word headwords that either exist in the WordNet lemma list (pinned in server/wordnet-resource.json) or carry an ECDICT exam tag, which keeps the resource at a few MiB instead of the full 340 万条. Everything is verified: source sha256, subset size within the pinned tolerance and the phonetic character set (no character outside the IPA alphabet may survive). Usage: python scripts/dict_prepare.py # download + build into .local/dictionaries python scripts/dict_prepare.py --cache-dir DIR # reuse already downloaded sources python scripts/dict_prepare.py --keep-going # still write the artifact if counts drift """ from __future__ import annotations import argparse import gzip import hashlib import json import re import struct import sys import urllib.error import urllib.parse import urllib.request import zipfile from pathlib import Path ROOT = Path(__file__).resolve().parents[1] PIN_FILE = ROOT / "server" / "zh-dictionary-resource.json" DEFAULT_OUT = ROOT / ".local" / "dictionaries" / "zh-dict-v1.zip" DEFAULT_CACHE = ROOT / ".local" / "dictionaries" / "cache" EXAM_TAG_RE = re.compile(r"\((?:[^)]*?(研|四|六|托|专|雅|高|中|初))[^)]*?\)") PHONETIC_RE = re.compile(r"^\*?\[([^\]\u4e00-\u9fff]*)\]") CJK_RE = re.compile(r"[\u3400-\u4dbf\u4e00-\u9fff\uf900-\ufaff]") SPACE_RE = re.compile(r"\s+") # ARPAbet (CMUdict, American English) -> IPA, a fixed documented convention. ARPABET_CONSONANTS = { "B": "b", "CH": "t͡ʃ", "D": "d", "DH": "ð", "F": "f", "G": "ɡ", "HH": "h", "JH": "d͡ʒ", "K": "k", "L": "l", "M": "m", "N": "n", "NG": "ŋ", "P": "p", "R": "ɹ", "S": "s", "SH": "ʃ", "T": "t", "TH": "θ", "V": "v", "W": "w", "Y": "j", "Z": "z", "ZH": "ʒ", } ARPABET_VOWELS = { "AA": ("ɑ", "ː"), "AE": ("æ", ""), "AH": ("ʌ", ""), "AO": ("ɔ", "ː"), "AW": ("a", "ʊ"), "AY": ("a", "ɪ"), "EH": ("e", ""), "ER": ("ɜ", "ː"), "EY": ("e", "ɪ"), "IH": ("ɪ", ""), "IY": ("i", "ː"), "OW": ("o", "ʊ"), "OY": ("ɔ", "ɪ"), "UH": ("ʊ", ""), "UW": ("u", "ː"), } STRESS_MARKS = {"1": "ˈ", "2": "ˌ", "0": ""} # Only character-level substitutions whose meaning is verifiable from the data. ECDICT_CHAR_MAP = { "'": "ˈ", ",": "ˌ", ".": "ˌ", "ˊ": "ˈ", ":": "ː", "ә": "ə", "ε": "e", # ә is U+04D9, ε U+03B5 " ": "", "-": "", "=": "", ";": "", "^": "", # ^ is a stray separator mid-word } IPA_ALPHABET = set( "abcdefghijklmnopqrstuvwxyz" "æɑɒɔəɜɛɪʊʌʃʒθðŋɹɡɚɝt͡ʃd͡ʒ" "ːˈˌ" ) def log(message: str) -> None: print(message, flush=True) def digest(path: Path) -> str: hasher = hashlib.sha256() with path.open("rb") as handle: for block in iter(lambda: handle.read(1024 * 1024), b""): hasher.update(block) return hasher.hexdigest() def download(source: dict, cache_dir: Path) -> Path: """Download a pinned source, trying every mirror, and verify sha256 + size.""" name = source["name"] target = cache_dir / source_filename(source["source"]) urls = [source["source"]] + list(source.get("mirrors", [])) if target.exists() and digest(target) == source["sha256"]: log(f"[cache] {name}: {target.name} ({target.stat().st_size} 字节, sha256 匹配)") return target last_error = None for url in urls: log(f"[download] {name}: {url}") try: request = urllib.request.Request(url, headers={"User-Agent": "lexgo-dict-prepare/1"}) with urllib.request.urlopen(request, timeout=300) as response, target.open("wb") as handle: while True: block = response.read(1024 * 256) if not block: break handle.write(block) except (urllib.error.URLError, TimeoutError, OSError) as error: last_error = error log(f"[download] {name}: 失败 {type(error).__name__}: {error}") continue found = digest(target) if found == source["sha256"]: log(f"[download] {name}: 完成 {target.stat().st_size} 字节, sha256 匹配") return target log(f"[download] {name}: sha256 不匹配 ({found[:16]}… != {source['sha256'][:16]}…),尝试下一个通道") raise SystemExit(f"{name}: 所有通道都无法取得通过校验的资源(最后错误 {last_error})") def source_filename(url: str) -> str: return Path(urllib.parse.urlparse(url).path).name def wordnet_lemmas(archive: Path) -> set: lemmas = set() with zipfile.ZipFile(archive) as bundle: for name in bundle.namelist(): if "/index." not in name: continue for line in bundle.read(name).decode("latin-1").splitlines(): if not line.strip() or line.startswith(" "): continue lemmas.add(line.split(" ", 1)[0].lower().replace("_", " ")) return lemmas def cmudict_entries(path: Path) -> dict: entries = {} for line in path.read_text(encoding="latin-1").splitlines(): if line.startswith(";;;") or not line.strip(): continue head, _, rest = line.partition(" ") head = head.split("(", 1)[0].lower() if head and rest.strip() and head not in entries: entries[head] = rest.split() return entries def stardict_entries(archive: Path): with zipfile.ZipFile(archive) as bundle: names = bundle.namelist() ifo_name = next((n for n in names if n.endswith(".ifo")), None) idx_name = next((n for n in names if n.endswith(".idx")), None) dict_name = next((n for n in names if n.endswith(".dict")), None) if not (ifo_name and idx_name and dict_name): raise SystemExit("ECDICT 归档缺少 .ifo/.idx/.dict") ifo = bundle.read(ifo_name).decode("utf-8", "replace") sequence = "" for line in ifo.splitlines(): if line.startswith("sametypesequence="): sequence = line.split("=", 1)[1].strip() if sequence != "m": raise SystemExit(f"未预期的 StarDict sametypesequence: {sequence!r}(只处理纯文本 m)") index = bundle.read(idx_name) payload = bundle.read(dict_name) offset = 0 while offset < len(index): end = index.index(b"\x00", offset) word = index[offset:end].decode("utf-8", "replace") start, size = struct.unpack(">II", index[end + 1:end + 9]) yield word, payload[start:start + size].decode("utf-8", "replace") offset = end + 9 def arpabet_to_ipa(phonemes) -> str: vowel_count = sum(1 for phoneme in phonemes if phoneme.rstrip("012").strip() in ARPABET_VOWELS) parts = [] for phoneme in phonemes: stress = "" base = phoneme if phoneme and phoneme[-1].isdigit(): base, digit = phoneme[:-1], phoneme[-1] stress = STRESS_MARKS.get(digit, "") if base in ARPABET_VOWELS and vowel_count == 1: # A monosyllable carries no stress mark in IPA (cat is /kæt/, not /kˈæt/). stress = "" if base in ARPABET_CONSONANTS: parts.append(ARPABET_CONSONANTS[base]) elif base in ARPABET_VOWELS: head, tail = ARPABET_VOWELS[base] if stress == "": # Unstressed vowels are short: curiosity ends /əti/, not /ətiː/. tail = "" if base == "AH": head = "ə" elif base == "ER": head = "ɚ" parts.append(stress + head + tail) else: return "" return "".join(parts) def normalise_ecdict_phonetic(raw: str) -> str: text = raw.strip() # A phonetic that starts with the stray caret has lost its first sound (grok is stored as # "^rɔk"): showing "rɔk" would be a wrong transcription, so the whole value is dropped. if text.startswith("^"): return "" for old, new in ECDICT_CHAR_MAP.items(): text = text.replace(old, new) return SPACE_RE.sub("", text) def build_subset(lemmas: set, exam_entries) -> dict: subset = {} for word, body in exam_entries: lower = word.lower() if " " in word or not lower.isalpha() or not lower.islower() or lower in subset: continue if lower not in lemmas and not EXAM_TAG_RE.search(body): continue subset[lower] = body return subset def translation_of(body: str) -> str: lines = [] for line in body.splitlines(): stripped = PHONETIC_RE.sub("", line).strip() if not stripped or not CJK_RE.search(stripped): continue # "(研四六托 4518/5059)" is an exam-list rank and "[时态] guarded, guarding" is a word-form # list: neither is a meaning, so they stay out of the gloss. if stripped.startswith("(") and stripped.endswith(")"): continue if stripped.startswith("[时态]"): continue lines.append(stripped) return "\n".join(lines) def main() -> int: parser = argparse.ArgumentParser(description="构建 LexGo 英汉词典资源 (#40)") parser.add_argument("--out", default=str(DEFAULT_OUT)) parser.add_argument("--cache-dir", default=str(DEFAULT_CACHE)) parser.add_argument("--keep-going", action="store_true", help="计数漂移时仍产出,只在 manifest 标记") args = parser.parse_args() pin = json.loads(PIN_FILE.read_text(encoding="utf-8")) sources = {source["role"]: source for source in pin["sources"]} cache_dir = Path(args.cache_dir) cache_dir.mkdir(parents=True, exist_ok=True) ecdict_zip = download(sources["chinese-gloss"], cache_dir) cmudict_file = download(sources["ipa-phonetic"], cache_dir) wordnet_zip = download(sources["subset-wordlist"], cache_dir) log("[parse] WordNet lemma 词表") lemmas = wordnet_lemmas(wordnet_zip) log(f"[parse] WordNet lemma: {len(lemmas)}") log("[parse] CMUdict 音素") cmu = cmudict_entries(cmudict_file) log(f"[parse] CMUdict 词条: {len(cmu)}") log("[build] 裁剪 ECDICT 子集") subset = build_subset(lemmas, stardict_entries(ecdict_zip)) log(f"[build] 子集词条: {len(subset)}") records = [] stats = {"cmudict": 0, "ecdict": 0, "none": 0, "droppedPhonetic": 0, "empty": 0} offenders = {} for word in sorted(subset): body = subset[word] translation = translation_of(body) phonetic = "" source_name = "" if word in cmu: phonetic = arpabet_to_ipa(cmu[word]) source_name = "cmudict" if phonetic else "" if not phonetic: match = PHONETIC_RE.match(body) candidate = normalise_ecdict_phonetic(match.group(1)) if match else "" if candidate: phonetic = candidate source_name = "ecdict" if phonetic: unknown = sorted({char for char in phonetic if char not in IPA_ALPHABET}) if unknown: stats["droppedPhonetic"] += 1 for char in unknown: offenders[char] = offenders.get(char, 0) + 1 phonetic = "" source_name = "" if not translation and not phonetic: # An entry that lost both its gloss and its phonetic answers nothing, so it is dropped # instead of occupying a lookup slot with a bare headword. stats["empty"] += 1 continue stats[source_name or "none"] += 1 records.append({ "w": word, "t": translation, "p": phonetic, "ps": source_name, }) if offenders: log(f"[check] 丢弃的非 IPA 字符: {sorted((c, n) for c, n in offenders.items())}") if stats["droppedPhonetic"]: log(f"[check] 因字符不合法丢弃音标 {stats['droppedPhonetic']} 条(不做语义猜测)") if stats["empty"]: log(f"[check] 既无释义也无音标丢弃 {stats['empty']} 条") expected = pin["subset"]["expectedEntries"] tolerance = pin["subset"]["toleranceEntries"] drift = abs(len(records) - expected) log(f"[check] 条目数 {len(records)}(pin 期望 {expected} ± {tolerance})") if drift > tolerance and not args.keep_going: raise SystemExit("子集条目数超出 pin 容差,拒绝产出;确认数据源变化后用 --keep-going 重新评估") if len(records) == 0: raise SystemExit("子集为空,拒绝产出") manifest = { "format": pin["format"], "version": pin["version"], "language": pin["language"], "provider": pin["provider"], "entries": len(records), "phonetic": {"cmudict": stats["cmudict"], "ecdict": stats["ecdict"], "none": stats["none"]}, "droppedPhonetic": stats["droppedPhonetic"], "droppedEmpty": stats["empty"], "subsetRule": pin["subset"]["rule"], "sources": [ {"name": s["name"], "role": s["role"], "sha256": s["sha256"], "license": s["license"]} for s in pin["sources"] ], "preparedBy": "scripts/dict_prepare.py", } destination = Path(args.out) destination.parent.mkdir(parents=True, exist_ok=True) with zipfile.ZipFile(destination, "w", zipfile.ZIP_DEFLATED) as bundle: payload = "\n".join(json.dumps(record, ensure_ascii=False, sort_keys=True) for record in records) bundle.writestr("entries.jsonl.gz", gzip.compress(payload.encode("utf-8"))) bundle.writestr("manifest.json", json.dumps(manifest, ensure_ascii=False, indent=2)) log(f"[write] {destination} ({destination.stat().st_size} 字节)") log(f"[write] sha256 {digest(destination)}") log(f"[stats] CMUdict IPA {stats['cmudict']} / ECDICT 记法 {stats['ecdict']} / 无音标 {stats['none']}") return 0 if __name__ == "__main__": sys.exit(main())