"""Builds an independent word bank: CEFR-J decides which words and at what level, the downloaded word books supply the IPA and the Chinese gloss. CEFR-J carries no glosses at all, so nothing here overrides it -- a word only enters the bank if the level pool already contains it. """ import csv, glob, json, os, re, sys import openpyxl POOLS = '/Users/shenlei/Work/English/kouyu_english/tool/courses/pools' # The books write parts of speech as `n.` / `vt.`; CEFR-J spells them out. POS_MAP = { 'n': 'noun', 'pron': 'pronoun', 'v': 'verb', 'vt': 'verb', 'vi': 'verb', 'aux': 'verb', 'modal': 'verb', 'adj': 'adjective', 'adv': 'adverb', 'prep': 'preposition', 'conj': 'conjunction', 'art': 'determiner', 'det': 'determiner', 'num': 'number', 'int': 'interjection', } def load_pool(): """{word form: (headword, pos set, level)} for A1..B1, content words only.""" pool = {} for level in ('A1', 'A2', 'B1'): for row in csv.DictReader(open(f'{POOLS}/{level.lower()}-word-pool.csv')): if row.get('status') != 'pool' or row.get('function_word') == 'yes': continue head = row['headword'].strip() for form in re.split(r'[/,]', head): form = form.strip().lower() if not form: continue entry = pool.setdefault(form, {'headword': head, 'pos': set(), 'level': level}) if row.get('pos'): entry['pos'].add(row['pos'].strip()) return pool def clean_gloss(raw, wanted_pos): """Picks one part of speech out of a stacked gloss and splits it into senses a beginner needs. The books ship dictionary dumps: `apple` arrives as "苹果,苹果树,苹果似的东西;[美俚]炸弹,手榴弹...".""" if not raw: return '' lines = [l.strip() for l in str(raw).split('\n') if l.strip()] picked = None for line in lines: m = re.match(r'^([a-z]+)\.\s*(.+)$', line, re.I) if not m: continue pos = POS_MAP.get(m.group(1).lower()) if pos and wanted_pos and pos in wanted_pos: picked = m.group(2) break if picked is None: picked = m.group(2) if picked is None: picked = lines[0] senses = [] for sense in re.split(r'[;;]', picked): sense = sense.strip() # `[美俚]炸弹` and `<美>支票` are register labels on a sense a learner # will never need; drop the whole sense, not just the label. if re.match(r'^\s*[\[<【]', sense): continue sense = re.sub(r'[((][^))]*[))]', '', sense) sense = re.sub(r'[\[【][^\]】]*[\]】]', '', sense) # "苹果,苹果树,苹果似的东西" is one sense listed three ways. sense = re.split(r'[,,]', sense)[0].strip(' 。.、') if sense: senses.append(sense) return list(dict.fromkeys(senses))[:3] def main(): pool = load_pool() bank, seen, skipped = {}, 0, 0 for path in sorted(glob.glob('src/*.xlsx')): source = os.path.basename(path) wb = openpyxl.load_workbook(path, read_only=True) for row in wb.worksheets[0].iter_rows(min_row=2, values_only=True): if not row or not row[0]: continue seen += 1 word = str(row[0]).strip() key = word.lower() hit = pool.get(key) if hit is None: skipped += 1 continue senses = clean_gloss(row[3] if len(row) > 3 else '', hit['pos']) if not senses: skipped += 1 continue uk = str(row[1] or '').strip() us = str(row[2] or '').strip() prev = bank.get(key) # Keep the first hit, but let a later book fill in a missing IPA. if prev: if not prev['ipa'] and (us or uk): prev['ipa'] = us or uk continue bank[key] = { 'id': 'V-' + re.sub(r'[^a-z0-9]+', '-', key).strip('-'), 'en': word, # One sense is what a quiz option should say; the rest are kept # for the word's detail line. 'zh': senses[0], 'more': ';'.join(senses[1:]), 'ipa': us or uk, 'level': hit['level'], 'source': source, } wb.close() out = [bank[k] for k in sorted(bank)] json.dump(out, open('wordbank.json', 'w'), ensure_ascii=False, indent=1) from collections import Counter by_level = Counter(w['level'] for w in out) print(f'读入词条 {seen},未匹配/丢弃 {skipped}') print(f'词库产出 {len(out)} ' + ' '.join(f'{k} {by_level[k]}' for k in ('A1', 'A2', 'B1'))) for level in ('A1', 'A2', 'B1'): total = len({f for f, v in pool.items() if v['level'] == level}) print(f' {level} 词池覆盖 {by_level[level]}/{total} = {by_level[level]/total:.0%}') print(f' 无音标 {sum(1 for w in out if not w["ipa"])}') print('\n样例:') for w in out[:8]: print(f' {w["en"]:12} {w["ipa"]:18} {w["zh"]:10} [{w["level"]}] {w["more"]}') main()