## 独立词库 理解词原先只能跟着课程单元走,学完 A0 十课词汇量只增加约 22 个实词, 不足以解决"记不住单词"。新增一份独立词库 assets/words/wordbank.json (2748 词,A1–B1),挂进 receptiveWordRegistry 的合成单元 bank-A1/A2/B1, 完全复用理解词已有的状态机,不依赖课程进度,第一天就能用。 数据来源、许可与合成规则记在 tool/words/DATA-NOTE.md:CEFR-J 定等级、 公开词书提供音标、AI 重写全部释义并生成例句、OpenSubtitles 提供口语词频。 词书部分为 CC BY-NC-SA 4.0 且上游权利不明,仅供个人非商用; 若要分发或上架,须替换音标那一列。 ## 背单词机制 - 间隔阶梯 1/3/7/15/30/60/120 天,连续答对上一级,答错回第一级。 原先首次答对后要等 7 天才复习,正是"第二天就忘"的成因。 - 每日新词上限(10 分钟 8 个 / 20 分钟 15 个 / 30 分钟 20 个)。 阶梯第一级是次日,今天引入的新词就是明天的工作量。 - 新词按口语频率发放,不再按字母序 —— A1 从 a.m./ability 变成 no/not/know/just。 - 三个方向按层级轮转:看词(英→中)→ 听词(音→中)→ 想词(中→英)。 想词题仍是选择题,不要求产出,理解词定位不变,不进升级分母。 - 单词页独立成 tab,首页今日任务卡下方给一张认词入口卡。 ## 用法对照 课程 JSON 增加 usage 字段(when/reply/swap/confuse):一个句型用在什么场合、 对方通常怎么答、还能怎么说、跟哪个学过的句型容易混。 知道 How are you? 的意思,不等于知道它不是用来问名字的。 ## 复习流 - 当日快闪(recap)独立成队列,不占复习预算,也不计入积压。 - 只发放当日预算内的量,其余保持到期状态等下次,不悄悄丢弃或改期。 - 答错的项隔几题后回来,而不是立刻重问。 测试 296 通过,flutter analyze 干净。 Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
132 lines
5.1 KiB
Python
132 lines
5.1 KiB
Python
"""Builds an independent word bank: CEFR-J decides which words and at what
|
||
level, the downloaded word books supply the IPA and the Chinese gloss.
|
||
|
||
CEFR-J carries no glosses at all, so nothing here overrides it -- a word only
|
||
enters the bank if the level pool already contains it.
|
||
"""
|
||
import csv, glob, json, os, re, sys
|
||
import openpyxl
|
||
|
||
POOLS = '/Users/shenlei/Work/English/kouyu_english/tool/courses/pools'
|
||
|
||
# The books write parts of speech as `n.` / `vt.`; CEFR-J spells them out.
|
||
POS_MAP = {
|
||
'n': 'noun', 'pron': 'pronoun', 'v': 'verb', 'vt': 'verb', 'vi': 'verb',
|
||
'aux': 'verb', 'modal': 'verb', 'adj': 'adjective', 'adv': 'adverb',
|
||
'prep': 'preposition', 'conj': 'conjunction', 'art': 'determiner',
|
||
'det': 'determiner', 'num': 'number', 'int': 'interjection',
|
||
}
|
||
|
||
|
||
def load_pool():
|
||
"""{word form: (headword, pos set, level)} for A1..B1, content words only."""
|
||
pool = {}
|
||
for level in ('A1', 'A2', 'B1'):
|
||
for row in csv.DictReader(open(f'{POOLS}/{level.lower()}-word-pool.csv')):
|
||
if row.get('status') != 'pool' or row.get('function_word') == 'yes':
|
||
continue
|
||
head = row['headword'].strip()
|
||
for form in re.split(r'[/,]', head):
|
||
form = form.strip().lower()
|
||
if not form:
|
||
continue
|
||
entry = pool.setdefault(form, {'headword': head, 'pos': set(), 'level': level})
|
||
if row.get('pos'):
|
||
entry['pos'].add(row['pos'].strip())
|
||
return pool
|
||
|
||
|
||
def clean_gloss(raw, wanted_pos):
|
||
"""Picks one part of speech out of a stacked gloss and splits it into senses a
|
||
beginner needs. The books ship dictionary dumps: `apple` arrives as
|
||
"苹果,苹果树,苹果似的东西;[美俚]炸弹,手榴弹..."."""
|
||
if not raw:
|
||
return ''
|
||
lines = [l.strip() for l in str(raw).split('\n') if l.strip()]
|
||
picked = None
|
||
for line in lines:
|
||
m = re.match(r'^([a-z]+)\.\s*(.+)$', line, re.I)
|
||
if not m:
|
||
continue
|
||
pos = POS_MAP.get(m.group(1).lower())
|
||
if pos and wanted_pos and pos in wanted_pos:
|
||
picked = m.group(2)
|
||
break
|
||
if picked is None:
|
||
picked = m.group(2)
|
||
if picked is None:
|
||
picked = lines[0]
|
||
senses = []
|
||
for sense in re.split(r'[;;]', picked):
|
||
sense = sense.strip()
|
||
# `[美俚]炸弹` and `<美>支票` are register labels on a sense a learner
|
||
# will never need; drop the whole sense, not just the label.
|
||
if re.match(r'^\s*[\[<【]', sense):
|
||
continue
|
||
sense = re.sub(r'[((][^))]*[))]', '', sense)
|
||
sense = re.sub(r'[\[【][^\]】]*[\]】]', '', sense)
|
||
# "苹果,苹果树,苹果似的东西" is one sense listed three ways.
|
||
sense = re.split(r'[,,]', sense)[0].strip(' 。.、')
|
||
if sense:
|
||
senses.append(sense)
|
||
return list(dict.fromkeys(senses))[:3]
|
||
|
||
|
||
def main():
|
||
pool = load_pool()
|
||
bank, seen, skipped = {}, 0, 0
|
||
for path in sorted(glob.glob('src/*.xlsx')):
|
||
source = os.path.basename(path)
|
||
wb = openpyxl.load_workbook(path, read_only=True)
|
||
for row in wb.worksheets[0].iter_rows(min_row=2, values_only=True):
|
||
if not row or not row[0]:
|
||
continue
|
||
seen += 1
|
||
word = str(row[0]).strip()
|
||
key = word.lower()
|
||
hit = pool.get(key)
|
||
if hit is None:
|
||
skipped += 1
|
||
continue
|
||
senses = clean_gloss(row[3] if len(row) > 3 else '', hit['pos'])
|
||
if not senses:
|
||
skipped += 1
|
||
continue
|
||
uk = str(row[1] or '').strip()
|
||
us = str(row[2] or '').strip()
|
||
prev = bank.get(key)
|
||
# Keep the first hit, but let a later book fill in a missing IPA.
|
||
if prev:
|
||
if not prev['ipa'] and (us or uk):
|
||
prev['ipa'] = us or uk
|
||
continue
|
||
bank[key] = {
|
||
'id': 'V-' + re.sub(r'[^a-z0-9]+', '-', key).strip('-'),
|
||
'en': word,
|
||
# One sense is what a quiz option should say; the rest are kept
|
||
# for the word's detail line.
|
||
'zh': senses[0],
|
||
'more': ';'.join(senses[1:]),
|
||
'ipa': us or uk,
|
||
'level': hit['level'], 'source': source,
|
||
}
|
||
wb.close()
|
||
|
||
out = [bank[k] for k in sorted(bank)]
|
||
json.dump(out, open('wordbank.json', 'w'), ensure_ascii=False, indent=1)
|
||
|
||
from collections import Counter
|
||
by_level = Counter(w['level'] for w in out)
|
||
print(f'读入词条 {seen},未匹配/丢弃 {skipped}')
|
||
print(f'词库产出 {len(out)} ' + ' '.join(f'{k} {by_level[k]}' for k in ('A1', 'A2', 'B1')))
|
||
for level in ('A1', 'A2', 'B1'):
|
||
total = len({f for f, v in pool.items() if v['level'] == level})
|
||
print(f' {level} 词池覆盖 {by_level[level]}/{total} = {by_level[level]/total:.0%}')
|
||
print(f' 无音标 {sum(1 for w in out if not w["ipa"])}')
|
||
print('\n样例:')
|
||
for w in out[:8]:
|
||
print(f' {w["en"]:12} {w["ipa"]:18} {w["zh"]:10} [{w["level"]}] {w["more"]}')
|
||
|
||
|
||
main()
|