"""Merges the AI rewrite back into assets/words/wordbank.json. Anything the model returned is checked before it lands: a gloss that is empty, parenthesised or sentence-long is worse than the word book's version, so the old value is kept and reported instead of being written over. """ import json import os import re import sys import tempfile ROOT = os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) BANK = os.path.join(ROOT, 'assets/words/wordbank.json') BAD_GLOSS = re.compile(r'[()()\[\]【】<>]') def usable_gloss(text): text = (text or '').strip() return bool(text) and len(text) <= 10 and not BAD_GLOSS.search(text) def mentions(word, sentence): """Whether the sentence visibly uses the word. This cannot be a hard check: `teach` shows up as "taught" and `bus stop` as two words, so a prefix match has no way to confirm them. It only flags a sentence as worth a human glance -- nothing is dropped over it. """ parts = [re.sub(r'[^a-z]', '', part) for part in word.lower().split()] stem = max(parts, key=len)[:3] return bool(stem) and stem in sentence.lower() def main(): cache_path = os.environ.get( 'ENRICH_CACHE', os.path.join(tempfile.gettempdir(), 'enrich_cache.jsonl') ) rows = {} with open(cache_path) as handle: for line in handle: row = json.loads(line) rows[row['id']] = row with open(BANK) as handle: bank = json.load(handle) kept = {'gloss': 0, 'example': 0, 'missing': 0, 'bad_gloss': 0, 'no_example': 0} unclear = [] for word in bank['words']: row = rows.get(word['id']) if row is None: kept['missing'] += 1 continue if usable_gloss(row.get('zh')): word['zh'] = row['zh'].strip() more = (row.get('more') or '').strip() senses = [s for s in more.split(';') if usable_gloss(s) and s != word['zh']] if senses: word['more'] = ';'.join(senses[:2]) else: word.pop('more', None) kept['gloss'] += 1 else: kept['bad_gloss'] += 1 sentence = (row.get('ex') or '').strip() if not sentence: kept['no_example'] += 1 continue word['ex'] = sentence word['exZh'] = (row.get('ex_zh') or '').strip() kept['example'] += 1 if not mentions(word['en'], sentence): unclear.append('%s | %s' % (word['en'], sentence)) print(json.dumps(kept, indent=2)) if unclear: print('%d sentences to eyeball:' % len(unclear)) for line in unclear: print(' ' + line) if '--dry-run' in sys.argv: return bank['source'] = ( 'CEFR-J Vocabulary Profile 1.5 (levels) + 公开词书 (音标) + AI 重写 (释义与例句)' ) with open(BANK, 'w') as handle: json.dump(bank, handle, ensure_ascii=False, separators=(',', ':')) handle.write('\n') if __name__ == '__main__': main()