#!/usr/bin/env python3 """Migrate section INDEX.md to CANON v1: | # | Group | EN title | Author(s) | Status | File | Usage: python3 tools/migrate_indexes.py
[--dry-run]""" import re, sys, os, glob NAMES = {'01-fundamentals': 'Fundamentals', '02-dreams': 'Dreams', '03-myths-fairy-tales': 'Myths & Fairy Tales', '04-pictures': 'Pictures'} COLMAP = {'block': 'group', 'subsection': 'group', 'group': 'group', 'en title': 'title', 'title': 'title', 'item': 'title', 'book': 'title', 'author(s)': 'author', 'author(s) ': 'author', 'authors': 'author', 'author': 'author', 'status': 'status', 'file': 'file'} def main(): sec = sys.argv[1] dry = '--dry-run' in sys.argv p = f'sections/{sec}/INDEX.md' c = open(p).read() lines = c.split('\n') # find table ti = next(i for i, l in enumerate(lines) if l.startswith('| #') or l.startswith('|#')) header = [x.strip() for x in lines[ti].strip().strip('|').split('|')] colidx = {} for i, h in enumerate(header): key = COLMAP.get(h.lower()) if key: colidx[key] = i rows, ti_end = [], ti for i in range(ti+1, len(lines)): if not lines[i].startswith('|'): ti_end = i break cells = [x.strip() for x in lines[i].strip().strip('|').split('|')] if len(cells) >= 2 and not (set(cells[0]) <= set('- ') or cells[0] == '#'): rows.append(cells) else: ti_end = len(lines) intro = '\n'.join(lines[1:ti]).strip() tail = '\n'.join(lines[ti_end:]).strip() out_rows = [] for cells in rows: num = cells[0] group = cells[colidx.get('group', 1)] if 'group' in colidx else '' group = group or 'β€”' title = cells[colidx.get('title')] author = cells[colidx.get('author')] if 'author' in colidx else '' status = cells[colidx.get('status')] if 'status' in colidx else '' filecol = cells[colidx.get('file')] if 'file' in colidx else '' m = re.match(r'\[(.+?)\]\((.+?)\)', title) if m: title, filecol = m.group(1), m.group(2) if not filecol: cand = glob.glob(f'sections/{sec}/{num}-*.md') filecol = os.path.basename(cand[0]) if cand else '' elif not filecol.startswith(num + '-'): cand = glob.glob(f'sections/{sec}/{num}-{filecol}') filecol = os.path.basename(cand[0]) if cand else f'{num}-{filecol}' mark = status[:1] if mark not in 'βœ…πŸ”ΆβŒβ¬œπŸ”Ž': mm = re.search(r'[βœ…πŸ”ΆβŒβ¬œπŸ”Ž]', status) mark = mm.group(0) if mm else '?' out_rows.append(f'| {num} | {group} | {title} | {author} | {mark} | {filecol} |') new = f'# Section {sec[:2]} β€” {NAMES[sec]}: INDEX\n\n{intro}\n\n' new += '| # | Group | EN title | Author(s) | Status | File |\n|---|-------|----------|-----------|--------|------|\n' new += '\n'.join(out_rows) + '\n' if tail: # keep trailing sections (## ...) only tlines = tail.split('\n') i0 = next((i for i, l in enumerate(tlines) if l.startswith('## ')), None) if i0 is not None: new += '\n' + '\n'.join(tlines[i0:]).strip() + '\n' if dry: print(new) else: open(p, 'w').write(new) # verify against cards bad = [] for r in out_rows: num = r.split('|')[1].strip() mark = r.split('|')[5].strip() card = glob.glob(f'sections/{sec}/{num}-*.md') if not card: bad.append(f'{num}: no card'); continue cm = re.search(r'\*\*Status:\*\* (.)', open(card[0]).read()) if cm and cm.group(1) != mark: bad.append(f'{num}: index {mark} != card {cm.group(1)}') print(f'wrote {p} ({len(out_rows)} rows); mismatches: {bad if bad else 0}') if __name__ == '__main__': main()