89 lines
3.8 KiB
Python
89 lines
3.8 KiB
Python
#!/usr/bin/env python3
|
|
"""Migrate section INDEX.md to CANON v1: | # | Group | EN title | Author(s) | Status | File |
|
|
Usage: python3 tools/migrate_indexes.py <section> [--dry-run]"""
|
|
import re, sys, os, glob
|
|
|
|
NAMES = {'01-fundamentals': 'Fundamentals', '02-dreams': 'Dreams',
|
|
'03-myths-fairy-tales': 'Myths & Fairy Tales', '04-pictures': 'Pictures'}
|
|
COLMAP = {'block': 'group', 'subsection': 'group', 'group': 'group',
|
|
'en title': 'title', 'title': 'title', 'item': 'title', 'book': 'title',
|
|
'author(s)': 'author', 'author(s) ': 'author', 'authors': 'author', 'author': 'author',
|
|
'status': 'status', 'file': 'file'}
|
|
|
|
def main():
|
|
sec = sys.argv[1]
|
|
dry = '--dry-run' in sys.argv
|
|
p = f'sections/{sec}/INDEX.md'
|
|
c = open(p).read()
|
|
lines = c.split('\n')
|
|
# find table
|
|
ti = next(i for i, l in enumerate(lines) if l.startswith('| #') or l.startswith('|#'))
|
|
header = [x.strip() for x in lines[ti].strip().strip('|').split('|')]
|
|
colidx = {}
|
|
for i, h in enumerate(header):
|
|
key = COLMAP.get(h.lower())
|
|
if key:
|
|
colidx[key] = i
|
|
rows, ti_end = [], ti
|
|
for i in range(ti+1, len(lines)):
|
|
if not lines[i].startswith('|'):
|
|
ti_end = i
|
|
break
|
|
cells = [x.strip() for x in lines[i].strip().strip('|').split('|')]
|
|
if len(cells) >= 2 and not (set(cells[0]) <= set('- ') or cells[0] == '#'):
|
|
rows.append(cells)
|
|
else:
|
|
ti_end = len(lines)
|
|
intro = '\n'.join(lines[1:ti]).strip()
|
|
tail = '\n'.join(lines[ti_end:]).strip()
|
|
out_rows = []
|
|
for cells in rows:
|
|
num = cells[0]
|
|
group = cells[colidx.get('group', 1)] if 'group' in colidx else ''
|
|
group = group or '—'
|
|
title = cells[colidx.get('title')]
|
|
author = cells[colidx.get('author')] if 'author' in colidx else ''
|
|
status = cells[colidx.get('status')] if 'status' in colidx else ''
|
|
filecol = cells[colidx.get('file')] if 'file' in colidx else ''
|
|
m = re.match(r'\[(.+?)\]\((.+?)\)', title)
|
|
if m:
|
|
title, filecol = m.group(1), m.group(2)
|
|
if not filecol:
|
|
cand = glob.glob(f'sections/{sec}/{num}-*.md')
|
|
filecol = os.path.basename(cand[0]) if cand else ''
|
|
elif not filecol.startswith(num + '-'):
|
|
cand = glob.glob(f'sections/{sec}/{num}-{filecol}')
|
|
filecol = os.path.basename(cand[0]) if cand else f'{num}-{filecol}'
|
|
mark = status[:1]
|
|
if mark not in '✅🔶❌⬜🔎':
|
|
mm = re.search(r'[✅🔶❌⬜🔎]', status)
|
|
mark = mm.group(0) if mm else '?'
|
|
out_rows.append(f'| {num} | {group} | {title} | {author} | {mark} | {filecol} |')
|
|
new = f'# Section {sec[:2]} — {NAMES[sec]}: INDEX\n\n{intro}\n\n'
|
|
new += '| # | Group | EN title | Author(s) | Status | File |\n|---|-------|----------|-----------|--------|------|\n'
|
|
new += '\n'.join(out_rows) + '\n'
|
|
if tail:
|
|
# keep trailing sections (## ...) only
|
|
tlines = tail.split('\n')
|
|
i0 = next((i for i, l in enumerate(tlines) if l.startswith('## ')), None)
|
|
if i0 is not None:
|
|
new += '\n' + '\n'.join(tlines[i0:]).strip() + '\n'
|
|
if dry:
|
|
print(new)
|
|
else:
|
|
open(p, 'w').write(new)
|
|
# verify against cards
|
|
bad = []
|
|
for r in out_rows:
|
|
num = r.split('|')[1].strip()
|
|
mark = r.split('|')[5].strip()
|
|
card = glob.glob(f'sections/{sec}/{num}-*.md')
|
|
if not card:
|
|
bad.append(f'{num}: no card'); continue
|
|
cm = re.search(r'\*\*Status:\*\* (.)', open(card[0]).read())
|
|
if cm and cm.group(1) != mark:
|
|
bad.append(f'{num}: index {mark} != card {cm.group(1)}')
|
|
print(f'wrote {p} ({len(out_rows)} rows); mismatches: {bad if bad else 0}')
|
|
|
|
if __name__ == '__main__':
|
|
main()
|