unification: INDEX → CANON v1 (| # | Group | EN title | Author(s) | Status | File |), 219 rows, 0 status mismatches; sec02 legacy triage table dropped (dup of main table + cards)

This commit is contained in:
Dmitry Kokorin 2026-09-21 23:48:53 +03:00
parent 3225039d88
commit 62bb11a99d
5 changed files with 319 additions and 294 deletions

89
tools/migrate_indexes.py Normal file
View file

@ -0,0 +1,89 @@
#!/usr/bin/env python3
"""Migrate section INDEX.md to CANON v1: | # | Group | EN title | Author(s) | Status | File |
Usage: python3 tools/migrate_indexes.py <section> [--dry-run]"""
import re, sys, os, glob
NAMES = {'01-fundamentals': 'Fundamentals', '02-dreams': 'Dreams',
'03-myths-fairy-tales': 'Myths & Fairy Tales', '04-pictures': 'Pictures'}
COLMAP = {'block': 'group', 'subsection': 'group', 'group': 'group',
'en title': 'title', 'title': 'title', 'item': 'title', 'book': 'title',
'author(s)': 'author', 'author(s) ': 'author', 'authors': 'author', 'author': 'author',
'status': 'status', 'file': 'file'}
def main():
sec = sys.argv[1]
dry = '--dry-run' in sys.argv
p = f'sections/{sec}/INDEX.md'
c = open(p).read()
lines = c.split('\n')
# find table
ti = next(i for i, l in enumerate(lines) if l.startswith('| #') or l.startswith('|#'))
header = [x.strip() for x in lines[ti].strip().strip('|').split('|')]
colidx = {}
for i, h in enumerate(header):
key = COLMAP.get(h.lower())
if key:
colidx[key] = i
rows, ti_end = [], ti
for i in range(ti+1, len(lines)):
if not lines[i].startswith('|'):
ti_end = i
break
cells = [x.strip() for x in lines[i].strip().strip('|').split('|')]
if len(cells) >= 2 and not (set(cells[0]) <= set('- ') or cells[0] == '#'):
rows.append(cells)
else:
ti_end = len(lines)
intro = '\n'.join(lines[1:ti]).strip()
tail = '\n'.join(lines[ti_end:]).strip()
out_rows = []
for cells in rows:
num = cells[0]
group = cells[colidx.get('group', 1)] if 'group' in colidx else ''
group = group or '—'
title = cells[colidx.get('title')]
author = cells[colidx.get('author')] if 'author' in colidx else ''
status = cells[colidx.get('status')] if 'status' in colidx else ''
filecol = cells[colidx.get('file')] if 'file' in colidx else ''
m = re.match(r'\[(.+?)\]\((.+?)\)', title)
if m:
title, filecol = m.group(1), m.group(2)
if not filecol:
cand = glob.glob(f'sections/{sec}/{num}-*.md')
filecol = os.path.basename(cand[0]) if cand else ''
elif not filecol.startswith(num + '-'):
cand = glob.glob(f'sections/{sec}/{num}-{filecol}')
filecol = os.path.basename(cand[0]) if cand else f'{num}-{filecol}'
mark = status[:1]
if mark not in '✅🔶❌⬜🔎':
mm = re.search(r'[✅🔶❌⬜🔎]', status)
mark = mm.group(0) if mm else '?'
out_rows.append(f'| {num} | {group} | {title} | {author} | {mark} | {filecol} |')
new = f'# Section {sec[:2]} — {NAMES[sec]}: INDEX\n\n{intro}\n\n'
new += '| # | Group | EN title | Author(s) | Status | File |\n|---|-------|----------|-----------|--------|------|\n'
new += '\n'.join(out_rows) + '\n'
if tail:
# keep trailing sections (## ...) only
tlines = tail.split('\n')
i0 = next((i for i, l in enumerate(tlines) if l.startswith('## ')), None)
if i0 is not None:
new += '\n' + '\n'.join(tlines[i0:]).strip() + '\n'
if dry:
print(new)
else:
open(p, 'w').write(new)
# verify against cards
bad = []
for r in out_rows:
num = r.split('|')[1].strip()
mark = r.split('|')[5].strip()
card = glob.glob(f'sections/{sec}/{num}-*.md')
if not card:
bad.append(f'{num}: no card'); continue
cm = re.search(r'\*\*Status:\*\* (.)', open(card[0]).read())
if cm and cm.group(1) != mark:
bad.append(f'{num}: index {mark} != card {cm.group(1)}')
print(f'wrote {p} ({len(out_rows)} rows); mismatches: {bad if bad else 0}')
if __name__ == '__main__':
main()