#!/usr/bin/env python3 """Normalize author dossiers to CANON v1 (lossless, mechanical). Usage: python3 tools/migrate_dossiers.py [--dry-run]""" import re, sys, glob, os def migrate(p, dry): c = open(p, encoding='utf-8').read() o = c # A1: H1 dates (19xx-19yy) / (19xx) → [..] c = re.sub(r'^(# .+?) \(((?:19|18)\d{2})(?:[-–](?:19|20)\d{2})?\)\s*$', r'\1 [\2\3]'.replace('\\3', lambda m: ''), c, count=1, flags=re.M) if False else c m = re.match(r'^# (.+?) \(((?:19|18)\d{2}(?:[-–](?:19|20)\d{2})?)\)\s*$', c, re.M) if m: c = c.replace(f'# {m.group(1)} ({m.group(2)})', f'# {m.group(1)} [{m.group(2)}]', 1) # open-ended (LIVING authors): (19xx-) → [19xx-] — never close an open range m = re.match(r'^# (.+?) \(((?:19|18)\d{2})-\)\s*$', c, re.M) if m: c = c.replace(f'# {m.group(1)} ({m.group(2)}-)', f'# {m.group(1)} [{m.group(2)}-]', 1) # A2: homonym heading variants c = re.sub(r'^## Homonym warning \([0-9-]+\)\s*$', '## Homonym warnings', c, flags=re.M) c = re.sub(r'^## Homonyms / notes\s*$', '## Homonym warnings', c, flags=re.M) # A3: ## RU name (RSL query) → bold field c = re.sub(r'^## RU name \(RSL query\)\s*\n+(.+?)\n', r'**RU name (canonical):** \1\n', c, flags=re.M) c = re.sub(r'^## RU name \([^)]*\)\s*\n+(.+?)\n', r'**RU name (canonical):** \1\n', c, flags=re.M) # A4: RU editions / RU works sections → Verified RU works for m in re.finditer(r'^## (RU editions|RU works)[^\n]*\s*$', c, re.M): pass def merge_ru(m): title = m.group(0) if '## Verified RU works' in c: return f'### {title[3:] if title.startswith("## ") else title}' return '## Verified RU works' c = re.sub(r'^## RU editions[^\n]*$', merge_ru, c, flags=re.M) c = re.sub(r'^## RU works[^\n]*$', merge_ru, c, flags=re.M) # A5: Round sections → under Notes (or Notes itself) for m in re.finditer(r'^## (Round [^\n]+)\s*$', c, re.M): title = m.group(1) if '## Notes' in c: c = c.replace(f'## {title}\n', f'### {title}\n', 1) else: c = c.replace(f'## {title}\n', f'## Notes\n\n*{title}*\n', 1) # A6: drop Dossier status line c = re.sub(r'\*\*Dossier status:\*\* [^\n]*\n', '', c) # A7: EN identity → Field / identity c = re.sub(r'^\*\*EN identity:\*\*', '**Field / identity:**', c, flags=re.M) c = re.sub(r'^\*\*Identity:\*\*', '**Field / identity:**', c, flags=re.M) c = re.sub(r'^\*\*Identity \(verified[^)]*\):\*\*', '**Field / identity:**', c, flags=re.M) # collapse 3+ blank lines c = re.sub(r'\n{3,}', '\n\n', c) if c != o and not dry: open(p, 'w', encoding='utf-8').write(c) return c != o def validate(p): c = open(p, encoding='utf-8').read() iss = [] if not c.startswith('# '): iss.append('no H1') if c.count('## Verified RU works') > 1: iss.append(f"Verified RU works x{c.count('## Verified RU works')}") if c.count('## Notes') > 1: iss.append(f"Notes x{c.count('## Notes')}") if c.count('## Homonym warnings') > 1: iss.append('Homonym warnings dup') if c.count('## List items (ISAP Zurich)') > 1: iss.append('List items dup') import re as _re for t in (r'^## RU name \(RSL query\)', r'^\*\*Dossier status:\*\*', r'^\*\*EN identity:\*\*', r'^## Round ', r'^## RU editions', r'^## RU works'): if _re.search(t, c, _re.M): iss.append(f'leftover: {t}') return iss def main(): dry = '--dry-run' in sys.argv n = 0 for p in sorted(glob.glob('authors/*.md')): if migrate(p, dry): n += 1 print(f'changed: {n} / {len(glob.glob("authors/*.md"))}') bad = 0 for p in sorted(glob.glob('authors/*.md')): for i in validate(p): bad += 1 print(f'{p}: {i}') print(f'validation issues: {bad}') if __name__ == '__main__': main()