91 lines
3.9 KiB
Python
91 lines
3.9 KiB
Python
#!/usr/bin/env python3
|
||
"""Normalize author dossiers to CANON v1 (lossless, mechanical).
|
||
Usage: python3 tools/migrate_dossiers.py [--dry-run]"""
|
||
import re, sys, glob, os
|
||
|
||
def migrate(p, dry):
|
||
c = open(p, encoding='utf-8').read()
|
||
o = c
|
||
# A1: H1 dates (19xx-19yy) / (19xx) → [..]
|
||
c = re.sub(r'^(# .+?) \(((?:19|18)\d{2})(?:[-–](?:19|20)\d{2})?\)\s*$',
|
||
r'\1 [\2\3]'.replace('\\3', lambda m: ''), c, count=1, flags=re.M) if False else c
|
||
m = re.match(r'^# (.+?) \(((?:19|18)\d{2}(?:[-–](?:19|20)\d{2})?)\)\s*$', c, re.M)
|
||
if m:
|
||
c = c.replace(f'# {m.group(1)} ({m.group(2)})', f'# {m.group(1)} [{m.group(2)}]', 1)
|
||
# open-ended (LIVING authors): (19xx-) → [19xx-] — never close an open range
|
||
m = re.match(r'^# (.+?) \(((?:19|18)\d{2})-\)\s*$', c, re.M)
|
||
if m:
|
||
c = c.replace(f'# {m.group(1)} ({m.group(2)}-)', f'# {m.group(1)} [{m.group(2)}-]', 1)
|
||
# A2: homonym heading variants
|
||
c = re.sub(r'^## Homonym warning \([0-9-]+\)\s*$', '## Homonym warnings', c, flags=re.M)
|
||
c = re.sub(r'^## Homonyms / notes\s*$', '## Homonym warnings', c, flags=re.M)
|
||
# A3: ## RU name (RSL query) → bold field
|
||
c = re.sub(r'^## RU name \(RSL query\)\s*\n+(.+?)\n',
|
||
r'**RU name (canonical):** \1\n', c, flags=re.M)
|
||
c = re.sub(r'^## RU name \([^)]*\)\s*\n+(.+?)\n',
|
||
r'**RU name (canonical):** \1\n', c, flags=re.M)
|
||
# A4: RU editions / RU works sections → Verified RU works
|
||
for m in re.finditer(r'^## (RU editions|RU works)[^\n]*\s*$', c, re.M):
|
||
pass
|
||
def merge_ru(m):
|
||
title = m.group(0)
|
||
if '## Verified RU works' in c:
|
||
return f'### {title[3:] if title.startswith("## ") else title}'
|
||
return '## Verified RU works'
|
||
c = re.sub(r'^## RU editions[^\n]*$', merge_ru, c, flags=re.M)
|
||
c = re.sub(r'^## RU works[^\n]*$', merge_ru, c, flags=re.M)
|
||
# A5: Round sections → under Notes (or Notes itself)
|
||
for m in re.finditer(r'^## (Round [^\n]+)\s*$', c, re.M):
|
||
title = m.group(1)
|
||
if '## Notes' in c:
|
||
c = c.replace(f'## {title}\n', f'### {title}\n', 1)
|
||
else:
|
||
c = c.replace(f'## {title}\n', f'## Notes\n\n*{title}*\n', 1)
|
||
# A6: drop Dossier status line
|
||
c = re.sub(r'\*\*Dossier status:\*\* [^\n]*\n', '', c)
|
||
# A7: EN identity → Field / identity
|
||
c = re.sub(r'^\*\*EN identity:\*\*', '**Field / identity:**', c, flags=re.M)
|
||
c = re.sub(r'^\*\*Identity:\*\*', '**Field / identity:**', c, flags=re.M)
|
||
c = re.sub(r'^\*\*Identity \(verified[^)]*\):\*\*', '**Field / identity:**', c, flags=re.M)
|
||
# collapse 3+ blank lines
|
||
c = re.sub(r'\n{3,}', '\n\n', c)
|
||
if c != o and not dry:
|
||
open(p, 'w', encoding='utf-8').write(c)
|
||
return c != o
|
||
|
||
def validate(p):
|
||
c = open(p, encoding='utf-8').read()
|
||
iss = []
|
||
if not c.startswith('# '):
|
||
iss.append('no H1')
|
||
if c.count('## Verified RU works') > 1:
|
||
iss.append(f"Verified RU works x{c.count('## Verified RU works')}")
|
||
if c.count('## Notes') > 1:
|
||
iss.append(f"Notes x{c.count('## Notes')}")
|
||
if c.count('## Homonym warnings') > 1:
|
||
iss.append('Homonym warnings dup')
|
||
if c.count('## List items (ISAP Zurich)') > 1:
|
||
iss.append('List items dup')
|
||
import re as _re
|
||
for t in (r'^## RU name \(RSL query\)', r'^\*\*Dossier status:\*\*', r'^\*\*EN identity:\*\*',
|
||
r'^## Round ', r'^## RU editions', r'^## RU works'):
|
||
if _re.search(t, c, _re.M):
|
||
iss.append(f'leftover: {t}')
|
||
return iss
|
||
|
||
def main():
|
||
dry = '--dry-run' in sys.argv
|
||
n = 0
|
||
for p in sorted(glob.glob('authors/*.md')):
|
||
if migrate(p, dry):
|
||
n += 1
|
||
print(f'changed: {n} / {len(glob.glob("authors/*.md"))}')
|
||
bad = 0
|
||
for p in sorted(glob.glob('authors/*.md')):
|
||
for i in validate(p):
|
||
bad += 1
|
||
print(f'{p}: {i}')
|
||
print(f'validation issues: {bad}')
|
||
|
||
if __name__ == '__main__':
|
||
main()
|