jung/tools/migrate_dossiers.py

91 lines
3.9 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""Normalize author dossiers to CANON v1 (lossless, mechanical).
Usage: python3 tools/migrate_dossiers.py [--dry-run]"""
import re, sys, glob, os
def migrate(p, dry):
c = open(p, encoding='utf-8').read()
o = c
# A1: H1 dates (19xx-19yy) / (19xx) → [..]
c = re.sub(r'^(# .+?) \(((?:19|18)\d{2})(?:[-–](?:19|20)\d{2})?\)\s*$',
r'\1 [\2\3]'.replace('\\3', lambda m: ''), c, count=1, flags=re.M) if False else c
m = re.match(r'^# (.+?) \(((?:19|18)\d{2}(?:[-–](?:19|20)\d{2})?)\)\s*$', c, re.M)
if m:
c = c.replace(f'# {m.group(1)} ({m.group(2)})', f'# {m.group(1)} [{m.group(2)}]', 1)
# open-ended (LIVING authors): (19xx-) → [19xx-] — never close an open range
m = re.match(r'^# (.+?) \(((?:19|18)\d{2})-\)\s*$', c, re.M)
if m:
c = c.replace(f'# {m.group(1)} ({m.group(2)}-)', f'# {m.group(1)} [{m.group(2)}-]', 1)
# A2: homonym heading variants
c = re.sub(r'^## Homonym warning \([0-9-]+\)\s*$', '## Homonym warnings', c, flags=re.M)
c = re.sub(r'^## Homonyms / notes\s*$', '## Homonym warnings', c, flags=re.M)
# A3: ## RU name (RSL query) → bold field
c = re.sub(r'^## RU name \(RSL query\)\s*\n+(.+?)\n',
r'**RU name (canonical):** \1\n', c, flags=re.M)
c = re.sub(r'^## RU name \([^)]*\)\s*\n+(.+?)\n',
r'**RU name (canonical):** \1\n', c, flags=re.M)
# A4: RU editions / RU works sections → Verified RU works
for m in re.finditer(r'^## (RU editions|RU works)[^\n]*\s*$', c, re.M):
pass
def merge_ru(m):
title = m.group(0)
if '## Verified RU works' in c:
return f'### {title[3:] if title.startswith("## ") else title}'
return '## Verified RU works'
c = re.sub(r'^## RU editions[^\n]*$', merge_ru, c, flags=re.M)
c = re.sub(r'^## RU works[^\n]*$', merge_ru, c, flags=re.M)
# A5: Round sections → under Notes (or Notes itself)
for m in re.finditer(r'^## (Round [^\n]+)\s*$', c, re.M):
title = m.group(1)
if '## Notes' in c:
c = c.replace(f'## {title}\n', f'### {title}\n', 1)
else:
c = c.replace(f'## {title}\n', f'## Notes\n\n*{title}*\n', 1)
# A6: drop Dossier status line
c = re.sub(r'\*\*Dossier status:\*\* [^\n]*\n', '', c)
# A7: EN identity → Field / identity
c = re.sub(r'^\*\*EN identity:\*\*', '**Field / identity:**', c, flags=re.M)
c = re.sub(r'^\*\*Identity:\*\*', '**Field / identity:**', c, flags=re.M)
c = re.sub(r'^\*\*Identity \(verified[^)]*\):\*\*', '**Field / identity:**', c, flags=re.M)
# collapse 3+ blank lines
c = re.sub(r'\n{3,}', '\n\n', c)
if c != o and not dry:
open(p, 'w', encoding='utf-8').write(c)
return c != o
def validate(p):
c = open(p, encoding='utf-8').read()
iss = []
if not c.startswith('# '):
iss.append('no H1')
if c.count('## Verified RU works') > 1:
iss.append(f"Verified RU works x{c.count('## Verified RU works')}")
if c.count('## Notes') > 1:
iss.append(f"Notes x{c.count('## Notes')}")
if c.count('## Homonym warnings') > 1:
iss.append('Homonym warnings dup')
if c.count('## List items (ISAP Zurich)') > 1:
iss.append('List items dup')
import re as _re
for t in (r'^## RU name \(RSL query\)', r'^\*\*Dossier status:\*\*', r'^\*\*EN identity:\*\*',
r'^## Round ', r'^## RU editions', r'^## RU works'):
if _re.search(t, c, _re.M):
iss.append(f'leftover: {t}')
return iss
def main():
dry = '--dry-run' in sys.argv
n = 0
for p in sorted(glob.glob('authors/*.md')):
if migrate(p, dry):
n += 1
print(f'changed: {n} / {len(glob.glob("authors/*.md"))}')
bad = 0
for p in sorted(glob.glob('authors/*.md')):
for i in validate(p):
bad += 1
print(f'{p}: {i}')
print(f'validation issues: {bad}')
if __name__ == '__main__':
main()