unification: dossiers → CANON v1 (104/142 changed: H1 [dates] incl. open-ended for living, Homonym warnings unified, RU-name sections → canonical field, Round logs → Notes, Dossier status dropped)

This commit is contained in:
Dmitry Kokorin 2026-09-21 23:59:49 +03:00
parent 680bf45ee1
commit 31cb0dea70
105 changed files with 159 additions and 192 deletions

91
tools/migrate_dossiers.py Normal file
View file

@ -0,0 +1,91 @@
#!/usr/bin/env python3
"""Normalize author dossiers to CANON v1 (lossless, mechanical).
Usage: python3 tools/migrate_dossiers.py [--dry-run]"""
import re, sys, glob, os
def migrate(p, dry):
c = open(p, encoding='utf-8').read()
o = c
# A1: H1 dates (19xx-19yy) / (19xx) → [..]
c = re.sub(r'^(# .+?) \(((?:19|18)\d{2})(?:[-–](?:19|20)\d{2})?\)\s*$',
r'\1 [\2\3]'.replace('\\3', lambda m: ''), c, count=1, flags=re.M) if False else c
m = re.match(r'^# (.+?) \(((?:19|18)\d{2}(?:[-–](?:19|20)\d{2})?)\)\s*$', c, re.M)
if m:
c = c.replace(f'# {m.group(1)} ({m.group(2)})', f'# {m.group(1)} [{m.group(2)}]', 1)
# open-ended (LIVING authors): (19xx-) → [19xx-] — never close an open range
m = re.match(r'^# (.+?) \(((?:19|18)\d{2})-\)\s*$', c, re.M)
if m:
c = c.replace(f'# {m.group(1)} ({m.group(2)}-)', f'# {m.group(1)} [{m.group(2)}-]', 1)
# A2: homonym heading variants
c = re.sub(r'^## Homonym warning \([0-9-]+\)\s*$', '## Homonym warnings', c, flags=re.M)
c = re.sub(r'^## Homonyms / notes\s*$', '## Homonym warnings', c, flags=re.M)
# A3: ## RU name (RSL query) → bold field
c = re.sub(r'^## RU name \(RSL query\)\s*\n+(.+?)\n',
r'**RU name (canonical):** \1\n', c, flags=re.M)
c = re.sub(r'^## RU name \([^)]*\)\s*\n+(.+?)\n',
r'**RU name (canonical):** \1\n', c, flags=re.M)
# A4: RU editions / RU works sections → Verified RU works
for m in re.finditer(r'^## (RU editions|RU works)[^\n]*\s*$', c, re.M):
pass
def merge_ru(m):
title = m.group(0)
if '## Verified RU works' in c:
return f'### {title[3:] if title.startswith("## ") else title}'
return '## Verified RU works'
c = re.sub(r'^## RU editions[^\n]*$', merge_ru, c, flags=re.M)
c = re.sub(r'^## RU works[^\n]*$', merge_ru, c, flags=re.M)
# A5: Round sections → under Notes (or Notes itself)
for m in re.finditer(r'^## (Round [^\n]+)\s*$', c, re.M):
title = m.group(1)
if '## Notes' in c:
c = c.replace(f'## {title}\n', f'### {title}\n', 1)
else:
c = c.replace(f'## {title}\n', f'## Notes\n\n*{title}*\n', 1)
# A6: drop Dossier status line
c = re.sub(r'\*\*Dossier status:\*\* [^\n]*\n', '', c)
# A7: EN identity → Field / identity
c = re.sub(r'^\*\*EN identity:\*\*', '**Field / identity:**', c, flags=re.M)
c = re.sub(r'^\*\*Identity:\*\*', '**Field / identity:**', c, flags=re.M)
c = re.sub(r'^\*\*Identity \(verified[^)]*\):\*\*', '**Field / identity:**', c, flags=re.M)
# collapse 3+ blank lines
c = re.sub(r'\n{3,}', '\n\n', c)
if c != o and not dry:
open(p, 'w', encoding='utf-8').write(c)
return c != o
def validate(p):
c = open(p, encoding='utf-8').read()
iss = []
if not c.startswith('# '):
iss.append('no H1')
if c.count('## Verified RU works') > 1:
iss.append(f"Verified RU works x{c.count('## Verified RU works')}")
if c.count('## Notes') > 1:
iss.append(f"Notes x{c.count('## Notes')}")
if c.count('## Homonym warnings') > 1:
iss.append('Homonym warnings dup')
if c.count('## List items (ISAP Zurich)') > 1:
iss.append('List items dup')
import re as _re
for t in (r'^## RU name \(RSL query\)', r'^\*\*Dossier status:\*\*', r'^\*\*EN identity:\*\*',
r'^## Round ', r'^## RU editions', r'^## RU works'):
if _re.search(t, c, _re.M):
iss.append(f'leftover: {t}')
return iss
def main():
dry = '--dry-run' in sys.argv
n = 0
for p in sorted(glob.glob('authors/*.md')):
if migrate(p, dry):
n += 1
print(f'changed: {n} / {len(glob.glob("authors/*.md"))}')
bad = 0
for p in sorted(glob.glob('authors/*.md')):
for i in validate(p):
bad += 1
print(f'{p}: {i}')
print(f'validation issues: {bad}')
if __name__ == '__main__':
main()