#!/usr/bin/env python3 """Migrate book cards to CANON v3 (see AGENTS.md). Lossless where possible. Usage: python3 tools/migrate_cards_v3.py --section 01-fundamentals [--dry-run NN ...]""" import re, sys, os, glob, argparse BOILER_LIBGEN = re.compile(r'^[-*]?\s*\((none|not searched|not searched yet|not checked|not checked yet|trade-only|empty|TBC|нет|—)\)?\s*$', re.I) SEC_DATES = { # section completion / verification dates (for gate lines) '01-fundamentals': '2026-07-17', '02-dreams': '2026-07-18', '03-myths-fairy-tales': '2026-09-20', '04-pictures': '2026-07-15', } SEC_NAMES = { '01-fundamentals': '01 Fundamentals', '02-dreams': '02 Dreams', '03-myths-fairy-tales': '03 Myths & Fairy Tales', '04-pictures': '04 Pictures', } def parse_manifest_files(manifest_path): """Return {item_num: [(file, size_str, source), ...]} for a section MANIFEST.""" items = {} if not os.path.exists(manifest_path): return items c = open(manifest_path, encoding='utf-8').read() fname_re = r'([0-9]{2}-[a-zA-Z0-9\-]+(?:\.[a-zA-Z0-9]+)*\.(?:pdf|fb2|epub|doc|docx|chm|djvu|txt|zip))' # sec01/02: table rows | file | size | source | note | for m in re.finditer(r'^\|\s*' + fname_re + r'\s*\|\s*([^|]+)\|\s*([^|]+)\|', c, re.M): f, size, src = m.group(1), m.group(2).strip(), m.group(3).strip() if re.search(r'-(ru|en)[.\-]', f): items.setdefault(f[:2], []).append((f, size, src)) # sec03: per-item subsections with bullets for m in re.finditer(r'^### (\d{2}) —.*?\n(.*?)(?=\n### |\n## |\Z)', c, re.M | re.S): num, body = m.group(1), m.group(2) for b in re.finditer(r'^- `([^`]+)` \(([^)]+)\)', body, re.M): fname, size = b.group(1), b.group(2) items.setdefault(fname[:2], []).append((fname, size, 'MANIFEST')) # sec04: per-item tables | file | bytes | ed/f | verify | for m in re.finditer(r'^## (\d{2}) —.*?\n(.*?)(?=\n## |\Z)', c, re.M | re.S): num, body = m.group(1), m.group(2) for row in re.finditer(r'^\|\s*' + fname_re + r'\s*\|\s*(\d[\d ]*?)\s*\|\s*([^|]+)\|', body, re.M): fname, sz, ids = row.group(1), f"{int(row.group(2).replace(' ', ''))//1048576} MB", row.group(3).strip() fids = re.search(r'/(\d+)$', ids) src = f'lg f/{fids.group(1)}' if fids else ids items.setdefault(fname[:2], []).append((fname, sz, src)) # dedupe by filename; prefer human-readable source (lg f/N, flib b/N, ia, MANIFEST) good = re.compile(r'^(lg f|flib b|ia |MANIFEST|CarlJung)') for num, lst in items.items(): seen = {} for f, s, r in lst: if f not in seen or (good.match(r) and not good.match(seen[f][2])): seen[f] = (f, s, r) items[num] = list(seen.values()) return items def h1_title(c): m = re.match(r'^# (.*)$', c, re.M) return m.group(1).strip() if m else '' def is_dash(v): if v is None: return True v = v.strip() return v in ('—', '-', '— (Phase 0: to resolve)', '') or v.startswith('— (Phase 0') def human_size(s): s = s.strip() m = re.match(r'^([\d\s]+)$', s) if m: b = int(m.group(1).replace(' ', '')) return f'{b//1048576} MB' if b >= 1048576 else f'{b//1024} KB' return s def migrate_card(path, files_by_item, sec, dry=False): c = open(path, encoding='utf-8').read() num = os.path.basename(path)[:2] # split header (before first ##) and sections m = re.search(r'\n## ', c) header, rest = c[:m.start()], c[m.start()+1:] def fld(name): mm = re.search(rf'\*\*{re.escape(name)}:\*\* (.*)', header) return mm.group(1).strip() if mm else None title = h1_title(c) author = fld('Author(s)') or '—' shelf = fld('Shelf mark') section = fld('Section') status = fld('Status') or '⬜' pub_en = fld('Publisher (EN)') isbn_en = fld('EN ISBN') rli = fld('Reading list item(s)') notes_extra = [] if rli: notes_extra.append(f'- Reading list item(s): {rli}') # --- EN editions table en_block = '' if not is_dash(pub_en) or (isbn_en and not is_dash(isbn_en)): pub = pub_en or '—' year = '—' mm = re.match(r'^(.*),\s*(\d{4}[a-z]?)\s*(\(.*\))?$', pub) if mm: pub, year = mm.group(1).strip(), mm.group(2) isbn = '—' if is_dash(isbn_en) else isbn_en en_block = ( '\n## Editions — EN\n\n' '| Title / edition | Publisher | Year | Pages | ISBN | Notes |\n' '|-----------------|-----------|------|-------|------|-------|\n' f'| {title} | {pub} | {year} | — | {isbn} | |\n') # --- gather sections secs = {} for sm in re.finditer(r'^## (.+?)\n(.*?)(?=^## |\Z)', rest, re.M | re.S): secs[sm.group(1).strip()] = sm.group(2).strip('\n') ru_body = secs.get('Russian editions') libgen_body = secs.get('Libgen') rsl_body = secs.get('RSL records') notes_body = secs.get('Notes', '') # --- Editions — RU ru_block = '' if ru_body is not None: tb = re.search(r'\| RU title \|.*?(?=\n\n|\Z)', ru_body, re.S) if tb: tbl = tb.group(0).rstrip() data_rows = [l for l in tbl.split('\n') if l.startswith('|') and 'RU title' not in l and not re.match(r'^\|[-\s|]+\|$', l) and '(not searched' not in l and not re.match(r'^\|\s*\(?\s*(none|n/a|—|-)\s*\)?\s*\|', l)] if data_rows: ru_block = '\n## Editions — RU\n\n' + tbl + '\n' else: vdate = re.search(r'verified (\d{4}-\d{2}-\d{2})', status) vdate = vdate.group(1) if vdate else SEC_DATES.get(sec, '') ru_block = f'\n## Editions — RU\n\n— нет (verified {vdate}, gate: MISSING.md §{num})\n' else: vdate = re.search(r'verified (\d{4}-\d{2}-\d{2})', status) vdate = vdate.group(1) if vdate else SEC_DATES.get(sec, '') ru_block = f'\n## Editions — RU\n\n— нет (verified {vdate}, gate: MISSING.md §{num})\n' # --- Downloads dl_block = '' files = files_by_item.get(num, []) if files: rows = [] for fname, size, src in files: lang = 'RU' if re.search(r'-ru[.\-]', fname) else ('EN' if re.search(r'-en[.\-]', fname) else '?') rows.append(f'| {lang} | {fname} | {human_size(size)} | {src} |') dl_block = ('## Downloads\n\n| lang | file | size | source |\n' '|------|------|------|--------|\n' + '\n'.join(rows) + '\n') # libgen body residue → Notes (ids not covered by the table, or no table at all) if libgen_body: src_ids = set(re.findall(r'(?:flib b|lg f)/\d+', ' '.join(r for _, _, r in files))) keep = [] for l in libgen_body.split('\n'): s = l.strip() if not s or BOILER_LIBGEN.match(s): continue if re.match(r'^(?:- )?(RU|EN) (file|files):', s): ids = set(re.findall(r'(?:flib b|lg f)/\d+', s)) if not ids or ids <= src_ids: continue # no info beyond the table keep.append(l) if keep: notes_extra.append('- (was Libgen): ' + ' '.join(k.lstrip('- ').strip() for k in keep)) # --- Catalog records cat_block = '' if rsl_body is not None: body = rsl_body.strip() inner = re.sub(r'^```json\n?|```$', '', body).strip() if not inner or inner == '// not searched yet' or inner == '// not searched': sweep = f'data/sweeps/{sec}/{num}-{os.path.basename(path)[:-3]}.md' if os.path.exists(sweep): inner = f'РГБ: (свип: {sweep})' else: vdate = re.search(r'verified (\d{4}-\d{2}-\d{2})', status) inner = f'no records (verified {vdate.group(1) if vdate else SEC_DATES.get(sec, "")})' else: inner = re.sub(r'^//\s*', 'РГБ: ', inner, count=1) if not inner.startswith(('РГБ', 'no records')): inner = 'РГБ: ' + inner cat_block = f'## Catalog records\n\n```\n{inner}\n```\n' # --- Notes (merge extra sections) extra_secs = [] for name in list(secs): if name in ('Russian editions', 'Libgen', 'RSL records', 'Notes'): continue extra_secs.append(f'{name}: {secs[name].strip()}') notes_lines = [l for l in notes_body.split('\n') if l.strip() != '- (empty)' and not re.match(r'^-?.*not searched yet\.?\s*$', l.strip())] notes = '\n'.join(notes_lines + notes_extra + [f'- (was {e})' for e in extra_secs]).strip() # --- assemble out = f'# {title}\n\n' out += f'**Author(s):** {author}\n' out += f'**Shelf mark:** {shelf if shelf is not None else "—"}\n' out += f'**Section:** {section if section is not None else SEC_NAMES.get(sec, sec)}\n' out += f'**Status:** {status}\n' out += en_block out += ru_block if dl_block: out += '\n' + dl_block if cat_block: out += '\n' + cat_block if notes: out += f'\n## Notes\n\n{notes}\n' if dry: print('=' * 70) print(f'DRY-RUN {path}') print(out) return open(path, 'w', encoding='utf-8').write(out) def main(): ap = argparse.ArgumentParser() ap.add_argument('--section', required=True) ap.add_argument('--dry-run', nargs='*', default=None) a = ap.parse_args() secdir = f'sections/{a.section}' man = f'downloads/{a.section}/MANIFEST.md' files = parse_manifest_files(man) cards = sorted(glob.glob(f'{secdir}/[0-9][0-9]-*.md')) for p in cards: if a.dry_run is not None and os.path.basename(p)[:2] not in a.dry_run: continue migrate_card(p, files, a.section, dry=(a.dry_run is not None)) if not a.dry_run: print(f'migrated {len(cards)} cards in {secdir}') if __name__ == '__main__': main()