#!/usr/bin/env python3 """Migrate section MANIFESTs to CANON v1 (AGENTS.md). Per-item tables `| file | size | source | verify |`, grouped by item number, source taken from the cards' Downloads tables (single source of truth). Usage: python3 tools/migrate_manifests.py
[--dry-run]""" import re, sys, os, glob, collections def card_titles(sec): t = {} for p in glob.glob(f'sections/{sec}/[0-9][0-9]-*.md'): num = os.path.basename(p)[:2] m = re.match(r'^# (.*)', open(p).read(), re.M) t[num] = m.group(1).strip() if m else num return t def card_sources(sec): """file -> source, from the cards' Downloads tables.""" s = {} for p in glob.glob(f'sections/{sec}/[0-9][0-9]-*.md'): c = open(p).read() dm = re.search(r'^## Downloads\n(.*?)(?=^## |\Z)', c, re.M | re.S) if not dm: continue for row in re.finditer(r'^\|\s*(?:EN|RU|\?)\s*\|\s*([0-9]{2}-[a-zA-Z0-9\.\-]+)\s*\|\s*[^|]*\|\s*([^|]+)\|\s*$', dm.group(1), re.M): s[row.group(1)] = row.group(2).strip() return s def human(b): return f'{b//1048576} MB' if b >= 1048576 else f'{b//1024} KB' def parse_old(manifest): """Return (items, not_dl, cross, en_unavail, notes, misc) — items: {num: [(file,size,src,note)]}""" c = open(manifest, encoding='utf-8').read() intro = re.split(r'^## ', c, maxsplit=1, flags=re.M)[0] intro = intro.split('\n', 1)[1].strip() if '\n' in intro else '' items = collections.defaultdict(list) # generic table rows: | file | size | source-or-bytes | rest | for m in re.finditer(r'^\|\s*([0-9]{2}-[a-zA-Z0-9\-]+(?:\.[a-zA-Z0-9]+)*\.(?:pdf|fb2|epub|doc|docx|chm|djvu|txt|zip|azw3))\s*\|\s*([^|]+)\|\s*([^|]*)\|\s*([^|]*?)\s*\|\s*$', c, re.M): f, size, col3, note = m.group(1), m.group(2).strip(), m.group(3).strip(), m.group(4).strip() size = human(int(size.replace(' ', ''))) if re.fullmatch(r'[\d ]+', size) else size src = col3 if not re.fullmatch(r'[\d/ ]+', col3) else '' # sec04 raw ed/f id items[f[:2]].append((f, size, src, note)) # sec03 bullet rows for m in re.finditer(r'^- `([0-9]{2}-[^`]+)` \(([^)]+)\)', c, re.M): f, size = m.group(1), m.group(2).strip() if not any(x[0] == f for x in items[f[:2]]): items[f[:2]].append((f, size, '', '')) # keep everything from "## Not downloadable" onward as raw blocks by heading blocks = {} for m in re.finditer(r'^## (.+?)\n(.*?)(?=^## |\Z)', c, re.M | re.S): name, body = m.group(1).strip(), m.group(2).strip() blocks[name] = body # ### Cross-section sub-block (inside the big Files block) must survive cross = re.search(r'### Cross-section[^\n]*\n(.*?)(?=^### |^## |\Z)', c, re.M | re.S) if cross: blocks['Cross-section (files live in other sections)'] = cross.group(1).strip() return items, blocks, intro # one-off source resolutions (2026-09-21): byte-verified against disk sizes / card notes OVERRIDES = { '01-fundamentals': { '01-cw6-psychological-types-en.pdf': ('ia CarlJungCollectedWorks', 5996798, 'Princeton 2014 — 2026-07-17 (MANIFEST-gap, byte-verified)'), '07-memories-dreams-reflections-en.pdf': ('ia MemoriesDreamsReflectionsCarlJung_201811', 1894281, '2026-07-17 (MANIFEST-gap, byte-verified)'), '32-cw12-psych-alchemy-en.pdf': ('lg (2026-07-17, id не записан)', 28634679, 'MANIFEST-gap'), '49-personality-individuation-en.epub': ('lg f/106948495', 759848, '2026-09-20 sweep'), '02-cw7-two-essays-en-princeton-1966.pdf': ('lg (2026-07-17, id не записан)', None, ''), '03-cw8-structure-dynamics-en-princeton-1972.pdf': ('lg (2026-07-17, id не записан)', None, ''), '06-cw18-symbolic-life-en-princeton-1978.pdf': ('lg (2026-07-17, id не записан)', None, ''), '34-cw10-civilization-en-princeton-1964.pdf': ('lg (2026-07-17, id не записан)', None, ''), }, '04-pictures': { '03-red-book-en-facsimile-norton-2009.pdf': ('lg f/91819237', None, 'last p404 = red epigraph page (book ends) \u2713'), '03-red-book-en-facsimile-2009-b.pdf': ('lg f/91545973', None, '= \u00abLiber Novus\u2026\u00bb Penguin 2009 text ed \u2713'), '03-red-book-en-reader-2009.pdf': ('lg (2026-07-15, id не записан)', None, ''), '03-red-book-ru-2009.pdf': ('lg (2026-07-15, id не записан)', None, ''), '05-abt-territory-symbol-ru-kastalia-2013.pdf': ('lg f/93127234', None, 'scan, trailer intact \u2713'), '28-man-symbols-en-doubleday-1969-a.pdf': ('lg f/91481666', None, 'pdftext: \u201cThe first and only work\u2026\u201d \u2713'), '28-man-symbols-en-doubleday-1969-b.pdf': ('lg f/91577995', None, 'size exact, %PDF \u2713'), '28-man-symbols-en-doubleday-1969-c.pdf': ('lg f/98995698', None, 'size exact \u2713'), '28-man-symbols-en-anchor-1988.pdf': ('lg f/96629120', None, 'size exact \u2713'), '30-kandinsky-spiritual-ru-buksmart-2020-t1.pdf': ('lg f/94051405', None, 'trailer intact \u2713'), '30-kandinsky-spiritual-ru-buksmart-2020-t2.pdf': ('lg f/94051663', None, 'trailer intact \u2713'), }, } def main(): sec = sys.argv[1] dry = '--dry-run' in sys.argv mp = f'downloads/{sec}/MANIFEST.md' titles, sources = card_titles(sec), card_sources(sec) items, blocks, intro = parse_old(mp) for f, (src_o, size_b, note_o) in OVERRIDES.get(sec, {}).items(): num = f[:2] if not size_b: p = os.path.join(f'downloads/{sec}', f) if os.path.exists(p): size_b = os.path.getsize(p) size_s = f'{size_b//1048576} MB' if size_b and size_b >= 1048576 else (f'{size_b//1024} KB' if size_b else '—') idx = [i for i, x in enumerate(items.get(num, [])) if x[0] == f] if idx: items[num][idx[0]] = (f, size_s, src_o, note_o or items[num][idx[0]][3]) else: items.setdefault(num, []).append((f, size_s, src_o, note_o)) # totals from disk disk = glob.glob(f'downloads/{sec}/*') disk = [d for d in disk if not d.endswith(('.md', '.part'))] nru = sum(1 for d in disk if re.search(r'-ru[.\-]', os.path.basename(d))) nen = sum(1 for d in disk if re.search(r'-en[.\-]', os.path.basename(d))) total_mb = sum(os.path.getsize(d) for d in disk) // 1048576 tot = f'{len(disk)} files, {total_mb/1024:.1f} GB (RU {nru} / EN {nen})' # sec04 per-item headings: keep status tail for the new ### lines item_tails = {} for name in list(blocks): m = re.match(r'^(\d{2}) — (.+?)(?: — (.+))?$', name) if m: item_tails[m.group(1)] = m.group(3) or m.group(2) names = {'01-fundamentals': '01 (Fundamentals)', '02-dreams': '02 (Dreams)', '03-myths-fairy-tales': '03 (Myths & Fairy Tales)', '04-pictures': '04 (Pictures)'} out = f'# Downloads — Section {names.get(sec, sec)} — FINAL 2026-09-21\n\n' out += '## Totals\n\n' + tot + '\n\n' out += '## Files (per item)\n' for num in sorted(items): rows = items[num] n = len(rows) tail = f' — {item_tails[num]}' if num in item_tails else '' out += f'\n### {num} — {titles.get(num, num)} — {n} file{"s" if n>1 else ""}{tail}\n\n' out += '| file | size | source | verify |\n|---|---|---|---|\n' for f, size, src, note in rows: card = sources.get(f, '') if card == 'MANIFEST': card = '' if not src or src == '?' or src == 'MANIFEST': src = card if card and card not in ('?', 'MANIFEST') else '—' if note == f: note = '' out += f'| {f} | {size} | {src} | {note or "—"} |\n' if intro: if 'Notes' in blocks: blocks['Notes'] = intro + '\n\n' + blocks['Notes'] # trailing blocks: keep verbatim; skip regrouped file tables and per-item sections conv = '' if 'Notes' in blocks else f'\n## Conventions\n\n{intro}\n' for name in sorted(blocks): if re.match(r'^(Files|Totals)', name) or re.match(r'^\d{2} —', name): continue body = blocks[name] out += f'\n## {name}\n\n{body}\n' if conv: out += conv txt = out if dry: print(txt) else: open(mp, 'w', encoding='utf-8').write(txt) print(f'wrote {mp} ({len(items)} items, {tot})') if __name__ == '__main__': main()