jung/tools/migrate_manifests.py

145 lines
7.1 KiB
Python

#!/usr/bin/env python3
"""Migrate section MANIFESTs to CANON v1 (AGENTS.md).
Per-item tables `| file | size | source | verify |`, grouped by item number,
source taken from the cards' Downloads tables (single source of truth).
Usage: python3 tools/migrate_manifests.py <section> [--dry-run]"""
import re, sys, os, glob, collections
def card_titles(sec):
t = {}
for p in glob.glob(f'sections/{sec}/[0-9][0-9]-*.md'):
num = os.path.basename(p)[:2]
m = re.match(r'^# (.*)', open(p).read(), re.M)
t[num] = m.group(1).strip() if m else num
return t
def card_sources(sec):
"""file -> source, from the cards' Downloads tables."""
s = {}
for p in glob.glob(f'sections/{sec}/[0-9][0-9]-*.md'):
c = open(p).read()
dm = re.search(r'^## Downloads\n(.*?)(?=^## |\Z)', c, re.M | re.S)
if not dm:
continue
for row in re.finditer(r'^\|\s*(?:EN|RU|\?)\s*\|\s*([0-9]{2}-[a-zA-Z0-9\.\-]+)\s*\|\s*[^|]*\|\s*([^|]+)\|\s*$', dm.group(1), re.M):
s[row.group(1)] = row.group(2).strip()
return s
def human(b):
return f'{b//1048576} MB' if b >= 1048576 else f'{b//1024} KB'
def parse_old(manifest):
"""Return (items, not_dl, cross, en_unavail, notes, misc) — items: {num: [(file,size,src,note)]}"""
c = open(manifest, encoding='utf-8').read()
intro = re.split(r'^## ', c, maxsplit=1, flags=re.M)[0]
intro = intro.split('\n', 1)[1].strip() if '\n' in intro else ''
items = collections.defaultdict(list)
# generic table rows: | file | size | source-or-bytes | rest |
for m in re.finditer(r'^\|\s*([0-9]{2}-[a-zA-Z0-9\-]+(?:\.[a-zA-Z0-9]+)*\.(?:pdf|fb2|epub|doc|docx|chm|djvu|txt|zip|azw3))\s*\|\s*([^|]+)\|\s*([^|]*)\|\s*([^|]*?)\s*\|\s*$', c, re.M):
f, size, col3, note = m.group(1), m.group(2).strip(), m.group(3).strip(), m.group(4).strip()
size = human(int(size.replace(' ', ''))) if re.fullmatch(r'[\d ]+', size) else size
src = col3 if not re.fullmatch(r'[\d/ ]+', col3) else '' # sec04 raw ed/f id
items[f[:2]].append((f, size, src, note))
# sec03 bullet rows
for m in re.finditer(r'^- `([0-9]{2}-[^`]+)` \(([^)]+)\)', c, re.M):
f, size = m.group(1), m.group(2).strip()
if not any(x[0] == f for x in items[f[:2]]):
items[f[:2]].append((f, size, '', ''))
# keep everything from "## Not downloadable" onward as raw blocks by heading
blocks = {}
for m in re.finditer(r'^## (.+?)\n(.*?)(?=^## |\Z)', c, re.M | re.S):
name, body = m.group(1).strip(), m.group(2).strip()
blocks[name] = body
# ### Cross-section sub-block (inside the big Files block) must survive
cross = re.search(r'### Cross-section[^\n]*\n(.*?)(?=^### |^## |\Z)', c, re.M | re.S)
if cross:
blocks['Cross-section (files live in other sections)'] = cross.group(1).strip()
return items, blocks, intro
# one-off source resolutions (2026-09-21): byte-verified against disk sizes / card notes
OVERRIDES = {
'01-fundamentals': {
'01-cw6-psychological-types-en.pdf': ('ia CarlJungCollectedWorks', 5996798, 'Princeton 2014 — 2026-07-17 (MANIFEST-gap, byte-verified)'),
'07-memories-dreams-reflections-en.pdf': ('ia MemoriesDreamsReflectionsCarlJung_201811', 1894281, '2026-07-17 (MANIFEST-gap, byte-verified)'),
'32-cw12-psych-alchemy-en.pdf': ('lg (2026-07-17, id не записан)', 28634679, 'MANIFEST-gap'),
'49-personality-individuation-en.epub': ('lg f/106948495', 759848, '2026-09-20 sweep'),
'02-cw7-two-essays-en-princeton-1966.pdf': ('lg (2026-07-17, id не записан)', None, ''),
'03-cw8-structure-dynamics-en-princeton-1972.pdf': ('lg (2026-07-17, id не записан)', None, ''),
'06-cw18-symbolic-life-en-princeton-1978.pdf': ('lg (2026-07-17, id не записан)', None, ''),
'34-cw10-civilization-en-princeton-1964.pdf': ('lg (2026-07-17, id не записан)', None, ''),
},
}
def main():
sec = sys.argv[1]
dry = '--dry-run' in sys.argv
mp = f'downloads/{sec}/MANIFEST.md'
titles, sources = card_titles(sec), card_sources(sec)
items, blocks, intro = parse_old(mp)
for f, (src_o, size_b, note_o) in OVERRIDES.get(sec, {}).items():
num = f[:2]
if not size_b:
p = os.path.join(f'downloads/{sec}', f)
if os.path.exists(p):
size_b = os.path.getsize(p)
size_s = f'{size_b//1048576} MB' if size_b and size_b >= 1048576 else (f'{size_b//1024} KB' if size_b else '—')
idx = [i for i, x in enumerate(items.get(num, [])) if x[0] == f]
if idx:
items[num][idx[0]] = (f, size_s, src_o, note_o or items[num][idx[0]][3])
else:
items.setdefault(num, []).append((f, size_s, src_o, note_o))
# totals from disk
disk = glob.glob(f'downloads/{sec}/*')
disk = [d for d in disk if not d.endswith(('.md', '.part'))]
nru = sum(1 for d in disk if re.search(r'-ru[.\-]', os.path.basename(d)))
nen = sum(1 for d in disk if re.search(r'-en[.\-]', os.path.basename(d)))
total_mb = sum(os.path.getsize(d) for d in disk) // 1048576
tot = f'{len(disk)} files, {total_mb/1024:.1f} GB (RU {nru} / EN {nen})'
# sec04 per-item headings: keep status tail for the new ### lines
item_tails = {}
for name in list(blocks):
m = re.match(r'^(\d{2}) — (.+?)(?: — (.+))?$', name)
if m:
item_tails[m.group(1)] = m.group(3) or m.group(2)
names = {'01-fundamentals': '01 (Fundamentals)', '02-dreams': '02 (Dreams)',
'03-myths-fairy-tales': '03 (Myths & Fairy Tales)', '04-pictures': '04 (Pictures)'}
out = f'# Downloads — Section {names.get(sec, sec)} — FINAL 2026-09-21\n\n'
out += '## Totals\n\n' + tot + '\n\n'
out += '## Files (per item)\n'
for num in sorted(items):
rows = items[num]
n = len(rows)
tail = f' — {item_tails[num]}' if num in item_tails else ''
out += f'\n### {num} — {titles.get(num, num)} — {n} file{"s" if n>1 else ""}{tail}\n\n'
out += '| file | size | source | verify |\n|---|---|---|---|\n'
for f, size, src, note in rows:
card = sources.get(f, '')
if card == 'MANIFEST':
card = ''
if not src or src == '?' or src == 'MANIFEST':
src = card if card and card not in ('?', 'MANIFEST') else '—'
if note == f:
note = ''
out += f'| {f} | {size} | {src} | {note or "—"} |\n'
if intro:
if 'Notes' in blocks:
blocks['Notes'] = intro + '\n\n' + blocks['Notes']
# trailing blocks: keep verbatim; skip regrouped file tables and per-item sections
conv = '' if 'Notes' in blocks else f'\n## Conventions\n\n{intro}\n'
for name in sorted(blocks):
if re.match(r'^(Files|Totals)', name) or re.match(r'^\d{2} —', name):
continue
body = blocks[name]
out += f'\n## {name}\n\n{body}\n'
if conv:
out += conv
txt = out
if dry:
print(txt)
else:
open(mp, 'w', encoding='utf-8').write(txt)
print(f'wrote {mp} ({len(items)} items, {tot})')
if __name__ == '__main__':
main()