158 lines
8.3 KiB
Python
158 lines
8.3 KiB
Python
#!/usr/bin/env python3
|
|
"""Migrate section MANIFESTs to CANON v1 (AGENTS.md).
|
|
Per-item tables `| file | size | source | verify |`, grouped by item number,
|
|
source taken from the cards' Downloads tables (single source of truth).
|
|
Usage: python3 tools/migrate_manifests.py <section> [--dry-run]"""
|
|
import re, sys, os, glob, collections
|
|
|
|
def card_titles(sec):
|
|
t = {}
|
|
for p in glob.glob(f'sections/{sec}/[0-9][0-9]-*.md'):
|
|
num = os.path.basename(p)[:2]
|
|
m = re.match(r'^# (.*)', open(p).read(), re.M)
|
|
t[num] = m.group(1).strip() if m else num
|
|
return t
|
|
|
|
def card_sources(sec):
|
|
"""file -> source, from the cards' Downloads tables."""
|
|
s = {}
|
|
for p in glob.glob(f'sections/{sec}/[0-9][0-9]-*.md'):
|
|
c = open(p).read()
|
|
dm = re.search(r'^## Downloads\n(.*?)(?=^## |\Z)', c, re.M | re.S)
|
|
if not dm:
|
|
continue
|
|
for row in re.finditer(r'^\|\s*(?:EN|RU|\?)\s*\|\s*([0-9]{2}-[a-zA-Z0-9\.\-]+)\s*\|\s*[^|]*\|\s*([^|]+)\|\s*$', dm.group(1), re.M):
|
|
s[row.group(1)] = row.group(2).strip()
|
|
return s
|
|
|
|
def human(b):
|
|
return f'{b//1048576} MB' if b >= 1048576 else f'{b//1024} KB'
|
|
|
|
def parse_old(manifest):
|
|
"""Return (items, not_dl, cross, en_unavail, notes, misc) — items: {num: [(file,size,src,note)]}"""
|
|
c = open(manifest, encoding='utf-8').read()
|
|
intro = re.split(r'^## ', c, maxsplit=1, flags=re.M)[0]
|
|
intro = intro.split('\n', 1)[1].strip() if '\n' in intro else ''
|
|
items = collections.defaultdict(list)
|
|
# generic table rows: | file | size | source-or-bytes | rest |
|
|
for m in re.finditer(r'^\|\s*([0-9]{2}-[a-zA-Z0-9\-]+(?:\.[a-zA-Z0-9]+)*\.(?:pdf|fb2|epub|doc|docx|chm|djvu|txt|zip|azw3))\s*\|\s*([^|]+)\|\s*([^|]*)\|\s*([^|]*?)\s*\|\s*$', c, re.M):
|
|
f, size, col3, note = m.group(1), m.group(2).strip(), m.group(3).strip(), m.group(4).strip()
|
|
size = human(int(size.replace(' ', ''))) if re.fullmatch(r'[\d ]+', size) else size
|
|
src = col3 if not re.fullmatch(r'[\d/ ]+', col3) else '' # sec04 raw ed/f id
|
|
items[f[:2]].append((f, size, src, note))
|
|
# sec03 bullet rows
|
|
for m in re.finditer(r'^- `([0-9]{2}-[^`]+)` \(([^)]+)\)', c, re.M):
|
|
f, size = m.group(1), m.group(2).strip()
|
|
if not any(x[0] == f for x in items[f[:2]]):
|
|
items[f[:2]].append((f, size, '', ''))
|
|
# keep everything from "## Not downloadable" onward as raw blocks by heading
|
|
blocks = {}
|
|
for m in re.finditer(r'^## (.+?)\n(.*?)(?=^## |\Z)', c, re.M | re.S):
|
|
name, body = m.group(1).strip(), m.group(2).strip()
|
|
blocks[name] = body
|
|
# ### Cross-section sub-block (inside the big Files block) must survive
|
|
cross = re.search(r'### Cross-section[^\n]*\n(.*?)(?=^### |^## |\Z)', c, re.M | re.S)
|
|
if cross:
|
|
blocks['Cross-section (files live in other sections)'] = cross.group(1).strip()
|
|
return items, blocks, intro
|
|
|
|
# one-off source resolutions (2026-09-21): byte-verified against disk sizes / card notes
|
|
OVERRIDES = {
|
|
'01-fundamentals': {
|
|
'01-cw6-psychological-types-en.pdf': ('ia CarlJungCollectedWorks', 5996798, 'Princeton 2014 — 2026-07-17 (MANIFEST-gap, byte-verified)'),
|
|
'07-memories-dreams-reflections-en.pdf': ('ia MemoriesDreamsReflectionsCarlJung_201811', 1894281, '2026-07-17 (MANIFEST-gap, byte-verified)'),
|
|
'32-cw12-psych-alchemy-en.pdf': ('lg (2026-07-17, id не записан)', 28634679, 'MANIFEST-gap'),
|
|
'49-personality-individuation-en.epub': ('lg f/106948495', 759848, '2026-09-20 sweep'),
|
|
'02-cw7-two-essays-en-princeton-1966.pdf': ('lg (2026-07-17, id не записан)', None, ''),
|
|
'03-cw8-structure-dynamics-en-princeton-1972.pdf': ('lg (2026-07-17, id не записан)', None, ''),
|
|
'06-cw18-symbolic-life-en-princeton-1978.pdf': ('lg (2026-07-17, id не записан)', None, ''),
|
|
'34-cw10-civilization-en-princeton-1964.pdf': ('lg (2026-07-17, id не записан)', None, ''),
|
|
},
|
|
'04-pictures': {
|
|
'03-red-book-en-facsimile-norton-2009.pdf': ('lg f/91819237', None, 'last p404 = red epigraph page (book ends) \u2713'),
|
|
'03-red-book-en-facsimile-2009-b.pdf': ('lg f/91545973', None, '= \u00abLiber Novus\u2026\u00bb Penguin 2009 text ed \u2713'),
|
|
'03-red-book-en-reader-2009.pdf': ('lg (2026-07-15, id не записан)', None, ''),
|
|
'03-red-book-ru-2009.pdf': ('lg (2026-07-15, id не записан)', None, ''),
|
|
'05-abt-territory-symbol-ru-kastalia-2013.pdf': ('lg f/93127234', None, 'scan, trailer intact \u2713'),
|
|
'28-man-symbols-en-doubleday-1969-a.pdf': ('lg f/91481666', None, 'pdftext: \u201cThe first and only work\u2026\u201d \u2713'),
|
|
'28-man-symbols-en-doubleday-1969-b.pdf': ('lg f/91577995', None, 'size exact, %PDF \u2713'),
|
|
'28-man-symbols-en-doubleday-1969-c.pdf': ('lg f/98995698', None, 'size exact \u2713'),
|
|
'28-man-symbols-en-anchor-1988.pdf': ('lg f/96629120', None, 'size exact \u2713'),
|
|
'30-kandinsky-spiritual-ru-buksmart-2020-t1.pdf': ('lg f/94051405', None, 'trailer intact \u2713'),
|
|
'30-kandinsky-spiritual-ru-buksmart-2020-t2.pdf': ('lg f/94051663', None, 'trailer intact \u2713'),
|
|
},
|
|
}
|
|
|
|
def main():
|
|
sec = sys.argv[1]
|
|
dry = '--dry-run' in sys.argv
|
|
mp = f'downloads/{sec}/MANIFEST.md'
|
|
titles, sources = card_titles(sec), card_sources(sec)
|
|
items, blocks, intro = parse_old(mp)
|
|
for f, (src_o, size_b, note_o) in OVERRIDES.get(sec, {}).items():
|
|
num = f[:2]
|
|
if not size_b:
|
|
p = os.path.join(f'downloads/{sec}', f)
|
|
if os.path.exists(p):
|
|
size_b = os.path.getsize(p)
|
|
size_s = f'{size_b//1048576} MB' if size_b and size_b >= 1048576 else (f'{size_b//1024} KB' if size_b else '—')
|
|
idx = [i for i, x in enumerate(items.get(num, [])) if x[0] == f]
|
|
if idx:
|
|
items[num][idx[0]] = (f, size_s, src_o, note_o or items[num][idx[0]][3])
|
|
else:
|
|
items.setdefault(num, []).append((f, size_s, src_o, note_o))
|
|
# totals from disk
|
|
disk = glob.glob(f'downloads/{sec}/*')
|
|
disk = [d for d in disk if not d.endswith(('.md', '.part'))]
|
|
nru = sum(1 for d in disk if re.search(r'-ru[.\-]', os.path.basename(d)))
|
|
nen = sum(1 for d in disk if re.search(r'-en[.\-]', os.path.basename(d)))
|
|
total_mb = sum(os.path.getsize(d) for d in disk) // 1048576
|
|
tot = f'{len(disk)} files, {total_mb/1024:.1f} GB (RU {nru} / EN {nen})'
|
|
|
|
# sec04 per-item headings: keep status tail for the new ### lines
|
|
item_tails = {}
|
|
for name in list(blocks):
|
|
m = re.match(r'^(\d{2}) — (.+?)(?: — (.+))?$', name)
|
|
if m:
|
|
item_tails[m.group(1)] = m.group(3) or m.group(2)
|
|
names = {'01-fundamentals': '01 (Fundamentals)', '02-dreams': '02 (Dreams)',
|
|
'03-myths-fairy-tales': '03 (Myths & Fairy Tales)', '04-pictures': '04 (Pictures)'}
|
|
out = f'# Downloads — Section {names.get(sec, sec)} — FINAL 2026-09-21\n\n'
|
|
out += '## Totals\n\n' + tot + '\n\n'
|
|
out += '## Files (per item)\n'
|
|
for num in sorted(items):
|
|
rows = items[num]
|
|
n = len(rows)
|
|
tail = f' — {item_tails[num]}' if num in item_tails else ''
|
|
out += f'\n### {num} — {titles.get(num, num)} — {n} file{"s" if n>1 else ""}{tail}\n\n'
|
|
out += '| file | size | source | verify |\n|---|---|---|---|\n'
|
|
for f, size, src, note in rows:
|
|
card = sources.get(f, '')
|
|
if card == 'MANIFEST':
|
|
card = ''
|
|
if not src or src == '?' or src == 'MANIFEST':
|
|
src = card if card and card not in ('?', 'MANIFEST') else '—'
|
|
if note == f:
|
|
note = ''
|
|
out += f'| {f} | {size} | {src} | {note or "—"} |\n'
|
|
if intro:
|
|
if 'Notes' in blocks:
|
|
blocks['Notes'] = intro + '\n\n' + blocks['Notes']
|
|
# trailing blocks: keep verbatim; skip regrouped file tables and per-item sections
|
|
conv = '' if 'Notes' in blocks else f'\n## Conventions\n\n{intro}\n'
|
|
for name in sorted(blocks):
|
|
if re.match(r'^(Files|Totals)', name) or re.match(r'^\d{2} —', name):
|
|
continue
|
|
body = blocks[name]
|
|
out += f'\n## {name}\n\n{body}\n'
|
|
if conv:
|
|
out += conv
|
|
txt = out
|
|
if dry:
|
|
print(txt)
|
|
else:
|
|
open(mp, 'w', encoding='utf-8').write(txt)
|
|
print(f'wrote {mp} ({len(items)} items, {tot})')
|
|
|
|
if __name__ == '__main__':
|
|
main()
|