#!/usr/bin/env python3 """dedup_files.py — one-time cross-section file dedup (2026-09-25). Rule (user, 2026-09-25): a book already downloaded in an earlier section is NOT re-downloaded in a later one — the later section's MANIFEST/card gets a cross-ref line "secNN: " instead. Removes 17 duplicated files (identical md5), rewrites the affected MANIFEST rows + totals, card Downloads tables, adds dated notes. Run: python3 tools/dedup_files.py (dry-run prints; add --apply to execute) """ import os, re, sys, subprocess BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) DL = os.path.join(BASE, 'downloads') APPLY = '--apply' in sys.argv # (section_dir, removed_file) -> (canonical_section_dir, kept_file) DEDUP = { ('02-dreams', '10-cw18-symbolic-life-en-1953.pdf'): ('01-fundamentals', '06-cw18-symbolic-life-en-princeton-1978.pdf'), ('02-dreams', '03-cw8-structure-dynamics-en.pdf'): ('01-fundamentals', '03-cw8-structure-dynamics-en-princeton-1972.pdf'), ('02-dreams', '02-cw7-two-essays-en.pdf'): ('01-fundamentals', '02-cw7-two-essays-en-princeton-1966.pdf'), ('02-dreams', '01-cw5-symbols-en.pdf'): ('01-fundamentals', '33-cw5-symbols-transformation-en-princeton-1954.pdf'), ('02-dreams', '35-ego-archetype-en-1974.pdf'): ('01-fundamentals', '10-ego-archetype-en-shambhala-1992.pdf'), ('04-pictures', '01-cw9-archetypes-en-routledge-1981.pdf'): ('02-dreams', '04-cw9-1-archetypes-en-1981.pdf'), ('05-ethnology', '12-eliade-myth-and-reality-en-1963.pdf'): ('03-myths-fairy-tales', '24-myth-and-reality-en-1963.pdf'), ('05-ethnology', '06-campbell-masks-of-god-primitive-mythology-en-2018.epub'): ('03-myths-fairy-tales', '48-masks-god-v1-primitive-en.epub'), ('05-ethnology', '06-campbell-masks-of-god-ru-2019-t1-c1.pdf'): ('03-myths-fairy-tales', '48-maski-boga-ru-t1c1.pdf'), ('05-ethnology', '06-campbell-masks-of-god-ru-2019-t1-c2.pdf'): ('03-myths-fairy-tales', '48-maski-boga-ru-t1c2.pdf'), ('05-ethnology', '06-campbell-masks-of-god-ru-2021-t2-c2.pdf'): ('03-myths-fairy-tales', '48-maski-boga-ru-t2c2.pdf'), ('05-ethnology', '06-campbell-masks-of-god-ru-t2-c1.pdf'): ('03-myths-fairy-tales', '48-maski-boga-ru-t2c1.pdf'), ('05-ethnology', '13-eliade-rites-and-symbols-of-initiation-en-1958.pdf'): ('03-myths-fairy-tales', '26-rites-symbols-en-1958.pdf'), ('05-ethnology', '02-memories-dreams-reflections-ru.fb2'): ('01-fundamentals', '07-memories-dreams-reflections-ru.b.fb2'), ('05-ethnology', '01-cw10-the-archaic-man-ru-2023.fb2'): ('01-fundamentals', '34-cw10-civilization-ru-2023.b.fb2'), ('06-religion', '14-solomon-self-transformation-en-2007.pdf'): ('01-fundamentals', '51-self-transformation-en-karnac-2007.pdf'), # mislabel: identical to #41's file (content = "English Fairy Tales", the #41 book) ('03-myths-fairy-tales', '44-more-english-fairy-tales-en-bodley-1968.pdf'): ('03-myths-fairy-tales', '41-jacobs-english-ft-en-1968.pdf'), } NOTE = 'dedup 2026-09-25: файл дублирует {c} (md5-идентичен) — файл удалён, cross-ref' def main(): by_sec = {} for (sec, f), (csec, cf) in DEDUP.items(): by_sec.setdefault(sec, []).append((f, csec, cf)) for sec, items in sorted(by_sec.items()): # 1) delete files for f, csec, cf in items: p = os.path.join(DL, sec, f) if not os.path.exists(p): print(f" MISSING: {p}") continue if APPLY: os.remove(p) subprocess.run(['git', 'add', '-f', '--', p], cwd=BASE, capture_output=True) print(f" rm {sec}/{f} (kept: {csec}/{cf})") # 2) MANIFEST: replace rows mf = os.path.join(DL, sec, 'MANIFEST.md') txt = open(mf, encoding='utf8').read() for f, csec, cf in items: cnum = csec[:2] new_tab = f"| — (cross-ref sec{cnum}: {cf}) | — | sec{cnum} file | {NOTE.format(c=cnum)} |" new_lang = f"| — (cross-ref sec{cnum}: {cf}) | — | sec{cnum} file | {NOTE.format(c=cnum)} |" row = re.search(r'^\| ' + re.escape(f) + r' \|.*$', txt, re.M) if row: txt = txt.replace(row.group(0), new_tab) continue row = re.search(r'^\| (EN|RU|DE) \| ' + re.escape(f) + r' \|.*$', txt, re.M) if row: txt = txt.replace(row.group(0), f"| {row.group(1)} | {new_lang[2:]}") continue row = re.search(r'^- `' + re.escape(f) + r'`[^(]*\(.*?\) —.*$', txt, re.M) if row: txt = txt.replace(row.group(0), f"- ~~{f}~~ — {NOTE.format(c=cnum)}: `{cf}` (sec{cnum})") continue print(f" !! MANIFEST row not found: {f}") # totals line: subtract nothing precisely; append note instead m = re.search(r'^## Totals\n\n(.+)$', txt, re.M) if m: txt = txt.replace(m.group(1), m.group(1) + f' — {len(items)} файла удалено как дубли (dedup 2026-09-25)') if APPLY: open(mf, 'w', encoding='utf8').write(txt) print(f" MANIFEST {sec}: {len(items)} строк -> cross-ref") # 3) cards: Downloads tables for f, csec, cf in items: num = f[:2] card = None for p in os.listdir(os.path.join(BASE, 'sections', sec)): if p.startswith(num + '-') and p.endswith('.md'): card = os.path.join(BASE, 'sections', sec, p) if not card: print(f" !! card not found for {num} in {sec}") continue ctxt = open(card, encoding='utf8').read() row = re.search(r'^\| (EN|RU|DE) \| ' + re.escape(f) + r' \|.*$', ctxt, re.M) if not row: # maybe listed in Notes instead print(f" card {card.split('/')[-1]}: Downloads row not found (check manually)") continue lang = row.group(1) cnum = csec[:2] new = (f"| {lang} | — (cross-ref sec{cnum}: {cf}) | — | sec{cnum} file " f"(dedup 2026-09-25: дубликат md5-идентичен) |") ctxt = ctxt.replace(row.group(0), new) if APPLY: open(card, 'w', encoding='utf8').write(ctxt) print(f" card {card.split('/')[-1]}: {f} -> cross-ref sec{cnum}") if APPLY: print("\nAPPLIED") else: print("\nDRY RUN — re-run with --apply") if __name__ == '__main__': main()