jung/tools/dedup_files.py
Dmitry Kokorin 7d3794560e cross-section dedup system (user tasks 2+3, 2026-09-25):
- tools/xref.py: master book list for ALL 12 sections (RAW 07-12 scraped: sections/07/RAW.md
  + data/isap-raw/), cross-ref matcher (CW gate + distinctive tokens + years + Cyrillic
  transliteration), stale MANIFEST 'Not downloadable' checker; data/MASTER-LIST.md generated
  (17 book-dupes in 01-06 found, e.g. sec01#51=sec06#14 Solomon)
- file-level dedup: 17 md5-identical files removed from later sections (kept in first),
  MANIFESTs/cards -> cross-ref rows, totals updated (sec01 102 / 02 37 / 03 125 / 04 38 /
  05 38 / 06 37), dedup rule in AGENTS.md + MANIFEST Notes
- sec05 MANIFEST phantom line '03 Aion' removed (Aion not in sec05 list; files in sec01)
- sec03 #44: 1968 Bodley file was #41's book (content 'English Fairy Tales') — removed,
  #44 keeps ABC-CLIO 2002
2026-09-25 23:25:05 +03:00

118 lines
6.4 KiB
Python

#!/usr/bin/env python3
"""dedup_files.py — one-time cross-section file dedup (2026-09-25).
Rule (user, 2026-09-25): a book already downloaded in an earlier section is NOT
re-downloaded in a later one — the later section's MANIFEST/card gets a
cross-ref line "secNN: <file>" instead.
Removes 17 duplicated files (identical md5), rewrites the affected
MANIFEST rows + totals, card Downloads tables, adds dated notes.
Run: python3 tools/dedup_files.py (dry-run prints; add --apply to execute)
"""
import os, re, sys, subprocess
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
DL = os.path.join(BASE, 'downloads')
APPLY = '--apply' in sys.argv
# (section_dir, removed_file) -> (canonical_section_dir, kept_file)
DEDUP = {
('02-dreams', '10-cw18-symbolic-life-en-1953.pdf'): ('01-fundamentals', '06-cw18-symbolic-life-en-princeton-1978.pdf'),
('02-dreams', '03-cw8-structure-dynamics-en.pdf'): ('01-fundamentals', '03-cw8-structure-dynamics-en-princeton-1972.pdf'),
('02-dreams', '02-cw7-two-essays-en.pdf'): ('01-fundamentals', '02-cw7-two-essays-en-princeton-1966.pdf'),
('02-dreams', '01-cw5-symbols-en.pdf'): ('01-fundamentals', '33-cw5-symbols-transformation-en-princeton-1954.pdf'),
('02-dreams', '35-ego-archetype-en-1974.pdf'): ('01-fundamentals', '10-ego-archetype-en-shambhala-1992.pdf'),
('04-pictures', '01-cw9-archetypes-en-routledge-1981.pdf'): ('02-dreams', '04-cw9-1-archetypes-en-1981.pdf'),
('05-ethnology', '12-eliade-myth-and-reality-en-1963.pdf'): ('03-myths-fairy-tales', '24-myth-and-reality-en-1963.pdf'),
('05-ethnology', '06-campbell-masks-of-god-primitive-mythology-en-2018.epub'): ('03-myths-fairy-tales', '48-masks-god-v1-primitive-en.epub'),
('05-ethnology', '06-campbell-masks-of-god-ru-2019-t1-c1.pdf'): ('03-myths-fairy-tales', '48-maski-boga-ru-t1c1.pdf'),
('05-ethnology', '06-campbell-masks-of-god-ru-2019-t1-c2.pdf'): ('03-myths-fairy-tales', '48-maski-boga-ru-t1c2.pdf'),
('05-ethnology', '06-campbell-masks-of-god-ru-2021-t2-c2.pdf'): ('03-myths-fairy-tales', '48-maski-boga-ru-t2c2.pdf'),
('05-ethnology', '06-campbell-masks-of-god-ru-t2-c1.pdf'): ('03-myths-fairy-tales', '48-maski-boga-ru-t2c1.pdf'),
('05-ethnology', '13-eliade-rites-and-symbols-of-initiation-en-1958.pdf'): ('03-myths-fairy-tales', '26-rites-symbols-en-1958.pdf'),
('05-ethnology', '02-memories-dreams-reflections-ru.fb2'): ('01-fundamentals', '07-memories-dreams-reflections-ru.b.fb2'),
('05-ethnology', '01-cw10-the-archaic-man-ru-2023.fb2'): ('01-fundamentals', '34-cw10-civilization-ru-2023.b.fb2'),
('06-religion', '14-solomon-self-transformation-en-2007.pdf'): ('01-fundamentals', '51-self-transformation-en-karnac-2007.pdf'),
# mislabel: identical to #41's file (content = "English Fairy Tales", the #41 book)
('03-myths-fairy-tales', '44-more-english-fairy-tales-en-bodley-1968.pdf'): ('03-myths-fairy-tales', '41-jacobs-english-ft-en-1968.pdf'),
}
NOTE = 'dedup 2026-09-25: файл дублирует {c} (md5-идентичен) — файл удалён, cross-ref'
def main():
by_sec = {}
for (sec, f), (csec, cf) in DEDUP.items():
by_sec.setdefault(sec, []).append((f, csec, cf))
for sec, items in sorted(by_sec.items()):
# 1) delete files
for f, csec, cf in items:
p = os.path.join(DL, sec, f)
if not os.path.exists(p):
print(f" MISSING: {p}")
continue
if APPLY:
os.remove(p)
subprocess.run(['git', 'add', '-f', '--', p], cwd=BASE, capture_output=True)
print(f" rm {sec}/{f} (kept: {csec}/{cf})")
# 2) MANIFEST: replace rows
mf = os.path.join(DL, sec, 'MANIFEST.md')
txt = open(mf, encoding='utf8').read()
for f, csec, cf in items:
cnum = csec[:2]
new_tab = f"| — (cross-ref sec{cnum}: {cf}) | — | sec{cnum} file | {NOTE.format(c=cnum)} |"
new_lang = f"| — (cross-ref sec{cnum}: {cf}) | — | sec{cnum} file | {NOTE.format(c=cnum)} |"
row = re.search(r'^\| ' + re.escape(f) + r' \|.*$', txt, re.M)
if row:
txt = txt.replace(row.group(0), new_tab)
continue
row = re.search(r'^\| (EN|RU|DE) \| ' + re.escape(f) + r' \|.*$', txt, re.M)
if row:
txt = txt.replace(row.group(0), f"| {row.group(1)} | {new_lang[2:]}")
continue
row = re.search(r'^- `' + re.escape(f) + r'`[^(]*\(.*?\) —.*$', txt, re.M)
if row:
txt = txt.replace(row.group(0), f"- ~~{f}~~ — {NOTE.format(c=cnum)}: `{cf}` (sec{cnum})")
continue
print(f" !! MANIFEST row not found: {f}")
# totals line: subtract nothing precisely; append note instead
m = re.search(r'^## Totals\n\n(.+)$', txt, re.M)
if m:
txt = txt.replace(m.group(1), m.group(1) + f' — {len(items)} файла удалено как дубли (dedup 2026-09-25)')
if APPLY:
open(mf, 'w', encoding='utf8').write(txt)
print(f" MANIFEST {sec}: {len(items)} строк -> cross-ref")
# 3) cards: Downloads tables
for f, csec, cf in items:
num = f[:2]
card = None
for p in os.listdir(os.path.join(BASE, 'sections', sec)):
if p.startswith(num + '-') and p.endswith('.md'):
card = os.path.join(BASE, 'sections', sec, p)
if not card:
print(f" !! card not found for {num} in {sec}")
continue
ctxt = open(card, encoding='utf8').read()
row = re.search(r'^\| (EN|RU|DE) \| ' + re.escape(f) + r' \|.*$', ctxt, re.M)
if not row:
# maybe listed in Notes instead
print(f" card {card.split('/')[-1]}: Downloads row not found (check manually)")
continue
lang = row.group(1)
cnum = csec[:2]
new = (f"| {lang} | — (cross-ref sec{cnum}: {cf}) | — | sec{cnum} file "
f"(dedup 2026-09-25: дубликат md5-идентичен) |")
ctxt = ctxt.replace(row.group(0), new)
if APPLY:
open(card, 'w', encoding='utf8').write(ctxt)
print(f" card {card.split('/')[-1]}: {f} -> cross-ref sec{cnum}")
if APPLY:
print("\nAPPLIED")
else:
print("\nDRY RUN — re-run with --apply")
if __name__ == '__main__':
main()