- tools/xref.py: master book list for ALL 12 sections (RAW 07-12 scraped: sections/07/RAW.md + data/isap-raw/), cross-ref matcher (CW gate + distinctive tokens + years + Cyrillic transliteration), stale MANIFEST 'Not downloadable' checker; data/MASTER-LIST.md generated (17 book-dupes in 01-06 found, e.g. sec01#51=sec06#14 Solomon) - file-level dedup: 17 md5-identical files removed from later sections (kept in first), MANIFESTs/cards -> cross-ref rows, totals updated (sec01 102 / 02 37 / 03 125 / 04 38 / 05 38 / 06 37), dedup rule in AGENTS.md + MANIFEST Notes - sec05 MANIFEST phantom line '03 Aion' removed (Aion not in sec05 list; files in sec01) - sec03 #44: 1968 Bodley file was #41's book (content 'English Fairy Tales') — removed, #44 keeps ABC-CLIO 2002
118 lines
6.4 KiB
Python
118 lines
6.4 KiB
Python
#!/usr/bin/env python3
|
|
"""dedup_files.py — one-time cross-section file dedup (2026-09-25).
|
|
|
|
Rule (user, 2026-09-25): a book already downloaded in an earlier section is NOT
|
|
re-downloaded in a later one — the later section's MANIFEST/card gets a
|
|
cross-ref line "secNN: <file>" instead.
|
|
|
|
Removes 17 duplicated files (identical md5), rewrites the affected
|
|
MANIFEST rows + totals, card Downloads tables, adds dated notes.
|
|
Run: python3 tools/dedup_files.py (dry-run prints; add --apply to execute)
|
|
"""
|
|
import os, re, sys, subprocess
|
|
|
|
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
DL = os.path.join(BASE, 'downloads')
|
|
APPLY = '--apply' in sys.argv
|
|
|
|
# (section_dir, removed_file) -> (canonical_section_dir, kept_file)
|
|
DEDUP = {
|
|
('02-dreams', '10-cw18-symbolic-life-en-1953.pdf'): ('01-fundamentals', '06-cw18-symbolic-life-en-princeton-1978.pdf'),
|
|
('02-dreams', '03-cw8-structure-dynamics-en.pdf'): ('01-fundamentals', '03-cw8-structure-dynamics-en-princeton-1972.pdf'),
|
|
('02-dreams', '02-cw7-two-essays-en.pdf'): ('01-fundamentals', '02-cw7-two-essays-en-princeton-1966.pdf'),
|
|
('02-dreams', '01-cw5-symbols-en.pdf'): ('01-fundamentals', '33-cw5-symbols-transformation-en-princeton-1954.pdf'),
|
|
('02-dreams', '35-ego-archetype-en-1974.pdf'): ('01-fundamentals', '10-ego-archetype-en-shambhala-1992.pdf'),
|
|
('04-pictures', '01-cw9-archetypes-en-routledge-1981.pdf'): ('02-dreams', '04-cw9-1-archetypes-en-1981.pdf'),
|
|
('05-ethnology', '12-eliade-myth-and-reality-en-1963.pdf'): ('03-myths-fairy-tales', '24-myth-and-reality-en-1963.pdf'),
|
|
('05-ethnology', '06-campbell-masks-of-god-primitive-mythology-en-2018.epub'): ('03-myths-fairy-tales', '48-masks-god-v1-primitive-en.epub'),
|
|
('05-ethnology', '06-campbell-masks-of-god-ru-2019-t1-c1.pdf'): ('03-myths-fairy-tales', '48-maski-boga-ru-t1c1.pdf'),
|
|
('05-ethnology', '06-campbell-masks-of-god-ru-2019-t1-c2.pdf'): ('03-myths-fairy-tales', '48-maski-boga-ru-t1c2.pdf'),
|
|
('05-ethnology', '06-campbell-masks-of-god-ru-2021-t2-c2.pdf'): ('03-myths-fairy-tales', '48-maski-boga-ru-t2c2.pdf'),
|
|
('05-ethnology', '06-campbell-masks-of-god-ru-t2-c1.pdf'): ('03-myths-fairy-tales', '48-maski-boga-ru-t2c1.pdf'),
|
|
('05-ethnology', '13-eliade-rites-and-symbols-of-initiation-en-1958.pdf'): ('03-myths-fairy-tales', '26-rites-symbols-en-1958.pdf'),
|
|
('05-ethnology', '02-memories-dreams-reflections-ru.fb2'): ('01-fundamentals', '07-memories-dreams-reflections-ru.b.fb2'),
|
|
('05-ethnology', '01-cw10-the-archaic-man-ru-2023.fb2'): ('01-fundamentals', '34-cw10-civilization-ru-2023.b.fb2'),
|
|
('06-religion', '14-solomon-self-transformation-en-2007.pdf'): ('01-fundamentals', '51-self-transformation-en-karnac-2007.pdf'),
|
|
# mislabel: identical to #41's file (content = "English Fairy Tales", the #41 book)
|
|
('03-myths-fairy-tales', '44-more-english-fairy-tales-en-bodley-1968.pdf'): ('03-myths-fairy-tales', '41-jacobs-english-ft-en-1968.pdf'),
|
|
}
|
|
|
|
NOTE = 'dedup 2026-09-25: файл дублирует {c} (md5-идентичен) — файл удалён, cross-ref'
|
|
|
|
def main():
|
|
by_sec = {}
|
|
for (sec, f), (csec, cf) in DEDUP.items():
|
|
by_sec.setdefault(sec, []).append((f, csec, cf))
|
|
|
|
for sec, items in sorted(by_sec.items()):
|
|
# 1) delete files
|
|
for f, csec, cf in items:
|
|
p = os.path.join(DL, sec, f)
|
|
if not os.path.exists(p):
|
|
print(f" MISSING: {p}")
|
|
continue
|
|
if APPLY:
|
|
os.remove(p)
|
|
subprocess.run(['git', 'add', '-f', '--', p], cwd=BASE, capture_output=True)
|
|
print(f" rm {sec}/{f} (kept: {csec}/{cf})")
|
|
|
|
# 2) MANIFEST: replace rows
|
|
mf = os.path.join(DL, sec, 'MANIFEST.md')
|
|
txt = open(mf, encoding='utf8').read()
|
|
for f, csec, cf in items:
|
|
cnum = csec[:2]
|
|
new_tab = f"| — (cross-ref sec{cnum}: {cf}) | — | sec{cnum} file | {NOTE.format(c=cnum)} |"
|
|
new_lang = f"| — (cross-ref sec{cnum}: {cf}) | — | sec{cnum} file | {NOTE.format(c=cnum)} |"
|
|
row = re.search(r'^\| ' + re.escape(f) + r' \|.*$', txt, re.M)
|
|
if row:
|
|
txt = txt.replace(row.group(0), new_tab)
|
|
continue
|
|
row = re.search(r'^\| (EN|RU|DE) \| ' + re.escape(f) + r' \|.*$', txt, re.M)
|
|
if row:
|
|
txt = txt.replace(row.group(0), f"| {row.group(1)} | {new_lang[2:]}")
|
|
continue
|
|
row = re.search(r'^- `' + re.escape(f) + r'`[^(]*\(.*?\) —.*$', txt, re.M)
|
|
if row:
|
|
txt = txt.replace(row.group(0), f"- ~~{f}~~ — {NOTE.format(c=cnum)}: `{cf}` (sec{cnum})")
|
|
continue
|
|
print(f" !! MANIFEST row not found: {f}")
|
|
# totals line: subtract nothing precisely; append note instead
|
|
m = re.search(r'^## Totals\n\n(.+)$', txt, re.M)
|
|
if m:
|
|
txt = txt.replace(m.group(1), m.group(1) + f' — {len(items)} файла удалено как дубли (dedup 2026-09-25)')
|
|
if APPLY:
|
|
open(mf, 'w', encoding='utf8').write(txt)
|
|
print(f" MANIFEST {sec}: {len(items)} строк -> cross-ref")
|
|
|
|
# 3) cards: Downloads tables
|
|
for f, csec, cf in items:
|
|
num = f[:2]
|
|
card = None
|
|
for p in os.listdir(os.path.join(BASE, 'sections', sec)):
|
|
if p.startswith(num + '-') and p.endswith('.md'):
|
|
card = os.path.join(BASE, 'sections', sec, p)
|
|
if not card:
|
|
print(f" !! card not found for {num} in {sec}")
|
|
continue
|
|
ctxt = open(card, encoding='utf8').read()
|
|
row = re.search(r'^\| (EN|RU|DE) \| ' + re.escape(f) + r' \|.*$', ctxt, re.M)
|
|
if not row:
|
|
# maybe listed in Notes instead
|
|
print(f" card {card.split('/')[-1]}: Downloads row not found (check manually)")
|
|
continue
|
|
lang = row.group(1)
|
|
cnum = csec[:2]
|
|
new = (f"| {lang} | — (cross-ref sec{cnum}: {cf}) | — | sec{cnum} file "
|
|
f"(dedup 2026-09-25: дубликат md5-идентичен) |")
|
|
ctxt = ctxt.replace(row.group(0), new)
|
|
if APPLY:
|
|
open(card, 'w', encoding='utf8').write(ctxt)
|
|
print(f" card {card.split('/')[-1]}: {f} -> cross-ref sec{cnum}")
|
|
|
|
if APPLY:
|
|
print("\nAPPLIED")
|
|
else:
|
|
print("\nDRY RUN — re-run with --apply")
|
|
|
|
if __name__ == '__main__':
|
|
main()
|