cross-section dedup system (user tasks 2+3, 2026-09-25):
- tools/xref.py: master book list for ALL 12 sections (RAW 07-12 scraped: sections/07/RAW.md + data/isap-raw/), cross-ref matcher (CW gate + distinctive tokens + years + Cyrillic transliteration), stale MANIFEST 'Not downloadable' checker; data/MASTER-LIST.md generated (17 book-dupes in 01-06 found, e.g. sec01#51=sec06#14 Solomon) - file-level dedup: 17 md5-identical files removed from later sections (kept in first), MANIFESTs/cards -> cross-ref rows, totals updated (sec01 102 / 02 37 / 03 125 / 04 38 / 05 38 / 06 37), dedup rule in AGENTS.md + MANIFEST Notes - sec05 MANIFEST phantom line '03 Aion' removed (Aion not in sec05 list; files in sec01) - sec03 #44: 1968 Bodley file was #41's book (content 'English Fairy Tales') — removed, #44 keeps ABC-CLIO 2002
This commit is contained in:
parent
4274488410
commit
7d3794560e
46 changed files with 1831 additions and 21889 deletions
118
tools/dedup_files.py
Normal file
118
tools/dedup_files.py
Normal file
|
|
@ -0,0 +1,118 @@
|
|||
#!/usr/bin/env python3
|
||||
"""dedup_files.py — one-time cross-section file dedup (2026-09-25).
|
||||
|
||||
Rule (user, 2026-09-25): a book already downloaded in an earlier section is NOT
|
||||
re-downloaded in a later one — the later section's MANIFEST/card gets a
|
||||
cross-ref line "secNN: <file>" instead.
|
||||
|
||||
Removes 17 duplicated files (identical md5), rewrites the affected
|
||||
MANIFEST rows + totals, card Downloads tables, adds dated notes.
|
||||
Run: python3 tools/dedup_files.py (dry-run prints; add --apply to execute)
|
||||
"""
|
||||
import os, re, sys, subprocess
|
||||
|
||||
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
DL = os.path.join(BASE, 'downloads')
|
||||
APPLY = '--apply' in sys.argv
|
||||
|
||||
# (section_dir, removed_file) -> (canonical_section_dir, kept_file)
|
||||
DEDUP = {
|
||||
('02-dreams', '10-cw18-symbolic-life-en-1953.pdf'): ('01-fundamentals', '06-cw18-symbolic-life-en-princeton-1978.pdf'),
|
||||
('02-dreams', '03-cw8-structure-dynamics-en.pdf'): ('01-fundamentals', '03-cw8-structure-dynamics-en-princeton-1972.pdf'),
|
||||
('02-dreams', '02-cw7-two-essays-en.pdf'): ('01-fundamentals', '02-cw7-two-essays-en-princeton-1966.pdf'),
|
||||
('02-dreams', '01-cw5-symbols-en.pdf'): ('01-fundamentals', '33-cw5-symbols-transformation-en-princeton-1954.pdf'),
|
||||
('02-dreams', '35-ego-archetype-en-1974.pdf'): ('01-fundamentals', '10-ego-archetype-en-shambhala-1992.pdf'),
|
||||
('04-pictures', '01-cw9-archetypes-en-routledge-1981.pdf'): ('02-dreams', '04-cw9-1-archetypes-en-1981.pdf'),
|
||||
('05-ethnology', '12-eliade-myth-and-reality-en-1963.pdf'): ('03-myths-fairy-tales', '24-myth-and-reality-en-1963.pdf'),
|
||||
('05-ethnology', '06-campbell-masks-of-god-primitive-mythology-en-2018.epub'): ('03-myths-fairy-tales', '48-masks-god-v1-primitive-en.epub'),
|
||||
('05-ethnology', '06-campbell-masks-of-god-ru-2019-t1-c1.pdf'): ('03-myths-fairy-tales', '48-maski-boga-ru-t1c1.pdf'),
|
||||
('05-ethnology', '06-campbell-masks-of-god-ru-2019-t1-c2.pdf'): ('03-myths-fairy-tales', '48-maski-boga-ru-t1c2.pdf'),
|
||||
('05-ethnology', '06-campbell-masks-of-god-ru-2021-t2-c2.pdf'): ('03-myths-fairy-tales', '48-maski-boga-ru-t2c2.pdf'),
|
||||
('05-ethnology', '06-campbell-masks-of-god-ru-t2-c1.pdf'): ('03-myths-fairy-tales', '48-maski-boga-ru-t2c1.pdf'),
|
||||
('05-ethnology', '13-eliade-rites-and-symbols-of-initiation-en-1958.pdf'): ('03-myths-fairy-tales', '26-rites-symbols-en-1958.pdf'),
|
||||
('05-ethnology', '02-memories-dreams-reflections-ru.fb2'): ('01-fundamentals', '07-memories-dreams-reflections-ru.b.fb2'),
|
||||
('05-ethnology', '01-cw10-the-archaic-man-ru-2023.fb2'): ('01-fundamentals', '34-cw10-civilization-ru-2023.b.fb2'),
|
||||
('06-religion', '14-solomon-self-transformation-en-2007.pdf'): ('01-fundamentals', '51-self-transformation-en-karnac-2007.pdf'),
|
||||
# mislabel: identical to #41's file (content = "English Fairy Tales", the #41 book)
|
||||
('03-myths-fairy-tales', '44-more-english-fairy-tales-en-bodley-1968.pdf'): ('03-myths-fairy-tales', '41-jacobs-english-ft-en-1968.pdf'),
|
||||
}
|
||||
|
||||
NOTE = 'dedup 2026-09-25: файл дублирует {c} (md5-идентичен) — файл удалён, cross-ref'
|
||||
|
||||
def main():
|
||||
by_sec = {}
|
||||
for (sec, f), (csec, cf) in DEDUP.items():
|
||||
by_sec.setdefault(sec, []).append((f, csec, cf))
|
||||
|
||||
for sec, items in sorted(by_sec.items()):
|
||||
# 1) delete files
|
||||
for f, csec, cf in items:
|
||||
p = os.path.join(DL, sec, f)
|
||||
if not os.path.exists(p):
|
||||
print(f" MISSING: {p}")
|
||||
continue
|
||||
if APPLY:
|
||||
os.remove(p)
|
||||
subprocess.run(['git', 'add', '-f', '--', p], cwd=BASE, capture_output=True)
|
||||
print(f" rm {sec}/{f} (kept: {csec}/{cf})")
|
||||
|
||||
# 2) MANIFEST: replace rows
|
||||
mf = os.path.join(DL, sec, 'MANIFEST.md')
|
||||
txt = open(mf, encoding='utf8').read()
|
||||
for f, csec, cf in items:
|
||||
cnum = csec[:2]
|
||||
new_tab = f"| — (cross-ref sec{cnum}: {cf}) | — | sec{cnum} file | {NOTE.format(c=cnum)} |"
|
||||
new_lang = f"| — (cross-ref sec{cnum}: {cf}) | — | sec{cnum} file | {NOTE.format(c=cnum)} |"
|
||||
row = re.search(r'^\| ' + re.escape(f) + r' \|.*$', txt, re.M)
|
||||
if row:
|
||||
txt = txt.replace(row.group(0), new_tab)
|
||||
continue
|
||||
row = re.search(r'^\| (EN|RU|DE) \| ' + re.escape(f) + r' \|.*$', txt, re.M)
|
||||
if row:
|
||||
txt = txt.replace(row.group(0), f"| {row.group(1)} | {new_lang[2:]}")
|
||||
continue
|
||||
row = re.search(r'^- `' + re.escape(f) + r'`[^(]*\(.*?\) —.*$', txt, re.M)
|
||||
if row:
|
||||
txt = txt.replace(row.group(0), f"- ~~{f}~~ — {NOTE.format(c=cnum)}: `{cf}` (sec{cnum})")
|
||||
continue
|
||||
print(f" !! MANIFEST row not found: {f}")
|
||||
# totals line: subtract nothing precisely; append note instead
|
||||
m = re.search(r'^## Totals\n\n(.+)$', txt, re.M)
|
||||
if m:
|
||||
txt = txt.replace(m.group(1), m.group(1) + f' — {len(items)} файла удалено как дубли (dedup 2026-09-25)')
|
||||
if APPLY:
|
||||
open(mf, 'w', encoding='utf8').write(txt)
|
||||
print(f" MANIFEST {sec}: {len(items)} строк -> cross-ref")
|
||||
|
||||
# 3) cards: Downloads tables
|
||||
for f, csec, cf in items:
|
||||
num = f[:2]
|
||||
card = None
|
||||
for p in os.listdir(os.path.join(BASE, 'sections', sec)):
|
||||
if p.startswith(num + '-') and p.endswith('.md'):
|
||||
card = os.path.join(BASE, 'sections', sec, p)
|
||||
if not card:
|
||||
print(f" !! card not found for {num} in {sec}")
|
||||
continue
|
||||
ctxt = open(card, encoding='utf8').read()
|
||||
row = re.search(r'^\| (EN|RU|DE) \| ' + re.escape(f) + r' \|.*$', ctxt, re.M)
|
||||
if not row:
|
||||
# maybe listed in Notes instead
|
||||
print(f" card {card.split('/')[-1]}: Downloads row not found (check manually)")
|
||||
continue
|
||||
lang = row.group(1)
|
||||
cnum = csec[:2]
|
||||
new = (f"| {lang} | — (cross-ref sec{cnum}: {cf}) | — | sec{cnum} file "
|
||||
f"(dedup 2026-09-25: дубликат md5-идентичен) |")
|
||||
ctxt = ctxt.replace(row.group(0), new)
|
||||
if APPLY:
|
||||
open(card, 'w', encoding='utf8').write(ctxt)
|
||||
print(f" card {card.split('/')[-1]}: {f} -> cross-ref sec{cnum}")
|
||||
|
||||
if APPLY:
|
||||
print("\nAPPLIED")
|
||||
else:
|
||||
print("\nDRY RUN — re-run with --apply")
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
422
tools/xref.py
Normal file
422
tools/xref.py
Normal file
|
|
@ -0,0 +1,422 @@
|
|||
#!/usr/bin/env python3
|
||||
"""xref.py — master book list + cross-section dedup + manifest freshness check.
|
||||
|
||||
Builds the single source of truth for "which book belongs to which section"
|
||||
across ALL 12 ISAP reading-list sections, so every book is searched exactly once:
|
||||
|
||||
1. Sections 01-06: canonical entries come from the CARDS (sections/0N/NN-*.md).
|
||||
2. Sections 07-12: entries parsed from RAW lists (sections/0N/RAW.md or
|
||||
data/isap-raw/NN-*.txt).
|
||||
3. Every 07-12 entry is matched to an existing 01-06 card (or to an earlier
|
||||
07-12 entry) -> "canonical" pointer; unmatched entries = NEW (to search).
|
||||
4. Every "Not downloadable" line in downloads/0N/MANIFEST.md is checked: if the
|
||||
same book's FILES live in another section's download dir -> STALE line
|
||||
(should be a cross-ref), otherwise KEEP.
|
||||
|
||||
Outputs:
|
||||
data/MASTER-LIST.md — the master list (all sections, canonical pointers)
|
||||
console — stale manifest line report (actionable)
|
||||
|
||||
Run: python3 tools/xref.py (report only)
|
||||
python3 tools/xref.py --write (rewrite data/MASTER-LIST.md)
|
||||
"""
|
||||
import re, os, sys, glob, unicodedata, datetime, itertools
|
||||
|
||||
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
STOP = {'the','a','an','of','in','and','to','on','for','from','by','with','its','his','her',
|
||||
'cw','jce','xje','pa','je','pf','xw','vol','pp','part','chap','chapter','esp','en','ru',
|
||||
'st','ltd','int','world','studies'}
|
||||
|
||||
def norm(s):
|
||||
s = unicodedata.normalize('NFKD', s or '')
|
||||
s = s.lower()
|
||||
s = re.sub(r'(\d)-(\d)', r'\1 \2', s) # 1936-1940 -> "1936 1940"
|
||||
s = re.sub(r'[^a-z0-9 ]', ' ', s)
|
||||
return set(w for w in s.split() if len(w) > 3 and w not in STOP)
|
||||
|
||||
def overlap(a, b):
|
||||
if not a or not b:
|
||||
return 0.0, 0
|
||||
inter = a & b
|
||||
return len(inter) / min(len(a), len(b)), len(inter)
|
||||
|
||||
# ---------------- RAW parsing (07-12) ----------------
|
||||
CW_LINE = re.compile(r'^\s{0,8}CW\s*(\d+/\w{1,2}|\d+|-?S\d?)\s+(\S.*)$')
|
||||
# author-start: one or more "Surname, I." groups, then >=2 spaces, then title
|
||||
AUTHOR_LINE = re.compile(
|
||||
r'^(\s{0,8})'
|
||||
r'([A-Z][A-Za-z\u00C0-\u017F\-]+)' # 2: surname
|
||||
r'(?:\s*,\s*[A-Za-z.\-]{1,9})?' # initials blob ", M.-L.v."
|
||||
r'(?:\s*,\s*[A-Z][A-Za-z\u00C0-\u017F\-]+\s*,\s*[A-Za-z.\-]{1,9})?' # ", Surname, I."
|
||||
r'\s*,?'
|
||||
r'\s{2,}(\S.*)$')
|
||||
CONT_AUTHOR = re.compile(r'^\s{8,16}([A-Z]\.)\s*$') # lone initial (surname on prev line)
|
||||
CITY_START = re.compile(
|
||||
r'^(New\s+York|London|Boston|Toronto|York|Woodstock|Hove|Ithaca|Cambridge|Tokyo|Geneva|Wilmette|'
|
||||
r'Hillsdale|Northvale|Cham|College\s+Station|Arles|Arlington|Z\u00fcrich|Chicago|Asheville|'
|
||||
r'Oxford|Basel|Paris|Berlin|Princeton|Syracuse|New\s+Haven|New\s+Brunswick|Heidelberg|Amsterdam)\b|'
|
||||
r'^(Publishing|Switzerland|Press,|Books,|University)')
|
||||
SECTION_HDR = re.compile(r'^\s{0,4}[AB]\s*\.?\s*\d*\.?\s+[A-Z]|\d{2}\s+[A-Z][A-Z ]{4,}$')
|
||||
|
||||
def parse_raw(path):
|
||||
"""-> list of entries: {sec, n, author, title, cw, raw}"""
|
||||
lines = open(path, encoding='utf8').read().splitlines()
|
||||
entries, cur = [], None
|
||||
def flush():
|
||||
nonlocal cur
|
||||
if cur and (cur['title'] or cur['cw']):
|
||||
entries.append(cur)
|
||||
cur = None
|
||||
for ln in lines:
|
||||
stripped = ln.strip()
|
||||
if not stripped:
|
||||
flush()
|
||||
continue
|
||||
indent = len(ln) - len(ln.lstrip())
|
||||
if SECTION_HDR.match(stripped) and indent <= 4:
|
||||
flush()
|
||||
continue
|
||||
if stripped.startswith('* required'):
|
||||
flush()
|
||||
continue
|
||||
m = CW_LINE.match(ln)
|
||||
if m:
|
||||
flush()
|
||||
vol = m.group(1).replace('/', '').upper()
|
||||
vol = vol.replace('II', '2').replace('I', '1')
|
||||
cur = {'cw': vol, 'author': 'Jung', 'title': m.group(2), 'raw': stripped}
|
||||
continue
|
||||
m = AUTHOR_LINE.match(ln)
|
||||
if m and not stripped.startswith(('"', '*', '—', '.', 'In:')):
|
||||
if cur and CITY_START.match(m.group(3)):
|
||||
# 2nd author + "City: Publisher, Year" line of the CURRENT entry
|
||||
cur['author'] += ', ' + m.group(2).strip().rstrip(',')
|
||||
cur['title'] += ' ' + m.group(3)
|
||||
continue
|
||||
flush()
|
||||
author = m.group(2).strip().rstrip(',')
|
||||
cur = {'cw': None, 'author': author, 'title': m.group(3), 'raw': stripped}
|
||||
continue
|
||||
m = CONT_AUTHOR.match(ln)
|
||||
if m and cur and ',' not in cur['author'] and not re.search(r'\.$', cur['author']):
|
||||
cur['author'] += ' ' + m.group(1)
|
||||
continue
|
||||
# continuation line — or a NEW book by the same author (bib of current entry is complete)
|
||||
if cur and (indent >= 8 or cur['cw']):
|
||||
if (cur['cw'] is None and indent >= 8 and len(stripped) >= 15
|
||||
and not stripped.startswith(('"', '*', '—', '.', 'In:', 'Part', 'Appendix',
|
||||
'Chap', 'Ch.', 'Vol', 'Note', 'Notes'))
|
||||
and not re.match(r'^[A-Z]{1,4}[\-:,.]?\w{0,6}$', stripped)
|
||||
and re.search(r'[a-z]{3,}', stripped)
|
||||
and re.search(r'\b(?:New\s+York|London|Boston|Toronto|Hove|Ithaca|Woodstock|'
|
||||
r'Cambridge|Tokyo|Geneva|Wilmette|Hillsdale|Northvale|Cham|Arles|'
|
||||
r'Arlington|Chicago|Asheville|Oxford|Basel|Paris|Berlin|Princeton|'
|
||||
r'Syracuse|New\s+Haven|Heidelberg|Amsterdam|Sigtuna)\s*:\s*'
|
||||
r'[A-Za-z][^,]{0,50}?\,\s*\d{4}', cur['title'])):
|
||||
prev_author = cur['author']
|
||||
flush()
|
||||
cur = {'cw': None, 'author': prev_author, 'title': stripped, 'raw': stripped}
|
||||
continue
|
||||
cur['title'] += ' ' + stripped
|
||||
flush()
|
||||
out = []
|
||||
for i, e in enumerate(entries, 1):
|
||||
e['n'] = i
|
||||
t = re.sub(r'\b(JCE|XJE|PA|JE|PF|XW)[\-:]?\w{1,6}\b\.?', ' ', e['title'])
|
||||
t = re.sub(r'\bopen access\b\.?', ' ', t, flags=re.I)
|
||||
# strip "City: Publisher, Year" publisher tail
|
||||
t = re.sub(r'\b(?:New\s+York|London|Boston|Toronto|New\s+Brunswick|Woodstock|Hove|Ithaca|'
|
||||
r'Cambridge|Tokyo|Geneva|Wilmette|Hillsdale|Northvale|Cham|College\s+Station|Arles|'
|
||||
r'Arlington|Chicago|Asheville|Oxford|Basel|Paris|Berlin|Princeton|Syracuse|New\s+Haven|'
|
||||
r'Heidelberg|Amsterdam|Sigtuna|Sigtuna)\s*:\s*[^.,]{0,60}?\,\s*\d{4}', ' ', t)
|
||||
t = re.sub(r'\s+', ' ', t).strip(' .')
|
||||
e['title'] = t
|
||||
out.append(e)
|
||||
return out
|
||||
|
||||
# ---------------- CARDS (01-06) ----------------
|
||||
def card_entries():
|
||||
out = []
|
||||
for path in sorted(glob.glob(os.path.join(BASE, 'sections', '0[1-6]-*', '??-*.md'))):
|
||||
txt = open(path, encoding='utf8').read()
|
||||
h1 = re.search(r'^# (.+)$', txt, re.M)
|
||||
am = re.search(r'\*\*Author\(s?\):\*\*\s*(.+)', txt, re.I)
|
||||
st = re.search(r'\*\*Status:\*\*\s*([✅🔶❌⬜🔎])', txt)
|
||||
if not h1:
|
||||
continue
|
||||
parts = os.path.dirname(path).split('/')
|
||||
sec = parts[-1]
|
||||
item = os.path.basename(path)[:2]
|
||||
title = h1.group(1).split('(')[0]
|
||||
cw = None
|
||||
m = re.search(r'CW\s*(\d+)\s*/?\s*([IVX]+)?', title, re.I)
|
||||
if m:
|
||||
rom = {'I': 1, 'II': 2, 'III': 3, 'IV': 4, 'V': 5}
|
||||
cw = m.group(1) + (str(rom.get(m.group(2).upper())) if m.group(2) else '')
|
||||
if not cw:
|
||||
fm = re.search(r'\bcw(9)[-_]([12])|\bcw(\d{1,2})', os.path.basename(path))
|
||||
if fm:
|
||||
cw = ((fm.group(1) or '') + (fm.group(2) or '')) or fm.group(3)
|
||||
out.append({
|
||||
'sec': sec[:2], 'n': item, 'author': (am.group(1) if am else '').strip(),
|
||||
'title': title, 'tok': norm(title), 'cw': cw, 'status': st.group(1) if st else '?',
|
||||
'file': os.path.join('sections', os.path.dirname(path).split('/')[-1], os.path.basename(path)),
|
||||
})
|
||||
return out
|
||||
|
||||
TRANSLIT = str.maketrans({
|
||||
'а':'a','б':'b','в':'v','г':'g','д':'d','е':'e','ё':'e','ж':'zh','з':'z','и':'i','й':'y',
|
||||
'к':'k','л':'l','м':'m','н':'n','о':'o','п':'p','р':'r','с':'s','т':'t','у':'u','ф':'f',
|
||||
'х':'kh','ц':'ts','ч':'ch','ш':'sh','щ':'shch','ы':'y','ь':'','э':'e','ю':'yu','я':'ya'})
|
||||
|
||||
def author_key(name):
|
||||
"""first surname token, normalized (Cyrillic -> Latin)"""
|
||||
if not name:
|
||||
return ''
|
||||
s = name.split(',')[0] if ',' in name else name
|
||||
s = s.translate(TRANSLIT)
|
||||
toks = [t.rstrip("'") for t in re.findall(r"[a-z']+", s.lower())
|
||||
if t.rstrip("'") not in STOP and t not in ('v', 'von', 'der', 'de', 'la')]
|
||||
return toks[-1] if toks else ''
|
||||
|
||||
# ---------------- matching ----------------
|
||||
GENERIC = {'psychology', 'analytical', 'psychotherapy', 'psychological', 'psychoanalysis',
|
||||
'jung', 'dream', 'dreams', 'unconscious', 'myth', 'myths', 'mythology',
|
||||
'analysis', 'psychic', 'spirit', 'soul', 'self', 'ego', 'mind', 'man', 'human',
|
||||
'transformation', 'transforming', 'goddess', 'gods', 'god', 'testament',
|
||||
'modern', 'contemporary', 'understanding', 'introduction', 'guide', 'handbook',
|
||||
'studies', 'essays', 'meaning', 'meanings', 'sacred', 'religion', 'cultural',
|
||||
'initiation', 'book', 'books', 'volume', 'volumes', 'text', 'texts',
|
||||
'image', 'images', 'evolution', 'greeks', 'greek', 'romans', 'roman',
|
||||
'journey', 'hero', 'heroes', 'tale', 'tales', 'story', 'stories',
|
||||
'fairy', 'folk', 'folklore', 'world', 'works', 'notes', 'lecture', 'lectures'}
|
||||
PUBLISHERS = {'suny', 'routledge', 'penguin', 'spring', 'shambhala', 'karnac', 'chiron', 'sigo',
|
||||
'pantheon', 'wiley', 'basic', 'press', 'ast', 'eksmo', 'bollingen', 'princeton',
|
||||
'cornell', 'thames', 'hudson', 'bodley', 'oxford', 'cambridge', 'harvard',
|
||||
'inner', 'city', 'harper', 'row', 'viking', 'harcourt', 'bradford', 'acorn',
|
||||
'lindisfarne', 'free', 'association', 'daimon', 'philemon', 'abe', 'brill',
|
||||
'harcourt', 'brunner', 'taylor', 'frank', 'casemate', 'essex', 'palgrave'}
|
||||
|
||||
def years(s):
|
||||
return set(re.findall(r'\b(?:19|20)\d{2}\b', s or ''))
|
||||
|
||||
def score_pair(e_tok, e_cw, e_author, c_tok, c_cw, c_author):
|
||||
"""shared identity score for two (title, cw, author) pairs; 0 = reject"""
|
||||
inter = e_tok & c_tok
|
||||
if e_cw and c_cw and e_cw == c_cw:
|
||||
if not inter:
|
||||
return 0.9 # same CW volume, Jung bypasses nothing else
|
||||
elif len(inter) < 2:
|
||||
return 0.0
|
||||
ak = author_key(e_author)
|
||||
cak = author_key(c_author)
|
||||
same_author = ('jung' in (c_author or '').lower() and (e_author or '').startswith('Jung')) or \
|
||||
(cak and ak and (cak == ak or cak in ak or ak in cak))
|
||||
distinct = inter - GENERIC
|
||||
if not same_author:
|
||||
if not (e_cw and c_cw):
|
||||
# lenient: author unreadable (e.g. Cyrillic on card) + very distinctive shared title
|
||||
if len(distinct) >= 3:
|
||||
pass # fall through to scoring
|
||||
else:
|
||||
return 0.0
|
||||
if e_cw and c_cw and e_cw != c_cw:
|
||||
return 0.0
|
||||
# contradictory years (both have years, none shared) = different editions/books
|
||||
ye, yc = years(' '.join(e_tok)), years(' '.join(c_tok))
|
||||
if ye and yc and not (ye & yc):
|
||||
return 0.0
|
||||
if e_cw and c_cw:
|
||||
return 1.0 # CW volume match + >=2 shared tokens
|
||||
if len(distinct) >= 2:
|
||||
return len(distinct) / max(1, len(inter))
|
||||
if len(c_tok - GENERIC) == 0 and cak == ak and not (e_tok - GENERIC) and len(inter) >= 2:
|
||||
return 0.6 # card title is all-generic (e.g. "The Psychology of C.G. Jung")
|
||||
# short title, one distinctive token shared (e.g. "From Freud to Jung")
|
||||
if len(distinct) == 1 and min(len(e_tok), len(c_tok)) <= 4:
|
||||
return 0.5
|
||||
# author in Cyrillic (can't compare reliably) + very distinctive shared title
|
||||
if not same_author and len(distinct) >= 3:
|
||||
return 0.7
|
||||
return 0.0
|
||||
|
||||
def match_entry_to_card(e, cards):
|
||||
best, best_score = None, 0
|
||||
for c in cards:
|
||||
sc = score_pair(norm(e['title']), e['cw'], e['author'], c['tok'], c['cw'], c['author'])
|
||||
if sc > best_score:
|
||||
best, best_score = c, sc
|
||||
return best
|
||||
|
||||
# ---------------- downloads index ----------------
|
||||
def download_index():
|
||||
idx = {}
|
||||
for d in glob.glob(os.path.join(BASE, 'downloads', '0[1-6]-*')):
|
||||
sec = os.path.basename(d)[:2]
|
||||
files = []
|
||||
for f in os.listdir(d):
|
||||
if os.path.isfile(os.path.join(d, f)) and not f.startswith('.'):
|
||||
files.append((f, norm(f)))
|
||||
idx[sec] = files
|
||||
return idx
|
||||
|
||||
def manifest_stale(sec, title, dl_idx, cards):
|
||||
"""(verdict, evidence) for a 'Not downloadable' line"""
|
||||
tok = {t for t in norm(title) if not re.match(r'^(19|20)\d{2}$', t)}
|
||||
if not tok:
|
||||
return 'SKIP', ''
|
||||
for s2, files in dl_idx.items():
|
||||
if s2 == sec:
|
||||
continue
|
||||
for f, ftok in files:
|
||||
ftok = {t for t in ftok
|
||||
if not re.match(r'^(19|20)\d{2}$', t) and t not in PUBLISHERS}
|
||||
inter = tok & ftok
|
||||
if inter - GENERIC: # at least one distinctive shared token
|
||||
return 'STALE', f'sec{s2} file: {f}'
|
||||
for c in cards:
|
||||
if c['sec'] == sec or c['status'] != '✅':
|
||||
continue
|
||||
ctok = {t for t in c['tok'] if not re.match(r'^(19|20)\d{2}$', t)}
|
||||
inter = tok & ctok
|
||||
if inter - GENERIC:
|
||||
return 'CHECK', f'sec{c["sec"]}#{c["n"]} card ✅ ({c["file"]})'
|
||||
return 'KEEP', ''
|
||||
|
||||
# ---------------- main ----------------
|
||||
def main():
|
||||
write = '--write' in sys.argv
|
||||
cards = card_entries()
|
||||
dl_idx = download_index()
|
||||
|
||||
raw_secs = {}
|
||||
for sec, name, path in [
|
||||
('07', 'Complexes & Association Experiment', 'sections/07-complexes/RAW.md'),
|
||||
('08', 'Developmental Psychology', 'data/isap-raw/08-developmental-psychology.txt'),
|
||||
('09', 'Comparison of Psychodynamic Concepts', 'data/isap-raw/09-comparison-of-psychodynamic-concepts.txt'),
|
||||
('10', 'Psychopathology & Psychiatry', 'data/isap-raw/10-psychopathology-psychiatry.txt'),
|
||||
('11', 'The Individuation Process', 'data/isap-raw/11-individuation-process.txt'),
|
||||
('12', 'Practical Case', 'data/isap-raw/12-practical-case.txt'),
|
||||
]:
|
||||
p = os.path.join(BASE, path)
|
||||
if os.path.exists(p):
|
||||
raw_secs[sec] = (name, parse_raw(p))
|
||||
|
||||
matched = {}
|
||||
virt = [] # virtual cards from earlier raw entries: (tok, cw, author, canonical_card)
|
||||
for sec in sorted(raw_secs):
|
||||
for e in raw_secs[sec][1]:
|
||||
e['sec'] = sec
|
||||
c = match_entry_to_card(e, cards)
|
||||
if not c:
|
||||
best_v, best_vs = None, 0
|
||||
for (vtok, vcw, vauthor, vcard) in virt:
|
||||
vs = score_pair(norm(e['title']), e['cw'], e['author'], vtok, vcw, vauthor)
|
||||
if vs > best_vs:
|
||||
best_v, best_vs = vcard, vs
|
||||
c = best_v
|
||||
matched[(sec, e['n'])] = {'kind': 'card' if c else 'new', 'ref': c, 'entry': e}
|
||||
if c:
|
||||
virt.append((norm(e['title']), e['cw'], e['author'], c))
|
||||
|
||||
print("=" * 78)
|
||||
print("RAW 07-12 entries vs existing cards")
|
||||
print("=" * 78)
|
||||
for sec in sorted(raw_secs):
|
||||
name, entries = raw_secs[sec]
|
||||
n_new = n_xref = 0
|
||||
print(f"\n## Section {sec} — {name} ({len(entries)} entries)")
|
||||
for e in entries:
|
||||
m = matched[(sec, e['n'])]
|
||||
t = e['title'][:60]
|
||||
if m['kind'] == 'card':
|
||||
c = m['ref']
|
||||
n_xref += 1
|
||||
print(f" [{e['n']:>2}] {t:60s} -> sec{c['sec']}#{c['n']} ({c['status']})")
|
||||
else:
|
||||
n_new += 1
|
||||
print(f" [{e['n']:>2}] {t:60s} NEW ({e['author']})")
|
||||
print(f" xref={n_xref} new={n_new}")
|
||||
|
||||
print("\n" + "=" * 78)
|
||||
print("MANIFEST 'Not downloadable' — stale lines (book/files exist elsewhere)")
|
||||
print("=" * 78)
|
||||
for mf in sorted(glob.glob(os.path.join(BASE, 'downloads', '0[1-6]-*', 'MANIFEST.md'))):
|
||||
sec = os.path.basename(os.path.dirname(mf))[:2]
|
||||
txt = open(mf, encoding='utf8').read()
|
||||
m = re.search(r'## Not downloadable.*?(?=\n## |\Z)', txt, re.S)
|
||||
if not m:
|
||||
continue
|
||||
stale = []
|
||||
for line in m.group(0).splitlines():
|
||||
if not line.startswith('|') or '---' in line:
|
||||
continue
|
||||
cells = [c.strip() for c in line.strip('|').split('|')]
|
||||
if len(cells) < 2:
|
||||
continue
|
||||
first = cells[0]
|
||||
tm = re.match(r'^(\d{2})\s+(\S.*)$', first)
|
||||
title = tm.group(2) if tm else (cells[1] if len(cells) > 1 else '')
|
||||
title = re.sub(r'\((EN|RU)\)\s*', '', title)
|
||||
title = re.sub(r'^(EN|RU):\s*', '', title)
|
||||
title = re.sub(r'\([^)]*\)', '', title) # author/edition parenthetical
|
||||
title = title.strip()
|
||||
verdict, ev = manifest_stale(sec, title, dl_idx, cards)
|
||||
if verdict in ('STALE', 'CHECK'):
|
||||
stale.append((line.strip(), verdict, ev))
|
||||
if stale:
|
||||
print(f"\n### {mf.split(BASE)[1]}")
|
||||
for line, v, ev in stale:
|
||||
print(f" [{v:5s}] {line[:95]}\n -> {ev}")
|
||||
|
||||
if write:
|
||||
out = os.path.join(BASE, 'data', 'MASTER-LIST.md')
|
||||
L = ['# Master book list — ISAP Zurich reading list (sections 01–12)',
|
||||
'',
|
||||
f'Generated: {datetime.date.today().isoformat()} by `tools/xref.py` — single source of truth',
|
||||
'for cross-section dedup.',
|
||||
'**Rule:** a book listed in several sections is researched once — at its CANONICAL card',
|
||||
'(first section it appears in). Later sections get cross-refs, no new search/download.',
|
||||
'']
|
||||
also = {}
|
||||
for (sec, n), m in matched.items():
|
||||
if m['kind'] == 'card':
|
||||
c = m['ref']
|
||||
also.setdefault((c['sec'], c['n']), []).append(f'sec{sec}#{n}')
|
||||
# 01-06 cross-section duplicates (canonical = earliest section)
|
||||
for c1, c2 in itertools.combinations(cards, 2):
|
||||
if c1['sec'] == c2['sec']:
|
||||
continue
|
||||
if score_pair(c1['tok'], c1['cw'], c1['author'], c2['tok'], c2['cw'], c2['author']) >= 0.5:
|
||||
first, second = (c1, c2) if c1['sec'] < c2['sec'] else (c2, c1)
|
||||
also.setdefault((first['sec'], first['n']), []).append(f"sec{second['sec']}#{second['n']}")
|
||||
for sec in ['01', '02', '03', '04', '05', '06']:
|
||||
sec_cards = [c for c in cards if c['sec'] == sec]
|
||||
L.append(f'## Section {sec} — {len(sec_cards)} items (canonical cards)')
|
||||
L.append('')
|
||||
L.append('| # | EN title | Status | Also listed in |')
|
||||
L.append('|---|----------|--------|----------------|')
|
||||
for c in sorted(sec_cards, key=lambda x: x['n']):
|
||||
a = also.get((c['sec'], c['n']), [])
|
||||
L.append(f'| {c["n"]} | {c["title"][:70]} | {c["status"]} | {", ".join(sorted(a)) or "—"} |')
|
||||
L.append('')
|
||||
for sec in sorted(raw_secs):
|
||||
name, entries = raw_secs[sec]
|
||||
L.append(f'## Section {sec} — {name} (RAW, provisional numbering)')
|
||||
L.append('')
|
||||
L.append('| # | EN title | Author | Canonical | New? |')
|
||||
L.append('|---|----------|--------|-----------|------|')
|
||||
for e in entries:
|
||||
m = matched[(sec, e['n'])]
|
||||
if m['kind'] == 'card':
|
||||
c = m['ref']
|
||||
L.append(f'| {e["n"]} | {e["title"][:60]} | {e["author"]} | sec{c["sec"]}#{c["n"]} ({c["status"]}) | — |')
|
||||
else:
|
||||
L.append(f'| {e["n"]} | {e["title"][:60]} | {e["author"]} | — | **NEW** |')
|
||||
L.append('')
|
||||
open(out, 'w', encoding='utf8').write('\n'.join(L))
|
||||
print(f'\nWROTE {out}')
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
Loading…
Add table
Add a link
Reference in a new issue