#!/usr/bin/env python3 """xref.py — master book list + cross-section dedup + manifest freshness check. Builds the single source of truth for "which book belongs to which section" across ALL 12 ISAP reading-list sections, so every book is searched exactly once: 1. Sections 01-06: canonical entries come from the CARDS (sections/0N/NN-*.md). 2. Sections 07-12: entries parsed from RAW lists (sections/0N/RAW.md or data/isap-raw/NN-*.txt). 3. Every 07-12 entry is matched to an existing 01-06 card (or to an earlier 07-12 entry) -> "canonical" pointer; unmatched entries = NEW (to search). 4. Every "Not downloadable" line in downloads/0N/MANIFEST.md is checked: if the same book's FILES live in another section's download dir -> STALE line (should be a cross-ref), otherwise KEEP. Outputs: data/MASTER-LIST.md — the master list (all sections, canonical pointers) console — stale manifest line report (actionable) Run: python3 tools/xref.py (report only) python3 tools/xref.py --write (rewrite data/MASTER-LIST.md) """ import re, os, sys, glob, unicodedata, datetime, itertools BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) STOP = {'the','a','an','of','in','and','to','on','for','from','by','with','its','his','her', 'cw','jce','xje','pa','je','pf','xw','vol','pp','part','chap','chapter','esp','en','ru', 'st','ltd','int','world','studies'} def norm(s): s = unicodedata.normalize('NFKD', s or '') s = s.lower() s = re.sub(r'(\d)-(\d)', r'\1 \2', s) # 1936-1940 -> "1936 1940" s = re.sub(r'[^a-z0-9 ]', ' ', s) return set(w for w in s.split() if len(w) > 3 and w not in STOP) def overlap(a, b): if not a or not b: return 0.0, 0 inter = a & b return len(inter) / min(len(a), len(b)), len(inter) # ---------------- RAW parsing (07-12) ---------------- CW_LINE = re.compile(r'^\s{0,8}CW\s*(\d+/\w{1,2}|\d+|-?S\d?)\s+(\S.*)$') # author-start: one or more "Surname, I." groups, then >=2 spaces, then title AUTHOR_LINE = re.compile( r'^(\s{0,8})' r'([A-Z][A-Za-z\u00C0-\u017F\-]+)' # 2: surname r'(?:\s*,\s*[A-Za-z.\-](?:\s?[A-Za-z.\-]){0,8}|\s+[A-Z]\.)?' # initials ", C. A." or space-initial "Kalsched D." r'(?:\s*,\s*[A-Z][A-Za-z\u00C0-\u017F\-]+\s*,\s*[A-Za-z.\-](?:\s?[A-Za-z.\-]){0,8}){0,3}' # ", Surname, I." x3 r'\s*,?' r'\s{2,}(\S.*)$') CONT_AUTHOR = re.compile(r'^\s{8,16}([A-Z]\.)\s*$') # lone initial (surname on prev line) CITY_START = re.compile( r'^(New\s+York|London|Boston|Toronto|York|Woodstock|Hove|Ithaca|Cambridge|Tokyo|Geneva|Wilmette|' r'Hillsdale|Northvale|Cham|College\s+Station|Arles|Arlington|Z\u00fcrich|Chicago|Asheville|' r'Oxford|Basel|Paris|Berlin|Princeton|Syracuse|New\s+Haven|New\s+Brunswick|Heidelberg|Amsterdam)\b|' r'^(Publishing|Switzerland|Press,|Books,|University)') SECTION_HDR = re.compile(r'^\s{0,4}[AB]\s*\.?\s*\d*\.?\s+[A-Z]|\d{2}\s+[A-Z][A-Z ]{4,}$') def parse_raw(path): """-> list of entries: {sec, n, author, title, cw, raw}""" lines = open(path, encoding='utf8').read().splitlines() entries, cur = [], None def flush(): nonlocal cur if cur and (cur['title'] or cur['cw']): entries.append(cur) cur = None for ln in lines: stripped = ln.strip() if not stripped: flush() continue indent = len(ln) - len(ln.lstrip()) if SECTION_HDR.match(stripped) and indent <= 4: flush() continue if stripped.startswith('* required'): flush() continue m = CW_LINE.match(ln) if m: flush() vol = m.group(1).replace('/', '').upper() vol = vol.replace('II', '2').replace('I', '1') cur = {'cw': vol, 'author': 'Jung', 'title': m.group(2), 'raw': stripped} continue m = AUTHOR_LINE.match(ln) if m and not stripped.startswith(('"', '*', '—', '.', 'In:', '“', '‘', '«')): if cur and cur['author'].endswith('-'): # hyphen-split surname across lines sp = re.search(r'\s{2,}', ln.lstrip()) author_part = (ln.lstrip()[:sp.start()].rstrip() if sp else ln.lstrip().rstrip()).rstrip(',') cur['author'] = cur['author'].rstrip('-') + '-' + re.sub(r'\s+', ' ', author_part).strip() cur['title'] += ' ' + m.group(3).strip() continue if cur and CITY_START.match(m.group(3)): # 2nd author + "City: Publisher, Year" line of the CURRENT entry cur['author'] += ', ' + m.group(2).strip().rstrip(',') cur['title'] += ' ' + m.group(3) continue if cur and re.match(r'^\d{4}\.?\s*$', m.group(3).strip()): # continuation author line: "Pine, F., 1975." (bib tail, no title) cur['author'] += ', ' + m.group(2).strip().rstrip(',') cur['title'] += ' ' + m.group(3).strip() continue if cur and not re.search(r'\d{4}', cur['title']) and re.search(r'\b(in|of|the|and|a|to|for|on|between|with)$', cur['title']): # multi-author entry split across lines (Hersh/Caligor/Yeomans case) sp = re.search(r'\s{2,}', ln.lstrip()) author_part = (ln.lstrip()[:sp.start()].rstrip() if sp else ln.lstrip().rstrip()).rstrip(',') cur['author'] += ', ' + re.sub(r'\s+', ' ', author_part).strip() cur['title'] += ' ' + m.group(3).strip() continue flush() author = m.group(2).strip().rstrip(',') cur = {'cw': None, 'author': author, 'title': m.group(3), 'raw': stripped} continue m = CONT_AUTHOR.match(ln) if m and cur and ',' not in cur['author'] and not re.search(r'\.$', cur['author']): cur['author'] += ' ' + m.group(1) continue # continuation line — or a NEW book by the same author (bib of current entry is complete) if cur and (indent >= 8 or cur['cw']): if (cur['cw'] is None and indent >= 8 and len(stripped) >= 15 and not stripped.startswith(('"', '*', '—', '.', 'In:', 'Part', 'Appendix', 'Chap', 'Ch.', 'Vol', 'Note', 'Notes', '“', '‘', '«')) and not re.match(r'^[A-Z]{1,4}[\-:,.]?\w{0,6}$', stripped) and re.search(r'[a-z]{3,}', stripped) and (re.search(r'\b(?:New\s+York|London|Boston|Toronto|Hove|Ithaca|Woodstock|' r'Cambridge|Tokyo|Geneva|Wilmette|Hillsdale|Northvale|Cham|Arles|' r'Arlington|Chicago|Asheville|Oxford|Basel|Paris|Berlin|Princeton|' r'Syracuse|New\s+Haven|Heidelberg|Amsterdam|Sigtuna)\s*:\s*' r'[A-Za-z][^,]{0,50}?\,\s*\d{4}', cur['title']) or re.search(r'(?:19|20)\d{2}', cur['title']))): prev_author = cur['author'] flush() cur = {'cw': None, 'author': prev_author, 'title': stripped, 'raw': stripped} continue cur['title'] += ' ' + stripped flush() out = [] for i, e in enumerate(entries, 1): e['n'] = i t = re.sub(r'\b(JCE|XJE|PA|JE|PF|XW)[\-:]?\w{1,6}\b\.?', ' ', e['title']) t = re.sub(r'\bopen access\b\.?', ' ', t, flags=re.I) # strip "City: Publisher, Year" publisher tail t = re.sub(r'\b(?:New\s+York|London|Boston|Toronto|New\s+Brunswick|Woodstock|Hove|Ithaca|' r'Cambridge|Tokyo|Geneva|Wilmette|Hillsdale|Northvale|Cham|College\s+Station|Arles|' r'Arlington|Chicago|Asheville|Oxford|Basel|Paris|Berlin|Princeton|Syracuse|New\s+Haven|' r'Heidelberg|Amsterdam|Sigtuna|Sigtuna)\s*:\s*[^.,]{0,60}?\,\s*\d{4}', ' ', t) t = re.sub(r'\s+', ' ', t).strip(' .') e['title'] = t out.append(e) return out # ---------------- CARDS (01-06) ---------------- def card_entries(): out = [] for path in sorted(glob.glob(os.path.join(BASE, 'sections', '0[1-9]-*', '??-*.md'))): txt = open(path, encoding='utf8').read() h1 = re.search(r'^# (.+)$', txt, re.M) am = re.search(r'\*\*Author\(s?\):\*\*\s*(.+)', txt, re.I) st = re.search(r'\*\*Status:\*\*\s*([✅🔶❌⬜🔎])', txt) if not h1: continue parts = os.path.dirname(path).split('/') sec = parts[-1] item = os.path.basename(path)[:2] title = h1.group(1).split('(')[0] cw = None m = re.search(r'CW\s*(\d+)\s*/?\s*([IVX]+)?', title, re.I) if m: rom = {'I': 1, 'II': 2, 'III': 3, 'IV': 4, 'V': 5} cw = m.group(1) + (str(rom.get(m.group(2).upper())) if m.group(2) else '') if not cw: fm = re.search(r'\bcw(9)[-_]([12])|\bcw(\d{1,2})', os.path.basename(path)) if fm: cw = ((fm.group(1) or '') + (fm.group(2) or '')) or fm.group(3) out.append({ 'sec': sec[:2], 'n': item, 'author': (am.group(1) if am else '').strip(), 'title': title, 'tok': norm(title), 'cw': cw, 'status': st.group(1) if st else '?', 'file': os.path.join('sections', os.path.dirname(path).split('/')[-1], os.path.basename(path)), }) return out TRANSLIT = str.maketrans({ 'а':'a','б':'b','в':'v','г':'g','д':'d','е':'e','ё':'e','ж':'zh','з':'z','и':'i','й':'y', 'к':'k','л':'l','м':'m','н':'n','о':'o','п':'p','р':'r','с':'s','т':'t','у':'u','ф':'f', 'х':'kh','ц':'ts','ч':'ch','ш':'sh','щ':'shch','ы':'y','ь':'','э':'e','ю':'yu','я':'ya'}) def author_key(name): """first surname token, normalized (Cyrillic -> Latin)""" if not name: return '' s = name.split(',')[0] if ',' in name else name s = s.translate(TRANSLIT) toks = [t.rstrip("'") for t in re.findall(r"[a-z']+", s.lower()) if t.rstrip("'") not in STOP and t not in ('v', 'von', 'der', 'de', 'la')] return toks[-1] if toks else '' # ---------------- matching ---------------- GENERIC = {'psychology', 'analytical', 'psychotherapy', 'psychological', 'psychoanalysis', 'jung', 'dream', 'dreams', 'unconscious', 'myth', 'myths', 'mythology', 'analysis', 'psychic', 'spirit', 'soul', 'self', 'ego', 'mind', 'man', 'human', 'transformation', 'transforming', 'goddess', 'gods', 'god', 'testament', 'modern', 'contemporary', 'understanding', 'introduction', 'guide', 'handbook', 'studies', 'essays', 'meaning', 'meanings', 'sacred', 'religion', 'cultural', 'initiation', 'book', 'books', 'volume', 'volumes', 'text', 'texts', 'image', 'images', 'evolution', 'greeks', 'greek', 'romans', 'roman', 'journey', 'hero', 'heroes', 'tale', 'tales', 'story', 'stories', 'fairy', 'folk', 'folklore', 'world', 'works', 'notes', 'lecture', 'lectures'} PUBLISHERS = {'suny', 'routledge', 'penguin', 'spring', 'shambhala', 'karnac', 'chiron', 'sigo', 'pantheon', 'wiley', 'basic', 'press', 'ast', 'eksmo', 'bollingen', 'princeton', 'cornell', 'thames', 'hudson', 'bodley', 'oxford', 'cambridge', 'harvard', 'inner', 'city', 'harper', 'row', 'viking', 'harcourt', 'bradford', 'acorn', 'lindisfarne', 'free', 'association', 'daimon', 'philemon', 'abe', 'brill', 'harcourt', 'brunner', 'taylor', 'frank', 'casemate', 'essex', 'palgrave'} def years(s): return set(re.findall(r'\b(?:19|20)\d{2}\b', s or '')) def score_pair(e_tok, e_cw, e_author, c_tok, c_cw, c_author): """shared identity score for two (title, cw, author) pairs; 0 = reject""" inter = e_tok & c_tok if e_cw and c_cw and e_cw == c_cw: if not inter: return 0.9 # same CW volume, Jung bypasses nothing else elif len(inter) < 2: return 0.0 ak = author_key(e_author) cak = author_key(c_author) same_author = ('jung' in (c_author or '').lower() and (e_author or '').startswith('Jung')) or \ (cak and ak and (cak == ak or cak in ak or ak in cak)) distinct = inter - GENERIC if not same_author: if not (e_cw and c_cw): # lenient: author unreadable (e.g. Cyrillic on card) + very distinctive shared title if len(distinct) >= 3: pass # fall through to scoring else: return 0.0 if e_cw and c_cw and e_cw != c_cw: return 0.0 # contradictory years (both have years, none shared) = different editions/books ye, yc = years(' '.join(e_tok)), years(' '.join(c_tok)) if ye and yc and not (ye & yc): return 0.0 if e_cw and c_cw: return 1.0 # CW volume match + >=2 shared tokens if len(distinct) >= 2: # min-set overlap ratio (+ eps by inter size to break subset ties: Vol.I vs Vol.II) return len(inter) / max(1, min(len(e_tok), len(c_tok))) + len(inter) / 1000.0 if len(c_tok - GENERIC) == 0 and cak == ak and not (e_tok - GENERIC) and len(inter) >= 2: return 0.6 # card title is all-generic (e.g. "The Psychology of C.G. Jung") # short title, one distinctive token shared (e.g. "From Freud to Jung") if len(distinct) == 1 and min(len(e_tok), len(c_tok)) <= 4: return 0.5 # author in Cyrillic (can't compare reliably) + very distinctive shared title if not same_author and len(distinct) >= 3: return 0.7 return 0.0 def match_entry_to_card(e, cards): best, best_score = None, 0 for c in cards: sc = score_pair(norm(e['title']), e['cw'], e['author'], c['tok'], c['cw'], c['author']) if sc > best_score: best, best_score = c, sc return best # ---------------- downloads index ---------------- def download_index(): idx = {} for d in glob.glob(os.path.join(BASE, 'downloads', '0[1-6]-*')): sec = os.path.basename(d)[:2] files = [] for f in os.listdir(d): if os.path.isfile(os.path.join(d, f)) and not f.startswith('.'): files.append((f, norm(f))) idx[sec] = files return idx def manifest_stale(sec, title, dl_idx, cards): """(verdict, evidence) for a 'Not downloadable' line""" tok = {t for t in norm(title) if not re.match(r'^(19|20)\d{2}$', t)} if not tok: return 'SKIP', '' for s2, files in dl_idx.items(): if s2 == sec: continue for f, ftok in files: ftok = {t for t in ftok if not re.match(r'^(19|20)\d{2}$', t) and t not in PUBLISHERS} inter = tok & ftok if len(inter - GENERIC) >= 2: # at least two distinctive shared tokens return 'STALE', f'sec{s2} file: {f}' for c in cards: if c['sec'] == sec or c['status'] != '✅': continue ctok = {t for t in c['tok'] if not re.match(r'^(19|20)\d{2}$', t)} inter = tok & ctok if len(inter - GENERIC) >= 2: # 1 distinctive token = too weak (Stein «treatment» ≠ Kohut) return 'CHECK', f'sec{c["sec"]}#{c["n"]} card ✅ ({c["file"]})' return 'KEEP', '' # ---------------- main ---------------- def main(): write = '--write' in sys.argv cards = card_entries() dl_idx = download_index() raw_secs = {} for sec, name, path in [ ('07', 'Complexes & Association Experiment', 'sections/07-complexes/RAW.md'), ('08', 'Developmental Psychology', 'sections/08-developmental/RAW.md'), ('09', 'Comparison of Psychodynamic Concepts', 'data/isap-raw/09-comparison-of-psychodynamic-concepts.txt'), ('10', 'Psychopathology & Psychiatry', 'data/isap-raw/10-psychopathology-psychiatry.txt'), ('11', 'The Individuation Process', 'data/isap-raw/11-individuation-process.txt'), ('12', 'Practical Case', 'data/isap-raw/12-practical-case.txt'), ]: p = os.path.join(BASE, path) if os.path.exists(p): raw_secs[sec] = (name, parse_raw(p)) matched = {} virt = [] # virtual cards from earlier raw entries: (tok, cw, author, canonical_card) for sec in sorted(raw_secs): for e in raw_secs[sec][1]: e['sec'] = sec c = match_entry_to_card(e, cards) if not c: best_v, best_vs = None, 0 for (vtok, vcw, vauthor, vcard) in virt: vs = score_pair(norm(e['title']), e['cw'], e['author'], vtok, vcw, vauthor) if vs > best_vs: best_v, best_vs = vcard, vs c = best_v matched[(sec, e['n'])] = {'kind': 'card' if c else 'new', 'ref': c, 'entry': e} if c: virt.append((norm(e['title']), e['cw'], e['author'], c)) print("=" * 78) print("RAW 07-12 entries vs existing cards") print("=" * 78) for sec in sorted(raw_secs): name, entries = raw_secs[sec] n_new = n_xref = 0 print(f"\n## Section {sec} — {name} ({len(entries)} entries)") for e in entries: m = matched[(sec, e['n'])] t = e['title'][:60] if m['kind'] == 'card': c = m['ref'] n_xref += 1 print(f" [{e['n']:>2}] {t:60s} -> sec{c['sec']}#{c['n']} ({c['status']})") else: n_new += 1 print(f" [{e['n']:>2}] {t:60s} NEW ({e['author']})") print(f" xref={n_xref} new={n_new}") print("\n" + "=" * 78) print("MANIFEST 'Not downloadable' — stale lines (book/files exist elsewhere)") print("=" * 78) for mf in sorted(glob.glob(os.path.join(BASE, 'downloads', '0[1-6]-*', 'MANIFEST.md'))): sec = os.path.basename(os.path.dirname(mf))[:2] txt = open(mf, encoding='utf8').read() m = re.search(r'## Not downloadable.*?(?=\n## |\Z)', txt, re.S) if not m: continue stale = [] for line in m.group(0).splitlines(): if not line.startswith('|') or '---' in line: continue cells = [c.strip() for c in line.strip('|').split('|')] if len(cells) < 2: continue first = cells[0] tm = re.match(r'^(\d{2})\s+(\S.*)$', first) title = tm.group(2) if tm else (cells[1] if len(cells) > 1 else '') title = re.sub(r'\((EN|RU)\)\s*', '', title) title = re.sub(r'^(EN|RU):\s*', '', title) title = re.sub(r'\([^)]*\)', '', title) # author/edition parenthetical title = title.strip() verdict, ev = manifest_stale(sec, title, dl_idx, cards) if verdict in ('STALE', 'CHECK'): stale.append((line.strip(), verdict, ev)) if stale: print(f"\n### {mf.split(BASE)[1]}") for line, v, ev in stale: print(f" [{v:5s}] {line[:95]}\n -> {ev}") if write: out = os.path.join(BASE, 'data', 'MASTER-LIST.md') L = ['# Master book list — ISAP Zurich reading list (sections 01–12)', '', f'Generated: {datetime.date.today().isoformat()} by `tools/xref.py` — single source of truth', 'for cross-section dedup.', '**Rule:** a book listed in several sections is researched once — at its CANONICAL card', '(first section it appears in). Later sections get cross-refs, no new search/download.', ''] also = {} for (sec, n), m in matched.items(): if m['kind'] == 'card': c = m['ref'] also.setdefault((c['sec'], c['n']), []).append(f'sec{sec}#{n}') # 01-06 cross-section duplicates (canonical = earliest section) for c1, c2 in itertools.combinations(cards, 2): if c1['sec'] == c2['sec']: continue if score_pair(c1['tok'], c1['cw'], c1['author'], c2['tok'], c2['cw'], c2['author']) >= 0.5: first, second = (c1, c2) if c1['sec'] < c2['sec'] else (c2, c1) also.setdefault((first['sec'], first['n']), []).append(f"sec{second['sec']}#{second['n']}") for sec in ['01', '02', '03', '04', '05', '06']: sec_cards = [c for c in cards if c['sec'] == sec] L.append(f'## Section {sec} — {len(sec_cards)} items (canonical cards)') L.append('') L.append('| # | EN title | Status | Also listed in |') L.append('|---|----------|--------|----------------|') for c in sorted(sec_cards, key=lambda x: x['n']): a = also.get((c['sec'], c['n']), []) L.append(f'| {c["n"]} | {c["title"][:70]} | {c["status"]} | {", ".join(sorted(a)) or "—"} |') L.append('') for sec in sorted(raw_secs): name, entries = raw_secs[sec] L.append(f'## Section {sec} — {name} (RAW, provisional numbering)') L.append('') L.append('| # | EN title | Author | Canonical | New? |') L.append('|---|----------|--------|-----------|------|') for e in entries: m = matched[(sec, e['n'])] if m['kind'] == 'card': c = m['ref'] L.append(f'| {e["n"]} | {e["title"][:60]} | {e["author"]} | sec{c["sec"]}#{c["n"]} ({c["status"]}) | — |') else: L.append(f'| {e["n"]} | {e["title"][:60]} | {e["author"]} | — | **NEW** |') L.append('') open(out, 'w', encoding='utf8').write('\n'.join(L)) print(f'\nWROTE {out}') if __name__ == '__main__': main()