jung/tools/xref.py
Dmitry Kokorin e20e2ba1a5 sec09 FINAL: downloads (41 files, 361MB: EN 27 + RU 14) + MANIFEST + SUMMARY + MISSING + AGENTS.md
- All 41 files content-verified (pdfinfo + first-page text; epubs xhtml; fb2 titles)
- 08 Fonagy = EN original (renamed -en-); Wirtz #38 = BOOK (Spring Journal Books 2014, 358pp)
- Bonus: Kawai Myōe (Castalia 2018 part 1, flib x2) + TFP Clinical Guide (Yeomans/Clarikin/Kernberg 2018, DIFFERENT TFP book)
- xref.py stale-check: 1-token distinctive match too weak (Stein/Kohut 'treatment' false positive) -> >=2 required
- .gitattributes: +sec05/07/08/09 LFS dir rules
- Linter 0, xref stale 0
2026-09-27 14:11:05 +03:00

443 lines
22 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""xref.py — master book list + cross-section dedup + manifest freshness check.
Builds the single source of truth for "which book belongs to which section"
across ALL 12 ISAP reading-list sections, so every book is searched exactly once:
1. Sections 01-06: canonical entries come from the CARDS (sections/0N/NN-*.md).
2. Sections 07-12: entries parsed from RAW lists (sections/0N/RAW.md or
data/isap-raw/NN-*.txt).
3. Every 07-12 entry is matched to an existing 01-06 card (or to an earlier
07-12 entry) -> "canonical" pointer; unmatched entries = NEW (to search).
4. Every "Not downloadable" line in downloads/0N/MANIFEST.md is checked: if the
same book's FILES live in another section's download dir -> STALE line
(should be a cross-ref), otherwise KEEP.
Outputs:
data/MASTER-LIST.md — the master list (all sections, canonical pointers)
console — stale manifest line report (actionable)
Run: python3 tools/xref.py (report only)
python3 tools/xref.py --write (rewrite data/MASTER-LIST.md)
"""
import re, os, sys, glob, unicodedata, datetime, itertools
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
STOP = {'the','a','an','of','in','and','to','on','for','from','by','with','its','his','her',
'cw','jce','xje','pa','je','pf','xw','vol','pp','part','chap','chapter','esp','en','ru',
'st','ltd','int','world','studies'}
def norm(s):
s = unicodedata.normalize('NFKD', s or '')
s = s.lower()
s = re.sub(r'(\d)-(\d)', r'\1 \2', s) # 1936-1940 -> "1936 1940"
s = re.sub(r'[^a-z0-9 ]', ' ', s)
return set(w for w in s.split() if len(w) > 3 and w not in STOP)
def overlap(a, b):
if not a or not b:
return 0.0, 0
inter = a & b
return len(inter) / min(len(a), len(b)), len(inter)
# ---------------- RAW parsing (07-12) ----------------
CW_LINE = re.compile(r'^\s{0,8}CW\s*(\d+/\w{1,2}|\d+|-?S\d?)\s+(\S.*)$')
# author-start: one or more "Surname, I." groups, then >=2 spaces, then title
AUTHOR_LINE = re.compile(
r'^(\s{0,8})'
r'([A-Z][A-Za-z\u00C0-\u017F\-]+)' # 2: surname
r'(?:\s*,\s*[A-Za-z.\-](?:\s?[A-Za-z.\-]){0,8}|\s+[A-Z]\.)?' # initials ", C. A." or space-initial "Kalsched D."
r'(?:\s*,\s*[A-Z][A-Za-z\u00C0-\u017F\-]+\s*,\s*[A-Za-z.\-](?:\s?[A-Za-z.\-]){0,8}){0,3}' # ", Surname, I." x3
r'\s*,?'
r'\s{2,}(\S.*)$')
CONT_AUTHOR = re.compile(r'^\s{8,16}([A-Z]\.)\s*$') # lone initial (surname on prev line)
CITY_START = re.compile(
r'^(New\s+York|London|Boston|Toronto|York|Woodstock|Hove|Ithaca|Cambridge|Tokyo|Geneva|Wilmette|'
r'Hillsdale|Northvale|Cham|College\s+Station|Arles|Arlington|Z\u00fcrich|Chicago|Asheville|'
r'Oxford|Basel|Paris|Berlin|Princeton|Syracuse|New\s+Haven|New\s+Brunswick|Heidelberg|Amsterdam)\b|'
r'^(Publishing|Switzerland|Press,|Books,|University)')
SECTION_HDR = re.compile(r'^\s{0,4}[AB]\s*\.?\s*\d*\.?\s+[A-Z]|\d{2}\s+[A-Z][A-Z ]{4,}$')
def parse_raw(path):
"""-> list of entries: {sec, n, author, title, cw, raw}"""
lines = open(path, encoding='utf8').read().splitlines()
entries, cur = [], None
def flush():
nonlocal cur
if cur and (cur['title'] or cur['cw']):
entries.append(cur)
cur = None
for ln in lines:
stripped = ln.strip()
if not stripped:
flush()
continue
indent = len(ln) - len(ln.lstrip())
if SECTION_HDR.match(stripped) and indent <= 4:
flush()
continue
if stripped.startswith('* required'):
flush()
continue
m = CW_LINE.match(ln)
if m:
flush()
vol = m.group(1).replace('/', '').upper()
vol = vol.replace('II', '2').replace('I', '1')
cur = {'cw': vol, 'author': 'Jung', 'title': m.group(2), 'raw': stripped}
continue
m = AUTHOR_LINE.match(ln)
if m and not stripped.startswith(('"', '*', '—', '.', 'In:', '“', '‘', '«')):
if cur and cur['author'].endswith('-'):
# hyphen-split surname across lines
sp = re.search(r'\s{2,}', ln.lstrip())
author_part = (ln.lstrip()[:sp.start()].rstrip() if sp else ln.lstrip().rstrip()).rstrip(',')
cur['author'] = cur['author'].rstrip('-') + '-' + re.sub(r'\s+', ' ', author_part).strip()
cur['title'] += ' ' + m.group(3).strip()
continue
if cur and CITY_START.match(m.group(3)):
# 2nd author + "City: Publisher, Year" line of the CURRENT entry
cur['author'] += ', ' + m.group(2).strip().rstrip(',')
cur['title'] += ' ' + m.group(3)
continue
if cur and re.match(r'^\d{4}\.?\s*$', m.group(3).strip()):
# continuation author line: "Pine, F., 1975." (bib tail, no title)
cur['author'] += ', ' + m.group(2).strip().rstrip(',')
cur['title'] += ' ' + m.group(3).strip()
continue
if cur and not re.search(r'\d{4}', cur['title']) and re.search(r'\b(in|of|the|and|a|to|for|on|between|with)$', cur['title']):
# multi-author entry split across lines (Hersh/Caligor/Yeomans case)
sp = re.search(r'\s{2,}', ln.lstrip())
author_part = (ln.lstrip()[:sp.start()].rstrip() if sp else ln.lstrip().rstrip()).rstrip(',')
cur['author'] += ', ' + re.sub(r'\s+', ' ', author_part).strip()
cur['title'] += ' ' + m.group(3).strip()
continue
flush()
author = m.group(2).strip().rstrip(',')
cur = {'cw': None, 'author': author, 'title': m.group(3), 'raw': stripped}
continue
m = CONT_AUTHOR.match(ln)
if m and cur and ',' not in cur['author'] and not re.search(r'\.$', cur['author']):
cur['author'] += ' ' + m.group(1)
continue
# continuation line — or a NEW book by the same author (bib of current entry is complete)
if cur and (indent >= 8 or cur['cw']):
if (cur['cw'] is None and indent >= 8 and len(stripped) >= 15
and not stripped.startswith(('"', '*', '—', '.', 'In:', 'Part', 'Appendix',
'Chap', 'Ch.', 'Vol', 'Note', 'Notes', '“', '‘', '«'))
and not re.match(r'^[A-Z]{1,4}[\-:,.]?\w{0,6}$', stripped)
and re.search(r'[a-z]{3,}', stripped)
and (re.search(r'\b(?:New\s+York|London|Boston|Toronto|Hove|Ithaca|Woodstock|'
r'Cambridge|Tokyo|Geneva|Wilmette|Hillsdale|Northvale|Cham|Arles|'
r'Arlington|Chicago|Asheville|Oxford|Basel|Paris|Berlin|Princeton|'
r'Syracuse|New\s+Haven|Heidelberg|Amsterdam|Sigtuna)\s*:\s*'
r'[A-Za-z][^,]{0,50}?\,\s*\d{4}', cur['title'])
or re.search(r'(?:19|20)\d{2}', cur['title']))):
prev_author = cur['author']
flush()
cur = {'cw': None, 'author': prev_author, 'title': stripped, 'raw': stripped}
continue
cur['title'] += ' ' + stripped
flush()
out = []
for i, e in enumerate(entries, 1):
e['n'] = i
t = re.sub(r'\b(JCE|XJE|PA|JE|PF|XW)[\-:]?\w{1,6}\b\.?', ' ', e['title'])
t = re.sub(r'\bopen access\b\.?', ' ', t, flags=re.I)
# strip "City: Publisher, Year" publisher tail
t = re.sub(r'\b(?:New\s+York|London|Boston|Toronto|New\s+Brunswick|Woodstock|Hove|Ithaca|'
r'Cambridge|Tokyo|Geneva|Wilmette|Hillsdale|Northvale|Cham|College\s+Station|Arles|'
r'Arlington|Chicago|Asheville|Oxford|Basel|Paris|Berlin|Princeton|Syracuse|New\s+Haven|'
r'Heidelberg|Amsterdam|Sigtuna|Sigtuna)\s*:\s*[^.,]{0,60}?\,\s*\d{4}', ' ', t)
t = re.sub(r'\s+', ' ', t).strip(' .')
e['title'] = t
out.append(e)
return out
# ---------------- CARDS (01-06) ----------------
def card_entries():
out = []
for path in sorted(glob.glob(os.path.join(BASE, 'sections', '0[1-9]-*', '??-*.md'))):
txt = open(path, encoding='utf8').read()
h1 = re.search(r'^# (.+)$', txt, re.M)
am = re.search(r'\*\*Author\(s?\):\*\*\s*(.+)', txt, re.I)
st = re.search(r'\*\*Status:\*\*\s*([✅🔶❌⬜🔎])', txt)
if not h1:
continue
parts = os.path.dirname(path).split('/')
sec = parts[-1]
item = os.path.basename(path)[:2]
title = h1.group(1).split('(')[0]
cw = None
m = re.search(r'CW\s*(\d+)\s*/?\s*([IVX]+)?', title, re.I)
if m:
rom = {'I': 1, 'II': 2, 'III': 3, 'IV': 4, 'V': 5}
cw = m.group(1) + (str(rom.get(m.group(2).upper())) if m.group(2) else '')
if not cw:
fm = re.search(r'\bcw(9)[-_]([12])|\bcw(\d{1,2})', os.path.basename(path))
if fm:
cw = ((fm.group(1) or '') + (fm.group(2) or '')) or fm.group(3)
out.append({
'sec': sec[:2], 'n': item, 'author': (am.group(1) if am else '').strip(),
'title': title, 'tok': norm(title), 'cw': cw, 'status': st.group(1) if st else '?',
'file': os.path.join('sections', os.path.dirname(path).split('/')[-1], os.path.basename(path)),
})
return out
TRANSLIT = str.maketrans({
'а':'a','б':'b','в':'v','г':'g','д':'d','е':'e','ё':'e','ж':'zh','з':'z','и':'i','й':'y',
'к':'k','л':'l','м':'m','н':'n','о':'o','п':'p','р':'r','с':'s','т':'t','у':'u','ф':'f',
'х':'kh','ц':'ts','ч':'ch','ш':'sh','щ':'shch','ы':'y','ь':'','э':'e','ю':'yu','я':'ya'})
def author_key(name):
"""first surname token, normalized (Cyrillic -> Latin)"""
if not name:
return ''
s = name.split(',')[0] if ',' in name else name
s = s.translate(TRANSLIT)
toks = [t.rstrip("'") for t in re.findall(r"[a-z']+", s.lower())
if t.rstrip("'") not in STOP and t not in ('v', 'von', 'der', 'de', 'la')]
return toks[-1] if toks else ''
# ---------------- matching ----------------
GENERIC = {'psychology', 'analytical', 'psychotherapy', 'psychological', 'psychoanalysis',
'jung', 'dream', 'dreams', 'unconscious', 'myth', 'myths', 'mythology',
'analysis', 'psychic', 'spirit', 'soul', 'self', 'ego', 'mind', 'man', 'human',
'transformation', 'transforming', 'goddess', 'gods', 'god', 'testament',
'modern', 'contemporary', 'understanding', 'introduction', 'guide', 'handbook',
'studies', 'essays', 'meaning', 'meanings', 'sacred', 'religion', 'cultural',
'initiation', 'book', 'books', 'volume', 'volumes', 'text', 'texts',
'image', 'images', 'evolution', 'greeks', 'greek', 'romans', 'roman',
'journey', 'hero', 'heroes', 'tale', 'tales', 'story', 'stories',
'fairy', 'folk', 'folklore', 'world', 'works', 'notes', 'lecture', 'lectures'}
PUBLISHERS = {'suny', 'routledge', 'penguin', 'spring', 'shambhala', 'karnac', 'chiron', 'sigo',
'pantheon', 'wiley', 'basic', 'press', 'ast', 'eksmo', 'bollingen', 'princeton',
'cornell', 'thames', 'hudson', 'bodley', 'oxford', 'cambridge', 'harvard',
'inner', 'city', 'harper', 'row', 'viking', 'harcourt', 'bradford', 'acorn',
'lindisfarne', 'free', 'association', 'daimon', 'philemon', 'abe', 'brill',
'harcourt', 'brunner', 'taylor', 'frank', 'casemate', 'essex', 'palgrave'}
def years(s):
return set(re.findall(r'\b(?:19|20)\d{2}\b', s or ''))
def score_pair(e_tok, e_cw, e_author, c_tok, c_cw, c_author):
"""shared identity score for two (title, cw, author) pairs; 0 = reject"""
inter = e_tok & c_tok
if e_cw and c_cw and e_cw == c_cw:
if not inter:
return 0.9 # same CW volume, Jung bypasses nothing else
elif len(inter) < 2:
return 0.0
ak = author_key(e_author)
cak = author_key(c_author)
same_author = ('jung' in (c_author or '').lower() and (e_author or '').startswith('Jung')) or \
(cak and ak and (cak == ak or cak in ak or ak in cak))
distinct = inter - GENERIC
if not same_author:
if not (e_cw and c_cw):
# lenient: author unreadable (e.g. Cyrillic on card) + very distinctive shared title
if len(distinct) >= 3:
pass # fall through to scoring
else:
return 0.0
if e_cw and c_cw and e_cw != c_cw:
return 0.0
# contradictory years (both have years, none shared) = different editions/books
ye, yc = years(' '.join(e_tok)), years(' '.join(c_tok))
if ye and yc and not (ye & yc):
return 0.0
if e_cw and c_cw:
return 1.0 # CW volume match + >=2 shared tokens
if len(distinct) >= 2:
# min-set overlap ratio (+ eps by inter size to break subset ties: Vol.I vs Vol.II)
return len(inter) / max(1, min(len(e_tok), len(c_tok))) + len(inter) / 1000.0
if len(c_tok - GENERIC) == 0 and cak == ak and not (e_tok - GENERIC) and len(inter) >= 2:
return 0.6 # card title is all-generic (e.g. "The Psychology of C.G. Jung")
# short title, one distinctive token shared (e.g. "From Freud to Jung")
if len(distinct) == 1 and min(len(e_tok), len(c_tok)) <= 4:
return 0.5
# author in Cyrillic (can't compare reliably) + very distinctive shared title
if not same_author and len(distinct) >= 3:
return 0.7
return 0.0
def match_entry_to_card(e, cards):
best, best_score = None, 0
for c in cards:
sc = score_pair(norm(e['title']), e['cw'], e['author'], c['tok'], c['cw'], c['author'])
if sc > best_score:
best, best_score = c, sc
return best
# ---------------- downloads index ----------------
def download_index():
idx = {}
for d in glob.glob(os.path.join(BASE, 'downloads', '0[1-6]-*')):
sec = os.path.basename(d)[:2]
files = []
for f in os.listdir(d):
if os.path.isfile(os.path.join(d, f)) and not f.startswith('.'):
files.append((f, norm(f)))
idx[sec] = files
return idx
def manifest_stale(sec, title, dl_idx, cards):
"""(verdict, evidence) for a 'Not downloadable' line"""
tok = {t for t in norm(title) if not re.match(r'^(19|20)\d{2}$', t)}
if not tok:
return 'SKIP', ''
for s2, files in dl_idx.items():
if s2 == sec:
continue
for f, ftok in files:
ftok = {t for t in ftok
if not re.match(r'^(19|20)\d{2}$', t) and t not in PUBLISHERS}
inter = tok & ftok
if len(inter - GENERIC) >= 2: # at least two distinctive shared tokens
return 'STALE', f'sec{s2} file: {f}'
for c in cards:
if c['sec'] == sec or c['status'] != '✅':
continue
ctok = {t for t in c['tok'] if not re.match(r'^(19|20)\d{2}$', t)}
inter = tok & ctok
if len(inter - GENERIC) >= 2: # 1 distinctive token = too weak (Stein «treatment» ≠ Kohut)
return 'CHECK', f'sec{c["sec"]}#{c["n"]} card ✅ ({c["file"]})'
return 'KEEP', ''
# ---------------- main ----------------
def main():
write = '--write' in sys.argv
cards = card_entries()
dl_idx = download_index()
raw_secs = {}
for sec, name, path in [
('07', 'Complexes & Association Experiment', 'sections/07-complexes/RAW.md'),
('08', 'Developmental Psychology', 'sections/08-developmental/RAW.md'),
('09', 'Comparison of Psychodynamic Concepts', 'data/isap-raw/09-comparison-of-psychodynamic-concepts.txt'),
('10', 'Psychopathology & Psychiatry', 'data/isap-raw/10-psychopathology-psychiatry.txt'),
('11', 'The Individuation Process', 'data/isap-raw/11-individuation-process.txt'),
('12', 'Practical Case', 'data/isap-raw/12-practical-case.txt'),
]:
p = os.path.join(BASE, path)
if os.path.exists(p):
raw_secs[sec] = (name, parse_raw(p))
matched = {}
virt = [] # virtual cards from earlier raw entries: (tok, cw, author, canonical_card)
for sec in sorted(raw_secs):
for e in raw_secs[sec][1]:
e['sec'] = sec
c = match_entry_to_card(e, cards)
if not c:
best_v, best_vs = None, 0
for (vtok, vcw, vauthor, vcard) in virt:
vs = score_pair(norm(e['title']), e['cw'], e['author'], vtok, vcw, vauthor)
if vs > best_vs:
best_v, best_vs = vcard, vs
c = best_v
matched[(sec, e['n'])] = {'kind': 'card' if c else 'new', 'ref': c, 'entry': e}
if c:
virt.append((norm(e['title']), e['cw'], e['author'], c))
print("=" * 78)
print("RAW 07-12 entries vs existing cards")
print("=" * 78)
for sec in sorted(raw_secs):
name, entries = raw_secs[sec]
n_new = n_xref = 0
print(f"\n## Section {sec} — {name} ({len(entries)} entries)")
for e in entries:
m = matched[(sec, e['n'])]
t = e['title'][:60]
if m['kind'] == 'card':
c = m['ref']
n_xref += 1
print(f" [{e['n']:>2}] {t:60s} -> sec{c['sec']}#{c['n']} ({c['status']})")
else:
n_new += 1
print(f" [{e['n']:>2}] {t:60s} NEW ({e['author']})")
print(f" xref={n_xref} new={n_new}")
print("\n" + "=" * 78)
print("MANIFEST 'Not downloadable' — stale lines (book/files exist elsewhere)")
print("=" * 78)
for mf in sorted(glob.glob(os.path.join(BASE, 'downloads', '0[1-6]-*', 'MANIFEST.md'))):
sec = os.path.basename(os.path.dirname(mf))[:2]
txt = open(mf, encoding='utf8').read()
m = re.search(r'## Not downloadable.*?(?=\n## |\Z)', txt, re.S)
if not m:
continue
stale = []
for line in m.group(0).splitlines():
if not line.startswith('|') or '---' in line:
continue
cells = [c.strip() for c in line.strip('|').split('|')]
if len(cells) < 2:
continue
first = cells[0]
tm = re.match(r'^(\d{2})\s+(\S.*)$', first)
title = tm.group(2) if tm else (cells[1] if len(cells) > 1 else '')
title = re.sub(r'\((EN|RU)\)\s*', '', title)
title = re.sub(r'^(EN|RU):\s*', '', title)
title = re.sub(r'\([^)]*\)', '', title) # author/edition parenthetical
title = title.strip()
verdict, ev = manifest_stale(sec, title, dl_idx, cards)
if verdict in ('STALE', 'CHECK'):
stale.append((line.strip(), verdict, ev))
if stale:
print(f"\n### {mf.split(BASE)[1]}")
for line, v, ev in stale:
print(f" [{v:5s}] {line[:95]}\n -> {ev}")
if write:
out = os.path.join(BASE, 'data', 'MASTER-LIST.md')
L = ['# Master book list — ISAP Zurich reading list (sections 01–12)',
'',
f'Generated: {datetime.date.today().isoformat()} by `tools/xref.py` — single source of truth',
'for cross-section dedup.',
'**Rule:** a book listed in several sections is researched once — at its CANONICAL card',
'(first section it appears in). Later sections get cross-refs, no new search/download.',
'']
also = {}
for (sec, n), m in matched.items():
if m['kind'] == 'card':
c = m['ref']
also.setdefault((c['sec'], c['n']), []).append(f'sec{sec}#{n}')
# 01-06 cross-section duplicates (canonical = earliest section)
for c1, c2 in itertools.combinations(cards, 2):
if c1['sec'] == c2['sec']:
continue
if score_pair(c1['tok'], c1['cw'], c1['author'], c2['tok'], c2['cw'], c2['author']) >= 0.5:
first, second = (c1, c2) if c1['sec'] < c2['sec'] else (c2, c1)
also.setdefault((first['sec'], first['n']), []).append(f"sec{second['sec']}#{second['n']}")
for sec in ['01', '02', '03', '04', '05', '06']:
sec_cards = [c for c in cards if c['sec'] == sec]
L.append(f'## Section {sec} — {len(sec_cards)} items (canonical cards)')
L.append('')
L.append('| # | EN title | Status | Also listed in |')
L.append('|---|----------|--------|----------------|')
for c in sorted(sec_cards, key=lambda x: x['n']):
a = also.get((c['sec'], c['n']), [])
L.append(f'| {c["n"]} | {c["title"][:70]} | {c["status"]} | {", ".join(sorted(a)) or "—"} |')
L.append('')
for sec in sorted(raw_secs):
name, entries = raw_secs[sec]
L.append(f'## Section {sec} — {name} (RAW, provisional numbering)')
L.append('')
L.append('| # | EN title | Author | Canonical | New? |')
L.append('|---|----------|--------|-----------|------|')
for e in entries:
m = matched[(sec, e['n'])]
if m['kind'] == 'card':
c = m['ref']
L.append(f'| {e["n"]} | {e["title"][:60]} | {e["author"]} | sec{c["sec"]}#{c["n"]} ({c["status"]}) | — |')
else:
L.append(f'| {e["n"]} | {e["title"][:60]} | {e["author"]} | — | **NEW** |')
L.append('')
open(out, 'w', encoding='utf8').write('\n'.join(L))
print(f'\nWROTE {out}')
if __name__ == '__main__':
main()