cross-section dedup system (user tasks 2+3, 2026-09-25):

- tools/xref.py: master book list for ALL 12 sections (RAW 07-12 scraped: sections/07/RAW.md
  + data/isap-raw/), cross-ref matcher (CW gate + distinctive tokens + years + Cyrillic
  transliteration), stale MANIFEST 'Not downloadable' checker; data/MASTER-LIST.md generated
  (17 book-dupes in 01-06 found, e.g. sec01#51=sec06#14 Solomon)
- file-level dedup: 17 md5-identical files removed from later sections (kept in first),
  MANIFESTs/cards -> cross-ref rows, totals updated (sec01 102 / 02 37 / 03 125 / 04 38 /
  05 38 / 06 37), dedup rule in AGENTS.md + MANIFEST Notes
- sec05 MANIFEST phantom line '03 Aion' removed (Aion not in sec05 list; files in sec01)
- sec03 #44: 1968 Bodley file was #41's book (content 'English Fairy Tales') — removed,
  #44 keeps ABC-CLIO 2002
This commit is contained in:
Dmitry Kokorin 2026-09-25 23:25:05 +03:00
parent 4274488410
commit 7d3794560e
46 changed files with 1831 additions and 21889 deletions

422
tools/xref.py Normal file
View file

@ -0,0 +1,422 @@
#!/usr/bin/env python3
"""xref.py — master book list + cross-section dedup + manifest freshness check.
Builds the single source of truth for "which book belongs to which section"
across ALL 12 ISAP reading-list sections, so every book is searched exactly once:
1. Sections 01-06: canonical entries come from the CARDS (sections/0N/NN-*.md).
2. Sections 07-12: entries parsed from RAW lists (sections/0N/RAW.md or
data/isap-raw/NN-*.txt).
3. Every 07-12 entry is matched to an existing 01-06 card (or to an earlier
07-12 entry) -> "canonical" pointer; unmatched entries = NEW (to search).
4. Every "Not downloadable" line in downloads/0N/MANIFEST.md is checked: if the
same book's FILES live in another section's download dir -> STALE line
(should be a cross-ref), otherwise KEEP.
Outputs:
data/MASTER-LIST.md — the master list (all sections, canonical pointers)
console — stale manifest line report (actionable)
Run: python3 tools/xref.py (report only)
python3 tools/xref.py --write (rewrite data/MASTER-LIST.md)
"""
import re, os, sys, glob, unicodedata, datetime, itertools
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
STOP = {'the','a','an','of','in','and','to','on','for','from','by','with','its','his','her',
'cw','jce','xje','pa','je','pf','xw','vol','pp','part','chap','chapter','esp','en','ru',
'st','ltd','int','world','studies'}
def norm(s):
s = unicodedata.normalize('NFKD', s or '')
s = s.lower()
s = re.sub(r'(\d)-(\d)', r'\1 \2', s) # 1936-1940 -> "1936 1940"
s = re.sub(r'[^a-z0-9 ]', ' ', s)
return set(w for w in s.split() if len(w) > 3 and w not in STOP)
def overlap(a, b):
if not a or not b:
return 0.0, 0
inter = a & b
return len(inter) / min(len(a), len(b)), len(inter)
# ---------------- RAW parsing (07-12) ----------------
CW_LINE = re.compile(r'^\s{0,8}CW\s*(\d+/\w{1,2}|\d+|-?S\d?)\s+(\S.*)$')
# author-start: one or more "Surname, I." groups, then >=2 spaces, then title
AUTHOR_LINE = re.compile(
r'^(\s{0,8})'
r'([A-Z][A-Za-z\u00C0-\u017F\-]+)' # 2: surname
r'(?:\s*,\s*[A-Za-z.\-]{1,9})?' # initials blob ", M.-L.v."
r'(?:\s*,\s*[A-Z][A-Za-z\u00C0-\u017F\-]+\s*,\s*[A-Za-z.\-]{1,9})?' # ", Surname, I."
r'\s*,?'
r'\s{2,}(\S.*)$')
CONT_AUTHOR = re.compile(r'^\s{8,16}([A-Z]\.)\s*$') # lone initial (surname on prev line)
CITY_START = re.compile(
r'^(New\s+York|London|Boston|Toronto|York|Woodstock|Hove|Ithaca|Cambridge|Tokyo|Geneva|Wilmette|'
r'Hillsdale|Northvale|Cham|College\s+Station|Arles|Arlington|Z\u00fcrich|Chicago|Asheville|'
r'Oxford|Basel|Paris|Berlin|Princeton|Syracuse|New\s+Haven|New\s+Brunswick|Heidelberg|Amsterdam)\b|'
r'^(Publishing|Switzerland|Press,|Books,|University)')
SECTION_HDR = re.compile(r'^\s{0,4}[AB]\s*\.?\s*\d*\.?\s+[A-Z]|\d{2}\s+[A-Z][A-Z ]{4,}$')
def parse_raw(path):
"""-> list of entries: {sec, n, author, title, cw, raw}"""
lines = open(path, encoding='utf8').read().splitlines()
entries, cur = [], None
def flush():
nonlocal cur
if cur and (cur['title'] or cur['cw']):
entries.append(cur)
cur = None
for ln in lines:
stripped = ln.strip()
if not stripped:
flush()
continue
indent = len(ln) - len(ln.lstrip())
if SECTION_HDR.match(stripped) and indent <= 4:
flush()
continue
if stripped.startswith('* required'):
flush()
continue
m = CW_LINE.match(ln)
if m:
flush()
vol = m.group(1).replace('/', '').upper()
vol = vol.replace('II', '2').replace('I', '1')
cur = {'cw': vol, 'author': 'Jung', 'title': m.group(2), 'raw': stripped}
continue
m = AUTHOR_LINE.match(ln)
if m and not stripped.startswith(('"', '*', '—', '.', 'In:')):
if cur and CITY_START.match(m.group(3)):
# 2nd author + "City: Publisher, Year" line of the CURRENT entry
cur['author'] += ', ' + m.group(2).strip().rstrip(',')
cur['title'] += ' ' + m.group(3)
continue
flush()
author = m.group(2).strip().rstrip(',')
cur = {'cw': None, 'author': author, 'title': m.group(3), 'raw': stripped}
continue
m = CONT_AUTHOR.match(ln)
if m and cur and ',' not in cur['author'] and not re.search(r'\.$', cur['author']):
cur['author'] += ' ' + m.group(1)
continue
# continuation line — or a NEW book by the same author (bib of current entry is complete)
if cur and (indent >= 8 or cur['cw']):
if (cur['cw'] is None and indent >= 8 and len(stripped) >= 15
and not stripped.startswith(('"', '*', '—', '.', 'In:', 'Part', 'Appendix',
'Chap', 'Ch.', 'Vol', 'Note', 'Notes'))
and not re.match(r'^[A-Z]{1,4}[\-:,.]?\w{0,6}$', stripped)
and re.search(r'[a-z]{3,}', stripped)
and re.search(r'\b(?:New\s+York|London|Boston|Toronto|Hove|Ithaca|Woodstock|'
r'Cambridge|Tokyo|Geneva|Wilmette|Hillsdale|Northvale|Cham|Arles|'
r'Arlington|Chicago|Asheville|Oxford|Basel|Paris|Berlin|Princeton|'
r'Syracuse|New\s+Haven|Heidelberg|Amsterdam|Sigtuna)\s*:\s*'
r'[A-Za-z][^,]{0,50}?\,\s*\d{4}', cur['title'])):
prev_author = cur['author']
flush()
cur = {'cw': None, 'author': prev_author, 'title': stripped, 'raw': stripped}
continue
cur['title'] += ' ' + stripped
flush()
out = []
for i, e in enumerate(entries, 1):
e['n'] = i
t = re.sub(r'\b(JCE|XJE|PA|JE|PF|XW)[\-:]?\w{1,6}\b\.?', ' ', e['title'])
t = re.sub(r'\bopen access\b\.?', ' ', t, flags=re.I)
# strip "City: Publisher, Year" publisher tail
t = re.sub(r'\b(?:New\s+York|London|Boston|Toronto|New\s+Brunswick|Woodstock|Hove|Ithaca|'
r'Cambridge|Tokyo|Geneva|Wilmette|Hillsdale|Northvale|Cham|College\s+Station|Arles|'
r'Arlington|Chicago|Asheville|Oxford|Basel|Paris|Berlin|Princeton|Syracuse|New\s+Haven|'
r'Heidelberg|Amsterdam|Sigtuna|Sigtuna)\s*:\s*[^.,]{0,60}?\,\s*\d{4}', ' ', t)
t = re.sub(r'\s+', ' ', t).strip(' .')
e['title'] = t
out.append(e)
return out
# ---------------- CARDS (01-06) ----------------
def card_entries():
out = []
for path in sorted(glob.glob(os.path.join(BASE, 'sections', '0[1-6]-*', '??-*.md'))):
txt = open(path, encoding='utf8').read()
h1 = re.search(r'^# (.+)$', txt, re.M)
am = re.search(r'\*\*Author\(s?\):\*\*\s*(.+)', txt, re.I)
st = re.search(r'\*\*Status:\*\*\s*([✅🔶❌⬜🔎])', txt)
if not h1:
continue
parts = os.path.dirname(path).split('/')
sec = parts[-1]
item = os.path.basename(path)[:2]
title = h1.group(1).split('(')[0]
cw = None
m = re.search(r'CW\s*(\d+)\s*/?\s*([IVX]+)?', title, re.I)
if m:
rom = {'I': 1, 'II': 2, 'III': 3, 'IV': 4, 'V': 5}
cw = m.group(1) + (str(rom.get(m.group(2).upper())) if m.group(2) else '')
if not cw:
fm = re.search(r'\bcw(9)[-_]([12])|\bcw(\d{1,2})', os.path.basename(path))
if fm:
cw = ((fm.group(1) or '') + (fm.group(2) or '')) or fm.group(3)
out.append({
'sec': sec[:2], 'n': item, 'author': (am.group(1) if am else '').strip(),
'title': title, 'tok': norm(title), 'cw': cw, 'status': st.group(1) if st else '?',
'file': os.path.join('sections', os.path.dirname(path).split('/')[-1], os.path.basename(path)),
})
return out
TRANSLIT = str.maketrans({
'а':'a','б':'b','в':'v','г':'g','д':'d','е':'e','ё':'e','ж':'zh','з':'z','и':'i','й':'y',
'к':'k','л':'l','м':'m','н':'n','о':'o','п':'p','р':'r','с':'s','т':'t','у':'u','ф':'f',
'х':'kh','ц':'ts','ч':'ch','ш':'sh','щ':'shch','ы':'y','ь':'','э':'e','ю':'yu','я':'ya'})
def author_key(name):
"""first surname token, normalized (Cyrillic -> Latin)"""
if not name:
return ''
s = name.split(',')[0] if ',' in name else name
s = s.translate(TRANSLIT)
toks = [t.rstrip("'") for t in re.findall(r"[a-z']+", s.lower())
if t.rstrip("'") not in STOP and t not in ('v', 'von', 'der', 'de', 'la')]
return toks[-1] if toks else ''
# ---------------- matching ----------------
GENERIC = {'psychology', 'analytical', 'psychotherapy', 'psychological', 'psychoanalysis',
'jung', 'dream', 'dreams', 'unconscious', 'myth', 'myths', 'mythology',
'analysis', 'psychic', 'spirit', 'soul', 'self', 'ego', 'mind', 'man', 'human',
'transformation', 'transforming', 'goddess', 'gods', 'god', 'testament',
'modern', 'contemporary', 'understanding', 'introduction', 'guide', 'handbook',
'studies', 'essays', 'meaning', 'meanings', 'sacred', 'religion', 'cultural',
'initiation', 'book', 'books', 'volume', 'volumes', 'text', 'texts',
'image', 'images', 'evolution', 'greeks', 'greek', 'romans', 'roman',
'journey', 'hero', 'heroes', 'tale', 'tales', 'story', 'stories',
'fairy', 'folk', 'folklore', 'world', 'works', 'notes', 'lecture', 'lectures'}
PUBLISHERS = {'suny', 'routledge', 'penguin', 'spring', 'shambhala', 'karnac', 'chiron', 'sigo',
'pantheon', 'wiley', 'basic', 'press', 'ast', 'eksmo', 'bollingen', 'princeton',
'cornell', 'thames', 'hudson', 'bodley', 'oxford', 'cambridge', 'harvard',
'inner', 'city', 'harper', 'row', 'viking', 'harcourt', 'bradford', 'acorn',
'lindisfarne', 'free', 'association', 'daimon', 'philemon', 'abe', 'brill',
'harcourt', 'brunner', 'taylor', 'frank', 'casemate', 'essex', 'palgrave'}
def years(s):
return set(re.findall(r'\b(?:19|20)\d{2}\b', s or ''))
def score_pair(e_tok, e_cw, e_author, c_tok, c_cw, c_author):
"""shared identity score for two (title, cw, author) pairs; 0 = reject"""
inter = e_tok & c_tok
if e_cw and c_cw and e_cw == c_cw:
if not inter:
return 0.9 # same CW volume, Jung bypasses nothing else
elif len(inter) < 2:
return 0.0
ak = author_key(e_author)
cak = author_key(c_author)
same_author = ('jung' in (c_author or '').lower() and (e_author or '').startswith('Jung')) or \
(cak and ak and (cak == ak or cak in ak or ak in cak))
distinct = inter - GENERIC
if not same_author:
if not (e_cw and c_cw):
# lenient: author unreadable (e.g. Cyrillic on card) + very distinctive shared title
if len(distinct) >= 3:
pass # fall through to scoring
else:
return 0.0
if e_cw and c_cw and e_cw != c_cw:
return 0.0
# contradictory years (both have years, none shared) = different editions/books
ye, yc = years(' '.join(e_tok)), years(' '.join(c_tok))
if ye and yc and not (ye & yc):
return 0.0
if e_cw and c_cw:
return 1.0 # CW volume match + >=2 shared tokens
if len(distinct) >= 2:
return len(distinct) / max(1, len(inter))
if len(c_tok - GENERIC) == 0 and cak == ak and not (e_tok - GENERIC) and len(inter) >= 2:
return 0.6 # card title is all-generic (e.g. "The Psychology of C.G. Jung")
# short title, one distinctive token shared (e.g. "From Freud to Jung")
if len(distinct) == 1 and min(len(e_tok), len(c_tok)) <= 4:
return 0.5
# author in Cyrillic (can't compare reliably) + very distinctive shared title
if not same_author and len(distinct) >= 3:
return 0.7
return 0.0
def match_entry_to_card(e, cards):
best, best_score = None, 0
for c in cards:
sc = score_pair(norm(e['title']), e['cw'], e['author'], c['tok'], c['cw'], c['author'])
if sc > best_score:
best, best_score = c, sc
return best
# ---------------- downloads index ----------------
def download_index():
idx = {}
for d in glob.glob(os.path.join(BASE, 'downloads', '0[1-6]-*')):
sec = os.path.basename(d)[:2]
files = []
for f in os.listdir(d):
if os.path.isfile(os.path.join(d, f)) and not f.startswith('.'):
files.append((f, norm(f)))
idx[sec] = files
return idx
def manifest_stale(sec, title, dl_idx, cards):
"""(verdict, evidence) for a 'Not downloadable' line"""
tok = {t for t in norm(title) if not re.match(r'^(19|20)\d{2}$', t)}
if not tok:
return 'SKIP', ''
for s2, files in dl_idx.items():
if s2 == sec:
continue
for f, ftok in files:
ftok = {t for t in ftok
if not re.match(r'^(19|20)\d{2}$', t) and t not in PUBLISHERS}
inter = tok & ftok
if inter - GENERIC: # at least one distinctive shared token
return 'STALE', f'sec{s2} file: {f}'
for c in cards:
if c['sec'] == sec or c['status'] != '✅':
continue
ctok = {t for t in c['tok'] if not re.match(r'^(19|20)\d{2}$', t)}
inter = tok & ctok
if inter - GENERIC:
return 'CHECK', f'sec{c["sec"]}#{c["n"]} card ✅ ({c["file"]})'
return 'KEEP', ''
# ---------------- main ----------------
def main():
write = '--write' in sys.argv
cards = card_entries()
dl_idx = download_index()
raw_secs = {}
for sec, name, path in [
('07', 'Complexes & Association Experiment', 'sections/07-complexes/RAW.md'),
('08', 'Developmental Psychology', 'data/isap-raw/08-developmental-psychology.txt'),
('09', 'Comparison of Psychodynamic Concepts', 'data/isap-raw/09-comparison-of-psychodynamic-concepts.txt'),
('10', 'Psychopathology & Psychiatry', 'data/isap-raw/10-psychopathology-psychiatry.txt'),
('11', 'The Individuation Process', 'data/isap-raw/11-individuation-process.txt'),
('12', 'Practical Case', 'data/isap-raw/12-practical-case.txt'),
]:
p = os.path.join(BASE, path)
if os.path.exists(p):
raw_secs[sec] = (name, parse_raw(p))
matched = {}
virt = [] # virtual cards from earlier raw entries: (tok, cw, author, canonical_card)
for sec in sorted(raw_secs):
for e in raw_secs[sec][1]:
e['sec'] = sec
c = match_entry_to_card(e, cards)
if not c:
best_v, best_vs = None, 0
for (vtok, vcw, vauthor, vcard) in virt:
vs = score_pair(norm(e['title']), e['cw'], e['author'], vtok, vcw, vauthor)
if vs > best_vs:
best_v, best_vs = vcard, vs
c = best_v
matched[(sec, e['n'])] = {'kind': 'card' if c else 'new', 'ref': c, 'entry': e}
if c:
virt.append((norm(e['title']), e['cw'], e['author'], c))
print("=" * 78)
print("RAW 07-12 entries vs existing cards")
print("=" * 78)
for sec in sorted(raw_secs):
name, entries = raw_secs[sec]
n_new = n_xref = 0
print(f"\n## Section {sec} — {name} ({len(entries)} entries)")
for e in entries:
m = matched[(sec, e['n'])]
t = e['title'][:60]
if m['kind'] == 'card':
c = m['ref']
n_xref += 1
print(f" [{e['n']:>2}] {t:60s} -> sec{c['sec']}#{c['n']} ({c['status']})")
else:
n_new += 1
print(f" [{e['n']:>2}] {t:60s} NEW ({e['author']})")
print(f" xref={n_xref} new={n_new}")
print("\n" + "=" * 78)
print("MANIFEST 'Not downloadable' — stale lines (book/files exist elsewhere)")
print("=" * 78)
for mf in sorted(glob.glob(os.path.join(BASE, 'downloads', '0[1-6]-*', 'MANIFEST.md'))):
sec = os.path.basename(os.path.dirname(mf))[:2]
txt = open(mf, encoding='utf8').read()
m = re.search(r'## Not downloadable.*?(?=\n## |\Z)', txt, re.S)
if not m:
continue
stale = []
for line in m.group(0).splitlines():
if not line.startswith('|') or '---' in line:
continue
cells = [c.strip() for c in line.strip('|').split('|')]
if len(cells) < 2:
continue
first = cells[0]
tm = re.match(r'^(\d{2})\s+(\S.*)$', first)
title = tm.group(2) if tm else (cells[1] if len(cells) > 1 else '')
title = re.sub(r'\((EN|RU)\)\s*', '', title)
title = re.sub(r'^(EN|RU):\s*', '', title)
title = re.sub(r'\([^)]*\)', '', title) # author/edition parenthetical
title = title.strip()
verdict, ev = manifest_stale(sec, title, dl_idx, cards)
if verdict in ('STALE', 'CHECK'):
stale.append((line.strip(), verdict, ev))
if stale:
print(f"\n### {mf.split(BASE)[1]}")
for line, v, ev in stale:
print(f" [{v:5s}] {line[:95]}\n -> {ev}")
if write:
out = os.path.join(BASE, 'data', 'MASTER-LIST.md')
L = ['# Master book list — ISAP Zurich reading list (sections 01–12)',
'',
f'Generated: {datetime.date.today().isoformat()} by `tools/xref.py` — single source of truth',
'for cross-section dedup.',
'**Rule:** a book listed in several sections is researched once — at its CANONICAL card',
'(first section it appears in). Later sections get cross-refs, no new search/download.',
'']
also = {}
for (sec, n), m in matched.items():
if m['kind'] == 'card':
c = m['ref']
also.setdefault((c['sec'], c['n']), []).append(f'sec{sec}#{n}')
# 01-06 cross-section duplicates (canonical = earliest section)
for c1, c2 in itertools.combinations(cards, 2):
if c1['sec'] == c2['sec']:
continue
if score_pair(c1['tok'], c1['cw'], c1['author'], c2['tok'], c2['cw'], c2['author']) >= 0.5:
first, second = (c1, c2) if c1['sec'] < c2['sec'] else (c2, c1)
also.setdefault((first['sec'], first['n']), []).append(f"sec{second['sec']}#{second['n']}")
for sec in ['01', '02', '03', '04', '05', '06']:
sec_cards = [c for c in cards if c['sec'] == sec]
L.append(f'## Section {sec} — {len(sec_cards)} items (canonical cards)')
L.append('')
L.append('| # | EN title | Status | Also listed in |')
L.append('|---|----------|--------|----------------|')
for c in sorted(sec_cards, key=lambda x: x['n']):
a = also.get((c['sec'], c['n']), [])
L.append(f'| {c["n"]} | {c["title"][:70]} | {c["status"]} | {", ".join(sorted(a)) or "—"} |')
L.append('')
for sec in sorted(raw_secs):
name, entries = raw_secs[sec]
L.append(f'## Section {sec} — {name} (RAW, provisional numbering)')
L.append('')
L.append('| # | EN title | Author | Canonical | New? |')
L.append('|---|----------|--------|-----------|------|')
for e in entries:
m = matched[(sec, e['n'])]
if m['kind'] == 'card':
c = m['ref']
L.append(f'| {e["n"]} | {e["title"][:60]} | {e["author"]} | sec{c["sec"]}#{c["n"]} ({c["status"]}) | — |')
else:
L.append(f'| {e["n"]} | {e["title"][:60]} | {e["author"]} | — | **NEW** |')
L.append('')
open(out, 'w', encoding='utf8').write('\n'.join(L))
print(f'\nWROTE {out}')
if __name__ == '__main__':
main()