- All 41 files content-verified (pdfinfo + first-page text; epubs xhtml; fb2 titles) - 08 Fonagy = EN original (renamed -en-); Wirtz #38 = BOOK (Spring Journal Books 2014, 358pp) - Bonus: Kawai Myōe (Castalia 2018 part 1, flib x2) + TFP Clinical Guide (Yeomans/Clarikin/Kernberg 2018, DIFFERENT TFP book) - xref.py stale-check: 1-token distinctive match too weak (Stein/Kohut 'treatment' false positive) -> >=2 required - .gitattributes: +sec05/07/08/09 LFS dir rules - Linter 0, xref stale 0
443 lines
22 KiB
Python
443 lines
22 KiB
Python
#!/usr/bin/env python3
|
||
"""xref.py — master book list + cross-section dedup + manifest freshness check.
|
||
|
||
Builds the single source of truth for "which book belongs to which section"
|
||
across ALL 12 ISAP reading-list sections, so every book is searched exactly once:
|
||
|
||
1. Sections 01-06: canonical entries come from the CARDS (sections/0N/NN-*.md).
|
||
2. Sections 07-12: entries parsed from RAW lists (sections/0N/RAW.md or
|
||
data/isap-raw/NN-*.txt).
|
||
3. Every 07-12 entry is matched to an existing 01-06 card (or to an earlier
|
||
07-12 entry) -> "canonical" pointer; unmatched entries = NEW (to search).
|
||
4. Every "Not downloadable" line in downloads/0N/MANIFEST.md is checked: if the
|
||
same book's FILES live in another section's download dir -> STALE line
|
||
(should be a cross-ref), otherwise KEEP.
|
||
|
||
Outputs:
|
||
data/MASTER-LIST.md — the master list (all sections, canonical pointers)
|
||
console — stale manifest line report (actionable)
|
||
|
||
Run: python3 tools/xref.py (report only)
|
||
python3 tools/xref.py --write (rewrite data/MASTER-LIST.md)
|
||
"""
|
||
import re, os, sys, glob, unicodedata, datetime, itertools
|
||
|
||
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||
|
||
STOP = {'the','a','an','of','in','and','to','on','for','from','by','with','its','his','her',
|
||
'cw','jce','xje','pa','je','pf','xw','vol','pp','part','chap','chapter','esp','en','ru',
|
||
'st','ltd','int','world','studies'}
|
||
|
||
def norm(s):
|
||
s = unicodedata.normalize('NFKD', s or '')
|
||
s = s.lower()
|
||
s = re.sub(r'(\d)-(\d)', r'\1 \2', s) # 1936-1940 -> "1936 1940"
|
||
s = re.sub(r'[^a-z0-9 ]', ' ', s)
|
||
return set(w for w in s.split() if len(w) > 3 and w not in STOP)
|
||
|
||
def overlap(a, b):
|
||
if not a or not b:
|
||
return 0.0, 0
|
||
inter = a & b
|
||
return len(inter) / min(len(a), len(b)), len(inter)
|
||
|
||
# ---------------- RAW parsing (07-12) ----------------
|
||
CW_LINE = re.compile(r'^\s{0,8}CW\s*(\d+/\w{1,2}|\d+|-?S\d?)\s+(\S.*)$')
|
||
# author-start: one or more "Surname, I." groups, then >=2 spaces, then title
|
||
AUTHOR_LINE = re.compile(
|
||
r'^(\s{0,8})'
|
||
r'([A-Z][A-Za-z\u00C0-\u017F\-]+)' # 2: surname
|
||
r'(?:\s*,\s*[A-Za-z.\-](?:\s?[A-Za-z.\-]){0,8}|\s+[A-Z]\.)?' # initials ", C. A." or space-initial "Kalsched D."
|
||
r'(?:\s*,\s*[A-Z][A-Za-z\u00C0-\u017F\-]+\s*,\s*[A-Za-z.\-](?:\s?[A-Za-z.\-]){0,8}){0,3}' # ", Surname, I." x3
|
||
r'\s*,?'
|
||
r'\s{2,}(\S.*)$')
|
||
CONT_AUTHOR = re.compile(r'^\s{8,16}([A-Z]\.)\s*$') # lone initial (surname on prev line)
|
||
CITY_START = re.compile(
|
||
r'^(New\s+York|London|Boston|Toronto|York|Woodstock|Hove|Ithaca|Cambridge|Tokyo|Geneva|Wilmette|'
|
||
r'Hillsdale|Northvale|Cham|College\s+Station|Arles|Arlington|Z\u00fcrich|Chicago|Asheville|'
|
||
r'Oxford|Basel|Paris|Berlin|Princeton|Syracuse|New\s+Haven|New\s+Brunswick|Heidelberg|Amsterdam)\b|'
|
||
r'^(Publishing|Switzerland|Press,|Books,|University)')
|
||
SECTION_HDR = re.compile(r'^\s{0,4}[AB]\s*\.?\s*\d*\.?\s+[A-Z]|\d{2}\s+[A-Z][A-Z ]{4,}$')
|
||
|
||
def parse_raw(path):
|
||
"""-> list of entries: {sec, n, author, title, cw, raw}"""
|
||
lines = open(path, encoding='utf8').read().splitlines()
|
||
entries, cur = [], None
|
||
def flush():
|
||
nonlocal cur
|
||
if cur and (cur['title'] or cur['cw']):
|
||
entries.append(cur)
|
||
cur = None
|
||
for ln in lines:
|
||
stripped = ln.strip()
|
||
if not stripped:
|
||
flush()
|
||
continue
|
||
indent = len(ln) - len(ln.lstrip())
|
||
if SECTION_HDR.match(stripped) and indent <= 4:
|
||
flush()
|
||
continue
|
||
if stripped.startswith('* required'):
|
||
flush()
|
||
continue
|
||
m = CW_LINE.match(ln)
|
||
if m:
|
||
flush()
|
||
vol = m.group(1).replace('/', '').upper()
|
||
vol = vol.replace('II', '2').replace('I', '1')
|
||
cur = {'cw': vol, 'author': 'Jung', 'title': m.group(2), 'raw': stripped}
|
||
continue
|
||
m = AUTHOR_LINE.match(ln)
|
||
if m and not stripped.startswith(('"', '*', '—', '.', 'In:', '“', '‘', '«')):
|
||
if cur and cur['author'].endswith('-'):
|
||
# hyphen-split surname across lines
|
||
sp = re.search(r'\s{2,}', ln.lstrip())
|
||
author_part = (ln.lstrip()[:sp.start()].rstrip() if sp else ln.lstrip().rstrip()).rstrip(',')
|
||
cur['author'] = cur['author'].rstrip('-') + '-' + re.sub(r'\s+', ' ', author_part).strip()
|
||
cur['title'] += ' ' + m.group(3).strip()
|
||
continue
|
||
if cur and CITY_START.match(m.group(3)):
|
||
# 2nd author + "City: Publisher, Year" line of the CURRENT entry
|
||
cur['author'] += ', ' + m.group(2).strip().rstrip(',')
|
||
cur['title'] += ' ' + m.group(3)
|
||
continue
|
||
if cur and re.match(r'^\d{4}\.?\s*$', m.group(3).strip()):
|
||
# continuation author line: "Pine, F., 1975." (bib tail, no title)
|
||
cur['author'] += ', ' + m.group(2).strip().rstrip(',')
|
||
cur['title'] += ' ' + m.group(3).strip()
|
||
continue
|
||
if cur and not re.search(r'\d{4}', cur['title']) and re.search(r'\b(in|of|the|and|a|to|for|on|between|with)$', cur['title']):
|
||
# multi-author entry split across lines (Hersh/Caligor/Yeomans case)
|
||
sp = re.search(r'\s{2,}', ln.lstrip())
|
||
author_part = (ln.lstrip()[:sp.start()].rstrip() if sp else ln.lstrip().rstrip()).rstrip(',')
|
||
cur['author'] += ', ' + re.sub(r'\s+', ' ', author_part).strip()
|
||
cur['title'] += ' ' + m.group(3).strip()
|
||
continue
|
||
flush()
|
||
author = m.group(2).strip().rstrip(',')
|
||
cur = {'cw': None, 'author': author, 'title': m.group(3), 'raw': stripped}
|
||
continue
|
||
m = CONT_AUTHOR.match(ln)
|
||
if m and cur and ',' not in cur['author'] and not re.search(r'\.$', cur['author']):
|
||
cur['author'] += ' ' + m.group(1)
|
||
continue
|
||
# continuation line — or a NEW book by the same author (bib of current entry is complete)
|
||
if cur and (indent >= 8 or cur['cw']):
|
||
if (cur['cw'] is None and indent >= 8 and len(stripped) >= 15
|
||
and not stripped.startswith(('"', '*', '—', '.', 'In:', 'Part', 'Appendix',
|
||
'Chap', 'Ch.', 'Vol', 'Note', 'Notes', '“', '‘', '«'))
|
||
and not re.match(r'^[A-Z]{1,4}[\-:,.]?\w{0,6}$', stripped)
|
||
and re.search(r'[a-z]{3,}', stripped)
|
||
and (re.search(r'\b(?:New\s+York|London|Boston|Toronto|Hove|Ithaca|Woodstock|'
|
||
r'Cambridge|Tokyo|Geneva|Wilmette|Hillsdale|Northvale|Cham|Arles|'
|
||
r'Arlington|Chicago|Asheville|Oxford|Basel|Paris|Berlin|Princeton|'
|
||
r'Syracuse|New\s+Haven|Heidelberg|Amsterdam|Sigtuna)\s*:\s*'
|
||
r'[A-Za-z][^,]{0,50}?\,\s*\d{4}', cur['title'])
|
||
or re.search(r'(?:19|20)\d{2}', cur['title']))):
|
||
prev_author = cur['author']
|
||
flush()
|
||
cur = {'cw': None, 'author': prev_author, 'title': stripped, 'raw': stripped}
|
||
continue
|
||
cur['title'] += ' ' + stripped
|
||
flush()
|
||
out = []
|
||
for i, e in enumerate(entries, 1):
|
||
e['n'] = i
|
||
t = re.sub(r'\b(JCE|XJE|PA|JE|PF|XW)[\-:]?\w{1,6}\b\.?', ' ', e['title'])
|
||
t = re.sub(r'\bopen access\b\.?', ' ', t, flags=re.I)
|
||
# strip "City: Publisher, Year" publisher tail
|
||
t = re.sub(r'\b(?:New\s+York|London|Boston|Toronto|New\s+Brunswick|Woodstock|Hove|Ithaca|'
|
||
r'Cambridge|Tokyo|Geneva|Wilmette|Hillsdale|Northvale|Cham|College\s+Station|Arles|'
|
||
r'Arlington|Chicago|Asheville|Oxford|Basel|Paris|Berlin|Princeton|Syracuse|New\s+Haven|'
|
||
r'Heidelberg|Amsterdam|Sigtuna|Sigtuna)\s*:\s*[^.,]{0,60}?\,\s*\d{4}', ' ', t)
|
||
t = re.sub(r'\s+', ' ', t).strip(' .')
|
||
e['title'] = t
|
||
out.append(e)
|
||
return out
|
||
|
||
# ---------------- CARDS (01-06) ----------------
|
||
def card_entries():
|
||
out = []
|
||
for path in sorted(glob.glob(os.path.join(BASE, 'sections', '0[1-9]-*', '??-*.md'))):
|
||
txt = open(path, encoding='utf8').read()
|
||
h1 = re.search(r'^# (.+)$', txt, re.M)
|
||
am = re.search(r'\*\*Author\(s?\):\*\*\s*(.+)', txt, re.I)
|
||
st = re.search(r'\*\*Status:\*\*\s*([✅🔶❌⬜🔎])', txt)
|
||
if not h1:
|
||
continue
|
||
parts = os.path.dirname(path).split('/')
|
||
sec = parts[-1]
|
||
item = os.path.basename(path)[:2]
|
||
title = h1.group(1).split('(')[0]
|
||
cw = None
|
||
m = re.search(r'CW\s*(\d+)\s*/?\s*([IVX]+)?', title, re.I)
|
||
if m:
|
||
rom = {'I': 1, 'II': 2, 'III': 3, 'IV': 4, 'V': 5}
|
||
cw = m.group(1) + (str(rom.get(m.group(2).upper())) if m.group(2) else '')
|
||
if not cw:
|
||
fm = re.search(r'\bcw(9)[-_]([12])|\bcw(\d{1,2})', os.path.basename(path))
|
||
if fm:
|
||
cw = ((fm.group(1) or '') + (fm.group(2) or '')) or fm.group(3)
|
||
out.append({
|
||
'sec': sec[:2], 'n': item, 'author': (am.group(1) if am else '').strip(),
|
||
'title': title, 'tok': norm(title), 'cw': cw, 'status': st.group(1) if st else '?',
|
||
'file': os.path.join('sections', os.path.dirname(path).split('/')[-1], os.path.basename(path)),
|
||
})
|
||
return out
|
||
|
||
TRANSLIT = str.maketrans({
|
||
'а':'a','б':'b','в':'v','г':'g','д':'d','е':'e','ё':'e','ж':'zh','з':'z','и':'i','й':'y',
|
||
'к':'k','л':'l','м':'m','н':'n','о':'o','п':'p','р':'r','с':'s','т':'t','у':'u','ф':'f',
|
||
'х':'kh','ц':'ts','ч':'ch','ш':'sh','щ':'shch','ы':'y','ь':'','э':'e','ю':'yu','я':'ya'})
|
||
|
||
def author_key(name):
|
||
"""first surname token, normalized (Cyrillic -> Latin)"""
|
||
if not name:
|
||
return ''
|
||
s = name.split(',')[0] if ',' in name else name
|
||
s = s.translate(TRANSLIT)
|
||
toks = [t.rstrip("'") for t in re.findall(r"[a-z']+", s.lower())
|
||
if t.rstrip("'") not in STOP and t not in ('v', 'von', 'der', 'de', 'la')]
|
||
return toks[-1] if toks else ''
|
||
|
||
# ---------------- matching ----------------
|
||
GENERIC = {'psychology', 'analytical', 'psychotherapy', 'psychological', 'psychoanalysis',
|
||
'jung', 'dream', 'dreams', 'unconscious', 'myth', 'myths', 'mythology',
|
||
'analysis', 'psychic', 'spirit', 'soul', 'self', 'ego', 'mind', 'man', 'human',
|
||
'transformation', 'transforming', 'goddess', 'gods', 'god', 'testament',
|
||
'modern', 'contemporary', 'understanding', 'introduction', 'guide', 'handbook',
|
||
'studies', 'essays', 'meaning', 'meanings', 'sacred', 'religion', 'cultural',
|
||
'initiation', 'book', 'books', 'volume', 'volumes', 'text', 'texts',
|
||
'image', 'images', 'evolution', 'greeks', 'greek', 'romans', 'roman',
|
||
'journey', 'hero', 'heroes', 'tale', 'tales', 'story', 'stories',
|
||
'fairy', 'folk', 'folklore', 'world', 'works', 'notes', 'lecture', 'lectures'}
|
||
PUBLISHERS = {'suny', 'routledge', 'penguin', 'spring', 'shambhala', 'karnac', 'chiron', 'sigo',
|
||
'pantheon', 'wiley', 'basic', 'press', 'ast', 'eksmo', 'bollingen', 'princeton',
|
||
'cornell', 'thames', 'hudson', 'bodley', 'oxford', 'cambridge', 'harvard',
|
||
'inner', 'city', 'harper', 'row', 'viking', 'harcourt', 'bradford', 'acorn',
|
||
'lindisfarne', 'free', 'association', 'daimon', 'philemon', 'abe', 'brill',
|
||
'harcourt', 'brunner', 'taylor', 'frank', 'casemate', 'essex', 'palgrave'}
|
||
|
||
def years(s):
|
||
return set(re.findall(r'\b(?:19|20)\d{2}\b', s or ''))
|
||
|
||
def score_pair(e_tok, e_cw, e_author, c_tok, c_cw, c_author):
|
||
"""shared identity score for two (title, cw, author) pairs; 0 = reject"""
|
||
inter = e_tok & c_tok
|
||
if e_cw and c_cw and e_cw == c_cw:
|
||
if not inter:
|
||
return 0.9 # same CW volume, Jung bypasses nothing else
|
||
elif len(inter) < 2:
|
||
return 0.0
|
||
ak = author_key(e_author)
|
||
cak = author_key(c_author)
|
||
same_author = ('jung' in (c_author or '').lower() and (e_author or '').startswith('Jung')) or \
|
||
(cak and ak and (cak == ak or cak in ak or ak in cak))
|
||
distinct = inter - GENERIC
|
||
if not same_author:
|
||
if not (e_cw and c_cw):
|
||
# lenient: author unreadable (e.g. Cyrillic on card) + very distinctive shared title
|
||
if len(distinct) >= 3:
|
||
pass # fall through to scoring
|
||
else:
|
||
return 0.0
|
||
if e_cw and c_cw and e_cw != c_cw:
|
||
return 0.0
|
||
# contradictory years (both have years, none shared) = different editions/books
|
||
ye, yc = years(' '.join(e_tok)), years(' '.join(c_tok))
|
||
if ye and yc and not (ye & yc):
|
||
return 0.0
|
||
if e_cw and c_cw:
|
||
return 1.0 # CW volume match + >=2 shared tokens
|
||
if len(distinct) >= 2:
|
||
# min-set overlap ratio (+ eps by inter size to break subset ties: Vol.I vs Vol.II)
|
||
return len(inter) / max(1, min(len(e_tok), len(c_tok))) + len(inter) / 1000.0
|
||
if len(c_tok - GENERIC) == 0 and cak == ak and not (e_tok - GENERIC) and len(inter) >= 2:
|
||
return 0.6 # card title is all-generic (e.g. "The Psychology of C.G. Jung")
|
||
# short title, one distinctive token shared (e.g. "From Freud to Jung")
|
||
if len(distinct) == 1 and min(len(e_tok), len(c_tok)) <= 4:
|
||
return 0.5
|
||
# author in Cyrillic (can't compare reliably) + very distinctive shared title
|
||
if not same_author and len(distinct) >= 3:
|
||
return 0.7
|
||
return 0.0
|
||
|
||
def match_entry_to_card(e, cards):
|
||
best, best_score = None, 0
|
||
for c in cards:
|
||
sc = score_pair(norm(e['title']), e['cw'], e['author'], c['tok'], c['cw'], c['author'])
|
||
if sc > best_score:
|
||
best, best_score = c, sc
|
||
return best
|
||
|
||
# ---------------- downloads index ----------------
|
||
def download_index():
|
||
idx = {}
|
||
for d in glob.glob(os.path.join(BASE, 'downloads', '0[1-6]-*')):
|
||
sec = os.path.basename(d)[:2]
|
||
files = []
|
||
for f in os.listdir(d):
|
||
if os.path.isfile(os.path.join(d, f)) and not f.startswith('.'):
|
||
files.append((f, norm(f)))
|
||
idx[sec] = files
|
||
return idx
|
||
|
||
def manifest_stale(sec, title, dl_idx, cards):
|
||
"""(verdict, evidence) for a 'Not downloadable' line"""
|
||
tok = {t for t in norm(title) if not re.match(r'^(19|20)\d{2}$', t)}
|
||
if not tok:
|
||
return 'SKIP', ''
|
||
for s2, files in dl_idx.items():
|
||
if s2 == sec:
|
||
continue
|
||
for f, ftok in files:
|
||
ftok = {t for t in ftok
|
||
if not re.match(r'^(19|20)\d{2}$', t) and t not in PUBLISHERS}
|
||
inter = tok & ftok
|
||
if len(inter - GENERIC) >= 2: # at least two distinctive shared tokens
|
||
return 'STALE', f'sec{s2} file: {f}'
|
||
for c in cards:
|
||
if c['sec'] == sec or c['status'] != '✅':
|
||
continue
|
||
ctok = {t for t in c['tok'] if not re.match(r'^(19|20)\d{2}$', t)}
|
||
inter = tok & ctok
|
||
if len(inter - GENERIC) >= 2: # 1 distinctive token = too weak (Stein «treatment» ≠ Kohut)
|
||
return 'CHECK', f'sec{c["sec"]}#{c["n"]} card ✅ ({c["file"]})'
|
||
return 'KEEP', ''
|
||
|
||
# ---------------- main ----------------
|
||
def main():
|
||
write = '--write' in sys.argv
|
||
cards = card_entries()
|
||
dl_idx = download_index()
|
||
|
||
raw_secs = {}
|
||
for sec, name, path in [
|
||
('07', 'Complexes & Association Experiment', 'sections/07-complexes/RAW.md'),
|
||
('08', 'Developmental Psychology', 'sections/08-developmental/RAW.md'),
|
||
('09', 'Comparison of Psychodynamic Concepts', 'data/isap-raw/09-comparison-of-psychodynamic-concepts.txt'),
|
||
('10', 'Psychopathology & Psychiatry', 'data/isap-raw/10-psychopathology-psychiatry.txt'),
|
||
('11', 'The Individuation Process', 'data/isap-raw/11-individuation-process.txt'),
|
||
('12', 'Practical Case', 'data/isap-raw/12-practical-case.txt'),
|
||
]:
|
||
p = os.path.join(BASE, path)
|
||
if os.path.exists(p):
|
||
raw_secs[sec] = (name, parse_raw(p))
|
||
|
||
matched = {}
|
||
virt = [] # virtual cards from earlier raw entries: (tok, cw, author, canonical_card)
|
||
for sec in sorted(raw_secs):
|
||
for e in raw_secs[sec][1]:
|
||
e['sec'] = sec
|
||
c = match_entry_to_card(e, cards)
|
||
if not c:
|
||
best_v, best_vs = None, 0
|
||
for (vtok, vcw, vauthor, vcard) in virt:
|
||
vs = score_pair(norm(e['title']), e['cw'], e['author'], vtok, vcw, vauthor)
|
||
if vs > best_vs:
|
||
best_v, best_vs = vcard, vs
|
||
c = best_v
|
||
matched[(sec, e['n'])] = {'kind': 'card' if c else 'new', 'ref': c, 'entry': e}
|
||
if c:
|
||
virt.append((norm(e['title']), e['cw'], e['author'], c))
|
||
|
||
print("=" * 78)
|
||
print("RAW 07-12 entries vs existing cards")
|
||
print("=" * 78)
|
||
for sec in sorted(raw_secs):
|
||
name, entries = raw_secs[sec]
|
||
n_new = n_xref = 0
|
||
print(f"\n## Section {sec} — {name} ({len(entries)} entries)")
|
||
for e in entries:
|
||
m = matched[(sec, e['n'])]
|
||
t = e['title'][:60]
|
||
if m['kind'] == 'card':
|
||
c = m['ref']
|
||
n_xref += 1
|
||
print(f" [{e['n']:>2}] {t:60s} -> sec{c['sec']}#{c['n']} ({c['status']})")
|
||
else:
|
||
n_new += 1
|
||
print(f" [{e['n']:>2}] {t:60s} NEW ({e['author']})")
|
||
print(f" xref={n_xref} new={n_new}")
|
||
|
||
print("\n" + "=" * 78)
|
||
print("MANIFEST 'Not downloadable' — stale lines (book/files exist elsewhere)")
|
||
print("=" * 78)
|
||
for mf in sorted(glob.glob(os.path.join(BASE, 'downloads', '0[1-6]-*', 'MANIFEST.md'))):
|
||
sec = os.path.basename(os.path.dirname(mf))[:2]
|
||
txt = open(mf, encoding='utf8').read()
|
||
m = re.search(r'## Not downloadable.*?(?=\n## |\Z)', txt, re.S)
|
||
if not m:
|
||
continue
|
||
stale = []
|
||
for line in m.group(0).splitlines():
|
||
if not line.startswith('|') or '---' in line:
|
||
continue
|
||
cells = [c.strip() for c in line.strip('|').split('|')]
|
||
if len(cells) < 2:
|
||
continue
|
||
first = cells[0]
|
||
tm = re.match(r'^(\d{2})\s+(\S.*)$', first)
|
||
title = tm.group(2) if tm else (cells[1] if len(cells) > 1 else '')
|
||
title = re.sub(r'\((EN|RU)\)\s*', '', title)
|
||
title = re.sub(r'^(EN|RU):\s*', '', title)
|
||
title = re.sub(r'\([^)]*\)', '', title) # author/edition parenthetical
|
||
title = title.strip()
|
||
verdict, ev = manifest_stale(sec, title, dl_idx, cards)
|
||
if verdict in ('STALE', 'CHECK'):
|
||
stale.append((line.strip(), verdict, ev))
|
||
if stale:
|
||
print(f"\n### {mf.split(BASE)[1]}")
|
||
for line, v, ev in stale:
|
||
print(f" [{v:5s}] {line[:95]}\n -> {ev}")
|
||
|
||
if write:
|
||
out = os.path.join(BASE, 'data', 'MASTER-LIST.md')
|
||
L = ['# Master book list — ISAP Zurich reading list (sections 01–12)',
|
||
'',
|
||
f'Generated: {datetime.date.today().isoformat()} by `tools/xref.py` — single source of truth',
|
||
'for cross-section dedup.',
|
||
'**Rule:** a book listed in several sections is researched once — at its CANONICAL card',
|
||
'(first section it appears in). Later sections get cross-refs, no new search/download.',
|
||
'']
|
||
also = {}
|
||
for (sec, n), m in matched.items():
|
||
if m['kind'] == 'card':
|
||
c = m['ref']
|
||
also.setdefault((c['sec'], c['n']), []).append(f'sec{sec}#{n}')
|
||
# 01-06 cross-section duplicates (canonical = earliest section)
|
||
for c1, c2 in itertools.combinations(cards, 2):
|
||
if c1['sec'] == c2['sec']:
|
||
continue
|
||
if score_pair(c1['tok'], c1['cw'], c1['author'], c2['tok'], c2['cw'], c2['author']) >= 0.5:
|
||
first, second = (c1, c2) if c1['sec'] < c2['sec'] else (c2, c1)
|
||
also.setdefault((first['sec'], first['n']), []).append(f"sec{second['sec']}#{second['n']}")
|
||
for sec in ['01', '02', '03', '04', '05', '06']:
|
||
sec_cards = [c for c in cards if c['sec'] == sec]
|
||
L.append(f'## Section {sec} — {len(sec_cards)} items (canonical cards)')
|
||
L.append('')
|
||
L.append('| # | EN title | Status | Also listed in |')
|
||
L.append('|---|----------|--------|----------------|')
|
||
for c in sorted(sec_cards, key=lambda x: x['n']):
|
||
a = also.get((c['sec'], c['n']), [])
|
||
L.append(f'| {c["n"]} | {c["title"][:70]} | {c["status"]} | {", ".join(sorted(a)) or "—"} |')
|
||
L.append('')
|
||
for sec in sorted(raw_secs):
|
||
name, entries = raw_secs[sec]
|
||
L.append(f'## Section {sec} — {name} (RAW, provisional numbering)')
|
||
L.append('')
|
||
L.append('| # | EN title | Author | Canonical | New? |')
|
||
L.append('|---|----------|--------|-----------|------|')
|
||
for e in entries:
|
||
m = matched[(sec, e['n'])]
|
||
if m['kind'] == 'card':
|
||
c = m['ref']
|
||
L.append(f'| {e["n"]} | {e["title"][:60]} | {e["author"]} | sec{c["sec"]}#{c["n"]} ({c["status"]}) | — |')
|
||
else:
|
||
L.append(f'| {e["n"]} | {e["title"][:60]} | {e["author"]} | — | **NEW** |')
|
||
L.append('')
|
||
open(out, 'w', encoding='utf8').write('\n'.join(L))
|
||
print(f'\nWROTE {out}')
|
||
|
||
if __name__ == '__main__':
|
||
main()
|