xref.py: parser fixes (space-initial authors, hyphen-split surnames, multi-author/multi-work splits, 07-08 cards in pool, min-set scoring)

- AUTHOR_LINE: 'Kalsched D.' (space-initial) now recognized as entry start
- hyphen-split surname across lines: 'Schwartz-' + 'Salant, N.'
- multi-author entries split across lines (Hersh/Caligor/Yeomans)
- multi-work author blocks now split per book (Jacoby sec08: Shame is separate)
- card_entries: glob 0[1-9] (was 0[1-6]) — sec07/08 cards now matchable
  (fixed 5 matcher misses: Bowlby I/II, Winnicott, Jacoby, Knox)
- score_pair: min-set overlap ratio + eps by inter size (subset tie-break: Vol.I vs Vol.II)
- curly quotes excluded from new-book split trigger
sec09 table: 19 auto-xref (18 true + 1 fragment), 19 NEW
This commit is contained in:
Dmitry Kokorin 2026-09-27 11:43:53 +03:00
parent db2ee4ab53
commit 5e07978171
2 changed files with 226 additions and 200 deletions

View file

@ -47,7 +47,7 @@ CW_LINE = re.compile(r'^\s{0,8}CW\s*(\d+/\w{1,2}|\d+|-?S\d?)\s+(\S.*)$')
AUTHOR_LINE = re.compile(
r'^(\s{0,8})'
r'([A-Z][A-Za-z\u00C0-\u017F\-]+)' # 2: surname
r'(?:\s*,\s*[A-Za-z.\-](?:\s?[A-Za-z.\-]){0,8})?' # initials blob ", C. A."
r'(?:\s*,\s*[A-Za-z.\-](?:\s?[A-Za-z.\-]){0,8}|\s+[A-Z]\.)?' # initials ", C. A." or space-initial "Kalsched D."
r'(?:\s*,\s*[A-Z][A-Za-z\u00C0-\u017F\-]+\s*,\s*[A-Za-z.\-](?:\s?[A-Za-z.\-]){0,8}){0,3}' # ", Surname, I." x3
r'\s*,?'
r'\s{2,}(\S.*)$')
@ -88,7 +88,14 @@ def parse_raw(path):
cur = {'cw': vol, 'author': 'Jung', 'title': m.group(2), 'raw': stripped}
continue
m = AUTHOR_LINE.match(ln)
if m and not stripped.startswith(('"', '*', '—', '.', 'In:')):
if m and not stripped.startswith(('"', '*', '—', '.', 'In:', '“', '‘', '«')):
if cur and cur['author'].endswith('-'):
# hyphen-split surname across lines
sp = re.search(r'\s{2,}', ln.lstrip())
author_part = (ln.lstrip()[:sp.start()].rstrip() if sp else ln.lstrip().rstrip()).rstrip(',')
cur['author'] = cur['author'].rstrip('-') + '-' + re.sub(r'\s+', ' ', author_part).strip()
cur['title'] += ' ' + m.group(3).strip()
continue
if cur and CITY_START.match(m.group(3)):
# 2nd author + "City: Publisher, Year" line of the CURRENT entry
cur['author'] += ', ' + m.group(2).strip().rstrip(',')
@ -99,6 +106,13 @@ def parse_raw(path):
cur['author'] += ', ' + m.group(2).strip().rstrip(',')
cur['title'] += ' ' + m.group(3).strip()
continue
if cur and not re.search(r'\d{4}', cur['title']) and re.search(r'\b(in|of|the|and|a|to|for|on|between|with)$', cur['title']):
# multi-author entry split across lines (Hersh/Caligor/Yeomans case)
sp = re.search(r'\s{2,}', ln.lstrip())
author_part = (ln.lstrip()[:sp.start()].rstrip() if sp else ln.lstrip().rstrip()).rstrip(',')
cur['author'] += ', ' + re.sub(r'\s+', ' ', author_part).strip()
cur['title'] += ' ' + m.group(3).strip()
continue
flush()
author = m.group(2).strip().rstrip(',')
cur = {'cw': None, 'author': author, 'title': m.group(3), 'raw': stripped}
@ -111,14 +125,15 @@ def parse_raw(path):
if cur and (indent >= 8 or cur['cw']):
if (cur['cw'] is None and indent >= 8 and len(stripped) >= 15
and not stripped.startswith(('"', '*', '—', '.', 'In:', 'Part', 'Appendix',
'Chap', 'Ch.', 'Vol', 'Note', 'Notes'))
'Chap', 'Ch.', 'Vol', 'Note', 'Notes', '“', '‘', '«'))
and not re.match(r'^[A-Z]{1,4}[\-:,.]?\w{0,6}$', stripped)
and re.search(r'[a-z]{3,}', stripped)
and re.search(r'\b(?:New\s+York|London|Boston|Toronto|Hove|Ithaca|Woodstock|'
and (re.search(r'\b(?:New\s+York|London|Boston|Toronto|Hove|Ithaca|Woodstock|'
r'Cambridge|Tokyo|Geneva|Wilmette|Hillsdale|Northvale|Cham|Arles|'
r'Arlington|Chicago|Asheville|Oxford|Basel|Paris|Berlin|Princeton|'
r'Syracuse|New\s+Haven|Heidelberg|Amsterdam|Sigtuna)\s*:\s*'
r'[A-Za-z][^,]{0,50}?\,\s*\d{4}', cur['title'])):
r'[A-Za-z][^,]{0,50}?\,\s*\d{4}', cur['title'])
or re.search(r'(?:19|20)\d{2}', cur['title']))):
prev_author = cur['author']
flush()
cur = {'cw': None, 'author': prev_author, 'title': stripped, 'raw': stripped}
@ -143,7 +158,7 @@ def parse_raw(path):
# ---------------- CARDS (01-06) ----------------
def card_entries():
out = []
for path in sorted(glob.glob(os.path.join(BASE, 'sections', '0[1-6]-*', '??-*.md'))):
for path in sorted(glob.glob(os.path.join(BASE, 'sections', '0[1-9]-*', '??-*.md'))):
txt = open(path, encoding='utf8').read()
h1 = re.search(r'^# (.+)$', txt, re.M)
am = re.search(r'\*\*Author\(s?\):\*\*\s*(.+)', txt, re.I)
@ -235,7 +250,8 @@ def score_pair(e_tok, e_cw, e_author, c_tok, c_cw, c_author):
if e_cw and c_cw:
return 1.0 # CW volume match + >=2 shared tokens
if len(distinct) >= 2:
return len(distinct) / max(1, len(inter))
# min-set overlap ratio (+ eps by inter size to break subset ties: Vol.I vs Vol.II)
return len(inter) / max(1, min(len(e_tok), len(c_tok))) + len(inter) / 1000.0
if len(c_tok - GENERIC) == 0 and cak == ak and not (e_tok - GENERIC) and len(inter) >= 2:
return 0.6 # card title is all-generic (e.g. "The Psychology of C.G. Jung")
# short title, one distinctive token shared (e.g. "From Freud to Jung")