jung/tools/verify_dossiers.py
Dmitry Kokorin bc1f91d9be authors: systemic identity audit — Wiki anchors + 9 date/kinship corrections
User caught: Emma Jung dossier said '1877-1965, Jung's eldest daughter'
(actually 1882-1955, his WIFE, Q124210). Systemic response:
- New MANDATORY field **Wiki anchor:** (QID + URL) in dossier canon (AGENTS.md)
- tools/verify_dossiers.py: resumable Wikidata audit (bulk wbgetentities for
  anchored, search+CANDIDATE review for the rest, 429 backoff, TSV report)
- 82 dossiers now anchored; 76 marked no-confirmed-QID (WIKI-CHECK notes)
- Date/kinship corrections found by the audit:
  * Emma Jung: wife not daughter, 1882-1955 (Q124210)
  * Verena Kast: b. 1943 not 1951, father Walter Kast not 'Hans Kast' (Q2515260)
  * James Hillman: 1926 not 1941 (Q934774)
  * Jolande Jacobi: 1890-1973 not 1902-1980 (Q88022)
  * Maria Czaplicka: d. 1921 not 1939 (Q532804)
  * Henri Ellenberger: 1905-1993 not 1900-1987 (Q115111)
  * Edward F. Edinger: 1922-1998 not 1905-2001 (Q5819984, en.wiki)
  * Hans Dieckmann: d. 2007 not 2005 (Q59526701)
  * Heinrich Zimmer: 1890-1943, anchored to the Indologist Q215922
    (not the Celtist Q76954)
- Homonym traps documented in TSV notes (John Hill=botanist, Karl
  Koch=hacker, Barbara Meier=model, Andrew Samuels=soccer, etc.)
2026-09-24 10:29:54 +03:00

165 lines
No EOL
7 KiB
Python

#!/usr/bin/env python3
"""verify_dossiers.py — audit author dossiers against Wikidata.
Pass 1: dossiers that already have a **Wiki anchor:** QID -> exact entity fetch
(wbgetentities, no search => no homonyms) -> compare birth/death years + labels.
Pass 2: dossiers without a QID -> wbsearchentities on the name from the slug ->
top candidate -> CANDIDATE row (NOT auto-written; review manually, then add the
anchor line to the dossier).
Name reconstruction: slug <lastname>-<firstname...> => "Firstname... Lastname".
Run: python3 tools/verify_dossiers.py [--write-candidates]
Output: data/dossier-check.tsv (columns: file, slug, name, dossier_dates, qid,
wd_born, wd_died, wd_label, status, note)
"""
import re, sys, time, csv, urllib.parse, urllib.request, json, glob, os
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
UA = {'User-Agent': 'jung-research/1.0 (personal research script)'}
def get(params):
url = 'https://www.wikidata.org/w/api.php?' + urllib.parse.urlencode(params)
req = urllib.request.Request(url, headers=UA)
with urllib.request.urlopen(req, timeout=30) as r:
return json.loads(r.read().decode('utf-8'))
def wb_get(qids):
"""bulk entities: qids list -> {qid: entity}"""
out = {}
for i in range(0, len(qids), 50):
chunk = qids[i:i+50]
d = get({'action': 'wbgetentities', 'ids': '|'.join(chunk),
'props': 'labels|claims', 'languages': 'en|ru', 'format': 'json'})
for q, e in d.get('entities', {}).items():
if e.get('missing'):
continue
lbl = e.get('labels', {}).get('en', {}).get('value') or e.get('labels', {}).get('ru', {}).get('value', '?')
born = died = ''
for claim in e.get('claims', {}).get('P569', []):
v = claim.get('mainsnak', {}).get('datavalue', {}).get('value', {})
if v.get('time'): born = v['time'][1:5]
for claim in e.get('claims', {}).get('P570', []):
v = claim.get('mainsnak', {}).get('datavalue', {}).get('value', {})
if v.get('time'): died = v['time'][1:5]
out[q] = {'label': lbl, 'born': born, 'died': died}
time.sleep(1)
return out
def wb_search(name, limit=3):
for attempt in range(3):
try:
d = get({'action': 'wbsearchentities', 'search': name, 'language': 'en',
'limit': limit, 'type': 'item', 'format': 'json'})
return [(r['id'], r.get('label'), r.get('description', '')) for r in d.get('search', [])]
except Exception as e:
if '429' in str(e) and attempt < 2:
time.sleep(30 + attempt * 30)
elif attempt < 2:
time.sleep(10)
else:
raise
return []
def dossier_qid(txt):
m = re.search(r'\*\*Wiki anchor:\*\*\s*\**\s*(Q\d+)', txt)
return m.group(1) if m else None
def dump(rows, out):
with open(out, 'w', newline='') as fh:
w = csv.writer(fh, delimiter='\t')
w.writerow(['file', 'name', 'dossier_dates', 'qid', 'wd_born', 'wd_died', 'wd_label', 'status', 'note'])
w.writerows(rows)
def main():
write_mode = '--write-candidates' in sys.argv
rows = []
pending = [] # (file, name) to search
for f in sorted(glob.glob(os.path.join(BASE, 'authors', '*.md'))):
txt = open(f).read()
slug = os.path.basename(f)[:-3]
# name from H1 "# Last, First [dates]" (canonical), else slug
m1 = re.search(r'^# (.+?)(?:\s*\[|$)', txt, re.M)
if m1 and ',' in m1.group(1):
last, first = m1.group(1).split(',', 1)
name = (first.strip() + ' ' + last.strip())
else:
parts = slug.split('-', 1)
name = (parts[1] + ' ' + parts[0]).title() if len(parts) == 2 else slug
# fix obvious multi-word (e.g. marie-louise-von -> "Marie-louise Von")
md = re.search(r'\*\*Dates:\*\*\s*(.+)', txt)
dates = md.group(1).strip()[:40] if md else ''
qid = dossier_qid(txt)
if qid:
pending.append((f, name, qid, dates))
else:
pending.append((f, name, None, dates))
# exact QIDs first
qids = sorted({p[2] for p in pending if p[2]})
ents = wb_get(qids) if qids else {}
rows2 = []
out = os.path.join(BASE, 'data', 'dossier-check.tsv')
# resume: load previous report, keep valid statuses (MATCH / CANDIDATE / NO-QID)
prev = {}
if os.path.exists(out):
with open(out) as fh:
rd = csv.DictReader(fh, delimiter='\t')
for r in rd:
if r['status'] in ('MATCH', 'CANDIDATE', 'NO-QID'):
prev[r['file']] = r
for f, name, qid, dates in pending:
rel = os.path.relpath(f, BASE)
if rel in prev:
p = prev[rel]
rows2.append((rel, p['name'], p['dossier_dates'], p['qid'], p['wd_born'],
p['wd_died'], p['wd_label'], p['status'], p['note']))
continue
rows2.append(('...', '...', '...', '...', '...', '...', '...', 'PENDING', ''))
dump(rows2, out) # incremental: resumable if killed
rows2.pop()
if qid:
e = ents.get(qid, {})
wborn, wdied, lbl = e.get('born', ''), e.get('died', ''), e.get('label', '?')
# compare: extract first 4-digit years from dossier dates
dy = re.findall(r'(1[6-9]\d\d|20\d\d)', dates)
note = ''
status = 'MATCH'
if wborn and dy and wborn[:4] not in dy:
status, note = 'DATE-DIFF', f'dossier {dy} vs wd {wborn}'
elif not wborn and not dy:
status = 'OK'
rows2.append((rel, name, dates, qid, wborn, wdied, lbl, status, note))
dump(rows2, out)
continue
# search pass
try:
cands = wb_search(name, 3)
except Exception as ex:
rows2.append((rel, name, dates, '', '', '', '', 'SEARCH-ERR', str(ex)[:60]))
dump(rows2, out)
continue
if not cands:
rows2.append((rel, name, dates, '', '', '', '', 'NO-QID', 'no candidate'))
else:
cqid, clbl, cdesc = cands[0]
rows2.append((rel, name, dates, cqid, '', '', clbl,
'CANDIDATE', f'desc: {cdesc[:50]} | alts: ' + '; '.join(c[1] for c in cands[1:])))
dump(rows2, out)
time.sleep(3)
rows2.sort(key=lambda r: r[7])
with open(out, 'w', newline='') as fh:
w = csv.writer(fh, delimiter='\t')
w.writerow(['file', 'name', 'dossier_dates', 'qid', 'wd_born', 'wd_died', 'wd_label', 'status', 'note'])
w.writerows(rows2)
# print summary by status
from collections import Counter
c = Counter(r[7] for r in rows2)
print(' '.join(f'{k}={v}' for k, v in c.most_common()))
for r in rows2:
if r[7] in ('DATE-DIFF', 'CANDIDATE', 'NO-QID', 'SEARCH-ERR'):
print(f"{r[7]:11} {r[0]:35} '{r[1]}' dossier=[{r[2]}] qid={r[3]} wd={r[4]}-{r[5]} label='{r[6]}' {r[8]}")
print(f"\nreport: {out}")
if __name__ == '__main__':
main()