User caught: Emma Jung dossier said '1877-1965, Jung's eldest daughter'
(actually 1882-1955, his WIFE, Q124210). Systemic response:
- New MANDATORY field **Wiki anchor:** (QID + URL) in dossier canon (AGENTS.md)
- tools/verify_dossiers.py: resumable Wikidata audit (bulk wbgetentities for
anchored, search+CANDIDATE review for the rest, 429 backoff, TSV report)
- 82 dossiers now anchored; 76 marked no-confirmed-QID (WIKI-CHECK notes)
- Date/kinship corrections found by the audit:
* Emma Jung: wife not daughter, 1882-1955 (Q124210)
* Verena Kast: b. 1943 not 1951, father Walter Kast not 'Hans Kast' (Q2515260)
* James Hillman: 1926 not 1941 (Q934774)
* Jolande Jacobi: 1890-1973 not 1902-1980 (Q88022)
* Maria Czaplicka: d. 1921 not 1939 (Q532804)
* Henri Ellenberger: 1905-1993 not 1900-1987 (Q115111)
* Edward F. Edinger: 1922-1998 not 1905-2001 (Q5819984, en.wiki)
* Hans Dieckmann: d. 2007 not 2005 (Q59526701)
* Heinrich Zimmer: 1890-1943, anchored to the Indologist Q215922
(not the Celtist Q76954)
- Homonym traps documented in TSV notes (John Hill=botanist, Karl
Koch=hacker, Barbara Meier=model, Andrew Samuels=soccer, etc.)
165 lines
No EOL
7 KiB
Python
165 lines
No EOL
7 KiB
Python
#!/usr/bin/env python3
|
|
"""verify_dossiers.py — audit author dossiers against Wikidata.
|
|
|
|
Pass 1: dossiers that already have a **Wiki anchor:** QID -> exact entity fetch
|
|
(wbgetentities, no search => no homonyms) -> compare birth/death years + labels.
|
|
Pass 2: dossiers without a QID -> wbsearchentities on the name from the slug ->
|
|
top candidate -> CANDIDATE row (NOT auto-written; review manually, then add the
|
|
anchor line to the dossier).
|
|
|
|
Name reconstruction: slug <lastname>-<firstname...> => "Firstname... Lastname".
|
|
Run: python3 tools/verify_dossiers.py [--write-candidates]
|
|
Output: data/dossier-check.tsv (columns: file, slug, name, dossier_dates, qid,
|
|
wd_born, wd_died, wd_label, status, note)
|
|
"""
|
|
import re, sys, time, csv, urllib.parse, urllib.request, json, glob, os
|
|
|
|
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
UA = {'User-Agent': 'jung-research/1.0 (personal research script)'}
|
|
|
|
def get(params):
|
|
url = 'https://www.wikidata.org/w/api.php?' + urllib.parse.urlencode(params)
|
|
req = urllib.request.Request(url, headers=UA)
|
|
with urllib.request.urlopen(req, timeout=30) as r:
|
|
return json.loads(r.read().decode('utf-8'))
|
|
|
|
def wb_get(qids):
|
|
"""bulk entities: qids list -> {qid: entity}"""
|
|
out = {}
|
|
for i in range(0, len(qids), 50):
|
|
chunk = qids[i:i+50]
|
|
d = get({'action': 'wbgetentities', 'ids': '|'.join(chunk),
|
|
'props': 'labels|claims', 'languages': 'en|ru', 'format': 'json'})
|
|
for q, e in d.get('entities', {}).items():
|
|
if e.get('missing'):
|
|
continue
|
|
lbl = e.get('labels', {}).get('en', {}).get('value') or e.get('labels', {}).get('ru', {}).get('value', '?')
|
|
born = died = ''
|
|
for claim in e.get('claims', {}).get('P569', []):
|
|
v = claim.get('mainsnak', {}).get('datavalue', {}).get('value', {})
|
|
if v.get('time'): born = v['time'][1:5]
|
|
for claim in e.get('claims', {}).get('P570', []):
|
|
v = claim.get('mainsnak', {}).get('datavalue', {}).get('value', {})
|
|
if v.get('time'): died = v['time'][1:5]
|
|
out[q] = {'label': lbl, 'born': born, 'died': died}
|
|
time.sleep(1)
|
|
return out
|
|
|
|
def wb_search(name, limit=3):
|
|
for attempt in range(3):
|
|
try:
|
|
d = get({'action': 'wbsearchentities', 'search': name, 'language': 'en',
|
|
'limit': limit, 'type': 'item', 'format': 'json'})
|
|
return [(r['id'], r.get('label'), r.get('description', '')) for r in d.get('search', [])]
|
|
except Exception as e:
|
|
if '429' in str(e) and attempt < 2:
|
|
time.sleep(30 + attempt * 30)
|
|
elif attempt < 2:
|
|
time.sleep(10)
|
|
else:
|
|
raise
|
|
return []
|
|
|
|
def dossier_qid(txt):
|
|
m = re.search(r'\*\*Wiki anchor:\*\*\s*\**\s*(Q\d+)', txt)
|
|
return m.group(1) if m else None
|
|
|
|
def dump(rows, out):
|
|
with open(out, 'w', newline='') as fh:
|
|
w = csv.writer(fh, delimiter='\t')
|
|
w.writerow(['file', 'name', 'dossier_dates', 'qid', 'wd_born', 'wd_died', 'wd_label', 'status', 'note'])
|
|
w.writerows(rows)
|
|
|
|
def main():
|
|
write_mode = '--write-candidates' in sys.argv
|
|
rows = []
|
|
pending = [] # (file, name) to search
|
|
for f in sorted(glob.glob(os.path.join(BASE, 'authors', '*.md'))):
|
|
txt = open(f).read()
|
|
slug = os.path.basename(f)[:-3]
|
|
# name from H1 "# Last, First [dates]" (canonical), else slug
|
|
m1 = re.search(r'^# (.+?)(?:\s*\[|$)', txt, re.M)
|
|
if m1 and ',' in m1.group(1):
|
|
last, first = m1.group(1).split(',', 1)
|
|
name = (first.strip() + ' ' + last.strip())
|
|
else:
|
|
parts = slug.split('-', 1)
|
|
name = (parts[1] + ' ' + parts[0]).title() if len(parts) == 2 else slug
|
|
# fix obvious multi-word (e.g. marie-louise-von -> "Marie-louise Von")
|
|
md = re.search(r'\*\*Dates:\*\*\s*(.+)', txt)
|
|
dates = md.group(1).strip()[:40] if md else ''
|
|
qid = dossier_qid(txt)
|
|
if qid:
|
|
pending.append((f, name, qid, dates))
|
|
else:
|
|
pending.append((f, name, None, dates))
|
|
|
|
# exact QIDs first
|
|
qids = sorted({p[2] for p in pending if p[2]})
|
|
ents = wb_get(qids) if qids else {}
|
|
rows2 = []
|
|
out = os.path.join(BASE, 'data', 'dossier-check.tsv')
|
|
# resume: load previous report, keep valid statuses (MATCH / CANDIDATE / NO-QID)
|
|
prev = {}
|
|
if os.path.exists(out):
|
|
with open(out) as fh:
|
|
rd = csv.DictReader(fh, delimiter='\t')
|
|
for r in rd:
|
|
if r['status'] in ('MATCH', 'CANDIDATE', 'NO-QID'):
|
|
prev[r['file']] = r
|
|
for f, name, qid, dates in pending:
|
|
rel = os.path.relpath(f, BASE)
|
|
if rel in prev:
|
|
p = prev[rel]
|
|
rows2.append((rel, p['name'], p['dossier_dates'], p['qid'], p['wd_born'],
|
|
p['wd_died'], p['wd_label'], p['status'], p['note']))
|
|
continue
|
|
rows2.append(('...', '...', '...', '...', '...', '...', '...', 'PENDING', ''))
|
|
dump(rows2, out) # incremental: resumable if killed
|
|
rows2.pop()
|
|
if qid:
|
|
e = ents.get(qid, {})
|
|
wborn, wdied, lbl = e.get('born', ''), e.get('died', ''), e.get('label', '?')
|
|
# compare: extract first 4-digit years from dossier dates
|
|
dy = re.findall(r'(1[6-9]\d\d|20\d\d)', dates)
|
|
note = ''
|
|
status = 'MATCH'
|
|
if wborn and dy and wborn[:4] not in dy:
|
|
status, note = 'DATE-DIFF', f'dossier {dy} vs wd {wborn}'
|
|
elif not wborn and not dy:
|
|
status = 'OK'
|
|
rows2.append((rel, name, dates, qid, wborn, wdied, lbl, status, note))
|
|
dump(rows2, out)
|
|
continue
|
|
# search pass
|
|
try:
|
|
cands = wb_search(name, 3)
|
|
except Exception as ex:
|
|
rows2.append((rel, name, dates, '', '', '', '', 'SEARCH-ERR', str(ex)[:60]))
|
|
dump(rows2, out)
|
|
continue
|
|
if not cands:
|
|
rows2.append((rel, name, dates, '', '', '', '', 'NO-QID', 'no candidate'))
|
|
else:
|
|
cqid, clbl, cdesc = cands[0]
|
|
rows2.append((rel, name, dates, cqid, '', '', clbl,
|
|
'CANDIDATE', f'desc: {cdesc[:50]} | alts: ' + '; '.join(c[1] for c in cands[1:])))
|
|
dump(rows2, out)
|
|
time.sleep(3)
|
|
|
|
rows2.sort(key=lambda r: r[7])
|
|
with open(out, 'w', newline='') as fh:
|
|
w = csv.writer(fh, delimiter='\t')
|
|
w.writerow(['file', 'name', 'dossier_dates', 'qid', 'wd_born', 'wd_died', 'wd_label', 'status', 'note'])
|
|
w.writerows(rows2)
|
|
# print summary by status
|
|
from collections import Counter
|
|
c = Counter(r[7] for r in rows2)
|
|
print(' '.join(f'{k}={v}' for k, v in c.most_common()))
|
|
for r in rows2:
|
|
if r[7] in ('DATE-DIFF', 'CANDIDATE', 'NO-QID', 'SEARCH-ERR'):
|
|
print(f"{r[7]:11} {r[0]:35} '{r[1]}' dossier=[{r[2]}] qid={r[3]} wd={r[4]}-{r[5]} label='{r[6]}' {r[8]}")
|
|
print(f"\nreport: {out}")
|
|
|
|
if __name__ == '__main__':
|
|
main() |