authors: systemic identity audit — Wiki anchors + 9 date/kinship corrections
User caught: Emma Jung dossier said '1877-1965, Jung's eldest daughter'
(actually 1882-1955, his WIFE, Q124210). Systemic response:
- New MANDATORY field **Wiki anchor:** (QID + URL) in dossier canon (AGENTS.md)
- tools/verify_dossiers.py: resumable Wikidata audit (bulk wbgetentities for
anchored, search+CANDIDATE review for the rest, 429 backoff, TSV report)
- 82 dossiers now anchored; 76 marked no-confirmed-QID (WIKI-CHECK notes)
- Date/kinship corrections found by the audit:
* Emma Jung: wife not daughter, 1882-1955 (Q124210)
* Verena Kast: b. 1943 not 1951, father Walter Kast not 'Hans Kast' (Q2515260)
* James Hillman: 1926 not 1941 (Q934774)
* Jolande Jacobi: 1890-1973 not 1902-1980 (Q88022)
* Maria Czaplicka: d. 1921 not 1939 (Q532804)
* Henri Ellenberger: 1905-1993 not 1900-1987 (Q115111)
* Edward F. Edinger: 1922-1998 not 1905-2001 (Q5819984, en.wiki)
* Hans Dieckmann: d. 2007 not 2005 (Q59526701)
* Heinrich Zimmer: 1890-1943, anchored to the Indologist Q215922
(not the Celtist Q76954)
- Homonym traps documented in TSV notes (John Hill=botanist, Karl
Koch=hacker, Barbara Meier=model, Andrew Samuels=soccer, etc.)
This commit is contained in:
parent
2a2699d61a
commit
bc1f91d9be
116 changed files with 523 additions and 15 deletions
165
tools/verify_dossiers.py
Normal file
165
tools/verify_dossiers.py
Normal file
|
|
@ -0,0 +1,165 @@
|
|||
#!/usr/bin/env python3
|
||||
"""verify_dossiers.py — audit author dossiers against Wikidata.
|
||||
|
||||
Pass 1: dossiers that already have a **Wiki anchor:** QID -> exact entity fetch
|
||||
(wbgetentities, no search => no homonyms) -> compare birth/death years + labels.
|
||||
Pass 2: dossiers without a QID -> wbsearchentities on the name from the slug ->
|
||||
top candidate -> CANDIDATE row (NOT auto-written; review manually, then add the
|
||||
anchor line to the dossier).
|
||||
|
||||
Name reconstruction: slug <lastname>-<firstname...> => "Firstname... Lastname".
|
||||
Run: python3 tools/verify_dossiers.py [--write-candidates]
|
||||
Output: data/dossier-check.tsv (columns: file, slug, name, dossier_dates, qid,
|
||||
wd_born, wd_died, wd_label, status, note)
|
||||
"""
|
||||
import re, sys, time, csv, urllib.parse, urllib.request, json, glob, os
|
||||
|
||||
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
UA = {'User-Agent': 'jung-research/1.0 (personal research script)'}
|
||||
|
||||
def get(params):
|
||||
url = 'https://www.wikidata.org/w/api.php?' + urllib.parse.urlencode(params)
|
||||
req = urllib.request.Request(url, headers=UA)
|
||||
with urllib.request.urlopen(req, timeout=30) as r:
|
||||
return json.loads(r.read().decode('utf-8'))
|
||||
|
||||
def wb_get(qids):
|
||||
"""bulk entities: qids list -> {qid: entity}"""
|
||||
out = {}
|
||||
for i in range(0, len(qids), 50):
|
||||
chunk = qids[i:i+50]
|
||||
d = get({'action': 'wbgetentities', 'ids': '|'.join(chunk),
|
||||
'props': 'labels|claims', 'languages': 'en|ru', 'format': 'json'})
|
||||
for q, e in d.get('entities', {}).items():
|
||||
if e.get('missing'):
|
||||
continue
|
||||
lbl = e.get('labels', {}).get('en', {}).get('value') or e.get('labels', {}).get('ru', {}).get('value', '?')
|
||||
born = died = ''
|
||||
for claim in e.get('claims', {}).get('P569', []):
|
||||
v = claim.get('mainsnak', {}).get('datavalue', {}).get('value', {})
|
||||
if v.get('time'): born = v['time'][1:5]
|
||||
for claim in e.get('claims', {}).get('P570', []):
|
||||
v = claim.get('mainsnak', {}).get('datavalue', {}).get('value', {})
|
||||
if v.get('time'): died = v['time'][1:5]
|
||||
out[q] = {'label': lbl, 'born': born, 'died': died}
|
||||
time.sleep(1)
|
||||
return out
|
||||
|
||||
def wb_search(name, limit=3):
|
||||
for attempt in range(3):
|
||||
try:
|
||||
d = get({'action': 'wbsearchentities', 'search': name, 'language': 'en',
|
||||
'limit': limit, 'type': 'item', 'format': 'json'})
|
||||
return [(r['id'], r.get('label'), r.get('description', '')) for r in d.get('search', [])]
|
||||
except Exception as e:
|
||||
if '429' in str(e) and attempt < 2:
|
||||
time.sleep(30 + attempt * 30)
|
||||
elif attempt < 2:
|
||||
time.sleep(10)
|
||||
else:
|
||||
raise
|
||||
return []
|
||||
|
||||
def dossier_qid(txt):
|
||||
m = re.search(r'\*\*Wiki anchor:\*\*\s*\**\s*(Q\d+)', txt)
|
||||
return m.group(1) if m else None
|
||||
|
||||
def dump(rows, out):
|
||||
with open(out, 'w', newline='') as fh:
|
||||
w = csv.writer(fh, delimiter='\t')
|
||||
w.writerow(['file', 'name', 'dossier_dates', 'qid', 'wd_born', 'wd_died', 'wd_label', 'status', 'note'])
|
||||
w.writerows(rows)
|
||||
|
||||
def main():
|
||||
write_mode = '--write-candidates' in sys.argv
|
||||
rows = []
|
||||
pending = [] # (file, name) to search
|
||||
for f in sorted(glob.glob(os.path.join(BASE, 'authors', '*.md'))):
|
||||
txt = open(f).read()
|
||||
slug = os.path.basename(f)[:-3]
|
||||
# name from H1 "# Last, First [dates]" (canonical), else slug
|
||||
m1 = re.search(r'^# (.+?)(?:\s*\[|$)', txt, re.M)
|
||||
if m1 and ',' in m1.group(1):
|
||||
last, first = m1.group(1).split(',', 1)
|
||||
name = (first.strip() + ' ' + last.strip())
|
||||
else:
|
||||
parts = slug.split('-', 1)
|
||||
name = (parts[1] + ' ' + parts[0]).title() if len(parts) == 2 else slug
|
||||
# fix obvious multi-word (e.g. marie-louise-von -> "Marie-louise Von")
|
||||
md = re.search(r'\*\*Dates:\*\*\s*(.+)', txt)
|
||||
dates = md.group(1).strip()[:40] if md else ''
|
||||
qid = dossier_qid(txt)
|
||||
if qid:
|
||||
pending.append((f, name, qid, dates))
|
||||
else:
|
||||
pending.append((f, name, None, dates))
|
||||
|
||||
# exact QIDs first
|
||||
qids = sorted({p[2] for p in pending if p[2]})
|
||||
ents = wb_get(qids) if qids else {}
|
||||
rows2 = []
|
||||
out = os.path.join(BASE, 'data', 'dossier-check.tsv')
|
||||
# resume: load previous report, keep valid statuses (MATCH / CANDIDATE / NO-QID)
|
||||
prev = {}
|
||||
if os.path.exists(out):
|
||||
with open(out) as fh:
|
||||
rd = csv.DictReader(fh, delimiter='\t')
|
||||
for r in rd:
|
||||
if r['status'] in ('MATCH', 'CANDIDATE', 'NO-QID'):
|
||||
prev[r['file']] = r
|
||||
for f, name, qid, dates in pending:
|
||||
rel = os.path.relpath(f, BASE)
|
||||
if rel in prev:
|
||||
p = prev[rel]
|
||||
rows2.append((rel, p['name'], p['dossier_dates'], p['qid'], p['wd_born'],
|
||||
p['wd_died'], p['wd_label'], p['status'], p['note']))
|
||||
continue
|
||||
rows2.append(('...', '...', '...', '...', '...', '...', '...', 'PENDING', ''))
|
||||
dump(rows2, out) # incremental: resumable if killed
|
||||
rows2.pop()
|
||||
if qid:
|
||||
e = ents.get(qid, {})
|
||||
wborn, wdied, lbl = e.get('born', ''), e.get('died', ''), e.get('label', '?')
|
||||
# compare: extract first 4-digit years from dossier dates
|
||||
dy = re.findall(r'(1[6-9]\d\d|20\d\d)', dates)
|
||||
note = ''
|
||||
status = 'MATCH'
|
||||
if wborn and dy and wborn[:4] not in dy:
|
||||
status, note = 'DATE-DIFF', f'dossier {dy} vs wd {wborn}'
|
||||
elif not wborn and not dy:
|
||||
status = 'OK'
|
||||
rows2.append((rel, name, dates, qid, wborn, wdied, lbl, status, note))
|
||||
dump(rows2, out)
|
||||
continue
|
||||
# search pass
|
||||
try:
|
||||
cands = wb_search(name, 3)
|
||||
except Exception as ex:
|
||||
rows2.append((rel, name, dates, '', '', '', '', 'SEARCH-ERR', str(ex)[:60]))
|
||||
dump(rows2, out)
|
||||
continue
|
||||
if not cands:
|
||||
rows2.append((rel, name, dates, '', '', '', '', 'NO-QID', 'no candidate'))
|
||||
else:
|
||||
cqid, clbl, cdesc = cands[0]
|
||||
rows2.append((rel, name, dates, cqid, '', '', clbl,
|
||||
'CANDIDATE', f'desc: {cdesc[:50]} | alts: ' + '; '.join(c[1] for c in cands[1:])))
|
||||
dump(rows2, out)
|
||||
time.sleep(3)
|
||||
|
||||
rows2.sort(key=lambda r: r[7])
|
||||
with open(out, 'w', newline='') as fh:
|
||||
w = csv.writer(fh, delimiter='\t')
|
||||
w.writerow(['file', 'name', 'dossier_dates', 'qid', 'wd_born', 'wd_died', 'wd_label', 'status', 'note'])
|
||||
w.writerows(rows2)
|
||||
# print summary by status
|
||||
from collections import Counter
|
||||
c = Counter(r[7] for r in rows2)
|
||||
print(' '.join(f'{k}={v}' for k, v in c.most_common()))
|
||||
for r in rows2:
|
||||
if r[7] in ('DATE-DIFF', 'CANDIDATE', 'NO-QID', 'SEARCH-ERR'):
|
||||
print(f"{r[7]:11} {r[0]:35} '{r[1]}' dossier=[{r[2]}] qid={r[3]} wd={r[4]}-{r[5]} label='{r[6]}' {r[8]}")
|
||||
print(f"\nreport: {out}")
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
Loading…
Add table
Add a link
Reference in a new issue