#!/usr/bin/env python3 """audit_anchors.py — reliable wiki-anchor audit for authors/*.md. For each dossier: extract H1 name, **Wiki anchor:** QID + URL, then fetch the QID EXACTLY (wbgetentities, no search) and compare the real EN/RU label with the dossier author. Reports: OK — QID label matches the author MISMATCH — QID resolves to a DIFFERENT person (wrong anchor!) NO-ITEM — QID missing on Wikidata URL-MISMATCH — the URL in the anchor line doesn't match the QID NO-ANCHOR — no QID in the anchor line Outputs data/anchor-audit.tsv + a console report. Read-only (no writes). """ import re, sys, time, csv, glob, os, json, urllib.parse, urllib.request BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) UA = {'User-Agent': 'jung-research/1.0 (personal research script)'} def get(params): url = 'https://www.wikidata.org/w/api.php?' + urllib.parse.urlencode(params) req = urllib.request.Request(url, headers=UA) with urllib.request.urlopen(req, timeout=30) as r: return json.loads(r.read().decode('utf-8')) def wb_labels(qids): out = {} for i in range(0, len(qids), 50): chunk = qids[i:i+50] d = get({'action': 'wbgetentities', 'ids': '|'.join(chunk), 'props': 'labels|descriptions', 'languages': 'en|ru', 'format': 'json'}) for q, e in d.get('entities', {}).items(): if e.get('missing'): out[q] = None continue out[q] = { 'labels': {lang: v.get('value', '') for lang, v in e.get('labels', {}).items() if v.get('value')}, 'desc': e.get('descriptions', {}).get('en', {}).get('value', ''), } time.sleep(1.5) return out STOP = set('the and of van von de la le di du der den das st ste saint st.-'.split()) def tokens(name): return {w for w in re.findall(r"[a-zа-яё']+", (name or '').lower()) if w not in STOP} def name_match(a, b): """surname+given overlap test (order-agnostic, >=1 shared significant token AND the longer name's tokens mostly covered).""" ta, tb = tokens(a), tokens(b) if not ta or not tb: return False shared = ta & tb if not shared: return False # require at least the surname: the longest token shared, or 2 tokens shared return len(shared) >= 2 or max(shared, key=len) in (max(ta, key=len), max(tb, key=len)) def main(): rows = [] qids = {} for path in sorted(glob.glob(os.path.join(BASE, 'authors', '*.md'))): txt = open(path, encoding='utf8').read() h1 = re.search(r'^# (.+)$', txt, re.M) h1 = h1.group(1).strip() if h1 else os.path.basename(path) m = re.search(r'\*\*Wiki anchor:\*\*\s*(.+)', txt) line = m.group(1).strip() if m else '' qm = re.search(r'(Q\d+)', line) um = re.search(r'\((https?://[^)\s]+)\)', line) qid = qm.group(1) if qm else '' url = um.group(1) if um else '' # EN name: prefer "(...)" in H1 else H1 itself; strip dates en = h1 pm = re.search(r'\(([^)]+)\)\s*$', h1) if pm and re.search(r'[a-z]', pm.group(1)): en = pm.group(1) en = re.sub(r'[\[\(].*?[\]\)]', ' ', en) rows.append({'file': 'authors/' + os.path.basename(path), 'h1': h1, 'en': en.strip(), 'qid': qid, 'url': url, 'anchor_line': line}) if qid: qids.setdefault(qid, []).append(rows[-1]) print(f"parsed {len(rows)} dossiers, {len(qids)} unique QIDs", file=sys.stderr) ents = wb_labels(sorted(qids)) out = os.path.join(BASE, 'data', 'anchor-audit.tsv') with open(out, 'w', newline='', encoding='utf8') as f: w = csv.writer(f, delimiter='\t') w.writerow(['file', 'en_name', 'qid', 'wd_en', 'wd_ru', 'wd_desc', 'status', 'url_ok']) for r in rows: q = r['qid'] if not q: st, e = 'NO-ANCHOR', None elif q not in ents: st, e = 'FETCH-ERR', None elif ents[q] is None: st, e = 'NO-ITEM', None else: e = ents[q] ok_en = (name_match(r['en'], e['labels'].get('en', '')) or name_match(r['h1'], e['labels'].get('en', '')) or name_match(r['h1'], e['labels'].get('ru', '')) or any(name_match(r['en'], v) or name_match(r['h1'], v) for k, v in e['labels'].items() if k not in ('en', 'ru'))) st = 'OK' if ok_en else 'MISMATCH' # URL check if q and r['url']: if r['url'].rstrip('/').endswith(q): url_ok = 'Y' elif 'wikidata' in r['url']: url_ok = 'OTHER-QID' else: url_ok = 'NOT-WIKIDATA' else: url_ok = '' w.writerow([r['file'], r['en'], q, e['labels'].get('en', '') if e else '', e['labels'].get('ru', '') if e else '', (e['desc'][:80] if e else ''), st, url_ok]) r['status'], r['url_ok'], r['wd'] = st, url_ok, e bad = [r for r in rows if r['status'] in ('MISMATCH', 'NO-ITEM') or r['url_ok'] in ('OTHER-QID', 'NOT-WIKIDATA')] print(f"\n=== PROBLEMS ({len(bad)}) — {out}\n") for r in sorted(bad, key=lambda x: (x['status'], x['file'])): e = r.get('wd') or {} print(f"[{r['status']:10s}][{r['url_ok'] or '-':11s}] {r['file']:40s} qid={r['qid'] or '-':12s} | dossier: {r['en'][:34]:34s} | WD: {(e.get('labels',{}).get('en') or e.get('labels',{}).get('mul') or e.get('labels',{}).get('de') or '?')[:34]} / {(e.get('labels',{}).get('ru') or '')[:20]}") stats = {} for r in rows: stats[r['status']] = stats.get(r['status'], 0) + 1 print("\n=== STATS:", stats) if __name__ == '__main__': main()