jung/tools/audit_anchors.py
Dmitry Kokorin 4274488410 authors: wiki-anchor audit — 1 wrong QID fixed (Homer: Q327970=Uzhevych → Q6691),
9 wikipedia URLs normalized to wikidata, tools/audit_anchors.py (all-language label match)
2026-09-25 22:39:01 +03:00

129 lines
5.8 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""audit_anchors.py — reliable wiki-anchor audit for authors/*.md.
For each dossier: extract H1 name, **Wiki anchor:** QID + URL, then fetch the
QID EXACTLY (wbgetentities, no search) and compare the real EN/RU label with
the dossier author. Reports:
OK — QID label matches the author
MISMATCH — QID resolves to a DIFFERENT person (wrong anchor!)
NO-ITEM — QID missing on Wikidata
URL-MISMATCH — the URL in the anchor line doesn't match the QID
NO-ANCHOR — no QID in the anchor line
Outputs data/anchor-audit.tsv + a console report. Read-only (no writes).
"""
import re, sys, time, csv, glob, os, json, urllib.parse, urllib.request
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
UA = {'User-Agent': 'jung-research/1.0 (personal research script)'}
def get(params):
url = 'https://www.wikidata.org/w/api.php?' + urllib.parse.urlencode(params)
req = urllib.request.Request(url, headers=UA)
with urllib.request.urlopen(req, timeout=30) as r:
return json.loads(r.read().decode('utf-8'))
def wb_labels(qids):
out = {}
for i in range(0, len(qids), 50):
chunk = qids[i:i+50]
d = get({'action': 'wbgetentities', 'ids': '|'.join(chunk),
'props': 'labels|descriptions', 'languages': 'en|ru', 'format': 'json'})
for q, e in d.get('entities', {}).items():
if e.get('missing'):
out[q] = None
continue
out[q] = {
'labels': {lang: v.get('value', '') for lang, v in e.get('labels', {}).items() if v.get('value')},
'desc': e.get('descriptions', {}).get('en', {}).get('value', ''),
}
time.sleep(1.5)
return out
STOP = set('the and of van von de la le di du der den das st ste saint st.-'.split())
def tokens(name):
return {w for w in re.findall(r"[a-zа-яё']+", (name or '').lower()) if w not in STOP}
def name_match(a, b):
"""surname+given overlap test (order-agnostic, >=1 shared significant token
AND the longer name's tokens mostly covered)."""
ta, tb = tokens(a), tokens(b)
if not ta or not tb:
return False
shared = ta & tb
if not shared:
return False
# require at least the surname: the longest token shared, or 2 tokens shared
return len(shared) >= 2 or max(shared, key=len) in (max(ta, key=len), max(tb, key=len))
def main():
rows = []
qids = {}
for path in sorted(glob.glob(os.path.join(BASE, 'authors', '*.md'))):
txt = open(path, encoding='utf8').read()
h1 = re.search(r'^# (.+)$', txt, re.M)
h1 = h1.group(1).strip() if h1 else os.path.basename(path)
m = re.search(r'\*\*Wiki anchor:\*\*\s*(.+)', txt)
line = m.group(1).strip() if m else ''
qm = re.search(r'(Q\d+)', line)
um = re.search(r'\((https?://[^)\s]+)\)', line)
qid = qm.group(1) if qm else ''
url = um.group(1) if um else ''
# EN name: prefer "(...)" in H1 else H1 itself; strip dates
en = h1
pm = re.search(r'\(([^)]+)\)\s*$', h1)
if pm and re.search(r'[a-z]', pm.group(1)):
en = pm.group(1)
en = re.sub(r'[\[\(].*?[\]\)]', ' ', en)
rows.append({'file': 'authors/' + os.path.basename(path), 'h1': h1, 'en': en.strip(),
'qid': qid, 'url': url, 'anchor_line': line})
if qid:
qids.setdefault(qid, []).append(rows[-1])
print(f"parsed {len(rows)} dossiers, {len(qids)} unique QIDs", file=sys.stderr)
ents = wb_labels(sorted(qids))
out = os.path.join(BASE, 'data', 'anchor-audit.tsv')
with open(out, 'w', newline='', encoding='utf8') as f:
w = csv.writer(f, delimiter='\t')
w.writerow(['file', 'en_name', 'qid', 'wd_en', 'wd_ru', 'wd_desc', 'status', 'url_ok'])
for r in rows:
q = r['qid']
if not q:
st, e = 'NO-ANCHOR', None
elif q not in ents:
st, e = 'FETCH-ERR', None
elif ents[q] is None:
st, e = 'NO-ITEM', None
else:
e = ents[q]
ok_en = (name_match(r['en'], e['labels'].get('en', '')) or name_match(r['h1'], e['labels'].get('en', ''))
or name_match(r['h1'], e['labels'].get('ru', ''))
or any(name_match(r['en'], v) or name_match(r['h1'], v) for k, v in e['labels'].items() if k not in ('en', 'ru')))
st = 'OK' if ok_en else 'MISMATCH'
# URL check
if q and r['url']:
if r['url'].rstrip('/').endswith(q):
url_ok = 'Y'
elif 'wikidata' in r['url']:
url_ok = 'OTHER-QID'
else:
url_ok = 'NOT-WIKIDATA'
else:
url_ok = ''
w.writerow([r['file'], r['en'], q,
e['labels'].get('en', '') if e else '', e['labels'].get('ru', '') if e else '',
(e['desc'][:80] if e else ''), st, url_ok])
r['status'], r['url_ok'], r['wd'] = st, url_ok, e
bad = [r for r in rows if r['status'] in ('MISMATCH', 'NO-ITEM') or r['url_ok'] in ('OTHER-QID', 'NOT-WIKIDATA')]
print(f"\n=== PROBLEMS ({len(bad)}) — {out}\n")
for r in sorted(bad, key=lambda x: (x['status'], x['file'])):
e = r.get('wd') or {}
print(f"[{r['status']:10s}][{r['url_ok'] or '-':11s}] {r['file']:40s} qid={r['qid'] or '-':12s} | dossier: {r['en'][:34]:34s} | WD: {(e.get('labels',{}).get('en') or e.get('labels',{}).get('mul') or e.get('labels',{}).get('de') or '?')[:34]} / {(e.get('labels',{}).get('ru') or '')[:20]}")
stats = {}
for r in rows:
stats[r['status']] = stats.get(r['status'], 0) + 1
print("\n=== STATS:", stats)
if __name__ == '__main__':
main()