authors: wiki-anchor audit — 1 wrong QID fixed (Homer: Q327970=Uzhevych → Q6691),

9 wikipedia URLs normalized to wikidata, tools/audit_anchors.py (all-language label match)
This commit is contained in:
Dmitry Kokorin 2026-09-25 22:39:01 +03:00
parent c818363dfa
commit 4274488410
12 changed files with 310 additions and 10 deletions

129
tools/audit_anchors.py Normal file
View file

@ -0,0 +1,129 @@
#!/usr/bin/env python3
"""audit_anchors.py — reliable wiki-anchor audit for authors/*.md.
For each dossier: extract H1 name, **Wiki anchor:** QID + URL, then fetch the
QID EXACTLY (wbgetentities, no search) and compare the real EN/RU label with
the dossier author. Reports:
OK — QID label matches the author
MISMATCH — QID resolves to a DIFFERENT person (wrong anchor!)
NO-ITEM — QID missing on Wikidata
URL-MISMATCH — the URL in the anchor line doesn't match the QID
NO-ANCHOR — no QID in the anchor line
Outputs data/anchor-audit.tsv + a console report. Read-only (no writes).
"""
import re, sys, time, csv, glob, os, json, urllib.parse, urllib.request
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
UA = {'User-Agent': 'jung-research/1.0 (personal research script)'}
def get(params):
url = 'https://www.wikidata.org/w/api.php?' + urllib.parse.urlencode(params)
req = urllib.request.Request(url, headers=UA)
with urllib.request.urlopen(req, timeout=30) as r:
return json.loads(r.read().decode('utf-8'))
def wb_labels(qids):
out = {}
for i in range(0, len(qids), 50):
chunk = qids[i:i+50]
d = get({'action': 'wbgetentities', 'ids': '|'.join(chunk),
'props': 'labels|descriptions', 'languages': 'en|ru', 'format': 'json'})
for q, e in d.get('entities', {}).items():
if e.get('missing'):
out[q] = None
continue
out[q] = {
'labels': {lang: v.get('value', '') for lang, v in e.get('labels', {}).items() if v.get('value')},
'desc': e.get('descriptions', {}).get('en', {}).get('value', ''),
}
time.sleep(1.5)
return out
STOP = set('the and of van von de la le di du der den das st ste saint st.-'.split())
def tokens(name):
return {w for w in re.findall(r"[a-zа-яё']+", (name or '').lower()) if w not in STOP}
def name_match(a, b):
"""surname+given overlap test (order-agnostic, >=1 shared significant token
AND the longer name's tokens mostly covered)."""
ta, tb = tokens(a), tokens(b)
if not ta or not tb:
return False
shared = ta & tb
if not shared:
return False
# require at least the surname: the longest token shared, or 2 tokens shared
return len(shared) >= 2 or max(shared, key=len) in (max(ta, key=len), max(tb, key=len))
def main():
rows = []
qids = {}
for path in sorted(glob.glob(os.path.join(BASE, 'authors', '*.md'))):
txt = open(path, encoding='utf8').read()
h1 = re.search(r'^# (.+)$', txt, re.M)
h1 = h1.group(1).strip() if h1 else os.path.basename(path)
m = re.search(r'\*\*Wiki anchor:\*\*\s*(.+)', txt)
line = m.group(1).strip() if m else ''
qm = re.search(r'(Q\d+)', line)
um = re.search(r'\((https?://[^)\s]+)\)', line)
qid = qm.group(1) if qm else ''
url = um.group(1) if um else ''
# EN name: prefer "(...)" in H1 else H1 itself; strip dates
en = h1
pm = re.search(r'\(([^)]+)\)\s*$', h1)
if pm and re.search(r'[a-z]', pm.group(1)):
en = pm.group(1)
en = re.sub(r'[\[\(].*?[\]\)]', ' ', en)
rows.append({'file': 'authors/' + os.path.basename(path), 'h1': h1, 'en': en.strip(),
'qid': qid, 'url': url, 'anchor_line': line})
if qid:
qids.setdefault(qid, []).append(rows[-1])
print(f"parsed {len(rows)} dossiers, {len(qids)} unique QIDs", file=sys.stderr)
ents = wb_labels(sorted(qids))
out = os.path.join(BASE, 'data', 'anchor-audit.tsv')
with open(out, 'w', newline='', encoding='utf8') as f:
w = csv.writer(f, delimiter='\t')
w.writerow(['file', 'en_name', 'qid', 'wd_en', 'wd_ru', 'wd_desc', 'status', 'url_ok'])
for r in rows:
q = r['qid']
if not q:
st, e = 'NO-ANCHOR', None
elif q not in ents:
st, e = 'FETCH-ERR', None
elif ents[q] is None:
st, e = 'NO-ITEM', None
else:
e = ents[q]
ok_en = (name_match(r['en'], e['labels'].get('en', '')) or name_match(r['h1'], e['labels'].get('en', ''))
or name_match(r['h1'], e['labels'].get('ru', ''))
or any(name_match(r['en'], v) or name_match(r['h1'], v) for k, v in e['labels'].items() if k not in ('en', 'ru')))
st = 'OK' if ok_en else 'MISMATCH'
# URL check
if q and r['url']:
if r['url'].rstrip('/').endswith(q):
url_ok = 'Y'
elif 'wikidata' in r['url']:
url_ok = 'OTHER-QID'
else:
url_ok = 'NOT-WIKIDATA'
else:
url_ok = ''
w.writerow([r['file'], r['en'], q,
e['labels'].get('en', '') if e else '', e['labels'].get('ru', '') if e else '',
(e['desc'][:80] if e else ''), st, url_ok])
r['status'], r['url_ok'], r['wd'] = st, url_ok, e
bad = [r for r in rows if r['status'] in ('MISMATCH', 'NO-ITEM') or r['url_ok'] in ('OTHER-QID', 'NOT-WIKIDATA')]
print(f"\n=== PROBLEMS ({len(bad)}) — {out}\n")
for r in sorted(bad, key=lambda x: (x['status'], x['file'])):
e = r.get('wd') or {}
print(f"[{r['status']:10s}][{r['url_ok'] or '-':11s}] {r['file']:40s} qid={r['qid'] or '-':12s} | dossier: {r['en'][:34]:34s} | WD: {(e.get('labels',{}).get('en') or e.get('labels',{}).get('mul') or e.get('labels',{}).get('de') or '?')[:34]} / {(e.get('labels',{}).get('ru') or '')[:20]}")
stats = {}
for r in rows:
stats[r['status']] = stats.get(r['status'], 0) + 1
print("\n=== STATS:", stats)
if __name__ == '__main__':
main()