#!/usr/bin/env python3 """doi.py — article-item pipeline: author+title → Crossref DOI → libgen availability. For reading-list items that are JOURNAL ARTICLES (not books), the DOI is the most reliable identifier: Crossref/OpenAlex resolve it from author+title, and libgen.vg indexes many papers BY DOI (req= — verified 2026-09-26: Krieger 10.1111/1468-5922.12544 → exactly ed 84462696). usage: doi.py "Lastname" "Title words..." [--year YYYY] [--no-libgen] Steps: 1. Crossref /works?query.bibliographic=TITLE&query.author=LAST (rows=8). 2. Rank: title token overlap + year proximity (if --year). 3. For top hits with DOI: libgen req= → count edition links (4s polite). 4. Print candidates: title | journal | author | year | DOI | LG: n (ed ids). Limits: pre-1997 works have no DOIs (1975 Hill paper = unresolvable — fall back to title sweeps). Conference papers in APA PsycEXTRA (10.1037/e*) are often noise. After a hit: content-verify the downloaded file (first page: author+journal+pages must match RAW; must be the paper, not a review of it). """ import json, re, sys, time, urllib.parse, urllib.request UA = {'User-Agent': 'jung-research/1.0 (mailto:dmitry@kokorin.org)'} LG_UA = {'User-Agent': 'Mozilla/5.0 (X11; Linux x86_64) Firefox/130.0'} def crossref(last, title, year=None): q = {'query.bibliographic': title, 'query.author': last, 'rows': 8} if year: q['filter'] = f'from-pub-date:{year-2},until-pub-date:{year+2}' url = f"https://api.crossref.org/works?{urllib.parse.urlencode(q)}" req = urllib.request.Request(url, headers=UA) d = json.load(urllib.request.urlopen(req, timeout=40)) out = [] for it in d['message']['items']: doi = it.get('DOI') if not doi: continue a = it.get('author', [{}])[0] fam = a.get('family', '?') yr = (it.get('issued', {}).get('date-parts') or [['?']])[0][0] src = (it.get('container-title') or [''])[0] t = it.get('title', ['?'])[0] # token overlap score want = set(re.findall(r'\w+', title.lower())) got = set(re.findall(r'\w+', (t + ' ' + src).lower())) score = len(want & got) / max(1, len(want)) out.append({'title': re.sub(r'<[^>]+>', '', t), 'journal': src, 'author': fam, 'year': yr, 'doi': doi, 'score': score}) out.sort(key=lambda x: (-x['score'], abs((x['year'] or 0) - (year or 0)) if year else 0)) return out def libgen_doi(doi): url = f"https://libgen.vg/index.php?req={urllib.parse.quote(doi)}&res=25" req = urllib.request.Request(url, headers=LG_UA) try: h = urllib.request.urlopen(req, timeout=40).read().decode('utf-8', 'replace') except Exception as e: return f"ERR {e}" ids = sorted(set(re.findall(r'edition\.php\?id=(\d+)', h))) return f"{len(ids)} ({', '.join(ids[:5])})" if ids else "0" def main(): args = [a for a in sys.argv[1:] if not a.startswith('--')] flags = [a for a in sys.argv[1:] if a.startswith('--')] year = None if '--year' in sys.argv[1:]: year = int(sys.argv[1:][sys.argv[1:].index('--year') + 1]) if len(args) < 2: print(__doc__) sys.exit(1) last, title = args[0], ' '.join(args[1:]) hits = crossref(last, title, year) if not hits: print("(no Crossref hits with DOI)") return for h in hits[:6]: line = (f"[{h['score']:.2f}] {h['title'][:70]} | {h['journal'][:35]} | " f"{h['author']} | {h['year']} | {h['doi']}") if '--no-libgen' not in flags: time.sleep(3) line += f" | LG: {libgen_doi(h['doi'])}" print(line) if __name__ == '__main__': main()