sec07: rescue wave 2 — CW3 RU (АСТ 2025 via flib .su) + 7 EN files (file-level libgen + DOI) + tools/doi.py
- CW3 (#02) = ✅: «Психогенез душевных болезней» АСТ 2025 (flibusta.su b/390112; .su ≠ .is по контенту; калка «психических» вела в тупик; найден браузером пользователя) - EN rescues (file-level libgen objects[]=f, подсказки пользователя): CW2 1981 pdf 662pp + 1973 epub, CW3 1960 pdf 325pp + 1982 epub, Bovensiepen JAP 2006 (ed 15358298), Krieger JAP 2019 (ed 84462696 — ссылка пользователя), Meier I JAP 2019 (ed 84462697) — все контент-верифицированы (статьи: автор+журнал+abstract = RAW, не рецензии) - tools/doi.py: Crossref author+title → DOI → libgen req=DOI (контроли 3/3); DOI Meier I верифицирован = 10.1111/1468-5922.12545 - CW2 RU: доп. свип (ast.ru server-side search, litres, Yandex Books, Neoclassic) = 0 - AGENTS.md: sources 10c (DOI pipeline) + 10d (flib .su); status line - итог sec07: 4 ✅ / 23 ❌ RU, 20 EN files (87 MB), linter 0, xref stale 0
This commit is contained in:
parent
08f885b172
commit
188e00d2fb
19 changed files with 295 additions and 95 deletions
90
tools/doi.py
Executable file
90
tools/doi.py
Executable file
|
|
@ -0,0 +1,90 @@
|
|||
#!/usr/bin/env python3
|
||||
"""doi.py — article-item pipeline: author+title → Crossref DOI → libgen availability.
|
||||
|
||||
For reading-list items that are JOURNAL ARTICLES (not books), the DOI is the most
|
||||
reliable identifier: Crossref/OpenAlex resolve it from author+title, and libgen.vg
|
||||
indexes many papers BY DOI (req=<doi> — verified 2026-09-26: Krieger
|
||||
10.1111/1468-5922.12544 → exactly ed 84462696).
|
||||
|
||||
usage: doi.py "Lastname" "Title words..." [--year YYYY] [--no-libgen]
|
||||
|
||||
Steps:
|
||||
1. Crossref /works?query.bibliographic=TITLE&query.author=LAST (rows=8).
|
||||
2. Rank: title token overlap + year proximity (if --year).
|
||||
3. For top hits with DOI: libgen req=<DOI> → count edition links (4s polite).
|
||||
4. Print candidates: title | journal | author | year | DOI | LG: n (ed ids).
|
||||
|
||||
Limits: pre-1997 works have no DOIs (1975 Hill paper = unresolvable — fall back to
|
||||
title sweeps). Conference papers in APA PsycEXTRA (10.1037/e*) are often noise.
|
||||
After a hit: content-verify the downloaded file (first page: author+journal+pages
|
||||
must match RAW; must be the paper, not a review of it).
|
||||
"""
|
||||
import json, re, sys, time, urllib.parse, urllib.request
|
||||
|
||||
UA = {'User-Agent': 'jung-research/1.0 (mailto:dmitry@kokorin.org)'}
|
||||
LG_UA = {'User-Agent': 'Mozilla/5.0 (X11; Linux x86_64) Firefox/130.0'}
|
||||
|
||||
|
||||
def crossref(last, title, year=None):
|
||||
q = {'query.bibliographic': title, 'query.author': last, 'rows': 8}
|
||||
if year:
|
||||
q['filter'] = f'from-pub-date:{year-2},until-pub-date:{year+2}'
|
||||
url = f"https://api.crossref.org/works?{urllib.parse.urlencode(q)}"
|
||||
req = urllib.request.Request(url, headers=UA)
|
||||
d = json.load(urllib.request.urlopen(req, timeout=40))
|
||||
out = []
|
||||
for it in d['message']['items']:
|
||||
doi = it.get('DOI')
|
||||
if not doi:
|
||||
continue
|
||||
a = it.get('author', [{}])[0]
|
||||
fam = a.get('family', '?')
|
||||
yr = (it.get('issued', {}).get('date-parts') or [['?']])[0][0]
|
||||
src = (it.get('container-title') or [''])[0]
|
||||
t = it.get('title', ['?'])[0]
|
||||
# token overlap score
|
||||
want = set(re.findall(r'\w+', title.lower()))
|
||||
got = set(re.findall(r'\w+', (t + ' ' + src).lower()))
|
||||
score = len(want & got) / max(1, len(want))
|
||||
out.append({'title': re.sub(r'<[^>]+>', '', t), 'journal': src, 'author': fam,
|
||||
'year': yr, 'doi': doi, 'score': score})
|
||||
out.sort(key=lambda x: (-x['score'], abs((x['year'] or 0) - (year or 0)) if year else 0))
|
||||
return out
|
||||
|
||||
|
||||
def libgen_doi(doi):
|
||||
url = f"https://libgen.vg/index.php?req={urllib.parse.quote(doi)}&res=25"
|
||||
req = urllib.request.Request(url, headers=LG_UA)
|
||||
try:
|
||||
h = urllib.request.urlopen(req, timeout=40).read().decode('utf-8', 'replace')
|
||||
except Exception as e:
|
||||
return f"ERR {e}"
|
||||
ids = sorted(set(re.findall(r'edition\.php\?id=(\d+)', h)))
|
||||
return f"{len(ids)} ({', '.join(ids[:5])})" if ids else "0"
|
||||
|
||||
|
||||
def main():
|
||||
args = [a for a in sys.argv[1:] if not a.startswith('--')]
|
||||
flags = [a for a in sys.argv[1:] if a.startswith('--')]
|
||||
year = None
|
||||
if '--year' in sys.argv[1:]:
|
||||
year = int(sys.argv[1:][sys.argv[1:].index('--year') + 1])
|
||||
if len(args) < 2:
|
||||
print(__doc__)
|
||||
sys.exit(1)
|
||||
last, title = args[0], ' '.join(args[1:])
|
||||
hits = crossref(last, title, year)
|
||||
if not hits:
|
||||
print("(no Crossref hits with DOI)")
|
||||
return
|
||||
for h in hits[:6]:
|
||||
line = (f"[{h['score']:.2f}] {h['title'][:70]} | {h['journal'][:35]} | "
|
||||
f"{h['author']} | {h['year']} | {h['doi']}")
|
||||
if '--no-libgen' not in flags:
|
||||
time.sleep(3)
|
||||
line += f" | LG: {libgen_doi(h['doi'])}"
|
||||
print(line)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
Loading…
Add table
Add a link
Reference in a new issue