- tools/rutitles.py: EN->RU token translation (data/ru-dict.tsv, 540+ pairs, Kogito/Castalia
conventions) + Jaccard diff against data/ru-titles.jsonl (692 titles: Kogito 518, Litres 24,
OPP 55, flibusta a/5272+57639+118921+122883+193690+193689). Matching is lemmatized
(pymorphy3) — case endings handled: 'великой матери' -> 'Великая мать' 1.25.
'The Great Mother' -> 'Великая мать' 1.25 top hit; 'The Symbolic Quest' -> correct negative.
- tools/sweep.py: batch driver for Phases 1-2 (rsl+lg+alib+flib+cogito+SearXNG-OZON-snippet
parse+rutitles diff per item; --nlr optional). Query log data/queries.log (JSONL).
Fixed: cogito nav-menu leak (parse bx_product_item only), libgen robot-block (lg.py curl fallback).
- tools/oa.py: OpenAlex + Crossref (Phase 0 identity/ISBN, no key; found Margaret Wilkinson,
Karen Evers-Fahey, Symbolic Quest Princeton ISBN).
- data/sweeps/01-fundamentals/input.tsv: 16 ❌ items loaded; background sweep running.
- AGENTS.md: source 10b (oa.py), flibusta .su = reduced mirror (dropped from pipeline),
pipeline v3 section (rutitles/oa/sweep/pymorphy3 note: pymorphy2 broken on py3.12).
- data/ru-titles.jsonl committed as the RU market universe asset.
98 lines
3.7 KiB
Python
Executable file
98 lines
3.7 KiB
Python
Executable file
#!/usr/bin/env python3
|
|
"""OpenAlex + Crossref — Phase 0 identity/ISBN resolution (free, no key).
|
|
|
|
Usage:
|
|
oa.py "Exact EN Title" # OpenAlex work search (names, year, DOI, ISBN via locations)
|
|
oa.py --crossref "Title words" # Crossref bibliographic search (author, ISBN, publisher)
|
|
oa.py --author "Last, First" # OpenAlex author entity (ID, works count, cited)
|
|
oa.py both "Title" # both APIs in one go
|
|
|
|
OpenAlex: https://api.openalex.org/works?search=... (no key, 10 req/s pool)
|
|
Crossref: https://api.crossref.org/works?query.bibliographic=... (polite pool ok)
|
|
|
|
Notes:
|
|
- OpenAlex 'display_name' + authorships[].author.display_name = full EN names
|
|
(better than OL for modern academic books; OL still primary for pre-1990).
|
|
- Crossref ISBN list = direct EN ISBN candidates (13-digit, Routledge/Springer/Karnac).
|
|
- DNS here is flaky for python sockets in some contexts; use urllib (works).
|
|
"""
|
|
import json, sys, time, urllib.parse, urllib.request
|
|
|
|
UA = "jung-ru-editions-research/1.0 (mailto:dmitry@kokorin.org)"
|
|
|
|
def get(url):
|
|
req = urllib.request.Request(url, headers={"User-Agent": UA})
|
|
with urllib.request.urlopen(req, timeout=45) as r:
|
|
return json.loads(r.read().decode("utf-8", "replace"))
|
|
|
|
def openalex_works(q, n=5):
|
|
url = "https://api.openalex.org/works?search=%s&per-page=%d" % (urllib.parse.quote(q), n)
|
|
d = get(url)
|
|
out = []
|
|
for w in d.get("results", []):
|
|
auths = [a["author"]["display_name"] for a in w.get("authorships", [])]
|
|
isbns = set()
|
|
for l in w.get("locations", []):
|
|
if l.get("pdf") or l.get("landing_page_url"):
|
|
pass
|
|
# ISBNs often in biblio
|
|
bib = w.get("biblio") or {}
|
|
ids = w.get("ids", {})
|
|
doi = ids.get("doi")
|
|
out.append({
|
|
"title": w.get("display_name"),
|
|
"authors": auths,
|
|
"year": w.get("publication_year"),
|
|
"doi": doi,
|
|
"publisher": (w.get("primary_location") or {}).get("source", {}).get("display_name") if w.get("primary_location") else None,
|
|
"openalex_id": w.get("id"),
|
|
})
|
|
return out
|
|
|
|
def crossref(q, n=3):
|
|
url = "https://api.crossref.org/works?query.bibliographic=%s&rows=%d" % (urllib.parse.quote(q), n)
|
|
d = get(url)
|
|
out = []
|
|
for m in d.get("message", {}).get("items", []):
|
|
auths = [("%s %s" % (a.get("given", ""), a.get("family", ""))).strip() for a in m.get("author", [])]
|
|
out.append({
|
|
"title": (m.get("title") or ["?"])[0],
|
|
"authors": auths,
|
|
"isbn": m.get("ISBN") or [],
|
|
"issn": m.get("ISSN") or [],
|
|
"publisher": m.get("publisher"),
|
|
"year": (m.get("issued", {}).get("date-parts") or [[None]])[0][0],
|
|
"doi": m.get("DOI"),
|
|
})
|
|
return out
|
|
|
|
def openalex_author(q):
|
|
url = "https://api.openalex.org/authors?search=%s&per-page=5" % urllib.parse.quote(q)
|
|
d = get(url)
|
|
out = []
|
|
for a in d.get("results", []):
|
|
out.append({
|
|
"name": a.get("display_name"),
|
|
"id": a.get("id"),
|
|
"works": a.get("works_count"),
|
|
"cited": a.get("cited_by_count"),
|
|
})
|
|
return out
|
|
|
|
def main():
|
|
if len(sys.argv) < 3:
|
|
print(__doc__); sys.exit(1)
|
|
mode, q = sys.argv[1], " ".join(sys.argv[2:])
|
|
if mode == "both":
|
|
for r in openalex_works(q): print("OA ", r)
|
|
time.sleep(1)
|
|
for r in crossref(q): print("CR ", r)
|
|
elif mode == "--crossref":
|
|
for r in crossref(q): print(r)
|
|
elif mode == "--author":
|
|
for r in openalex_author(q): print(r)
|
|
else:
|
|
for r in openalex_works(q): print(r)
|
|
|
|
if __name__ == "__main__":
|
|
main()
|