jung/tools/oa.py

98 lines
3.6 KiB
Python
Executable file

#!/usr/bin/env python3
"""OpenAlex + Crossref — Phase 0 identity/ISBN resolution (free, no key).
Usage:
oa.py "Exact EN Title" # OpenAlex work search (names, year, DOI, ISBN via locations)
oa.py --crossref "Title words" # Crossref bibliographic search (author, ISBN, publisher)
oa.py --author "Last, First" # OpenAlex author entity (ID, works count, cited)
oa.py both "Title" # both APIs in one go
OpenAlex: https://api.openalex.org/works?search=... (no key, 10 req/s pool)
Crossref: https://api.crossref.org/works?query.bibliographic=... (polite pool ok)
Notes:
- OpenAlex 'display_name' + authorships[].author.display_name = full EN names
(better than OL for modern academic books; OL still primary for pre-1990).
- Crossref ISBN list = direct EN ISBN candidates (13-digit, Routledge/Springer/Karnac).
- DNS here is flaky for python sockets in some contexts; use urllib (works).
"""
import json, sys, time, urllib.parse, urllib.request
UA = "jung-ru-editions-research/1.0 (mailto:dmitry@kokorin.org)"
def get(url):
req = urllib.request.Request(url, headers={"User-Agent": UA})
with urllib.request.urlopen(req, timeout=45) as r:
return json.loads(r.read().decode("utf-8", "replace"))
def openalex_works(q, n=5):
url = "https://api.openalex.org/works?search=%s&per-page=%d" % (urllib.parse.quote(q), n)
d = get(url)
out = []
for w in d.get("results", []):
auths = [a["author"]["display_name"] for a in w.get("authorships", [])]
isbns = set()
for l in w.get("locations", []):
if l.get("pdf") or l.get("landing_page_url"):
pass
# ISBNs often in biblio
bib = w.get("biblio") or {}
ids = w.get("ids", {})
doi = ids.get("doi")
out.append({
"title": w.get("display_name"),
"authors": auths,
"year": w.get("publication_year"),
"doi": doi,
"publisher": (((w.get("primary_location") or {}).get("source")) or {}).get("display_name"),
"openalex_id": w.get("id"),
})
return out
def crossref(q, n=3):
url = "https://api.crossref.org/works?query.bibliographic=%s&rows=%d" % (urllib.parse.quote(q), n)
d = get(url)
out = []
for m in d.get("message", {}).get("items", []):
auths = [("%s %s" % (a.get("given", ""), a.get("family", ""))).strip() for a in m.get("author", [])]
out.append({
"title": (m.get("title") or ["?"])[0],
"authors": auths,
"isbn": m.get("ISBN") or [],
"issn": m.get("ISSN") or [],
"publisher": m.get("publisher"),
"year": (m.get("issued", {}).get("date-parts") or [[None]])[0][0],
"doi": m.get("DOI"),
})
return out
def openalex_author(q):
url = "https://api.openalex.org/authors?search=%s&per-page=5" % urllib.parse.quote(q)
d = get(url)
out = []
for a in d.get("results", []):
out.append({
"name": a.get("display_name"),
"id": a.get("id"),
"works": a.get("works_count"),
"cited": a.get("cited_by_count"),
})
return out
def main():
if len(sys.argv) < 3:
print(__doc__); sys.exit(1)
mode, q = sys.argv[1], " ".join(sys.argv[2:])
if mode == "both":
for r in openalex_works(q): print("OA ", r)
time.sleep(1)
for r in crossref(q): print("CR ", r)
elif mode == "--crossref":
for r in crossref(q): print(r)
elif mode == "--author":
for r in openalex_author(q): print(r)
else:
for r in openalex_works(q): print(r)
if __name__ == "__main__":
main()