#!/usr/bin/env python3 """OpenAlex + Crossref — Phase 0 identity/ISBN resolution (free, no key). Usage: oa.py "Exact EN Title" # OpenAlex work search (names, year, DOI, ISBN via locations) oa.py --crossref "Title words" # Crossref bibliographic search (author, ISBN, publisher) oa.py --author "Last, First" # OpenAlex author entity (ID, works count, cited) oa.py both "Title" # both APIs in one go OpenAlex: https://api.openalex.org/works?search=... (no key, 10 req/s pool) Crossref: https://api.crossref.org/works?query.bibliographic=... (polite pool ok) Notes: - OpenAlex 'display_name' + authorships[].author.display_name = full EN names (better than OL for modern academic books; OL still primary for pre-1990). - Crossref ISBN list = direct EN ISBN candidates (13-digit, Routledge/Springer/Karnac). - DNS here is flaky for python sockets in some contexts; use urllib (works). """ import json, sys, time, urllib.parse, urllib.request UA = "jung-ru-editions-research/1.0 (mailto:dmitry@kokorin.org)" def get(url): req = urllib.request.Request(url, headers={"User-Agent": UA}) with urllib.request.urlopen(req, timeout=45) as r: return json.loads(r.read().decode("utf-8", "replace")) def openalex_works(q, n=5): url = "https://api.openalex.org/works?search=%s&per-page=%d" % (urllib.parse.quote(q), n) d = get(url) out = [] for w in d.get("results", []): auths = [a["author"]["display_name"] for a in w.get("authorships", [])] isbns = set() for l in w.get("locations", []): if l.get("pdf") or l.get("landing_page_url"): pass # ISBNs often in biblio bib = w.get("biblio") or {} ids = w.get("ids", {}) doi = ids.get("doi") out.append({ "title": w.get("display_name"), "authors": auths, "year": w.get("publication_year"), "doi": doi, "publisher": (((w.get("primary_location") or {}).get("source")) or {}).get("display_name"), "openalex_id": w.get("id"), }) return out def crossref(q, n=3): url = "https://api.crossref.org/works?query.bibliographic=%s&rows=%d" % (urllib.parse.quote(q), n) d = get(url) out = [] for m in d.get("message", {}).get("items", []): auths = [("%s %s" % (a.get("given", ""), a.get("family", ""))).strip() for a in m.get("author", [])] out.append({ "title": (m.get("title") or ["?"])[0], "authors": auths, "isbn": m.get("ISBN") or [], "issn": m.get("ISSN") or [], "publisher": m.get("publisher"), "year": (m.get("issued", {}).get("date-parts") or [[None]])[0][0], "doi": m.get("DOI"), }) return out def openalex_author(q): url = "https://api.openalex.org/authors?search=%s&per-page=5" % urllib.parse.quote(q) d = get(url) out = [] for a in d.get("results", []): out.append({ "name": a.get("display_name"), "id": a.get("id"), "works": a.get("works_count"), "cited": a.get("cited_by_count"), }) return out def main(): if len(sys.argv) < 3: print(__doc__); sys.exit(1) mode, q = sys.argv[1], " ".join(sys.argv[2:]) if mode == "both": for r in openalex_works(q): print("OA ", r) time.sleep(1) for r in crossref(q): print("CR ", r) elif mode == "--crossref": for r in crossref(q): print(r) elif mode == "--author": for r in openalex_author(q): print(r) else: for r in openalex_works(q): print(r) if __name__ == "__main__": main()