EN-side research pillars: tools/ol.py (Open Library) + tools/wd.py (Wikidata), Google Books ISBN pattern, archive.org verification, HSE shop-directory + NikBook/Skifiya/beta2alpha sources, workflow v2 (full names first, reverse publisher sweep, community lists, content verification gate); full EN author names resolved for all 39 items (AUTHORS-EN.md); CONTEXT round 3 (Neumann cogito finds, 1959 disambiguation)
This commit is contained in:
parent
77d39ea000
commit
c0a3ac0f2f
29 changed files with 281 additions and 34 deletions
49
tools/ol.py
Executable file
49
tools/ol.py
Executable file
|
|
@ -0,0 +1,49 @@
|
|||
#!/usr/bin/env python3
|
||||
"""Open Library search — the "English RSL" for author works lists.
|
||||
|
||||
Usage: tools/ol.py "Neumann, Erich" [title_words] [limit]
|
||||
|
||||
Output: one line per title: title | years | languages | sample publishers | sample ISBN.
|
||||
Open Library aggregates all editions of a work into one doc (author_name, publish_year[],
|
||||
language[], publisher[], isbn[]) — i.e. it answers "what did this EN author write" and
|
||||
"which languages/years exist" — the missing half of our RSL (RU-only) pipeline.
|
||||
NO RU editions are indexed (checked: language=rus → 0 for known RU books) — RU hunting
|
||||
still goes through RSL/libgen/flibusta.
|
||||
"""
|
||||
import json, sys, time, urllib.parse, urllib.request
|
||||
|
||||
UA = "research-tools/1.0 (jung booklist research; contact: dmitry@kokorin.org)"
|
||||
|
||||
def ol_search(q, fields="title,author_name,publish_year,language,publisher,isbn,original_title", limit=40):
|
||||
url = "https://openlibrary.org/search.json?" + urllib.parse.urlencode({
|
||||
"q": q, "fields": fields, "limit": min(limit, 100)})
|
||||
req = urllib.request.Request(url, headers={"User-Agent": UA})
|
||||
with urllib.request.urlopen(req, timeout=40) as r:
|
||||
return json.load(r)
|
||||
|
||||
def main():
|
||||
if len(sys.argv) < 2:
|
||||
sys.exit(__doc__)
|
||||
first = sys.argv[1]
|
||||
# raw query passthrough (e.g. 'title:"..."'); else treat as author name
|
||||
if first.startswith(("title:", "subject:", "isbn:", "publisher:")):
|
||||
q = " ".join(sys.argv[1:-1]) if sys.argv[-1].isdigit() else " ".join(sys.argv[1:])
|
||||
else:
|
||||
rest = " ".join(sys.argv[2:-1]) if sys.argv[-1].isdigit() else " ".join(sys.argv[2:])
|
||||
q = f"author:{first}" + (f" title:{rest}" if rest else "")
|
||||
limit = int(sys.argv[-1]) if sys.argv[-1].isdigit() else 40
|
||||
d = ol_search(q, limit=limit)
|
||||
print(f"numFound: {d.get('numFound')}")
|
||||
for doc in d.get("docs", []):
|
||||
years = sorted(set(doc.get("publish_year") or []))
|
||||
ys = f"{years[0]}-{years[-1]}" if len(years) > 1 else (str(years[0]) if years else "?")
|
||||
langs = ",".join(doc.get("language") or [])
|
||||
pubs = sorted(set(p[:30] for p in (doc.get("publisher") or [])))[:3]
|
||||
isbns = (doc.get("isbn") or [])[:2]
|
||||
ot = doc.get("original_title") or ""
|
||||
an = ", ".join(doc.get("author_name") or [])
|
||||
print(f"- {doc.get('title')} | {an} | {ys} | {langs} | {'; '.join(pubs)} | {isbns} | {doc.get('key','')} | {ot}")
|
||||
time.sleep(1)
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Loading…
Add table
Add a link
Reference in a new issue