EN-side research pillars: tools/ol.py (Open Library) + tools/wd.py (Wikidata), Google Books ISBN pattern, archive.org verification, HSE shop-directory + NikBook/Skifiya/beta2alpha sources, workflow v2 (full names first, reverse publisher sweep, community lists, content verification gate); full EN author names resolved for all 39 items (AUTHORS-EN.md); CONTEXT round 3 (Neumann cogito finds, 1959 disambiguation)

This commit is contained in:
Dmitry Kokorin 2026-09-09 00:10:25 +03:00
parent 77d39ea000
commit c0a3ac0f2f
29 changed files with 281 additions and 34 deletions

49
tools/ol.py Executable file
View file

@ -0,0 +1,49 @@
#!/usr/bin/env python3
"""Open Library search — the "English RSL" for author works lists.
Usage: tools/ol.py "Neumann, Erich" [title_words] [limit]
Output: one line per title: title | years | languages | sample publishers | sample ISBN.
Open Library aggregates all editions of a work into one doc (author_name, publish_year[],
language[], publisher[], isbn[]) — i.e. it answers "what did this EN author write" and
"which languages/years exist" — the missing half of our RSL (RU-only) pipeline.
NO RU editions are indexed (checked: language=rus → 0 for known RU books) — RU hunting
still goes through RSL/libgen/flibusta.
"""
import json, sys, time, urllib.parse, urllib.request
UA = "research-tools/1.0 (jung booklist research; contact: dmitry@kokorin.org)"
def ol_search(q, fields="title,author_name,publish_year,language,publisher,isbn,original_title", limit=40):
url = "https://openlibrary.org/search.json?" + urllib.parse.urlencode({
"q": q, "fields": fields, "limit": min(limit, 100)})
req = urllib.request.Request(url, headers={"User-Agent": UA})
with urllib.request.urlopen(req, timeout=40) as r:
return json.load(r)
def main():
if len(sys.argv) < 2:
sys.exit(__doc__)
first = sys.argv[1]
# raw query passthrough (e.g. 'title:"..."'); else treat as author name
if first.startswith(("title:", "subject:", "isbn:", "publisher:")):
q = " ".join(sys.argv[1:-1]) if sys.argv[-1].isdigit() else " ".join(sys.argv[1:])
else:
rest = " ".join(sys.argv[2:-1]) if sys.argv[-1].isdigit() else " ".join(sys.argv[2:])
q = f"author:{first}" + (f" title:{rest}" if rest else "")
limit = int(sys.argv[-1]) if sys.argv[-1].isdigit() else 40
d = ol_search(q, limit=limit)
print(f"numFound: {d.get('numFound')}")
for doc in d.get("docs", []):
years = sorted(set(doc.get("publish_year") or []))
ys = f"{years[0]}-{years[-1]}" if len(years) > 1 else (str(years[0]) if years else "?")
langs = ",".join(doc.get("language") or [])
pubs = sorted(set(p[:30] for p in (doc.get("publisher") or [])))[:3]
isbns = (doc.get("isbn") or [])[:2]
ot = doc.get("original_title") or ""
an = ", ".join(doc.get("author_name") or [])
print(f"- {doc.get('title')} | {an} | {ys} | {langs} | {'; '.join(pubs)} | {isbns} | {doc.get('key','')} | {ot}")
time.sleep(1)
if __name__ == "__main__":
main()