EN-side research pillars: tools/ol.py (Open Library) + tools/wd.py (Wikidata), Google Books ISBN pattern, archive.org verification, HSE shop-directory + NikBook/Skifiya/beta2alpha sources, workflow v2 (full names first, reverse publisher sweep, community lists, content verification gate); full EN author names resolved for all 39 items (AUTHORS-EN.md); CONTEXT round 3 (Neumann cogito finds, 1959 disambiguation)
This commit is contained in:
parent
a9a0dda62a
commit
653f97f384
29 changed files with 281 additions and 34 deletions
49
tools/ol.py
Executable file
49
tools/ol.py
Executable file
|
|
@ -0,0 +1,49 @@
|
|||
#!/usr/bin/env python3
|
||||
"""Open Library search — the "English RSL" for author works lists.
|
||||
|
||||
Usage: tools/ol.py "Neumann, Erich" [title_words] [limit]
|
||||
|
||||
Output: one line per title: title | years | languages | sample publishers | sample ISBN.
|
||||
Open Library aggregates all editions of a work into one doc (author_name, publish_year[],
|
||||
language[], publisher[], isbn[]) — i.e. it answers "what did this EN author write" and
|
||||
"which languages/years exist" — the missing half of our RSL (RU-only) pipeline.
|
||||
NO RU editions are indexed (checked: language=rus → 0 for known RU books) — RU hunting
|
||||
still goes through RSL/libgen/flibusta.
|
||||
"""
|
||||
import json, sys, time, urllib.parse, urllib.request
|
||||
|
||||
UA = "research-tools/1.0 (jung booklist research; contact: dmitry@kokorin.org)"
|
||||
|
||||
def ol_search(q, fields="title,author_name,publish_year,language,publisher,isbn,original_title", limit=40):
|
||||
url = "https://openlibrary.org/search.json?" + urllib.parse.urlencode({
|
||||
"q": q, "fields": fields, "limit": min(limit, 100)})
|
||||
req = urllib.request.Request(url, headers={"User-Agent": UA})
|
||||
with urllib.request.urlopen(req, timeout=40) as r:
|
||||
return json.load(r)
|
||||
|
||||
def main():
|
||||
if len(sys.argv) < 2:
|
||||
sys.exit(__doc__)
|
||||
first = sys.argv[1]
|
||||
# raw query passthrough (e.g. 'title:"..."'); else treat as author name
|
||||
if first.startswith(("title:", "subject:", "isbn:", "publisher:")):
|
||||
q = " ".join(sys.argv[1:-1]) if sys.argv[-1].isdigit() else " ".join(sys.argv[1:])
|
||||
else:
|
||||
rest = " ".join(sys.argv[2:-1]) if sys.argv[-1].isdigit() else " ".join(sys.argv[2:])
|
||||
q = f"author:{first}" + (f" title:{rest}" if rest else "")
|
||||
limit = int(sys.argv[-1]) if sys.argv[-1].isdigit() else 40
|
||||
d = ol_search(q, limit=limit)
|
||||
print(f"numFound: {d.get('numFound')}")
|
||||
for doc in d.get("docs", []):
|
||||
years = sorted(set(doc.get("publish_year") or []))
|
||||
ys = f"{years[0]}-{years[-1]}" if len(years) > 1 else (str(years[0]) if years else "?")
|
||||
langs = ",".join(doc.get("language") or [])
|
||||
pubs = sorted(set(p[:30] for p in (doc.get("publisher") or [])))[:3]
|
||||
isbns = (doc.get("isbn") or [])[:2]
|
||||
ot = doc.get("original_title") or ""
|
||||
an = ", ".join(doc.get("author_name") or [])
|
||||
print(f"- {doc.get('title')} | {an} | {ys} | {langs} | {'; '.join(pubs)} | {isbns} | {doc.get('key','')} | {ot}")
|
||||
time.sleep(1)
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
51
tools/wd.py
Executable file
51
tools/wd.py
Executable file
|
|
@ -0,0 +1,51 @@
|
|||
#!/usr/bin/env python3
|
||||
"""Wikidata identity check — EN author -> life dates + OFFICIAL RU name.
|
||||
|
||||
Usage: tools/wd.py "Erich Neumann" [top]
|
||||
|
||||
Resolves the Q-entity, prints birth/death (P569/P570), EN/DE/RU labels, RU aliases
|
||||
(P725-equivalent via aliases block), and the P50 works list (often sparse).
|
||||
Primary value: (1) disambiguation via life dates, (2) the canonical RU label — the
|
||||
single most reliable RU spelling for RSL/libgen queries.
|
||||
"""
|
||||
import json, sys, urllib.parse, urllib.request
|
||||
|
||||
UA = "research-tools/1.0 (jung booklist research; contact: dmitry@kokorin.org)"
|
||||
|
||||
def api(params):
|
||||
url = "https://www.wikidata.org/w/api.php?" + urllib.parse.urlencode(params)
|
||||
req = urllib.request.Request(url, headers={"User-Agent": UA})
|
||||
with urllib.request.urlopen(req, timeout=40) as r:
|
||||
return json.load(r)
|
||||
|
||||
def main():
|
||||
if len(sys.argv) < 2:
|
||||
sys.exit(__doc__)
|
||||
name = " ".join(sys.argv[1:-1]) if len(sys.argv) > 2 and sys.argv[-1].isdigit() else " ".join(sys.argv[1:])
|
||||
top = int(sys.argv[-1]) if len(sys.argv) > 2 and sys.argv[-1].isdigit() else 3
|
||||
d = api({"action": "wbsearchentities", "search": name, "language": "en",
|
||||
"limit": top, "format": "json"})
|
||||
for r in d.get("search", []):
|
||||
qid = r["id"]
|
||||
label = r.get("label")
|
||||
desc = (r.get("description") or "")[:100]
|
||||
try:
|
||||
ed = api({"action": "wbgetentities", "ids": qid, "format": "json"})
|
||||
ent = ed["entities"][qid]
|
||||
c = ent.get("claims", {})
|
||||
bd = c.get("P569", [{}])[0].get("mainsnak", {}).get("datavalue", {}).get("value", {}).get("time", "?")
|
||||
dd = c.get("P570", [{}])[0].get("mainsnak", {}).get("datavalue", {}).get("value", {}).get("time", "?")
|
||||
labels = {k: v["value"] for k, v in ent.get("labels", {}).items() if k in ("en", "de", "ru")}
|
||||
ru_al = [a["value"] for a in ent.get("aliases", {}).get("ru", [])]
|
||||
works = [w["mainsnak"]["datavalue"]["value"].get("id") for w in c.get("P50", [])]
|
||||
print(f"{qid} | {label} | {desc}")
|
||||
print(f" born/died: {bd} / {dd}")
|
||||
print(f" labels: {labels}")
|
||||
if ru_al: print(f" RU aliases: {ru_al}")
|
||||
if works: print(f" P50 works: {works[:15]}")
|
||||
except Exception as e:
|
||||
print(f"{qid} | {label} | (entity fetch failed: {e})")
|
||||
print()
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Loading…
Add table
Add a link
Reference in a new issue