pipeline v3: RU-title candidates (rutitles+ru-dict+pymorphy3), batch sweep driver, OpenAlex/Crossref

- tools/rutitles.py: EN->RU token translation (data/ru-dict.tsv, 540+ pairs, Kogito/Castalia
  conventions) + Jaccard diff against data/ru-titles.jsonl (692 titles: Kogito 518, Litres 24,
  OPP 55, flibusta a/5272+57639+118921+122883+193690+193689). Matching is lemmatized
  (pymorphy3) — case endings handled: 'великой матери' -> 'Великая мать' 1.25.
  'The Great Mother' -> 'Великая мать' 1.25 top hit; 'The Symbolic Quest' -> correct negative.
- tools/sweep.py: batch driver for Phases 1-2 (rsl+lg+alib+flib+cogito+SearXNG-OZON-snippet
  parse+rutitles diff per item; --nlr optional). Query log data/queries.log (JSONL).
  Fixed: cogito nav-menu leak (parse bx_product_item only), libgen robot-block (lg.py curl fallback).
- tools/oa.py: OpenAlex + Crossref (Phase 0 identity/ISBN, no key; found Margaret Wilkinson,
  Karen Evers-Fahey, Symbolic Quest Princeton ISBN).
- data/sweeps/01-fundamentals/input.tsv: 16 ❌ items loaded; background sweep running.
- AGENTS.md: source 10b (oa.py), flibusta .su = reduced mirror (dropped from pipeline),
  pipeline v3 section (rutitles/oa/sweep/pymorphy3 note: pymorphy2 broken on py3.12).
- data/ru-titles.jsonl committed as the RU market universe asset.
This commit is contained in:
Dmitry Kokorin 2026-09-18 10:56:36 +03:00
parent 2157d058df
commit 5441dddbb8
11 changed files with 2101 additions and 3 deletions

View file

@ -10,10 +10,24 @@ def fetch(url, tries=3):
for i in range(tries):
try:
req = urllib.request.Request(url, headers={"User-Agent": UA, "Accept-Language": "ru-RU,ru;q=0.9"})
return urllib.request.urlopen(req, timeout=90).read().decode("utf-8", "replace")
html = urllib.request.urlopen(req, timeout=90).read().decode("utf-8", "replace")
# robot-block detection: nginx default page has no req= result table
if 'libgen' not in html.lower()[:3000] and 'Search Result' not in html and 'result' not in html.lower()[:3000]:
time.sleep(15)
continue
return html
except Exception as e:
last = e
time.sleep(5 + 5 * i)
# curl fallback (works when urllib is fingerprinted/blocked)
import subprocess
try:
r = subprocess.run(["curl", "-s", "-A", UA, "--max-time", "90", url],
capture_output=True, text=True, timeout=120)
if r.stdout:
return r.stdout
except Exception:
pass
raise SystemExit(f"fetch failed after {tries} tries: {last}")
def main():

98
tools/oa.py Executable file
View file

@ -0,0 +1,98 @@
#!/usr/bin/env python3
"""OpenAlex + Crossref — Phase 0 identity/ISBN resolution (free, no key).
Usage:
oa.py "Exact EN Title" # OpenAlex work search (names, year, DOI, ISBN via locations)
oa.py --crossref "Title words" # Crossref bibliographic search (author, ISBN, publisher)
oa.py --author "Last, First" # OpenAlex author entity (ID, works count, cited)
oa.py both "Title" # both APIs in one go
OpenAlex: https://api.openalex.org/works?search=... (no key, 10 req/s pool)
Crossref: https://api.crossref.org/works?query.bibliographic=... (polite pool ok)
Notes:
- OpenAlex 'display_name' + authorships[].author.display_name = full EN names
(better than OL for modern academic books; OL still primary for pre-1990).
- Crossref ISBN list = direct EN ISBN candidates (13-digit, Routledge/Springer/Karnac).
- DNS here is flaky for python sockets in some contexts; use urllib (works).
"""
import json, sys, time, urllib.parse, urllib.request
UA = "jung-ru-editions-research/1.0 (mailto:dmitry@kokorin.org)"
def get(url):
req = urllib.request.Request(url, headers={"User-Agent": UA})
with urllib.request.urlopen(req, timeout=45) as r:
return json.loads(r.read().decode("utf-8", "replace"))
def openalex_works(q, n=5):
url = "https://api.openalex.org/works?search=%s&per-page=%d" % (urllib.parse.quote(q), n)
d = get(url)
out = []
for w in d.get("results", []):
auths = [a["author"]["display_name"] for a in w.get("authorships", [])]
isbns = set()
for l in w.get("locations", []):
if l.get("pdf") or l.get("landing_page_url"):
pass
# ISBNs often in biblio
bib = w.get("biblio") or {}
ids = w.get("ids", {})
doi = ids.get("doi")
out.append({
"title": w.get("display_name"),
"authors": auths,
"year": w.get("publication_year"),
"doi": doi,
"publisher": (w.get("primary_location") or {}).get("source", {}).get("display_name") if w.get("primary_location") else None,
"openalex_id": w.get("id"),
})
return out
def crossref(q, n=3):
url = "https://api.crossref.org/works?query.bibliographic=%s&rows=%d" % (urllib.parse.quote(q), n)
d = get(url)
out = []
for m in d.get("message", {}).get("items", []):
auths = [("%s %s" % (a.get("given", ""), a.get("family", ""))).strip() for a in m.get("author", [])]
out.append({
"title": (m.get("title") or ["?"])[0],
"authors": auths,
"isbn": m.get("ISBN") or [],
"issn": m.get("ISSN") or [],
"publisher": m.get("publisher"),
"year": (m.get("issued", {}).get("date-parts") or [[None]])[0][0],
"doi": m.get("DOI"),
})
return out
def openalex_author(q):
url = "https://api.openalex.org/authors?search=%s&per-page=5" % urllib.parse.quote(q)
d = get(url)
out = []
for a in d.get("results", []):
out.append({
"name": a.get("display_name"),
"id": a.get("id"),
"works": a.get("works_count"),
"cited": a.get("cited_by_count"),
})
return out
def main():
if len(sys.argv) < 3:
print(__doc__); sys.exit(1)
mode, q = sys.argv[1], " ".join(sys.argv[2:])
if mode == "both":
for r in openalex_works(q): print("OA ", r)
time.sleep(1)
for r in crossref(q): print("CR ", r)
elif mode == "--crossref":
for r in crossref(q): print(r)
elif mode == "--author":
for r in openalex_author(q): print(r)
else:
for r in openalex_works(q): print(r)
if __name__ == "__main__":
main()

205
tools/rutitles.py Executable file
View file

@ -0,0 +1,205 @@
#!/usr/bin/env python3
"""RU-title candidates: EN->RU token translation + match against known RU title DB.
Usage:
rutitles.py build # rebuild data/ru-titles.jsonl from data/catalogs/*
rutitles.py translate "The Symbolic Quest" # EN title -> RU token bag (+unknown EN words)
rutitles.py match "символический поиск" [N] # top-N fuzzy matches in the RU title DB
rutitles.py both "The Symbolic Quest" # translate + match in one go
DB sources (data/catalogs/*.md, title lines after the header):
cogito Jungian series (518), litres 140409, OPP/MISP list.
Extend: `rutitles.py addflib <authorId>` appends flibusta.is authorall book titles.
Matching: token Jaccard over normalized lowercase tokens (stopwords removed).
Order-agnostic — RU word order irrelevant. This is a DIFF against the known
RU market universe, NOT a translation check — verify identity on real hits.
"""
import json, os, re, sys, difflib
try:
import pymorphy3
_MORPH = pymorphy3.MorphAnalyzer()
except Exception:
_MORPH = None
def lemma(w):
if _MORPH and w.isalpha() and "а" <= w[0] <= "я":
try:
return _MORPH.parse(w)[0].normal_form
except Exception:
return w
return w
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
DICT = os.path.join(ROOT, "data", "ru-dict.tsv")
DB = os.path.join(ROOT, "data", "ru-titles.jsonl")
CATS = os.path.join(ROOT, "data", "catalogs")
STOP = set("""в вв во не на наа наа не при по от из до к к к к у за для с со и и и или но а же ли бы то что это тот эта этот эта эти те они он она мы вы они ее его их наш твой мой ваш их их мой твой его ее их наш ваш свой чей какой какой какие какое какое каких каких какие какие какой какой какой какой какая какое какие сколько сколько многие несколько весь все всё весь все всё весь все всё каждый каждый каждая каждое какие любой любая любое любые любой любая любое любые некий некая некое некие иной иная иное иные иной иная иное иные чужой чужая чужое чужие другой другая другое другие другой другая другое другие сам сама само сами сами сами сам сама само сами сам сам сама само сами""".split())
STOP = set(w for w in STOP if len(w) > 1)
STOP |= {"в", "на", "во", "о", "об", "с", "со", "из", "от", "по", "к", "у", "за", "и", "а", "но",
"не", "что", "который", "которая", "которое", "которые", "для", "про", "же", "ли"}
def load_dict():
phrases, words = [], {}
for line in open(DICT, encoding="utf-8"):
line = line.rstrip("\n")
if not line or line.startswith("#") or "\t" not in line:
continue
k, v = line.split("\t", 1)
k, v = k.strip(), v.strip()
if " " in k:
phrases.append((k.lower(), v))
else:
words[k.lower()] = v
phrases.sort(key=lambda kv: -len(kv[0]))
return phrases, words
def norm_tokens(s):
s = s.lower()
s = re.sub(r"[«»\"'’\-–—().,;:!?/\\|]", " ", s)
toks = [t for t in s.split() if t]
return [lemma(t) for t in toks if t not in STOP]
PHR, WORD = load_dict()
def translate(title):
"""EN title -> (ru_token_bag, [unknown_en_words])"""
text = title.lower().strip()
text = re.sub(r"[«»\"'’\-–—().,;:!?/\\|]", " ", text)
tokens, unknown = [], []
# phrases first (longest)
for ph, ru in PHR:
pat = re.compile(r"(?<![a-z])" + re.escape(ph) + r"(?![a-z])")
m = pat.search(text)
if m:
tokens.append(ru)
text = text[:m.start()] + " " * (m.end() - m.start()) + text[m.end():]
for w in text.split():
w = w.strip()
if not w or w in STOP or not w.isalpha():
continue
if w in WORD:
tokens.append(WORD[w])
elif re.fullmatch(r"[a-z]+", w):
unknown.append(w)
# expand variants for matching: keep all
return tokens, unknown
def load_db():
if not os.path.exists(DB):
return []
return [json.loads(l) for l in open(DB, encoding="utf-8") if l.strip()]
def build():
rows, seen = [], set()
# preserve non-catalog entries (flibusta author lists added via addflib)
if os.path.exists(DB):
for l in open(DB, encoding="utf-8"):
if l.strip():
r = json.loads(l)
if r["src"].startswith("flib:"):
rows.append(r)
seen.add(r["norm"])
for fn in sorted(os.listdir(CATS)):
if not fn.endswith((".md", ".txt")):
continue
src = os.path.basename(fn)
for line in open(os.path.join(CATS, fn), encoding="utf-8", errors="replace"):
line = line.strip()
if not line or line.startswith(("#", "-", "Источник", "Дата", "##")):
continue
if len(line) < 8:
continue
key = " ".join(norm_tokens(line))
if key in seen:
continue
seen.add(key)
rows.append({"title": line, "norm": key, "src": src})
with open(DB, "w", encoding="utf-8") as f:
for r in rows:
f.write(json.dumps(r, ensure_ascii=False) + "\n")
print("DB built: %d titles from %d catalogs" % (len(rows), len(os.listdir(CATS))))
def match(needle, n=10):
nt = set(norm_tokens(needle))
if not nt:
print("empty needle"); return
rows = load_db()
scored = []
for r in rows:
rt = set(r["norm"].split())
if not rt:
continue
inter = nt & rt
union = nt | rt
j = len(inter) / len(union) if union else 0
# boost: all needle tokens present
if nt <= rt:
j += 0.25
if j >= 0.25:
scored.append((j, r))
scored.sort(key=lambda x: -x[0])
for j, r in scored[:n]:
print("%.2f [%s] %s" % (j, r["src"][:12], r["title"][:100]))
def both(title, n=10):
tokens, unknown = translate(title)
print("RU tokens:", " | ".join(tokens))
if unknown:
print("UNTRANSLATED:", " ".join(unknown))
if tokens:
print("--- matches ---")
match(" ".join(tokens), n)
def addflib(author_id):
import subprocess
out = subprocess.run([sys.executable, os.path.join(ROOT, "tools", "flib.py"), "authorall", str(author_id)],
capture_output=True, text=True, timeout=600).stdout
rows = [json.loads(l) for l in open(DB, encoding="utf-8") if l.strip()] if os.path.exists(DB) else []
seen = set(r["norm"] for r in rows)
added = 0
for line in out.splitlines():
# flib.py authorall: "b/<id> <year> <fmt> <author> — <title>[ [пер. X]]"
m = re.match(r"^b/(\d+)\s+(\d+|\?)\s+\S+\s+([^—]+?)\s*—\s*(.+)$", line)
if not m:
continue
title = re.sub(r"\s*\[пер\. .+\]$", "", m.group(4)).strip()
key = " ".join(norm_tokens(title))
if not key or key in seen:
continue
seen.add(key)
rows.append({"title": title, "norm": key, "src": "flib:a%s" % author_id})
added += 1
with open(DB, "w", encoding="utf-8") as f:
for r in rows:
f.write(json.dumps(r, ensure_ascii=False) + "\n")
print("flib a/%s: +%d titles (DB now %d)" % (author_id, added, len(rows)))
def main():
if len(sys.argv) < 2:
print(__doc__); sys.exit(1)
cmd = sys.argv[1]
if cmd == "build":
build()
elif cmd == "translate":
tokens, unknown = translate(" ".join(sys.argv[2:]))
print(" | ".join(tokens))
if unknown:
print("UNTRANSLATED:", " ".join(unknown))
elif cmd == "match":
args = sys.argv[2:]
n = 10
if args and args[-1].isdigit():
n, args = int(args[-1]), args[:-1]
match(" ".join(args), n)
elif cmd == "both":
both(" ".join(sys.argv[2:]), int(sys.argv[3]) if len(sys.argv) > 3 else 10)
elif cmd == "addflib":
addflib(sys.argv[2])
else:
print(__doc__); sys.exit(1)
if __name__ == "__main__":
main()

205
tools/sweep.py Normal file
View file

@ -0,0 +1,205 @@
#!/usr/bin/env python3
"""sweep.py — batch driver for pipeline Phases 1-2 (mechanical part).
Input: data/sweeps/<sec>/input.tsv with columns (tab-separated):
nn slug en_title ru_surnames('|'-separated variants) ru_title_candidates('|'-separated)
(e.g. 30 the-symbolic-quest The Symbolic Quest Уайтмонт|Уитмонт символический поиск|символическая погоня)
Usage:
sweep.py 01 [--nlr] [--only 30 37 46] [--skip-existing]
Reads sections/01-fundamentals/INDEX.md statuses; by default sweeps items NOT marked ✅.
--only restricts to given nn's; --skip-existing skips items with an output file.
Per item (polite delays): rsl, lg, alib(surname + title), flib books, cogito search,
SearXNG (with OZON snippet parse: year/pages/ISBN), rutitles diff.
--nlr adds НРБ (slow, CRW-rendered, ~15s/query).
Query log: data/queries.log (JSONL). Output: data/sweeps/<sec>/<nn>-<slug>.md.
"""
import json, os, re, sys, time, subprocess, urllib.parse, urllib.request
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
UA = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"}
SX = "http://localhost:8888/search"
LOG = os.path.join(ROOT, "data", "queries.log")
def log(source, item, query, hits, summary=""):
rec = {"ts": time.strftime("%Y-%m-%dT%H:%M:%S"), "item": item, "src": source,
"q": query, "hits": hits, "sum": summary[:200]}
with open(LOG, "a", encoding="utf-8") as f:
f.write(json.dumps(rec, ensure_ascii=False) + "\n")
def run(cmd, timeout=180):
try:
r = subprocess.run(cmd, capture_output=True, text=True, timeout=timeout)
return (r.stdout or "") + (r.stderr or "")
except Exception as e:
return "(error: %s)" % e
def sh_run(*cmd, timeout=180):
return run(list(cmd), timeout)
def rsl(q):
out = sh_run(sys.executable, os.path.join(ROOT, "tools", "rsl.py"), q)
hits = len(re.findall(r"^\[\d+\]", out, re.M))
return out, hits
def lg(q):
out = sh_run(sys.executable, os.path.join(ROOT, "tools", "lg.py"), q)
hits = out.count("libgen.vg/edition.php?id=")
return out, hits
def alib(q):
out = sh_run(sys.executable, os.path.join(ROOT, "tools", "alib.py"), q)
hits = out.count("Купить") + len(re.findall(r"\d+\.(?:\d+)?\s+руб", out))
return out, hits
def flib(q):
out = sh_run(sys.executable, os.path.join(ROOT, "tools", "flib.py"), "books", q)
hits = len(re.findall(r"^b/\d+", out, re.M))
return out, hits
def cogito(q):
url = "https://cogito-shop.com/search/?q=" + urllib.parse.quote(q)
out = run(["curl", "-sL", "-A", UA["User-Agent"], "--max-time", "60", url])
# product cards only (bx_product_item divs); the first bx_product_list block is the nav menu
items = re.split(r'<div class="bx_product_item[^"]*"', out)
prods = []
for blk in items[1:]:
m = re.search(r'<a[^>]*href="(/catalog/[^"]+)"[^>]*>(?:\s|<[^>]+>)*([^<]{15,120})', blk[:6000])
if m and m.group(2).strip():
prods.append(m.group(2).strip())
seen, lines = set(), []
for x in prods:
if x not in seen:
seen.add(x); lines.append(x)
return "\n".join(lines), len(lines)
def sx_ozon(q):
url = SX + "?q=" + urllib.parse.quote(q) + "&format=json&language=ru"
try:
req = urllib.request.Request(url, headers=UA)
d = json.loads(urllib.request.urlopen(req, timeout=45).read().decode("utf-8", "replace"))
except Exception as e:
return "(error: %s)" % e, 0
lines, oz = [], 0
for r in d.get("results", [])[:12]:
urlr = r.get("url", "")
title = r.get("title", "")
snip = re.sub(r"<[^>]+>", " ", r.get("content", ""))
lines.append("- %s\n %s" % (title[:110], snip[:220]))
if "ozon.ru" in urlr:
oz += 1
for m in re.finditer(r"(19|20)\d{2}\s*г?\.\s*—?\s*(\d{2,4})\s+стр", snip):
lines.append(" >> OZON meta: %s, %s стр" % (m.group(0)[:30], m.group(2)))
for m in re.finditer(r"ISBN[:\s]*(978-[\d\-]{11,14})", snip):
lines.append(" >> OZON ISBN: " + m.group(1))
return "\n".join(lines), oz
def nlr(q):
out = sh_run(sys.executable, os.path.join(ROOT, "tools", "nlr.py"), q, timeout=300)
hits = len(re.findall(r"^\d+\.", out, re.M))
return out, hits
def rutitles(en_title):
return sh_run(sys.executable, os.path.join(ROOT, "tools", "rutitles.py"), "both", en_title, "8")
def index_statuses(secdir):
st = {}
p = os.path.join(ROOT, "sections", secdir, "INDEX.md")
for m in re.finditer(r"^\|\s*(\d{2})\s*\|[^|]*\|[^|]*\|[^|]*\|\s*(✅|🔶|❌|⬜)", open(p, encoding="utf-8").read(), re.M):
st[m.group(1)] = m.group(2)
return st
def main():
args = sys.argv[1:]
sec = args[0] if args else None
if not sec or not re.fullmatch(r"\d{2}-[a-z0-9\-]+", sec):
print(__doc__); sys.exit(1)
do_nlr = "--nlr" in args
only = []
if "--only" in args:
only = args[args.index("--only") + 1: args.index("--only") + 2 + 999]
only = [x for x in only if x.isdigit()]
skip_existing = "--skip-existing" in args
secdir = os.path.join(ROOT, "sections", sec)
inp = os.path.join(ROOT, "data", "sweeps", sec, "input.tsv")
outdir = os.path.join(ROOT, "data", "sweeps", sec)
os.makedirs(outdir, exist_ok=True)
if not os.path.exists(inp):
sys.exit("no %s (create input.tsv: nn slug en_title ru_surnames ru_title_candidates)" % inp)
st = index_statuses(sec)
for line in open(inp, encoding="utf-8"):
line = line.strip()
if not line or line.startswith("#"):
continue
parts = line.split("\t")
while len(parts) < 5:
parts.append("")
nn, slug, en_title, surnames, rtitles = parts[:5]
if only and nn not in only:
continue
if not only and st.get(nn) == "✅":
continue
op = os.path.join(outdir, "%s-%s.md" % (nn, slug))
if skip_existing and os.path.exists(op):
print("skip %s (exists)" % nn)
continue
print("== %s %s (%s)" % (nn, en_title, st.get(nn, "?")))
rep = ["# Sweep %s — %s" % (nn, en_title), "",
"Run: %s" % time.strftime("%Y-%m-%d %H:%M"), "", "status: %s" % st.get(nn, "?"), ""]
surnames = [s for s in surnames.split("|") if s]
rtitles = [t for t in rtitles.split("|") if t]
for s in surnames:
print(" rsl:", s)
out, h = rsl(s); time.sleep(3)
rep.append("## RSL «%s» — %d" % (s, h)); rep.append(out.strip()[:3000]); rep.append("")
log("rsl", nn, s, h)
print(" lg:", s)
out, h = lg(s); time.sleep(4)
rep.append("## libgen «%s» — %d" % (s, h)); rep.append(out.strip()[:3000]); rep.append("")
log("libgen", nn, s, h)
print(" alib:", s)
out, h = alib(s); time.sleep(2)
rep.append("## alib «%s» — %d" % (s, h)); rep.append(out.strip()[:2000]); rep.append("")
log("alib", nn, s, h)
print(" flib:", s)
out, h = flib(s); time.sleep(2)
rep.append("## flibusta «%s» — %d" % (s, h)); rep.append(out.strip()[:2000]); rep.append("")
log("flibusta", nn, s, h)
print(" cogito:", s)
out, h = cogito(s); time.sleep(2)
rep.append("## cogito «%s» — %d" % (s, h)); rep.append(out.strip()[:2000]); rep.append("")
log("cogito", nn, s, h)
for t in rtitles:
print(" sx:", t)
out, h = sx_ozon(t); time.sleep(2)
rep.append("## SearXNG/OZON «%s» — %d ozon" % (t, h)); rep.append(out.strip()[:3000]); rep.append("")
log("searxng", nn, t, h)
print(" alib(title):", t)
out, h = alib(t); time.sleep(2)
rep.append("## alib title «%s» — %d" % (t, h)); rep.append(out.strip()[:2000]); rep.append("")
log("alib-title", nn, t, h)
print(" rutitles:")
out = rutitles(en_title)
rep.append("## rutitles diff (EN→RU + DB match)"); rep.append(out.strip()[:2500]); rep.append("")
log("rutitles", nn, en_title, 0)
if do_nlr and surnames:
print(" nlr:", surnames[0])
out, h = nlr(surnames[0])
rep.append("## NLR «%s» — %d" % (surnames[0], h)); rep.append(out.strip()[:2500]); rep.append("")
log("nlr", nn, surnames[0], h)
time.sleep(5)
open(op, "w", encoding="utf-8").write("\n".join(rep))
print(" -> %s" % op)
print("done")
if __name__ == "__main__":
main()