pipeline v3: RU-title candidates (rutitles+ru-dict+pymorphy3), batch sweep driver, OpenAlex/Crossref

- tools/rutitles.py: EN->RU token translation (data/ru-dict.tsv, 540+ pairs, Kogito/Castalia
  conventions) + Jaccard diff against data/ru-titles.jsonl (692 titles: Kogito 518, Litres 24,
  OPP 55, flibusta a/5272+57639+118921+122883+193690+193689). Matching is lemmatized
  (pymorphy3) — case endings handled: 'великой матери' -> 'Великая мать' 1.25.
  'The Great Mother' -> 'Великая мать' 1.25 top hit; 'The Symbolic Quest' -> correct negative.
- tools/sweep.py: batch driver for Phases 1-2 (rsl+lg+alib+flib+cogito+SearXNG-OZON-snippet
  parse+rutitles diff per item; --nlr optional). Query log data/queries.log (JSONL).
  Fixed: cogito nav-menu leak (parse bx_product_item only), libgen robot-block (lg.py curl fallback).
- tools/oa.py: OpenAlex + Crossref (Phase 0 identity/ISBN, no key; found Margaret Wilkinson,
  Karen Evers-Fahey, Symbolic Quest Princeton ISBN).
- data/sweeps/01-fundamentals/input.tsv: 16 ❌ items loaded; background sweep running.
- AGENTS.md: source 10b (oa.py), flibusta .su = reduced mirror (dropped from pipeline),
  pipeline v3 section (rutitles/oa/sweep/pymorphy3 note: pymorphy2 broken on py3.12).
- data/ru-titles.jsonl committed as the RU market universe asset.
This commit is contained in:
Dmitry Kokorin 2026-09-18 10:56:36 +03:00
parent 2157d058df
commit 5441dddbb8
11 changed files with 2101 additions and 3 deletions

205
tools/sweep.py Normal file
View file

@ -0,0 +1,205 @@
#!/usr/bin/env python3
"""sweep.py — batch driver for pipeline Phases 1-2 (mechanical part).
Input: data/sweeps/<sec>/input.tsv with columns (tab-separated):
nn slug en_title ru_surnames('|'-separated variants) ru_title_candidates('|'-separated)
(e.g. 30 the-symbolic-quest The Symbolic Quest Уайтмонт|Уитмонт символический поиск|символическая погоня)
Usage:
sweep.py 01 [--nlr] [--only 30 37 46] [--skip-existing]
Reads sections/01-fundamentals/INDEX.md statuses; by default sweeps items NOT marked ✅.
--only restricts to given nn's; --skip-existing skips items with an output file.
Per item (polite delays): rsl, lg, alib(surname + title), flib books, cogito search,
SearXNG (with OZON snippet parse: year/pages/ISBN), rutitles diff.
--nlr adds НРБ (slow, CRW-rendered, ~15s/query).
Query log: data/queries.log (JSONL). Output: data/sweeps/<sec>/<nn>-<slug>.md.
"""
import json, os, re, sys, time, subprocess, urllib.parse, urllib.request
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
UA = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"}
SX = "http://localhost:8888/search"
LOG = os.path.join(ROOT, "data", "queries.log")
def log(source, item, query, hits, summary=""):
rec = {"ts": time.strftime("%Y-%m-%dT%H:%M:%S"), "item": item, "src": source,
"q": query, "hits": hits, "sum": summary[:200]}
with open(LOG, "a", encoding="utf-8") as f:
f.write(json.dumps(rec, ensure_ascii=False) + "\n")
def run(cmd, timeout=180):
try:
r = subprocess.run(cmd, capture_output=True, text=True, timeout=timeout)
return (r.stdout or "") + (r.stderr or "")
except Exception as e:
return "(error: %s)" % e
def sh_run(*cmd, timeout=180):
return run(list(cmd), timeout)
def rsl(q):
out = sh_run(sys.executable, os.path.join(ROOT, "tools", "rsl.py"), q)
hits = len(re.findall(r"^\[\d+\]", out, re.M))
return out, hits
def lg(q):
out = sh_run(sys.executable, os.path.join(ROOT, "tools", "lg.py"), q)
hits = out.count("libgen.vg/edition.php?id=")
return out, hits
def alib(q):
out = sh_run(sys.executable, os.path.join(ROOT, "tools", "alib.py"), q)
hits = out.count("Купить") + len(re.findall(r"\d+\.(?:\d+)?\s+руб", out))
return out, hits
def flib(q):
out = sh_run(sys.executable, os.path.join(ROOT, "tools", "flib.py"), "books", q)
hits = len(re.findall(r"^b/\d+", out, re.M))
return out, hits
def cogito(q):
url = "https://cogito-shop.com/search/?q=" + urllib.parse.quote(q)
out = run(["curl", "-sL", "-A", UA["User-Agent"], "--max-time", "60", url])
# product cards only (bx_product_item divs); the first bx_product_list block is the nav menu
items = re.split(r'<div class="bx_product_item[^"]*"', out)
prods = []
for blk in items[1:]:
m = re.search(r'<a[^>]*href="(/catalog/[^"]+)"[^>]*>(?:\s|<[^>]+>)*([^<]{15,120})', blk[:6000])
if m and m.group(2).strip():
prods.append(m.group(2).strip())
seen, lines = set(), []
for x in prods:
if x not in seen:
seen.add(x); lines.append(x)
return "\n".join(lines), len(lines)
def sx_ozon(q):
url = SX + "?q=" + urllib.parse.quote(q) + "&format=json&language=ru"
try:
req = urllib.request.Request(url, headers=UA)
d = json.loads(urllib.request.urlopen(req, timeout=45).read().decode("utf-8", "replace"))
except Exception as e:
return "(error: %s)" % e, 0
lines, oz = [], 0
for r in d.get("results", [])[:12]:
urlr = r.get("url", "")
title = r.get("title", "")
snip = re.sub(r"<[^>]+>", " ", r.get("content", ""))
lines.append("- %s\n %s" % (title[:110], snip[:220]))
if "ozon.ru" in urlr:
oz += 1
for m in re.finditer(r"(19|20)\d{2}\s*г?\.\s*—?\s*(\d{2,4})\s+стр", snip):
lines.append(" >> OZON meta: %s, %s стр" % (m.group(0)[:30], m.group(2)))
for m in re.finditer(r"ISBN[:\s]*(978-[\d\-]{11,14})", snip):
lines.append(" >> OZON ISBN: " + m.group(1))
return "\n".join(lines), oz
def nlr(q):
out = sh_run(sys.executable, os.path.join(ROOT, "tools", "nlr.py"), q, timeout=300)
hits = len(re.findall(r"^\d+\.", out, re.M))
return out, hits
def rutitles(en_title):
return sh_run(sys.executable, os.path.join(ROOT, "tools", "rutitles.py"), "both", en_title, "8")
def index_statuses(secdir):
st = {}
p = os.path.join(ROOT, "sections", secdir, "INDEX.md")
for m in re.finditer(r"^\|\s*(\d{2})\s*\|[^|]*\|[^|]*\|[^|]*\|\s*(✅|🔶|❌|⬜)", open(p, encoding="utf-8").read(), re.M):
st[m.group(1)] = m.group(2)
return st
def main():
args = sys.argv[1:]
sec = args[0] if args else None
if not sec or not re.fullmatch(r"\d{2}-[a-z0-9\-]+", sec):
print(__doc__); sys.exit(1)
do_nlr = "--nlr" in args
only = []
if "--only" in args:
only = args[args.index("--only") + 1: args.index("--only") + 2 + 999]
only = [x for x in only if x.isdigit()]
skip_existing = "--skip-existing" in args
secdir = os.path.join(ROOT, "sections", sec)
inp = os.path.join(ROOT, "data", "sweeps", sec, "input.tsv")
outdir = os.path.join(ROOT, "data", "sweeps", sec)
os.makedirs(outdir, exist_ok=True)
if not os.path.exists(inp):
sys.exit("no %s (create input.tsv: nn slug en_title ru_surnames ru_title_candidates)" % inp)
st = index_statuses(sec)
for line in open(inp, encoding="utf-8"):
line = line.strip()
if not line or line.startswith("#"):
continue
parts = line.split("\t")
while len(parts) < 5:
parts.append("")
nn, slug, en_title, surnames, rtitles = parts[:5]
if only and nn not in only:
continue
if not only and st.get(nn) == "✅":
continue
op = os.path.join(outdir, "%s-%s.md" % (nn, slug))
if skip_existing and os.path.exists(op):
print("skip %s (exists)" % nn)
continue
print("== %s %s (%s)" % (nn, en_title, st.get(nn, "?")))
rep = ["# Sweep %s — %s" % (nn, en_title), "",
"Run: %s" % time.strftime("%Y-%m-%d %H:%M"), "", "status: %s" % st.get(nn, "?"), ""]
surnames = [s for s in surnames.split("|") if s]
rtitles = [t for t in rtitles.split("|") if t]
for s in surnames:
print(" rsl:", s)
out, h = rsl(s); time.sleep(3)
rep.append("## RSL «%s» — %d" % (s, h)); rep.append(out.strip()[:3000]); rep.append("")
log("rsl", nn, s, h)
print(" lg:", s)
out, h = lg(s); time.sleep(4)
rep.append("## libgen «%s» — %d" % (s, h)); rep.append(out.strip()[:3000]); rep.append("")
log("libgen", nn, s, h)
print(" alib:", s)
out, h = alib(s); time.sleep(2)
rep.append("## alib «%s» — %d" % (s, h)); rep.append(out.strip()[:2000]); rep.append("")
log("alib", nn, s, h)
print(" flib:", s)
out, h = flib(s); time.sleep(2)
rep.append("## flibusta «%s» — %d" % (s, h)); rep.append(out.strip()[:2000]); rep.append("")
log("flibusta", nn, s, h)
print(" cogito:", s)
out, h = cogito(s); time.sleep(2)
rep.append("## cogito «%s» — %d" % (s, h)); rep.append(out.strip()[:2000]); rep.append("")
log("cogito", nn, s, h)
for t in rtitles:
print(" sx:", t)
out, h = sx_ozon(t); time.sleep(2)
rep.append("## SearXNG/OZON «%s» — %d ozon" % (t, h)); rep.append(out.strip()[:3000]); rep.append("")
log("searxng", nn, t, h)
print(" alib(title):", t)
out, h = alib(t); time.sleep(2)
rep.append("## alib title «%s» — %d" % (t, h)); rep.append(out.strip()[:2000]); rep.append("")
log("alib-title", nn, t, h)
print(" rutitles:")
out = rutitles(en_title)
rep.append("## rutitles diff (EN→RU + DB match)"); rep.append(out.strip()[:2500]); rep.append("")
log("rutitles", nn, en_title, 0)
if do_nlr and surnames:
print(" nlr:", surnames[0])
out, h = nlr(surnames[0])
rep.append("## NLR «%s» — %d" % (surnames[0], h)); rep.append(out.strip()[:2500]); rep.append("")
log("nlr", nn, surnames[0], h)
time.sleep(5)
open(op, "w", encoding="utf-8").write("\n".join(rep))
print(" -> %s" % op)
print("done")
if __name__ == "__main__":
main()