- tools/rutitles.py: EN->RU token translation (data/ru-dict.tsv, 540+ pairs, Kogito/Castalia
conventions) + Jaccard diff against data/ru-titles.jsonl (692 titles: Kogito 518, Litres 24,
OPP 55, flibusta a/5272+57639+118921+122883+193690+193689). Matching is lemmatized
(pymorphy3) — case endings handled: 'великой матери' -> 'Великая мать' 1.25.
'The Great Mother' -> 'Великая мать' 1.25 top hit; 'The Symbolic Quest' -> correct negative.
- tools/sweep.py: batch driver for Phases 1-2 (rsl+lg+alib+flib+cogito+SearXNG-OZON-snippet
parse+rutitles diff per item; --nlr optional). Query log data/queries.log (JSONL).
Fixed: cogito nav-menu leak (parse bx_product_item only), libgen robot-block (lg.py curl fallback).
- tools/oa.py: OpenAlex + Crossref (Phase 0 identity/ISBN, no key; found Margaret Wilkinson,
Karen Evers-Fahey, Symbolic Quest Princeton ISBN).
- data/sweeps/01-fundamentals/input.tsv: 16 ❌ items loaded; background sweep running.
- AGENTS.md: source 10b (oa.py), flibusta .su = reduced mirror (dropped from pipeline),
pipeline v3 section (rutitles/oa/sweep/pymorphy3 note: pymorphy2 broken on py3.12).
- data/ru-titles.jsonl committed as the RU market universe asset.
205 lines
8.6 KiB
Python
205 lines
8.6 KiB
Python
#!/usr/bin/env python3
|
||
"""sweep.py — batch driver for pipeline Phases 1-2 (mechanical part).
|
||
|
||
Input: data/sweeps/<sec>/input.tsv with columns (tab-separated):
|
||
nn slug en_title ru_surnames('|'-separated variants) ru_title_candidates('|'-separated)
|
||
(e.g. 30 the-symbolic-quest The Symbolic Quest Уайтмонт|Уитмонт символический поиск|символическая погоня)
|
||
|
||
Usage:
|
||
sweep.py 01 [--nlr] [--only 30 37 46] [--skip-existing]
|
||
Reads sections/01-fundamentals/INDEX.md statuses; by default sweeps items NOT marked ✅.
|
||
--only restricts to given nn's; --skip-existing skips items with an output file.
|
||
|
||
Per item (polite delays): rsl, lg, alib(surname + title), flib books, cogito search,
|
||
SearXNG (with OZON snippet parse: year/pages/ISBN), rutitles diff.
|
||
--nlr adds НРБ (slow, CRW-rendered, ~15s/query).
|
||
Query log: data/queries.log (JSONL). Output: data/sweeps/<sec>/<nn>-<slug>.md.
|
||
"""
|
||
import json, os, re, sys, time, subprocess, urllib.parse, urllib.request
|
||
|
||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||
UA = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"}
|
||
SX = "http://localhost:8888/search"
|
||
LOG = os.path.join(ROOT, "data", "queries.log")
|
||
|
||
def log(source, item, query, hits, summary=""):
|
||
rec = {"ts": time.strftime("%Y-%m-%dT%H:%M:%S"), "item": item, "src": source,
|
||
"q": query, "hits": hits, "sum": summary[:200]}
|
||
with open(LOG, "a", encoding="utf-8") as f:
|
||
f.write(json.dumps(rec, ensure_ascii=False) + "\n")
|
||
|
||
def run(cmd, timeout=180):
|
||
try:
|
||
r = subprocess.run(cmd, capture_output=True, text=True, timeout=timeout)
|
||
return (r.stdout or "") + (r.stderr or "")
|
||
except Exception as e:
|
||
return "(error: %s)" % e
|
||
|
||
def sh_run(*cmd, timeout=180):
|
||
return run(list(cmd), timeout)
|
||
|
||
def rsl(q):
|
||
out = sh_run(sys.executable, os.path.join(ROOT, "tools", "rsl.py"), q)
|
||
hits = len(re.findall(r"^\[\d+\]", out, re.M))
|
||
return out, hits
|
||
|
||
def lg(q):
|
||
out = sh_run(sys.executable, os.path.join(ROOT, "tools", "lg.py"), q)
|
||
hits = out.count("libgen.vg/edition.php?id=")
|
||
return out, hits
|
||
|
||
def alib(q):
|
||
out = sh_run(sys.executable, os.path.join(ROOT, "tools", "alib.py"), q)
|
||
hits = out.count("Купить") + len(re.findall(r"\d+\.(?:\d+)?\s+руб", out))
|
||
return out, hits
|
||
|
||
def flib(q):
|
||
out = sh_run(sys.executable, os.path.join(ROOT, "tools", "flib.py"), "books", q)
|
||
hits = len(re.findall(r"^b/\d+", out, re.M))
|
||
return out, hits
|
||
|
||
def cogito(q):
|
||
url = "https://cogito-shop.com/search/?q=" + urllib.parse.quote(q)
|
||
out = run(["curl", "-sL", "-A", UA["User-Agent"], "--max-time", "60", url])
|
||
# product cards only (bx_product_item divs); the first bx_product_list block is the nav menu
|
||
items = re.split(r'<div class="bx_product_item[^"]*"', out)
|
||
prods = []
|
||
for blk in items[1:]:
|
||
m = re.search(r'<a[^>]*href="(/catalog/[^"]+)"[^>]*>(?:\s|<[^>]+>)*([^<]{15,120})', blk[:6000])
|
||
if m and m.group(2).strip():
|
||
prods.append(m.group(2).strip())
|
||
seen, lines = set(), []
|
||
for x in prods:
|
||
if x not in seen:
|
||
seen.add(x); lines.append(x)
|
||
return "\n".join(lines), len(lines)
|
||
|
||
def sx_ozon(q):
|
||
url = SX + "?q=" + urllib.parse.quote(q) + "&format=json&language=ru"
|
||
try:
|
||
req = urllib.request.Request(url, headers=UA)
|
||
d = json.loads(urllib.request.urlopen(req, timeout=45).read().decode("utf-8", "replace"))
|
||
except Exception as e:
|
||
return "(error: %s)" % e, 0
|
||
lines, oz = [], 0
|
||
for r in d.get("results", [])[:12]:
|
||
urlr = r.get("url", "")
|
||
title = r.get("title", "")
|
||
snip = re.sub(r"<[^>]+>", " ", r.get("content", ""))
|
||
lines.append("- %s\n %s" % (title[:110], snip[:220]))
|
||
if "ozon.ru" in urlr:
|
||
oz += 1
|
||
for m in re.finditer(r"(19|20)\d{2}\s*г?\.\s*—?\s*(\d{2,4})\s+стр", snip):
|
||
lines.append(" >> OZON meta: %s, %s стр" % (m.group(0)[:30], m.group(2)))
|
||
for m in re.finditer(r"ISBN[:\s]*(978-[\d\-]{11,14})", snip):
|
||
lines.append(" >> OZON ISBN: " + m.group(1))
|
||
return "\n".join(lines), oz
|
||
|
||
def nlr(q):
|
||
out = sh_run(sys.executable, os.path.join(ROOT, "tools", "nlr.py"), q, timeout=300)
|
||
hits = len(re.findall(r"^\d+\.", out, re.M))
|
||
return out, hits
|
||
|
||
def rutitles(en_title):
|
||
return sh_run(sys.executable, os.path.join(ROOT, "tools", "rutitles.py"), "both", en_title, "8")
|
||
|
||
def index_statuses(secdir):
|
||
st = {}
|
||
p = os.path.join(ROOT, "sections", secdir, "INDEX.md")
|
||
for m in re.finditer(r"^\|\s*(\d{2})\s*\|[^|]*\|[^|]*\|[^|]*\|\s*(✅|🔶|❌|⬜)", open(p, encoding="utf-8").read(), re.M):
|
||
st[m.group(1)] = m.group(2)
|
||
return st
|
||
|
||
def main():
|
||
args = sys.argv[1:]
|
||
sec = args[0] if args else None
|
||
if not sec or not re.fullmatch(r"\d{2}-[a-z0-9\-]+", sec):
|
||
print(__doc__); sys.exit(1)
|
||
do_nlr = "--nlr" in args
|
||
only = []
|
||
if "--only" in args:
|
||
only = args[args.index("--only") + 1: args.index("--only") + 2 + 999]
|
||
only = [x for x in only if x.isdigit()]
|
||
skip_existing = "--skip-existing" in args
|
||
|
||
secdir = os.path.join(ROOT, "sections", sec)
|
||
inp = os.path.join(ROOT, "data", "sweeps", sec, "input.tsv")
|
||
outdir = os.path.join(ROOT, "data", "sweeps", sec)
|
||
os.makedirs(outdir, exist_ok=True)
|
||
if not os.path.exists(inp):
|
||
sys.exit("no %s (create input.tsv: nn slug en_title ru_surnames ru_title_candidates)" % inp)
|
||
|
||
st = index_statuses(sec)
|
||
for line in open(inp, encoding="utf-8"):
|
||
line = line.strip()
|
||
if not line or line.startswith("#"):
|
||
continue
|
||
parts = line.split("\t")
|
||
while len(parts) < 5:
|
||
parts.append("")
|
||
nn, slug, en_title, surnames, rtitles = parts[:5]
|
||
if only and nn not in only:
|
||
continue
|
||
if not only and st.get(nn) == "✅":
|
||
continue
|
||
op = os.path.join(outdir, "%s-%s.md" % (nn, slug))
|
||
if skip_existing and os.path.exists(op):
|
||
print("skip %s (exists)" % nn)
|
||
continue
|
||
print("== %s %s (%s)" % (nn, en_title, st.get(nn, "?")))
|
||
rep = ["# Sweep %s — %s" % (nn, en_title), "",
|
||
"Run: %s" % time.strftime("%Y-%m-%d %H:%M"), "", "status: %s" % st.get(nn, "?"), ""]
|
||
surnames = [s for s in surnames.split("|") if s]
|
||
rtitles = [t for t in rtitles.split("|") if t]
|
||
|
||
for s in surnames:
|
||
print(" rsl:", s)
|
||
out, h = rsl(s); time.sleep(3)
|
||
rep.append("## RSL «%s» — %d" % (s, h)); rep.append(out.strip()[:3000]); rep.append("")
|
||
log("rsl", nn, s, h)
|
||
print(" lg:", s)
|
||
out, h = lg(s); time.sleep(4)
|
||
rep.append("## libgen «%s» — %d" % (s, h)); rep.append(out.strip()[:3000]); rep.append("")
|
||
log("libgen", nn, s, h)
|
||
print(" alib:", s)
|
||
out, h = alib(s); time.sleep(2)
|
||
rep.append("## alib «%s» — %d" % (s, h)); rep.append(out.strip()[:2000]); rep.append("")
|
||
log("alib", nn, s, h)
|
||
print(" flib:", s)
|
||
out, h = flib(s); time.sleep(2)
|
||
rep.append("## flibusta «%s» — %d" % (s, h)); rep.append(out.strip()[:2000]); rep.append("")
|
||
log("flibusta", nn, s, h)
|
||
print(" cogito:", s)
|
||
out, h = cogito(s); time.sleep(2)
|
||
rep.append("## cogito «%s» — %d" % (s, h)); rep.append(out.strip()[:2000]); rep.append("")
|
||
log("cogito", nn, s, h)
|
||
|
||
for t in rtitles:
|
||
print(" sx:", t)
|
||
out, h = sx_ozon(t); time.sleep(2)
|
||
rep.append("## SearXNG/OZON «%s» — %d ozon" % (t, h)); rep.append(out.strip()[:3000]); rep.append("")
|
||
log("searxng", nn, t, h)
|
||
print(" alib(title):", t)
|
||
out, h = alib(t); time.sleep(2)
|
||
rep.append("## alib title «%s» — %d" % (t, h)); rep.append(out.strip()[:2000]); rep.append("")
|
||
log("alib-title", nn, t, h)
|
||
|
||
print(" rutitles:")
|
||
out = rutitles(en_title)
|
||
rep.append("## rutitles diff (EN→RU + DB match)"); rep.append(out.strip()[:2500]); rep.append("")
|
||
log("rutitles", nn, en_title, 0)
|
||
|
||
if do_nlr and surnames:
|
||
print(" nlr:", surnames[0])
|
||
out, h = nlr(surnames[0])
|
||
rep.append("## NLR «%s» — %d" % (surnames[0], h)); rep.append(out.strip()[:2500]); rep.append("")
|
||
log("nlr", nn, surnames[0], h)
|
||
time.sleep(5)
|
||
|
||
open(op, "w", encoding="utf-8").write("\n".join(rep))
|
||
print(" -> %s" % op)
|
||
|
||
print("done")
|
||
|
||
if __name__ == "__main__":
|
||
main()
|