#!/usr/bin/env python3 """sweep.py — batch driver for pipeline Phases 1-2 (mechanical part). Input: data/sweeps//input.tsv with columns (tab-separated): nn slug en_title ru_surnames('|'-separated variants) ru_title_candidates('|'-separated) (e.g. 30 the-symbolic-quest The Symbolic Quest Уайтмонт|Уитмонт символический поиск|символическая погоня) Usage: sweep.py 01 [--nlr] [--only 30 37 46] [--skip-existing] Reads sections/01-fundamentals/INDEX.md statuses; by default sweeps items NOT marked ✅. --only restricts to given nn's; --skip-existing skips items with an output file. Per item (polite delays): rsl, lg, alib(surname + title), flib books, cogito search, SearXNG (with OZON snippet parse: year/pages/ISBN), rutitles diff. --nlr adds НРБ (slow, CRW-rendered, ~15s/query). Query log: data/queries.log (JSONL). Output: data/sweeps//-.md. """ import json, os, re, sys, time, subprocess, urllib.parse, urllib.request ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) UA = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"} SX = "http://localhost:8888/search" LOG = os.path.join(ROOT, "data", "queries.log") def log(source, item, query, hits, summary=""): rec = {"ts": time.strftime("%Y-%m-%dT%H:%M:%S"), "item": item, "src": source, "q": query, "hits": hits, "sum": summary[:200]} with open(LOG, "a", encoding="utf-8") as f: f.write(json.dumps(rec, ensure_ascii=False) + "\n") def run(cmd, timeout=180): try: r = subprocess.run(cmd, capture_output=True, text=True, timeout=timeout) return (r.stdout or "") + (r.stderr or "") except Exception as e: return "(error: %s)" % e def sh_run(*cmd, timeout=180): return run(list(cmd), timeout) def rsl(q): out = sh_run(sys.executable, os.path.join(ROOT, "tools", "rsl.py"), q) hits = len(re.findall(r"^\[\d+\]", out, re.M)) return out, hits def lg(q): out = sh_run(sys.executable, os.path.join(ROOT, "tools", "lg.py"), q) hits = out.count("libgen.vg/edition.php?id=") return out, hits def alib(q): out = sh_run(sys.executable, os.path.join(ROOT, "tools", "alib.py"), q) hits = out.count("Купить") + len(re.findall(r"\d+\.(?:\d+)?\s+руб", out)) return out, hits def flib(q): out = sh_run(sys.executable, os.path.join(ROOT, "tools", "flib.py"), "books", q) hits = len(re.findall(r"^b/\d+", out, re.M)) return out, hits def cogito(q): url = "https://cogito-shop.com/search/?q=" + urllib.parse.quote(q) out = run(["curl", "-sL", "-A", UA["User-Agent"], "--max-time", "60", url]) # product cards only (bx_product_item divs); the first bx_product_list block is the nav menu items = re.split(r'
]*href="(/catalog/[^"]+)"[^>]*>(?:\s|<[^>]+>)*([^<]{15,120})', blk[:6000]) if m and m.group(2).strip(): prods.append(m.group(2).strip()) seen, lines = set(), [] for x in prods: if x not in seen: seen.add(x); lines.append(x) return "\n".join(lines), len(lines) def sx_ozon(q): url = SX + "?q=" + urllib.parse.quote(q) + "&format=json&language=ru" try: req = urllib.request.Request(url, headers=UA) d = json.loads(urllib.request.urlopen(req, timeout=45).read().decode("utf-8", "replace")) except Exception as e: return "(error: %s)" % e, 0 lines, oz = [], 0 for r in d.get("results", [])[:12]: urlr = r.get("url", "") title = r.get("title", "") snip = re.sub(r"<[^>]+>", " ", r.get("content", "")) lines.append("- %s\n %s" % (title[:110], snip[:220])) if "ozon.ru" in urlr: oz += 1 for m in re.finditer(r"(19|20)\d{2}\s*г?\.\s*—?\s*(\d{2,4})\s+стр", snip): lines.append(" >> OZON meta: %s, %s стр" % (m.group(0)[:30], m.group(2))) for m in re.finditer(r"ISBN[:\s]*(978-[\d\-]{11,14})", snip): lines.append(" >> OZON ISBN: " + m.group(1)) return "\n".join(lines), oz def nlr(q): out = sh_run(sys.executable, os.path.join(ROOT, "tools", "nlr.py"), q, timeout=300) hits = len(re.findall(r"^\d+\.", out, re.M)) return out, hits def rutitles(en_title): return sh_run(sys.executable, os.path.join(ROOT, "tools", "rutitles.py"), "both", en_title, "8") def index_statuses(secdir): st = {} p = os.path.join(ROOT, "sections", secdir, "INDEX.md") for m in re.finditer(r"^\|\s*(\d{2})\s*\|[^|]*\|[^|]*\|[^|]*\|\s*(✅|🔶|❌|⬜)", open(p, encoding="utf-8").read(), re.M): st[m.group(1)] = m.group(2) return st def main(): args = sys.argv[1:] sec = args[0] if args else None if not sec or not re.fullmatch(r"\d{2}-[a-z0-9\-]+", sec): print(__doc__); sys.exit(1) do_nlr = "--nlr" in args only = [] if "--only" in args: only = args[args.index("--only") + 1: args.index("--only") + 2 + 999] only = [x for x in only if x.isdigit()] skip_existing = "--skip-existing" in args secdir = os.path.join(ROOT, "sections", sec) inp = os.path.join(ROOT, "data", "sweeps", sec, "input.tsv") outdir = os.path.join(ROOT, "data", "sweeps", sec) os.makedirs(outdir, exist_ok=True) if not os.path.exists(inp): sys.exit("no %s (create input.tsv: nn slug en_title ru_surnames ru_title_candidates)" % inp) st = index_statuses(sec) for line in open(inp, encoding="utf-8"): line = line.strip() if not line or line.startswith("#"): continue parts = line.split("\t") while len(parts) < 5: parts.append("") nn, slug, en_title, surnames, rtitles = parts[:5] if only and nn not in only: continue if not only and st.get(nn) == "✅": continue op = os.path.join(outdir, "%s-%s.md" % (nn, slug)) if skip_existing and os.path.exists(op): print("skip %s (exists)" % nn) continue print("== %s %s (%s)" % (nn, en_title, st.get(nn, "?"))) rep = ["# Sweep %s — %s" % (nn, en_title), "", "Run: %s" % time.strftime("%Y-%m-%d %H:%M"), "", "status: %s" % st.get(nn, "?"), ""] surnames = [s for s in surnames.split("|") if s] rtitles = [t for t in rtitles.split("|") if t] for s in surnames: print(" rsl:", s) out, h = rsl(s); time.sleep(3) rep.append("## RSL «%s» — %d" % (s, h)); rep.append(out.strip()[:3000]); rep.append("") log("rsl", nn, s, h) print(" lg:", s) out, h = lg(s); time.sleep(4) rep.append("## libgen «%s» — %d" % (s, h)); rep.append(out.strip()[:3000]); rep.append("") log("libgen", nn, s, h) print(" alib:", s) out, h = alib(s); time.sleep(2) rep.append("## alib «%s» — %d" % (s, h)); rep.append(out.strip()[:2000]); rep.append("") log("alib", nn, s, h) print(" flib:", s) out, h = flib(s); time.sleep(2) rep.append("## flibusta «%s» — %d" % (s, h)); rep.append(out.strip()[:2000]); rep.append("") log("flibusta", nn, s, h) print(" cogito:", s) out, h = cogito(s); time.sleep(2) rep.append("## cogito «%s» — %d" % (s, h)); rep.append(out.strip()[:2000]); rep.append("") log("cogito", nn, s, h) for t in rtitles: print(" sx:", t) out, h = sx_ozon(t); time.sleep(20) # google rate-limit: keep <= ~3 req/min (2026-09-19) rep.append("## SearXNG/OZON «%s» — %d ozon" % (t, h)); rep.append(out.strip()[:3000]); rep.append("") log("searxng", nn, t, h) print(" alib(title):", t) out, h = alib(t); time.sleep(2) rep.append("## alib title «%s» — %d" % (t, h)); rep.append(out.strip()[:2000]); rep.append("") log("alib-title", nn, t, h) print(" rutitles:") out = rutitles(en_title) rep.append("## rutitles diff (EN→RU + DB match)"); rep.append(out.strip()[:2500]); rep.append("") log("rutitles", nn, en_title, 0) if do_nlr and surnames: print(" nlr:", surnames[0]) out, h = nlr(surnames[0]) rep.append("## NLR «%s» — %d" % (surnames[0], h)); rep.append(out.strip()[:2500]); rep.append("") log("nlr", nn, surnames[0], h) time.sleep(5) open(op, "w", encoding="utf-8").write("\n".join(rep)) print(" -> %s" % op) print("done") if __name__ == "__main__": main()