jung/tools/sweep.py

205 lines
8.7 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""sweep.py — batch driver for pipeline Phases 1-2 (mechanical part).
Input: data/sweeps/<sec>/input.tsv with columns (tab-separated):
nn slug en_title ru_surnames('|'-separated variants) ru_title_candidates('|'-separated)
(e.g. 30 the-symbolic-quest The Symbolic Quest Уайтмонт|Уитмонт символический поиск|символическая погоня)
Usage:
sweep.py 01 [--nlr] [--only 30 37 46] [--skip-existing]
Reads sections/01-fundamentals/INDEX.md statuses; by default sweeps items NOT marked ✅.
--only restricts to given nn's; --skip-existing skips items with an output file.
Per item (polite delays): rsl, lg, alib(surname + title), flib books, cogito search,
SearXNG (with OZON snippet parse: year/pages/ISBN), rutitles diff.
--nlr adds НРБ (slow, CRW-rendered, ~15s/query).
Query log: data/queries.log (JSONL). Output: data/sweeps/<sec>/<nn>-<slug>.md.
"""
import json, os, re, sys, time, subprocess, urllib.parse, urllib.request
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
UA = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"}
SX = "http://localhost:8888/search"
LOG = os.path.join(ROOT, "data", "queries.log")
def log(source, item, query, hits, summary=""):
rec = {"ts": time.strftime("%Y-%m-%dT%H:%M:%S"), "item": item, "src": source,
"q": query, "hits": hits, "sum": summary[:200]}
with open(LOG, "a", encoding="utf-8") as f:
f.write(json.dumps(rec, ensure_ascii=False) + "\n")
def run(cmd, timeout=180):
try:
r = subprocess.run(cmd, capture_output=True, text=True, timeout=timeout)
return (r.stdout or "") + (r.stderr or "")
except Exception as e:
return "(error: %s)" % e
def sh_run(*cmd, timeout=180):
return run(list(cmd), timeout)
def rsl(q):
out = sh_run(sys.executable, os.path.join(ROOT, "tools", "rsl.py"), q)
hits = len(re.findall(r"^\[\d+\]", out, re.M))
return out, hits
def lg(q):
out = sh_run(sys.executable, os.path.join(ROOT, "tools", "lg.py"), q)
hits = out.count("libgen.vg/edition.php?id=")
return out, hits
def alib(q):
out = sh_run(sys.executable, os.path.join(ROOT, "tools", "alib.py"), q)
hits = out.count("Купить") + len(re.findall(r"\d+\.(?:\d+)?\s+руб", out))
return out, hits
def flib(q):
out = sh_run(sys.executable, os.path.join(ROOT, "tools", "flib.py"), "books", q)
hits = len(re.findall(r"^b/\d+", out, re.M))
return out, hits
def cogito(q):
url = "https://cogito-shop.com/search/?q=" + urllib.parse.quote(q)
out = run(["curl", "-sL", "-A", UA["User-Agent"], "--max-time", "60", url])
# product cards only (bx_product_item divs); the first bx_product_list block is the nav menu
items = re.split(r'<div class="bx_product_item[^"]*"', out)
prods = []
for blk in items[1:]:
m = re.search(r'<a[^>]*href="(/catalog/[^"]+)"[^>]*>(?:\s|<[^>]+>)*([^<]{15,120})', blk[:6000])
if m and m.group(2).strip():
prods.append(m.group(2).strip())
seen, lines = set(), []
for x in prods:
if x not in seen:
seen.add(x); lines.append(x)
return "\n".join(lines), len(lines)
def sx_ozon(q):
url = SX + "?q=" + urllib.parse.quote(q) + "&format=json&language=ru"
try:
req = urllib.request.Request(url, headers=UA)
d = json.loads(urllib.request.urlopen(req, timeout=45).read().decode("utf-8", "replace"))
except Exception as e:
return "(error: %s)" % e, 0
lines, oz = [], 0
for r in d.get("results", [])[:12]:
urlr = r.get("url", "")
title = r.get("title", "")
snip = re.sub(r"<[^>]+>", " ", r.get("content", ""))
lines.append("- %s\n %s" % (title[:110], snip[:220]))
if "ozon.ru" in urlr:
oz += 1
for m in re.finditer(r"(19|20)\d{2}\s*г?\.\s*—?\s*(\d{2,4})\s+стр", snip):
lines.append(" >> OZON meta: %s, %s стр" % (m.group(0)[:30], m.group(2)))
for m in re.finditer(r"ISBN[:\s]*(978-[\d\-]{11,14})", snip):
lines.append(" >> OZON ISBN: " + m.group(1))
return "\n".join(lines), oz
def nlr(q):
out = sh_run(sys.executable, os.path.join(ROOT, "tools", "nlr.py"), q, timeout=300)
hits = len(re.findall(r"^\d+\.", out, re.M))
return out, hits
def rutitles(en_title):
return sh_run(sys.executable, os.path.join(ROOT, "tools", "rutitles.py"), "both", en_title, "8")
def index_statuses(secdir):
st = {}
p = os.path.join(ROOT, "sections", secdir, "INDEX.md")
for m in re.finditer(r"^\|\s*(\d{2})\s*\|[^|]*\|[^|]*\|[^|]*\|\s*(✅|🔶|❌|⬜)", open(p, encoding="utf-8").read(), re.M):
st[m.group(1)] = m.group(2)
return st
def main():
args = sys.argv[1:]
sec = args[0] if args else None
if not sec or not re.fullmatch(r"\d{2}-[a-z0-9\-]+", sec):
print(__doc__); sys.exit(1)
do_nlr = "--nlr" in args
only = []
if "--only" in args:
only = args[args.index("--only") + 1: args.index("--only") + 2 + 999]
only = [x for x in only if x.isdigit()]
skip_existing = "--skip-existing" in args
secdir = os.path.join(ROOT, "sections", sec)
inp = os.path.join(ROOT, "data", "sweeps", sec, "input.tsv")
outdir = os.path.join(ROOT, "data", "sweeps", sec)
os.makedirs(outdir, exist_ok=True)
if not os.path.exists(inp):
sys.exit("no %s (create input.tsv: nn slug en_title ru_surnames ru_title_candidates)" % inp)
st = index_statuses(sec)
for line in open(inp, encoding="utf-8"):
line = line.strip()
if not line or line.startswith("#"):
continue
parts = line.split("\t")
while len(parts) < 5:
parts.append("")
nn, slug, en_title, surnames, rtitles = parts[:5]
if only and nn not in only:
continue
if not only and st.get(nn) == "✅":
continue
op = os.path.join(outdir, "%s-%s.md" % (nn, slug))
if skip_existing and os.path.exists(op):
print("skip %s (exists)" % nn)
continue
print("== %s %s (%s)" % (nn, en_title, st.get(nn, "?")))
rep = ["# Sweep %s — %s" % (nn, en_title), "",
"Run: %s" % time.strftime("%Y-%m-%d %H:%M"), "", "status: %s" % st.get(nn, "?"), ""]
surnames = [s for s in surnames.split("|") if s]
rtitles = [t for t in rtitles.split("|") if t]
for s in surnames:
print(" rsl:", s)
out, h = rsl(s); time.sleep(3)
rep.append("## RSL «%s» — %d" % (s, h)); rep.append(out.strip()[:3000]); rep.append("")
log("rsl", nn, s, h)
print(" lg:", s)
out, h = lg(s); time.sleep(4)
rep.append("## libgen «%s» — %d" % (s, h)); rep.append(out.strip()[:3000]); rep.append("")
log("libgen", nn, s, h)
print(" alib:", s)
out, h = alib(s); time.sleep(2)
rep.append("## alib «%s» — %d" % (s, h)); rep.append(out.strip()[:2000]); rep.append("")
log("alib", nn, s, h)
print(" flib:", s)
out, h = flib(s); time.sleep(2)
rep.append("## flibusta «%s» — %d" % (s, h)); rep.append(out.strip()[:2000]); rep.append("")
log("flibusta", nn, s, h)
print(" cogito:", s)
out, h = cogito(s); time.sleep(2)
rep.append("## cogito «%s» — %d" % (s, h)); rep.append(out.strip()[:2000]); rep.append("")
log("cogito", nn, s, h)
for t in rtitles:
print(" sx:", t)
out, h = sx_ozon(t); time.sleep(20) # google rate-limit: keep <= ~3 req/min (2026-09-19)
rep.append("## SearXNG/OZON «%s» — %d ozon" % (t, h)); rep.append(out.strip()[:3000]); rep.append("")
log("searxng", nn, t, h)
print(" alib(title):", t)
out, h = alib(t); time.sleep(2)
rep.append("## alib title «%s» — %d" % (t, h)); rep.append(out.strip()[:2000]); rep.append("")
log("alib-title", nn, t, h)
print(" rutitles:")
out = rutitles(en_title)
rep.append("## rutitles diff (EN→RU + DB match)"); rep.append(out.strip()[:2500]); rep.append("")
log("rutitles", nn, en_title, 0)
if do_nlr and surnames:
print(" nlr:", surnames[0])
out, h = nlr(surnames[0])
rep.append("## NLR «%s» — %d" % (surnames[0], h)); rep.append(out.strip()[:2500]); rep.append("")
log("nlr", nn, surnames[0], h)
time.sleep(5)
open(op, "w", encoding="utf-8").write("\n".join(rep))
print(" -> %s" % op)
print("done")
if __name__ == "__main__":
main()