- tools/rutitles.py: EN->RU token translation (data/ru-dict.tsv, 540+ pairs, Kogito/Castalia
conventions) + Jaccard diff against data/ru-titles.jsonl (692 titles: Kogito 518, Litres 24,
OPP 55, flibusta a/5272+57639+118921+122883+193690+193689). Matching is lemmatized
(pymorphy3) — case endings handled: 'великой матери' -> 'Великая мать' 1.25.
'The Great Mother' -> 'Великая мать' 1.25 top hit; 'The Symbolic Quest' -> correct negative.
- tools/sweep.py: batch driver for Phases 1-2 (rsl+lg+alib+flib+cogito+SearXNG-OZON-snippet
parse+rutitles diff per item; --nlr optional). Query log data/queries.log (JSONL).
Fixed: cogito nav-menu leak (parse bx_product_item only), libgen robot-block (lg.py curl fallback).
- tools/oa.py: OpenAlex + Crossref (Phase 0 identity/ISBN, no key; found Margaret Wilkinson,
Karen Evers-Fahey, Symbolic Quest Princeton ISBN).
- data/sweeps/01-fundamentals/input.tsv: 16 ❌ items loaded; background sweep running.
- AGENTS.md: source 10b (oa.py), flibusta .su = reduced mirror (dropped from pipeline),
pipeline v3 section (rutitles/oa/sweep/pymorphy3 note: pymorphy2 broken on py3.12).
- data/ru-titles.jsonl committed as the RU market universe asset.
68 lines
2.7 KiB
Python
68 lines
2.7 KiB
Python
#!/usr/bin/env python3
|
|
"""Search libgen.vg (Library Genesis). Usage: lg.sh "query" [limit]
|
|
Polite mode: one request per query."""
|
|
import sys, time, urllib.parse, urllib.request, re
|
|
|
|
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"
|
|
|
|
def fetch(url, tries=3):
|
|
last = None
|
|
for i in range(tries):
|
|
try:
|
|
req = urllib.request.Request(url, headers={"User-Agent": UA, "Accept-Language": "ru-RU,ru;q=0.9"})
|
|
html = urllib.request.urlopen(req, timeout=90).read().decode("utf-8", "replace")
|
|
# robot-block detection: nginx default page has no req= result table
|
|
if 'libgen' not in html.lower()[:3000] and 'Search Result' not in html and 'result' not in html.lower()[:3000]:
|
|
time.sleep(15)
|
|
continue
|
|
return html
|
|
except Exception as e:
|
|
last = e
|
|
time.sleep(5 + 5 * i)
|
|
# curl fallback (works when urllib is fingerprinted/blocked)
|
|
import subprocess
|
|
try:
|
|
r = subprocess.run(["curl", "-s", "-A", UA, "--max-time", "90", url],
|
|
capture_output=True, text=True, timeout=120)
|
|
if r.stdout:
|
|
return r.stdout
|
|
except Exception:
|
|
pass
|
|
raise SystemExit(f"fetch failed after {tries} tries: {last}")
|
|
|
|
def main():
|
|
q = sys.argv[1]
|
|
limit = int(sys.argv[2]) if len(sys.argv) > 2 else 10
|
|
url = "https://libgen.vg/index.php?" + urllib.parse.urlencode({
|
|
"req": q, "res": "100", "dlt": "0", "ln": "0",
|
|
"columns[]": ["t","a","s","y","p","i","l","x","sz"]
|
|
})
|
|
html = fetch(url)
|
|
# pagination hint
|
|
m = re.search(r'page=(\d+)', html)
|
|
if m and int(m.group(1)) > 1:
|
|
print(f"[more pages available, max page {m.group(1)} — rerun with &page=N]")
|
|
# parse result rows
|
|
rows = re.findall(r"<tr[^>]*>\s*<td[^>]*>.*?</tr>", html, re.S)
|
|
count = 0
|
|
for r in rows:
|
|
tds = re.findall(r"<td[^>]*>(.*?)</td>", r, re.S)
|
|
tds = [re.sub(r"<[^>]+>", "", t).replace("&","&").strip() for t in tds]
|
|
tds = [re.sub(r"\s+", " ", t) for t in tds]
|
|
if len(tds) < 6:
|
|
continue
|
|
# first cell is usually a link with id, title is the big cell
|
|
title = max(tds, key=len) if tds else ""
|
|
line = " | ".join(tds[:9])
|
|
print(line[:300])
|
|
count += 1
|
|
if count >= limit:
|
|
break
|
|
if count == 0:
|
|
# maybe "No results" or error page
|
|
m = re.search(r"No results|Ничего не найдено|Error[^<]*", html)
|
|
print("(no results)" if m else "(unparsed — check manually)")
|
|
print(html[:500].replace("\n", " "))
|
|
|
|
if __name__ == "__main__":
|
|
main()
|