jung/tools/lg.py
Dmitry Kokorin b7ddfac17e sec06: key-vs-f_id discovery — 12 EN files recovered (6 rescues) + precheck v2
- USER FINDING: libgen edition files map {key: {f_id, md5}} — KEY and f_id VALUE
  are DIFFERENT file records; we were downloading/prechecking by the KEY (wrong file).
- Recovered by f_id: 09-t1 (Eliade UCP 1978), 13 Sungod (Cornell 2010), 14 Solomon
  (Karnac 2007), 17 God-Image (Inner Light 1992), 19 Hornung (UNC 1982), 20 Psyche in
  Scripture (1995), 23 Dionysos (Bollingen LXV/2 Princeton) + backups 12/16/24/25/26/27.
  All content-verified. '24 systemic mislinks' = our bug, not libgen.
- lg.py: objects[]=f,e,s,a,p,w + topics[] (file-level search; found Solomon md5) +
  md5 in output. lgdl.py: bymd5 subcommand.
- precheck v2: rev-value match = OK; locator words >=2 shared = OK; words present
  0 shared = BAD; hash/ISBN/empty locator = no signal (SUSPECT); digit-runs masked
  before tokenization (hex 'aaae' trap).
- MANIFEST/SUMMARY/cards/AGENTS.md updated. 34 files, 608 MB (28 backup still downloading).
2026-09-25 10:04:01 +03:00

75 lines
3.2 KiB
Python

#!/usr/bin/env python3
"""Search libgen.vg (Library Genesis). Usage: lg.sh "query" [limit]
Polite mode: one request per query."""
import sys, time, urllib.parse, urllib.request, re
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"
def fetch(url, tries=3):
last = None
for i in range(tries):
try:
req = urllib.request.Request(url, headers={"User-Agent": UA, "Accept-Language": "ru-RU,ru;q=0.9"})
html = urllib.request.urlopen(req, timeout=90).read().decode("utf-8", "replace")
# robot-block detection: nginx default page has no req= result table
if 'libgen' not in html.lower()[:3000] and 'Search Result' not in html and 'result' not in html.lower()[:3000]:
time.sleep(15)
continue
return html
except Exception as e:
last = e
time.sleep(5 + 5 * i)
# curl fallback (works when urllib is fingerprinted/blocked)
import subprocess
try:
r = subprocess.run(["curl", "-s", "-A", UA, "--max-time", "90", url],
capture_output=True, text=True, timeout=120)
if r.stdout:
return r.stdout
except Exception:
pass
raise SystemExit(f"fetch failed after {tries} tries: {last}")
def main():
q = sys.argv[1]
limit = int(sys.argv[2]) if len(sys.argv) > 2 else 10
# objects[] MUST include 'f' (files): editions-only search misses files that
# exist in libgen but are not linked to the right edition (verified 2026-09-25:
# Solomon "Self in Transformation" found by file search, missing in edition search).
params = {"req": q, "res": "100"}
params["columns[]"] = ["t","a","s","y","p","i"]
params["objects[]"] = ["f","e","s","a","p","w"]
params["topics[]"] = ["l","c","f","a","m","r","s"]
url = "https://libgen.vg/index.php?" + urllib.parse.urlencode(params, doseq=True)
html = fetch(url)
# pagination hint
m = re.search(r'page=(\d+)', html)
if m and int(m.group(1)) > 1:
print(f"[more pages available, max page {m.group(1)} — rerun with &page=N]")
# parse result rows
rows = re.findall(r"<tr[^>]*>\s*<td[^>]*>.*?</tr>", html, re.S)
count = 0
for r in rows:
tds = re.findall(r"<td[^>]*>(.*?)</td>", r, re.S)
tds = [re.sub(r"<[^>]+>", "", t).replace("&amp;","&").strip() for t in tds]
tds = [re.sub(r"\s+", " ", t) for t in tds]
if len(tds) < 6:
continue
# first cell is usually a link with id, title is the big cell
line = " | ".join(tds[:9])
# file rows: surface the md5 (ads.php link) — feed to tools/lgdl.py bymd5
md5s = re.findall(r'ads\.php\?md5=([0-9a-f]{32})', r)
if md5s:
line += f" [md5={md5s[0]}]"
print(line[:300])
count += 1
if count >= limit:
break
if count == 0:
# maybe "No results" or error page
m = re.search(r"No results|Ничего не найдено|Error[^<]*", html)
print("(no results)" if m else "(unparsed — check manually)")
print(html[:500].replace("\n", " "))
if __name__ == "__main__":
main()