- USER FINDING: libgen edition files map {key: {f_id, md5}} — KEY and f_id VALUE
are DIFFERENT file records; we were downloading/prechecking by the KEY (wrong file).
- Recovered by f_id: 09-t1 (Eliade UCP 1978), 13 Sungod (Cornell 2010), 14 Solomon
(Karnac 2007), 17 God-Image (Inner Light 1992), 19 Hornung (UNC 1982), 20 Psyche in
Scripture (1995), 23 Dionysos (Bollingen LXV/2 Princeton) + backups 12/16/24/25/26/27.
All content-verified. '24 systemic mislinks' = our bug, not libgen.
- lg.py: objects[]=f,e,s,a,p,w + topics[] (file-level search; found Solomon md5) +
md5 in output. lgdl.py: bymd5 subcommand.
- precheck v2: rev-value match = OK; locator words >=2 shared = OK; words present
0 shared = BAD; hash/ISBN/empty locator = no signal (SUSPECT); digit-runs masked
before tokenization (hex 'aaae' trap).
- MANIFEST/SUMMARY/cards/AGENTS.md updated. 34 files, 608 MB (28 backup still downloading).
75 lines
3.2 KiB
Python
75 lines
3.2 KiB
Python
#!/usr/bin/env python3
|
|
"""Search libgen.vg (Library Genesis). Usage: lg.sh "query" [limit]
|
|
Polite mode: one request per query."""
|
|
import sys, time, urllib.parse, urllib.request, re
|
|
|
|
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"
|
|
|
|
def fetch(url, tries=3):
|
|
last = None
|
|
for i in range(tries):
|
|
try:
|
|
req = urllib.request.Request(url, headers={"User-Agent": UA, "Accept-Language": "ru-RU,ru;q=0.9"})
|
|
html = urllib.request.urlopen(req, timeout=90).read().decode("utf-8", "replace")
|
|
# robot-block detection: nginx default page has no req= result table
|
|
if 'libgen' not in html.lower()[:3000] and 'Search Result' not in html and 'result' not in html.lower()[:3000]:
|
|
time.sleep(15)
|
|
continue
|
|
return html
|
|
except Exception as e:
|
|
last = e
|
|
time.sleep(5 + 5 * i)
|
|
# curl fallback (works when urllib is fingerprinted/blocked)
|
|
import subprocess
|
|
try:
|
|
r = subprocess.run(["curl", "-s", "-A", UA, "--max-time", "90", url],
|
|
capture_output=True, text=True, timeout=120)
|
|
if r.stdout:
|
|
return r.stdout
|
|
except Exception:
|
|
pass
|
|
raise SystemExit(f"fetch failed after {tries} tries: {last}")
|
|
|
|
def main():
|
|
q = sys.argv[1]
|
|
limit = int(sys.argv[2]) if len(sys.argv) > 2 else 10
|
|
# objects[] MUST include 'f' (files): editions-only search misses files that
|
|
# exist in libgen but are not linked to the right edition (verified 2026-09-25:
|
|
# Solomon "Self in Transformation" found by file search, missing in edition search).
|
|
params = {"req": q, "res": "100"}
|
|
params["columns[]"] = ["t","a","s","y","p","i"]
|
|
params["objects[]"] = ["f","e","s","a","p","w"]
|
|
params["topics[]"] = ["l","c","f","a","m","r","s"]
|
|
url = "https://libgen.vg/index.php?" + urllib.parse.urlencode(params, doseq=True)
|
|
html = fetch(url)
|
|
# pagination hint
|
|
m = re.search(r'page=(\d+)', html)
|
|
if m and int(m.group(1)) > 1:
|
|
print(f"[more pages available, max page {m.group(1)} — rerun with &page=N]")
|
|
# parse result rows
|
|
rows = re.findall(r"<tr[^>]*>\s*<td[^>]*>.*?</tr>", html, re.S)
|
|
count = 0
|
|
for r in rows:
|
|
tds = re.findall(r"<td[^>]*>(.*?)</td>", r, re.S)
|
|
tds = [re.sub(r"<[^>]+>", "", t).replace("&","&").strip() for t in tds]
|
|
tds = [re.sub(r"\s+", " ", t) for t in tds]
|
|
if len(tds) < 6:
|
|
continue
|
|
# first cell is usually a link with id, title is the big cell
|
|
line = " | ".join(tds[:9])
|
|
# file rows: surface the md5 (ads.php link) — feed to tools/lgdl.py bymd5
|
|
md5s = re.findall(r'ads\.php\?md5=([0-9a-f]{32})', r)
|
|
if md5s:
|
|
line += f" [md5={md5s[0]}]"
|
|
print(line[:300])
|
|
count += 1
|
|
if count >= limit:
|
|
break
|
|
if count == 0:
|
|
# maybe "No results" or error page
|
|
m = re.search(r"No results|Ничего не найдено|Error[^<]*", html)
|
|
print("(no results)" if m else "(unparsed — check manually)")
|
|
print(html[:500].replace("\n", " "))
|
|
|
|
if __name__ == "__main__":
|
|
main()
|