jung/tools/lgdl.py
Dmitry Kokorin 2a2699d61a sec05: EN downloads complete (30 files) + SUMMARY at-a-glance + lgdl queue infra
- 30/31 EN candidates downloaded via libgen, all content-verified
  (magic+pages+first-page text); 08b 37-pp fragment discarded
- EN gaps: 04/05/15/16/23 not found (rare academic), 03/14 ia-restricted
- MANIFEST.md: EN sections per item + not-downloadable extended
- 26 cards: Downloads tables w/ EN rows + EN NOT FOUND notes (6 cards)
- SUMMARY.md created (canon v1): at-a-glance table 29 items, EN->RU authors
- tools/lgdl.py: speed-probe + CDN-edge re-draw (per-request get.php
  302s to different storage hosts), connect timeout, stall wipe on first stall
- AGENTS.md: Lesson 9 (EN-queue infra: ads.php throttle, watchdog,
  unbuffered logs, dual-queue trap, AA 403, i2p verdict)
2026-09-24 00:04:13 +03:00

316 lines
13 KiB
Python

#!/usr/bin/env python3
"""libgen.vg downloader (the box-verified flow, 2026-07-14).
Pipeline (curl engine — Python's DNS is flaky on this box, curl is not):
search: tools/lg.py "query" -> edition ids
ed: json.php?object=e&ids=E -> title/author/publisher/year + files{f_id, md5}
f: json.php?object=f&addkeys=*&ids=F -> filesize, extension
dl: ads.php?md5=M (UA+cookie+referer) -> card HTML with
libgen.bz/get.php?md5=M&key=KEY -> one-time download URL
curl that (same cookie jar) -> real bytes
Usage:
tools/lgdl.py ed <edition_id> [more_ids...]
tools/lgdl.py dl <file_id> <out_dir> <target_name> [edition_id]
tools/lgdl.py md5 <file_id>
tools/lgdl.py precheck <file_id> <edition_id> [expected_pages]
precheck (no download, ~2 libgen calls): file record vs the target edition.
RED (exit 2) — reverse-map `editions` non-empty and does NOT contain the
target edition (file belongs to another book), or locator path is a
foreign-language batch (spa/English/fiction) for a Cyrillic book.
YELLOW(exit 1) — reverse map empty, or size implausible vs expected pages.
OK (exit 0) — reverse map matches target, or empty with plausible locator.
`dl` with the optional 4th arg runs precheck first and aborts on RED.
Verification on dl: final size must equal libgen filesize; magic-byte sniff
(fb2/pdf/epub/doc) must not be HTML. Card+key are re-fetched up to 3x.
Content-verify after dl (AGENTS.md lesson 8): pdfinfo pages vs RSL + first-page
text must contain the RU title/author. Politeness: ~4s between libgen requests.
"""
import json, os, re, subprocess, sys, time
UA = ("Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36")
BASE = "https://libgen.vg"
JAR = "/tmp/lgdl_cookies.txt"
def curl(url, out=None, referer=None, binary=False, timeout=240, tries=3, extra=None,
connect_timeout=None):
cmd = ["curl", "-s", "--fail", "--max-time", str(timeout),
"-c", JAR, "-b", JAR,
"-H", "User-Agent: " + UA,
"-H", "Accept: */*",
"-H", "Accept-Language: ru,en;q=0.9",
"-H", "Referer: " + (referer or BASE + "/"),
"-L"]
if connect_timeout:
cmd += ["--connect-timeout", str(connect_timeout)]
if extra:
cmd += extra
if out:
cmd.append("-o")
cmd.append(out)
cmd.append(url)
last = None
for i in range(tries):
p = subprocess.run(cmd, capture_output=True, timeout=timeout + 30)
if p.returncode == 0:
if out is None:
data = p.stdout
if not binary:
try:
data = data.decode("utf-8")
except UnicodeDecodeError:
data = data.decode("utf-8", "replace")
return data
return p.returncode
last = p.returncode
time.sleep(5 * (i + 1))
raise RuntimeError(f"curl failed (code {last}) after {tries} tries: {url}")
def jget(path):
return json.loads(curl(BASE + "/" + path))
def file_info(f_id):
d = jget(f"json.php?object=f&addkeys=*&ids={f_id}")
return d[str(f_id)]
def edition(e_id):
d = jget(f"json.php?object=e&ids={e_id}")
e = d[str(e_id)]
files = []
for rel, f in (e.get("files") or {}).items():
fi = file_info(f["f_id"])
time.sleep(1)
files.append({
"f_id": f["f_id"], "md5": f["md5"],
"ext": fi.get("extension"), "size": int(fi.get("filesize") or 0),
})
e["files"] = files
return e
def sniff(data: bytes) -> str:
if data[:5] == b"<?xml" and b"FictionBook" in data[:500]:
return "fb2"
if data[:5] == b"%PDF-":
return "pdf"
if data[:4] == b"PK\x03\x04":
return "epub/zip"
if data[:8] == b"\xd0\xcf\x11\xe0\xa1\xb1\x1a\xe1":
return "doc/ole"
if data[:8] in (b"AT&TFORM", b"AT&ToC\01", b"AT&ToC\02"):
return "djvu"
if b"<html" in data[:2000].lower():
return "HTML-ERROR"
return "unknown"
def download(f_id, out_dir, name, tries=16):
fi = file_info(f_id)
md5, size, ext = fi["md5"], int(fi.get("filesize") or 0), fi.get("extension") or "bin"
if not name.lower().endswith("." + ext.lower()):
name = f"{name}.{ext}"
os.makedirs(out_dir, exist_ok=True)
out = os.path.join(out_dir, name)
stall = 0
for t in range(1, tries + 1):
card = None
print(f" try {t}: card fetch start")
for c in range(5): # card fetch with backoff (libgen DB is flaky)
try:
card = curl(f"{BASE}/ads.php?md5={md5}",
referer=f"{BASE}/file.php?id={f_id}", tries=1,
timeout=90, connect_timeout=20)
break
except RuntimeError:
print(f" try {t}/{c+1}: card fetch failed — back off")
time.sleep(8 * (c + 1))
if card is None:
print(f" try {t}: card fetch failed 5x — back off harder")
time.sleep(30)
continue
m = re.search(r'(https?://[^"\']*get\.php\?md5=' + md5 + r'&key=[^"\']+)', card)
if not m:
print(f" try {t}: no get.php link in card ({len(card)}b) — retry")
time.sleep(4)
continue
url = m.group(1)
# speed probe: 6s range fetch on the fresh CDN edge; <15KB/s => this
# edge is wedged, re-draw a new get.php link (libgen 302s to different
# storage hosts per request)
probe_ok = False
for p in range(3):
try:
pr = subprocess.run(["curl", "-s", "--fail", "-o", "/dev/null",
"-r", "0-300000", "--max-time", "6",
"--connect-timeout", "10",
"-H", "User-Agent: " + UA,
"-H", "Referer: " + f"{BASE}/ads.php?md5={md5}",
"-w", "%{size_download} %{speed_download}",
url], capture_output=True, text=True, timeout=20)
if pr.returncode == 0:
parts = pr.stdout.split()
spd = float(parts[1]) if len(parts) > 1 else 0
print(f" try {t}: probe edge {p+1}: {spd/1024:.0f} KB/s")
if spd >= 15 * 1024:
probe_ok = True
break
except Exception:
pass
print(f" try {t}: edge {p+1} slow/wedged — re-drawing get.php link")
time.sleep(3)
try:
card2 = curl(f"{BASE}/ads.php?md5={md5}",
referer=f"{BASE}/file.php?id={f_id}", tries=1,
timeout=90, connect_timeout=20)
m2 = re.search(r'(https?://[^"\']*get\.php\?md5=' + md5 + r'&key=[^"\']+)', card2)
if m2:
url = m2.group(1)
except RuntimeError:
pass
if not probe_ok:
print(f" try {t}: all 3 edges slow — waiting out the CDN")
tmp = out + ".part"
prev = os.path.getsize(tmp) if os.path.exists(tmp) else 0
print(f" try {t}: data fetch start ({prev}b resumed)")
try:
rc = curl(url, out=tmp, referer=f"{BASE}/ads.php?md5={md5}", binary=True,
timeout=3600, connect_timeout=30,
# abort hung connections: <500B/s for 300s
extra=["-C", "-", "--speed-time", "300", "--speed-limit", "500"])
except RuntimeError as e:
print(f" try {t}: fetch error ({e}) — back off")
time.sleep(45)
continue
cur = os.path.getsize(tmp) if os.path.exists(tmp) else 0
# resume-hang detection: zero-progress resume at the same offset =>
# this CDN offset is wedged; wipe and start fresh (first stall is enough)
if 0 < prev == cur:
print(f" try {t}: resume stalled at {prev}b — wiping part, fresh start")
os.remove(tmp)
stall = 0
cur = 0
data = open(tmp, "rb").read()
kind = sniff(data)
ok_size = (size == 0) or (len(data) == size)
print(f" try {t}: {len(data)}b (expected {size}) kind={kind}")
if kind == "HTML-ERROR":
os.remove(tmp)
if kind == "HTML-ERROR" or not ok_size:
time.sleep(4)
continue
os.replace(tmp, out)
print(f" OK -> {out} ({len(data)}b, {kind})")
return out
print(f" FAILED after {tries} tries: file_id={f_id} md5={md5}")
return None
def precheck(f_id, e_id, expected_pages=None):
"""Mislinked-file gate. Returns 0=OK, 1=SUSPECT, 2=BAD (see module doc)."""
fi = file_info(f_id)
time.sleep(2)
ext, size = fi.get("extension"), int(fi.get("filesize") or 0)
locator = fi.get("locator") or ""
rev = {eid: int(x.get("e_id") or 0) for eid, x in (fi.get("editions") or {}).items()}
rev = {k: v for k, v in rev.items() if v}
verdict, reasons = "OK", []
# 1) reverse map: who does this file BELONG to?
rev_match = None
if rev:
if int(e_id) in rev.values():
rev_match = True
reasons.append(f"reverse-map matches target e_id {e_id}")
else:
verdict = "BAD"
owners = []
for owner_eid in set(rev.values()):
try:
oe = jget(f"json.php?object=e&addkeys=title,author,year&ids={owner_eid}")
o = oe.get(str(owner_eid), {})
owners.append(f"e_id {owner_eid} «{o.get('title')}» {o.get('author')} {o.get('year')}")
time.sleep(2)
except Exception:
owners.append(f"e_id {owner_eid} (?)")
reasons.append("reverse-map points to ANOTHER edition: " + "; ".join(owners))
else:
rev_match = False
verdict = "SUSPECT" if verdict == "OK" else verdict
reasons.append("reverse-map EMPTY (old file) — rely on locator + post-verify")
# 2) locator: foreign-language batch path. Only meaningful when the reverse
# map does NOT confirm the target (matching map = the edition itself comes
# from that batch — an origin note, not a mislink).
low = locator.lower()
foreign = re.search(r"(spa\\|spanish|english|epublibre|0day|\\fiction\\)", low)
cyrillic = re.search(r"[\u0400-\u04FF]", locator)
if locator and foreign and not cyrillic:
if rev_match is True:
reasons.append(f"origin note: edition comes from a foreign batch: {locator[:120]}")
else:
verdict = "BAD" if verdict == "OK" else verdict
reasons.append(f"locator is a foreign batch path: {locator[:120]}")
elif locator and rev_match is not True:
reasons.append(f"locator ok: {locator[:120]}")
# 3) size sanity vs expected pages (text formats are tiny: ~200B/pp;
# scans: 50KB..15MB/pp)
if expected_pages and size:
per_lo = 200 if ext in ("fb2", "epub", "txt", "html") else 10_000
lo, hi = expected_pages * per_lo, expected_pages * 15_000_000
if size < lo:
verdict = "SUSPECT" if verdict == "OK" else verdict
reasons.append(f"size {size}b < {lo}b for ~{expected_pages} pp (too small — fragment?)")
elif size > hi:
verdict = "SUSPECT" if verdict == "OK" else verdict
reasons.append(f"size {size}b > {hi}b for ~{expected_pages} pp (too big — scan mismatch?)")
print(f"precheck f_id={f_id} vs e_id={e_id}: {verdict} [{ext}, {size}b]")
for r in reasons:
print(f" - {r}")
return {"OK": 0, "SUSPECT": 1, "BAD": 2}[verdict]
def main():
if len(sys.argv) < 2:
sys.exit(__doc__)
cmd = sys.argv[1]
if cmd == "ed":
for e_id in sys.argv[2:]:
e = edition(e_id)
keep = {k: e[k] for k in ("title", "title_add", "author", "publisher",
"year", "pages", "series_name", "libgen_topic") if e.get(k)}
print(f"[{e_id}] {json.dumps(keep, ensure_ascii=False)}")
for f in e["files"]:
print(f" f_id={f['f_id']} {f['size']:>10}b {f['ext']:<6} md5={f['md5']}")
time.sleep(1)
elif cmd == "dl":
f_id, out_dir, name = sys.argv[2], sys.argv[3], sys.argv[4]
if len(sys.argv) > 5 and sys.argv[5].isdigit():
rc = precheck(f_id, sys.argv[5])
if rc == 2:
print(" ABORT: precheck RED (mislinked file) — pick another f_id/edition or flib")
sys.exit(2)
if rc == 1:
print(" precheck SUSPECT — downloading, MUST post-verify content")
time.sleep(3)
download(f_id, out_dir, name)
elif cmd == "precheck":
f_id, e_id = sys.argv[2], sys.argv[3]
pages = int(sys.argv[4]) if len(sys.argv) > 4 else None
sys.exit(precheck(f_id, e_id, pages))
elif cmd == "md5":
print(file_info(sys.argv[2])["md5"])
else:
sys.exit(__doc__)
if __name__ == "__main__":
main()