downloads: item 01 (Jung CW9/1) — 7 PDFs (5 EN Princeton/Routledge 2nd eds + 2 RU АСТ 2019/2020), all size-verified + content-verified; tools/lgdl.py downloader (ads.php card → get.php key → bytes, resume, backoff)
This commit is contained in:
parent
00af419b79
commit
e41464c2b4
2 changed files with 172 additions and 4 deletions
167
tools/lgdl.py
Normal file
167
tools/lgdl.py
Normal file
|
|
@ -0,0 +1,167 @@
|
|||
#!/usr/bin/env python3
|
||||
"""libgen.vg downloader (the box-verified flow, 2026-07-14).
|
||||
|
||||
Pipeline (curl engine — Python's DNS is flaky on this box, curl is not):
|
||||
search: tools/lg.py "query" -> edition ids
|
||||
ed: json.php?object=e&ids=E -> title/author/publisher/year + files{f_id, md5}
|
||||
f: json.php?object=f&addkeys=*&ids=F -> filesize, extension
|
||||
dl: ads.php?md5=M (UA+cookie+referer) -> card HTML with
|
||||
libgen.bz/get.php?md5=M&key=KEY -> one-time download URL
|
||||
curl that (same cookie jar) -> real bytes
|
||||
|
||||
Usage:
|
||||
tools/lgdl.py ed <edition_id> [more_ids...]
|
||||
tools/lgdl.py dl <file_id> <out_dir> <target_name>
|
||||
tools/lgdl.py md5 <file_id>
|
||||
|
||||
Verification on dl: final size must equal libgen filesize; magic-byte sniff
|
||||
(fb2/pdf/epub/doc) must not be HTML. Card+key are re-fetched up to 3x.
|
||||
Politeness: ~4s between libgen requests.
|
||||
"""
|
||||
import json, os, re, subprocess, sys, time
|
||||
|
||||
UA = ("Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
|
||||
"(KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36")
|
||||
BASE = "https://libgen.vg"
|
||||
JAR = "/tmp/lgdl_cookies.txt"
|
||||
|
||||
|
||||
def curl(url, out=None, referer=None, binary=False, timeout=240, tries=3, extra=None):
|
||||
cmd = ["curl", "-s", "--fail", "--max-time", str(timeout),
|
||||
"-c", JAR, "-b", JAR,
|
||||
"-H", "User-Agent: " + UA,
|
||||
"-H", "Accept: */*",
|
||||
"-H", "Accept-Language: ru,en;q=0.9",
|
||||
"-H", "Referer: " + (referer or BASE + "/"),
|
||||
"-L"]
|
||||
if extra:
|
||||
cmd += extra
|
||||
if out:
|
||||
cmd.append("-o")
|
||||
cmd.append(out)
|
||||
cmd.append(url)
|
||||
last = None
|
||||
for i in range(tries):
|
||||
p = subprocess.run(cmd, capture_output=True, timeout=timeout + 30)
|
||||
if p.returncode == 0:
|
||||
if out is None:
|
||||
data = p.stdout
|
||||
if not binary:
|
||||
try:
|
||||
data = data.decode("utf-8")
|
||||
except UnicodeDecodeError:
|
||||
data = data.decode("utf-8", "replace")
|
||||
return data
|
||||
return p.returncode
|
||||
last = p.returncode
|
||||
time.sleep(5 * (i + 1))
|
||||
raise RuntimeError(f"curl failed (code {last}) after {tries} tries: {url}")
|
||||
|
||||
|
||||
def jget(path):
|
||||
return json.loads(curl(BASE + "/" + path))
|
||||
|
||||
|
||||
def file_info(f_id):
|
||||
d = jget(f"json.php?object=f&addkeys=*&ids={f_id}")
|
||||
return d[str(f_id)]
|
||||
|
||||
|
||||
def edition(e_id):
|
||||
d = jget(f"json.php?object=e&ids={e_id}")
|
||||
e = d[str(e_id)]
|
||||
files = []
|
||||
for rel, f in (e.get("files") or {}).items():
|
||||
fi = file_info(f["f_id"])
|
||||
time.sleep(1)
|
||||
files.append({
|
||||
"f_id": f["f_id"], "md5": f["md5"],
|
||||
"ext": fi.get("extension"), "size": int(fi.get("filesize") or 0),
|
||||
})
|
||||
e["files"] = files
|
||||
return e
|
||||
|
||||
|
||||
def sniff(data: bytes) -> str:
|
||||
if data[:5] == b"<?xml" and b"FictionBook" in data[:500]:
|
||||
return "fb2"
|
||||
if data[:5] == b"%PDF-":
|
||||
return "pdf"
|
||||
if data[:4] == b"PK\x03\x04":
|
||||
return "epub/zip"
|
||||
if data[:8] == b"\xd0\xcf\x11\xe0\xa1\xb1\x1a\xe1":
|
||||
return "doc/ole"
|
||||
if b"<html" in data[:2000].lower():
|
||||
return "HTML-ERROR"
|
||||
return "unknown"
|
||||
|
||||
|
||||
def download(f_id, out_dir, name, tries=8):
|
||||
fi = file_info(f_id)
|
||||
md5, size, ext = fi["md5"], int(fi.get("filesize") or 0), fi.get("extension") or "bin"
|
||||
if not name.lower().endswith("." + ext.lower()):
|
||||
name = f"{name}.{ext}"
|
||||
os.makedirs(out_dir, exist_ok=True)
|
||||
out = os.path.join(out_dir, name)
|
||||
for t in range(1, tries + 1):
|
||||
card = None
|
||||
for c in range(5): # card fetch with backoff (libgen DB is flaky)
|
||||
try:
|
||||
card = curl(f"{BASE}/ads.php?md5={md5}",
|
||||
referer=f"{BASE}/file.php?id={f_id}", tries=1)
|
||||
break
|
||||
except RuntimeError:
|
||||
time.sleep(8 * (c + 1))
|
||||
if card is None:
|
||||
print(f" try {t}: card fetch failed 5x — back off harder")
|
||||
time.sleep(30)
|
||||
continue
|
||||
m = re.search(r'(https?://[^"\']*get\.php\?md5=' + md5 + r'&key=[^"\']+)', card)
|
||||
if not m:
|
||||
print(f" try {t}: no get.php link in card ({len(card)}b) — retry")
|
||||
time.sleep(4)
|
||||
continue
|
||||
url = m.group(1)
|
||||
tmp = out + ".part"
|
||||
rc = curl(url, out=tmp, referer=f"{BASE}/ads.php?md5={md5}", binary=True,
|
||||
timeout=900, extra=["-C", "-"])
|
||||
data = open(tmp, "rb").read()
|
||||
kind = sniff(data)
|
||||
ok_size = (size == 0) or (len(data) == size)
|
||||
print(f" try {t}: {len(data)}b (expected {size}) kind={kind}")
|
||||
if kind == "HTML-ERROR":
|
||||
os.remove(tmp)
|
||||
if kind == "HTML-ERROR" or not ok_size:
|
||||
time.sleep(4)
|
||||
continue
|
||||
os.replace(tmp, out)
|
||||
print(f" OK -> {out} ({len(data)}b, {kind})")
|
||||
return out
|
||||
print(f" FAILED after {tries} tries: file_id={f_id} md5={md5}")
|
||||
return None
|
||||
|
||||
|
||||
def main():
|
||||
if len(sys.argv) < 2:
|
||||
sys.exit(__doc__)
|
||||
cmd = sys.argv[1]
|
||||
if cmd == "ed":
|
||||
for e_id in sys.argv[2:]:
|
||||
e = edition(e_id)
|
||||
keep = {k: e[k] for k in ("title", "title_add", "author", "publisher",
|
||||
"year", "pages", "series_name", "libgen_topic") if e.get(k)}
|
||||
print(f"[{e_id}] {json.dumps(keep, ensure_ascii=False)}")
|
||||
for f in e["files"]:
|
||||
print(f" f_id={f['f_id']} {f['size']:>10}b {f['ext']:<6} md5={f['md5']}")
|
||||
time.sleep(1)
|
||||
elif cmd == "dl":
|
||||
f_id, out_dir, name = sys.argv[2], sys.argv[3], sys.argv[4]
|
||||
download(f_id, out_dir, name)
|
||||
elif cmd == "md5":
|
||||
print(file_info(sys.argv[2])["md5"])
|
||||
else:
|
||||
sys.exit(__doc__)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Loading…
Add table
Add a link
Reference in a new issue