- USER FINDING: libgen edition files map {key: {f_id, md5}} — KEY and f_id VALUE
are DIFFERENT file records; we were downloading/prechecking by the KEY (wrong file).
- Recovered by f_id: 09-t1 (Eliade UCP 1978), 13 Sungod (Cornell 2010), 14 Solomon
(Karnac 2007), 17 God-Image (Inner Light 1992), 19 Hornung (UNC 1982), 20 Psyche in
Scripture (1995), 23 Dionysos (Bollingen LXV/2 Princeton) + backups 12/16/24/25/26/27.
All content-verified. '24 systemic mislinks' = our bug, not libgen.
- lg.py: objects[]=f,e,s,a,p,w + topics[] (file-level search; found Solomon md5) +
md5 in output. lgdl.py: bymd5 subcommand.
- precheck v2: rev-value match = OK; locator words >=2 shared = OK; words present
0 shared = BAD; hash/ISBN/empty locator = no signal (SUSPECT); digit-runs masked
before tokenization (hex 'aaae' trap).
- MANIFEST/SUMMARY/cards/AGENTS.md updated. 34 files, 608 MB (28 backup still downloading).
368 lines
16 KiB
Python
368 lines
16 KiB
Python
#!/usr/bin/env python3
|
|
"""libgen.vg downloader (the box-verified flow, 2026-07-14).
|
|
|
|
Pipeline (curl engine — Python's DNS is flaky on this box, curl is not):
|
|
search: tools/lg.py "query" -> edition ids
|
|
ed: json.php?object=e&ids=E -> title/author/publisher/year + files{f_id, md5}
|
|
f: json.php?object=f&addkeys=*&ids=F -> filesize, extension
|
|
dl: ads.php?md5=M (UA+cookie+referer) -> card HTML with
|
|
libgen.bz/get.php?md5=M&key=KEY -> one-time download URL
|
|
curl that (same cookie jar) -> real bytes
|
|
|
|
Usage:
|
|
tools/lgdl.py ed <edition_id> [more_ids...]
|
|
tools/lgdl.py dl <file_id> <out_dir> <target_name> [edition_id]
|
|
tools/lgdl.py md5 <file_id>
|
|
tools/lgdl.py precheck <file_id> <edition_id> [expected_pages]
|
|
|
|
precheck (no download, ~2 libgen calls): file record vs the target edition.
|
|
The ONLY reliable ownership signal is the file `locator` (the importer's
|
|
original Windows path — it names the real book). libgen's `file.editions`
|
|
reverse-map uses internal IDs that do NOT resolve via object=e (they return
|
|
unrelated books), so it is logged as a hint but never gates the verdict.
|
|
RED (exit 2) — locator non-empty and shares NO content word with the target
|
|
title/author (locator names a different book), or a foreign-language
|
|
batch path for a Cyrillic target.
|
|
OK (exit 0) — locator non-empty and shares content words with the target.
|
|
SUSPECT(exit 1) — locator empty (no signal), too few target tokens to judge,
|
|
or size implausible vs expected pages. ALWAYS post-verify content.
|
|
`dl` with the optional 4th arg runs precheck first and aborts on RED.
|
|
|
|
Verification on dl: final size must equal libgen filesize; magic-byte sniff
|
|
(fb2/pdf/epub/doc) must not be HTML. Card+key are re-fetched up to 3x.
|
|
Content-verify after dl (AGENTS.md lesson 8): pdfinfo pages vs RSL + first-page
|
|
text must contain the RU title/author. Politeness: ~4s between libgen requests.
|
|
"""
|
|
import json, os, re, subprocess, sys, time
|
|
|
|
UA = ("Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
|
|
"(KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36")
|
|
BASE = "https://libgen.vg"
|
|
JAR = "/tmp/lgdl_cookies.txt"
|
|
|
|
|
|
def curl(url, out=None, referer=None, binary=False, timeout=240, tries=3, extra=None,
|
|
connect_timeout=None):
|
|
cmd = ["curl", "-s", "--fail", "--max-time", str(timeout),
|
|
"-c", JAR, "-b", JAR,
|
|
"-H", "User-Agent: " + UA,
|
|
"-H", "Accept: */*",
|
|
"-H", "Accept-Language: ru,en;q=0.9",
|
|
"-H", "Referer: " + (referer or BASE + "/"),
|
|
"-L"]
|
|
if connect_timeout:
|
|
cmd += ["--connect-timeout", str(connect_timeout)]
|
|
if extra:
|
|
cmd += extra
|
|
if out:
|
|
cmd.append("-o")
|
|
cmd.append(out)
|
|
cmd.append(url)
|
|
last = None
|
|
for i in range(tries):
|
|
p = subprocess.run(cmd, capture_output=True, timeout=timeout + 30)
|
|
if p.returncode == 0:
|
|
if out is None:
|
|
data = p.stdout
|
|
if not binary:
|
|
try:
|
|
data = data.decode("utf-8")
|
|
except UnicodeDecodeError:
|
|
data = data.decode("utf-8", "replace")
|
|
return data
|
|
return p.returncode
|
|
last = p.returncode
|
|
time.sleep(5 * (i + 1))
|
|
raise RuntimeError(f"curl failed (code {last}) after {tries} tries: {url}")
|
|
|
|
|
|
def jget(path):
|
|
return json.loads(curl(BASE + "/" + path))
|
|
|
|
|
|
def file_info(f_id):
|
|
d = jget(f"json.php?object=f&addkeys=*&ids={f_id}")
|
|
return d[str(f_id)]
|
|
|
|
|
|
def edition(e_id):
|
|
d = jget(f"json.php?object=e&ids={e_id}")
|
|
e = d[str(e_id)]
|
|
files = []
|
|
for rel, f in (e.get("files") or {}).items():
|
|
fi = file_info(f["f_id"])
|
|
time.sleep(1)
|
|
files.append({
|
|
"f_id": f["f_id"], "md5": f["md5"],
|
|
"ext": fi.get("extension"), "size": int(fi.get("filesize") or 0),
|
|
})
|
|
e["files"] = files
|
|
return e
|
|
|
|
|
|
def sniff(data: bytes) -> str:
|
|
if data[:5] == b"<?xml" and b"FictionBook" in data[:500]:
|
|
return "fb2"
|
|
if data[:5] == b"%PDF-":
|
|
return "pdf"
|
|
if data[:4] == b"PK\x03\x04":
|
|
return "epub/zip"
|
|
if data[:8] == b"\xd0\xcf\x11\xe0\xa1\xb1\x1a\xe1":
|
|
return "doc/ole"
|
|
if data[:8] in (b"AT&TFORM", b"AT&ToC\01", b"AT&ToC\02"):
|
|
return "djvu"
|
|
if b"<html" in data[:2000].lower():
|
|
return "HTML-ERROR"
|
|
return "unknown"
|
|
|
|
|
|
def download(f_id, out_dir, name, tries=16):
|
|
fi = file_info(f_id)
|
|
md5, size, ext = fi["md5"], int(fi.get("filesize") or 0), fi.get("extension") or "bin"
|
|
if not name.lower().endswith("." + ext.lower()):
|
|
name = f"{name}.{ext}"
|
|
os.makedirs(out_dir, exist_ok=True)
|
|
out = os.path.join(out_dir, name)
|
|
stall = 0
|
|
for t in range(1, tries + 1):
|
|
card = None
|
|
print(f" try {t}: card fetch start")
|
|
for c in range(5): # card fetch with backoff (libgen DB is flaky)
|
|
try:
|
|
card = curl(f"{BASE}/ads.php?md5={md5}",
|
|
referer=f"{BASE}/file.php?id={f_id}", tries=1,
|
|
timeout=90, connect_timeout=20)
|
|
break
|
|
except RuntimeError:
|
|
print(f" try {t}/{c+1}: card fetch failed — back off")
|
|
time.sleep(8 * (c + 1))
|
|
if card is None:
|
|
print(f" try {t}: card fetch failed 5x — back off harder")
|
|
time.sleep(30)
|
|
continue
|
|
m = re.search(r'(https?://[^"\']*get\.php\?md5=' + md5 + r'&key=[^"\']+)', card)
|
|
if not m:
|
|
print(f" try {t}: no get.php link in card ({len(card)}b) — retry")
|
|
time.sleep(4)
|
|
continue
|
|
url = m.group(1)
|
|
# speed probe: 6s range fetch on the fresh CDN edge; <15KB/s => this
|
|
# edge is wedged, re-draw a new get.php link (libgen 302s to different
|
|
# storage hosts per request)
|
|
probe_ok = False
|
|
for p in range(3):
|
|
try:
|
|
pr = subprocess.run(["curl", "-s", "--fail", "-o", "/dev/null",
|
|
"-r", "0-300000", "--max-time", "6",
|
|
"--connect-timeout", "10",
|
|
"-H", "User-Agent: " + UA,
|
|
"-H", "Referer: " + f"{BASE}/ads.php?md5={md5}",
|
|
"-w", "%{size_download} %{speed_download}",
|
|
url], capture_output=True, text=True, timeout=20)
|
|
if pr.returncode == 0:
|
|
parts = pr.stdout.split()
|
|
spd = float(parts[1]) if len(parts) > 1 else 0
|
|
print(f" try {t}: probe edge {p+1}: {spd/1024:.0f} KB/s")
|
|
if spd >= 15 * 1024:
|
|
probe_ok = True
|
|
break
|
|
except Exception:
|
|
pass
|
|
print(f" try {t}: edge {p+1} slow/wedged — re-drawing get.php link")
|
|
time.sleep(3)
|
|
try:
|
|
card2 = curl(f"{BASE}/ads.php?md5={md5}",
|
|
referer=f"{BASE}/file.php?id={f_id}", tries=1,
|
|
timeout=90, connect_timeout=20)
|
|
m2 = re.search(r'(https?://[^"\']*get\.php\?md5=' + md5 + r'&key=[^"\']+)', card2)
|
|
if m2:
|
|
url = m2.group(1)
|
|
except RuntimeError:
|
|
pass
|
|
if not probe_ok:
|
|
print(f" try {t}: all 3 edges slow — waiting out the CDN")
|
|
tmp = out + ".part"
|
|
prev = os.path.getsize(tmp) if os.path.exists(tmp) else 0
|
|
print(f" try {t}: data fetch start ({prev}b resumed)")
|
|
try:
|
|
rc = curl(url, out=tmp, referer=f"{BASE}/ads.php?md5={md5}", binary=True,
|
|
timeout=3600, connect_timeout=30,
|
|
# abort hung connections: <500B/s for 300s
|
|
extra=["-C", "-", "--speed-time", "300", "--speed-limit", "500"])
|
|
except RuntimeError as e:
|
|
print(f" try {t}: fetch error ({e}) — back off")
|
|
time.sleep(45)
|
|
continue
|
|
cur = os.path.getsize(tmp) if os.path.exists(tmp) else 0
|
|
# resume-hang detection: zero-progress resume at the same offset =>
|
|
# this CDN offset is wedged; wipe and start fresh (first stall is enough)
|
|
if 0 < prev == cur:
|
|
print(f" try {t}: resume stalled at {prev}b — wiping part, fresh start")
|
|
os.remove(tmp)
|
|
stall = 0
|
|
cur = 0
|
|
data = open(tmp, "rb").read()
|
|
kind = sniff(data)
|
|
ok_size = (size == 0) or (len(data) == size)
|
|
print(f" try {t}: {len(data)}b (expected {size}) kind={kind}")
|
|
if kind == "HTML-ERROR":
|
|
os.remove(tmp)
|
|
if kind == "HTML-ERROR" or not ok_size:
|
|
time.sleep(4)
|
|
continue
|
|
os.replace(tmp, out)
|
|
print(f" OK -> {out} ({len(data)}b, {kind})")
|
|
return out
|
|
print(f" FAILED after {tries} tries: file_id={f_id} md5={md5}")
|
|
return None
|
|
|
|
|
|
# stop-ish / generic tokens to ignore when matching locator vs target
|
|
_STOP = set("the of and in to a an for with from on by edition new second first "
|
|
"vol volume pdf epub fb2 doc djvu mobi txt part pt series book "
|
|
"translated translation revised updated complete full original"
|
|
.split())
|
|
|
|
def _tokens(s):
|
|
return {w for w in re.findall(r"[a-z\u0400-\u04ff]{4,}", (s or "").lower())
|
|
if w not in _STOP}
|
|
|
|
def _loc_tokens(s):
|
|
# locator words: mask alphanumeric runs containing digits (IDs/hashes/ISBNs)
|
|
# so "d1e2f4...aaae..." does not yield fake word 'aaae'
|
|
masked = re.sub(r"[a-z0-9]+\d[a-z0-9]*", " ", (s or "").lower())
|
|
return _tokens(masked)
|
|
|
|
def precheck(f_id, e_id, expected_pages=None):
|
|
"""Mislinked-file gate. Returns 0=OK, 1=SUSPECT, 2=BAD (see module doc).
|
|
|
|
CRITICAL: pass the edition files-map VALUE (`f_id`), not the map key —
|
|
the key is a different (often stale/mislinked) file record (verified
|
|
2026-09-25: Solomon — key 92709688 = Russian book, f_id 93253044 = real book).
|
|
|
|
Signals:
|
|
1) reverse map: for a real f_id record the `editions` values ARE the public
|
|
edition IDs — value == target is a strong OK signal; for a key's record
|
|
it contains other numbers (no match => weak negative, never hard BAD).
|
|
2) locator: words in the locator vs target title/author words.
|
|
- no content words (hash/ISBN/number) => no signal (SUSPECT)
|
|
- >=2 shared words (or 1 for a short target) => OK
|
|
- words present, none shared => BAD (names a different book)
|
|
ALWAYS content-verify after download (ground truth).
|
|
"""
|
|
fi = file_info(f_id)
|
|
time.sleep(2)
|
|
ext, size = fi.get("extension"), int(fi.get("filesize") or 0)
|
|
locator = (fi.get("locator") or "").strip()
|
|
reasons = []
|
|
|
|
rev = {k: v.get("e_id") for k, v in (fi.get("editions") or {}).items()}
|
|
rev_match = int(e_id) in {int(v) for v in rev.values() if v}
|
|
if rev:
|
|
reasons.append(f"reverse-map: {rev} {'MATCHES target' if rev_match else '(no target)'}")
|
|
|
|
try:
|
|
ed = jget(f"json.php?object=e&addkeys=title,author&ids={e_id}").get(str(e_id), {})
|
|
time.sleep(2)
|
|
target = f"{ed.get('title','')} {ed.get('author','')}"
|
|
except Exception:
|
|
target = ""
|
|
t_tok, l_tok = _tokens(target), _loc_tokens(locator)
|
|
shared = t_tok & l_tok
|
|
|
|
if not locator or not l_tok:
|
|
loc_signal = "none"
|
|
reasons.append(f"locator has no content words (hash/ISBN/empty) — no signal: {locator[:90]}")
|
|
elif len(shared) >= 2 or (len(shared) == 1 and len(t_tok) <= 3):
|
|
loc_signal = "match"
|
|
reasons.append(f"locator names the target (shared: {sorted(shared)}): {locator[:120]}")
|
|
else:
|
|
loc_signal = "mismatch"
|
|
low = locator.lower()
|
|
target_cyr = re.search(r"[\u0400-\u04FF]", target)
|
|
cyrillic_loc = re.search(r"[\u0400-\u04FF]", locator)
|
|
foreign = re.search(r"(spa\\|spanish|english|epublibre|0day|\\fiction\\)", low)
|
|
if target_cyr and not cyrillic_loc:
|
|
reasons.append(f"Cyrillic target but locator is non-Cyrillic: {locator[:120]}")
|
|
elif foreign:
|
|
reasons.append(f"locator is a foreign batch naming another book: {locator[:120]}")
|
|
else:
|
|
reasons.append(f"locator names a DIFFERENT book (no shared words): {locator[:120]}")
|
|
|
|
if rev_match and loc_signal != "mismatch":
|
|
verdict = "OK"
|
|
elif loc_signal == "match":
|
|
verdict = "OK"
|
|
elif loc_signal == "mismatch" and not rev_match:
|
|
verdict = "BAD"
|
|
else:
|
|
verdict = "SUSPECT" # none/none, or contradiction (rev match vs locator mismatch)
|
|
if rev_match and loc_signal == "mismatch":
|
|
reasons.append("CONTRADICTION: reverse-map matches but locator names another book — verify content")
|
|
|
|
# size sanity vs expected pages (text ~200B/pp; scans 50KB..15MB/pp)
|
|
if expected_pages and size:
|
|
per_lo = 200 if ext in ("fb2", "epub", "txt", "html") else 10_000
|
|
lo, hi = expected_pages * per_lo, expected_pages * 15_000_000
|
|
if size < lo:
|
|
if verdict == "OK":
|
|
verdict = "SUSPECT"
|
|
reasons.append(f"size {size}b < {lo}b for ~{expected_pages} pp (too small — fragment?)")
|
|
elif size > hi:
|
|
if verdict == "OK":
|
|
verdict = "SUSPECT"
|
|
reasons.append(f"size {size}b > {hi}b for ~{expected_pages} pp (too big — scan mismatch?)")
|
|
|
|
print(f"precheck f_id={f_id} vs e_id={e_id}: {verdict} [{ext}, {size}b]")
|
|
for r in reasons:
|
|
print(f" - {r}")
|
|
return {"OK": 0, "SUSPECT": 1, "BAD": 2}[verdict]
|
|
|
|
|
|
def main():
|
|
if len(sys.argv) < 2:
|
|
sys.exit(__doc__)
|
|
cmd = sys.argv[1]
|
|
if cmd == "ed":
|
|
for e_id in sys.argv[2:]:
|
|
e = edition(e_id)
|
|
keep = {k: e[k] for k in ("title", "title_add", "author", "publisher",
|
|
"year", "pages", "series_name", "libgen_topic") if e.get(k)}
|
|
print(f"[{e_id}] {json.dumps(keep, ensure_ascii=False)}")
|
|
for f in e["files"]:
|
|
print(f" f_id={f['f_id']} {f['size']:>10}b {f['ext']:<6} md5={f['md5']}")
|
|
time.sleep(1)
|
|
elif cmd == "dl":
|
|
f_id, out_dir, name = sys.argv[2], sys.argv[3], sys.argv[4]
|
|
if len(sys.argv) > 5 and sys.argv[5].isdigit():
|
|
rc = precheck(f_id, sys.argv[5])
|
|
if rc == 2:
|
|
print(" ABORT: precheck RED (mislinked file) — pick another f_id/edition or flib")
|
|
sys.exit(2)
|
|
if rc == 1:
|
|
print(" precheck SUSPECT — downloading, MUST post-verify content")
|
|
time.sleep(3)
|
|
download(f_id, out_dir, name)
|
|
elif cmd == "precheck":
|
|
f_id, e_id = sys.argv[2], sys.argv[3]
|
|
pages = int(sys.argv[4]) if len(sys.argv) > 4 else None
|
|
sys.exit(precheck(f_id, e_id, pages))
|
|
elif cmd == "md5":
|
|
print(file_info(sys.argv[2])["md5"])
|
|
elif cmd == "bymd5":
|
|
# md5 -> f_id + metadata (for files found via lg.py file search)
|
|
d = jget(f"json.php?object=f&addkeys=*&md5={sys.argv[2]}")
|
|
if not d:
|
|
print("not found"); sys.exit(1)
|
|
fid = list(d.keys())[0]
|
|
f = d[fid]
|
|
print(f"f_id={fid} {f.get('extension')} {int(f.get('filesize') or 0)}b pages={f.get('pages')} ocr={f.get('ocr')} added={f.get('time_added')}")
|
|
print(f"locator: {f.get('locator','')}")
|
|
rev = {k: v.get('e_id') for k, v in (f.get('editions') or {}).items()}
|
|
if rev: print(f"editions (hint): {rev}")
|
|
else:
|
|
sys.exit(__doc__)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|