#!/usr/bin/env python3 """libgen.vg downloader (the box-verified flow, 2026-07-14). Pipeline (curl engine — Python's DNS is flaky on this box, curl is not): search: tools/lg.py "query" -> edition ids ed: json.php?object=e&ids=E -> title/author/publisher/year + files{f_id, md5} f: json.php?object=f&addkeys=*&ids=F -> filesize, extension dl: ads.php?md5=M (UA+cookie+referer) -> card HTML with libgen.bz/get.php?md5=M&key=KEY -> one-time download URL curl that (same cookie jar) -> real bytes Usage: tools/lgdl.py ed [more_ids...] tools/lgdl.py dl [edition_id] tools/lgdl.py md5 tools/lgdl.py precheck [expected_pages] precheck (no download, ~2 libgen calls): file record vs the target edition. The ONLY reliable ownership signal is the file `locator` (the importer's original Windows path — it names the real book). libgen's `file.editions` reverse-map uses internal IDs that do NOT resolve via object=e (they return unrelated books), so it is logged as a hint but never gates the verdict. RED (exit 2) — locator non-empty and shares NO content word with the target title/author (locator names a different book), or a foreign-language batch path for a Cyrillic target. OK (exit 0) — locator non-empty and shares content words with the target. SUSPECT(exit 1) — locator empty (no signal), too few target tokens to judge, or size implausible vs expected pages. ALWAYS post-verify content. `dl` with the optional 4th arg runs precheck first and aborts on RED. Verification on dl: final size must equal libgen filesize; magic-byte sniff (fb2/pdf/epub/doc) must not be HTML. Card+key are re-fetched up to 3x. Content-verify after dl (AGENTS.md lesson 8): pdfinfo pages vs RSL + first-page text must contain the RU title/author. Politeness: ~4s between libgen requests. """ import json, os, re, subprocess, sys, time UA = ("Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 " "(KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36") BASE = "https://libgen.vg" JAR = "/tmp/lgdl_cookies.txt" def curl(url, out=None, referer=None, binary=False, timeout=240, tries=3, extra=None, connect_timeout=None): cmd = ["curl", "-s", "--fail", "--max-time", str(timeout), "-c", JAR, "-b", JAR, "-H", "User-Agent: " + UA, "-H", "Accept: */*", "-H", "Accept-Language: ru,en;q=0.9", "-H", "Referer: " + (referer or BASE + "/"), "-L"] if connect_timeout: cmd += ["--connect-timeout", str(connect_timeout)] if extra: cmd += extra if out: cmd.append("-o") cmd.append(out) cmd.append(url) last = None for i in range(tries): p = subprocess.run(cmd, capture_output=True, timeout=timeout + 30) if p.returncode == 0: if out is None: data = p.stdout if not binary: try: data = data.decode("utf-8") except UnicodeDecodeError: data = data.decode("utf-8", "replace") return data return p.returncode last = p.returncode time.sleep(5 * (i + 1)) raise RuntimeError(f"curl failed (code {last}) after {tries} tries: {url}") def jget(path): return json.loads(curl(BASE + "/" + path)) def file_info(f_id): d = jget(f"json.php?object=f&addkeys=*&ids={f_id}") return d[str(f_id)] def edition(e_id): d = jget(f"json.php?object=e&ids={e_id}") e = d[str(e_id)] files = [] for rel, f in (e.get("files") or {}).items(): fi = file_info(f["f_id"]) time.sleep(1) files.append({ "f_id": f["f_id"], "md5": f["md5"], "ext": fi.get("extension"), "size": int(fi.get("filesize") or 0), }) e["files"] = files return e def sniff(data: bytes) -> str: if data[:5] == b" this # edge is wedged, re-draw a new get.php link (libgen 302s to different # storage hosts per request) probe_ok = False for p in range(3): try: pr = subprocess.run(["curl", "-s", "--fail", "-o", "/dev/null", "-r", "0-300000", "--max-time", "6", "--connect-timeout", "10", "-H", "User-Agent: " + UA, "-H", "Referer: " + f"{BASE}/ads.php?md5={md5}", "-w", "%{size_download} %{speed_download}", url], capture_output=True, text=True, timeout=20) if pr.returncode == 0: parts = pr.stdout.split() spd = float(parts[1]) if len(parts) > 1 else 0 print(f" try {t}: probe edge {p+1}: {spd/1024:.0f} KB/s") if spd >= 15 * 1024: probe_ok = True break except Exception: pass print(f" try {t}: edge {p+1} slow/wedged — re-drawing get.php link") time.sleep(3) try: card2 = curl(f"{BASE}/ads.php?md5={md5}", referer=f"{BASE}/file.php?id={f_id}", tries=1, timeout=90, connect_timeout=20) m2 = re.search(r'(https?://[^"\']*get\.php\?md5=' + md5 + r'&key=[^"\']+)', card2) if m2: url = m2.group(1) except RuntimeError: pass if not probe_ok: print(f" try {t}: all 3 edges slow — waiting out the CDN") tmp = out + ".part" prev = os.path.getsize(tmp) if os.path.exists(tmp) else 0 print(f" try {t}: data fetch start ({prev}b resumed)") try: rc = curl(url, out=tmp, referer=f"{BASE}/ads.php?md5={md5}", binary=True, timeout=3600, connect_timeout=30, # abort hung connections: <500B/s for 300s extra=["-C", "-", "--speed-time", "300", "--speed-limit", "500"]) except RuntimeError as e: print(f" try {t}: fetch error ({e}) — back off") time.sleep(45) continue cur = os.path.getsize(tmp) if os.path.exists(tmp) else 0 # resume-hang detection: zero-progress resume at the same offset => # this CDN offset is wedged; wipe and start fresh (first stall is enough) if 0 < prev == cur: print(f" try {t}: resume stalled at {prev}b — wiping part, fresh start") os.remove(tmp) stall = 0 cur = 0 data = open(tmp, "rb").read() kind = sniff(data) ok_size = (size == 0) or (len(data) == size) print(f" try {t}: {len(data)}b (expected {size}) kind={kind}") if kind == "HTML-ERROR": os.remove(tmp) if kind == "HTML-ERROR" or not ok_size: time.sleep(4) continue os.replace(tmp, out) print(f" OK -> {out} ({len(data)}b, {kind})") return out print(f" FAILED after {tries} tries: file_id={f_id} md5={md5}") return None # stop-ish / generic tokens to ignore when matching locator vs target _STOP = set("the of and in to a an for with from on by edition new second first " "vol volume pdf epub fb2 doc djvu mobi txt part pt series book " "translated translation revised updated complete full original" .split()) def _tokens(s): return {w for w in re.findall(r"[a-z\u0400-\u04ff]{4,}", (s or "").lower()) if w not in _STOP} def _loc_tokens(s): # locator words: mask alphanumeric runs containing digits (IDs/hashes/ISBNs) # so "d1e2f4...aaae..." does not yield fake word 'aaae' masked = re.sub(r"[a-z0-9]+\d[a-z0-9]*", " ", (s or "").lower()) return _tokens(masked) def precheck(f_id, e_id, expected_pages=None): """Mislinked-file gate. Returns 0=OK, 1=SUSPECT, 2=BAD (see module doc). CRITICAL: pass the edition files-map VALUE (`f_id`), not the map key — the key is a different (often stale/mislinked) file record (verified 2026-09-25: Solomon — key 92709688 = Russian book, f_id 93253044 = real book). Signals: 1) reverse map: for a real f_id record the `editions` values ARE the public edition IDs — value == target is a strong OK signal; for a key's record it contains other numbers (no match => weak negative, never hard BAD). 2) locator: words in the locator vs target title/author words. - no content words (hash/ISBN/number) => no signal (SUSPECT) - >=2 shared words (or 1 for a short target) => OK - words present, none shared => BAD (names a different book) ALWAYS content-verify after download (ground truth). """ fi = file_info(f_id) time.sleep(2) ext, size = fi.get("extension"), int(fi.get("filesize") or 0) locator = (fi.get("locator") or "").strip() reasons = [] rev = {k: v.get("e_id") for k, v in (fi.get("editions") or {}).items()} rev_match = int(e_id) in {int(v) for v in rev.values() if v} if rev: reasons.append(f"reverse-map: {rev} {'MATCHES target' if rev_match else '(no target)'}") try: ed = jget(f"json.php?object=e&addkeys=title,author&ids={e_id}").get(str(e_id), {}) time.sleep(2) target = f"{ed.get('title','')} {ed.get('author','')}" except Exception: target = "" t_tok, l_tok = _tokens(target), _loc_tokens(locator) shared = t_tok & l_tok if not locator or not l_tok: loc_signal = "none" reasons.append(f"locator has no content words (hash/ISBN/empty) — no signal: {locator[:90]}") elif len(shared) >= 2 or (len(shared) == 1 and len(t_tok) <= 3): loc_signal = "match" reasons.append(f"locator names the target (shared: {sorted(shared)}): {locator[:120]}") else: loc_signal = "mismatch" low = locator.lower() target_cyr = re.search(r"[\u0400-\u04FF]", target) cyrillic_loc = re.search(r"[\u0400-\u04FF]", locator) foreign = re.search(r"(spa\\|spanish|english|epublibre|0day|\\fiction\\)", low) if target_cyr and not cyrillic_loc: reasons.append(f"Cyrillic target but locator is non-Cyrillic: {locator[:120]}") elif foreign: reasons.append(f"locator is a foreign batch naming another book: {locator[:120]}") else: reasons.append(f"locator names a DIFFERENT book (no shared words): {locator[:120]}") if rev_match and loc_signal != "mismatch": verdict = "OK" elif loc_signal == "match": verdict = "OK" elif loc_signal == "mismatch" and not rev_match: verdict = "BAD" else: verdict = "SUSPECT" # none/none, or contradiction (rev match vs locator mismatch) if rev_match and loc_signal == "mismatch": reasons.append("CONTRADICTION: reverse-map matches but locator names another book — verify content") # size sanity vs expected pages (text ~200B/pp; scans 50KB..15MB/pp) if expected_pages and size: per_lo = 200 if ext in ("fb2", "epub", "txt", "html") else 10_000 lo, hi = expected_pages * per_lo, expected_pages * 15_000_000 if size < lo: if verdict == "OK": verdict = "SUSPECT" reasons.append(f"size {size}b < {lo}b for ~{expected_pages} pp (too small — fragment?)") elif size > hi: if verdict == "OK": verdict = "SUSPECT" reasons.append(f"size {size}b > {hi}b for ~{expected_pages} pp (too big — scan mismatch?)") print(f"precheck f_id={f_id} vs e_id={e_id}: {verdict} [{ext}, {size}b]") for r in reasons: print(f" - {r}") return {"OK": 0, "SUSPECT": 1, "BAD": 2}[verdict] def main(): if len(sys.argv) < 2: sys.exit(__doc__) cmd = sys.argv[1] if cmd == "ed": for e_id in sys.argv[2:]: e = edition(e_id) keep = {k: e[k] for k in ("title", "title_add", "author", "publisher", "year", "pages", "series_name", "libgen_topic") if e.get(k)} print(f"[{e_id}] {json.dumps(keep, ensure_ascii=False)}") for f in e["files"]: print(f" f_id={f['f_id']} {f['size']:>10}b {f['ext']:<6} md5={f['md5']}") time.sleep(1) elif cmd == "dl": f_id, out_dir, name = sys.argv[2], sys.argv[3], sys.argv[4] if len(sys.argv) > 5 and sys.argv[5].isdigit(): rc = precheck(f_id, sys.argv[5]) if rc == 2: print(" ABORT: precheck RED (mislinked file) — pick another f_id/edition or flib") sys.exit(2) if rc == 1: print(" precheck SUSPECT — downloading, MUST post-verify content") time.sleep(3) download(f_id, out_dir, name) elif cmd == "precheck": f_id, e_id = sys.argv[2], sys.argv[3] pages = int(sys.argv[4]) if len(sys.argv) > 4 else None sys.exit(precheck(f_id, e_id, pages)) elif cmd == "md5": print(file_info(sys.argv[2])["md5"]) elif cmd == "bymd5": # md5 -> f_id + metadata (for files found via lg.py file search) d = jget(f"json.php?object=f&addkeys=*&md5={sys.argv[2]}") if not d: print("not found"); sys.exit(1) fid = list(d.keys())[0] f = d[fid] print(f"f_id={fid} {f.get('extension')} {int(f.get('filesize') or 0)}b pages={f.get('pages')} ocr={f.get('ocr')} added={f.get('time_added')}") print(f"locator: {f.get('locator','')}") rev = {k: v.get('e_id') for k, v in (f.get('editions') or {}).items()} if rev: print(f"editions (hint): {rev}") else: sys.exit(__doc__) if __name__ == "__main__": main()