#!/usr/bin/env python3 """Anna's Archive — search + slow download (no account). Challenge: DDoS-Guard JS check + text captcha. Solved with camoufox (headless Firefox stealth fork). The captcha OCR (RapidOCR) is unreliable -> HYBRID: 1) auto OCR attempt (best effort), 2) on failure the captcha image is saved to /tmp/aa-captcha.png and the operator (agent with vision) reads it and passes --code, or `solve` prompts on stdin. After a passed challenge the DDG cookies (data/aa-cookies.json) let PLAIN CURL fetch search/book pages for days. Slow downloads (dl1.dlann.com, countdown) are plain HTTP with waits — no browser needed. Commands: aa.py solve [query] [--code XXXXX] [--auto] pass DDG challenge, save cookies aa.py search "query" [--n N] search, print rows: md5 | ext | size | year | title aa.py info book page: metadata + available files aa.py dl [--ext pdf] [--name FILE] slow download (countdown-aware) Politeness: one challenge pass per day is fine; searches ~2s apart. Requires: pip --user --break-system-packages camoufox[geoip] rapidocr_onnxruntime (python-pip packages, camoufox browser: python3 -m camoufox fetch) """ import base64 import io import json import os import random import re import subprocess import sys import time import urllib.parse ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) COOKIE_FILE = os.path.join(ROOT, "data", "aa-cookies.json") BASE = "https://annas-archive.gd" UA = ("Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 " "(KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36") CAPTCHA_PNG = "/tmp/aa-captcha.png" # ---------------------------------------------------------------- cookies (curl) def load_cookies(): if not os.path.exists(COOKIE_FILE): die("no cookies — run: aa.py solve first") with open(COOKIE_FILE) as fh: return json.load(fh) def curl(url, extra=None, out=None): """Plain curl with AA cookies. Returns (status, body_text).""" cookies = load_cookies() # Netscape-ish conversion via --cookie cookie_str = "; ".join(f'{c["name"]}={c["value"]}' for c in cookies) cmd = ["curl", "-s", "-L", "--max-time", "120", "-A", UA, "--cookie", cookie_str, "-w", "\n%{http_code}"] cmd += extra or [] cmd.append(url) r = subprocess.run(cmd, capture_output=True, text=True) body, _, status = r.stdout.rpartition("\n") if out: with open(out, "wb") as fh: fh.write(r.stdout.encode() if isinstance(body, str) else b"") return status, body def die(msg, code=1): print(f"aa.py: {msg}", file=sys.stderr) sys.exit(code) # ---------------------------------------------------------------- camoufox challenge def _find_srcdoc_captcha_frame(pg, timeout=45): t0 = time.time() while time.time() - t0 < timeout: for f in pg.frames: if "srcdoc" in (f.url or ""): try: if f.locator(".ddg-modal__captcha-image").count() > 0: return f except Exception: pass time.sleep(2) return None def _click_checkbox(pg): pg.mouse.move(random.randint(300, 600), random.randint(200, 400), steps=8) time.sleep(random.uniform(0.3, 0.6)) pg.mouse.move(822, 592, steps=25) time.sleep(random.uniform(0.3, 0.7)) pg.mouse.click(822, 592) def _get_captcha_png(fr): """Fetch raw captcha data-URL PNG from the frame; returns bytes or None.""" for _ in range(15): s = fr.evaluate( "() => { const e = document.querySelector('.ddg-modal__captcha-image');" " return e && e.src; }") if s and s.startswith("data:image"): return base64.b64decode(s.split(",", 1)[1]) time.sleep(1) return None def _auto_ocr(captcha_png): """Best-effort OCR. Returns (text, ok).""" try: import numpy as np from PIL import Image from rapidocr_onnxruntime import RapidOCR im = Image.open(io.BytesIO(captcha_png)).convert("RGB") arr = np.asarray(im) d = 255 - arr.min(axis=2) mask = (d > 60) im2 = Image.fromarray((255 - mask * 255).astype(np.uint8)) im2 = im2.resize((im2.width * 5, im2.height * 5), Image.LANCZOS) ocr = RapidOCR() res, _ = ocr(np.asarray(im2)) txt = "".join(r[1] for r in (res or [])).strip() conf = max([r[2] for r in (res or [])] + [0.0]) ok = len(txt) >= 4 and conf >= 0.55 return txt, ok except Exception as e: print(f" auto-ocr failed: {e}", file=sys.stderr) return "", False def _manual_code(refresh_fn, timeout=600): """Wait for operator code: file /tmp/aa-code (poll) or stdin. 'r' -> refresh.""" code_file = "/tmp/aa-code" if os.path.exists(code_file): os.remove(code_file) t0 = time.time() stdin_pending = None while time.time() - t0 < timeout: if os.path.exists(code_file): with open(code_file) as fh: return fh.read().strip() time.sleep(2) return None def solve(query=None, code=None, auto=True, max_rounds=4): from camoufox.sync_api import Camoufox q = urllib.parse.quote(query or "dreams") url = f"{BASE}/search?q={q}" with Camoufox(headless=True) as b: pg = b.new_page(viewport={"width": 1920, "height": 1080}) for rnd in range(1, max_rounds + 1): print(f"[solve] round {rnd}: {url}", file=sys.stderr) pg.goto(url, wait_until="domcontentloaded", timeout=90000) if "check=1" not in pg.url and "ddos-guard" not in pg.title().lower(): break # wait for manual check for _ in range(20): try: if "manual check" in pg.inner_text("body").lower(): break except Exception: pass time.sleep(3) _click_checkbox(pg) fr = _find_srcdoc_captcha_frame(pg) if not fr: print("[solve] captcha modal not found", file=sys.stderr) continue for sub in range(3): fr = None png = None for _ in range(5): fr = _find_srcdoc_captcha_frame(pg, timeout=2) if fr: png = _get_captcha_png(fr) if png: break time.sleep(2) if not png or not fr: break with open(CAPTCHA_PNG, "wb") as fh: fh.write(png) txt = None if code: txt = code code = None # one-shot elif auto: txt, ok = _auto_ocr(png) print(f" auto-ocr: {txt!r} (ok={ok})", file=sys.stderr) if not ok: txt = None if txt is None: # hybrid fallback: operator reads the PNG (agent workflow: # solve runs in background; agent reads /tmp/aa-captcha.png # and writes the code to /tmp/aa-code) print(f"[solve] captcha image -> {CAPTCHA_PNG}; " f"waiting for code in /tmp/aa-code ...", file=sys.stderr) ans = _manual_code(lambda: None) if ans is None: die("timeout waiting for /tmp/aa-code") if ans.lower() == "q": die("aborted") if ans.lower() == "r": fr.locator(".ddg-modal__refresh").click() time.sleep(2.5) continue if not ans: continue txt = ans fr.locator("input[name=captcha]").fill(txt) time.sleep(0.4) try: val = fr.locator("input[name=captcha]").input_value() print(f" field value after fill: {val!r}", file=sys.stderr) except Exception as e: print(f" readback failed: {e}", file=sys.stderr) fr.locator(".ddg-modal__submit").click() time.sleep(10) # diagnostics try: fr2 = _find_srcdoc_captcha_frame(pg, timeout=2) if fr2: try: val2 = fr2.locator("input[name=captcha]").input_value(timeout=1500) print(f" field value after submit: {val2!r}", file=sys.stderr) except Exception: pass try: err = fr2.locator(".ddg-modal__error").inner_text(timeout=1500) print(f" CAPTCHA ERROR: {err!r}", file=sys.stderr) except Exception: pass except Exception: pass pg.screenshot(path="/tmp/aa-after-verify.png") if "check=1" not in pg.url and "ddos-guard" not in pg.title().lower(): cookies = b.contexts[0].cookies() with open(COOKIE_FILE, "w") as fh: json.dump(cookies, fh) print(f"[solve] PASSED. cookies -> {COOKIE_FILE} ({len(cookies)})", file=sys.stderr) print(pg.url) return True print(" verify failed", file=sys.stderr) time.sleep(5) die("challenge not passed in N rounds") # ---------------------------------------------------------------- search / info (curl) def parse_results(html): """New AA search page (server-rendered). Rows: /md5/ anchors.""" rows = [] # result blocks: TITLE ... file info follows for m in re.finditer( r'href="/md5/([0-9a-f]{32})"[^>]*>([^<]{3,150})(.{0,1500}?)' r'(?=href="/md5/|$)', html, re.S): md5, title, rest = m.group(1), m.group(2).strip(), m.group(3) exts = re.findall(r'title="(\w{2,5})" aria-label', rest) or \ re.findall(r'>\s*(pdf|epub|mobi|fb2|djvu|doc|txt)\s*<', rest, re.I) size = re.search(r'([\d.]+\s*(?:MB|KB|GB))\s* /tmp/aa-search-debug.html") for i, r in enumerate(rows, 1): print(f'{i:2}. {r["md5"]} {"/".join(r["exts"]):14} {r["size"]:8} ' f'{r["year"]} {r["title"]}') return rows # ---------------------------------------------------------------- slow download def find_slow_links(md5): """Book page -> list of (ext, url) for slow downloads.""" status, html = curl(f"{BASE}/md5/{md5}") if status != "200" or "ddos-guard" in html[:2000].lower(): die(f"book page blocked (status {status}) — run: aa.py solve") links = re.findall( r'href="(https?://[^"]*(?:dlann|download)[^"]*)"[^>]*>\s*[^<]*?(\w{2,5})', html, re.I) # fallback: any external link containing the md5 if not links: links = re.findall(r'href="(https?://[^"]*' + md5 + r'[^"]*)"', html) return links, html def wait_countdown(html, base_url): """Parse meta-refresh / JS countdown; return seconds to wait.""" m = re.search(r'content="(\d+);\s*url=', html, re.I) if m: return int(m.group(1)) m = re.search(r'(?:countdown|timer|wait)["\']?\s*[:=]\s*(\d{1,4})\b', html, re.I) if m: return int(m.group(1)) m = re.search(r'seconds?[\'"]?\s*[:=]\s*(\d{1,4})', html, re.I) if m: return int(m.group(1)) return 0 def dl(md5, outdir, ext=None, name=None, max_wait=300): links, html = find_slow_links(md5) if not links: with open("/tmp/aa-book-debug.html", "w") as fh: fh.write(html) die("no download links found (debug -> /tmp/aa-book-debug.html)") # pick link chosen = None for u, e in links: if ext is None or e.lower() == ext: chosen = u break if chosen is None: chosen = links[0][0] print(f"[dl] slow link: {chosen}") # follow with countdown waits url = chosen for step in range(6): cmd = ["curl", "-s", "-L", "--max-time", "60", "-A", UA, "-w", "\n%{http_code}\n%{size_download}\n%{content_type}", url] r = subprocess.run(cmd, capture_output=True) body = r.stdout # parse trailing meta idx = body.rfind(b"\n200\n") head = body[:idx] if idx > 0 else body if len(head) > 200000: # looks like the actual file fname = name or f"{md5[:12]}.{ext or 'bin'}" path = os.path.join(outdir, fname) with open(path, "wb") as fh: fh.write(head) print(f"[dl] OK -> {path} ({len(head)} bytes)") return path status_m = re.search(rb'\n(4\d\d|5\d\d)\n', body) text = head.decode("utf-8", "ignore") wait = wait_countdown(text, url) redir = re.search(r'content="\d+;\s*url=([^"]+)"', text, re.I) print(f"[dl] step {step}: len={len(head)} wait={wait}s " f"redir={'yes' if redir else 'no'}") if redir: url = redir.group(1) if url.startswith("/"): url = "https://annas-archive.gd" + url continue if wait and wait <= max_wait: print(f"[dl] sleeping {wait}s ...") time.sleep(wait + 2) # re-request the same page (countdown pages usually self-redirect) url = chosen if step == 0 else url continue die(f"download stalled: {text[:300]!r}") die("too many redirects") # ---------------------------------------------------------------- cli if __name__ == "__main__": args = sys.argv[1:] if not args: die(__doc__) cmd, args = args[0], args[1:] if cmd == "solve": code = None query = None auto = True rest = [] i = 0 while i < len(args): if args[i] == "--code": code = args[i + 1]; i += 2 elif args[i] == "--no-auto": auto = False; i += 1 else: rest.append(args[i]); i += 1 query = " ".join(rest) or None solve(query, code, auto) elif cmd == "search": n = 15 q = None rest = [] i = 0 while i < len(args): if args[i] == "--n": n = int(args[i + 1]); i += 2 else: rest.append(args[i]); i += 1 q = " ".join(rest) search(q, n) elif cmd == "info": md5 = args[0] links, html = find_slow_links(md5) t = re.search(r"([^<]*)", html) print("title:", t.group(1) if t else "?") for u, e in links[:12]: print(f" {e:6} {u}") elif cmd == "dl": md5 = args[0] outdir = args[1] ext = None name = None i = 2 while i < len(args): if args[i] == "--ext": ext = args[i + 1]; i += 2 elif args[i] == "--name": name = args[i + 1]; i += 2 else: i += 1 dl(md5, outdir, ext, name) else: die(f"unknown command: {cmd}")