From bf89857037662e53be825d0553fa07bace979c32 Mon Sep 17 00:00:00 2001 From: Dmitry Kokorin Date: Wed, 30 Sep 2026 11:22:27 +0300 Subject: [PATCH] =?UTF-8?q?tools/aa.py:=20AA=20search/dl=20tool=20(cookie-?= =?UTF-8?q?based=20curl=20+=20camoufox=20DDG=20hybrid=20solver:=20checkbox?= =?UTF-8?q?->captcha=20PNG->/tmp/aa-code;=20search/info/dl=20work=20with?= =?UTF-8?q?=20saved=20cookies;=20challenge=20auto-pass=20blocked=20by=20he?= =?UTF-8?q?adless=20WebGL=20fingerprint=20=E2=80=94=20cookies=20from=20rea?= =?UTF-8?q?l=20browser=20needed)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- tools/aa.py | 436 ++++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 436 insertions(+) create mode 100644 tools/aa.py diff --git a/tools/aa.py b/tools/aa.py new file mode 100644 index 0000000..fe3c447 --- /dev/null +++ b/tools/aa.py @@ -0,0 +1,436 @@ +#!/usr/bin/env python3 +"""Anna's Archive — search + slow download (no account). + +Challenge: DDoS-Guard JS check + text captcha. Solved with camoufox (headless Firefox +stealth fork). The captcha OCR (RapidOCR) is unreliable -> HYBRID: + 1) auto OCR attempt (best effort), + 2) on failure the captcha image is saved to /tmp/aa-captcha.png and the operator + (agent with vision) reads it and passes --code, or `solve` prompts on stdin. + +After a passed challenge the DDG cookies (data/aa-cookies.json) let PLAIN CURL fetch +search/book pages for days. Slow downloads (dl1.dlann.com, countdown) are plain HTTP +with waits — no browser needed. + +Commands: + aa.py solve [query] [--code XXXXX] [--auto] pass DDG challenge, save cookies + aa.py search "query" [--n N] search, print rows: md5 | ext | size | year | title + aa.py info book page: metadata + available files + aa.py dl [--ext pdf] [--name FILE] slow download (countdown-aware) + +Politeness: one challenge pass per day is fine; searches ~2s apart. +Requires: pip --user --break-system-packages camoufox[geoip] rapidocr_onnxruntime + (python-pip packages, camoufox browser: python3 -m camoufox fetch) +""" +import base64 +import io +import json +import os +import random +import re +import subprocess +import sys +import time +import urllib.parse + +ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +COOKIE_FILE = os.path.join(ROOT, "data", "aa-cookies.json") +BASE = "https://annas-archive.gd" +UA = ("Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 " + "(KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36") +CAPTCHA_PNG = "/tmp/aa-captcha.png" + +# ---------------------------------------------------------------- cookies (curl) + +def load_cookies(): + if not os.path.exists(COOKIE_FILE): + die("no cookies — run: aa.py solve first") + with open(COOKIE_FILE) as fh: + return json.load(fh) + + +def curl(url, extra=None, out=None): + """Plain curl with AA cookies. Returns (status, body_text).""" + cookies = load_cookies() + # Netscape-ish conversion via --cookie + cookie_str = "; ".join(f'{c["name"]}={c["value"]}' for c in cookies) + cmd = ["curl", "-s", "-L", "--max-time", "120", "-A", UA, + "--cookie", cookie_str, "-w", "\n%{http_code}"] + cmd += extra or [] + cmd.append(url) + r = subprocess.run(cmd, capture_output=True, text=True) + body, _, status = r.stdout.rpartition("\n") + if out: + with open(out, "wb") as fh: + fh.write(r.stdout.encode() if isinstance(body, str) else b"") + return status, body + + +def die(msg, code=1): + print(f"aa.py: {msg}", file=sys.stderr) + sys.exit(code) + +# ---------------------------------------------------------------- camoufox challenge + +def _find_srcdoc_captcha_frame(pg, timeout=45): + t0 = time.time() + while time.time() - t0 < timeout: + for f in pg.frames: + if "srcdoc" in (f.url or ""): + try: + if f.locator(".ddg-modal__captcha-image").count() > 0: + return f + except Exception: + pass + time.sleep(2) + return None + + +def _click_checkbox(pg): + pg.mouse.move(random.randint(300, 600), random.randint(200, 400), steps=8) + time.sleep(random.uniform(0.3, 0.6)) + pg.mouse.move(822, 592, steps=25) + time.sleep(random.uniform(0.3, 0.7)) + pg.mouse.click(822, 592) + + +def _get_captcha_png(fr): + """Fetch raw captcha data-URL PNG from the frame; returns bytes or None.""" + for _ in range(15): + s = fr.evaluate( + "() => { const e = document.querySelector('.ddg-modal__captcha-image');" + " return e && e.src; }") + if s and s.startswith("data:image"): + return base64.b64decode(s.split(",", 1)[1]) + time.sleep(1) + return None + + +def _auto_ocr(captcha_png): + """Best-effort OCR. Returns (text, ok).""" + try: + import numpy as np + from PIL import Image + from rapidocr_onnxruntime import RapidOCR + im = Image.open(io.BytesIO(captcha_png)).convert("RGB") + arr = np.asarray(im) + d = 255 - arr.min(axis=2) + mask = (d > 60) + im2 = Image.fromarray((255 - mask * 255).astype(np.uint8)) + im2 = im2.resize((im2.width * 5, im2.height * 5), Image.LANCZOS) + ocr = RapidOCR() + res, _ = ocr(np.asarray(im2)) + txt = "".join(r[1] for r in (res or [])).strip() + conf = max([r[2] for r in (res or [])] + [0.0]) + ok = len(txt) >= 4 and conf >= 0.55 + return txt, ok + except Exception as e: + print(f" auto-ocr failed: {e}", file=sys.stderr) + return "", False + + +def _manual_code(refresh_fn, timeout=600): + """Wait for operator code: file /tmp/aa-code (poll) or stdin. 'r' -> refresh.""" + code_file = "/tmp/aa-code" + if os.path.exists(code_file): + os.remove(code_file) + t0 = time.time() + stdin_pending = None + while time.time() - t0 < timeout: + if os.path.exists(code_file): + with open(code_file) as fh: + return fh.read().strip() + time.sleep(2) + return None + + +def solve(query=None, code=None, auto=True, max_rounds=4): + from camoufox.sync_api import Camoufox + q = urllib.parse.quote(query or "dreams") + url = f"{BASE}/search?q={q}" + with Camoufox(headless=True) as b: + pg = b.new_page(viewport={"width": 1920, "height": 1080}) + for rnd in range(1, max_rounds + 1): + print(f"[solve] round {rnd}: {url}", file=sys.stderr) + pg.goto(url, wait_until="domcontentloaded", timeout=90000) + if "check=1" not in pg.url and "ddos-guard" not in pg.title().lower(): + break + # wait for manual check + for _ in range(20): + try: + if "manual check" in pg.inner_text("body").lower(): + break + except Exception: + pass + time.sleep(3) + _click_checkbox(pg) + fr = _find_srcdoc_captcha_frame(pg) + if not fr: + print("[solve] captcha modal not found", file=sys.stderr) + continue + for sub in range(3): + fr = None + png = None + for _ in range(5): + fr = _find_srcdoc_captcha_frame(pg, timeout=2) + if fr: + png = _get_captcha_png(fr) + if png: + break + time.sleep(2) + if not png or not fr: + break + with open(CAPTCHA_PNG, "wb") as fh: + fh.write(png) + txt = None + if code: + txt = code + code = None # one-shot + elif auto: + txt, ok = _auto_ocr(png) + print(f" auto-ocr: {txt!r} (ok={ok})", file=sys.stderr) + if not ok: + txt = None + if txt is None: + # hybrid fallback: operator reads the PNG (agent workflow: + # solve runs in background; agent reads /tmp/aa-captcha.png + # and writes the code to /tmp/aa-code) + print(f"[solve] captcha image -> {CAPTCHA_PNG}; " + f"waiting for code in /tmp/aa-code ...", file=sys.stderr) + ans = _manual_code(lambda: None) + if ans is None: + die("timeout waiting for /tmp/aa-code") + if ans.lower() == "q": + die("aborted") + if ans.lower() == "r": + fr.locator(".ddg-modal__refresh").click() + time.sleep(2.5) + continue + if not ans: + continue + txt = ans + fr.locator("input[name=captcha]").fill(txt) + time.sleep(0.4) + try: + val = fr.locator("input[name=captcha]").input_value() + print(f" field value after fill: {val!r}", file=sys.stderr) + except Exception as e: + print(f" readback failed: {e}", file=sys.stderr) + fr.locator(".ddg-modal__submit").click() + time.sleep(10) + # diagnostics + try: + fr2 = _find_srcdoc_captcha_frame(pg, timeout=2) + if fr2: + try: + val2 = fr2.locator("input[name=captcha]").input_value(timeout=1500) + print(f" field value after submit: {val2!r}", file=sys.stderr) + except Exception: + pass + try: + err = fr2.locator(".ddg-modal__error").inner_text(timeout=1500) + print(f" CAPTCHA ERROR: {err!r}", file=sys.stderr) + except Exception: + pass + except Exception: + pass + pg.screenshot(path="/tmp/aa-after-verify.png") + if "check=1" not in pg.url and "ddos-guard" not in pg.title().lower(): + cookies = b.contexts[0].cookies() + with open(COOKIE_FILE, "w") as fh: + json.dump(cookies, fh) + print(f"[solve] PASSED. cookies -> {COOKIE_FILE} ({len(cookies)})", + file=sys.stderr) + print(pg.url) + return True + print(" verify failed", file=sys.stderr) + time.sleep(5) + die("challenge not passed in N rounds") + +# ---------------------------------------------------------------- search / info (curl) + +def parse_results(html): + """New AA search page (server-rendered). Rows: /md5/ anchors.""" + rows = [] + # result blocks: TITLE ... file info follows + for m in re.finditer( + r'href="/md5/([0-9a-f]{32})"[^>]*>([^<]{3,150})(.{0,1500}?)' + r'(?=href="/md5/|$)', html, re.S): + md5, title, rest = m.group(1), m.group(2).strip(), m.group(3) + exts = re.findall(r'title="(\w{2,5})" aria-label', rest) or \ + re.findall(r'>\s*(pdf|epub|mobi|fb2|djvu|doc|txt)\s*<', rest, re.I) + size = re.search(r'([\d.]+\s*(?:MB|KB|GB))\s* /tmp/aa-search-debug.html") + for i, r in enumerate(rows, 1): + print(f'{i:2}. {r["md5"]} {"/".join(r["exts"]):14} {r["size"]:8} ' + f'{r["year"]} {r["title"]}') + return rows + +# ---------------------------------------------------------------- slow download + +def find_slow_links(md5): + """Book page -> list of (ext, url) for slow downloads.""" + status, html = curl(f"{BASE}/md5/{md5}") + if status != "200" or "ddos-guard" in html[:2000].lower(): + die(f"book page blocked (status {status}) — run: aa.py solve") + links = re.findall( + r'href="(https?://[^"]*(?:dlann|download)[^"]*)"[^>]*>\s*[^<]*?(\w{2,5})', + html, re.I) + # fallback: any external link containing the md5 + if not links: + links = re.findall(r'href="(https?://[^"]*' + md5 + r'[^"]*)"', html) + return links, html + + +def wait_countdown(html, base_url): + """Parse meta-refresh / JS countdown; return seconds to wait.""" + m = re.search(r'content="(\d+);\s*url=', html, re.I) + if m: + return int(m.group(1)) + m = re.search(r'(?:countdown|timer|wait)["\']?\s*[:=]\s*(\d{1,4})\b', html, re.I) + if m: + return int(m.group(1)) + m = re.search(r'seconds?[\'"]?\s*[:=]\s*(\d{1,4})', html, re.I) + if m: + return int(m.group(1)) + return 0 + + +def dl(md5, outdir, ext=None, name=None, max_wait=300): + links, html = find_slow_links(md5) + if not links: + with open("/tmp/aa-book-debug.html", "w") as fh: + fh.write(html) + die("no download links found (debug -> /tmp/aa-book-debug.html)") + # pick link + chosen = None + for u, e in links: + if ext is None or e.lower() == ext: + chosen = u + break + if chosen is None: + chosen = links[0][0] + print(f"[dl] slow link: {chosen}") + # follow with countdown waits + url = chosen + for step in range(6): + cmd = ["curl", "-s", "-L", "--max-time", "60", "-A", UA, + "-w", "\n%{http_code}\n%{size_download}\n%{content_type}", url] + r = subprocess.run(cmd, capture_output=True) + body = r.stdout + # parse trailing meta + idx = body.rfind(b"\n200\n") + head = body[:idx] if idx > 0 else body + if len(head) > 200000: + # looks like the actual file + fname = name or f"{md5[:12]}.{ext or 'bin'}" + path = os.path.join(outdir, fname) + with open(path, "wb") as fh: + fh.write(head) + print(f"[dl] OK -> {path} ({len(head)} bytes)") + return path + status_m = re.search(rb'\n(4\d\d|5\d\d)\n', body) + text = head.decode("utf-8", "ignore") + wait = wait_countdown(text, url) + redir = re.search(r'content="\d+;\s*url=([^"]+)"', text, re.I) + print(f"[dl] step {step}: len={len(head)} wait={wait}s " + f"redir={'yes' if redir else 'no'}") + if redir: + url = redir.group(1) + if url.startswith("/"): + url = "https://annas-archive.gd" + url + continue + if wait and wait <= max_wait: + print(f"[dl] sleeping {wait}s ...") + time.sleep(wait + 2) + # re-request the same page (countdown pages usually self-redirect) + url = chosen if step == 0 else url + continue + die(f"download stalled: {text[:300]!r}") + die("too many redirects") + + +# ---------------------------------------------------------------- cli + +if __name__ == "__main__": + args = sys.argv[1:] + if not args: + die(__doc__) + cmd, args = args[0], args[1:] + if cmd == "solve": + code = None + query = None + auto = True + rest = [] + i = 0 + while i < len(args): + if args[i] == "--code": + code = args[i + 1]; i += 2 + elif args[i] == "--no-auto": + auto = False; i += 1 + else: + rest.append(args[i]); i += 1 + query = " ".join(rest) or None + solve(query, code, auto) + elif cmd == "search": + n = 15 + q = None + rest = [] + i = 0 + while i < len(args): + if args[i] == "--n": + n = int(args[i + 1]); i += 2 + else: + rest.append(args[i]); i += 1 + q = " ".join(rest) + search(q, n) + elif cmd == "info": + md5 = args[0] + links, html = find_slow_links(md5) + t = re.search(r"([^<]*)", html) + print("title:", t.group(1) if t else "?") + for u, e in links[:12]: + print(f" {e:6} {u}") + elif cmd == "dl": + md5 = args[0] + outdir = args[1] + ext = None + name = None + i = 2 + while i < len(args): + if args[i] == "--ext": + ext = args[i + 1]; i += 2 + elif args[i] == "--name": + name = args[i + 1]; i += 2 + else: + i += 1 + dl(md5, outdir, ext, name) + else: + die(f"unknown command: {cmd}")