436 lines
16 KiB
Python
436 lines
16 KiB
Python
#!/usr/bin/env python3
|
|
"""Anna's Archive — search + slow download (no account).
|
|
|
|
Challenge: DDoS-Guard JS check + text captcha. Solved with camoufox (headless Firefox
|
|
stealth fork). The captcha OCR (RapidOCR) is unreliable -> HYBRID:
|
|
1) auto OCR attempt (best effort),
|
|
2) on failure the captcha image is saved to /tmp/aa-captcha.png and the operator
|
|
(agent with vision) reads it and passes --code, or `solve` prompts on stdin.
|
|
|
|
After a passed challenge the DDG cookies (data/aa-cookies.json) let PLAIN CURL fetch
|
|
search/book pages for days. Slow downloads (dl1.dlann.com, countdown) are plain HTTP
|
|
with waits — no browser needed.
|
|
|
|
Commands:
|
|
aa.py solve [query] [--code XXXXX] [--auto] pass DDG challenge, save cookies
|
|
aa.py search "query" [--n N] search, print rows: md5 | ext | size | year | title
|
|
aa.py info <md5> book page: metadata + available files
|
|
aa.py dl <md5> <outdir> [--ext pdf] [--name FILE] slow download (countdown-aware)
|
|
|
|
Politeness: one challenge pass per day is fine; searches ~2s apart.
|
|
Requires: pip --user --break-system-packages camoufox[geoip] rapidocr_onnxruntime
|
|
(python-pip packages, camoufox browser: python3 -m camoufox fetch)
|
|
"""
|
|
import base64
|
|
import io
|
|
import json
|
|
import os
|
|
import random
|
|
import re
|
|
import subprocess
|
|
import sys
|
|
import time
|
|
import urllib.parse
|
|
|
|
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
COOKIE_FILE = os.path.join(ROOT, "data", "aa-cookies.json")
|
|
BASE = "https://annas-archive.gd"
|
|
UA = ("Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 "
|
|
"(KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36")
|
|
CAPTCHA_PNG = "/tmp/aa-captcha.png"
|
|
|
|
# ---------------------------------------------------------------- cookies (curl)
|
|
|
|
def load_cookies():
|
|
if not os.path.exists(COOKIE_FILE):
|
|
die("no cookies — run: aa.py solve first")
|
|
with open(COOKIE_FILE) as fh:
|
|
return json.load(fh)
|
|
|
|
|
|
def curl(url, extra=None, out=None):
|
|
"""Plain curl with AA cookies. Returns (status, body_text)."""
|
|
cookies = load_cookies()
|
|
# Netscape-ish conversion via --cookie
|
|
cookie_str = "; ".join(f'{c["name"]}={c["value"]}' for c in cookies)
|
|
cmd = ["curl", "-s", "-L", "--max-time", "120", "-A", UA,
|
|
"--cookie", cookie_str, "-w", "\n%{http_code}"]
|
|
cmd += extra or []
|
|
cmd.append(url)
|
|
r = subprocess.run(cmd, capture_output=True, text=True)
|
|
body, _, status = r.stdout.rpartition("\n")
|
|
if out:
|
|
with open(out, "wb") as fh:
|
|
fh.write(r.stdout.encode() if isinstance(body, str) else b"")
|
|
return status, body
|
|
|
|
|
|
def die(msg, code=1):
|
|
print(f"aa.py: {msg}", file=sys.stderr)
|
|
sys.exit(code)
|
|
|
|
# ---------------------------------------------------------------- camoufox challenge
|
|
|
|
def _find_srcdoc_captcha_frame(pg, timeout=45):
|
|
t0 = time.time()
|
|
while time.time() - t0 < timeout:
|
|
for f in pg.frames:
|
|
if "srcdoc" in (f.url or ""):
|
|
try:
|
|
if f.locator(".ddg-modal__captcha-image").count() > 0:
|
|
return f
|
|
except Exception:
|
|
pass
|
|
time.sleep(2)
|
|
return None
|
|
|
|
|
|
def _click_checkbox(pg):
|
|
pg.mouse.move(random.randint(300, 600), random.randint(200, 400), steps=8)
|
|
time.sleep(random.uniform(0.3, 0.6))
|
|
pg.mouse.move(822, 592, steps=25)
|
|
time.sleep(random.uniform(0.3, 0.7))
|
|
pg.mouse.click(822, 592)
|
|
|
|
|
|
def _get_captcha_png(fr):
|
|
"""Fetch raw captcha data-URL PNG from the frame; returns bytes or None."""
|
|
for _ in range(15):
|
|
s = fr.evaluate(
|
|
"() => { const e = document.querySelector('.ddg-modal__captcha-image');"
|
|
" return e && e.src; }")
|
|
if s and s.startswith("data:image"):
|
|
return base64.b64decode(s.split(",", 1)[1])
|
|
time.sleep(1)
|
|
return None
|
|
|
|
|
|
def _auto_ocr(captcha_png):
|
|
"""Best-effort OCR. Returns (text, ok)."""
|
|
try:
|
|
import numpy as np
|
|
from PIL import Image
|
|
from rapidocr_onnxruntime import RapidOCR
|
|
im = Image.open(io.BytesIO(captcha_png)).convert("RGB")
|
|
arr = np.asarray(im)
|
|
d = 255 - arr.min(axis=2)
|
|
mask = (d > 60)
|
|
im2 = Image.fromarray((255 - mask * 255).astype(np.uint8))
|
|
im2 = im2.resize((im2.width * 5, im2.height * 5), Image.LANCZOS)
|
|
ocr = RapidOCR()
|
|
res, _ = ocr(np.asarray(im2))
|
|
txt = "".join(r[1] for r in (res or [])).strip()
|
|
conf = max([r[2] for r in (res or [])] + [0.0])
|
|
ok = len(txt) >= 4 and conf >= 0.55
|
|
return txt, ok
|
|
except Exception as e:
|
|
print(f" auto-ocr failed: {e}", file=sys.stderr)
|
|
return "", False
|
|
|
|
|
|
def _manual_code(refresh_fn, timeout=600):
|
|
"""Wait for operator code: file /tmp/aa-code (poll) or stdin. 'r' -> refresh."""
|
|
code_file = "/tmp/aa-code"
|
|
if os.path.exists(code_file):
|
|
os.remove(code_file)
|
|
t0 = time.time()
|
|
stdin_pending = None
|
|
while time.time() - t0 < timeout:
|
|
if os.path.exists(code_file):
|
|
with open(code_file) as fh:
|
|
return fh.read().strip()
|
|
time.sleep(2)
|
|
return None
|
|
|
|
|
|
def solve(query=None, code=None, auto=True, max_rounds=4):
|
|
from camoufox.sync_api import Camoufox
|
|
q = urllib.parse.quote(query or "dreams")
|
|
url = f"{BASE}/search?q={q}"
|
|
with Camoufox(headless=True) as b:
|
|
pg = b.new_page(viewport={"width": 1920, "height": 1080})
|
|
for rnd in range(1, max_rounds + 1):
|
|
print(f"[solve] round {rnd}: {url}", file=sys.stderr)
|
|
pg.goto(url, wait_until="domcontentloaded", timeout=90000)
|
|
if "check=1" not in pg.url and "ddos-guard" not in pg.title().lower():
|
|
break
|
|
# wait for manual check
|
|
for _ in range(20):
|
|
try:
|
|
if "manual check" in pg.inner_text("body").lower():
|
|
break
|
|
except Exception:
|
|
pass
|
|
time.sleep(3)
|
|
_click_checkbox(pg)
|
|
fr = _find_srcdoc_captcha_frame(pg)
|
|
if not fr:
|
|
print("[solve] captcha modal not found", file=sys.stderr)
|
|
continue
|
|
for sub in range(3):
|
|
fr = None
|
|
png = None
|
|
for _ in range(5):
|
|
fr = _find_srcdoc_captcha_frame(pg, timeout=2)
|
|
if fr:
|
|
png = _get_captcha_png(fr)
|
|
if png:
|
|
break
|
|
time.sleep(2)
|
|
if not png or not fr:
|
|
break
|
|
with open(CAPTCHA_PNG, "wb") as fh:
|
|
fh.write(png)
|
|
txt = None
|
|
if code:
|
|
txt = code
|
|
code = None # one-shot
|
|
elif auto:
|
|
txt, ok = _auto_ocr(png)
|
|
print(f" auto-ocr: {txt!r} (ok={ok})", file=sys.stderr)
|
|
if not ok:
|
|
txt = None
|
|
if txt is None:
|
|
# hybrid fallback: operator reads the PNG (agent workflow:
|
|
# solve runs in background; agent reads /tmp/aa-captcha.png
|
|
# and writes the code to /tmp/aa-code)
|
|
print(f"[solve] captcha image -> {CAPTCHA_PNG}; "
|
|
f"waiting for code in /tmp/aa-code ...", file=sys.stderr)
|
|
ans = _manual_code(lambda: None)
|
|
if ans is None:
|
|
die("timeout waiting for /tmp/aa-code")
|
|
if ans.lower() == "q":
|
|
die("aborted")
|
|
if ans.lower() == "r":
|
|
fr.locator(".ddg-modal__refresh").click()
|
|
time.sleep(2.5)
|
|
continue
|
|
if not ans:
|
|
continue
|
|
txt = ans
|
|
fr.locator("input[name=captcha]").fill(txt)
|
|
time.sleep(0.4)
|
|
try:
|
|
val = fr.locator("input[name=captcha]").input_value()
|
|
print(f" field value after fill: {val!r}", file=sys.stderr)
|
|
except Exception as e:
|
|
print(f" readback failed: {e}", file=sys.stderr)
|
|
fr.locator(".ddg-modal__submit").click()
|
|
time.sleep(10)
|
|
# diagnostics
|
|
try:
|
|
fr2 = _find_srcdoc_captcha_frame(pg, timeout=2)
|
|
if fr2:
|
|
try:
|
|
val2 = fr2.locator("input[name=captcha]").input_value(timeout=1500)
|
|
print(f" field value after submit: {val2!r}", file=sys.stderr)
|
|
except Exception:
|
|
pass
|
|
try:
|
|
err = fr2.locator(".ddg-modal__error").inner_text(timeout=1500)
|
|
print(f" CAPTCHA ERROR: {err!r}", file=sys.stderr)
|
|
except Exception:
|
|
pass
|
|
except Exception:
|
|
pass
|
|
pg.screenshot(path="/tmp/aa-after-verify.png")
|
|
if "check=1" not in pg.url and "ddos-guard" not in pg.title().lower():
|
|
cookies = b.contexts[0].cookies()
|
|
with open(COOKIE_FILE, "w") as fh:
|
|
json.dump(cookies, fh)
|
|
print(f"[solve] PASSED. cookies -> {COOKIE_FILE} ({len(cookies)})",
|
|
file=sys.stderr)
|
|
print(pg.url)
|
|
return True
|
|
print(" verify failed", file=sys.stderr)
|
|
time.sleep(5)
|
|
die("challenge not passed in N rounds")
|
|
|
|
# ---------------------------------------------------------------- search / info (curl)
|
|
|
|
def parse_results(html):
|
|
"""New AA search page (server-rendered). Rows: /md5/<hash> anchors."""
|
|
rows = []
|
|
# result blocks: <a ... href="/md5/HEX" ...>TITLE</a> ... file info follows
|
|
for m in re.finditer(
|
|
r'href="/md5/([0-9a-f]{32})"[^>]*>([^<]{3,150})</a>(.{0,1500}?)'
|
|
r'(?=href="/md5/|$)', html, re.S):
|
|
md5, title, rest = m.group(1), m.group(2).strip(), m.group(3)
|
|
exts = re.findall(r'title="(\w{2,5})" aria-label', rest) or \
|
|
re.findall(r'>\s*(pdf|epub|mobi|fb2|djvu|doc|txt)\s*<', rest, re.I)
|
|
size = re.search(r'([\d.]+\s*(?:MB|KB|GB))\s*</', rest)
|
|
year = re.search(r'\b(19\d{2}|20\d{2})\b', rest)
|
|
rows.append({
|
|
"md5": md5,
|
|
"title": title,
|
|
"exts": list(dict.fromkeys(e.lower() for e in exts))[:8],
|
|
"size": size.group(1) if size else "",
|
|
"year": year.group(1) if year else "",
|
|
})
|
|
# de-dup by md5
|
|
seen, out = set(), []
|
|
for r in rows:
|
|
if r["md5"] not in seen:
|
|
seen.add(r["md5"])
|
|
out.append(r)
|
|
return out
|
|
|
|
|
|
def search(query, n=15):
|
|
q = urllib.parse.quote(query)
|
|
status, html = curl(f"{BASE}/search?q={q}")
|
|
if status != "200" or "ddos-guard" in html[:2000].lower():
|
|
die(f"search blocked (status {status}) — run: aa.py solve")
|
|
rows = parse_results(html)[:n]
|
|
if not rows:
|
|
print("(no rows parsed — check the page manually or aa.py solve)")
|
|
# save for debug
|
|
with open("/tmp/aa-search-debug.html", "w") as fh:
|
|
fh.write(html)
|
|
print("debug html -> /tmp/aa-search-debug.html")
|
|
for i, r in enumerate(rows, 1):
|
|
print(f'{i:2}. {r["md5"]} {"/".join(r["exts"]):14} {r["size"]:8} '
|
|
f'{r["year"]} {r["title"]}')
|
|
return rows
|
|
|
|
# ---------------------------------------------------------------- slow download
|
|
|
|
def find_slow_links(md5):
|
|
"""Book page -> list of (ext, url) for slow downloads."""
|
|
status, html = curl(f"{BASE}/md5/{md5}")
|
|
if status != "200" or "ddos-guard" in html[:2000].lower():
|
|
die(f"book page blocked (status {status}) — run: aa.py solve")
|
|
links = re.findall(
|
|
r'href="(https?://[^"]*(?:dlann|download)[^"]*)"[^>]*>\s*[^<]*?(\w{2,5})',
|
|
html, re.I)
|
|
# fallback: any external link containing the md5
|
|
if not links:
|
|
links = re.findall(r'href="(https?://[^"]*' + md5 + r'[^"]*)"', html)
|
|
return links, html
|
|
|
|
|
|
def wait_countdown(html, base_url):
|
|
"""Parse meta-refresh / JS countdown; return seconds to wait."""
|
|
m = re.search(r'content="(\d+);\s*url=', html, re.I)
|
|
if m:
|
|
return int(m.group(1))
|
|
m = re.search(r'(?:countdown|timer|wait)["\']?\s*[:=]\s*(\d{1,4})\b', html, re.I)
|
|
if m:
|
|
return int(m.group(1))
|
|
m = re.search(r'seconds?[\'"]?\s*[:=]\s*(\d{1,4})', html, re.I)
|
|
if m:
|
|
return int(m.group(1))
|
|
return 0
|
|
|
|
|
|
def dl(md5, outdir, ext=None, name=None, max_wait=300):
|
|
links, html = find_slow_links(md5)
|
|
if not links:
|
|
with open("/tmp/aa-book-debug.html", "w") as fh:
|
|
fh.write(html)
|
|
die("no download links found (debug -> /tmp/aa-book-debug.html)")
|
|
# pick link
|
|
chosen = None
|
|
for u, e in links:
|
|
if ext is None or e.lower() == ext:
|
|
chosen = u
|
|
break
|
|
if chosen is None:
|
|
chosen = links[0][0]
|
|
print(f"[dl] slow link: {chosen}")
|
|
# follow with countdown waits
|
|
url = chosen
|
|
for step in range(6):
|
|
cmd = ["curl", "-s", "-L", "--max-time", "60", "-A", UA,
|
|
"-w", "\n%{http_code}\n%{size_download}\n%{content_type}", url]
|
|
r = subprocess.run(cmd, capture_output=True)
|
|
body = r.stdout
|
|
# parse trailing meta
|
|
idx = body.rfind(b"\n200\n")
|
|
head = body[:idx] if idx > 0 else body
|
|
if len(head) > 200000:
|
|
# looks like the actual file
|
|
fname = name or f"{md5[:12]}.{ext or 'bin'}"
|
|
path = os.path.join(outdir, fname)
|
|
with open(path, "wb") as fh:
|
|
fh.write(head)
|
|
print(f"[dl] OK -> {path} ({len(head)} bytes)")
|
|
return path
|
|
status_m = re.search(rb'\n(4\d\d|5\d\d)\n', body)
|
|
text = head.decode("utf-8", "ignore")
|
|
wait = wait_countdown(text, url)
|
|
redir = re.search(r'content="\d+;\s*url=([^"]+)"', text, re.I)
|
|
print(f"[dl] step {step}: len={len(head)} wait={wait}s "
|
|
f"redir={'yes' if redir else 'no'}")
|
|
if redir:
|
|
url = redir.group(1)
|
|
if url.startswith("/"):
|
|
url = "https://annas-archive.gd" + url
|
|
continue
|
|
if wait and wait <= max_wait:
|
|
print(f"[dl] sleeping {wait}s ...")
|
|
time.sleep(wait + 2)
|
|
# re-request the same page (countdown pages usually self-redirect)
|
|
url = chosen if step == 0 else url
|
|
continue
|
|
die(f"download stalled: {text[:300]!r}")
|
|
die("too many redirects")
|
|
|
|
|
|
# ---------------------------------------------------------------- cli
|
|
|
|
if __name__ == "__main__":
|
|
args = sys.argv[1:]
|
|
if not args:
|
|
die(__doc__)
|
|
cmd, args = args[0], args[1:]
|
|
if cmd == "solve":
|
|
code = None
|
|
query = None
|
|
auto = True
|
|
rest = []
|
|
i = 0
|
|
while i < len(args):
|
|
if args[i] == "--code":
|
|
code = args[i + 1]; i += 2
|
|
elif args[i] == "--no-auto":
|
|
auto = False; i += 1
|
|
else:
|
|
rest.append(args[i]); i += 1
|
|
query = " ".join(rest) or None
|
|
solve(query, code, auto)
|
|
elif cmd == "search":
|
|
n = 15
|
|
q = None
|
|
rest = []
|
|
i = 0
|
|
while i < len(args):
|
|
if args[i] == "--n":
|
|
n = int(args[i + 1]); i += 2
|
|
else:
|
|
rest.append(args[i]); i += 1
|
|
q = " ".join(rest)
|
|
search(q, n)
|
|
elif cmd == "info":
|
|
md5 = args[0]
|
|
links, html = find_slow_links(md5)
|
|
t = re.search(r"<title>([^<]*)</title>", html)
|
|
print("title:", t.group(1) if t else "?")
|
|
for u, e in links[:12]:
|
|
print(f" {e:6} {u}")
|
|
elif cmd == "dl":
|
|
md5 = args[0]
|
|
outdir = args[1]
|
|
ext = None
|
|
name = None
|
|
i = 2
|
|
while i < len(args):
|
|
if args[i] == "--ext":
|
|
ext = args[i + 1]; i += 2
|
|
elif args[i] == "--name":
|
|
name = args[i + 1]; i += 2
|
|
else:
|
|
i += 1
|
|
dl(md5, outdir, ext, name)
|
|
else:
|
|
die(f"unknown command: {cmd}")
|