jung/tools/aa.py

436 lines
16 KiB
Python

#!/usr/bin/env python3
"""Anna's Archive — search + slow download (no account).
Challenge: DDoS-Guard JS check + text captcha. Solved with camoufox (headless Firefox
stealth fork). The captcha OCR (RapidOCR) is unreliable -> HYBRID:
1) auto OCR attempt (best effort),
2) on failure the captcha image is saved to /tmp/aa-captcha.png and the operator
(agent with vision) reads it and passes --code, or `solve` prompts on stdin.
After a passed challenge the DDG cookies (data/aa-cookies.json) let PLAIN CURL fetch
search/book pages for days. Slow downloads (dl1.dlann.com, countdown) are plain HTTP
with waits — no browser needed.
Commands:
aa.py solve [query] [--code XXXXX] [--auto] pass DDG challenge, save cookies
aa.py search "query" [--n N] search, print rows: md5 | ext | size | year | title
aa.py info <md5> book page: metadata + available files
aa.py dl <md5> <outdir> [--ext pdf] [--name FILE] slow download (countdown-aware)
Politeness: one challenge pass per day is fine; searches ~2s apart.
Requires: pip --user --break-system-packages camoufox[geoip] rapidocr_onnxruntime
(python-pip packages, camoufox browser: python3 -m camoufox fetch)
"""
import base64
import io
import json
import os
import random
import re
import subprocess
import sys
import time
import urllib.parse
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
COOKIE_FILE = os.path.join(ROOT, "data", "aa-cookies.json")
BASE = "https://annas-archive.gd"
UA = ("Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36")
CAPTCHA_PNG = "/tmp/aa-captcha.png"
# ---------------------------------------------------------------- cookies (curl)
def load_cookies():
if not os.path.exists(COOKIE_FILE):
die("no cookies — run: aa.py solve first")
with open(COOKIE_FILE) as fh:
return json.load(fh)
def curl(url, extra=None, out=None):
"""Plain curl with AA cookies. Returns (status, body_text)."""
cookies = load_cookies()
# Netscape-ish conversion via --cookie
cookie_str = "; ".join(f'{c["name"]}={c["value"]}' for c in cookies)
cmd = ["curl", "-s", "-L", "--max-time", "120", "-A", UA,
"--cookie", cookie_str, "-w", "\n%{http_code}"]
cmd += extra or []
cmd.append(url)
r = subprocess.run(cmd, capture_output=True, text=True)
body, _, status = r.stdout.rpartition("\n")
if out:
with open(out, "wb") as fh:
fh.write(r.stdout.encode() if isinstance(body, str) else b"")
return status, body
def die(msg, code=1):
print(f"aa.py: {msg}", file=sys.stderr)
sys.exit(code)
# ---------------------------------------------------------------- camoufox challenge
def _find_srcdoc_captcha_frame(pg, timeout=45):
t0 = time.time()
while time.time() - t0 < timeout:
for f in pg.frames:
if "srcdoc" in (f.url or ""):
try:
if f.locator(".ddg-modal__captcha-image").count() > 0:
return f
except Exception:
pass
time.sleep(2)
return None
def _click_checkbox(pg):
pg.mouse.move(random.randint(300, 600), random.randint(200, 400), steps=8)
time.sleep(random.uniform(0.3, 0.6))
pg.mouse.move(822, 592, steps=25)
time.sleep(random.uniform(0.3, 0.7))
pg.mouse.click(822, 592)
def _get_captcha_png(fr):
"""Fetch raw captcha data-URL PNG from the frame; returns bytes or None."""
for _ in range(15):
s = fr.evaluate(
"() => { const e = document.querySelector('.ddg-modal__captcha-image');"
" return e && e.src; }")
if s and s.startswith("data:image"):
return base64.b64decode(s.split(",", 1)[1])
time.sleep(1)
return None
def _auto_ocr(captcha_png):
"""Best-effort OCR. Returns (text, ok)."""
try:
import numpy as np
from PIL import Image
from rapidocr_onnxruntime import RapidOCR
im = Image.open(io.BytesIO(captcha_png)).convert("RGB")
arr = np.asarray(im)
d = 255 - arr.min(axis=2)
mask = (d > 60)
im2 = Image.fromarray((255 - mask * 255).astype(np.uint8))
im2 = im2.resize((im2.width * 5, im2.height * 5), Image.LANCZOS)
ocr = RapidOCR()
res, _ = ocr(np.asarray(im2))
txt = "".join(r[1] for r in (res or [])).strip()
conf = max([r[2] for r in (res or [])] + [0.0])
ok = len(txt) >= 4 and conf >= 0.55
return txt, ok
except Exception as e:
print(f" auto-ocr failed: {e}", file=sys.stderr)
return "", False
def _manual_code(refresh_fn, timeout=600):
"""Wait for operator code: file /tmp/aa-code (poll) or stdin. 'r' -> refresh."""
code_file = "/tmp/aa-code"
if os.path.exists(code_file):
os.remove(code_file)
t0 = time.time()
stdin_pending = None
while time.time() - t0 < timeout:
if os.path.exists(code_file):
with open(code_file) as fh:
return fh.read().strip()
time.sleep(2)
return None
def solve(query=None, code=None, auto=True, max_rounds=4):
from camoufox.sync_api import Camoufox
q = urllib.parse.quote(query or "dreams")
url = f"{BASE}/search?q={q}"
with Camoufox(headless=True) as b:
pg = b.new_page(viewport={"width": 1920, "height": 1080})
for rnd in range(1, max_rounds + 1):
print(f"[solve] round {rnd}: {url}", file=sys.stderr)
pg.goto(url, wait_until="domcontentloaded", timeout=90000)
if "check=1" not in pg.url and "ddos-guard" not in pg.title().lower():
break
# wait for manual check
for _ in range(20):
try:
if "manual check" in pg.inner_text("body").lower():
break
except Exception:
pass
time.sleep(3)
_click_checkbox(pg)
fr = _find_srcdoc_captcha_frame(pg)
if not fr:
print("[solve] captcha modal not found", file=sys.stderr)
continue
for sub in range(3):
fr = None
png = None
for _ in range(5):
fr = _find_srcdoc_captcha_frame(pg, timeout=2)
if fr:
png = _get_captcha_png(fr)
if png:
break
time.sleep(2)
if not png or not fr:
break
with open(CAPTCHA_PNG, "wb") as fh:
fh.write(png)
txt = None
if code:
txt = code
code = None # one-shot
elif auto:
txt, ok = _auto_ocr(png)
print(f" auto-ocr: {txt!r} (ok={ok})", file=sys.stderr)
if not ok:
txt = None
if txt is None:
# hybrid fallback: operator reads the PNG (agent workflow:
# solve runs in background; agent reads /tmp/aa-captcha.png
# and writes the code to /tmp/aa-code)
print(f"[solve] captcha image -> {CAPTCHA_PNG}; "
f"waiting for code in /tmp/aa-code ...", file=sys.stderr)
ans = _manual_code(lambda: None)
if ans is None:
die("timeout waiting for /tmp/aa-code")
if ans.lower() == "q":
die("aborted")
if ans.lower() == "r":
fr.locator(".ddg-modal__refresh").click()
time.sleep(2.5)
continue
if not ans:
continue
txt = ans
fr.locator("input[name=captcha]").fill(txt)
time.sleep(0.4)
try:
val = fr.locator("input[name=captcha]").input_value()
print(f" field value after fill: {val!r}", file=sys.stderr)
except Exception as e:
print(f" readback failed: {e}", file=sys.stderr)
fr.locator(".ddg-modal__submit").click()
time.sleep(10)
# diagnostics
try:
fr2 = _find_srcdoc_captcha_frame(pg, timeout=2)
if fr2:
try:
val2 = fr2.locator("input[name=captcha]").input_value(timeout=1500)
print(f" field value after submit: {val2!r}", file=sys.stderr)
except Exception:
pass
try:
err = fr2.locator(".ddg-modal__error").inner_text(timeout=1500)
print(f" CAPTCHA ERROR: {err!r}", file=sys.stderr)
except Exception:
pass
except Exception:
pass
pg.screenshot(path="/tmp/aa-after-verify.png")
if "check=1" not in pg.url and "ddos-guard" not in pg.title().lower():
cookies = b.contexts[0].cookies()
with open(COOKIE_FILE, "w") as fh:
json.dump(cookies, fh)
print(f"[solve] PASSED. cookies -> {COOKIE_FILE} ({len(cookies)})",
file=sys.stderr)
print(pg.url)
return True
print(" verify failed", file=sys.stderr)
time.sleep(5)
die("challenge not passed in N rounds")
# ---------------------------------------------------------------- search / info (curl)
def parse_results(html):
"""New AA search page (server-rendered). Rows: /md5/<hash> anchors."""
rows = []
# result blocks: <a ... href="/md5/HEX" ...>TITLE</a> ... file info follows
for m in re.finditer(
r'href="/md5/([0-9a-f]{32})"[^>]*>([^<]{3,150})</a>(.{0,1500}?)'
r'(?=href="/md5/|$)', html, re.S):
md5, title, rest = m.group(1), m.group(2).strip(), m.group(3)
exts = re.findall(r'title="(\w{2,5})" aria-label', rest) or \
re.findall(r'>\s*(pdf|epub|mobi|fb2|djvu|doc|txt)\s*<', rest, re.I)
size = re.search(r'([\d.]+\s*(?:MB|KB|GB))\s*</', rest)
year = re.search(r'\b(19\d{2}|20\d{2})\b', rest)
rows.append({
"md5": md5,
"title": title,
"exts": list(dict.fromkeys(e.lower() for e in exts))[:8],
"size": size.group(1) if size else "",
"year": year.group(1) if year else "",
})
# de-dup by md5
seen, out = set(), []
for r in rows:
if r["md5"] not in seen:
seen.add(r["md5"])
out.append(r)
return out
def search(query, n=15):
q = urllib.parse.quote(query)
status, html = curl(f"{BASE}/search?q={q}")
if status != "200" or "ddos-guard" in html[:2000].lower():
die(f"search blocked (status {status}) — run: aa.py solve")
rows = parse_results(html)[:n]
if not rows:
print("(no rows parsed — check the page manually or aa.py solve)")
# save for debug
with open("/tmp/aa-search-debug.html", "w") as fh:
fh.write(html)
print("debug html -> /tmp/aa-search-debug.html")
for i, r in enumerate(rows, 1):
print(f'{i:2}. {r["md5"]} {"/".join(r["exts"]):14} {r["size"]:8} '
f'{r["year"]} {r["title"]}')
return rows
# ---------------------------------------------------------------- slow download
def find_slow_links(md5):
"""Book page -> list of (ext, url) for slow downloads."""
status, html = curl(f"{BASE}/md5/{md5}")
if status != "200" or "ddos-guard" in html[:2000].lower():
die(f"book page blocked (status {status}) — run: aa.py solve")
links = re.findall(
r'href="(https?://[^"]*(?:dlann|download)[^"]*)"[^>]*>\s*[^<]*?(\w{2,5})',
html, re.I)
# fallback: any external link containing the md5
if not links:
links = re.findall(r'href="(https?://[^"]*' + md5 + r'[^"]*)"', html)
return links, html
def wait_countdown(html, base_url):
"""Parse meta-refresh / JS countdown; return seconds to wait."""
m = re.search(r'content="(\d+);\s*url=', html, re.I)
if m:
return int(m.group(1))
m = re.search(r'(?:countdown|timer|wait)["\']?\s*[:=]\s*(\d{1,4})\b', html, re.I)
if m:
return int(m.group(1))
m = re.search(r'seconds?[\'"]?\s*[:=]\s*(\d{1,4})', html, re.I)
if m:
return int(m.group(1))
return 0
def dl(md5, outdir, ext=None, name=None, max_wait=300):
links, html = find_slow_links(md5)
if not links:
with open("/tmp/aa-book-debug.html", "w") as fh:
fh.write(html)
die("no download links found (debug -> /tmp/aa-book-debug.html)")
# pick link
chosen = None
for u, e in links:
if ext is None or e.lower() == ext:
chosen = u
break
if chosen is None:
chosen = links[0][0]
print(f"[dl] slow link: {chosen}")
# follow with countdown waits
url = chosen
for step in range(6):
cmd = ["curl", "-s", "-L", "--max-time", "60", "-A", UA,
"-w", "\n%{http_code}\n%{size_download}\n%{content_type}", url]
r = subprocess.run(cmd, capture_output=True)
body = r.stdout
# parse trailing meta
idx = body.rfind(b"\n200\n")
head = body[:idx] if idx > 0 else body
if len(head) > 200000:
# looks like the actual file
fname = name or f"{md5[:12]}.{ext or 'bin'}"
path = os.path.join(outdir, fname)
with open(path, "wb") as fh:
fh.write(head)
print(f"[dl] OK -> {path} ({len(head)} bytes)")
return path
status_m = re.search(rb'\n(4\d\d|5\d\d)\n', body)
text = head.decode("utf-8", "ignore")
wait = wait_countdown(text, url)
redir = re.search(r'content="\d+;\s*url=([^"]+)"', text, re.I)
print(f"[dl] step {step}: len={len(head)} wait={wait}s "
f"redir={'yes' if redir else 'no'}")
if redir:
url = redir.group(1)
if url.startswith("/"):
url = "https://annas-archive.gd" + url
continue
if wait and wait <= max_wait:
print(f"[dl] sleeping {wait}s ...")
time.sleep(wait + 2)
# re-request the same page (countdown pages usually self-redirect)
url = chosen if step == 0 else url
continue
die(f"download stalled: {text[:300]!r}")
die("too many redirects")
# ---------------------------------------------------------------- cli
if __name__ == "__main__":
args = sys.argv[1:]
if not args:
die(__doc__)
cmd, args = args[0], args[1:]
if cmd == "solve":
code = None
query = None
auto = True
rest = []
i = 0
while i < len(args):
if args[i] == "--code":
code = args[i + 1]; i += 2
elif args[i] == "--no-auto":
auto = False; i += 1
else:
rest.append(args[i]); i += 1
query = " ".join(rest) or None
solve(query, code, auto)
elif cmd == "search":
n = 15
q = None
rest = []
i = 0
while i < len(args):
if args[i] == "--n":
n = int(args[i + 1]); i += 2
else:
rest.append(args[i]); i += 1
q = " ".join(rest)
search(q, n)
elif cmd == "info":
md5 = args[0]
links, html = find_slow_links(md5)
t = re.search(r"<title>([^<]*)</title>", html)
print("title:", t.group(1) if t else "?")
for u, e in links[:12]:
print(f" {e:6} {u}")
elif cmd == "dl":
md5 = args[0]
outdir = args[1]
ext = None
name = None
i = 2
while i < len(args):
if args[i] == "--ext":
ext = args[i + 1]; i += 2
elif args[i] == "--name":
name = args[i + 1]; i += 2
else:
i += 1
dl(md5, outdir, ext, name)
else:
die(f"unknown command: {cmd}")