tools/aa.py: AA search/dl tool (cookie-based curl + camoufox DDG hybrid solver: checkbox->captcha PNG->/tmp/aa-code; search/info/dl work with saved cookies; challenge auto-pass blocked by headless WebGL fingerprint — cookies from real browser needed)
This commit is contained in:
parent
d3cf00c6ca
commit
bf89857037
1 changed files with 436 additions and 0 deletions
436
tools/aa.py
Normal file
436
tools/aa.py
Normal file
|
|
@ -0,0 +1,436 @@
|
|||
#!/usr/bin/env python3
|
||||
"""Anna's Archive — search + slow download (no account).
|
||||
|
||||
Challenge: DDoS-Guard JS check + text captcha. Solved with camoufox (headless Firefox
|
||||
stealth fork). The captcha OCR (RapidOCR) is unreliable -> HYBRID:
|
||||
1) auto OCR attempt (best effort),
|
||||
2) on failure the captcha image is saved to /tmp/aa-captcha.png and the operator
|
||||
(agent with vision) reads it and passes --code, or `solve` prompts on stdin.
|
||||
|
||||
After a passed challenge the DDG cookies (data/aa-cookies.json) let PLAIN CURL fetch
|
||||
search/book pages for days. Slow downloads (dl1.dlann.com, countdown) are plain HTTP
|
||||
with waits — no browser needed.
|
||||
|
||||
Commands:
|
||||
aa.py solve [query] [--code XXXXX] [--auto] pass DDG challenge, save cookies
|
||||
aa.py search "query" [--n N] search, print rows: md5 | ext | size | year | title
|
||||
aa.py info <md5> book page: metadata + available files
|
||||
aa.py dl <md5> <outdir> [--ext pdf] [--name FILE] slow download (countdown-aware)
|
||||
|
||||
Politeness: one challenge pass per day is fine; searches ~2s apart.
|
||||
Requires: pip --user --break-system-packages camoufox[geoip] rapidocr_onnxruntime
|
||||
(python-pip packages, camoufox browser: python3 -m camoufox fetch)
|
||||
"""
|
||||
import base64
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import random
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
import urllib.parse
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
COOKIE_FILE = os.path.join(ROOT, "data", "aa-cookies.json")
|
||||
BASE = "https://annas-archive.gd"
|
||||
UA = ("Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 "
|
||||
"(KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36")
|
||||
CAPTCHA_PNG = "/tmp/aa-captcha.png"
|
||||
|
||||
# ---------------------------------------------------------------- cookies (curl)
|
||||
|
||||
def load_cookies():
|
||||
if not os.path.exists(COOKIE_FILE):
|
||||
die("no cookies — run: aa.py solve first")
|
||||
with open(COOKIE_FILE) as fh:
|
||||
return json.load(fh)
|
||||
|
||||
|
||||
def curl(url, extra=None, out=None):
|
||||
"""Plain curl with AA cookies. Returns (status, body_text)."""
|
||||
cookies = load_cookies()
|
||||
# Netscape-ish conversion via --cookie
|
||||
cookie_str = "; ".join(f'{c["name"]}={c["value"]}' for c in cookies)
|
||||
cmd = ["curl", "-s", "-L", "--max-time", "120", "-A", UA,
|
||||
"--cookie", cookie_str, "-w", "\n%{http_code}"]
|
||||
cmd += extra or []
|
||||
cmd.append(url)
|
||||
r = subprocess.run(cmd, capture_output=True, text=True)
|
||||
body, _, status = r.stdout.rpartition("\n")
|
||||
if out:
|
||||
with open(out, "wb") as fh:
|
||||
fh.write(r.stdout.encode() if isinstance(body, str) else b"")
|
||||
return status, body
|
||||
|
||||
|
||||
def die(msg, code=1):
|
||||
print(f"aa.py: {msg}", file=sys.stderr)
|
||||
sys.exit(code)
|
||||
|
||||
# ---------------------------------------------------------------- camoufox challenge
|
||||
|
||||
def _find_srcdoc_captcha_frame(pg, timeout=45):
|
||||
t0 = time.time()
|
||||
while time.time() - t0 < timeout:
|
||||
for f in pg.frames:
|
||||
if "srcdoc" in (f.url or ""):
|
||||
try:
|
||||
if f.locator(".ddg-modal__captcha-image").count() > 0:
|
||||
return f
|
||||
except Exception:
|
||||
pass
|
||||
time.sleep(2)
|
||||
return None
|
||||
|
||||
|
||||
def _click_checkbox(pg):
|
||||
pg.mouse.move(random.randint(300, 600), random.randint(200, 400), steps=8)
|
||||
time.sleep(random.uniform(0.3, 0.6))
|
||||
pg.mouse.move(822, 592, steps=25)
|
||||
time.sleep(random.uniform(0.3, 0.7))
|
||||
pg.mouse.click(822, 592)
|
||||
|
||||
|
||||
def _get_captcha_png(fr):
|
||||
"""Fetch raw captcha data-URL PNG from the frame; returns bytes or None."""
|
||||
for _ in range(15):
|
||||
s = fr.evaluate(
|
||||
"() => { const e = document.querySelector('.ddg-modal__captcha-image');"
|
||||
" return e && e.src; }")
|
||||
if s and s.startswith("data:image"):
|
||||
return base64.b64decode(s.split(",", 1)[1])
|
||||
time.sleep(1)
|
||||
return None
|
||||
|
||||
|
||||
def _auto_ocr(captcha_png):
|
||||
"""Best-effort OCR. Returns (text, ok)."""
|
||||
try:
|
||||
import numpy as np
|
||||
from PIL import Image
|
||||
from rapidocr_onnxruntime import RapidOCR
|
||||
im = Image.open(io.BytesIO(captcha_png)).convert("RGB")
|
||||
arr = np.asarray(im)
|
||||
d = 255 - arr.min(axis=2)
|
||||
mask = (d > 60)
|
||||
im2 = Image.fromarray((255 - mask * 255).astype(np.uint8))
|
||||
im2 = im2.resize((im2.width * 5, im2.height * 5), Image.LANCZOS)
|
||||
ocr = RapidOCR()
|
||||
res, _ = ocr(np.asarray(im2))
|
||||
txt = "".join(r[1] for r in (res or [])).strip()
|
||||
conf = max([r[2] for r in (res or [])] + [0.0])
|
||||
ok = len(txt) >= 4 and conf >= 0.55
|
||||
return txt, ok
|
||||
except Exception as e:
|
||||
print(f" auto-ocr failed: {e}", file=sys.stderr)
|
||||
return "", False
|
||||
|
||||
|
||||
def _manual_code(refresh_fn, timeout=600):
|
||||
"""Wait for operator code: file /tmp/aa-code (poll) or stdin. 'r' -> refresh."""
|
||||
code_file = "/tmp/aa-code"
|
||||
if os.path.exists(code_file):
|
||||
os.remove(code_file)
|
||||
t0 = time.time()
|
||||
stdin_pending = None
|
||||
while time.time() - t0 < timeout:
|
||||
if os.path.exists(code_file):
|
||||
with open(code_file) as fh:
|
||||
return fh.read().strip()
|
||||
time.sleep(2)
|
||||
return None
|
||||
|
||||
|
||||
def solve(query=None, code=None, auto=True, max_rounds=4):
|
||||
from camoufox.sync_api import Camoufox
|
||||
q = urllib.parse.quote(query or "dreams")
|
||||
url = f"{BASE}/search?q={q}"
|
||||
with Camoufox(headless=True) as b:
|
||||
pg = b.new_page(viewport={"width": 1920, "height": 1080})
|
||||
for rnd in range(1, max_rounds + 1):
|
||||
print(f"[solve] round {rnd}: {url}", file=sys.stderr)
|
||||
pg.goto(url, wait_until="domcontentloaded", timeout=90000)
|
||||
if "check=1" not in pg.url and "ddos-guard" not in pg.title().lower():
|
||||
break
|
||||
# wait for manual check
|
||||
for _ in range(20):
|
||||
try:
|
||||
if "manual check" in pg.inner_text("body").lower():
|
||||
break
|
||||
except Exception:
|
||||
pass
|
||||
time.sleep(3)
|
||||
_click_checkbox(pg)
|
||||
fr = _find_srcdoc_captcha_frame(pg)
|
||||
if not fr:
|
||||
print("[solve] captcha modal not found", file=sys.stderr)
|
||||
continue
|
||||
for sub in range(3):
|
||||
fr = None
|
||||
png = None
|
||||
for _ in range(5):
|
||||
fr = _find_srcdoc_captcha_frame(pg, timeout=2)
|
||||
if fr:
|
||||
png = _get_captcha_png(fr)
|
||||
if png:
|
||||
break
|
||||
time.sleep(2)
|
||||
if not png or not fr:
|
||||
break
|
||||
with open(CAPTCHA_PNG, "wb") as fh:
|
||||
fh.write(png)
|
||||
txt = None
|
||||
if code:
|
||||
txt = code
|
||||
code = None # one-shot
|
||||
elif auto:
|
||||
txt, ok = _auto_ocr(png)
|
||||
print(f" auto-ocr: {txt!r} (ok={ok})", file=sys.stderr)
|
||||
if not ok:
|
||||
txt = None
|
||||
if txt is None:
|
||||
# hybrid fallback: operator reads the PNG (agent workflow:
|
||||
# solve runs in background; agent reads /tmp/aa-captcha.png
|
||||
# and writes the code to /tmp/aa-code)
|
||||
print(f"[solve] captcha image -> {CAPTCHA_PNG}; "
|
||||
f"waiting for code in /tmp/aa-code ...", file=sys.stderr)
|
||||
ans = _manual_code(lambda: None)
|
||||
if ans is None:
|
||||
die("timeout waiting for /tmp/aa-code")
|
||||
if ans.lower() == "q":
|
||||
die("aborted")
|
||||
if ans.lower() == "r":
|
||||
fr.locator(".ddg-modal__refresh").click()
|
||||
time.sleep(2.5)
|
||||
continue
|
||||
if not ans:
|
||||
continue
|
||||
txt = ans
|
||||
fr.locator("input[name=captcha]").fill(txt)
|
||||
time.sleep(0.4)
|
||||
try:
|
||||
val = fr.locator("input[name=captcha]").input_value()
|
||||
print(f" field value after fill: {val!r}", file=sys.stderr)
|
||||
except Exception as e:
|
||||
print(f" readback failed: {e}", file=sys.stderr)
|
||||
fr.locator(".ddg-modal__submit").click()
|
||||
time.sleep(10)
|
||||
# diagnostics
|
||||
try:
|
||||
fr2 = _find_srcdoc_captcha_frame(pg, timeout=2)
|
||||
if fr2:
|
||||
try:
|
||||
val2 = fr2.locator("input[name=captcha]").input_value(timeout=1500)
|
||||
print(f" field value after submit: {val2!r}", file=sys.stderr)
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
err = fr2.locator(".ddg-modal__error").inner_text(timeout=1500)
|
||||
print(f" CAPTCHA ERROR: {err!r}", file=sys.stderr)
|
||||
except Exception:
|
||||
pass
|
||||
except Exception:
|
||||
pass
|
||||
pg.screenshot(path="/tmp/aa-after-verify.png")
|
||||
if "check=1" not in pg.url and "ddos-guard" not in pg.title().lower():
|
||||
cookies = b.contexts[0].cookies()
|
||||
with open(COOKIE_FILE, "w") as fh:
|
||||
json.dump(cookies, fh)
|
||||
print(f"[solve] PASSED. cookies -> {COOKIE_FILE} ({len(cookies)})",
|
||||
file=sys.stderr)
|
||||
print(pg.url)
|
||||
return True
|
||||
print(" verify failed", file=sys.stderr)
|
||||
time.sleep(5)
|
||||
die("challenge not passed in N rounds")
|
||||
|
||||
# ---------------------------------------------------------------- search / info (curl)
|
||||
|
||||
def parse_results(html):
|
||||
"""New AA search page (server-rendered). Rows: /md5/<hash> anchors."""
|
||||
rows = []
|
||||
# result blocks: <a ... href="/md5/HEX" ...>TITLE</a> ... file info follows
|
||||
for m in re.finditer(
|
||||
r'href="/md5/([0-9a-f]{32})"[^>]*>([^<]{3,150})</a>(.{0,1500}?)'
|
||||
r'(?=href="/md5/|$)', html, re.S):
|
||||
md5, title, rest = m.group(1), m.group(2).strip(), m.group(3)
|
||||
exts = re.findall(r'title="(\w{2,5})" aria-label', rest) or \
|
||||
re.findall(r'>\s*(pdf|epub|mobi|fb2|djvu|doc|txt)\s*<', rest, re.I)
|
||||
size = re.search(r'([\d.]+\s*(?:MB|KB|GB))\s*</', rest)
|
||||
year = re.search(r'\b(19\d{2}|20\d{2})\b', rest)
|
||||
rows.append({
|
||||
"md5": md5,
|
||||
"title": title,
|
||||
"exts": list(dict.fromkeys(e.lower() for e in exts))[:8],
|
||||
"size": size.group(1) if size else "",
|
||||
"year": year.group(1) if year else "",
|
||||
})
|
||||
# de-dup by md5
|
||||
seen, out = set(), []
|
||||
for r in rows:
|
||||
if r["md5"] not in seen:
|
||||
seen.add(r["md5"])
|
||||
out.append(r)
|
||||
return out
|
||||
|
||||
|
||||
def search(query, n=15):
|
||||
q = urllib.parse.quote(query)
|
||||
status, html = curl(f"{BASE}/search?q={q}")
|
||||
if status != "200" or "ddos-guard" in html[:2000].lower():
|
||||
die(f"search blocked (status {status}) — run: aa.py solve")
|
||||
rows = parse_results(html)[:n]
|
||||
if not rows:
|
||||
print("(no rows parsed — check the page manually or aa.py solve)")
|
||||
# save for debug
|
||||
with open("/tmp/aa-search-debug.html", "w") as fh:
|
||||
fh.write(html)
|
||||
print("debug html -> /tmp/aa-search-debug.html")
|
||||
for i, r in enumerate(rows, 1):
|
||||
print(f'{i:2}. {r["md5"]} {"/".join(r["exts"]):14} {r["size"]:8} '
|
||||
f'{r["year"]} {r["title"]}')
|
||||
return rows
|
||||
|
||||
# ---------------------------------------------------------------- slow download
|
||||
|
||||
def find_slow_links(md5):
|
||||
"""Book page -> list of (ext, url) for slow downloads."""
|
||||
status, html = curl(f"{BASE}/md5/{md5}")
|
||||
if status != "200" or "ddos-guard" in html[:2000].lower():
|
||||
die(f"book page blocked (status {status}) — run: aa.py solve")
|
||||
links = re.findall(
|
||||
r'href="(https?://[^"]*(?:dlann|download)[^"]*)"[^>]*>\s*[^<]*?(\w{2,5})',
|
||||
html, re.I)
|
||||
# fallback: any external link containing the md5
|
||||
if not links:
|
||||
links = re.findall(r'href="(https?://[^"]*' + md5 + r'[^"]*)"', html)
|
||||
return links, html
|
||||
|
||||
|
||||
def wait_countdown(html, base_url):
|
||||
"""Parse meta-refresh / JS countdown; return seconds to wait."""
|
||||
m = re.search(r'content="(\d+);\s*url=', html, re.I)
|
||||
if m:
|
||||
return int(m.group(1))
|
||||
m = re.search(r'(?:countdown|timer|wait)["\']?\s*[:=]\s*(\d{1,4})\b', html, re.I)
|
||||
if m:
|
||||
return int(m.group(1))
|
||||
m = re.search(r'seconds?[\'"]?\s*[:=]\s*(\d{1,4})', html, re.I)
|
||||
if m:
|
||||
return int(m.group(1))
|
||||
return 0
|
||||
|
||||
|
||||
def dl(md5, outdir, ext=None, name=None, max_wait=300):
|
||||
links, html = find_slow_links(md5)
|
||||
if not links:
|
||||
with open("/tmp/aa-book-debug.html", "w") as fh:
|
||||
fh.write(html)
|
||||
die("no download links found (debug -> /tmp/aa-book-debug.html)")
|
||||
# pick link
|
||||
chosen = None
|
||||
for u, e in links:
|
||||
if ext is None or e.lower() == ext:
|
||||
chosen = u
|
||||
break
|
||||
if chosen is None:
|
||||
chosen = links[0][0]
|
||||
print(f"[dl] slow link: {chosen}")
|
||||
# follow with countdown waits
|
||||
url = chosen
|
||||
for step in range(6):
|
||||
cmd = ["curl", "-s", "-L", "--max-time", "60", "-A", UA,
|
||||
"-w", "\n%{http_code}\n%{size_download}\n%{content_type}", url]
|
||||
r = subprocess.run(cmd, capture_output=True)
|
||||
body = r.stdout
|
||||
# parse trailing meta
|
||||
idx = body.rfind(b"\n200\n")
|
||||
head = body[:idx] if idx > 0 else body
|
||||
if len(head) > 200000:
|
||||
# looks like the actual file
|
||||
fname = name or f"{md5[:12]}.{ext or 'bin'}"
|
||||
path = os.path.join(outdir, fname)
|
||||
with open(path, "wb") as fh:
|
||||
fh.write(head)
|
||||
print(f"[dl] OK -> {path} ({len(head)} bytes)")
|
||||
return path
|
||||
status_m = re.search(rb'\n(4\d\d|5\d\d)\n', body)
|
||||
text = head.decode("utf-8", "ignore")
|
||||
wait = wait_countdown(text, url)
|
||||
redir = re.search(r'content="\d+;\s*url=([^"]+)"', text, re.I)
|
||||
print(f"[dl] step {step}: len={len(head)} wait={wait}s "
|
||||
f"redir={'yes' if redir else 'no'}")
|
||||
if redir:
|
||||
url = redir.group(1)
|
||||
if url.startswith("/"):
|
||||
url = "https://annas-archive.gd" + url
|
||||
continue
|
||||
if wait and wait <= max_wait:
|
||||
print(f"[dl] sleeping {wait}s ...")
|
||||
time.sleep(wait + 2)
|
||||
# re-request the same page (countdown pages usually self-redirect)
|
||||
url = chosen if step == 0 else url
|
||||
continue
|
||||
die(f"download stalled: {text[:300]!r}")
|
||||
die("too many redirects")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------- cli
|
||||
|
||||
if __name__ == "__main__":
|
||||
args = sys.argv[1:]
|
||||
if not args:
|
||||
die(__doc__)
|
||||
cmd, args = args[0], args[1:]
|
||||
if cmd == "solve":
|
||||
code = None
|
||||
query = None
|
||||
auto = True
|
||||
rest = []
|
||||
i = 0
|
||||
while i < len(args):
|
||||
if args[i] == "--code":
|
||||
code = args[i + 1]; i += 2
|
||||
elif args[i] == "--no-auto":
|
||||
auto = False; i += 1
|
||||
else:
|
||||
rest.append(args[i]); i += 1
|
||||
query = " ".join(rest) or None
|
||||
solve(query, code, auto)
|
||||
elif cmd == "search":
|
||||
n = 15
|
||||
q = None
|
||||
rest = []
|
||||
i = 0
|
||||
while i < len(args):
|
||||
if args[i] == "--n":
|
||||
n = int(args[i + 1]); i += 2
|
||||
else:
|
||||
rest.append(args[i]); i += 1
|
||||
q = " ".join(rest)
|
||||
search(q, n)
|
||||
elif cmd == "info":
|
||||
md5 = args[0]
|
||||
links, html = find_slow_links(md5)
|
||||
t = re.search(r"<title>([^<]*)</title>", html)
|
||||
print("title:", t.group(1) if t else "?")
|
||||
for u, e in links[:12]:
|
||||
print(f" {e:6} {u}")
|
||||
elif cmd == "dl":
|
||||
md5 = args[0]
|
||||
outdir = args[1]
|
||||
ext = None
|
||||
name = None
|
||||
i = 2
|
||||
while i < len(args):
|
||||
if args[i] == "--ext":
|
||||
ext = args[i + 1]; i += 2
|
||||
elif args[i] == "--name":
|
||||
name = args[i + 1]; i += 2
|
||||
else:
|
||||
i += 1
|
||||
dl(md5, outdir, ext, name)
|
||||
else:
|
||||
die(f"unknown command: {cmd}")
|
||||
Loading…
Add table
Add a link
Reference in a new issue