#!/usr/bin/env python3 """alib.ru book-marketplace search (3.2M listings, RU+KZ sellers). Usage: tools/alib.py "query" Site specifics (verified 2026-07-09): - Endpoint: https://www.alib.ru/find3.php4?tfind= - Encoding: query must be **CP1251** percent-encoded (utf8 → mojibake "РќРѕР"). - Pages are CP1251. - Search is STEM-BASED (word matches all inflections — unlike libgen) + phrase mode with double quotes; also author-first, year-range, ISBN-digits, price-range fields. - Result rows: " || |Серия: ... <City> <Year> <Pages> <Binding>. (Seller X, city.) Цена: N руб." — i.e. listings, not library records: good for small-press / rare / out-of-print RU editions that shops don't carry. """ import re, sys, time, urllib.parse, urllib.request UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36" def alib(q): enc = urllib.parse.quote(q.encode("cp1251", errors="replace")) url = f"https://www.alib.ru/find3.php4?tfind={enc}" req = urllib.request.Request(url, headers={"User-Agent": UA}) with urllib.request.urlopen(req, timeout=40) as r: return r.read().decode("cp1251", errors="replace") def main(): if len(sys.argv) < 2: sys.exit(__doc__) q = " ".join(sys.argv[1:]) h = alib(q) n = len(re.findall(r'>\s*Купить\s*<', h)) print(f"listings: {n}") # each listing: text before 'Цена: N руб.' within the same row block for m in re.finditer(r'(.{150,420}?)\s*\(До заказа[^)]*\)\s*Цена:\s*(\d+)\s*руб', h, re.S): raw = m.group(1) # cut back to the category row boundary raw = raw[raw.rfind('</td>') + 5:] t = re.sub(r'<[^>]+>', ' ', raw) t = re.sub(r' ', ' ', t) t = re.sub(r'\s+', ' ', t).strip() print(f"- {t} [price {m.group(2)} руб]") print() time.sleep(2) if __name__ == "__main__": main()