alib.ru source (tools/alib.py, CP1251 stem-search marketplace): Turner handbook = Дипак 2015 ISBN 978-5-98580-069-2 (RSL 7609331); Furth 2nd ed 2014; Neumann market round 4 (КДУ/Питер/Маниф/Касталия editions, Angelika Lev bio 2021); notes #16/#26/#29 updated

This commit is contained in:
Dmitry Kokorin 2026-09-09 00:19:52 +03:00
parent 653f97f384
commit 5a448d8353
7 changed files with 96 additions and 4 deletions

47
tools/alib.py Executable file
View file

@ -0,0 +1,47 @@
#!/usr/bin/env python3
"""alib.ru book-marketplace search (3.2M listings, RU+KZ sellers).
Usage: tools/alib.py "query"
Site specifics (verified 2026-07-09):
- Endpoint: https://www.alib.ru/find3.php4?tfind=<query>
- Encoding: query must be **CP1251** percent-encoded (utf8 → mojibake "РќРѕР").
- Pages are CP1251.
- Search is STEM-BASED (word matches all inflections — unlike libgen) + phrase mode with
double quotes; also author-first, year-range, ISBN-digits, price-range fields.
- Result rows: "<cat> || <Author> <Title> |Серия: ... <City> <Year> <Pages> <Binding>.
(Seller X, city.) Цена: N руб." — i.e. listings, not library records: good for
small-press / rare / out-of-print RU editions that shops don't carry.
"""
import re, sys, time, urllib.parse, urllib.request
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"
def alib(q):
enc = urllib.parse.quote(q.encode("cp1251", errors="replace"))
url = f"https://www.alib.ru/find3.php4?tfind={enc}"
req = urllib.request.Request(url, headers={"User-Agent": UA})
with urllib.request.urlopen(req, timeout=40) as r:
return r.read().decode("cp1251", errors="replace")
def main():
if len(sys.argv) < 2:
sys.exit(__doc__)
q = " ".join(sys.argv[1:])
h = alib(q)
n = len(re.findall(r'>\s*Купить\s*<', h))
print(f"listings: {n}")
# each listing: text before 'Цена: N руб.' within the same row block
for m in re.finditer(r'(.{150,420}?)\s*\(До заказа[^)]*\)\s*Цена:\s*(\d+)\s*руб', h, re.S):
raw = m.group(1)
# cut back to the category row boundary
raw = raw[raw.rfind('</td>') + 5:]
t = re.sub(r'<[^>]+>', ' ', raw)
t = re.sub(r'&nbsp;', ' ', t)
t = re.sub(r'\s+', ' ', t).strip()
print(f"- {t} [price {m.group(2)} руб]")
print()
time.sleep(2)
if __name__ == "__main__":
main()