#!/usr/bin/env python3 """Flibusta (.is) OPDS catalog — search, author feeds, download links. Usage: tools/flib.py books "query" # book search (OPDS searchType=books) tools/flib.py authors "query" # author search (OPDS searchType=authors) tools/flib.py author # all books of author id (OPDS /opds/author//alphabet) Notes: - http://flibusta.is — OPDS catalog + direct file downloads (works 2026-07; user fixed access). - Entry fields: title, author (name + /a/), category, language, format, year (dc:issued), annotation (contains «Перевод: …» when present), acquisition links /b// (fb2, epub, mobi, html, txt; pdf/doc via «/b//download» on the /a/ or /b/ HTML pages). - Author HTML page: http://flibusta.is/a/ (title + full book list w/ format links). - Book HTML page: http://flibusta.is/b/. - Polite: 2.5s between requests. """ import re, sys, time, urllib.parse, urllib.request UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36" BASE = "http://flibusta.is" def fetch(url, tries=3): last = None for i in range(tries): try: req = urllib.request.Request(url, headers={"User-Agent": UA}) return urllib.request.urlopen(req, timeout=60).read().decode("utf-8", "replace") except Exception as e: last = e time.sleep(3 + 3 * i) raise SystemExit(f"fetch failed: {last}") def parse_entries(xml): out = [] for e in re.findall(r".*?", xml, re.S): def g(pat): m = re.search(pat, e, re.S) return m.group(1).strip() if m else "" title = g(r"(.*?)") aname = g(r"\s*(.*?)") aid = g(r"(/a/\d+)") year = g(r"(.*?)") fmt = g(r"(.*?)") lang = g(r"(.*?)") bid = g(r'(.*?)") trans = re.search(r"Перевод:?\s*]*>?\s*([^&<]+)", ann) if not trans: trans = re.search(r"Перевод:(?:
|]*>)?([^&]+?).*?", xml, re.S): aid = re.search(r"tag:author:(\d+)", e) t = re.search(r"(.*?)", e, re.S) n = re.search(r"(.*?)", e, re.S) rows.append((aid.group(1) if aid else "?", t.group(1).strip() if t else "?", n.group(1).strip() if n else "?")) for aid, name, n in rows: print(f"a/{aid} {name} — {n}") if not rows: print("(no results)") return rows = parse_entries(xml) if not rows: print("(no results)") return for r in rows: t = f" [пер. {r['trans']}]" if r["trans"] else "" print(f"b/{r['id']} {r['year'] or '?'} {r['fmt'] or '?':10s} {r['author'] or '?'} — {r['title'][:70]}{t}") print(f"--- {len(rows)} entries") if __name__ == "__main__": main()