- 45 RU + 5 EN + 1 preview downloaded (flibusta.is OPDS, libgen, archive.org) - 10 flib pdf stubs replaced via libgen (18,22,28,29,53) or fb2 twins - flib 'fb2+zip' files unpacked to plain fb2 - item 12 (Evers-Fahey): RG=Request-PDF; PagePlace 23pp preview saved+verified - evers-fahey dossier enriched (ofj.org bio) - flib.py: authorall (rel=next pagination) + author <page> - AGENTS.md: flib pagination/quirks + PagePlace preview tip - INDEX: phase 4 done line; 8 trade-only items listed in MANIFEST
106 lines
4.4 KiB
Python
Executable file
106 lines
4.4 KiB
Python
Executable file
#!/usr/bin/env python3
|
|
"""Flibusta (.is) OPDS catalog — search, author feeds, download links.
|
|
|
|
Usage:
|
|
tools/flib.py books "query" # book search (OPDS searchType=books)
|
|
tools/flib.py authors "query" # author search (OPDS searchType=authors)
|
|
tools/flib.py author <id> # all books of author id (OPDS /opds/author/<id>/alphabet)
|
|
|
|
Notes:
|
|
- http://flibusta.is — OPDS catalog + direct file downloads (works 2026-07; user fixed access).
|
|
- Entry fields: title, author (name + /a/<id>), category, language, format, year (dc:issued),
|
|
annotation (contains «Перевод: …» when present), acquisition links /b/<id>/<fmt>
|
|
(fb2, epub, mobi, html, txt; pdf/doc via «/b/<id>/download» on the /a/ or /b/ HTML pages).
|
|
- Author HTML page: http://flibusta.is/a/<id> (title + full book list w/ format links).
|
|
- Book HTML page: http://flibusta.is/b/<id>.
|
|
- Polite: 2.5s between requests.
|
|
"""
|
|
import re, sys, time, urllib.parse, urllib.request
|
|
|
|
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"
|
|
BASE = "http://flibusta.is"
|
|
|
|
def fetch(url, tries=3):
|
|
last = None
|
|
for i in range(tries):
|
|
try:
|
|
req = urllib.request.Request(url, headers={"User-Agent": UA})
|
|
return urllib.request.urlopen(req, timeout=60).read().decode("utf-8", "replace")
|
|
except Exception as e:
|
|
last = e
|
|
time.sleep(3 + 3 * i)
|
|
raise SystemExit(f"fetch failed: {last}")
|
|
|
|
def parse_entries(xml):
|
|
out = []
|
|
for e in re.findall(r"<entry>.*?</entry>", xml, re.S):
|
|
def g(pat):
|
|
m = re.search(pat, e, re.S)
|
|
return m.group(1).strip() if m else ""
|
|
title = g(r"<title>(.*?)</title>")
|
|
aname = g(r"<author>\s*<name>(.*?)</name>")
|
|
aid = g(r"<uri>(/a/\d+)</uri>")
|
|
year = g(r"<dc:issued>(.*?)</dc:issued>")
|
|
fmt = g(r"<dc:format>(.*?)</dc:format>")
|
|
lang = g(r"<dc:language>(.*?)</dc:language>")
|
|
bid = g(r'<link href="/b/(\d+)/[a-z0-9]+"')
|
|
ann = g(r"<content type=\"text/html\">(.*?)</content>")
|
|
trans = re.search(r"Перевод:?\s*</?[^>]*>?\s*([^&<]+)", ann)
|
|
if not trans:
|
|
trans = re.search(r"Перевод:(?:</br>|<br[^>]*>)?([^&]+?)<br", ann)
|
|
trans = trans.group(1).strip() if trans else ""
|
|
out.append(dict(id=bid, title=title, author=aname, aid=aid, year=year,
|
|
fmt=fmt, lang=lang, trans=trans))
|
|
return out
|
|
|
|
def main():
|
|
if len(sys.argv) < 3:
|
|
sys.exit(__doc__)
|
|
cmd, q = sys.argv[1], sys.argv[2]
|
|
if cmd == "books":
|
|
url = f"{BASE}/opds/search?searchType=books&searchTerm=" + urllib.parse.quote(q)
|
|
elif cmd == "authors":
|
|
url = f"{BASE}/opds/search?searchType=authors&searchTerm=" + urllib.parse.quote(q)
|
|
elif cmd == "author":
|
|
page = sys.argv[3] if len(sys.argv) > 3 else ""
|
|
url = f"{BASE}/opds/author/{q}/alphabet" + (f"/{page}" if page else "")
|
|
elif cmd == "authorall":
|
|
url = f"{BASE}/opds/author/{q}/alphabet"
|
|
else:
|
|
sys.exit("cmd: books|authors|author[page]|authorall")
|
|
if cmd == "authorall":
|
|
all_rows = []
|
|
xml = ""
|
|
while url and len(all_rows) < 600:
|
|
xml = fetch(url)
|
|
all_rows.extend(parse_entries(xml))
|
|
m = re.search(r'<link href="([^"]+)" rel="next"', xml)
|
|
url = (BASE + m.group(1)) if m else None
|
|
rows = all_rows
|
|
else:
|
|
xml = fetch(url)
|
|
rows = parse_entries(xml)
|
|
if cmd == "authors":
|
|
rows = []
|
|
for e in re.findall(r"<entry>.*?</entry>", xml, re.S):
|
|
aid = re.search(r"tag:author:(\d+)", e)
|
|
t = re.search(r"<title>(.*?)</title>", e, re.S)
|
|
n = re.search(r"<content type=\"text\">(.*?)</content>", e, re.S)
|
|
rows.append((aid.group(1) if aid else "?",
|
|
t.group(1).strip() if t else "?",
|
|
n.group(1).strip() if n else "?"))
|
|
for aid, name, n in rows:
|
|
print(f"a/{aid} {name} — {n}")
|
|
if not rows:
|
|
print("(no results)")
|
|
return
|
|
if not rows:
|
|
print("(no results)")
|
|
return
|
|
for r in rows:
|
|
t = f" [пер. {r['trans']}]" if r["trans"] else ""
|
|
print(f"b/{r['id']} {r['year'] or '?'} {r['fmt'] or '?':10s} {r['author'] or '?'} — {r['title'][:70]}{t}")
|
|
print(f"--- {len(rows)} entries")
|
|
|
|
if __name__ == "__main__":
|
|
main()
|