sec01: phase 4 downloads complete (50 files, 204MB) + item 12 EN preview + flib.py pagination

- 45 RU + 5 EN + 1 preview downloaded (flibusta.is OPDS, libgen, archive.org)
- 10 flib pdf stubs replaced via libgen (18,22,28,29,53) or fb2 twins
- flib 'fb2+zip' files unpacked to plain fb2
- item 12 (Evers-Fahey): RG=Request-PDF; PagePlace 23pp preview saved+verified
- evers-fahey dossier enriched (ofj.org bio)
- flib.py: authorall (rel=next pagination) + author <page>
- AGENTS.md: flib pagination/quirks + PagePlace preview tip
- INDEX: phase 4 done line; 8 trade-only items listed in MANIFEST
This commit is contained in:
Dmitry Kokorin 2026-09-17 10:41:43 +03:00
parent 5063bd8174
commit 32e505b2f9
8 changed files with 131 additions and 11 deletions

View file

@ -62,10 +62,24 @@ def main():
elif cmd == "authors":
url = f"{BASE}/opds/search?searchType=authors&searchTerm=" + urllib.parse.quote(q)
elif cmd == "author":
page = sys.argv[3] if len(sys.argv) > 3 else ""
url = f"{BASE}/opds/author/{q}/alphabet" + (f"/{page}" if page else "")
elif cmd == "authorall":
url = f"{BASE}/opds/author/{q}/alphabet"
else:
sys.exit("cmd: books|authors|author")
xml = fetch(url)
sys.exit("cmd: books|authors|author[page]|authorall")
if cmd == "authorall":
all_rows = []
xml = ""
while url and len(all_rows) < 600:
xml = fetch(url)
all_rows.extend(parse_entries(xml))
m = re.search(r'<link href="([^"]+)" rel="next"', xml)
url = (BASE + m.group(1)) if m else None
rows = all_rows
else:
xml = fetch(url)
rows = parse_entries(xml)
if cmd == "authors":
rows = []
for e in re.findall(r"<entry>.*?</entry>", xml, re.S):
@ -80,7 +94,6 @@ def main():
if not rows:
print("(no results)")
return
rows = parse_entries(xml)
if not rows:
print("(no results)")
return