sec01 phase 2 (round 1): 17/55 items resolved
- new ✅: 03 (Структура и динамика психического, Когито), 10 (Эго и архетип, 2 изд.), 13 (Лекции по юнговской типологии, b/509587), 29 (Юнговская карта души, Стайн Мюррей 2010) - ❌ with evidence: 02 (CW7), 05 (Aion — 'Эйон' 0), 16 (Shadow and Self), 21 (Animus and Anima), 38 (Woman in the Mirror), 47 (Way of All Women) - flibusta.is OPDS sweep (tools/flib.py): 30+ author feeds checked; bonus books logged (Edinger x10, Hillman x8, Harding x4, Henderson x2, Douglas Cambridge Handbook, Jung-E Grail Legend) - tools/ia.py (archive.org EN search/metadata/files); AGENTS.md sources 10+13 updated
This commit is contained in:
parent
fa3f98d9ad
commit
4420aef7c3
26 changed files with 351 additions and 119 deletions
93
tools/flib.py
Executable file
93
tools/flib.py
Executable file
|
|
@ -0,0 +1,93 @@
|
|||
#!/usr/bin/env python3
|
||||
"""Flibusta (.is) OPDS catalog — search, author feeds, download links.
|
||||
|
||||
Usage:
|
||||
tools/flib.py books "query" # book search (OPDS searchType=books)
|
||||
tools/flib.py authors "query" # author search (OPDS searchType=authors)
|
||||
tools/flib.py author <id> # all books of author id (OPDS /opds/author/<id>/alphabet)
|
||||
|
||||
Notes:
|
||||
- http://flibusta.is — OPDS catalog + direct file downloads (works 2026-07; user fixed access).
|
||||
- Entry fields: title, author (name + /a/<id>), category, language, format, year (dc:issued),
|
||||
annotation (contains «Перевод: …» when present), acquisition links /b/<id>/<fmt>
|
||||
(fb2, epub, mobi, html, txt; pdf/doc via «/b/<id>/download» on the /a/ or /b/ HTML pages).
|
||||
- Author HTML page: http://flibusta.is/a/<id> (title + full book list w/ format links).
|
||||
- Book HTML page: http://flibusta.is/b/<id>.
|
||||
- Polite: 2.5s between requests.
|
||||
"""
|
||||
import re, sys, time, urllib.parse, urllib.request
|
||||
|
||||
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"
|
||||
BASE = "http://flibusta.is"
|
||||
|
||||
def fetch(url, tries=3):
|
||||
last = None
|
||||
for i in range(tries):
|
||||
try:
|
||||
req = urllib.request.Request(url, headers={"User-Agent": UA})
|
||||
return urllib.request.urlopen(req, timeout=60).read().decode("utf-8", "replace")
|
||||
except Exception as e:
|
||||
last = e
|
||||
time.sleep(3 + 3 * i)
|
||||
raise SystemExit(f"fetch failed: {last}")
|
||||
|
||||
def parse_entries(xml):
|
||||
out = []
|
||||
for e in re.findall(r"<entry>.*?</entry>", xml, re.S):
|
||||
def g(pat):
|
||||
m = re.search(pat, e, re.S)
|
||||
return m.group(1).strip() if m else ""
|
||||
title = g(r"<title>(.*?)</title>")
|
||||
aname = g(r"<author>\s*<name>(.*?)</name>")
|
||||
aid = g(r"<uri>(/a/\d+)</uri>")
|
||||
year = g(r"<dc:issued>(.*?)</dc:issued>")
|
||||
fmt = g(r"<dc:format>(.*?)</dc:format>")
|
||||
lang = g(r"<dc:language>(.*?)</dc:language>")
|
||||
bid = g(r'<link href="/b/(\d+)/[a-z0-9]+"')
|
||||
ann = g(r"<content type=\"text/html\">(.*?)</content>")
|
||||
trans = re.search(r"Перевод:?\s*</?[^>]*>?\s*([^&<]+)", ann)
|
||||
if not trans:
|
||||
trans = re.search(r"Перевод:(?:</br>|<br[^>]*>)?([^&]+?)<br", ann)
|
||||
trans = trans.group(1).strip() if trans else ""
|
||||
out.append(dict(id=bid, title=title, author=aname, aid=aid, year=year,
|
||||
fmt=fmt, lang=lang, trans=trans))
|
||||
return out
|
||||
|
||||
def main():
|
||||
if len(sys.argv) < 3:
|
||||
sys.exit(__doc__)
|
||||
cmd, q = sys.argv[1], sys.argv[2]
|
||||
if cmd == "books":
|
||||
url = f"{BASE}/opds/search?searchType=books&searchTerm=" + urllib.parse.quote(q)
|
||||
elif cmd == "authors":
|
||||
url = f"{BASE}/opds/search?searchType=authors&searchTerm=" + urllib.parse.quote(q)
|
||||
elif cmd == "author":
|
||||
url = f"{BASE}/opds/author/{q}/alphabet"
|
||||
else:
|
||||
sys.exit("cmd: books|authors|author")
|
||||
xml = fetch(url)
|
||||
if cmd == "authors":
|
||||
rows = []
|
||||
for e in re.findall(r"<entry>.*?</entry>", xml, re.S):
|
||||
aid = re.search(r"tag:author:(\d+)", e)
|
||||
t = re.search(r"<title>(.*?)</title>", e, re.S)
|
||||
n = re.search(r"<content type=\"text\">(.*?)</content>", e, re.S)
|
||||
rows.append((aid.group(1) if aid else "?",
|
||||
t.group(1).strip() if t else "?",
|
||||
n.group(1).strip() if n else "?"))
|
||||
for aid, name, n in rows:
|
||||
print(f"a/{aid} {name} — {n}")
|
||||
if not rows:
|
||||
print("(no results)")
|
||||
return
|
||||
rows = parse_entries(xml)
|
||||
if not rows:
|
||||
print("(no results)")
|
||||
return
|
||||
for r in rows:
|
||||
t = f" [пер. {r['trans']}]" if r["trans"] else ""
|
||||
print(f"b/{r['id']} {r['year'] or '?'} {r['fmt'] or '?':10s} {r['author'] or '?'} — {r['title'][:70]}{t}")
|
||||
print(f"--- {len(rows)} entries")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
78
tools/ia.py
Executable file
78
tools/ia.py
Executable file
|
|
@ -0,0 +1,78 @@
|
|||
#!/usr/bin/env python3
|
||||
"""archive.org — EN full-text search + metadata + file lists ("English libgen").
|
||||
|
||||
Usage:
|
||||
tools/ia.py search "creator:(Carl Gustav Jung)" [rows] # advancedsearch, mediatype filtered
|
||||
tools/ia.py meta <identifier> # /metadata/<id> — files, size, format
|
||||
tools/ia.py title "Aion" # convenience: title:("...") AND mediatype:texts
|
||||
|
||||
Output (search): numFound + rows: identifier | title | year | downloads | restricted | mediatype.
|
||||
Output (meta): item fields + list of files (name, format, size) — pick _pdf/_epub/_fb2/_djvu.
|
||||
|
||||
Notes:
|
||||
- query syntax: Solr — creator:(...), title:(...), subject:(...), AND/OR/NOT; + = space.
|
||||
- access-restricted-item=true → controlled digital lending (borrow, no direct download).
|
||||
- creator: is loose (images/misattributions included) — add AND mediatype:texts in query
|
||||
(the tool appends it for search/title unless the query already has 'mediatype').
|
||||
- Politeness: 2s between calls. No key needed.
|
||||
"""
|
||||
import json, sys, time, urllib.parse, urllib.request
|
||||
|
||||
UA = "Mozilla/5.0 (research; contact: dmitry@kokorin.org)"
|
||||
|
||||
def fetch(url, tries=3):
|
||||
last = None
|
||||
for i in range(tries):
|
||||
try:
|
||||
req = urllib.request.Request(url, headers={"User-Agent": UA})
|
||||
return urllib.request.urlopen(req, timeout=60).read().decode("utf-8", "replace")
|
||||
except Exception as e:
|
||||
last = e
|
||||
time.sleep(3 + 3 * i)
|
||||
raise SystemExit(f"fetch failed: {last}")
|
||||
|
||||
FL = "identifier,title,year,downloads,access-restricted-item,mediatype"
|
||||
|
||||
def main():
|
||||
if len(sys.argv) < 3:
|
||||
sys.exit(__doc__)
|
||||
cmd, q = sys.argv[1], sys.argv[2]
|
||||
if cmd == "search":
|
||||
rows = int(sys.argv[3]) if len(sys.argv) > 3 else 20
|
||||
if "mediatype" not in q:
|
||||
q = q + " AND mediatype:texts"
|
||||
url = "https://archive.org/advancedsearch.php?" + urllib.parse.urlencode(
|
||||
{"q": q, "output": "json", "rows": rows, "fl[]": FL})
|
||||
d = json.loads(fetch(url))
|
||||
print("numFound:", d.get("response", {}).get("numFound"))
|
||||
for x in d.get("response", {}).get("docs", []):
|
||||
r = x.get("access-restricted-item", "")
|
||||
print(f" {x.get('identifier')[:60]:60s} {x.get('year') or '?':>5} "
|
||||
f"dl={x.get('downloads', 0):>7} {'RESTRICTED' if r == 'true' else 'open':10s} {x.get('title', '')[:60]}")
|
||||
elif cmd == "title":
|
||||
q = f'title:("{q}")'
|
||||
rows = 20
|
||||
if "mediatype" not in q:
|
||||
q = q + " AND mediatype:texts"
|
||||
url = "https://archive.org/advancedsearch.php?" + urllib.parse.urlencode(
|
||||
{"q": q, "output": "json", "rows": rows, "fl[]": FL})
|
||||
d = json.loads(fetch(url))
|
||||
print("numFound:", d.get("response", {}).get("numFound"))
|
||||
for x in d.get("response", {}).get("docs", []):
|
||||
r = x.get("access-restricted-item", "")
|
||||
print(f" {x.get('identifier')[:60]:60s} {x.get('year') or '?':>5} "
|
||||
f"dl={x.get('downloads', 0):>7} {'RESTRICTED' if r == 'true' else 'open':10s} {x.get('title', '')[:60]}")
|
||||
elif cmd == "meta":
|
||||
d = json.loads(fetch(f"https://archive.org/metadata/{q}"))
|
||||
meta = d.get("metadata", {})
|
||||
print("title:", meta.get("title"), "| creator:", meta.get("creator"), "| year:", meta.get("year"))
|
||||
print("restricted:", d.get("is_access_restricted"))
|
||||
for f in d.get("files", []):
|
||||
n, fmt, size = f.get("name"), f.get("format"), f.get("size", "?")
|
||||
if fmt in ("Text PDF", "EPUB", "FB3", "DJVU", "Full Text OCR", "Mobi", "DAISY") or "pdf" in str(fmt):
|
||||
print(f" {n[:70]:70s} {fmt:15s} {size}")
|
||||
else:
|
||||
sys.exit("cmd: search|title|meta")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Loading…
Add table
Add a link
Reference in a new issue