Init: research structure, tools (lg.py, rsl.py, sx.sh), AGENTS.md, CONTEXT.md, section 04 skeleton (39 book notes)
This commit is contained in:
commit
9b8cf58a4d
55 changed files with 1421 additions and 0 deletions
54
tools/lg.py
Normal file
54
tools/lg.py
Normal file
|
|
@ -0,0 +1,54 @@
|
|||
#!/usr/bin/env python3
|
||||
"""Search libgen.vg (Library Genesis). Usage: lg.sh "query" [limit]
|
||||
Polite mode: one request per query."""
|
||||
import sys, time, urllib.parse, urllib.request, re
|
||||
|
||||
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"
|
||||
|
||||
def fetch(url, tries=3):
|
||||
last = None
|
||||
for i in range(tries):
|
||||
try:
|
||||
req = urllib.request.Request(url, headers={"User-Agent": UA, "Accept-Language": "ru-RU,ru;q=0.9"})
|
||||
return urllib.request.urlopen(req, timeout=90).read().decode("utf-8", "replace")
|
||||
except Exception as e:
|
||||
last = e
|
||||
time.sleep(5 + 5 * i)
|
||||
raise SystemExit(f"fetch failed after {tries} tries: {last}")
|
||||
|
||||
def main():
|
||||
q = sys.argv[1]
|
||||
limit = int(sys.argv[2]) if len(sys.argv) > 2 else 10
|
||||
url = "https://libgen.vg/index.php?" + urllib.parse.urlencode({
|
||||
"req": q, "res": "100", "dlt": "0", "ln": "0",
|
||||
"columns[]": ["t","a","s","y","p","i","l","x","sz"]
|
||||
})
|
||||
html = fetch(url)
|
||||
# pagination hint
|
||||
m = re.search(r'page=(\d+)', html)
|
||||
if m and int(m.group(1)) > 1:
|
||||
print(f"[more pages available, max page {m.group(1)} — rerun with &page=N]")
|
||||
# parse result rows
|
||||
rows = re.findall(r"<tr[^>]*>\s*<td[^>]*>.*?</tr>", html, re.S)
|
||||
count = 0
|
||||
for r in rows:
|
||||
tds = re.findall(r"<td[^>]*>(.*?)</td>", r, re.S)
|
||||
tds = [re.sub(r"<[^>]+>", "", t).replace("&","&").strip() for t in tds]
|
||||
tds = [re.sub(r"\s+", " ", t) for t in tds]
|
||||
if len(tds) < 6:
|
||||
continue
|
||||
# first cell is usually a link with id, title is the big cell
|
||||
title = max(tds, key=len) if tds else ""
|
||||
line = " | ".join(tds[:9])
|
||||
print(line[:300])
|
||||
count += 1
|
||||
if count >= limit:
|
||||
break
|
||||
if count == 0:
|
||||
# maybe "No results" or error page
|
||||
m = re.search(r"No results|Ничего не найдено|Error[^<]*", html)
|
||||
print("(no results)" if m else "(unparsed — check manually)")
|
||||
print(html[:500].replace("\n", " "))
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
49
tools/rsl.py
Normal file
49
tools/rsl.py
Normal file
|
|
@ -0,0 +1,49 @@
|
|||
#!/usr/bin/env python3
|
||||
"""Search РГБ (Russian State Library) via libgen.vg biblioservice. JSON, paginated.
|
||||
Usage: rsl.py "query" [maxpages]
|
||||
Polite: 3s between page requests."""
|
||||
import sys, time, json, urllib.parse, urllib.request
|
||||
|
||||
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"
|
||||
|
||||
def fetch(url, tries=3):
|
||||
last = None
|
||||
for i in range(tries):
|
||||
try:
|
||||
req = urllib.request.Request(url, headers={"User-Agent": UA})
|
||||
return urllib.request.urlopen(req, timeout=90).read().decode("utf-8", "replace")
|
||||
except Exception as e:
|
||||
last = e
|
||||
time.sleep(5 + 5 * i)
|
||||
raise SystemExit(f"fetch failed: {last}")
|
||||
|
||||
def main():
|
||||
q = sys.argv[1]
|
||||
maxpages = int(sys.argv[2]) if len(sys.argv) > 2 else 4
|
||||
for page in range(1, maxpages + 1):
|
||||
url = "https://libgen.vg/biblioservice.php?" + urllib.parse.urlencode(
|
||||
{"value": q, "type": "rsl", "format": "json", "page": page})
|
||||
d = json.loads(fetch(url))
|
||||
recs = {k: v for k, v in d.items() if k != "error"}
|
||||
if not recs:
|
||||
if page == 1:
|
||||
print("(no records)")
|
||||
break
|
||||
for k in sorted(recs, key=int):
|
||||
r = recs[k]
|
||||
title = " ".join(r.get("title", [])) + (
|
||||
(" : " + r["title_add"]) if r.get("title_add") else "")
|
||||
author = "; ".join(r.get("author", []))
|
||||
pub = " ".join(r.get("publisher", []))
|
||||
city = " ".join(r.get("city", []))
|
||||
year = " ".join(r.get("year", []))
|
||||
pages = " ".join(r.get("pages", []))
|
||||
isbn = ", ".join(r.get("505_isbn", []))
|
||||
series = " ".join(r.get("series_name", []))
|
||||
print(f"[{r['id']}] {title}")
|
||||
print(f" {author}")
|
||||
print(f" {city} : {pub}, {year} | {pages} | {isbn}" + (f" | серия: {series}" if series else ""))
|
||||
time.sleep(3)
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
15
tools/sx.sh
Executable file
15
tools/sx.sh
Executable file
|
|
@ -0,0 +1,15 @@
|
|||
#!/bin/bash
|
||||
# SearXNG search helper: usage: ./sx.sh "query" [limit]
|
||||
q="$1"; lim="${2:-8}"
|
||||
curl -s --max-time 30 "http://localhost:8888/search" \
|
||||
--data-urlencode "q=$q" --data-urlencode "format=json" --data-urlencode "language=ru" \
|
||||
| python3 -c "
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
for i,r in enumerate(d.get('results',[])[:$lim],1):
|
||||
print(f'{i}. {r[\"title\"]}')
|
||||
print(f' {r[\"url\"]}')
|
||||
c=r.get('content','').replace(chr(10),' ')
|
||||
if c: print(f' {c[:220]}')
|
||||
print()
|
||||
"
|
||||
Loading…
Add table
Add a link
Reference in a new issue