libgen re-sweep of EN gaps in sections 01-06 (user request; improved sweep: file-level objects[]=f + full 100-row dumps) — 75 candidates, 7 rescues verified: - sec03 #15 Witches Ogres (Shambhala 1992, 236pp, cover OK) - sec03 #19 Hermes and His Children (Daimon 2020 2nd ed., epub ISBN in file) - sec03 #31 European Folktale (Indiana 1986, 204pp, cover OK) - sec03 #32 Fairytale as Art Form (Indiana 1987, 232pp, title OK) - sec03 #33 Once Upon a Time (Ungar 1976 1st ed., 204pp, title OK; RAW 'Indiana 1970' = ISAP typo) - sec03 #36 Irresistible Fairy Tale (Princeton 2012, epub OK) - sec03 #47 Heart of the Hunter (Vintage 2010, mobi) + sec02 #16 A Little Course in Dreams (Shambhala 1986, 137pp, verified) Canonical location fix: Eliade 'Myths, Dreams and Mysteries' files moved sec05 -> sec03 (first appearance); sec03 #25 card ❌ -> ✅ (RU was found in sec05 rescue 2026-09-21, card was stale). Still downloading (queue tools/lgdl-en0106-queue.sh): sec03 #50 (Atlas x4), #55, #57, sec04 #04 (Art of C.G. Jung 2018, 298MB), #17, sec01 #42 (Freud Unconscious Penguin). Rejected: 04#23 Encyclopedia (JSTOR article 467KB), 05#16 Singer 2023 (different book).
67 lines
3.6 KiB
Python
67 lines
3.6 KiB
Python
#!/usr/bin/env python3
|
|
"""Round 2: full-row dumps for high-hit queries; grep targets."""
|
|
import os, re, sys, time, urllib.parse
|
|
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
|
from lg import fetch
|
|
|
|
def fetch_url(url):
|
|
return fetch(url, tries=2)
|
|
|
|
def parse_rows(html):
|
|
rows = re.findall(r"<tr[^>]*>\s*<td[^>]*>.*?</tr>", html, re.S)
|
|
out = []
|
|
for r in rows:
|
|
tds = re.findall(r"<td[^>]*>(.*?)</td>", r, re.S)
|
|
tds = [re.sub(r"<[^>]+>", "", t).replace("&", "&").strip() for t in tds]
|
|
tds = [re.sub(r"\s+", " ", t) for t in tds]
|
|
if len(tds) >= 6:
|
|
md5s = re.findall(r"ads\.php\?md5=([0-9a-f]{32})", r)
|
|
ids = re.findall(r"edition\.php\?id=(\d+)", r)
|
|
out.append(" / ".join(tds[:9]) + (f" [md5={md5s[0]}]" if md5s else "") + (f" [ed={ids[0]}]" if ids else ""))
|
|
return out
|
|
|
|
# (sec, item, query, grep_pattern, label)
|
|
JOBS = [
|
|
("01","42","Freud unconscious", r"the unconscious|unconscious.*191[57]|191[57].*unconscious", "SE14"),
|
|
("01","52","Stevens on jung", r"jung", "On Jung Princeton"),
|
|
("02","20","way of dream", r"way of the dream|windrose|von franz|boa", "Windrose 1988"),
|
|
("03","47","Post heart hunter", r"heart of the hunter|van der post|van der post", "1961"),
|
|
("03","50","historical atlas mythology", r"atlas|campbell", "Campbell 2 vols"),
|
|
("03","53","folklore mythology legend", r"leach|fried|funk", "Leach/Fried"),
|
|
("03","58","ariadne clue", r"stevens|ariadne", "Princeton 1998"),
|
|
("03","59","motif index folklore", r"thompson|motif.index", "6 vols"),
|
|
("03","61","american indian mythology", r"burland|north american", "Barnes Noble 1996"),
|
|
("04","04","art of jung", r"art of (c\.? ?g\.? )?jung|hoerni|norton", "Norton 2018"),
|
|
("04","21","personality tree drawings", r"bolander|tree drawing", "Basic 1977"),
|
|
("04","34","depth psychology of art", r"mcniff|thomas|art", "CCT 1999"),
|
|
("04","35","integrating arts therapy", r"mcniff|integrating", "CCT 2004"),
|
|
("05","04","buhrmann", r"buhrmann|bürrmann|chiron", "Chiron 1993"),
|
|
("05","05","burleson", r"burleson", "1981"),
|
|
("05","14","thresholds of initiation", r"henderson|thresholds", "Henderson"),
|
|
("05","15","kirsch", r"kirsch.*jung|jung.*kirsch|suny", "SUNY 1997"),
|
|
("05","16","cultural complex", r"singer|cultural complex", "Singer"),
|
|
("06","04","idea of the numinous", r"numinous|otto", "Chiron 2006"),
|
|
("06","07","passion of perpetua", r"fran|perpetua", "von Franz 1980"),
|
|
("06","10","jung white letters", r"lammers|white letters|jung.white", "Routledge 2005"),
|
|
("06","21","satan old testament", r"satan|kluger|rivka", "1967"),
|
|
("06","22","religion greeks romans", r"cumont|greeks and romans", "Cumont"),
|
|
]
|
|
|
|
out = open("data/en-sweep-0106-r2.txt", "w", encoding="utf-8")
|
|
for sec, item, q, pat, label in JOBS:
|
|
url = ("https://libgen.vg/index.php?req=" + urllib.parse.quote(q) +
|
|
"&res=100&columns%5B%5D=t&columns%5B%5D=a&columns%5B%5D=s&columns%5B%5D=y&columns%5B%5D=p&columns%5B%5D=i"
|
|
"&objects%5B%5D=f&objects%5B%5D=e&objects%5B%5D=s&objects%5B%5D=a&objects%5B%5D=p&objects%5B%5D=w"
|
|
"&topics%5B%5D=l&topics%5B%5D=c&topics%5B%5D=f&topics%5B%5D=a&topics%5B%5D=m&topics%5B%5D=r&topics%5B%5D=s")
|
|
try:
|
|
rows = parse_rows(fetch_url(url))
|
|
except Exception as e:
|
|
rows = [f"ERR {e}"]
|
|
out.write(f"\n{'='*20} {sec}#{item} '{q}' ({label}) — {len(rows)} rows\n")
|
|
for r in rows:
|
|
out.write(r + "\n")
|
|
out.flush()
|
|
print(f"{sec}#{item}: {len(rows)} rows", flush=True)
|
|
time.sleep(4)
|
|
out.close()
|
|
print("R2 DONE")
|