#!/usr/bin/env python3
"""Round 2: full-row dumps for high-hit queries; grep targets."""
import os, re, sys, time, urllib.parse
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from lg import fetch
def fetch_url(url):
return fetch(url, tries=2)
def parse_rows(html):
rows = re.findall(r"
]*>\s*| ]*>.*? |
", html, re.S)
out = []
for r in rows:
tds = re.findall(r"]*>(.*?) | ", r, re.S)
tds = [re.sub(r"<[^>]+>", "", t).replace("&", "&").strip() for t in tds]
tds = [re.sub(r"\s+", " ", t) for t in tds]
if len(tds) >= 6:
md5s = re.findall(r"ads\.php\?md5=([0-9a-f]{32})", r)
ids = re.findall(r"edition\.php\?id=(\d+)", r)
out.append(" / ".join(tds[:9]) + (f" [md5={md5s[0]}]" if md5s else "") + (f" [ed={ids[0]}]" if ids else ""))
return out
# (sec, item, query, grep_pattern, label)
JOBS = [
("01","42","Freud unconscious", r"the unconscious|unconscious.*191[57]|191[57].*unconscious", "SE14"),
("01","52","Stevens on jung", r"jung", "On Jung Princeton"),
("02","20","way of dream", r"way of the dream|windrose|von franz|boa", "Windrose 1988"),
("03","47","Post heart hunter", r"heart of the hunter|van der post|van der post", "1961"),
("03","50","historical atlas mythology", r"atlas|campbell", "Campbell 2 vols"),
("03","53","folklore mythology legend", r"leach|fried|funk", "Leach/Fried"),
("03","58","ariadne clue", r"stevens|ariadne", "Princeton 1998"),
("03","59","motif index folklore", r"thompson|motif.index", "6 vols"),
("03","61","american indian mythology", r"burland|north american", "Barnes Noble 1996"),
("04","04","art of jung", r"art of (c\.? ?g\.? )?jung|hoerni|norton", "Norton 2018"),
("04","21","personality tree drawings", r"bolander|tree drawing", "Basic 1977"),
("04","34","depth psychology of art", r"mcniff|thomas|art", "CCT 1999"),
("04","35","integrating arts therapy", r"mcniff|integrating", "CCT 2004"),
("05","04","buhrmann", r"buhrmann|bürrmann|chiron", "Chiron 1993"),
("05","05","burleson", r"burleson", "1981"),
("05","14","thresholds of initiation", r"henderson|thresholds", "Henderson"),
("05","15","kirsch", r"kirsch.*jung|jung.*kirsch|suny", "SUNY 1997"),
("05","16","cultural complex", r"singer|cultural complex", "Singer"),
("06","04","idea of the numinous", r"numinous|otto", "Chiron 2006"),
("06","07","passion of perpetua", r"fran|perpetua", "von Franz 1980"),
("06","10","jung white letters", r"lammers|white letters|jung.white", "Routledge 2005"),
("06","21","satan old testament", r"satan|kluger|rivka", "1967"),
("06","22","religion greeks romans", r"cumont|greeks and romans", "Cumont"),
]
out = open("data/en-sweep-0106-r2.txt", "w", encoding="utf-8")
for sec, item, q, pat, label in JOBS:
url = ("https://libgen.vg/index.php?req=" + urllib.parse.quote(q) +
"&res=100&columns%5B%5D=t&columns%5B%5D=a&columns%5B%5D=s&columns%5B%5D=y&columns%5B%5D=p&columns%5B%5D=i"
"&objects%5B%5D=f&objects%5B%5D=e&objects%5B%5D=s&objects%5B%5D=a&objects%5B%5D=p&objects%5B%5D=w"
"&topics%5B%5D=l&topics%5B%5D=c&topics%5B%5D=f&topics%5B%5D=a&topics%5B%5D=m&topics%5B%5D=r&topics%5B%5D=s")
try:
rows = parse_rows(fetch_url(url))
except Exception as e:
rows = [f"ERR {e}"]
out.write(f"\n{'='*20} {sec}#{item} '{q}' ({label}) — {len(rows)} rows\n")
for r in rows:
out.write(r + "\n")
out.flush()
print(f"{sec}#{item}: {len(rows)} rows", flush=True)
time.sleep(4)
out.close()
print("R2 DONE")