#!/usr/bin/env python3 """Round 2: full-row dumps for high-hit queries; grep targets.""" import os, re, sys, time, urllib.parse sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) from lg import fetch def fetch_url(url): return fetch(url, tries=2) def parse_rows(html): rows = re.findall(r"]*>\s*]*>.*?", html, re.S) out = [] for r in rows: tds = re.findall(r"]*>(.*?)", r, re.S) tds = [re.sub(r"<[^>]+>", "", t).replace("&", "&").strip() for t in tds] tds = [re.sub(r"\s+", " ", t) for t in tds] if len(tds) >= 6: md5s = re.findall(r"ads\.php\?md5=([0-9a-f]{32})", r) ids = re.findall(r"edition\.php\?id=(\d+)", r) out.append(" / ".join(tds[:9]) + (f" [md5={md5s[0]}]" if md5s else "") + (f" [ed={ids[0]}]" if ids else "")) return out # (sec, item, query, grep_pattern, label) JOBS = [ ("01","42","Freud unconscious", r"the unconscious|unconscious.*191[57]|191[57].*unconscious", "SE14"), ("01","52","Stevens on jung", r"jung", "On Jung Princeton"), ("02","20","way of dream", r"way of the dream|windrose|von franz|boa", "Windrose 1988"), ("03","47","Post heart hunter", r"heart of the hunter|van der post|van der post", "1961"), ("03","50","historical atlas mythology", r"atlas|campbell", "Campbell 2 vols"), ("03","53","folklore mythology legend", r"leach|fried|funk", "Leach/Fried"), ("03","58","ariadne clue", r"stevens|ariadne", "Princeton 1998"), ("03","59","motif index folklore", r"thompson|motif.index", "6 vols"), ("03","61","american indian mythology", r"burland|north american", "Barnes Noble 1996"), ("04","04","art of jung", r"art of (c\.? ?g\.? )?jung|hoerni|norton", "Norton 2018"), ("04","21","personality tree drawings", r"bolander|tree drawing", "Basic 1977"), ("04","34","depth psychology of art", r"mcniff|thomas|art", "CCT 1999"), ("04","35","integrating arts therapy", r"mcniff|integrating", "CCT 2004"), ("05","04","buhrmann", r"buhrmann|bürrmann|chiron", "Chiron 1993"), ("05","05","burleson", r"burleson", "1981"), ("05","14","thresholds of initiation", r"henderson|thresholds", "Henderson"), ("05","15","kirsch", r"kirsch.*jung|jung.*kirsch|suny", "SUNY 1997"), ("05","16","cultural complex", r"singer|cultural complex", "Singer"), ("06","04","idea of the numinous", r"numinous|otto", "Chiron 2006"), ("06","07","passion of perpetua", r"fran|perpetua", "von Franz 1980"), ("06","10","jung white letters", r"lammers|white letters|jung.white", "Routledge 2005"), ("06","21","satan old testament", r"satan|kluger|rivka", "1967"), ("06","22","religion greeks romans", r"cumont|greeks and romans", "Cumont"), ] out = open("data/en-sweep-0106-r2.txt", "w", encoding="utf-8") for sec, item, q, pat, label in JOBS: url = ("https://libgen.vg/index.php?req=" + urllib.parse.quote(q) + "&res=100&columns%5B%5D=t&columns%5B%5D=a&columns%5B%5D=s&columns%5B%5D=y&columns%5B%5D=p&columns%5B%5D=i" "&objects%5B%5D=f&objects%5B%5D=e&objects%5B%5D=s&objects%5B%5D=a&objects%5B%5D=p&objects%5B%5D=w" "&topics%5B%5D=l&topics%5B%5D=c&topics%5B%5D=f&topics%5B%5D=a&topics%5B%5D=m&topics%5B%5D=r&topics%5B%5D=s") try: rows = parse_rows(fetch_url(url)) except Exception as e: rows = [f"ERR {e}"] out.write(f"\n{'='*20} {sec}#{item} '{q}' ({label}) — {len(rows)} rows\n") for r in rows: out.write(r + "\n") out.flush() print(f"{sec}#{item}: {len(rows)} rows", flush=True) time.sleep(4) out.close() print("R2 DONE")