#!/usr/bin/env python3 """Batch-validate ISBNs via isbnsearch.org (user-provided source, 2026-07-15). Response shapes (calibrated): 200 + "ISBN-13:"/"ISBN-10:" fields -> book found; parse title/author/publisher/year 200 + "Please Verify to Continue" -> rate-limit wall; cooldown 150s, retry 404 "Page Not Found" -> ISBN not in their DB Usage: tools/isbncheck.py items.tsv rows: item\tlang\tisbn (whitespace-separated, # comments) Resume: rows already OK in out.tsv are skipped. Politeness: random pause 8-18s between requests (curl engine). """ import random, re, subprocess, sys, time UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36" def fetch(isbn): url = f"https://isbnsearch.org/isbn/{isbn}" p = subprocess.run(["curl", "-s", "-A", UA, "--max-time", "60", "-w", "\nHTTP_CODE:%{http_code}", url], capture_output=True, timeout=90) body, _, code = p.stdout.decode("utf-8", "replace").rpartition("HTTP_CODE:") return code.strip(), body def parse(body): def field(label): m = re.search(r"]*>\s*" + re.escape(label) + r"\s*\s*]*>(.*?)", body, re.S) if not m: return "" return re.sub(r"<[^>]+>", " ", m.group(1)).strip() tm = re.search(r"ISBN\s+\S+\s*-\s*(.*?)", body, re.S) title = re.sub(r"<[^>]+>", " ", tm.group(1)).strip() if tm else "" pm = re.search(r"Published:?\s*\s*]*>\s*([\d]{4})", body) year = pm.group(1) if pm else "" return title, field("Author:") or field("Authors:"), field("Publisher:"), year def main(): items_file, out_file = sys.argv[1], sys.argv[2] rows = [] for line in open(items_file): line = line.strip() if not line or line.startswith("#"): continue rows.append(line.split("\t")) # resume: load existing OK rows (stale non-OK rows from wall-hits are dropped) done = set() if __import__("os").path.exists(out_file): for line in open(out_file): parts = line.rstrip("\n").split("\t") if len(parts) >= 4 and parts[3] == "OK": done.add((parts[0], parts[1], parts[2])) out = open(out_file, "w") out.write("item\tlang\tisbn\tstatus\ttitle\tauthor\tpublisher\tyear\n") if __import__("os").path.exists(out_file): for line in open(out_file): if line.startswith("item\t"): continue parts = line.rstrip("\n").split("\t") if len(parts) >= 4 and parts[3] == "OK": out.write(line) total = len(rows) n = 0 for i, (item, lang, isbn) in enumerate(rows): if (item, lang, isbn) in done: print(f"[{i+1}/{total}] {item}/{lang} {isbn} skip (OK)", flush=True) continue n += 1 status, fields = None, None for attempt in range(4): try: code, body = fetch(isbn) except Exception as e: print(f" {isbn}: fetch error ({e}), backoff", flush=True) time.sleep(15) continue if code == "200" and ("ISBN-13:" in body or "ISBN-10:" in body): status = "OK" fields = parse(body) break if code == "200" and "Please Verify to Continue" in body: print(f" {isbn}: VERIFY wall, cooldown 150s (attempt {attempt+1})", flush=True) time.sleep(150) continue status = code # 404 etc. break if status == "OK": title, author, pub, year = fields out.write(f"{item}\t{lang}\t{isbn}\tOK\t{title}\t{author}\t{pub}\t{year}\n") print(f"[{i+1}/{total}] {item}/{lang} {isbn} OK: {title[:60]}", flush=True) else: out.write(f"{item}\t{lang}\t{isbn}\t{status}\t-\t-\t-\t-\n") print(f"[{i+1}/{total}] {item}/{lang} {isbn} -> HTTP {status} (not found)", flush=True) out.flush() if i < total - 1: time.sleep(random.uniform(8, 18)) out.close() print(f"done -> {out_file}") if __name__ == "__main__": main()