107 lines
4.2 KiB
Python
107 lines
4.2 KiB
Python
#!/usr/bin/env python3
|
|
"""Batch-validate ISBNs via isbnsearch.org (user-provided source, 2026-07-15).
|
|
|
|
Response shapes (calibrated):
|
|
200 + "ISBN-13:"/"ISBN-10:" fields -> book found; parse title/author/publisher/year
|
|
200 + "Please Verify to Continue" -> rate-limit wall; cooldown 150s, retry
|
|
404 "Page Not Found" -> ISBN not in their DB
|
|
|
|
Usage:
|
|
tools/isbncheck.py <items.tsv> <out.tsv>
|
|
items.tsv rows: item\tlang\tisbn (whitespace-separated, # comments)
|
|
Resume: rows already OK in out.tsv are skipped.
|
|
Politeness: random pause 8-18s between requests (curl engine).
|
|
"""
|
|
import random, re, subprocess, sys, time
|
|
|
|
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36"
|
|
|
|
|
|
def fetch(isbn):
|
|
url = f"https://isbnsearch.org/isbn/{isbn}"
|
|
p = subprocess.run(["curl", "-s", "-A", UA, "--max-time", "60",
|
|
"-w", "\nHTTP_CODE:%{http_code}", url],
|
|
capture_output=True, timeout=90)
|
|
body, _, code = p.stdout.decode("utf-8", "replace").rpartition("HTTP_CODE:")
|
|
return code.strip(), body
|
|
|
|
|
|
def parse(body):
|
|
def field(label):
|
|
m = re.search(r"<td[^>]*>\s*" + re.escape(label) + r"\s*</td>\s*<td[^>]*>(.*?)</td>",
|
|
body, re.S)
|
|
if not m:
|
|
return ""
|
|
return re.sub(r"<[^>]+>", " ", m.group(1)).strip()
|
|
tm = re.search(r"<title>ISBN\s+\S+\s*-\s*(.*?)</title>", body, re.S)
|
|
title = re.sub(r"<[^>]+>", " ", tm.group(1)).strip() if tm else ""
|
|
pm = re.search(r"Published:?\s*</td>\s*<td[^>]*>\s*([\d]{4})", body)
|
|
year = pm.group(1) if pm else ""
|
|
return title, field("Author:") or field("Authors:"), field("Publisher:"), year
|
|
|
|
|
|
def main():
|
|
items_file, out_file = sys.argv[1], sys.argv[2]
|
|
rows = []
|
|
for line in open(items_file):
|
|
line = line.strip()
|
|
if not line or line.startswith("#"):
|
|
continue
|
|
rows.append(line.split("\t"))
|
|
# resume: load existing OK rows (stale non-OK rows from wall-hits are dropped)
|
|
done = set()
|
|
if __import__("os").path.exists(out_file):
|
|
for line in open(out_file):
|
|
parts = line.rstrip("\n").split("\t")
|
|
if len(parts) >= 4 and parts[3] == "OK":
|
|
done.add((parts[0], parts[1], parts[2]))
|
|
out = open(out_file, "w")
|
|
out.write("item\tlang\tisbn\tstatus\ttitle\tauthor\tpublisher\tyear\n")
|
|
if __import__("os").path.exists(out_file):
|
|
for line in open(out_file):
|
|
if line.startswith("item\t"):
|
|
continue
|
|
parts = line.rstrip("\n").split("\t")
|
|
if len(parts) >= 4 and parts[3] == "OK":
|
|
out.write(line)
|
|
total = len(rows)
|
|
n = 0
|
|
for i, (item, lang, isbn) in enumerate(rows):
|
|
if (item, lang, isbn) in done:
|
|
print(f"[{i+1}/{total}] {item}/{lang} {isbn} skip (OK)", flush=True)
|
|
continue
|
|
n += 1
|
|
status, fields = None, None
|
|
for attempt in range(4):
|
|
try:
|
|
code, body = fetch(isbn)
|
|
except Exception as e:
|
|
print(f" {isbn}: fetch error ({e}), backoff", flush=True)
|
|
time.sleep(15)
|
|
continue
|
|
if code == "200" and ("ISBN-13:" in body or "ISBN-10:" in body):
|
|
status = "OK"
|
|
fields = parse(body)
|
|
break
|
|
if code == "200" and "Please Verify to Continue" in body:
|
|
print(f" {isbn}: VERIFY wall, cooldown 150s (attempt {attempt+1})", flush=True)
|
|
time.sleep(150)
|
|
continue
|
|
status = code # 404 etc.
|
|
break
|
|
if status == "OK":
|
|
title, author, pub, year = fields
|
|
out.write(f"{item}\t{lang}\t{isbn}\tOK\t{title}\t{author}\t{pub}\t{year}\n")
|
|
print(f"[{i+1}/{total}] {item}/{lang} {isbn} OK: {title[:60]}", flush=True)
|
|
else:
|
|
out.write(f"{item}\t{lang}\t{isbn}\t{status}\t-\t-\t-\t-\n")
|
|
print(f"[{i+1}/{total}] {item}/{lang} {isbn} -> HTTP {status} (not found)", flush=True)
|
|
out.flush()
|
|
if i < total - 1:
|
|
time.sleep(random.uniform(8, 18))
|
|
out.close()
|
|
print(f"done -> {out_file}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|