#!/usr/bin/env python3 """ISBN validation pipeline (2026-07-15, replaces isbnsearch.org which rate-walls). Sources per ISBN: 1. STRUCT libgen.vg biblioservice type=isbn -> checksum error, publisher group name (free, no limits; ISBN-13 or ISBN-10 both accepted) 2. OL openlibrary.org search.json?isbn= -> title/author/year/publisher (best for EN) 3. GB books.google.com/books?vid=ISBN... -> tag "T - A - Google Книги" (200 = indexed w/ title+author, 404 = not in Google's index) 4. NLR tools/nlr.py free-text (RU books only, if 1-3 inconclusive) -> presence check Usage: tools/isbnval.py <items.tsv> <out.tsv> items.tsv: item\tlang\tisbn Resume: rows with non-BAD struct+verdict already in out.tsv are skipped. Politeness: random pauses 2-5s (GB/OL), ~3s (libgen). """ import json, random, re, subprocess, sys, time UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36" def curl_json(url, timeout=60): p = subprocess.run(["curl", "-s", "-A", UA, "--max-time", str(timeout), url], capture_output=True, timeout=timeout + 30) return json.loads(p.stdout.decode("utf-8", "replace")) def isbn13(isbn): d = re.sub(r"\D", "", isbn) if len(d) == 13: return d if len(d) == 10: d = "978" + d[:9] s = sum(int(c) * (1 if i % 2 == 0 else 3) for i, c in enumerate(d)) d += str((10 - s % 10) % 10) return d return None def struct_check(isbn): d = isbn13(isbn) or re.sub(r"\D", "", isbn) try: j = curl_json(f"https://libgen.vg/biblioservice.php?value={d}&type=isbn&format=json") r = j[0] if isinstance(j, list) else j if not r or r.get("error"): return ("BAD", r.get("error", "no response") if r else "no response", "") pubs = ";".join(p.get("name", "") for p in (r.get("isbn_pubgr_info") or [])) return ("OK", r.get("isbn_with_dashes", ""), pubs) except Exception as e: return ("ERR", str(e)[:60], "") def ol_check(isbn): d = re.sub(r"\D", "", isbn) try: j = curl_json(f"https://openlibrary.org/search.json?isbn={d}&limit=3") docs = j.get("docs") or [] if not docs: return "" x = docs[0] t = (x.get("title") or "").strip() a = ";".join((x.get("author_name") or [])[:2]) y = (x.get("publish_year") or [""])[0] return f"{t} | {a} | {y}" except Exception: return "" def gb_check(isbn): d = isbn13(isbn) if not d: return "" try: p = subprocess.run(["curl", "-s", "-L", "-A", UA, "--max-time", "60", "-w", "\nHTTP:%{http_code}", f"https://books.google.com/books?vid=ISBN{d}"], capture_output=True, timeout=90) out = p.stdout.decode("utf-8", "replace") body, _, code = out.rpartition("HTTP:") code = code.strip() if code != "200": return "" m = re.search(r"<title>(.*?)", body, re.S) if not m: return "" t = re.sub(r"<[^>]+>", " ", m.group(1)) t = re.sub(r"\s+", " ", t).strip() t = re.sub(r"\s*-\s*Google (Книги|Books)\s*$", "", t) return t[:160] except Exception: return "" def nlr_check(isbn): d = re.sub(r"\D", "", isbn) try: p = subprocess.run(["python3", "tools/nlr.py", d], capture_output=True, timeout=150) t = p.stdout.decode("utf-8", "replace") m = re.search(r"(\d+) rows", t) return m.group(1) if m else "?" except Exception: return "?" def main(): items_file, out_file = sys.argv[1], sys.argv[2] rows = [] for line in open(items_file): line = line.strip() if not line or line.startswith("#"): continue rows.append(line.split("\t")) done = set() import os if os.path.exists(out_file): for line in open(out_file): parts = line.rstrip("\n").split("\t") if len(parts) >= 4 and parts[3] == "OK": done.add((parts[0], parts[1], parts[2])) out = open(out_file, "w") out.write("item\tlang\tisbn\tstruct\tstruct_detail\tol\tgbooks\tnlr\tverdict\n") if os.path.exists(out_file): for line in open(out_file): parts = line.rstrip("\n").split("\t") if line.startswith("item\t") or (len(parts) >= 4 and parts[3] == "OK"): out.write(line) total = len(rows) for i, (item, lang, isbn) in enumerate(rows): if (item, lang, isbn) in done: print(f"[{i+1}/{total}] {item}/{lang} {isbn} skip", flush=True) continue s_stat, s_det, s_pub = struct_check(isbn) time.sleep(random.uniform(1.5, 3.0)) ol = ol_check(isbn) time.sleep(random.uniform(1.0, 2.5)) gb = gb_check(isbn) time.sleep(random.uniform(2.0, 5.0)) nlr = "" if lang == "RU" and not gb: nlr = nlr_check(isbn) time.sleep(random.uniform(2.0, 4.0)) found = bool(ol) or bool(gb) if s_stat == "BAD": verdict = "BAD-STRUCT" elif found: verdict = "VALID" else: verdict = "STRUCT-ONLY" out.write(f"{item}\t{lang}\t{isbn}\t{s_stat}\t{s_det}|{s_pub}\t{ol}\t{gb}\t{nlr}\t{verdict}\n") out.flush() print(f"[{i+1}/{total}] {item}/{lang} {isbn} {verdict} | gb={gb[:50]!r} | ol={ol[:50]!r} | pub={s_pub[:40]}", flush=True) out.close() print(f"done -> {out_file}") if __name__ == "__main__": main()