ISBN validation (58 ISBNs): tools/isbnval.py + data/isbn-validation.tsv; SUMMARY Table 4 + fixed 2 bad-check-digit typos and 2 publisher-prefix mismatches; AGENTS.md source 21
This commit is contained in:
parent
89c4845d0b
commit
4feff7a0f7
6 changed files with 458 additions and 15 deletions
107
tools/isbncheck.py
Normal file
107
tools/isbncheck.py
Normal file
|
|
@ -0,0 +1,107 @@
|
|||
#!/usr/bin/env python3
|
||||
"""Batch-validate ISBNs via isbnsearch.org (user-provided source, 2026-07-15).
|
||||
|
||||
Response shapes (calibrated):
|
||||
200 + "ISBN-13:"/"ISBN-10:" fields -> book found; parse title/author/publisher/year
|
||||
200 + "Please Verify to Continue" -> rate-limit wall; cooldown 150s, retry
|
||||
404 "Page Not Found" -> ISBN not in their DB
|
||||
|
||||
Usage:
|
||||
tools/isbncheck.py <items.tsv> <out.tsv>
|
||||
items.tsv rows: item\tlang\tisbn (whitespace-separated, # comments)
|
||||
Resume: rows already OK in out.tsv are skipped.
|
||||
Politeness: random pause 8-18s between requests (curl engine).
|
||||
"""
|
||||
import random, re, subprocess, sys, time
|
||||
|
||||
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36"
|
||||
|
||||
|
||||
def fetch(isbn):
|
||||
url = f"https://isbnsearch.org/isbn/{isbn}"
|
||||
p = subprocess.run(["curl", "-s", "-A", UA, "--max-time", "60",
|
||||
"-w", "\nHTTP_CODE:%{http_code}", url],
|
||||
capture_output=True, timeout=90)
|
||||
body, _, code = p.stdout.decode("utf-8", "replace").rpartition("HTTP_CODE:")
|
||||
return code.strip(), body
|
||||
|
||||
|
||||
def parse(body):
|
||||
def field(label):
|
||||
m = re.search(r"<td[^>]*>\s*" + re.escape(label) + r"\s*</td>\s*<td[^>]*>(.*?)</td>",
|
||||
body, re.S)
|
||||
if not m:
|
||||
return ""
|
||||
return re.sub(r"<[^>]+>", " ", m.group(1)).strip()
|
||||
tm = re.search(r"<title>ISBN\s+\S+\s*-\s*(.*?)</title>", body, re.S)
|
||||
title = re.sub(r"<[^>]+>", " ", tm.group(1)).strip() if tm else ""
|
||||
pm = re.search(r"Published:?\s*</td>\s*<td[^>]*>\s*([\d]{4})", body)
|
||||
year = pm.group(1) if pm else ""
|
||||
return title, field("Author:") or field("Authors:"), field("Publisher:"), year
|
||||
|
||||
|
||||
def main():
|
||||
items_file, out_file = sys.argv[1], sys.argv[2]
|
||||
rows = []
|
||||
for line in open(items_file):
|
||||
line = line.strip()
|
||||
if not line or line.startswith("#"):
|
||||
continue
|
||||
rows.append(line.split("\t"))
|
||||
# resume: load existing OK rows (stale non-OK rows from wall-hits are dropped)
|
||||
done = set()
|
||||
if __import__("os").path.exists(out_file):
|
||||
for line in open(out_file):
|
||||
parts = line.rstrip("\n").split("\t")
|
||||
if len(parts) >= 4 and parts[3] == "OK":
|
||||
done.add((parts[0], parts[1], parts[2]))
|
||||
out = open(out_file, "w")
|
||||
out.write("item\tlang\tisbn\tstatus\ttitle\tauthor\tpublisher\tyear\n")
|
||||
if __import__("os").path.exists(out_file):
|
||||
for line in open(out_file):
|
||||
if line.startswith("item\t"):
|
||||
continue
|
||||
parts = line.rstrip("\n").split("\t")
|
||||
if len(parts) >= 4 and parts[3] == "OK":
|
||||
out.write(line)
|
||||
total = len(rows)
|
||||
n = 0
|
||||
for i, (item, lang, isbn) in enumerate(rows):
|
||||
if (item, lang, isbn) in done:
|
||||
print(f"[{i+1}/{total}] {item}/{lang} {isbn} skip (OK)", flush=True)
|
||||
continue
|
||||
n += 1
|
||||
status, fields = None, None
|
||||
for attempt in range(4):
|
||||
try:
|
||||
code, body = fetch(isbn)
|
||||
except Exception as e:
|
||||
print(f" {isbn}: fetch error ({e}), backoff", flush=True)
|
||||
time.sleep(15)
|
||||
continue
|
||||
if code == "200" and ("ISBN-13:" in body or "ISBN-10:" in body):
|
||||
status = "OK"
|
||||
fields = parse(body)
|
||||
break
|
||||
if code == "200" and "Please Verify to Continue" in body:
|
||||
print(f" {isbn}: VERIFY wall, cooldown 150s (attempt {attempt+1})", flush=True)
|
||||
time.sleep(150)
|
||||
continue
|
||||
status = code # 404 etc.
|
||||
break
|
||||
if status == "OK":
|
||||
title, author, pub, year = fields
|
||||
out.write(f"{item}\t{lang}\t{isbn}\tOK\t{title}\t{author}\t{pub}\t{year}\n")
|
||||
print(f"[{i+1}/{total}] {item}/{lang} {isbn} OK: {title[:60]}", flush=True)
|
||||
else:
|
||||
out.write(f"{item}\t{lang}\t{isbn}\t{status}\t-\t-\t-\t-\n")
|
||||
print(f"[{i+1}/{total}] {item}/{lang} {isbn} -> HTTP {status} (not found)", flush=True)
|
||||
out.flush()
|
||||
if i < total - 1:
|
||||
time.sleep(random.uniform(8, 18))
|
||||
out.close()
|
||||
print(f"done -> {out_file}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
157
tools/isbnval.py
Normal file
157
tools/isbnval.py
Normal file
|
|
@ -0,0 +1,157 @@
|
|||
#!/usr/bin/env python3
|
||||
"""ISBN validation pipeline (2026-07-15, replaces isbnsearch.org which rate-walls).
|
||||
|
||||
Sources per ISBN:
|
||||
1. STRUCT libgen.vg biblioservice type=isbn -> checksum error, publisher group name
|
||||
(free, no limits; ISBN-13 or ISBN-10 both accepted)
|
||||
2. OL openlibrary.org search.json?isbn= -> title/author/year/publisher (best for EN)
|
||||
3. GB books.google.com/books?vid=ISBN... -> <title> tag "T - A - Google Книги"
|
||||
(200 = indexed w/ title+author, 404 = not in Google's index)
|
||||
4. NLR tools/nlr.py free-text (RU books only, if 1-3 inconclusive) -> presence check
|
||||
|
||||
Usage: tools/isbnval.py <items.tsv> <out.tsv>
|
||||
items.tsv: item\tlang\tisbn
|
||||
Resume: rows with non-BAD struct+verdict already in out.tsv are skipped.
|
||||
Politeness: random pauses 2-5s (GB/OL), ~3s (libgen).
|
||||
"""
|
||||
import json, random, re, subprocess, sys, time
|
||||
|
||||
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36"
|
||||
|
||||
|
||||
def curl_json(url, timeout=60):
|
||||
p = subprocess.run(["curl", "-s", "-A", UA, "--max-time", str(timeout), url],
|
||||
capture_output=True, timeout=timeout + 30)
|
||||
return json.loads(p.stdout.decode("utf-8", "replace"))
|
||||
|
||||
|
||||
def isbn13(isbn):
|
||||
d = re.sub(r"\D", "", isbn)
|
||||
if len(d) == 13:
|
||||
return d
|
||||
if len(d) == 10:
|
||||
d = "978" + d[:9]
|
||||
s = sum(int(c) * (1 if i % 2 == 0 else 3) for i, c in enumerate(d))
|
||||
d += str((10 - s % 10) % 10)
|
||||
return d
|
||||
return None
|
||||
|
||||
|
||||
def struct_check(isbn):
|
||||
d = isbn13(isbn) or re.sub(r"\D", "", isbn)
|
||||
try:
|
||||
j = curl_json(f"https://libgen.vg/biblioservice.php?value={d}&type=isbn&format=json")
|
||||
r = j[0] if isinstance(j, list) else j
|
||||
if not r or r.get("error"):
|
||||
return ("BAD", r.get("error", "no response") if r else "no response", "")
|
||||
pubs = ";".join(p.get("name", "") for p in (r.get("isbn_pubgr_info") or []))
|
||||
return ("OK", r.get("isbn_with_dashes", ""), pubs)
|
||||
except Exception as e:
|
||||
return ("ERR", str(e)[:60], "")
|
||||
|
||||
|
||||
def ol_check(isbn):
|
||||
d = re.sub(r"\D", "", isbn)
|
||||
try:
|
||||
j = curl_json(f"https://openlibrary.org/search.json?isbn={d}&limit=3")
|
||||
docs = j.get("docs") or []
|
||||
if not docs:
|
||||
return ""
|
||||
x = docs[0]
|
||||
t = (x.get("title") or "").strip()
|
||||
a = ";".join((x.get("author_name") or [])[:2])
|
||||
y = (x.get("publish_year") or [""])[0]
|
||||
return f"{t} | {a} | {y}"
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
|
||||
def gb_check(isbn):
|
||||
d = isbn13(isbn)
|
||||
if not d:
|
||||
return ""
|
||||
try:
|
||||
p = subprocess.run(["curl", "-s", "-L", "-A", UA, "--max-time", "60",
|
||||
"-w", "\nHTTP:%{http_code}",
|
||||
f"https://books.google.com/books?vid=ISBN{d}"],
|
||||
capture_output=True, timeout=90)
|
||||
out = p.stdout.decode("utf-8", "replace")
|
||||
body, _, code = out.rpartition("HTTP:")
|
||||
code = code.strip()
|
||||
if code != "200":
|
||||
return ""
|
||||
m = re.search(r"<title>(.*?)</title>", body, re.S)
|
||||
if not m:
|
||||
return ""
|
||||
t = re.sub(r"<[^>]+>", " ", m.group(1))
|
||||
t = re.sub(r"\s+", " ", t).strip()
|
||||
t = re.sub(r"\s*-\s*Google (Книги|Books)\s*$", "", t)
|
||||
return t[:160]
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
|
||||
def nlr_check(isbn):
|
||||
d = re.sub(r"\D", "", isbn)
|
||||
try:
|
||||
p = subprocess.run(["python3", "tools/nlr.py", d], capture_output=True, timeout=150)
|
||||
t = p.stdout.decode("utf-8", "replace")
|
||||
m = re.search(r"(\d+) rows", t)
|
||||
return m.group(1) if m else "?"
|
||||
except Exception:
|
||||
return "?"
|
||||
|
||||
|
||||
def main():
|
||||
items_file, out_file = sys.argv[1], sys.argv[2]
|
||||
rows = []
|
||||
for line in open(items_file):
|
||||
line = line.strip()
|
||||
if not line or line.startswith("#"):
|
||||
continue
|
||||
rows.append(line.split("\t"))
|
||||
done = set()
|
||||
import os
|
||||
if os.path.exists(out_file):
|
||||
for line in open(out_file):
|
||||
parts = line.rstrip("\n").split("\t")
|
||||
if len(parts) >= 4 and parts[3] == "OK":
|
||||
done.add((parts[0], parts[1], parts[2]))
|
||||
out = open(out_file, "w")
|
||||
out.write("item\tlang\tisbn\tstruct\tstruct_detail\tol\tgbooks\tnlr\tverdict\n")
|
||||
if os.path.exists(out_file):
|
||||
for line in open(out_file):
|
||||
parts = line.rstrip("\n").split("\t")
|
||||
if line.startswith("item\t") or (len(parts) >= 4 and parts[3] == "OK"):
|
||||
out.write(line)
|
||||
total = len(rows)
|
||||
for i, (item, lang, isbn) in enumerate(rows):
|
||||
if (item, lang, isbn) in done:
|
||||
print(f"[{i+1}/{total}] {item}/{lang} {isbn} skip", flush=True)
|
||||
continue
|
||||
s_stat, s_det, s_pub = struct_check(isbn)
|
||||
time.sleep(random.uniform(1.5, 3.0))
|
||||
ol = ol_check(isbn)
|
||||
time.sleep(random.uniform(1.0, 2.5))
|
||||
gb = gb_check(isbn)
|
||||
time.sleep(random.uniform(2.0, 5.0))
|
||||
nlr = ""
|
||||
if lang == "RU" and not gb:
|
||||
nlr = nlr_check(isbn)
|
||||
time.sleep(random.uniform(2.0, 4.0))
|
||||
found = bool(ol) or bool(gb)
|
||||
if s_stat == "BAD":
|
||||
verdict = "BAD-STRUCT"
|
||||
elif found:
|
||||
verdict = "VALID"
|
||||
else:
|
||||
verdict = "STRUCT-ONLY"
|
||||
out.write(f"{item}\t{lang}\t{isbn}\t{s_stat}\t{s_det}|{s_pub}\t{ol}\t{gb}\t{nlr}\t{verdict}\n")
|
||||
out.flush()
|
||||
print(f"[{i+1}/{total}] {item}/{lang} {isbn} {verdict} | gb={gb[:50]!r} | ol={ol[:50]!r} | pub={s_pub[:40]}", flush=True)
|
||||
out.close()
|
||||
print(f"done -> {out_file}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Loading…
Add table
Add a link
Reference in a new issue