#!/usr/bin/env python3 """MGU Scientific Library (nbmgu.ru) catalog search — server-rendered, curl-able. Discovered 2026-07-15 from the advanced-search page JS (functions.js CreateClientRequest): the "JS" search is a plain GET: https://nbmgu.ru/search/?adv=1&q1=&f1=&v1=0&cat=BOOK[&p=] Fields (f1..f3, AND-combined): Fields (f1..f3, AND-combined between rows): ANY все · AUT автор · COA колл. автор · TIT заглавие · KEY ключ. слова RUB рубрики · YEA год · PLA место · PUB изд-во · SER серия ISB ISBN · ISS ISSN · NBM индексы НБ МГУ Method (v1..v3, per row): 0=Слова (stemming, default) · 1=Словосочетание (phrase) · 2=Начинается с · 3=Дословно (exact). Results: "Всего: N", 20 rows/page (p=0-based). Row = GOST description text (title / authors; notes - [edition]) + "City : Publisher, Year" + shelf code. Detail page /order/storing.aspx?uid=... is holdings/order only (no full record). **ISB field NOT populated** (0 hits for 3 formats x 2 control ISBNs) — search by AUT/TIT/SER/PUB instead; ISBN validation stays with РГБ/Google/OL (tools/isbnval.py). Usage: tools/mgu.py "query" [FIELD=ANY] [pages=1] [method=0] tools/mgu.py --multi "AUT:Франц" "TIT:Число" ["TIT:точное:3"] (AND rows) """ import re, subprocess, sys, time UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36" BASE = "https://nbmgu.ru/search/" def search_url(pairs, p=0, cat="BOOK"): for attempt in range(3): args = ["curl", "-s", "-G", "-A", UA, "--max-time", "90", "-e", BASE] args += ["--data-urlencode", "adv=1"] # mirror the browser: always send slots 1..N (empties included) for i, pr in enumerate(pairs, 1): f, q = pr[0], pr[1] v = str(pr[2]) if len(pr) > 2 else "0" args += ["--data-urlencode", f"q{i}={q}", "--data-urlencode", f"f{i}={f}", "--data-urlencode", f"v{i}={v}"] if p: args += ["--data-urlencode", f"p={p}"] args += ["--data-urlencode", f"cat={cat}", BASE] try: out = subprocess.run(args, capture_output=True, timeout=120) if out.returncode != 0 or len(out.stdout) < 5000: print(f" (curl rc={out.returncode} size={len(out.stdout)} err={out.stderr[:120]!r})", file=sys.stderr) body = out.stdout.decode("utf-8", "replace") except Exception as e: print(f" (subprocess error: {e})", file=sys.stderr) body = "" if len(body) > 5000: return body time.sleep(4 * (attempt + 1)) return body def parse(html): total = "" m = re.search(r"Всего:\s*(\d+)", html) if m: total = m.group(1) rows = [] # NOTE: anchor title attr contains a raw '>' ("
"), so plain [^>]+ fails; # split on row start and rely on href being the LAST attribute of the anchor. for chunk in html.split('
  • ')[1:]: cm = re.search(r'

    \s*(\d+)\.\s*(.*?)', chunk, re.S) pm = re.search(r"

    \s*

    (.*?)

    ", chunk, re.S) sm = re.search(r"Шифр:\s*([^<]+)", chunk) if not (cm and pm): continue clean = lambda s: re.sub(r"\s+", " ", re.sub(r"<[^>]+>", " ", s)).strip() rows.append({ "n": cm.group(1), "desc": clean(cm.group(3)), "pubyear": clean(pm.group(1)), "shelf": clean(sm.group(1)) if sm else "", "url": "https://nbmgu.ru" + cm.group(2), }) return total, rows def main(): argv = sys.argv[1:] if not argv: sys.exit(__doc__) if argv[0] == "--multi": pairs = [] for spec in argv[1:]: parts = spec.split(":") f = parts[0].upper() q = parts[1] if len(parts) > 1 else "" v = int(parts[2]) if len(parts) > 2 else 0 pairs.append((f, q, v) if v else (f, q)) pmax = 1 else: q = argv[0] f = argv[1].upper() if len(argv) > 1 else "ANY" v = int(argv[3]) if len(argv) > 3 else 0 pairs = [(f, q, v) if v else (f, q)] pmax = int(argv[2]) if len(argv) > 2 else 1 all_rows = [] total = "" seen = set() for p in range(pmax): html = search_url(pairs, p=p) total, rows = parse(html) # out-of-range p falls back to page 0 on the server side — dedupe by url new = [r for r in rows if r["url"] not in seen] if not new: break seen.update(r["url"] for r in new) all_rows += new if total and len(all_rows) >= int(total): break time.sleep(3) print(f"### MGU: {pairs} — total: {total} (fetched {len(all_rows)})") for r in all_rows: print(f"{r['n']:>3}. {r['desc']}") print(f" {r['pubyear']} | шифр: {r['shelf']} | {r['url']}") if __name__ == "__main__": main()