diff --git a/AGENTS.md b/AGENTS.md index f21efa7..70da68f 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -139,6 +139,10 @@ sections/0N-/ # one .md per BOOK (chapter-level list items are merged int - `books.google.com/books?vid=ISBN<13digits>`: **plain curl works** — 200 + `"T - A - Google Книги"` if indexed, 404 if not. Good ISBN validator incl. RU (АСТ/Эксмо/БукСМарт hit; small-press 404). googleapis.com/books API stays 429 without a key. +- **Keep volume LOW**: Google may block agents on sustained querying (user warning 2026-07-15). + Budget: a few dozen requests per session, 2–5s random pauses, never in tight loops; treat as + an auxiliary cross-check, not a bulk source. If 429/403-redirect walls appear — stop and fall + back to OL/libgen/RSL. - `fetch_content` on the same URL → richer metadata: full author/editor names, edition, series. Resolved Ottmann=Klaus, Goldstein=Ralph, Brutsche=Paul, Elder=George, Moon=Beverly, Killick=Katherine, Pennington/Staples, Rowland=Susan, Bolander=Karen, Acton=Mary via this or OL. @@ -213,6 +217,20 @@ sections/0N-<name>/ # one .md per BOOK (chapter-level list items are merged int - **isbnsearch.org**: nice per-ISBN pages (author/publisher/year) but rate-walls after ~10 requests ("Please Verify to Continue", no recovery in 7 min) — not for batches. +### 22. MGU Scientific Library (nbmgu.ru) — server-rendered, curl-able (2026-07-15) +- Tool: `tools/mgu.py "query" [FIELD] [pages] [method]` or `--multi "AUT:X" "TIT:Y"`. + The "JS" advanced search is a plain GET: `/search/?adv=1&q1=<q>&f1=<F>&v1=<M>&cat=BOOK[&p=N]`. +- Fields: ANY/AUT/COA/TIT/KEY/RUB/YEA/PLA/PUB/SER/ISB/ISS/NBM; rows AND between q1..q3. + Method v: 0=Слова (stemming, default) · 1=Словосочетание · 2=Начинается с · 3=Дословно. +- "Всего: N" + GOST row (title/authors/notes) + "City : Publisher, Year" + shelf code + uid link. + 20 rows/page, p=0-based; **out-of-range p silently falls back to page 0** (dedupe by uid!). +- **ISB field NOT populated** (0 hits, 3 formats × control ISBNs) — no ISBN search; use + AUT/TIT/SER/PUB. `storing.aspx?uid=` page = holdings/order only (no full record). +- Value: second RU catalog (complements RSL/РНБ), strong **SER** sweep (e.g. «Библиотека + аналитической психологии» → full series list in one query), GOST rows w/ translators. +- Anchor titles in result rows contain a raw `>` (title="Хранение<br/>Заказ") — regexes must + not use [^>]+ across the anchor attrs. + ## Not usable / low value - libgen biblioservice worldcat/googlebooks/isbndb/udc (broken on this mirror) - imaton.com (МААП educational publisher, not translations) diff --git a/tools/__pycache__/mgu.cpython-312.pyc b/tools/__pycache__/mgu.cpython-312.pyc new file mode 100644 index 0000000..34acd79 Binary files /dev/null and b/tools/__pycache__/mgu.cpython-312.pyc differ diff --git a/tools/mgu.py b/tools/mgu.py new file mode 100644 index 0000000..a4a9433 --- /dev/null +++ b/tools/mgu.py @@ -0,0 +1,128 @@ +#!/usr/bin/env python3 +"""MGU Scientific Library (nbmgu.ru) catalog search — server-rendered, curl-able. + +Discovered 2026-07-15 from the advanced-search page JS (functions.js +CreateClientRequest): the "JS" search is a plain GET: + + https://nbmgu.ru/search/?adv=1&q1=<query>&f1=<FIELD>&v1=0&cat=BOOK[&p=<page>] + +Fields (f1..f3, AND-combined): + Fields (f1..f3, AND-combined between rows): + ANY все · AUT автор · COA колл. автор · TIT заглавие · KEY ключ. слова + RUB рубрики · YEA год · PLA место · PUB изд-во · SER серия + ISB ISBN · ISS ISSN · NBM индексы НБ МГУ +Method (v1..v3, per row): 0=Слова (stemming, default) · 1=Словосочетание (phrase) + · 2=Начинается с · 3=Дословно (exact). +Results: "Всего: N", 20 rows/page (p=0-based). Row = GOST description text +(title / authors; notes - [edition]) + "City : Publisher, Year" + shelf code. +Detail page /order/storing.aspx?uid=... is holdings/order only (no full record). +**ISB field NOT populated** (0 hits for 3 formats x 2 control ISBNs) — search by +AUT/TIT/SER/PUB instead; ISBN validation stays with РГБ/Google/OL (tools/isbnval.py). + +Usage: + tools/mgu.py "query" [FIELD=ANY] [pages=1] [method=0] + tools/mgu.py --multi "AUT:Франц" "TIT:Число" ["TIT:точное:3"] (AND rows) +""" +import re, subprocess, sys, time + +UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36" +BASE = "https://nbmgu.ru/search/" + + +def search_url(pairs, p=0, cat="BOOK"): + for attempt in range(3): + args = ["curl", "-s", "-G", "-A", UA, "--max-time", "90", "-e", BASE] + args += ["--data-urlencode", "adv=1"] + # mirror the browser: always send slots 1..N (empties included) + for i, pr in enumerate(pairs, 1): + f, q = pr[0], pr[1] + v = str(pr[2]) if len(pr) > 2 else "0" + args += ["--data-urlencode", f"q{i}={q}", + "--data-urlencode", f"f{i}={f}", + "--data-urlencode", f"v{i}={v}"] + if p: + args += ["--data-urlencode", f"p={p}"] + args += ["--data-urlencode", f"cat={cat}", BASE] + try: + out = subprocess.run(args, capture_output=True, timeout=120) + if out.returncode != 0 or len(out.stdout) < 5000: + print(f" (curl rc={out.returncode} size={len(out.stdout)} err={out.stderr[:120]!r})", + file=sys.stderr) + body = out.stdout.decode("utf-8", "replace") + except Exception as e: + print(f" (subprocess error: {e})", file=sys.stderr) + body = "" + if len(body) > 5000: + return body + time.sleep(4 * (attempt + 1)) + return body + + +def parse(html): + total = "" + m = re.search(r"Всего:\s*(\d+)", html) + if m: + total = m.group(1) + rows = [] + # NOTE: anchor title attr contains a raw '>' ("<br/>"), so plain [^>]+ fails; + # split on row start and rely on href being the LAST attribute of the anchor. + for chunk in html.split('<li class="result">')[1:]: + cm = re.search(r'<h2>\s*(\d+)\.\s*<a\b.*?href="(/order/storing\.aspx\?[^\"]+)"\s*>(.*?)</a>', + chunk, re.S) + pm = re.search(r"</h2>\s*<p>(.*?)</p>", chunk, re.S) + sm = re.search(r"Шифр:\s*([^<]+)", chunk) + if not (cm and pm): + continue + clean = lambda s: re.sub(r"\s+", " ", re.sub(r"<[^>]+>", " ", s)).strip() + rows.append({ + "n": cm.group(1), + "desc": clean(cm.group(3)), + "pubyear": clean(pm.group(1)), + "shelf": clean(sm.group(1)) if sm else "", + "url": "https://nbmgu.ru" + cm.group(2), + }) + return total, rows + + +def main(): + argv = sys.argv[1:] + if not argv: + sys.exit(__doc__) + if argv[0] == "--multi": + pairs = [] + for spec in argv[1:]: + parts = spec.split(":") + f = parts[0].upper() + q = parts[1] if len(parts) > 1 else "" + v = int(parts[2]) if len(parts) > 2 else 0 + pairs.append((f, q, v) if v else (f, q)) + pmax = 1 + else: + q = argv[0] + f = argv[1].upper() if len(argv) > 1 else "ANY" + v = int(argv[3]) if len(argv) > 3 else 0 + pairs = [(f, q, v) if v else (f, q)] + pmax = int(argv[2]) if len(argv) > 2 else 1 + all_rows = [] + total = "" + seen = set() + for p in range(pmax): + html = search_url(pairs, p=p) + total, rows = parse(html) + # out-of-range p falls back to page 0 on the server side — dedupe by url + new = [r for r in rows if r["url"] not in seen] + if not new: + break + seen.update(r["url"] for r in new) + all_rows += new + if total and len(all_rows) >= int(total): + break + time.sleep(3) + print(f"### MGU: {pairs} — total: {total} (fetched {len(all_rows)})") + for r in all_rows: + print(f"{r['n']:>3}. {r['desc']}") + print(f" {r['pubyear']} | шифр: {r['shelf']} | {r['url']}") + + +if __name__ == "__main__": + main()