jung/tools/mgu.py

128 lines
5.1 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""MGU Scientific Library (nbmgu.ru) catalog search — server-rendered, curl-able.
Discovered 2026-07-15 from the advanced-search page JS (functions.js
CreateClientRequest): the "JS" search is a plain GET:
https://nbmgu.ru/search/?adv=1&q1=<query>&f1=<FIELD>&v1=0&cat=BOOK[&p=<page>]
Fields (f1..f3, AND-combined):
Fields (f1..f3, AND-combined between rows):
ANY все · AUT автор · COA колл. автор · TIT заглавие · KEY ключ. слова
RUB рубрики · YEA год · PLA место · PUB изд-во · SER серия
ISB ISBN · ISS ISSN · NBM индексы НБ МГУ
Method (v1..v3, per row): 0=Слова (stemming, default) · 1=Словосочетание (phrase)
· 2=Начинается с · 3=Дословно (exact).
Results: "Всего: N", 20 rows/page (p=0-based). Row = GOST description text
(title / authors; notes - [edition]) + "City : Publisher, Year" + shelf code.
Detail page /order/storing.aspx?uid=... is holdings/order only (no full record).
**ISB field NOT populated** (0 hits for 3 formats x 2 control ISBNs) — search by
AUT/TIT/SER/PUB instead; ISBN validation stays with РГБ/Google/OL (tools/isbnval.py).
Usage:
tools/mgu.py "query" [FIELD=ANY] [pages=1] [method=0]
tools/mgu.py --multi "AUT:Франц" "TIT:Число" ["TIT:точное:3"] (AND rows)
"""
import re, subprocess, sys, time
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36"
BASE = "https://nbmgu.ru/search/"
def search_url(pairs, p=0, cat="BOOK"):
for attempt in range(3):
args = ["curl", "-s", "-G", "-A", UA, "--max-time", "90", "-e", BASE]
args += ["--data-urlencode", "adv=1"]
# mirror the browser: always send slots 1..N (empties included)
for i, pr in enumerate(pairs, 1):
f, q = pr[0], pr[1]
v = str(pr[2]) if len(pr) > 2 else "0"
args += ["--data-urlencode", f"q{i}={q}",
"--data-urlencode", f"f{i}={f}",
"--data-urlencode", f"v{i}={v}"]
if p:
args += ["--data-urlencode", f"p={p}"]
args += ["--data-urlencode", f"cat={cat}", BASE]
try:
out = subprocess.run(args, capture_output=True, timeout=120)
if out.returncode != 0 or len(out.stdout) < 5000:
print(f" (curl rc={out.returncode} size={len(out.stdout)} err={out.stderr[:120]!r})",
file=sys.stderr)
body = out.stdout.decode("utf-8", "replace")
except Exception as e:
print(f" (subprocess error: {e})", file=sys.stderr)
body = ""
if len(body) > 5000:
return body
time.sleep(4 * (attempt + 1))
return body
def parse(html):
total = ""
m = re.search(r"Всего:\s*(\d+)", html)
if m:
total = m.group(1)
rows = []
# NOTE: anchor title attr contains a raw '>' ("<br/>"), so plain [^>]+ fails;
# split on row start and rely on href being the LAST attribute of the anchor.
for chunk in html.split('<li class="result">')[1:]:
cm = re.search(r'<h2>\s*(\d+)\.\s*<a\b.*?href="(/order/storing\.aspx\?[^\"]+)"\s*>(.*?)</a>',
chunk, re.S)
pm = re.search(r"</h2>\s*<p>(.*?)</p>", chunk, re.S)
sm = re.search(r"Шифр:\s*([^<]+)", chunk)
if not (cm and pm):
continue
clean = lambda s: re.sub(r"\s+", " ", re.sub(r"<[^>]+>", " ", s)).strip()
rows.append({
"n": cm.group(1),
"desc": clean(cm.group(3)),
"pubyear": clean(pm.group(1)),
"shelf": clean(sm.group(1)) if sm else "",
"url": "https://nbmgu.ru" + cm.group(2),
})
return total, rows
def main():
argv = sys.argv[1:]
if not argv:
sys.exit(__doc__)
if argv[0] == "--multi":
pairs = []
for spec in argv[1:]:
parts = spec.split(":")
f = parts[0].upper()
q = parts[1] if len(parts) > 1 else ""
v = int(parts[2]) if len(parts) > 2 else 0
pairs.append((f, q, v) if v else (f, q))
pmax = 1
else:
q = argv[0]
f = argv[1].upper() if len(argv) > 1 else "ANY"
v = int(argv[3]) if len(argv) > 3 else 0
pairs = [(f, q, v) if v else (f, q)]
pmax = int(argv[2]) if len(argv) > 2 else 1
all_rows = []
total = ""
seen = set()
for p in range(pmax):
html = search_url(pairs, p=p)
total, rows = parse(html)
# out-of-range p falls back to page 0 on the server side — dedupe by url
new = [r for r in rows if r["url"] not in seen]
if not new:
break
seen.update(r["url"] for r in new)
all_rows += new
if total and len(all_rows) >= int(total):
break
time.sleep(3)
print(f"### MGU: {pairs} — total: {total} (fetched {len(all_rows)})")
for r in all_rows:
print(f"{r['n']:>3}. {r['desc']}")
print(f" {r['pubyear']} | шифр: {r['shelf']} | {r['url']}")
if __name__ == "__main__":
main()