#!/usr/bin/env python3 """NLR (РНБ, National Library of Russia) catalog via Primo (primo.nlr.ru). Usage: nlr.py "query" # free-text search, prints parsed results nlr.py "query" --full # also fetch full record (CRW) for the first hit nlr.py "query" --raw # print raw CRW markdown (debug) How it works: - Search is client-rendered; we render the search.do URL with local CRW (Firecrawl-compatible JS renderer, localhost:3000) and parse the markdown. - Result rows: ## [Title](display.do?...doc=07NLR_LMS#########...) ### Author (dates) ### Year *Holding (call no.)* - Count line: "Результаты 1 - 20 из *N*" Query semantics: free-text (all fields, words ANDed). For author sweeps use the surname; disambiguate via co-occurrence / life dates in the title line. Polite: CRW is local; still keep queries reasonable. """ import html import json import re import sys import urllib.parse import urllib.request def crw_scrape(url, wait=8000, timeout=150): payload = {"url": url, "formats": ["markdown"], "waitFor": wait} req = urllib.request.Request( "http://localhost:3000/v1/scrape", data=json.dumps(payload).encode(), headers={"Content-Type": "application/json"}, method="POST", ) with urllib.request.urlopen(req, timeout=timeout) as r: d = json.load(r) if not d.get("success"): raise RuntimeError("CRW failed: " + str(d)[:300]) return d["data"].get("markdown", "") def _parse_linked(md): """Format A: result rows contain '## [Title](display.do?...doc=ID...)' links.""" parts = re.split(r"(?=## \[)", md) out = [] for p in parts: if not p.startswith("## ["): continue tmatch = re.match(r"## \[(.*?)\]\(", p, re.S) if not tmatch: continue title = html.unescape(re.sub(r"<[^>]+>", "", tmatch.group(1))).strip() dmatch = re.search(r"doc=(07NLR_[A-Z0-9]+)", p) doc = dmatch.group(1) if dmatch else "?" headers = [html.unescape(h).strip() for h in re.findall(r"^### (.+)$", p, re.M)] author = headers[0] if headers else "" year = headers[1] if len(headers) > 1 else "" out.append((doc, title, author, year)) return out def _parse_plain(md): """Format B: flat lines — 'Material Type: X' / Title / Author... / Year / Holding.""" lines = [l.strip() for l in md.splitlines()] out = [] year_re = re.compile(r"^(cop\.)?\d{4}([-,]\d{3,4})?$") hold_re = re.compile(r"^(Хранится|Смотрите|Недоступно|Отметка|Бессрочно)") for i, l in enumerate(lines): if not l.startswith("Material Type:"): continue j = i + 1 while j < len(lines) and not lines[j]: j += 1 if j >= len(lines): continue title = html.unescape(lines[j]).strip() if title.startswith(("Record Details", "Рядом на полке", "Material Type")): continue k = j + 1 authors = [] year, holding = "", "" while k < len(lines): cur = lines[k] if year_re.match(cur): year = cur elif hold_re.match(cur): holding = html.unescape(cur).strip() break elif cur and not re.fullmatch(r"\d{1,3}", cur) and not cur.startswith(("Заказать", "Места", "Описание", "Отзывы", "Рядом", "Ещё", "Material Type", "Другие варианты")): a = html.unescape(cur) a = re.sub(r"NLR1[01]::RU\\NLR\\[Aa][Uu][Tt][Hh]\\\d+", "", a) a = re.sub(r"\s+\d{6,}(?=\s*$|,)", "", a) authors.append(a.strip()) k += 1 if year and holding: break author = "; ".join(authors[:2]) out.append(("?", title, author, year)) return out def search(query, raw=False): q = urllib.parse.quote(query) url = ("https://primo.nlr.ru/primo_library/libweb/action/search.do" f"?fn=search&ct=search&vl(freeText0)={q}&vid=07NLR_VU1&mode=Basic&initialSearch=true") md = crw_scrape(url) if raw: print(md) return m = re.search(r"Результаты[\s\d-]*из\s*\**\s*(\d+)", md) or re.search(r"из\s*\**\s*(\d+)", md) total = m.group(1) if m else "?" res = _parse_linked(md) or _parse_plain(md) print(f"### NLR: «{query}» — {len(res)} rows (page 1; stated total: {total})") for doc, title, author, year in res: line = f"- [{doc}] {title}" if author: line += f" — {author}" if year: line += f", {year}" print(line) if not res: print("(no parsed doc blocks — try --raw)") def main(): args = sys.argv[1:] if not args: print(__doc__) sys.exit(1) query = args[0] raw = "--raw" in args search(query, raw=raw) if __name__ == "__main__": main()