140 lines
5 KiB
Python
140 lines
5 KiB
Python
#!/usr/bin/env python3
|
|
"""NLR (РНБ, National Library of Russia) catalog via Primo (primo.nlr.ru).
|
|
|
|
Usage:
|
|
nlr.py "query" # free-text search, prints parsed results
|
|
nlr.py "query" --full # also fetch full record (CRW) for the first hit
|
|
nlr.py "query" --raw # print raw CRW markdown (debug)
|
|
|
|
How it works:
|
|
- Search is client-rendered; we render the search.do URL with local CRW
|
|
(Firecrawl-compatible JS renderer, localhost:3000) and parse the markdown.
|
|
- Result rows: ## [Title](display.do?...doc=07NLR_LMS#########...)
|
|
### Author (dates)
|
|
### Year
|
|
*Holding (call no.)*
|
|
- Count line: "Результаты 1 - 20 из *N*"
|
|
|
|
Query semantics: free-text (all fields, words ANDed). For author sweeps use
|
|
the surname; disambiguate via co-occurrence / life dates in the title line.
|
|
|
|
Polite: CRW is local; still keep queries reasonable.
|
|
"""
|
|
import html
|
|
import json
|
|
import re
|
|
import sys
|
|
import urllib.parse
|
|
import urllib.request
|
|
|
|
|
|
def crw_scrape(url, wait=8000, timeout=150):
|
|
payload = {"url": url, "formats": ["markdown"], "waitFor": wait}
|
|
req = urllib.request.Request(
|
|
"http://localhost:3000/v1/scrape",
|
|
data=json.dumps(payload).encode(),
|
|
headers={"Content-Type": "application/json"},
|
|
method="POST",
|
|
)
|
|
with urllib.request.urlopen(req, timeout=timeout) as r:
|
|
d = json.load(r)
|
|
if not d.get("success"):
|
|
raise RuntimeError("CRW failed: " + str(d)[:300])
|
|
return d["data"].get("markdown", "")
|
|
|
|
|
|
def _parse_linked(md):
|
|
"""Format A: result rows contain '## [Title](display.do?...doc=ID...)' links."""
|
|
parts = re.split(r"(?=## \[)", md)
|
|
out = []
|
|
for p in parts:
|
|
if not p.startswith("## ["):
|
|
continue
|
|
tmatch = re.match(r"## \[(.*?)\]\(", p, re.S)
|
|
if not tmatch:
|
|
continue
|
|
title = html.unescape(re.sub(r"<[^>]+>", "", tmatch.group(1))).strip()
|
|
dmatch = re.search(r"doc=(07NLR_[A-Z0-9]+)", p)
|
|
doc = dmatch.group(1) if dmatch else "?"
|
|
headers = [html.unescape(h).strip() for h in re.findall(r"^### (.+)$", p, re.M)]
|
|
author = headers[0] if headers else ""
|
|
year = headers[1] if len(headers) > 1 else ""
|
|
out.append((doc, title, author, year))
|
|
return out
|
|
|
|
|
|
def _parse_plain(md):
|
|
"""Format B: flat lines — 'Material Type: X' / Title / Author... / Year / Holding."""
|
|
lines = [l.strip() for l in md.splitlines()]
|
|
out = []
|
|
year_re = re.compile(r"^(cop\.)?\d{4}([-,]\d{3,4})?$")
|
|
hold_re = re.compile(r"^(Хранится|Смотрите|Недоступно|Отметка|Бессрочно)")
|
|
for i, l in enumerate(lines):
|
|
if not l.startswith("Material Type:"):
|
|
continue
|
|
j = i + 1
|
|
while j < len(lines) and not lines[j]:
|
|
j += 1
|
|
if j >= len(lines):
|
|
continue
|
|
title = html.unescape(lines[j]).strip()
|
|
if title.startswith(("Record Details", "Рядом на полке", "Material Type")):
|
|
continue
|
|
k = j + 1
|
|
authors = []
|
|
year, holding = "", ""
|
|
while k < len(lines):
|
|
cur = lines[k]
|
|
if year_re.match(cur):
|
|
year = cur
|
|
elif hold_re.match(cur):
|
|
holding = html.unescape(cur).strip()
|
|
break
|
|
elif cur and not re.fullmatch(r"\d{1,3}", cur) and not cur.startswith(("Заказать", "Места", "Описание", "Отзывы", "Рядом", "Ещё", "Material Type", "Другие варианты")):
|
|
a = html.unescape(cur)
|
|
a = re.sub(r"NLR1[01]::RU\\NLR\\[Aa][Uu][Tt][Hh]\\\d+", "", a)
|
|
a = re.sub(r"\s+\d{6,}(?=\s*$|,)", "", a)
|
|
authors.append(a.strip())
|
|
k += 1
|
|
if year and holding:
|
|
break
|
|
author = "; ".join(authors[:2])
|
|
out.append(("?", title, author, year))
|
|
return out
|
|
|
|
|
|
def search(query, raw=False):
|
|
q = urllib.parse.quote(query)
|
|
url = ("https://primo.nlr.ru/primo_library/libweb/action/search.do"
|
|
f"?fn=search&ct=search&vl(freeText0)={q}&vid=07NLR_VU1&mode=Basic&initialSearch=true")
|
|
md = crw_scrape(url)
|
|
if raw:
|
|
print(md)
|
|
return
|
|
m = re.search(r"Результаты[\s\d-]*из\s*\**\s*(\d+)", md) or re.search(r"из\s*\**\s*(\d+)", md)
|
|
total = m.group(1) if m else "?"
|
|
res = _parse_linked(md) or _parse_plain(md)
|
|
print(f"### NLR: «{query}» — {len(res)} rows (page 1; stated total: {total})")
|
|
for doc, title, author, year in res:
|
|
line = f"- [{doc}] {title}"
|
|
if author:
|
|
line += f" — {author}"
|
|
if year:
|
|
line += f", {year}"
|
|
print(line)
|
|
if not res:
|
|
print("(no parsed doc blocks — try --raw)")
|
|
|
|
|
|
def main():
|
|
args = sys.argv[1:]
|
|
if not args:
|
|
print(__doc__)
|
|
sys.exit(1)
|
|
query = args[0]
|
|
raw = "--raw" in args
|
|
search(query, raw=raw)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|