NLR sweep round 7: all section-04 surnames checked (tools/nlr.py, CRW+Primo); Cunningham 2018 identified; Dora Kalff confirmed absent from NLR (correct spelling Калфф); Schaverien 'Умирающий пациент' in NLR holdings

This commit is contained in:
Dmitry Kokorin 2026-09-09 12:38:57 +03:00
parent 52811eb793
commit 6decbd3146
2 changed files with 149 additions and 0 deletions

140
tools/nlr.py Normal file
View file

@ -0,0 +1,140 @@
#!/usr/bin/env python3
"""NLR (РНБ, National Library of Russia) catalog via Primo (primo.nlr.ru).
Usage:
nlr.py "query" # free-text search, prints parsed results
nlr.py "query" --full # also fetch full record (CRW) for the first hit
nlr.py "query" --raw # print raw CRW markdown (debug)
How it works:
- Search is client-rendered; we render the search.do URL with local CRW
(Firecrawl-compatible JS renderer, localhost:3000) and parse the markdown.
- Result rows: ## [Title](display.do?...doc=07NLR_LMS#########...)
### Author (dates)
### Year
*Holding (call no.)*
- Count line: "Результаты 1 - 20 из *N*"
Query semantics: free-text (all fields, words ANDed). For author sweeps use
the surname; disambiguate via co-occurrence / life dates in the title line.
Polite: CRW is local; still keep queries reasonable.
"""
import html
import json
import re
import sys
import urllib.parse
import urllib.request
def crw_scrape(url, wait=8000, timeout=150):
payload = {"url": url, "formats": ["markdown"], "waitFor": wait}
req = urllib.request.Request(
"http://localhost:3000/v1/scrape",
data=json.dumps(payload).encode(),
headers={"Content-Type": "application/json"},
method="POST",
)
with urllib.request.urlopen(req, timeout=timeout) as r:
d = json.load(r)
if not d.get("success"):
raise RuntimeError("CRW failed: " + str(d)[:300])
return d["data"].get("markdown", "")
def _parse_linked(md):
"""Format A: result rows contain '## [Title](display.do?...doc=ID...)' links."""
parts = re.split(r"(?=## \[)", md)
out = []
for p in parts:
if not p.startswith("## ["):
continue
tmatch = re.match(r"## \[(.*?)\]\(", p, re.S)
if not tmatch:
continue
title = html.unescape(re.sub(r"<[^>]+>", "", tmatch.group(1))).strip()
dmatch = re.search(r"doc=(07NLR_[A-Z0-9]+)", p)
doc = dmatch.group(1) if dmatch else "?"
headers = [html.unescape(h).strip() for h in re.findall(r"^### (.+)$", p, re.M)]
author = headers[0] if headers else ""
year = headers[1] if len(headers) > 1 else ""
out.append((doc, title, author, year))
return out
def _parse_plain(md):
"""Format B: flat lines — 'Material Type: X' / Title / Author... / Year / Holding."""
lines = [l.strip() for l in md.splitlines()]
out = []
year_re = re.compile(r"^(cop\.)?\d{4}([-,]\d{3,4})?$")
hold_re = re.compile(r"^(Хранится|Смотрите|Недоступно|Отметка|Бессрочно)")
for i, l in enumerate(lines):
if not l.startswith("Material Type:"):
continue
j = i + 1
while j < len(lines) and not lines[j]:
j += 1
if j >= len(lines):
continue
title = html.unescape(lines[j]).strip()
if title.startswith(("Record Details", "Рядом на полке", "Material Type")):
continue
k = j + 1
authors = []
year, holding = "", ""
while k < len(lines):
cur = lines[k]
if year_re.match(cur):
year = cur
elif hold_re.match(cur):
holding = html.unescape(cur).strip()
break
elif cur and not re.fullmatch(r"\d{1,3}", cur) and not cur.startswith(("Заказать", "Места", "Описание", "Отзывы", "Рядом", "Ещё", "Material Type", "Другие варианты")):
a = html.unescape(cur)
a = re.sub(r"NLR1[01]::RU\\NLR\\[Aa][Uu][Tt][Hh]\\\d+", "", a)
a = re.sub(r"\s+\d{6,}(?=\s*$|,)", "", a)
authors.append(a.strip())
k += 1
if year and holding:
break
author = "; ".join(authors[:2])
out.append(("?", title, author, year))
return out
def search(query, raw=False):
q = urllib.parse.quote(query)
url = ("https://primo.nlr.ru/primo_library/libweb/action/search.do"
f"?fn=search&ct=search&vl(freeText0)={q}&vid=07NLR_VU1&mode=Basic&initialSearch=true")
md = crw_scrape(url)
if raw:
print(md)
return
m = re.search(r"Результаты[\s\d-]*из\s*\**\s*(\d+)", md) or re.search(r"из\s*\**\s*(\d+)", md)
total = m.group(1) if m else "?"
res = _parse_linked(md) or _parse_plain(md)
print(f"### NLR: «{query}» — {len(res)} rows (page 1; stated total: {total})")
for doc, title, author, year in res:
line = f"- [{doc}] {title}"
if author:
line += f" — {author}"
if year:
line += f", {year}"
print(line)
if not res:
print("(no parsed doc blocks — try --raw)")
def main():
args = sys.argv[1:]
if not args:
print(__doc__)
sys.exit(1)
query = args[0]
raw = "--raw" in args
search(query, raw=raw)
if __name__ == "__main__":
main()