NLR sweep round 7: all section-04 surnames checked (tools/nlr.py, CRW+Primo); Cunningham 2018 identified; Dora Kalff confirmed absent from NLR (correct spelling Калфф); Schaverien 'Умирающий пациент' in NLR holdings
This commit is contained in:
parent
52811eb793
commit
6decbd3146
2 changed files with 149 additions and 0 deletions
|
|
@ -25,3 +25,12 @@ in RSL/cogito).
|
|||
- **«Руководство по сэндплей-терапии»** (The International Handbook of Sandplay Therapy,
|
||||
red. Barbara Turner et al.) — М.: **Дипак** (RU sandplay association), 2015, 647 с.,
|
||||
ISBN 978-5-98580-069-2, серия «Песочная терапия» (RSL 7609331; alib listing 648 с.).
|
||||
|
||||
## Round 5 (NLR sweep, 2026-07-09)
|
||||
|
||||
- **«Сэндплей и терапевтические отношения» = Линда Каннингем** (NOT Dora Kalff):
|
||||
Москва: [б. и.], 2018, 176 с., пер. с англ. М. Думчев и др. (NLR 07NLR_LMS011717294).
|
||||
- NLR holds no Дора Калфф at all: queries «Калфф» (0), «Сэндплей целительное
|
||||
воздействие» (0), «Сэндплей психотерапевтический подход» (0). NLR's whole sandplay shelf:
|
||||
Cunningham 2018, «Девять окон в целостность» (07NLR_LMS011694768), Turner handbook
|
||||
(07NLR_LMS010967377), «Перечень категорий анализа...» (07NLR_LMS011668396).
|
||||
|
|
|
|||
140
tools/nlr.py
Normal file
140
tools/nlr.py
Normal file
|
|
@ -0,0 +1,140 @@
|
|||
#!/usr/bin/env python3
|
||||
"""NLR (РНБ, National Library of Russia) catalog via Primo (primo.nlr.ru).
|
||||
|
||||
Usage:
|
||||
nlr.py "query" # free-text search, prints parsed results
|
||||
nlr.py "query" --full # also fetch full record (CRW) for the first hit
|
||||
nlr.py "query" --raw # print raw CRW markdown (debug)
|
||||
|
||||
How it works:
|
||||
- Search is client-rendered; we render the search.do URL with local CRW
|
||||
(Firecrawl-compatible JS renderer, localhost:3000) and parse the markdown.
|
||||
- Result rows: ## [Title](display.do?...doc=07NLR_LMS#########...)
|
||||
### Author (dates)
|
||||
### Year
|
||||
*Holding (call no.)*
|
||||
- Count line: "Результаты 1 - 20 из *N*"
|
||||
|
||||
Query semantics: free-text (all fields, words ANDed). For author sweeps use
|
||||
the surname; disambiguate via co-occurrence / life dates in the title line.
|
||||
|
||||
Polite: CRW is local; still keep queries reasonable.
|
||||
"""
|
||||
import html
|
||||
import json
|
||||
import re
|
||||
import sys
|
||||
import urllib.parse
|
||||
import urllib.request
|
||||
|
||||
|
||||
def crw_scrape(url, wait=8000, timeout=150):
|
||||
payload = {"url": url, "formats": ["markdown"], "waitFor": wait}
|
||||
req = urllib.request.Request(
|
||||
"http://localhost:3000/v1/scrape",
|
||||
data=json.dumps(payload).encode(),
|
||||
headers={"Content-Type": "application/json"},
|
||||
method="POST",
|
||||
)
|
||||
with urllib.request.urlopen(req, timeout=timeout) as r:
|
||||
d = json.load(r)
|
||||
if not d.get("success"):
|
||||
raise RuntimeError("CRW failed: " + str(d)[:300])
|
||||
return d["data"].get("markdown", "")
|
||||
|
||||
|
||||
def _parse_linked(md):
|
||||
"""Format A: result rows contain '## [Title](display.do?...doc=ID...)' links."""
|
||||
parts = re.split(r"(?=## \[)", md)
|
||||
out = []
|
||||
for p in parts:
|
||||
if not p.startswith("## ["):
|
||||
continue
|
||||
tmatch = re.match(r"## \[(.*?)\]\(", p, re.S)
|
||||
if not tmatch:
|
||||
continue
|
||||
title = html.unescape(re.sub(r"<[^>]+>", "", tmatch.group(1))).strip()
|
||||
dmatch = re.search(r"doc=(07NLR_[A-Z0-9]+)", p)
|
||||
doc = dmatch.group(1) if dmatch else "?"
|
||||
headers = [html.unescape(h).strip() for h in re.findall(r"^### (.+)$", p, re.M)]
|
||||
author = headers[0] if headers else ""
|
||||
year = headers[1] if len(headers) > 1 else ""
|
||||
out.append((doc, title, author, year))
|
||||
return out
|
||||
|
||||
|
||||
def _parse_plain(md):
|
||||
"""Format B: flat lines — 'Material Type: X' / Title / Author... / Year / Holding."""
|
||||
lines = [l.strip() for l in md.splitlines()]
|
||||
out = []
|
||||
year_re = re.compile(r"^(cop\.)?\d{4}([-,]\d{3,4})?$")
|
||||
hold_re = re.compile(r"^(Хранится|Смотрите|Недоступно|Отметка|Бессрочно)")
|
||||
for i, l in enumerate(lines):
|
||||
if not l.startswith("Material Type:"):
|
||||
continue
|
||||
j = i + 1
|
||||
while j < len(lines) and not lines[j]:
|
||||
j += 1
|
||||
if j >= len(lines):
|
||||
continue
|
||||
title = html.unescape(lines[j]).strip()
|
||||
if title.startswith(("Record Details", "Рядом на полке", "Material Type")):
|
||||
continue
|
||||
k = j + 1
|
||||
authors = []
|
||||
year, holding = "", ""
|
||||
while k < len(lines):
|
||||
cur = lines[k]
|
||||
if year_re.match(cur):
|
||||
year = cur
|
||||
elif hold_re.match(cur):
|
||||
holding = html.unescape(cur).strip()
|
||||
break
|
||||
elif cur and not re.fullmatch(r"\d{1,3}", cur) and not cur.startswith(("Заказать", "Места", "Описание", "Отзывы", "Рядом", "Ещё", "Material Type", "Другие варианты")):
|
||||
a = html.unescape(cur)
|
||||
a = re.sub(r"NLR1[01]::RU\\NLR\\[Aa][Uu][Tt][Hh]\\\d+", "", a)
|
||||
a = re.sub(r"\s+\d{6,}(?=\s*$|,)", "", a)
|
||||
authors.append(a.strip())
|
||||
k += 1
|
||||
if year and holding:
|
||||
break
|
||||
author = "; ".join(authors[:2])
|
||||
out.append(("?", title, author, year))
|
||||
return out
|
||||
|
||||
|
||||
def search(query, raw=False):
|
||||
q = urllib.parse.quote(query)
|
||||
url = ("https://primo.nlr.ru/primo_library/libweb/action/search.do"
|
||||
f"?fn=search&ct=search&vl(freeText0)={q}&vid=07NLR_VU1&mode=Basic&initialSearch=true")
|
||||
md = crw_scrape(url)
|
||||
if raw:
|
||||
print(md)
|
||||
return
|
||||
m = re.search(r"Результаты[\s\d-]*из\s*\**\s*(\d+)", md) or re.search(r"из\s*\**\s*(\d+)", md)
|
||||
total = m.group(1) if m else "?"
|
||||
res = _parse_linked(md) or _parse_plain(md)
|
||||
print(f"### NLR: «{query}» — {len(res)} rows (page 1; stated total: {total})")
|
||||
for doc, title, author, year in res:
|
||||
line = f"- [{doc}] {title}"
|
||||
if author:
|
||||
line += f" — {author}"
|
||||
if year:
|
||||
line += f", {year}"
|
||||
print(line)
|
||||
if not res:
|
||||
print("(no parsed doc blocks — try --raw)")
|
||||
|
||||
|
||||
def main():
|
||||
args = sys.argv[1:]
|
||||
if not args:
|
||||
print(__doc__)
|
||||
sys.exit(1)
|
||||
query = args[0]
|
||||
raw = "--raw" in args
|
||||
search(query, raw=raw)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Loading…
Add table
Add a link
Reference in a new issue