#38 FOUND: «Символизм цвета. Из архивов Эраноса 1972» (Клуб Касталия, ISBN 978-5-90811-054-9) — same Ottmann editor as Spring 2005; via castalia.ru sitemap sweep (Phase 2b)

This commit is contained in:
Dmitry Kokorin 2026-09-09 00:46:19 +03:00
parent 8b603f89fd
commit 6dc0738584
3 changed files with 65 additions and 6 deletions

51
tools/crw.py Normal file
View file

@ -0,0 +1,51 @@
#!/usr/bin/env python3
"""CRW (fastcrw.com) — local Firecrawl-compatible JS renderer.
Usage:
crw.py <url> # markdown
crw.py <url> --html # raw HTML
crw.py <url> --wait 5000 # waitFor ms for late-loading content
Endpoint: POST http://localhost:3000/v1/scrape
Payload: {"url": ..., "formats": ["markdown"] | ["html"], "waitFor": ms}
Use for JS-heavy sites curl can't parse (e.g. soznanie.ast-academy.ru —
rendered fine, it's the Intermodal Art Therapy festival site).
Note: Insales shops (castalia.ru) work with curl (JSON-LD), no CRW needed.
"""
import json
import sys
import urllib.request
def scrape(url, fmt="markdown", wait=None, timeout=120):
payload = {"url": url, "formats": [fmt]}
if wait:
payload["waitFor"] = wait
req = urllib.request.Request(
"http://localhost:3000/v1/scrape",
data=json.dumps(payload).encode(),
headers={"Content-Type": "application/json"},
method="POST",
)
with urllib.request.urlopen(req, timeout=timeout) as r:
d = json.load(r)
if not d.get("success"):
raise RuntimeError("CRW failed: " + str(d)[:300])
return d["data"]
if __name__ == "__main__":
args = sys.argv[1:]
if not args:
print(__doc__)
sys.exit(1)
url = args[0]
fmt = "html" if "--html" in args else "markdown"
wait = int(args[args.index("--wait") + 1]) if "--wait" in args else None
try:
d = scrape(url, fmt, wait)
print(d.get("markdown") or d.get("html", ""))
except Exception as e:
print("ERROR:", e)
sys.exit(2)