#!/usr/bin/env python3 """CRW (fastcrw.com) — local Firecrawl-compatible JS renderer. Usage: crw.py # markdown crw.py --html # raw HTML crw.py --wait 5000 # waitFor ms for late-loading content Endpoint: POST http://localhost:3000/v1/scrape Payload: {"url": ..., "formats": ["markdown"] | ["html"], "waitFor": ms} Use for JS-heavy sites curl can't parse (e.g. soznanie.ast-academy.ru — rendered fine, it's the Intermodal Art Therapy festival site). Note: Insales shops (castalia.ru) work with curl (JSON-LD), no CRW needed. """ import json import sys import urllib.request def scrape(url, fmt="markdown", wait=None, timeout=120): payload = {"url": url, "formats": [fmt]} if wait: payload["waitFor"] = wait req = urllib.request.Request( "http://localhost:3000/v1/scrape", data=json.dumps(payload).encode(), headers={"Content-Type": "application/json"}, method="POST", ) with urllib.request.urlopen(req, timeout=timeout) as r: d = json.load(r) if not d.get("success"): raise RuntimeError("CRW failed: " + str(d)[:300]) return d["data"] if __name__ == "__main__": args = sys.argv[1:] if not args: print(__doc__) sys.exit(1) url = args[0] fmt = "html" if "--html" in args else "markdown" wait = int(args[args.index("--wait") + 1]) if "--wait" in args else None try: d = scrape(url, fmt, wait) print(d.get("markdown") or d.get("html", "")) except Exception as e: print("ERROR:", e) sys.exit(2)