jung/tools/crw.py

51 lines
1.6 KiB
Python

#!/usr/bin/env python3
"""CRW (fastcrw.com) — local Firecrawl-compatible JS renderer.
Usage:
crw.py <url> # markdown
crw.py <url> --html # raw HTML
crw.py <url> --wait 5000 # waitFor ms for late-loading content
Endpoint: POST http://localhost:3000/v1/scrape
Payload: {"url": ..., "formats": ["markdown"] | ["html"], "waitFor": ms}
Use for JS-heavy sites curl can't parse (e.g. soznanie.ast-academy.ru —
rendered fine, it's the Intermodal Art Therapy festival site).
Note: Insales shops (castalia.ru) work with curl (JSON-LD), no CRW needed.
"""
import json
import sys
import urllib.request
def scrape(url, fmt="markdown", wait=None, timeout=120):
payload = {"url": url, "formats": [fmt]}
if wait:
payload["waitFor"] = wait
req = urllib.request.Request(
"http://localhost:3000/v1/scrape",
data=json.dumps(payload).encode(),
headers={"Content-Type": "application/json"},
method="POST",
)
with urllib.request.urlopen(req, timeout=timeout) as r:
d = json.load(r)
if not d.get("success"):
raise RuntimeError("CRW failed: " + str(d)[:300])
return d["data"]
if __name__ == "__main__":
args = sys.argv[1:]
if not args:
print(__doc__)
sys.exit(1)
url = args[0]
fmt = "html" if "--html" in args else "markdown"
wait = int(args[args.index("--wait") + 1]) if "--wait" in args else None
try:
d = scrape(url, fmt, wait)
print(d.get("markdown") or d.get("html", ""))
except Exception as e:
print("ERROR:", e)
sys.exit(2)