51 lines
1.6 KiB
Python
51 lines
1.6 KiB
Python
#!/usr/bin/env python3
|
|
"""CRW (fastcrw.com) — local Firecrawl-compatible JS renderer.
|
|
|
|
Usage:
|
|
crw.py <url> # markdown
|
|
crw.py <url> --html # raw HTML
|
|
crw.py <url> --wait 5000 # waitFor ms for late-loading content
|
|
|
|
Endpoint: POST http://localhost:3000/v1/scrape
|
|
Payload: {"url": ..., "formats": ["markdown"] | ["html"], "waitFor": ms}
|
|
|
|
Use for JS-heavy sites curl can't parse (e.g. soznanie.ast-academy.ru —
|
|
rendered fine, it's the Intermodal Art Therapy festival site).
|
|
Note: Insales shops (castalia.ru) work with curl (JSON-LD), no CRW needed.
|
|
"""
|
|
import json
|
|
import sys
|
|
import urllib.request
|
|
|
|
|
|
def scrape(url, fmt="markdown", wait=None, timeout=120):
|
|
payload = {"url": url, "formats": [fmt]}
|
|
if wait:
|
|
payload["waitFor"] = wait
|
|
req = urllib.request.Request(
|
|
"http://localhost:3000/v1/scrape",
|
|
data=json.dumps(payload).encode(),
|
|
headers={"Content-Type": "application/json"},
|
|
method="POST",
|
|
)
|
|
with urllib.request.urlopen(req, timeout=timeout) as r:
|
|
d = json.load(r)
|
|
if not d.get("success"):
|
|
raise RuntimeError("CRW failed: " + str(d)[:300])
|
|
return d["data"]
|
|
|
|
|
|
if __name__ == "__main__":
|
|
args = sys.argv[1:]
|
|
if not args:
|
|
print(__doc__)
|
|
sys.exit(1)
|
|
url = args[0]
|
|
fmt = "html" if "--html" in args else "markdown"
|
|
wait = int(args[args.index("--wait") + 1]) if "--wait" in args else None
|
|
try:
|
|
d = scrape(url, fmt, wait)
|
|
print(d.get("markdown") or d.get("html", ""))
|
|
except Exception as e:
|
|
print("ERROR:", e)
|
|
sys.exit(2)
|