"""ClawEngine client for Python. One file, no dependency beyond `requests`.

Copy it into your project, set CLAWENGINE_API_KEY, then:

    from clawengine import ClawEngine
    claw = ClawEngine()
    page = claw.extract("https://example.com/pricing", format="json")
    pages = claw.crawl("https://example.com/docs", path_prefix="/docs", limit=200)   # waits, follows `next`
    mon = claw.create_monitor("https://shop.example.com/p/1", interval="daily",
                              schema={"price": "number"}, alerts=[{"field": "price", "below": 49.99}])

Every method raises ClawEngineError with the API's error type and message.
Reference: https://clawengine.ai/docs
"""

import os
import time

import requests

__version__ = "1.0.0"


class ClawEngineError(Exception):
    def __init__(self, status, type_, message):
        super().__init__(f"{status} {type_}: {message}")
        self.status = status
        self.type = type_
        self.message = message


class ClawEngine:
    def __init__(self, api_key=None, base_url="https://clawengine.ai/v1", timeout=120):
        self.api_key = api_key or os.environ.get("CLAWENGINE_API_KEY")
        if not self.api_key:
            raise ValueError("Pass api_key or set CLAWENGINE_API_KEY")
        self.base_url = base_url.rstrip("/")
        self.timeout = timeout
        self.session = requests.Session()
        self.session.headers.update({"Authorization": f"Bearer {self.api_key}", "User-Agent": f"clawengine-python/{__version__}"})

    def _call(self, method, path, json=None, params=None, raw=False):
        url = path if path.startswith("http") else self.base_url + path
        resp = self.session.request(method, url, json=json, params=params, timeout=self.timeout)
        if resp.status_code >= 400:
            try:
                err = resp.json().get("error", {})
            except ValueError:
                err = {}
            raise ClawEngineError(resp.status_code, err.get("type", "http_error"), err.get("message", resp.text[:300]))
        return resp if raw else resp.json()

    # Pages

    def extract(self, url, **options):
        """One page: markdown (default), json, or typed fields with schema=..."""
        return self._call("POST", "/extract", json={"url": url, **options})

    def extract_many(self, urls, **options):
        """Up to 10 URLs in one call; returns the list of results in the order sent."""
        return self._call("POST", "/extract", json={"urls": list(urls), **options})["results"]

    # Crawls

    def start_crawl(self, url=None, **options):
        body = dict(options)
        if url is not None:
            body["url"] = url
        return self._call("POST", "/crawl", json=body)

    def get_crawl(self, crawl_id, skip=0):
        return self._call("GET", f"/crawl/{crawl_id}", params={"skip": skip} if skip else None)

    def wait_for_crawl(self, crawl_id, poll_seconds=3, timeout_seconds=3 * 3600):
        """Polls until the crawl is completed or failed, then returns every page."""
        deadline = time.time() + timeout_seconds
        while True:
            body = self.get_crawl(crawl_id)
            if body["status"] in ("completed", "failed"):
                break
            if time.time() > deadline:
                raise TimeoutError(f"crawl {crawl_id} still {body['status']}")
            time.sleep(poll_seconds)
        if body["status"] == "failed":
            raise ClawEngineError(200, "crawl_failed", body.get("error", "The crawl failed."))
        pages = list(body.get("pages", []))
        while body.get("next"):
            body = self._call("GET", body["next"])
            pages.extend(body.get("pages", []))
        return pages

    def crawl(self, url=None, **options):
        """Starts a crawl and returns all its pages when it is done."""
        return self.wait_for_crawl(self.start_crawl(url, **options)["id"])

    def llms_txt(self, crawl_id, full=False):
        return self._call("GET", f"/crawl/{crawl_id}/{'llms-full.txt' if full else 'llms.txt'}", raw=True).text

    def markdown_zip(self, crawl_id, path):
        """Saves one .md file per page (plus llms.txt and llms-full.txt) as a zip at `path`."""
        resp = self._call("GET", f"/crawl/{crawl_id}/files.zip", raw=True)
        with open(path, "wb") as fh:
            fh.write(resp.content)
        return path

    # Monitors

    def create_monitor(self, url, **options):
        return self._call("POST", "/monitors", json={"url": url, **options})

    def monitors(self):
        return self._call("GET", "/monitors")["monitors"]

    def get_monitor(self, monitor_id, skip=0):
        return self._call("GET", f"/monitors/{monitor_id}", params={"skip": skip} if skip else None)

    def get_snapshot(self, monitor_id, seq):
        return self._call("GET", f"/monitors/{monitor_id}/snapshots/{seq}")

    def update_monitor(self, monitor_id, **changes):
        return self._call("PATCH", f"/monitors/{monitor_id}", json=changes)

    def run_monitor(self, monitor_id):
        return self._call("POST", f"/monitors/{monitor_id}/run")

    def delete_monitor(self, monitor_id):
        return self._call("DELETE", f"/monitors/{monitor_id}")
