#!/usr/bin/env python3
"""One bounded Scraipe extraction. Standard library only; no automatic retries."""
import argparse
from datetime import datetime, timezone
import json
import os
import re
import sys
from urllib.error import HTTPError, URLError
from urllib.parse import urlsplit
from urllib.request import HTTPRedirectHandler, Request, build_opener


class ExtractionError(Exception):
    """A failed request or a result that must not enter the dataset."""


class NoRedirect(HTTPRedirectHandler):
    def redirect_request(self, req, fp, code, msg, headers, newurl):
        return None  # Never forward the API credential to a redirect target.


def safe_token(value):
    return str(value)[:120] if re.fullmatch(r"[A-Za-z0-9_.:-]{1,120}", str(value)) else "unknown"


def api_error(payload, status, retry_after=None):
    error = payload.get("error") if isinstance(payload, dict) else None
    error = error if isinstance(error, dict) else {}
    request_id = payload.get("requestId") if isinstance(payload, dict) else None
    return ExtractionError(
        f"HTTP {status}; reason={safe_token(error.get('reason'))}; "
        f"requestId={safe_token(request_id)}; "
        f"retryAfter={safe_token(error.get('retryAfter', retry_after))}. "
        "No retry was sent. Inspect the error and receipt before retrying."
    )


def extract(api_base, api_key, source_url, operation_key, timeout=60):
    base = urlsplit(api_base)
    local_http = base.scheme == "http" and base.hostname in {"localhost", "127.0.0.1", "::1"}
    if (base.scheme != "https" and not local_http) or not base.hostname or base.username or base.password or base.query or base.fragment or base.path not in {"", "/"}:
        raise ExtractionError("Use an HTTPS API origin without credentials, path or query. HTTP is allowed only for local tests.")
    source = urlsplit(source_url)
    if source.scheme not in {"http", "https"} or not source.hostname or source.username or source.password:
        raise ExtractionError("Use a complete HTTP(S) source URL without embedded credentials.")
    if not api_key or "\n" in api_key or "\r" in api_key:
        raise ExtractionError("Set SCRAIPE_API_KEY to a valid server-side key.")
    if not re.fullmatch(r"[A-Za-z0-9_-]{1,80}", operation_key):
        raise ExtractionError("Use a 1–80 character request ID containing letters, digits, underscores or hyphens.")

    body = {"url": source_url, "fields": ["title", "price"], "maxTokens": 2000, "egress": "direct"}
    request = Request(api_base.rstrip("/") + "/v1/scrape", data=json.dumps(body).encode(), headers={
        "Authorization": "Bearer " + api_key,
        "Content-Type": "application/json",
        "Accept": "application/json",
        "Idempotency-Key": operation_key,
    }, method="POST")
    opener = build_opener(NoRedirect())
    try:
        with opener.open(request, timeout=timeout) as response:
            raw = response.read(1_048_577)
    except HTTPError as error:
        with error:
            try:
                payload = json.loads(error.read(1_048_576))
            except (ValueError, UnicodeError):
                payload = {}
            raise api_error(payload, error.code, error.headers.get("Retry-After")) from None
    except (URLError, OSError):
        raise ExtractionError("Network request failed; outcome may be unknown. No retry was sent. Preserve the request ID and body for investigation.") from None
    if len(raw) > 1_048_576:
        raise ExtractionError("Response exceeds the local 1 MiB safety limit; no rows accepted. Outcome may be unknown.")
    try:
        payload = json.loads(raw)
    except (ValueError, UnicodeError):
        raise ExtractionError("Response is not JSON; no rows accepted. Outcome may be unknown.") from None
    if not isinstance(payload, dict) or payload.get("ok") is not True:
        raise api_error(payload, 200)
    if payload.get("truncated") is not False:
        raise ExtractionError("Result is incomplete or lacks a truncation flag; no rows accepted. Inspect the response cursor in the API playground before continuing.")
    rows = payload.get("data")
    if not isinstance(rows, list):
        raise ExtractionError("Expected a list of product records; no rows accepted.")
    for index, row in enumerate(rows):
        if not isinstance(row, dict) or any(not isinstance(row.get(field), str) or not row[field].strip() for field in ("title", "price")):
            raise ExtractionError(f"Record {index} needs nonempty title and price strings; no rows accepted.")
    # Preserve price text, including its currency. Do not guess a numeric currency conversion.
    return {
        "sourceUrl": source_url,
        "retrievedAt": datetime.now(timezone.utc).isoformat(),
        "operationKey": operation_key,
        "requestId": payload.get("requestId"),
        "rows": rows,
        "receipt": payload.get("meta"),
    }


def main():
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("url", help="A permitted public product-list page")
    parser.add_argument("--request-id", required=True, help="A new ID for each distinct operation; retain for an identical retry")
    args = parser.parse_args()
    try:
        result = extract(os.environ.get("SCRAIPE_API_URL", ""), os.environ.get("SCRAIPE_API_KEY", ""), args.url, args.request_id)
    except (ExtractionError, ValueError) as error:
        print(f"Extraction stopped: {error}", file=sys.stderr)
        return 1
    print(json.dumps(result, ensure_ascii=False, indent=2))
    return 0


if __name__ == "__main__":
    sys.exit(main())
