Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
16 changes: 16 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,22 @@ All notable changes to this project are documented here. Format follows
library with a stable API) reasonably can — a patch bump means "fixes", not
a promise that every flag and exit code is contractually frozen.

## [Unreleased]

### Fixed

- **Scraper API: `waitFor` is now sent as a JSON object.** Measured
2026-09-23 against `scraper.2captcha.com/tasks/sync`: the JSON-encoded
string this client sent (the form older docs described) is refused with
HTTP 422 "params.waitFor must be an object" -- and the task is still
billed ($0.0005) -- so every run with `--wait-text` or its sibling wait flags
failed with exit 5. The object form answers HTTP 200.
- **Scraper API: the target page's status is read from `http_code`.** The
response's `status` field is the API's own verdict string (`"success"`),
not the target site's HTTP code, so a target 403/503 was never seen by
this client. `http_code` (an int) is read now, with `status` kept as a
fallback only when it is an int.

## [0.5.1] — 2026-09-16

### Fixed
Expand Down
56 changes: 42 additions & 14 deletions scraper_api_client.py
Original file line number Diff line number Diff line change
Expand Up @@ -41,14 +41,24 @@
Content-Type: application/json
{"task_type": "scrape", "url": ..., "data_format": "raw",
"format": "json", "timeout": 1..120,
"waitFor": "<JSON *string*, not an object>",
"waitFor": {"text": "..."}, # an OBJECT -- see below
"cdpurl": "ws://user:pass@host:port" # optional
}
-> 200 {"status": 200, "headers": {...}, "body": "<!DOCTYPE html>..."}
-> 200 {"status": "success", "http_code": 200, "headers": {...},
"body": "<!DOCTYPE html>..."}

Note `waitFor` must be a JSON *string* (double-encoded), and the param is
spelled `cdpurl` (all lowercase) while `waitFor` is camelCase — that's
the API's own inconsistency, not a typo here.
Measured 2026-09-23 against the live endpoint: `waitFor` must be a JSON
OBJECT. Sent as a JSON-encoded string (which this client did until then,
following older docs) the API answers HTTP 422 "params.waitFor must be an
object" -- and still bills the task ($0.0005). The same request with an
object answers 200.

`status` in the response is the API's own verdict string ("success"), NOT
the target page's HTTP code; that is `http_code`. Reading `status` as the
page status meant a target 403/503 was never seen.

The param is spelled `cdpurl` (all lowercase) while `waitFor` is camelCase
— that's the API's own inconsistency, not a typo here.

Usage
-----
Expand Down Expand Up @@ -150,21 +160,39 @@ def _redact_debug_header(value: str) -> str:
_CREDS_IN_TEXT_RE.sub(r"\1***:***@", value))


def _build_wait_for(args) -> Optional[str]:
"""`waitFor` must be a JSON STRING (double-encoded), per the API docs.
Passing a nested object is silently wrong.
def _build_wait_for(args) -> Optional[dict]:
"""`waitFor` is sent as a JSON OBJECT. Measured 2026-09-23: the
JSON-encoded-string form this client used before is refused with HTTP
422 ("params.waitFor must be an object") and still billed; the object
form answers 200.

Default (no flag): wait for the DOM. On a challenge-protected page
that resolves instantly against the challenge page itself — which is
exactly the trap documented in this module's docstring, so
--wait-text/--wait-element exist to wait on something only the real
page can contain."""
if args.wait_text:
return json.dumps({"text": args.wait_text})
return {"text": args.wait_text}
if args.wait_element:
return json.dumps({"element": args.wait_element, "checkVisible": True})
return {"element": args.wait_element, "checkVisible": True}
if args.wait_state:
return json.dumps({"state": args.wait_state})
return {"state": args.wait_state}
return None


def _target_status(body: dict) -> Optional[int]:
"""The TARGET page's HTTP status, or None.

Measured 2026-09-23: the response carries the target's code as
`http_code` (an int) and the API's own verdict as `status` ("success").
Reading `status` handed the string "success" onward and a target 403
was never seen. `status` is kept as a fallback only when it is an int,
for the older response shape. bool is excluded: it is an int to Python.
"""
for key in ("http_code", "status"):
v = body.get(key)
if isinstance(v, int) and not isinstance(v, bool):
return v
return None


Expand All @@ -173,14 +201,14 @@ def fetch_html(args) -> str:
"task_type": "scrape",
"url": args.url,
"data_format": "raw", # we want HTML; product_parser does the rest
"format": "json", # so we get {"status", "headers", "body"}
"format": "json", # so we get {"status", "http_code", "headers", "body"}
"timeout": min(args.timeout, MAX_API_TIMEOUT),
}

wait_for = _build_wait_for(args)
if wait_for:
payload["waitFor"] = wait_for
logger.info("waitFor: %s", wait_for)
logger.info("waitFor: %s", json.dumps(wait_for))

if args.cdp_url:
payload["cdpurl"] = args.cdp_url
Expand Down Expand Up @@ -215,7 +243,7 @@ def fetch_html(args) -> str:

body = resp.json()
html = body.get("body") or ""
upstream_status = body.get("status")
upstream_status = _target_status(body)
logger.info("Upstream page status %s, %d bytes of HTML.", upstream_status, len(html))
return html

Expand Down
53 changes: 53 additions & 0 deletions smoke_test.py
Original file line number Diff line number Diff line change
Expand Up @@ -3965,6 +3965,58 @@ def check_x_debug_header_is_redacted(ok):
return ok


def check_scraper_api_waitfor_object_and_http_code(ok):
"""Measured 2026-09-23 against the live Scraper API: `waitFor` sent as
a JSON-encoded string is refused with HTTP 422 and still billed, and
the response's `status` is the API's verdict ("success") while the
target's own code is `http_code`. Drives the real fetch_html with
requests.post replaced, so no network and no money."""
import argparse
import logging
import scraper_api_client as sac
captured = {}

class _Resp:
status_code = 200
headers = {}
text = ""

def json(self):
return {"status": "success", "http_code": 403, "body": "<html></html>"}

def _fake_post(url, **kw):
captured["json"] = kw.get("json")
return _Resp()

statuses = []

class _H(logging.Handler):
def emit(self, record):
if str(record.msg).startswith("Upstream page status"):
statuses.append(record.args[0])

h = _H()
real_post = sac.requests.post
sac.requests.post = _fake_post
sac.logger.addHandler(h)
try:
args = argparse.Namespace(url="https://www.farfetch.com/shopping/kids/items.aspx",
key="k", timeout=60, cdp_url=None, wait_text="$",
wait_element=None, wait_state=None)
sac.fetch_html(args)
finally:
sac.requests.post = real_post
sac.logger.removeHandler(h)
wf = (captured.get("json") or {}).get("waitFor")
ok &= check("Scraper API: --wait-text sends waitFor as an OBJECT, not a JSON string "
"(a string is HTTP 422 and still billed, measured 2026-09-23)",
isinstance(wf, dict) and wf.get("text") == "$")
ok &= check("Scraper API: the target status is read from http_code (403), not the "
"API's own 'success' verdict",
statuses == [403])
return ok


def main() -> int:
ok = True

Expand Down Expand Up @@ -4155,6 +4207,7 @@ def main() -> int:
ok = check_sign_up_modal_selectors(ok)
ok = check_empty_result_contract(ok)
ok = check_x_debug_header_is_redacted(ok)
ok = check_scraper_api_waitfor_object_and_http_code(ok)
ok = check_blocked_vs_empty_exit_code(ok)
ok = check_akamai_s_refusal_page_the_2026_09_11_audit_s_p0(ok)
ok = check_page_content_mid_navigation(ok)
Expand Down
Loading