From c0a6112b56c4d0270f843acf4e89a72e8760d124 Mon Sep 17 00:00:00 2001 From: Petr Date: Wed, 23 Sep 2026 15:11:21 +0000 Subject: [PATCH] Scraper API: send waitFor as an object, read the target status from http_code Measured 2026-09-23 against scraper.2captcha.com/tasks/sync: waitFor sent as a JSON-encoded string is refused with HTTP 422 ("params.waitFor must be an object") and still billed; as an object it answers 200. The response's status field is the API's verdict ("success"); the target's HTTP code is http_code. Adds a no-network regression check, controlled red against the unfixed client. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 16 +++++++++++++ scraper_api_client.py | 56 ++++++++++++++++++++++++++++++++----------- smoke_test.py | 53 ++++++++++++++++++++++++++++++++++++++++ 3 files changed, 111 insertions(+), 14 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index b0ece5f..cea71c4 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,22 @@ All notable changes to this project are documented here. Format follows library with a stable API) reasonably can — a patch bump means "fixes", not a promise that every flag and exit code is contractually frozen. +## [Unreleased] + +### Fixed + +- **Scraper API: `waitFor` is now sent as a JSON object.** Measured + 2026-09-23 against `scraper.2captcha.com/tasks/sync`: the JSON-encoded + string this client sent (the form older docs described) is refused with + HTTP 422 "params.waitFor must be an object" -- and the task is still + billed ($0.0005) -- so every run with `--wait-text` or its sibling wait flags + failed with exit 5. The object form answers HTTP 200. +- **Scraper API: the target page's status is read from `http_code`.** The + response's `status` field is the API's own verdict string (`"success"`), + not the target site's HTTP code, so a target 403/503 was never seen by + this client. `http_code` (an int) is read now, with `status` kept as a + fallback only when it is an int. + ## [0.5.1] — 2026-09-16 ### Fixed diff --git a/scraper_api_client.py b/scraper_api_client.py index d51c698..73da693 100644 --- a/scraper_api_client.py +++ b/scraper_api_client.py @@ -41,14 +41,24 @@ Content-Type: application/json {"task_type": "scrape", "url": ..., "data_format": "raw", "format": "json", "timeout": 1..120, - "waitFor": "", + "waitFor": {"text": "..."}, # an OBJECT -- see below "cdpurl": "ws://user:pass@host:port" # optional } - -> 200 {"status": 200, "headers": {...}, "body": "..."} + -> 200 {"status": "success", "http_code": 200, "headers": {...}, + "body": "..."} -Note `waitFor` must be a JSON *string* (double-encoded), and the param is -spelled `cdpurl` (all lowercase) while `waitFor` is camelCase — that's -the API's own inconsistency, not a typo here. +Measured 2026-09-23 against the live endpoint: `waitFor` must be a JSON +OBJECT. Sent as a JSON-encoded string (which this client did until then, +following older docs) the API answers HTTP 422 "params.waitFor must be an +object" -- and still bills the task ($0.0005). The same request with an +object answers 200. + +`status` in the response is the API's own verdict string ("success"), NOT +the target page's HTTP code; that is `http_code`. Reading `status` as the +page status meant a target 403/503 was never seen. + +The param is spelled `cdpurl` (all lowercase) while `waitFor` is camelCase +— that's the API's own inconsistency, not a typo here. Usage ----- @@ -150,9 +160,11 @@ def _redact_debug_header(value: str) -> str: _CREDS_IN_TEXT_RE.sub(r"\1***:***@", value)) -def _build_wait_for(args) -> Optional[str]: - """`waitFor` must be a JSON STRING (double-encoded), per the API docs. - Passing a nested object is silently wrong. +def _build_wait_for(args) -> Optional[dict]: + """`waitFor` is sent as a JSON OBJECT. Measured 2026-09-23: the + JSON-encoded-string form this client used before is refused with HTTP + 422 ("params.waitFor must be an object") and still billed; the object + form answers 200. Default (no flag): wait for the DOM. On a challenge-protected page that resolves instantly against the challenge page itself — which is @@ -160,11 +172,27 @@ def _build_wait_for(args) -> Optional[str]: --wait-text/--wait-element exist to wait on something only the real page can contain.""" if args.wait_text: - return json.dumps({"text": args.wait_text}) + return {"text": args.wait_text} if args.wait_element: - return json.dumps({"element": args.wait_element, "checkVisible": True}) + return {"element": args.wait_element, "checkVisible": True} if args.wait_state: - return json.dumps({"state": args.wait_state}) + return {"state": args.wait_state} + return None + + +def _target_status(body: dict) -> Optional[int]: + """The TARGET page's HTTP status, or None. + + Measured 2026-09-23: the response carries the target's code as + `http_code` (an int) and the API's own verdict as `status` ("success"). + Reading `status` handed the string "success" onward and a target 403 + was never seen. `status` is kept as a fallback only when it is an int, + for the older response shape. bool is excluded: it is an int to Python. + """ + for key in ("http_code", "status"): + v = body.get(key) + if isinstance(v, int) and not isinstance(v, bool): + return v return None @@ -173,14 +201,14 @@ def fetch_html(args) -> str: "task_type": "scrape", "url": args.url, "data_format": "raw", # we want HTML; product_parser does the rest - "format": "json", # so we get {"status", "headers", "body"} + "format": "json", # so we get {"status", "http_code", "headers", "body"} "timeout": min(args.timeout, MAX_API_TIMEOUT), } wait_for = _build_wait_for(args) if wait_for: payload["waitFor"] = wait_for - logger.info("waitFor: %s", wait_for) + logger.info("waitFor: %s", json.dumps(wait_for)) if args.cdp_url: payload["cdpurl"] = args.cdp_url @@ -215,7 +243,7 @@ def fetch_html(args) -> str: body = resp.json() html = body.get("body") or "" - upstream_status = body.get("status") + upstream_status = _target_status(body) logger.info("Upstream page status %s, %d bytes of HTML.", upstream_status, len(html)) return html diff --git a/smoke_test.py b/smoke_test.py index 979a539..33d03a0 100644 --- a/smoke_test.py +++ b/smoke_test.py @@ -3965,6 +3965,58 @@ def check_x_debug_header_is_redacted(ok): return ok +def check_scraper_api_waitfor_object_and_http_code(ok): + """Measured 2026-09-23 against the live Scraper API: `waitFor` sent as + a JSON-encoded string is refused with HTTP 422 and still billed, and + the response's `status` is the API's verdict ("success") while the + target's own code is `http_code`. Drives the real fetch_html with + requests.post replaced, so no network and no money.""" + import argparse + import logging + import scraper_api_client as sac + captured = {} + + class _Resp: + status_code = 200 + headers = {} + text = "" + + def json(self): + return {"status": "success", "http_code": 403, "body": ""} + + def _fake_post(url, **kw): + captured["json"] = kw.get("json") + return _Resp() + + statuses = [] + + class _H(logging.Handler): + def emit(self, record): + if str(record.msg).startswith("Upstream page status"): + statuses.append(record.args[0]) + + h = _H() + real_post = sac.requests.post + sac.requests.post = _fake_post + sac.logger.addHandler(h) + try: + args = argparse.Namespace(url="https://www.farfetch.com/shopping/kids/items.aspx", + key="k", timeout=60, cdp_url=None, wait_text="$", + wait_element=None, wait_state=None) + sac.fetch_html(args) + finally: + sac.requests.post = real_post + sac.logger.removeHandler(h) + wf = (captured.get("json") or {}).get("waitFor") + ok &= check("Scraper API: --wait-text sends waitFor as an OBJECT, not a JSON string " + "(a string is HTTP 422 and still billed, measured 2026-09-23)", + isinstance(wf, dict) and wf.get("text") == "$") + ok &= check("Scraper API: the target status is read from http_code (403), not the " + "API's own 'success' verdict", + statuses == [403]) + return ok + + def main() -> int: ok = True @@ -4155,6 +4207,7 @@ def main() -> int: ok = check_sign_up_modal_selectors(ok) ok = check_empty_result_contract(ok) ok = check_x_debug_header_is_redacted(ok) + ok = check_scraper_api_waitfor_object_and_http_code(ok) ok = check_blocked_vs_empty_exit_code(ok) ok = check_akamai_s_refusal_page_the_2026_09_11_audit_s_p0(ok) ok = check_page_content_mid_navigation(ok)