diff --git a/.agents/skills/scrapingbee-cli/reference/amazon/product.md b/.agents/skills/scrapingbee-cli/reference/amazon/product.md index 002512e..0f67db0 100644 --- a/.agents/skills/scrapingbee-cli/reference/amazon/product.md +++ b/.agents/skills/scrapingbee-cli/reference/amazon/product.md @@ -21,6 +21,7 @@ scrapingbee amazon-product --output-file product.json B0DPDRNSXV --domain com | `--language` | string | e.g. en_US, es_US, fr_FR. | | `--currency` | string | USD, EUR, GBP, etc. | | `--add-html` | true/false | Include full HTML. | +| `--autoselect-variant` | true/false | Auto-select the default/most-popular variant (undocumented API param, verified accepted). | | `--light-request` | true/false | Light request. | | `--screenshot` | true/false | Take screenshot. | | `--tag` | string | Optional label included in API response headers. | diff --git a/.agents/skills/scrapingbee-cli/reference/google/overview.md b/.agents/skills/scrapingbee-cli/reference/google/overview.md index b371703..b6224f6 100644 --- a/.agents/skills/scrapingbee-cli/reference/google/overview.md +++ b/.agents/skills/scrapingbee-cli/reference/google/overview.md @@ -18,7 +18,8 @@ scrapingbee google --output-file serp.json "pizza new york" --country-code us | `--country-code` | string | ISO 3166-1 (e.g. us, gb, de). | | `--device` | string | `desktop` or `mobile`. | | `--page` | int | Page number (default 1). | -| `--pages` | int | Consecutive pages to fetch starting at `--page` (default 1, max 10; 3 or fewer recommended). Combined into one response; cost is per fetched page. | +| `--pages` | int | Consecutive pages to fetch starting at `--page` (default 1, max 10; 3 or fewer recommended). Combined into one response; cost is flat per request (10 light / 15 rendered) regardless of page count. | +| `--nb-results` | int | Requested results per page (undocumented API param, verified accepted; Google may return more or fewer). | | `--language` | string | Language code (e.g. en, fr, de). | | `--date-range` | string | `past-hour`, `past-day`, `past-week`, `past-month`, `past-year`. Restrict results by recency. | | `--nfpr` | true/false | Disable autocorrection. | diff --git a/.agents/skills/scrapingbee-cli/reference/scrape/options.md b/.agents/skills/scrapingbee-cli/reference/scrape/options.md index 98f2800..37ee290 100644 --- a/.agents/skills/scrapingbee-cli/reference/scrape/options.md +++ b/.agents/skills/scrapingbee-cli/reference/scrape/options.md @@ -71,7 +71,7 @@ Blocked? See [reference/proxy/strategies.md](reference/proxy/strategies.md). |-----------|------|-------------| | `--device` | desktop \| mobile | Device type (CLI validates). | | `--timeout` | int | Timeout ms (1000–140000). Scrape job timeout on ScrapingBee. The CLI sets the HTTP client (aiohttp) timeout to this value in seconds plus 30 s (for send/receive) so the client does not give up before the API responds. | -| `--custom-google` / `--transparent-status-code` | — | Google (15 credits), target status. | +| `--custom-google` / `--transparent-status-code` | — | Google (20 credits), target status. | | `--tag` | string | Optional label included in API response headers. | | `--mode` | auto | Auto-Mode: API picks the cheapest config that succeeds; charged only for the winning config. GET only. See [Auto-Mode](#auto-mode). | | `--max-cost` | int | Cap credits a request may cost (≥ 1). Requires `--mode auto`; omit = uncapped. | diff --git a/.agents/skills/scrapingbee-cli/reference/youtube/subtitles.md b/.agents/skills/scrapingbee-cli/reference/youtube/subtitles.md index 5b52fb7..985f28c 100644 --- a/.agents/skills/scrapingbee-cli/reference/youtube/subtitles.md +++ b/.agents/skills/scrapingbee-cli/reference/youtube/subtitles.md @@ -14,7 +14,7 @@ scrapingbee youtube-subtitles --output-file subtitles.json dQw4w9WgXcQ | Flag | Values | Notes | |------|--------|-------| -| `--language` | ISO language code (`en`, `fr`, ...) | A language with no matching subtitles returns 404. | +| `--language` | ISO language code (`en`, `fr`, ...) | A language with no matching subtitles returns HTTP 200 with an empty `subtitles` object (still 5 credits); the CLI prints a warning. | | `--subtitle-origin` | `auto-generated`, `uploader-provided` | Filter by subtitle source. | Plus global flags (`--output-file`, `--verbose`, `--output-dir`, `--concurrency`, `--retries`, `--backoff`). diff --git a/.github/skills/scrapingbee-cli/reference/amazon/product.md b/.github/skills/scrapingbee-cli/reference/amazon/product.md index 002512e..0f67db0 100644 --- a/.github/skills/scrapingbee-cli/reference/amazon/product.md +++ b/.github/skills/scrapingbee-cli/reference/amazon/product.md @@ -21,6 +21,7 @@ scrapingbee amazon-product --output-file product.json B0DPDRNSXV --domain com | `--language` | string | e.g. en_US, es_US, fr_FR. | | `--currency` | string | USD, EUR, GBP, etc. | | `--add-html` | true/false | Include full HTML. | +| `--autoselect-variant` | true/false | Auto-select the default/most-popular variant (undocumented API param, verified accepted). | | `--light-request` | true/false | Light request. | | `--screenshot` | true/false | Take screenshot. | | `--tag` | string | Optional label included in API response headers. | diff --git a/.github/skills/scrapingbee-cli/reference/google/overview.md b/.github/skills/scrapingbee-cli/reference/google/overview.md index b371703..b6224f6 100644 --- a/.github/skills/scrapingbee-cli/reference/google/overview.md +++ b/.github/skills/scrapingbee-cli/reference/google/overview.md @@ -18,7 +18,8 @@ scrapingbee google --output-file serp.json "pizza new york" --country-code us | `--country-code` | string | ISO 3166-1 (e.g. us, gb, de). | | `--device` | string | `desktop` or `mobile`. | | `--page` | int | Page number (default 1). | -| `--pages` | int | Consecutive pages to fetch starting at `--page` (default 1, max 10; 3 or fewer recommended). Combined into one response; cost is per fetched page. | +| `--pages` | int | Consecutive pages to fetch starting at `--page` (default 1, max 10; 3 or fewer recommended). Combined into one response; cost is flat per request (10 light / 15 rendered) regardless of page count. | +| `--nb-results` | int | Requested results per page (undocumented API param, verified accepted; Google may return more or fewer). | | `--language` | string | Language code (e.g. en, fr, de). | | `--date-range` | string | `past-hour`, `past-day`, `past-week`, `past-month`, `past-year`. Restrict results by recency. | | `--nfpr` | true/false | Disable autocorrection. | diff --git a/.github/skills/scrapingbee-cli/reference/scrape/options.md b/.github/skills/scrapingbee-cli/reference/scrape/options.md index 98f2800..37ee290 100644 --- a/.github/skills/scrapingbee-cli/reference/scrape/options.md +++ b/.github/skills/scrapingbee-cli/reference/scrape/options.md @@ -71,7 +71,7 @@ Blocked? See [reference/proxy/strategies.md](reference/proxy/strategies.md). |-----------|------|-------------| | `--device` | desktop \| mobile | Device type (CLI validates). | | `--timeout` | int | Timeout ms (1000–140000). Scrape job timeout on ScrapingBee. The CLI sets the HTTP client (aiohttp) timeout to this value in seconds plus 30 s (for send/receive) so the client does not give up before the API responds. | -| `--custom-google` / `--transparent-status-code` | — | Google (15 credits), target status. | +| `--custom-google` / `--transparent-status-code` | — | Google (20 credits), target status. | | `--tag` | string | Optional label included in API response headers. | | `--mode` | auto | Auto-Mode: API picks the cheapest config that succeeds; charged only for the winning config. GET only. See [Auto-Mode](#auto-mode). | | `--max-cost` | int | Cap credits a request may cost (≥ 1). Requires `--mode auto`; omit = uncapped. | diff --git a/.github/skills/scrapingbee-cli/reference/youtube/subtitles.md b/.github/skills/scrapingbee-cli/reference/youtube/subtitles.md index 5b52fb7..985f28c 100644 --- a/.github/skills/scrapingbee-cli/reference/youtube/subtitles.md +++ b/.github/skills/scrapingbee-cli/reference/youtube/subtitles.md @@ -14,7 +14,7 @@ scrapingbee youtube-subtitles --output-file subtitles.json dQw4w9WgXcQ | Flag | Values | Notes | |------|--------|-------| -| `--language` | ISO language code (`en`, `fr`, ...) | A language with no matching subtitles returns 404. | +| `--language` | ISO language code (`en`, `fr`, ...) | A language with no matching subtitles returns HTTP 200 with an empty `subtitles` object (still 5 credits); the CLI prints a warning. | | `--subtitle-origin` | `auto-generated`, `uploader-provided` | Filter by subtitle source. | Plus global flags (`--output-file`, `--verbose`, `--output-dir`, `--concurrency`, `--retries`, `--backoff`). diff --git a/.kiro/skills/scrapingbee-cli/reference/amazon/product.md b/.kiro/skills/scrapingbee-cli/reference/amazon/product.md index 002512e..0f67db0 100644 --- a/.kiro/skills/scrapingbee-cli/reference/amazon/product.md +++ b/.kiro/skills/scrapingbee-cli/reference/amazon/product.md @@ -21,6 +21,7 @@ scrapingbee amazon-product --output-file product.json B0DPDRNSXV --domain com | `--language` | string | e.g. en_US, es_US, fr_FR. | | `--currency` | string | USD, EUR, GBP, etc. | | `--add-html` | true/false | Include full HTML. | +| `--autoselect-variant` | true/false | Auto-select the default/most-popular variant (undocumented API param, verified accepted). | | `--light-request` | true/false | Light request. | | `--screenshot` | true/false | Take screenshot. | | `--tag` | string | Optional label included in API response headers. | diff --git a/.kiro/skills/scrapingbee-cli/reference/google/overview.md b/.kiro/skills/scrapingbee-cli/reference/google/overview.md index b371703..b6224f6 100644 --- a/.kiro/skills/scrapingbee-cli/reference/google/overview.md +++ b/.kiro/skills/scrapingbee-cli/reference/google/overview.md @@ -18,7 +18,8 @@ scrapingbee google --output-file serp.json "pizza new york" --country-code us | `--country-code` | string | ISO 3166-1 (e.g. us, gb, de). | | `--device` | string | `desktop` or `mobile`. | | `--page` | int | Page number (default 1). | -| `--pages` | int | Consecutive pages to fetch starting at `--page` (default 1, max 10; 3 or fewer recommended). Combined into one response; cost is per fetched page. | +| `--pages` | int | Consecutive pages to fetch starting at `--page` (default 1, max 10; 3 or fewer recommended). Combined into one response; cost is flat per request (10 light / 15 rendered) regardless of page count. | +| `--nb-results` | int | Requested results per page (undocumented API param, verified accepted; Google may return more or fewer). | | `--language` | string | Language code (e.g. en, fr, de). | | `--date-range` | string | `past-hour`, `past-day`, `past-week`, `past-month`, `past-year`. Restrict results by recency. | | `--nfpr` | true/false | Disable autocorrection. | diff --git a/.kiro/skills/scrapingbee-cli/reference/scrape/options.md b/.kiro/skills/scrapingbee-cli/reference/scrape/options.md index 98f2800..37ee290 100644 --- a/.kiro/skills/scrapingbee-cli/reference/scrape/options.md +++ b/.kiro/skills/scrapingbee-cli/reference/scrape/options.md @@ -71,7 +71,7 @@ Blocked? See [reference/proxy/strategies.md](reference/proxy/strategies.md). |-----------|------|-------------| | `--device` | desktop \| mobile | Device type (CLI validates). | | `--timeout` | int | Timeout ms (1000–140000). Scrape job timeout on ScrapingBee. The CLI sets the HTTP client (aiohttp) timeout to this value in seconds plus 30 s (for send/receive) so the client does not give up before the API responds. | -| `--custom-google` / `--transparent-status-code` | — | Google (15 credits), target status. | +| `--custom-google` / `--transparent-status-code` | — | Google (20 credits), target status. | | `--tag` | string | Optional label included in API response headers. | | `--mode` | auto | Auto-Mode: API picks the cheapest config that succeeds; charged only for the winning config. GET only. See [Auto-Mode](#auto-mode). | | `--max-cost` | int | Cap credits a request may cost (≥ 1). Requires `--mode auto`; omit = uncapped. | diff --git a/.kiro/skills/scrapingbee-cli/reference/youtube/subtitles.md b/.kiro/skills/scrapingbee-cli/reference/youtube/subtitles.md index 5b52fb7..985f28c 100644 --- a/.kiro/skills/scrapingbee-cli/reference/youtube/subtitles.md +++ b/.kiro/skills/scrapingbee-cli/reference/youtube/subtitles.md @@ -14,7 +14,7 @@ scrapingbee youtube-subtitles --output-file subtitles.json dQw4w9WgXcQ | Flag | Values | Notes | |------|--------|-------| -| `--language` | ISO language code (`en`, `fr`, ...) | A language with no matching subtitles returns 404. | +| `--language` | ISO language code (`en`, `fr`, ...) | A language with no matching subtitles returns HTTP 200 with an empty `subtitles` object (still 5 credits); the CLI prints a warning. | | `--subtitle-origin` | `auto-generated`, `uploader-provided` | Filter by subtitle source. | Plus global flags (`--output-file`, `--verbose`, `--output-dir`, `--concurrency`, `--retries`, `--backoff`). diff --git a/.opencode/skills/scrapingbee-cli/reference/amazon/product.md b/.opencode/skills/scrapingbee-cli/reference/amazon/product.md index 002512e..0f67db0 100644 --- a/.opencode/skills/scrapingbee-cli/reference/amazon/product.md +++ b/.opencode/skills/scrapingbee-cli/reference/amazon/product.md @@ -21,6 +21,7 @@ scrapingbee amazon-product --output-file product.json B0DPDRNSXV --domain com | `--language` | string | e.g. en_US, es_US, fr_FR. | | `--currency` | string | USD, EUR, GBP, etc. | | `--add-html` | true/false | Include full HTML. | +| `--autoselect-variant` | true/false | Auto-select the default/most-popular variant (undocumented API param, verified accepted). | | `--light-request` | true/false | Light request. | | `--screenshot` | true/false | Take screenshot. | | `--tag` | string | Optional label included in API response headers. | diff --git a/.opencode/skills/scrapingbee-cli/reference/google/overview.md b/.opencode/skills/scrapingbee-cli/reference/google/overview.md index b371703..b6224f6 100644 --- a/.opencode/skills/scrapingbee-cli/reference/google/overview.md +++ b/.opencode/skills/scrapingbee-cli/reference/google/overview.md @@ -18,7 +18,8 @@ scrapingbee google --output-file serp.json "pizza new york" --country-code us | `--country-code` | string | ISO 3166-1 (e.g. us, gb, de). | | `--device` | string | `desktop` or `mobile`. | | `--page` | int | Page number (default 1). | -| `--pages` | int | Consecutive pages to fetch starting at `--page` (default 1, max 10; 3 or fewer recommended). Combined into one response; cost is per fetched page. | +| `--pages` | int | Consecutive pages to fetch starting at `--page` (default 1, max 10; 3 or fewer recommended). Combined into one response; cost is flat per request (10 light / 15 rendered) regardless of page count. | +| `--nb-results` | int | Requested results per page (undocumented API param, verified accepted; Google may return more or fewer). | | `--language` | string | Language code (e.g. en, fr, de). | | `--date-range` | string | `past-hour`, `past-day`, `past-week`, `past-month`, `past-year`. Restrict results by recency. | | `--nfpr` | true/false | Disable autocorrection. | diff --git a/.opencode/skills/scrapingbee-cli/reference/scrape/options.md b/.opencode/skills/scrapingbee-cli/reference/scrape/options.md index 98f2800..37ee290 100644 --- a/.opencode/skills/scrapingbee-cli/reference/scrape/options.md +++ b/.opencode/skills/scrapingbee-cli/reference/scrape/options.md @@ -71,7 +71,7 @@ Blocked? See [reference/proxy/strategies.md](reference/proxy/strategies.md). |-----------|------|-------------| | `--device` | desktop \| mobile | Device type (CLI validates). | | `--timeout` | int | Timeout ms (1000–140000). Scrape job timeout on ScrapingBee. The CLI sets the HTTP client (aiohttp) timeout to this value in seconds plus 30 s (for send/receive) so the client does not give up before the API responds. | -| `--custom-google` / `--transparent-status-code` | — | Google (15 credits), target status. | +| `--custom-google` / `--transparent-status-code` | — | Google (20 credits), target status. | | `--tag` | string | Optional label included in API response headers. | | `--mode` | auto | Auto-Mode: API picks the cheapest config that succeeds; charged only for the winning config. GET only. See [Auto-Mode](#auto-mode). | | `--max-cost` | int | Cap credits a request may cost (≥ 1). Requires `--mode auto`; omit = uncapped. | diff --git a/.opencode/skills/scrapingbee-cli/reference/youtube/subtitles.md b/.opencode/skills/scrapingbee-cli/reference/youtube/subtitles.md index 5b52fb7..985f28c 100644 --- a/.opencode/skills/scrapingbee-cli/reference/youtube/subtitles.md +++ b/.opencode/skills/scrapingbee-cli/reference/youtube/subtitles.md @@ -14,7 +14,7 @@ scrapingbee youtube-subtitles --output-file subtitles.json dQw4w9WgXcQ | Flag | Values | Notes | |------|--------|-------| -| `--language` | ISO language code (`en`, `fr`, ...) | A language with no matching subtitles returns 404. | +| `--language` | ISO language code (`en`, `fr`, ...) | A language with no matching subtitles returns HTTP 200 with an empty `subtitles` object (still 5 credits); the CLI prints a warning. | | `--subtitle-origin` | `auto-generated`, `uploader-provided` | Filter by subtitle source. | Plus global flags (`--output-file`, `--verbose`, `--output-dir`, `--concurrency`, `--retries`, `--backoff`). diff --git a/CHANGELOG.md b/CHANGELOG.md index 0897119..97e283a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,7 +5,7 @@ All notable changes to this project are documented in this file. The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). -## [1.6.0] - TBD +## [1.6.0] - 2026-08-24 ### Added @@ -13,8 +13,19 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - **`--max-cost` on `scrape`** — cap the credits a request may cost (integer ≥ 1). Requires `--mode auto`; omit for an uncapped budget. Forwarded to the API as `max_cost` when set, omitted otherwise. - The verbose output (`-v`) now surfaces the `Spb-auto-cost` response header as `Auto Credit Cost` (the credits actually charged for the winning Auto-Mode config), alongside the existing `Credit Cost`. - **`youtube-subtitles` command** — fetch video captions/transcripts from the YouTube Subtitles API (5 credits per request). Accepts a video ID or full YouTube URL, `--language` (ISO code) and `--subtitle-origin` (`auto-generated` / `uploader-provided`), and supports batch via `--input-file` like the other YouTube commands. -- **`--pages` on `google`** — fetch up to 10 consecutive result pages starting at `--page` in a single combined response (3 or fewer recommended; cost is per fetched page). +- **`--pages` on `google`** — fetch up to 10 consecutive result pages starting at `--page` in a single combined response (3 or fewer recommended). Cost is flat per request — 10 credits light / 15 rendered — regardless of page count. - **`--search-type ads` on `google`** — classic-result structure optimized for paid-ad visibility. +- **`--nb-results` on `google`** — requested number of results per page. Undocumented API parameter, verified accepted by the API (Google may return more or fewer results than requested). +- **`--autoselect-variant` on `amazon-product`** — auto-select the default/most-popular product variant, matching the existing `amazon-search` flag. Undocumented API parameter, verified accepted by the API. + +### Changed + +- **Header-based authorization** — all API requests now authenticate via the `Authorization: Bearer` header instead of the deprecated `api_key` query parameter, so the key no longer appears in request URLs (or anything that logs them). `crawl` is the one exception: its Scrapy middleware (`scrapy-scrapingbee`) still builds `api_key` URLs and will migrate separately. + +### Fixed + +- **`-H` headers were silently dropped on POST/PUT** — custom headers are now `Spb-`-prefixed on every method (idempotently), which is the only form the API forwards to the target. Previously the prefix was only added on GET, so POST/PUT headers never reached the target — and a user `Authorization` header could clobber the CLI's own API authentication. Already-prefixed headers are passed through unchanged, so `-H "Spb-X: 1"` no longer double-prefixes. +- **Empty subtitles warned about** — `youtube-subtitles` with a `--language`/`--subtitle-origin` that matches nothing returns HTTP 200 with an empty `subtitles` object (not 404) and still charges 5 credits; the CLI now prints a warning instead of silent empty JSON. ## [1.5.1] - 2026-07-20 diff --git a/plugins/scrapingbee-cli/skills/scrapingbee-cli/reference/amazon/product.md b/plugins/scrapingbee-cli/skills/scrapingbee-cli/reference/amazon/product.md index 002512e..0f67db0 100644 --- a/plugins/scrapingbee-cli/skills/scrapingbee-cli/reference/amazon/product.md +++ b/plugins/scrapingbee-cli/skills/scrapingbee-cli/reference/amazon/product.md @@ -21,6 +21,7 @@ scrapingbee amazon-product --output-file product.json B0DPDRNSXV --domain com | `--language` | string | e.g. en_US, es_US, fr_FR. | | `--currency` | string | USD, EUR, GBP, etc. | | `--add-html` | true/false | Include full HTML. | +| `--autoselect-variant` | true/false | Auto-select the default/most-popular variant (undocumented API param, verified accepted). | | `--light-request` | true/false | Light request. | | `--screenshot` | true/false | Take screenshot. | | `--tag` | string | Optional label included in API response headers. | diff --git a/plugins/scrapingbee-cli/skills/scrapingbee-cli/reference/google/overview.md b/plugins/scrapingbee-cli/skills/scrapingbee-cli/reference/google/overview.md index b371703..b6224f6 100644 --- a/plugins/scrapingbee-cli/skills/scrapingbee-cli/reference/google/overview.md +++ b/plugins/scrapingbee-cli/skills/scrapingbee-cli/reference/google/overview.md @@ -18,7 +18,8 @@ scrapingbee google --output-file serp.json "pizza new york" --country-code us | `--country-code` | string | ISO 3166-1 (e.g. us, gb, de). | | `--device` | string | `desktop` or `mobile`. | | `--page` | int | Page number (default 1). | -| `--pages` | int | Consecutive pages to fetch starting at `--page` (default 1, max 10; 3 or fewer recommended). Combined into one response; cost is per fetched page. | +| `--pages` | int | Consecutive pages to fetch starting at `--page` (default 1, max 10; 3 or fewer recommended). Combined into one response; cost is flat per request (10 light / 15 rendered) regardless of page count. | +| `--nb-results` | int | Requested results per page (undocumented API param, verified accepted; Google may return more or fewer). | | `--language` | string | Language code (e.g. en, fr, de). | | `--date-range` | string | `past-hour`, `past-day`, `past-week`, `past-month`, `past-year`. Restrict results by recency. | | `--nfpr` | true/false | Disable autocorrection. | diff --git a/plugins/scrapingbee-cli/skills/scrapingbee-cli/reference/scrape/options.md b/plugins/scrapingbee-cli/skills/scrapingbee-cli/reference/scrape/options.md index 98f2800..37ee290 100644 --- a/plugins/scrapingbee-cli/skills/scrapingbee-cli/reference/scrape/options.md +++ b/plugins/scrapingbee-cli/skills/scrapingbee-cli/reference/scrape/options.md @@ -71,7 +71,7 @@ Blocked? See [reference/proxy/strategies.md](reference/proxy/strategies.md). |-----------|------|-------------| | `--device` | desktop \| mobile | Device type (CLI validates). | | `--timeout` | int | Timeout ms (1000–140000). Scrape job timeout on ScrapingBee. The CLI sets the HTTP client (aiohttp) timeout to this value in seconds plus 30 s (for send/receive) so the client does not give up before the API responds. | -| `--custom-google` / `--transparent-status-code` | — | Google (15 credits), target status. | +| `--custom-google` / `--transparent-status-code` | — | Google (20 credits), target status. | | `--tag` | string | Optional label included in API response headers. | | `--mode` | auto | Auto-Mode: API picks the cheapest config that succeeds; charged only for the winning config. GET only. See [Auto-Mode](#auto-mode). | | `--max-cost` | int | Cap credits a request may cost (≥ 1). Requires `--mode auto`; omit = uncapped. | diff --git a/plugins/scrapingbee-cli/skills/scrapingbee-cli/reference/youtube/subtitles.md b/plugins/scrapingbee-cli/skills/scrapingbee-cli/reference/youtube/subtitles.md index 5b52fb7..985f28c 100644 --- a/plugins/scrapingbee-cli/skills/scrapingbee-cli/reference/youtube/subtitles.md +++ b/plugins/scrapingbee-cli/skills/scrapingbee-cli/reference/youtube/subtitles.md @@ -14,7 +14,7 @@ scrapingbee youtube-subtitles --output-file subtitles.json dQw4w9WgXcQ | Flag | Values | Notes | |------|--------|-------| -| `--language` | ISO language code (`en`, `fr`, ...) | A language with no matching subtitles returns 404. | +| `--language` | ISO language code (`en`, `fr`, ...) | A language with no matching subtitles returns HTTP 200 with an empty `subtitles` object (still 5 credits); the CLI prints a warning. | | `--subtitle-origin` | `auto-generated`, `uploader-provided` | Filter by subtitle source. | Plus global flags (`--output-file`, `--verbose`, `--output-dir`, `--concurrency`, `--retries`, `--backoff`). diff --git a/src/scrapingbee_cli/client.py b/src/scrapingbee_cli/client.py index 5233454..ea4c25a 100644 --- a/src/scrapingbee_cli/client.py +++ b/src/scrapingbee_cli/client.py @@ -43,10 +43,13 @@ async def __aenter__(self) -> Client: else aiohttp.TCPConnector(ssl=ssl_context) ) timeout = aiohttp.ClientTimeout(total=self.timeout) + # Header-based auth: the api_key query parameter is deprecated for + # new integrations, so authenticate every request via the session. + session_headers = user_agent_headers() | {"Authorization": f"Bearer {self.api_key}"} self._session = aiohttp.ClientSession( connector=connector, timeout=timeout, - headers=user_agent_headers(), + headers=session_headers, ) return self @@ -67,7 +70,6 @@ async def _get( headers: dict[str, str] | None = None, ) -> tuple[bytes, dict, int]: params = _clean_params(params) - params.setdefault("api_key", self.api_key) url = f"{self.base_url}{path}" if path else self.base_url session = self._ensure_session() req_kwargs: dict[str, Any] = {"params": params} @@ -114,7 +116,6 @@ async def _request( headers: dict[str, str] | None = None, ) -> tuple[bytes, dict, int]: params = _clean_params(params) - params.setdefault("api_key", self.api_key) url = f"{self.base_url}{path}" if path else self.base_url session = self._ensure_session() req_kwargs: dict[str, Any] = {"params": params} @@ -231,10 +232,15 @@ async def scrape( method_upper = (method or "GET").upper() req_headers: dict[str, str] | None = None if custom_headers: - if method_upper == "GET": - req_headers = {f"Spb-{k}": v for k, v in custom_headers.items()} - else: - req_headers = dict(custom_headers) + # Always Spb-prefix (idempotently): the API only forwards prefixed + # headers — it strips the prefix in both forward modes and drops + # raw custom headers on POST/PUT — and a raw user Authorization + # header would replace the session's Bearer key (per-request + # headers win over session headers in aiohttp). + req_headers = { + (k if k.lower().startswith("spb-") else f"Spb-{k}"): v + for k, v in custom_headers.items() + } last_error: BaseException | None = None last_result: tuple[bytes, dict, int] = (b"", {}, 500) for attempt in range(max(0, retries) + 1): @@ -243,22 +249,16 @@ async def scrape( body_out, out_headers, status = await self._get("", params, headers=req_headers) else: params_clean = _clean_params(params) - params_clean["api_key"] = self.api_key - # ScrapingBee API expects POST to it as application/x-www-form-urlencoded + # ScrapingBee API expects POST to it as application/x-www-form-urlencoded; + # the user's Content-Type reaches the target via Spb-Content-Type. content_type = "application/x-www-form-urlencoded; charset=utf-8" - # Don't send user's Content-Type to ScrapingBee; forward via params if needed - req_headers_send = ( - {k: v for k, v in req_headers.items() if k.lower() != "content-type"} - if req_headers - else None - ) body_out, out_headers, status = await self._request( method_upper, "", params_clean, data=body, content_type=content_type, - headers=req_headers_send, + headers=req_headers, ) last_result = (body_out, out_headers, status) if status < 500 or attempt >= max(0, retries): @@ -279,7 +279,7 @@ async def usage( ) -> tuple[bytes, dict, int]: return await self._get_with_retry( "/usage", - {"api_key": self.api_key}, + {}, retries=retries, backoff=backoff, ) @@ -292,6 +292,7 @@ async def google_search( device: str | None = None, page: int | None = None, pages: int | None = None, + nb_results: int | None = None, language: str | None = None, nfpr: bool | None = None, extra_params: str | None = None, @@ -315,6 +316,7 @@ async def google_search( "device": device, "page": page if page is not None else None, "pages": pages if pages is not None else None, + "nb_results": nb_results if nb_results is not None else None, "language": language, "nfpr": self._bool(nfpr), "extra_params": extra_params, @@ -369,6 +371,7 @@ async def amazon_product( zip_code: str | None = None, language: str | None = None, currency: str | None = None, + autoselect_variant: bool | None = None, add_html: bool | None = None, light_request: bool | None = None, screenshot: bool | None = None, @@ -384,6 +387,7 @@ async def amazon_product( "zip_code": zip_code, "language": language, "currency": currency, + "autoselect_variant": self._bool(autoselect_variant), "add_html": self._bool(add_html), "light_request": self._bool(light_request), "screenshot": self._bool(screenshot), diff --git a/src/scrapingbee_cli/commands/amazon.py b/src/scrapingbee_cli/commands/amazon.py index 00a5806..0835863 100644 --- a/src/scrapingbee_cli/commands/amazon.py +++ b/src/scrapingbee_cli/commands/amazon.py @@ -62,6 +62,15 @@ ) @optgroup.option("--currency", type=str, default=None, help="Currency code (e.g. USD, EUR, GBP).") @optgroup.group("Output", help="Response format options") +@optgroup.option( + "--autoselect-variant", + type=BOOL_STR, + default=None, + help=( + "Auto-select the default/most-popular product variant (true/false). " + "Undocumented API parameter — verified accepted by the API." + ), +) @optgroup.option( "--add-html", type=BOOL_STR, default=None, help="Include full HTML in response (true/false)." ) @@ -86,6 +95,7 @@ def amazon_product_cmd( zip_code: str | None, language: str | None, currency: str | None, + autoselect_variant: str | None, add_html: str | None, light_request: str | None, screenshot: str | None, @@ -135,6 +145,7 @@ async def api_call(client, a): zip_code=zip_code, language=language, currency=currency, + autoselect_variant=parse_bool(autoselect_variant), add_html=parse_bool(add_html), light_request=parse_bool(light_request), screenshot=parse_bool(screenshot), @@ -178,6 +189,7 @@ async def _single() -> None: zip_code=zip_code, language=language, currency=currency, + autoselect_variant=parse_bool(autoselect_variant), add_html=parse_bool(add_html), light_request=parse_bool(light_request), screenshot=parse_bool(screenshot), diff --git a/src/scrapingbee_cli/commands/google.py b/src/scrapingbee_cli/commands/google.py index 508d499..2ad1dee 100644 --- a/src/scrapingbee_cli/commands/google.py +++ b/src/scrapingbee_cli/commands/google.py @@ -93,6 +93,15 @@ def _warn_empty_organic(data: bytes, search_type: str | None) -> None: "3 or fewer recommended). Results are combined into one response." ), ) +@optgroup.option( + "--nb-results", + type=int, + default=None, + help=( + "Requested number of results per page (undocumented API parameter — " + "verified accepted by the API; Google may return more or fewer)." + ), +) @optgroup.option( "--language", type=str, @@ -178,6 +187,7 @@ def google_cmd( device: str | None, page: int | None, pages: int | None, + nb_results: int | None, language: str | None, nfpr: str | None, extra_params: str | None, @@ -206,6 +216,7 @@ def google_cmd( raise SystemExit(1) _validate_page(page) _validate_pages(pages) + _validate_page(nb_results, name="nb-results") _validate_price_range(min_price, max_price) _validate_geolocation(latitude, longitude, radius) @@ -239,6 +250,7 @@ async def api_call(client, q): device=device, page=page, pages=pages, + nb_results=nb_results, language=language, nfpr=parse_bool(nfpr), extra_params=extra_params, @@ -290,6 +302,7 @@ async def _single() -> None: device=device, page=page, pages=pages, + nb_results=nb_results, language=language, nfpr=parse_bool(nfpr), extra_params=extra_params, diff --git a/src/scrapingbee_cli/commands/scrape.py b/src/scrapingbee_cli/commands/scrape.py index c23e0e3..3b3259e 100644 --- a/src/scrapingbee_cli/commands/scrape.py +++ b/src/scrapingbee_cli/commands/scrape.py @@ -173,7 +173,7 @@ def _apply_chunking(url: str, data: bytes, chunk_size: int, chunk_overlap: int) "--forward-headers", type=BOOL_STR, default=None, - help="Forward custom headers to target (true/false). Use -H with Spb- prefix for GET.", + help="Forward custom headers to target (true/false). -H headers are prefixed automatically.", ) @optgroup.option( "--forward-headers-pure", diff --git a/src/scrapingbee_cli/commands/youtube.py b/src/scrapingbee_cli/commands/youtube.py index f550d17..6648fc9 100644 --- a/src/scrapingbee_cli/commands/youtube.py +++ b/src/scrapingbee_cli/commands/youtube.py @@ -421,6 +421,40 @@ async def _single() -> None: asyncio.run(_single()) +def _warn_empty_subtitles(data: bytes, language: str | None, subtitle_origin: str | None) -> None: + """Warn when the subtitles response is empty — the request still cost credits. + + A language with no matching subtitles returns HTTP 200 with an empty + ``subtitles`` object (not 404), so without this the user gets silent + empty JSON. + """ + import json as _json + + try: + obj = _json.loads(data) + except Exception: + return + if not isinstance(obj, dict): + return + subs = obj.get("subtitles") + if not isinstance(subs, dict): + return + if any(v for v in subs.values() if v): + return + filters = [ + f"--language {language}" if language else None, + f"--subtitle-origin {subtitle_origin}" if subtitle_origin else None, + ] + hint = " and ".join(f for f in filters if f) + click.echo( + "Warning: no subtitles found" + + (f" matching {hint}" if hint else " for this video") + + " — the request still used credits." + + (" Try omitting the filter(s) to see what languages exist." if hint else ""), + err=True, + ) + + @click.command("youtube-subtitles") @click.argument("video_id", required=False) @click.option( @@ -531,6 +565,7 @@ async def _single() -> None: backoff=float(obj.get("backoff") or 2.0), ) check_api_response(data, status_code) + _warn_empty_subtitles(data, language, norm_val(subtitle_origin)) write_output( data, headers, diff --git a/tests/integration/helpers.py b/tests/integration/helpers.py index 3fb5e86..dae566b 100644 --- a/tests/integration/helpers.py +++ b/tests/integration/helpers.py @@ -194,6 +194,8 @@ def build_api_matrix_tests( tests.append(("youtube-metadata", base + ["youtube-metadata", "dQw4w9WgXcQ"], api_timeout)) + tests.append(("youtube-subtitles", base + ["youtube-subtitles", "dQw4w9WgXcQ"], api_timeout)) + tests.append(("chatgpt", base + ["chatgpt", "Say hello"], chatgpt_timeout)) # Gemini is also an LLM source; reuse the longer chatgpt_timeout. diff --git a/tests/run_e2e_tests.py b/tests/run_e2e_tests.py index 2558341..561e423 100644 --- a/tests/run_e2e_tests.py +++ b/tests/run_e2e_tests.py @@ -1607,6 +1607,22 @@ def build_tests(fx: dict[str, str]) -> list[Test]: ), ] + # ── YB: youtube-subtitles ───────────────────────────────────────────────── + tests += [ + Test( + "YB-01", + "youtube-subtitles dQw4w9WgXcQ", + ["youtube-subtitles", "dQw4w9WgXcQ"], + json_key("subtitles"), + ), + Test( + "YB-02", + "youtube-subtitles --language en", + ["youtube-subtitles", "dQw4w9WgXcQ", "--language", "en"], + combined_checks(json_key("subtitles"), stdout_contains("strangers")), + ), + ] + # ── CG: chatgpt ─────────────────────────────────────────────────────────── tests += [ Test( diff --git a/tests/unit/test_cli.py b/tests/unit/test_cli.py index dffa99d..a844b09 100644 --- a/tests/unit/test_cli.py +++ b/tests/unit/test_cli.py @@ -23,6 +23,7 @@ YOUTUBE_UPLOAD_DATE, _extract_video_id, _normalize_youtube_search, + _warn_empty_subtitles, ) @@ -260,6 +261,20 @@ def test_google_pages_option(self): assert code == 0 assert "--pages" in out + def test_google_nb_results_option(self): + from tests.conftest import cli_run + + code, out, _ = cli_run(["google", "--help"]) + assert code == 0 + assert "--nb-results" in out + + def test_amazon_product_autoselect_variant_option(self): + from tests.conftest import cli_run + + code, out, _ = cli_run(["amazon-product", "--help"]) + assert code == 0 + assert "--autoselect-variant" in out + class TestExtractFieldValues: """Tests for _extract_field_values().""" @@ -447,6 +462,40 @@ def test_other_fields_preserved(self): assert d["search"] == "rick" +class TestWarnEmptySubtitles: + """A missing language returns HTTP 200 with empty subtitles (not 404) and + still charges credits — the CLI must warn instead of printing silent + empty JSON. Verified live 2026-09-04: --language fr on dQw4w9WgXcQ -> + 200, {"subtitles":{}}, 5 credits.""" + + def test_warns_on_fully_empty_subtitles(self, capsys): + _warn_empty_subtitles(b'{"subtitles": {}}', "fr", None) + err = capsys.readouterr().err + assert "no subtitles found" in err + assert "--language fr" in err + + def test_warns_on_empty_origin_buckets(self, capsys): + _warn_empty_subtitles( + b'{"subtitles": {"auto_generated": {}, "uploader_provided": {}}}', + None, + "uploader_provided", + ) + err = capsys.readouterr().err + assert "no subtitles found" in err + assert "--subtitle-origin uploader_provided" in err + + def test_no_warning_when_subtitles_present(self, capsys): + _warn_empty_subtitles( + b'{"subtitles": {"auto_generated": {"en": [{"start_ms": "0"}]}}}', "en", None + ) + assert capsys.readouterr().err == "" + + def test_silent_on_non_json_or_unexpected_shape(self, capsys): + _warn_empty_subtitles(b"not json", None, None) + _warn_empty_subtitles(b'{"other": 1}', None, None) + assert capsys.readouterr().err == "" + + class TestYouTubeDurationAlias: """Tests for shell-safe duration aliases (short/medium/long).""" diff --git a/tests/unit/test_client.py b/tests/unit/test_client.py index c8966bf..2aac0fc 100644 --- a/tests/unit/test_client.py +++ b/tests/unit/test_client.py @@ -378,6 +378,134 @@ async def fake_get(path, params, headers=None): asyncio.run(run()) +class TestHeaderAuth: + """The client authenticates via Authorization: Bearer, not the deprecated api_key param.""" + + def test_session_has_bearer_authorization(self): + async def run(): + async with Client("fake-key") as client: + assert client._ensure_session().headers.get("Authorization") == "Bearer fake-key" + + asyncio.run(run()) + + def test_get_sends_no_api_key_param(self): + async def run(): + client = Client("fake-key") + captured: dict = {} + + async def fake_get(path, params, headers=None): + captured["params"] = _clean_params(params) + return (b"{}", {}, 200) + + with patch.object(client, "_get", new=AsyncMock(side_effect=fake_get)): + await client.scrape("https://example.com", retries=0) + assert "api_key" not in captured["params"] + + asyncio.run(run()) + + def test_post_sends_no_api_key_param(self): + async def run(): + client = Client("fake-key") + captured: dict = {} + + async def fake_request( + method, path, params, data=None, content_type=None, headers=None + ): + captured["params"] = dict(params) + return (b"{}", {}, 200) + + with patch.object(client, "_request", new=AsyncMock(side_effect=fake_request)): + await client.scrape("https://example.com", method="post", body="x=1", retries=0) + assert "api_key" not in captured["params"] + + asyncio.run(run()) + + def test_usage_sends_no_api_key_param(self): + async def run(): + client = Client("fake-key") + captured: dict = {} + + async def fake_get(path, params, headers=None): + captured["params"] = _clean_params(params) + return (b"{}", {}, 200) + + with patch.object(client, "_get", new=AsyncMock(side_effect=fake_get)): + await client.usage(retries=0) + assert "api_key" not in captured["params"] + + asyncio.run(run()) + + +class TestGoogleNbResults: + """Tests that google_search forwards nb_results only when set.""" + + def test_nb_results_sent_when_set(self): + async def run(): + client = Client("fake-key") + captured: dict = {} + + async def fake_get(path, params, headers=None): + captured["params"] = _clean_params(params) + return (b"{}", {}, 200) + + with patch.object(client, "_get", new=AsyncMock(side_effect=fake_get)): + await client.google_search("coffee", nb_results=3, retries=0) + assert captured["params"].get("nb_results") == 3 + + asyncio.run(run()) + + def test_nb_results_omitted_when_unset(self): + async def run(): + client = Client("fake-key") + captured: dict = {} + + async def fake_get(path, params, headers=None): + captured["params"] = _clean_params(params) + return (b"{}", {}, 200) + + with patch.object(client, "_get", new=AsyncMock(side_effect=fake_get)): + await client.google_search("coffee", retries=0) + assert "nb_results" not in captured["params"] + + asyncio.run(run()) + + +class TestAmazonProductAutoselectVariant: + """Tests that amazon_product forwards autoselect_variant only when set.""" + + def test_sent_when_set(self): + async def run(): + client = Client("fake-key") + captured: dict = {} + + async def fake_get(path, params, headers=None): + captured["path"] = path + captured["params"] = _clean_params(params) + return (b"{}", {}, 200) + + with patch.object(client, "_get", new=AsyncMock(side_effect=fake_get)): + await client.amazon_product("B000000000", autoselect_variant=True, retries=0) + assert captured["path"] == "/amazon/product" + assert captured["params"].get("autoselect_variant") == "true" + + asyncio.run(run()) + + def test_omitted_when_unset(self): + async def run(): + client = Client("fake-key") + captured: dict = {} + + async def fake_get(path, params, headers=None): + captured["params"] = _clean_params(params) + return (b"{}", {}, 200) + + with patch.object(client, "_get", new=AsyncMock(side_effect=fake_get)): + await client.amazon_product("B000000000", retries=0) + assert "autoselect_variant" not in captured["params"] + + asyncio.run(run()) + + class TestGoogleDateRange: """Tests that google_search forwards date_range only when set.""" diff --git a/tests/unit/test_error_responses.py b/tests/unit/test_error_responses.py index f4e6f69..06dd6f2 100644 --- a/tests/unit/test_error_responses.py +++ b/tests/unit/test_error_responses.py @@ -94,6 +94,11 @@ def _mock_client_cls(method_name: str, status_code: int, body: bytes = b'{"error "scrapingbee_cli.commands.youtube.Client", "youtube_metadata", ), + ( + ["youtube-subtitles", "dQw4w9WgXcQ"], + "scrapingbee_cli.commands.youtube.Client", + "youtube_subtitles", + ), ( ["chatgpt", "hello"], "scrapingbee_cli.commands.chatgpt.Client", diff --git a/tests/unit/test_post_auth_header.py b/tests/unit/test_post_auth_header.py new file mode 100644 index 0000000..7db0b80 --- /dev/null +++ b/tests/unit/test_post_auth_header.py @@ -0,0 +1,86 @@ +"""Regression tests: user -H headers must never clobber the API's Bearer auth. + +With header-based auth (SCR-585), the session carries ``Authorization: Bearer +``. aiohttp lets per-request headers override session headers on the +same (case-insensitive) key, so passing a user's ``-H "Authorization: …"`` +through raw on POST/PUT replaced our key and the API rejected the request +(400/401). Custom headers are therefore always ``Spb-``-prefixed — which is +also the only form the API forwards to the target (raw custom headers were +silently dropped on POST/PUT). + +These tests run the real Client against a local fake API endpoint and assert +on the headers that actually arrive on the wire. +""" + +from __future__ import annotations + +import asyncio + +from aiohttp import web + +from scrapingbee_cli.client import Client + + +def _run_against_fake_api(method, custom_headers): + """Return the (headers, query) the fake ScrapingBee endpoint received.""" + seen = {} + + async def endpoint(request): + seen["headers"] = dict(request.headers) + seen["query"] = dict(request.query) + return web.json_response({"ok": True}) + + async def run(): + app = web.Application() + app.router.add_route("*", "/", endpoint) + runner = web.AppRunner(app) + await runner.setup() + site = web.TCPSite(runner, "127.0.0.1", 0) + await site.start() + port = runner.addresses[0][1] + try: + async with Client("fake-key", base_url=f"http://127.0.0.1:{port}") as client: + await client.scrape( + "https://example.com", + method=method, + body="x=1" if method != "get" else None, + custom_headers=custom_headers, + forward_headers_pure=True, + retries=0, + ) + finally: + await runner.cleanup() + + asyncio.run(run()) + return seen["headers"], seen["query"] + + +class TestUserAuthorizationHeaderDoesNotClobberApiAuth: + def test_post_with_user_basic_auth_header_still_authenticates(self): + headers, _ = _run_against_fake_api("post", {"Authorization": "Basic Zm9vOmJhcg=="}) + assert headers.get("Authorization") == "Bearer fake-key" + assert headers.get("Spb-Authorization") == "Basic Zm9vOmJhcg==" + + def test_put_with_user_token_header_still_authenticates(self): + headers, _ = _run_against_fake_api("put", {"Authorization": "Token abc"}) + assert headers.get("Authorization") == "Bearer fake-key" + assert headers.get("Spb-Authorization") == "Token abc" + + def test_get_with_user_auth_header_is_prefixed_and_keeps_bearer(self): + headers, _ = _run_against_fake_api("get", {"Authorization": "Basic Zm9vOmJhcg=="}) + assert headers.get("Authorization") == "Bearer fake-key" + assert headers.get("Spb-Authorization") == "Basic Zm9vOmJhcg==" + + +class TestSpbPrefixing: + def test_already_prefixed_header_is_not_double_prefixed(self): + headers, _ = _run_against_fake_api("post", {"Spb-X-Custom": "1"}) + assert headers.get("Spb-X-Custom") == "1" + assert "Spb-Spb-X-Custom" not in headers + + def test_user_content_type_goes_to_target_api_content_type_intact(self): + # The user's Content-Type must travel as Spb-Content-Type (for the + # target), while the request to the API itself stays form-encoded. + headers, _ = _run_against_fake_api("post", {"Content-Type": "application/json"}) + assert headers.get("Spb-Content-Type") == "application/json" + assert headers.get("Content-Type", "").startswith("application/x-www-form-urlencoded")