22
33from __future__ import annotations
44
5- from typing import Dict , Union , Optional
5+ from typing import Dict , Union , Iterable , Optional
66from typing_extensions import Literal
77
88import httpx
@@ -1391,6 +1391,7 @@ def web_scrape_html(
13911391 self ,
13921392 * ,
13931393 url : str ,
1394+ actions : Optional [Iterable [web_web_scrape_html_params .Action ]] | Omit = omit ,
13941395 country : Literal [
13951396 "ad" ,
13961397 "ae" ,
@@ -1617,12 +1618,18 @@ def web_scrape_html(
16171618 extra_body : Body | None = None ,
16181619 timeout : float | httpx .Timeout | None | NotGiven = not_given ,
16191620 ) -> WebWebScrapeHTMLResponse :
1620- """
1621- Scrapes the given URL and returns the raw HTML content of the page.
1621+ """Scrapes the given URL and returns the raw HTML content of the page.
1622+
1623+ The base
1624+ request costs 1 credit; requests with browser actions cost 2 credits.
16221625
16231626 Args:
16241627 url: Full URL to scrape (must include http:// or https:// protocol)
16251628
1629+ actions: Optional browser actions executed in array order after the page loads and before
1630+ content is captured. Requires a paid plan. Send a JSON array in the query
1631+ parameter. Maximum: 5 actions.
1632+
16261633 country: Two-letter ISO 3166-1 alpha-2 country code identifying a supported Context.dev
16271634 residential proxy exit location. Must be one of Context.dev's supported
16281635 countries. When provided, Context.dev fetches the target page from that country.
@@ -1690,6 +1697,7 @@ def web_scrape_html(
16901697 query = maybe_transform (
16911698 {
16921699 "url" : url ,
1700+ "actions" : actions ,
16931701 "country" : country ,
16941702 "exclude_selectors" : exclude_selectors ,
16951703 "headers" : headers ,
@@ -1714,6 +1722,7 @@ def web_scrape_images(
17141722 self ,
17151723 * ,
17161724 url : str ,
1725+ actions : Optional [Iterable [web_web_scrape_images_params .Action ]] | Omit = omit ,
17171726 dedupe : Union [bool , Literal ["true" , "false" ]] | Omit = omit ,
17181727 enrichment : Optional [web_web_scrape_images_params .Enrichment ] | Omit = omit ,
17191728 headers : Dict [str , str ] | Omit = omit ,
@@ -1731,12 +1740,17 @@ def web_scrape_images(
17311740 """
17321741 Extract image assets from a web page, including standard URLs, inline SVGs, data
17331742 URIs, responsive image sources, metadata, CSS backgrounds, video posters, and
1734- embeds. The base request costs 1 credit. When enrichment is enabled, the entire
1735- call costs 5 credits.
1743+ embeds. The base request costs 1 credit, or 2 credits with browser actions. When
1744+ enrichment is enabled, the entire call costs 5 credits, including requests that
1745+ also use actions.
17361746
17371747 Args:
17381748 url: Page URL to inspect. Must include http:// or https://.
17391749
1750+ actions: Optional browser actions executed in array order after the page loads and before
1751+ content is captured. Requires a paid plan. Send a JSON array in the query
1752+ parameter. Maximum: 5 actions.
1753+
17401754 dedupe: When true, visually duplicate images are removed: every image is loaded and
17411755 perceptually hashed, and only the highest-resolution copy of each duplicate
17421756 group is kept. Images that cannot be downloaded or hashed are kept. Default:
@@ -1781,6 +1795,7 @@ def web_scrape_images(
17811795 query = maybe_transform (
17821796 {
17831797 "url" : url ,
1798+ "actions" : actions ,
17841799 "dedupe" : dedupe ,
17851800 "enrichment" : enrichment ,
17861801 "headers" : headers ,
@@ -1799,6 +1814,7 @@ def web_scrape_md(
17991814 self ,
18001815 * ,
18011816 url : str ,
1817+ actions : Optional [Iterable [web_web_scrape_md_params .Action ]] | Omit = omit ,
18021818 country : Literal [
18031819 "ad" ,
18041820 "ae" ,
@@ -2036,21 +2052,25 @@ def web_scrape_md(
20362052
20372053 ### Billing & errors
20382054
2039- | HTTP status | Billed? | Meaning |
2040- | ----------- | -------------- | ---------------------------------------------------------------------------------------- |
2041- | 200 | Yes — 1 credit | Successful scrape, including a zero-length result when includeSelectors matched nothing |
2042- | 400 | No | Invalid input, skipped PDF, or the page could not be scraped |
2043- | 401 / 403 | No | Invalid/disabled key, insufficient permissions, or credits exhausted; inspect error_code |
2044- | 404 | No | Target page returned or fingerprinted as not found |
2045- | 408 | No | Request timed out |
2046- | 415 | No | Unsupported content type |
2047- | 429 | No | Per-minute rate limit exceeded; honor Retry-After |
2048- | 500 | No | Internal error |
2055+ | HTTP status | Billed? | Meaning |
2056+ | ----------- | ----------------------------------------- | ---------------------------------------------------------------------------------------- |
2057+ | 200 | Yes — 1 credit, or 2 credits with actions | Successful scrape, including a zero-length result when includeSelectors matched nothing |
2058+ | 400 | No | Invalid input, skipped PDF, or the page could not be scraped |
2059+ | 401 / 403 | No | Invalid/disabled key, insufficient permissions, or credits exhausted; inspect error_code |
2060+ | 404 | No | Target page returned or fingerprinted as not found |
2061+ | 408 | No | Request timed out |
2062+ | 415 | No | Unsupported content type |
2063+ | 429 | No | Per-minute rate limit exceeded; honor Retry-After |
2064+ | 500 | No | Internal error |
20492065
20502066 Args:
20512067 url: Full URL to scrape into LLM usable Markdown (must include http:// or https://
20522068 protocol)
20532069
2070+ actions: Optional browser actions executed in array order after the page loads and before
2071+ content is captured. Requires a paid plan. Send a JSON array in the query
2072+ parameter. Maximum: 5 actions.
2073+
20542074 country: Two-letter ISO 3166-1 alpha-2 country code identifying a supported Context.dev
20552075 residential proxy exit location. Must be one of Context.dev's supported
20562076 countries. When provided, Context.dev fetches the target page from that country.
@@ -2123,6 +2143,7 @@ def web_scrape_md(
21232143 query = maybe_transform (
21242144 {
21252145 "url" : url ,
2146+ "actions" : actions ,
21262147 "country" : country ,
21272148 "exclude_selectors" : exclude_selectors ,
21282149 "headers" : headers ,
@@ -3574,6 +3595,7 @@ async def web_scrape_html(
35743595 self ,
35753596 * ,
35763597 url : str ,
3598+ actions : Optional [Iterable [web_web_scrape_html_params .Action ]] | Omit = omit ,
35773599 country : Literal [
35783600 "ad" ,
35793601 "ae" ,
@@ -3800,12 +3822,18 @@ async def web_scrape_html(
38003822 extra_body : Body | None = None ,
38013823 timeout : float | httpx .Timeout | None | NotGiven = not_given ,
38023824 ) -> WebWebScrapeHTMLResponse :
3803- """
3804- Scrapes the given URL and returns the raw HTML content of the page.
3825+ """Scrapes the given URL and returns the raw HTML content of the page.
3826+
3827+ The base
3828+ request costs 1 credit; requests with browser actions cost 2 credits.
38053829
38063830 Args:
38073831 url: Full URL to scrape (must include http:// or https:// protocol)
38083832
3833+ actions: Optional browser actions executed in array order after the page loads and before
3834+ content is captured. Requires a paid plan. Send a JSON array in the query
3835+ parameter. Maximum: 5 actions.
3836+
38093837 country: Two-letter ISO 3166-1 alpha-2 country code identifying a supported Context.dev
38103838 residential proxy exit location. Must be one of Context.dev's supported
38113839 countries. When provided, Context.dev fetches the target page from that country.
@@ -3873,6 +3901,7 @@ async def web_scrape_html(
38733901 query = await async_maybe_transform (
38743902 {
38753903 "url" : url ,
3904+ "actions" : actions ,
38763905 "country" : country ,
38773906 "exclude_selectors" : exclude_selectors ,
38783907 "headers" : headers ,
@@ -3897,6 +3926,7 @@ async def web_scrape_images(
38973926 self ,
38983927 * ,
38993928 url : str ,
3929+ actions : Optional [Iterable [web_web_scrape_images_params .Action ]] | Omit = omit ,
39003930 dedupe : Union [bool , Literal ["true" , "false" ]] | Omit = omit ,
39013931 enrichment : Optional [web_web_scrape_images_params .Enrichment ] | Omit = omit ,
39023932 headers : Dict [str , str ] | Omit = omit ,
@@ -3914,12 +3944,17 @@ async def web_scrape_images(
39143944 """
39153945 Extract image assets from a web page, including standard URLs, inline SVGs, data
39163946 URIs, responsive image sources, metadata, CSS backgrounds, video posters, and
3917- embeds. The base request costs 1 credit. When enrichment is enabled, the entire
3918- call costs 5 credits.
3947+ embeds. The base request costs 1 credit, or 2 credits with browser actions. When
3948+ enrichment is enabled, the entire call costs 5 credits, including requests that
3949+ also use actions.
39193950
39203951 Args:
39213952 url: Page URL to inspect. Must include http:// or https://.
39223953
3954+ actions: Optional browser actions executed in array order after the page loads and before
3955+ content is captured. Requires a paid plan. Send a JSON array in the query
3956+ parameter. Maximum: 5 actions.
3957+
39233958 dedupe: When true, visually duplicate images are removed: every image is loaded and
39243959 perceptually hashed, and only the highest-resolution copy of each duplicate
39253960 group is kept. Images that cannot be downloaded or hashed are kept. Default:
@@ -3964,6 +3999,7 @@ async def web_scrape_images(
39643999 query = await async_maybe_transform (
39654000 {
39664001 "url" : url ,
4002+ "actions" : actions ,
39674003 "dedupe" : dedupe ,
39684004 "enrichment" : enrichment ,
39694005 "headers" : headers ,
@@ -3982,6 +4018,7 @@ async def web_scrape_md(
39824018 self ,
39834019 * ,
39844020 url : str ,
4021+ actions : Optional [Iterable [web_web_scrape_md_params .Action ]] | Omit = omit ,
39854022 country : Literal [
39864023 "ad" ,
39874024 "ae" ,
@@ -4219,21 +4256,25 @@ async def web_scrape_md(
42194256
42204257 ### Billing & errors
42214258
4222- | HTTP status | Billed? | Meaning |
4223- | ----------- | -------------- | ---------------------------------------------------------------------------------------- |
4224- | 200 | Yes — 1 credit | Successful scrape, including a zero-length result when includeSelectors matched nothing |
4225- | 400 | No | Invalid input, skipped PDF, or the page could not be scraped |
4226- | 401 / 403 | No | Invalid/disabled key, insufficient permissions, or credits exhausted; inspect error_code |
4227- | 404 | No | Target page returned or fingerprinted as not found |
4228- | 408 | No | Request timed out |
4229- | 415 | No | Unsupported content type |
4230- | 429 | No | Per-minute rate limit exceeded; honor Retry-After |
4231- | 500 | No | Internal error |
4259+ | HTTP status | Billed? | Meaning |
4260+ | ----------- | ----------------------------------------- | ---------------------------------------------------------------------------------------- |
4261+ | 200 | Yes — 1 credit, or 2 credits with actions | Successful scrape, including a zero-length result when includeSelectors matched nothing |
4262+ | 400 | No | Invalid input, skipped PDF, or the page could not be scraped |
4263+ | 401 / 403 | No | Invalid/disabled key, insufficient permissions, or credits exhausted; inspect error_code |
4264+ | 404 | No | Target page returned or fingerprinted as not found |
4265+ | 408 | No | Request timed out |
4266+ | 415 | No | Unsupported content type |
4267+ | 429 | No | Per-minute rate limit exceeded; honor Retry-After |
4268+ | 500 | No | Internal error |
42324269
42334270 Args:
42344271 url: Full URL to scrape into LLM usable Markdown (must include http:// or https://
42354272 protocol)
42364273
4274+ actions: Optional browser actions executed in array order after the page loads and before
4275+ content is captured. Requires a paid plan. Send a JSON array in the query
4276+ parameter. Maximum: 5 actions.
4277+
42374278 country: Two-letter ISO 3166-1 alpha-2 country code identifying a supported Context.dev
42384279 residential proxy exit location. Must be one of Context.dev's supported
42394280 countries. When provided, Context.dev fetches the target page from that country.
@@ -4306,6 +4347,7 @@ async def web_scrape_md(
43064347 query = await async_maybe_transform (
43074348 {
43084349 "url" : url ,
4350+ "actions" : actions ,
43094351 "country" : country ,
43104352 "exclude_selectors" : exclude_selectors ,
43114353 "headers" : headers ,
0 commit comments