Skip to content

Commit cfec05f

Browse files
author
stlc-bot
committed
feat(scrape): add unified scrape API (#1182)
Stainless-Generated-From: 3e47848454ce16dfa04feab91529465422e6b86f
1 parent 542c639 commit cfec05f

7 files changed

Lines changed: 980 additions & 1 deletion

File tree

‎.stats.yml‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1 +1 @@
1-
configured_endpoints: 50
1+
configured_endpoints: 51

‎api.md‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -21,6 +21,7 @@ from context.dev.types import (
2121
WebExtractCompetitorsResponse,
2222
WebExtractFontsResponse,
2323
WebExtractStyleguideResponse,
24+
WebScrapeResponse,
2425
WebScreenshotResponse,
2526
WebSearchResponse,
2627
WebWebCrawlMdResponse,
@@ -40,6 +41,7 @@ Methods:
4041
- <code title="get /web/competitors">client.web.<a href="./src/context/dev/resources/web.py">extract_competitors</a>(\*\*<a href="src/context/dev/types/web_extract_competitors_params.py">params</a>) -> <a href="./src/context/dev/types/web_extract_competitors_response.py">WebExtractCompetitorsResponse</a></code>
4142
- <code title="get /web/fonts">client.web.<a href="./src/context/dev/resources/web.py">extract_fonts</a>(\*\*<a href="src/context/dev/types/web_extract_fonts_params.py">params</a>) -> <a href="./src/context/dev/types/web_extract_fonts_response.py">WebExtractFontsResponse</a></code>
4243
- <code title="get /web/styleguide">client.web.<a href="./src/context/dev/resources/web.py">extract_styleguide</a>(\*\*<a href="src/context/dev/types/web_extract_styleguide_params.py">params</a>) -> <a href="./src/context/dev/types/web_extract_styleguide_response.py">WebExtractStyleguideResponse</a></code>
44+
- <code title="post /web/scrape">client.web.<a href="./src/context/dev/resources/web.py">scrape</a>(\*\*<a href="src/context/dev/types/web_scrape_params.py">params</a>) -> <a href="./src/context/dev/types/web_scrape_response.py">WebScrapeResponse</a></code>
4345
- <code title="get /web/screenshot">client.web.<a href="./src/context/dev/resources/web.py">screenshot</a>(\*\*<a href="src/context/dev/types/web_screenshot_params.py">params</a>) -> <a href="./src/context/dev/types/web_screenshot_response.py">WebScreenshotResponse</a></code>
4446
- <code title="post /web/search">client.web.<a href="./src/context/dev/resources/web.py">search</a>(\*\*<a href="src/context/dev/types/web_search_params.py">params</a>) -> <a href="./src/context/dev/types/web_search_response.py">WebSearchResponse</a></code>
4547
- <code title="post /web/crawl">client.web.<a href="./src/context/dev/resources/web.py">web_crawl_md</a>(\*\*<a href="src/context/dev/types/web_web_crawl_md_params.py">params</a>) -> <a href="./src/context/dev/types/web_web_crawl_md_response.py">WebWebCrawlMdResponse</a></code>

‎src/context/dev/resources/web.py‎

Lines changed: 188 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -8,6 +8,7 @@
88
import httpx
99

1010
from ..types import (
11+
web_scrape_params,
1112
web_search_params,
1213
web_answers_params,
1314
web_extract_params,
@@ -34,6 +35,7 @@
3435
async_to_streamed_response_wrapper,
3536
)
3637
from .._base_client import make_request_options
38+
from ..types.web_scrape_response import WebScrapeResponse
3739
from ..types.web_search_response import WebSearchResponse
3840
from ..types.web_answers_response import WebAnswersResponse
3941
from ..types.web_extract_response import WebExtractResponse
@@ -496,6 +498,93 @@ def extract_styleguide(
496498
cast_to=WebExtractStyleguideResponse,
497499
)
498500

501+
def scrape(
502+
self,
503+
*,
504+
formats: web_scrape_params.Formats,
505+
url: str,
506+
image_params: web_scrape_params.ImageParams | Omit = omit,
507+
markdown_params: web_scrape_params.MarkdownParams | Omit = omit,
508+
max_age_ms: int | Omit = omit,
509+
parse_params: web_scrape_params.ParseParams | Omit = omit,
510+
screenshot_params: web_scrape_params.ScreenshotParams | Omit = omit,
511+
shared_params: web_scrape_params.SharedParams | Omit = omit,
512+
tags: SequenceNotStr[str] | Omit = omit,
513+
timeout_ms: int | Omit = omit,
514+
zdr: Literal["enabled", "disabled"] | Omit = omit,
515+
# Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
516+
# The extra values given here take precedence over values defined on the client or passed to this method.
517+
extra_headers: Headers | None = None,
518+
extra_query: Query | None = None,
519+
extra_body: Body | None = None,
520+
timeout: float | httpx.Timeout | None | NotGiven = not_given,
521+
) -> WebScrapeResponse:
522+
"""Capture the requested formats from one page visit.
523+
524+
Shared settings apply once.
525+
HTML-only requests use the existing fast acquisition path. One credit per
526+
capture, or two with browser actions; PDF OCR adds one credit per recovered
527+
page. Original response bytes and screenshots are limited to 20 MiB each,
528+
screenshots to 40 megapixels, and the combined browser capture to 60 MiB.
529+
530+
Args:
531+
formats: Outputs to return. Enable at least one; omitted formats are false.
532+
533+
url: The URL to scrape.
534+
535+
image_params: Image options. Requires formats.images: true.
536+
537+
markdown_params: Markdown options. Requires formats.markdown: true.
538+
539+
max_age_ms: Maximum age for the entire capture, including bytes. Defaults to 1 day; 0
540+
fetches fresh. Captures with hosted image files refresh after 23 hours.
541+
542+
parse_params: Required when formats.parse is true.
543+
544+
screenshot_params: Screenshot options. Requires formats.screenshot: true.
545+
546+
shared_params: Shared browser and content settings. Content filters leave screenshots and
547+
original bytes unchanged.
548+
549+
tags: Labels for tracking request usage. Not retained when zdr is enabled.
550+
551+
timeout_ms: Total deadline, including navigation, actions, waiting, and all outputs.
552+
553+
zdr: Zero data retention. Bypasses caches and uploads; excludes request/response
554+
content and tags from logs. Must be enabled for your organization.
555+
556+
extra_headers: Send extra headers
557+
558+
extra_query: Add additional query parameters to the request
559+
560+
extra_body: Add additional JSON properties to the request
561+
562+
timeout: Override the client-level default timeout for this request, in seconds
563+
"""
564+
return self._post(
565+
"/web/scrape",
566+
body=maybe_transform(
567+
{
568+
"formats": formats,
569+
"url": url,
570+
"image_params": image_params,
571+
"markdown_params": markdown_params,
572+
"max_age_ms": max_age_ms,
573+
"parse_params": parse_params,
574+
"screenshot_params": screenshot_params,
575+
"shared_params": shared_params,
576+
"tags": tags,
577+
"timeout_ms": timeout_ms,
578+
"zdr": zdr,
579+
},
580+
web_scrape_params.WebScrapeParams,
581+
),
582+
options=make_request_options(
583+
extra_headers=extra_headers, extra_query=extra_query, extra_body=extra_body, timeout=timeout
584+
),
585+
cast_to=WebScrapeResponse,
586+
)
587+
499588
def screenshot(
500589
self,
501590
*,
@@ -3761,6 +3850,93 @@ async def extract_styleguide(
37613850
cast_to=WebExtractStyleguideResponse,
37623851
)
37633852

3853+
async def scrape(
3854+
self,
3855+
*,
3856+
formats: web_scrape_params.Formats,
3857+
url: str,
3858+
image_params: web_scrape_params.ImageParams | Omit = omit,
3859+
markdown_params: web_scrape_params.MarkdownParams | Omit = omit,
3860+
max_age_ms: int | Omit = omit,
3861+
parse_params: web_scrape_params.ParseParams | Omit = omit,
3862+
screenshot_params: web_scrape_params.ScreenshotParams | Omit = omit,
3863+
shared_params: web_scrape_params.SharedParams | Omit = omit,
3864+
tags: SequenceNotStr[str] | Omit = omit,
3865+
timeout_ms: int | Omit = omit,
3866+
zdr: Literal["enabled", "disabled"] | Omit = omit,
3867+
# Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
3868+
# The extra values given here take precedence over values defined on the client or passed to this method.
3869+
extra_headers: Headers | None = None,
3870+
extra_query: Query | None = None,
3871+
extra_body: Body | None = None,
3872+
timeout: float | httpx.Timeout | None | NotGiven = not_given,
3873+
) -> WebScrapeResponse:
3874+
"""Capture the requested formats from one page visit.
3875+
3876+
Shared settings apply once.
3877+
HTML-only requests use the existing fast acquisition path. One credit per
3878+
capture, or two with browser actions; PDF OCR adds one credit per recovered
3879+
page. Original response bytes and screenshots are limited to 20 MiB each,
3880+
screenshots to 40 megapixels, and the combined browser capture to 60 MiB.
3881+
3882+
Args:
3883+
formats: Outputs to return. Enable at least one; omitted formats are false.
3884+
3885+
url: The URL to scrape.
3886+
3887+
image_params: Image options. Requires formats.images: true.
3888+
3889+
markdown_params: Markdown options. Requires formats.markdown: true.
3890+
3891+
max_age_ms: Maximum age for the entire capture, including bytes. Defaults to 1 day; 0
3892+
fetches fresh. Captures with hosted image files refresh after 23 hours.
3893+
3894+
parse_params: Required when formats.parse is true.
3895+
3896+
screenshot_params: Screenshot options. Requires formats.screenshot: true.
3897+
3898+
shared_params: Shared browser and content settings. Content filters leave screenshots and
3899+
original bytes unchanged.
3900+
3901+
tags: Labels for tracking request usage. Not retained when zdr is enabled.
3902+
3903+
timeout_ms: Total deadline, including navigation, actions, waiting, and all outputs.
3904+
3905+
zdr: Zero data retention. Bypasses caches and uploads; excludes request/response
3906+
content and tags from logs. Must be enabled for your organization.
3907+
3908+
extra_headers: Send extra headers
3909+
3910+
extra_query: Add additional query parameters to the request
3911+
3912+
extra_body: Add additional JSON properties to the request
3913+
3914+
timeout: Override the client-level default timeout for this request, in seconds
3915+
"""
3916+
return await self._post(
3917+
"/web/scrape",
3918+
body=await async_maybe_transform(
3919+
{
3920+
"formats": formats,
3921+
"url": url,
3922+
"image_params": image_params,
3923+
"markdown_params": markdown_params,
3924+
"max_age_ms": max_age_ms,
3925+
"parse_params": parse_params,
3926+
"screenshot_params": screenshot_params,
3927+
"shared_params": shared_params,
3928+
"tags": tags,
3929+
"timeout_ms": timeout_ms,
3930+
"zdr": zdr,
3931+
},
3932+
web_scrape_params.WebScrapeParams,
3933+
),
3934+
options=make_request_options(
3935+
extra_headers=extra_headers, extra_query=extra_query, extra_body=extra_body, timeout=timeout
3936+
),
3937+
cast_to=WebScrapeResponse,
3938+
)
3939+
37643940
async def screenshot(
37653941
self,
37663942
*,
@@ -6601,6 +6777,9 @@ def __init__(self, web: WebResource) -> None:
66016777
self.extract_styleguide = to_raw_response_wrapper(
66026778
web.extract_styleguide,
66036779
)
6780+
self.scrape = to_raw_response_wrapper(
6781+
web.scrape,
6782+
)
66046783
self.screenshot = to_raw_response_wrapper(
66056784
web.screenshot,
66066785
)
@@ -6649,6 +6828,9 @@ def __init__(self, web: AsyncWebResource) -> None:
66496828
self.extract_styleguide = async_to_raw_response_wrapper(
66506829
web.extract_styleguide,
66516830
)
6831+
self.scrape = async_to_raw_response_wrapper(
6832+
web.scrape,
6833+
)
66526834
self.screenshot = async_to_raw_response_wrapper(
66536835
web.screenshot,
66546836
)
@@ -6697,6 +6879,9 @@ def __init__(self, web: WebResource) -> None:
66976879
self.extract_styleguide = to_streamed_response_wrapper(
66986880
web.extract_styleguide,
66996881
)
6882+
self.scrape = to_streamed_response_wrapper(
6883+
web.scrape,
6884+
)
67006885
self.screenshot = to_streamed_response_wrapper(
67016886
web.screenshot,
67026887
)
@@ -6745,6 +6930,9 @@ def __init__(self, web: AsyncWebResource) -> None:
67456930
self.extract_styleguide = async_to_streamed_response_wrapper(
67466931
web.extract_styleguide,
67476932
)
6933+
self.scrape = async_to_streamed_response_wrapper(
6934+
web.scrape,
6935+
)
67486936
self.screenshot = async_to_streamed_response_wrapper(
67496937
web.screenshot,
67506938
)

‎src/context/dev/types/__init__.py‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -11,6 +11,7 @@
1111
from .webhook_delivery import WebhookDelivery as WebhookDelivery
1212
from .batch_list_params import BatchListParams as BatchListParams
1313
from .log_list_response import LogListResponse as LogListResponse
14+
from .web_scrape_params import WebScrapeParams as WebScrapeParams
1415
from .web_search_params import WebSearchParams as WebSearchParams
1516
from .news_search_params import NewsSearchParams as NewsSearchParams
1617
from .retry_config_param import RetryConfigParam as RetryConfigParam
@@ -21,6 +22,7 @@
2122
from .brand_search_params import BrandSearchParams as BrandSearchParams
2223
from .monitor_list_params import MonitorListParams as MonitorListParams
2324
from .parse_handle_params import ParseHandleParams as ParseHandleParams
25+
from .web_scrape_response import WebScrapeResponse as WebScrapeResponse
2426
from .web_search_response import WebSearchResponse as WebSearchResponse
2527
from .monitor_run_response import MonitorRunResponse as MonitorRunResponse
2628
from .news_search_response import NewsSearchResponse as NewsSearchResponse

0 commit comments

Comments
 (0)