|
8 | 8 | import httpx |
9 | 9 |
|
10 | 10 | from ..types import ( |
| 11 | + web_scrape_params, |
11 | 12 | web_search_params, |
12 | 13 | web_answers_params, |
13 | 14 | web_extract_params, |
|
34 | 35 | async_to_streamed_response_wrapper, |
35 | 36 | ) |
36 | 37 | from .._base_client import make_request_options |
| 38 | +from ..types.web_scrape_response import WebScrapeResponse |
37 | 39 | from ..types.web_search_response import WebSearchResponse |
38 | 40 | from ..types.web_answers_response import WebAnswersResponse |
39 | 41 | from ..types.web_extract_response import WebExtractResponse |
@@ -496,6 +498,93 @@ def extract_styleguide( |
496 | 498 | cast_to=WebExtractStyleguideResponse, |
497 | 499 | ) |
498 | 500 |
|
| 501 | + def scrape( |
| 502 | + self, |
| 503 | + *, |
| 504 | + formats: web_scrape_params.Formats, |
| 505 | + url: str, |
| 506 | + image_params: web_scrape_params.ImageParams | Omit = omit, |
| 507 | + markdown_params: web_scrape_params.MarkdownParams | Omit = omit, |
| 508 | + max_age_ms: int | Omit = omit, |
| 509 | + parse_params: web_scrape_params.ParseParams | Omit = omit, |
| 510 | + screenshot_params: web_scrape_params.ScreenshotParams | Omit = omit, |
| 511 | + shared_params: web_scrape_params.SharedParams | Omit = omit, |
| 512 | + tags: SequenceNotStr[str] | Omit = omit, |
| 513 | + timeout_ms: int | Omit = omit, |
| 514 | + zdr: Literal["enabled", "disabled"] | Omit = omit, |
| 515 | + # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs. |
| 516 | + # The extra values given here take precedence over values defined on the client or passed to this method. |
| 517 | + extra_headers: Headers | None = None, |
| 518 | + extra_query: Query | None = None, |
| 519 | + extra_body: Body | None = None, |
| 520 | + timeout: float | httpx.Timeout | None | NotGiven = not_given, |
| 521 | + ) -> WebScrapeResponse: |
| 522 | + """Capture the requested formats from one page visit. |
| 523 | +
|
| 524 | + Shared settings apply once. |
| 525 | + HTML-only requests use the existing fast acquisition path. One credit per |
| 526 | + capture, or two with browser actions; PDF OCR adds one credit per recovered |
| 527 | + page. Original response bytes and screenshots are limited to 20 MiB each, |
| 528 | + screenshots to 40 megapixels, and the combined browser capture to 60 MiB. |
| 529 | +
|
| 530 | + Args: |
| 531 | + formats: Outputs to return. Enable at least one; omitted formats are false. |
| 532 | +
|
| 533 | + url: The URL to scrape. |
| 534 | +
|
| 535 | + image_params: Image options. Requires formats.images: true. |
| 536 | +
|
| 537 | + markdown_params: Markdown options. Requires formats.markdown: true. |
| 538 | +
|
| 539 | + max_age_ms: Maximum age for the entire capture, including bytes. Defaults to 1 day; 0 |
| 540 | + fetches fresh. Captures with hosted image files refresh after 23 hours. |
| 541 | +
|
| 542 | + parse_params: Required when formats.parse is true. |
| 543 | +
|
| 544 | + screenshot_params: Screenshot options. Requires formats.screenshot: true. |
| 545 | +
|
| 546 | + shared_params: Shared browser and content settings. Content filters leave screenshots and |
| 547 | + original bytes unchanged. |
| 548 | +
|
| 549 | + tags: Labels for tracking request usage. Not retained when zdr is enabled. |
| 550 | +
|
| 551 | + timeout_ms: Total deadline, including navigation, actions, waiting, and all outputs. |
| 552 | +
|
| 553 | + zdr: Zero data retention. Bypasses caches and uploads; excludes request/response |
| 554 | + content and tags from logs. Must be enabled for your organization. |
| 555 | +
|
| 556 | + extra_headers: Send extra headers |
| 557 | +
|
| 558 | + extra_query: Add additional query parameters to the request |
| 559 | +
|
| 560 | + extra_body: Add additional JSON properties to the request |
| 561 | +
|
| 562 | + timeout: Override the client-level default timeout for this request, in seconds |
| 563 | + """ |
| 564 | + return self._post( |
| 565 | + "/web/scrape", |
| 566 | + body=maybe_transform( |
| 567 | + { |
| 568 | + "formats": formats, |
| 569 | + "url": url, |
| 570 | + "image_params": image_params, |
| 571 | + "markdown_params": markdown_params, |
| 572 | + "max_age_ms": max_age_ms, |
| 573 | + "parse_params": parse_params, |
| 574 | + "screenshot_params": screenshot_params, |
| 575 | + "shared_params": shared_params, |
| 576 | + "tags": tags, |
| 577 | + "timeout_ms": timeout_ms, |
| 578 | + "zdr": zdr, |
| 579 | + }, |
| 580 | + web_scrape_params.WebScrapeParams, |
| 581 | + ), |
| 582 | + options=make_request_options( |
| 583 | + extra_headers=extra_headers, extra_query=extra_query, extra_body=extra_body, timeout=timeout |
| 584 | + ), |
| 585 | + cast_to=WebScrapeResponse, |
| 586 | + ) |
| 587 | + |
499 | 588 | def screenshot( |
500 | 589 | self, |
501 | 590 | *, |
@@ -3761,6 +3850,93 @@ async def extract_styleguide( |
3761 | 3850 | cast_to=WebExtractStyleguideResponse, |
3762 | 3851 | ) |
3763 | 3852 |
|
| 3853 | + async def scrape( |
| 3854 | + self, |
| 3855 | + *, |
| 3856 | + formats: web_scrape_params.Formats, |
| 3857 | + url: str, |
| 3858 | + image_params: web_scrape_params.ImageParams | Omit = omit, |
| 3859 | + markdown_params: web_scrape_params.MarkdownParams | Omit = omit, |
| 3860 | + max_age_ms: int | Omit = omit, |
| 3861 | + parse_params: web_scrape_params.ParseParams | Omit = omit, |
| 3862 | + screenshot_params: web_scrape_params.ScreenshotParams | Omit = omit, |
| 3863 | + shared_params: web_scrape_params.SharedParams | Omit = omit, |
| 3864 | + tags: SequenceNotStr[str] | Omit = omit, |
| 3865 | + timeout_ms: int | Omit = omit, |
| 3866 | + zdr: Literal["enabled", "disabled"] | Omit = omit, |
| 3867 | + # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs. |
| 3868 | + # The extra values given here take precedence over values defined on the client or passed to this method. |
| 3869 | + extra_headers: Headers | None = None, |
| 3870 | + extra_query: Query | None = None, |
| 3871 | + extra_body: Body | None = None, |
| 3872 | + timeout: float | httpx.Timeout | None | NotGiven = not_given, |
| 3873 | + ) -> WebScrapeResponse: |
| 3874 | + """Capture the requested formats from one page visit. |
| 3875 | +
|
| 3876 | + Shared settings apply once. |
| 3877 | + HTML-only requests use the existing fast acquisition path. One credit per |
| 3878 | + capture, or two with browser actions; PDF OCR adds one credit per recovered |
| 3879 | + page. Original response bytes and screenshots are limited to 20 MiB each, |
| 3880 | + screenshots to 40 megapixels, and the combined browser capture to 60 MiB. |
| 3881 | +
|
| 3882 | + Args: |
| 3883 | + formats: Outputs to return. Enable at least one; omitted formats are false. |
| 3884 | +
|
| 3885 | + url: The URL to scrape. |
| 3886 | +
|
| 3887 | + image_params: Image options. Requires formats.images: true. |
| 3888 | +
|
| 3889 | + markdown_params: Markdown options. Requires formats.markdown: true. |
| 3890 | +
|
| 3891 | + max_age_ms: Maximum age for the entire capture, including bytes. Defaults to 1 day; 0 |
| 3892 | + fetches fresh. Captures with hosted image files refresh after 23 hours. |
| 3893 | +
|
| 3894 | + parse_params: Required when formats.parse is true. |
| 3895 | +
|
| 3896 | + screenshot_params: Screenshot options. Requires formats.screenshot: true. |
| 3897 | +
|
| 3898 | + shared_params: Shared browser and content settings. Content filters leave screenshots and |
| 3899 | + original bytes unchanged. |
| 3900 | +
|
| 3901 | + tags: Labels for tracking request usage. Not retained when zdr is enabled. |
| 3902 | +
|
| 3903 | + timeout_ms: Total deadline, including navigation, actions, waiting, and all outputs. |
| 3904 | +
|
| 3905 | + zdr: Zero data retention. Bypasses caches and uploads; excludes request/response |
| 3906 | + content and tags from logs. Must be enabled for your organization. |
| 3907 | +
|
| 3908 | + extra_headers: Send extra headers |
| 3909 | +
|
| 3910 | + extra_query: Add additional query parameters to the request |
| 3911 | +
|
| 3912 | + extra_body: Add additional JSON properties to the request |
| 3913 | +
|
| 3914 | + timeout: Override the client-level default timeout for this request, in seconds |
| 3915 | + """ |
| 3916 | + return await self._post( |
| 3917 | + "/web/scrape", |
| 3918 | + body=await async_maybe_transform( |
| 3919 | + { |
| 3920 | + "formats": formats, |
| 3921 | + "url": url, |
| 3922 | + "image_params": image_params, |
| 3923 | + "markdown_params": markdown_params, |
| 3924 | + "max_age_ms": max_age_ms, |
| 3925 | + "parse_params": parse_params, |
| 3926 | + "screenshot_params": screenshot_params, |
| 3927 | + "shared_params": shared_params, |
| 3928 | + "tags": tags, |
| 3929 | + "timeout_ms": timeout_ms, |
| 3930 | + "zdr": zdr, |
| 3931 | + }, |
| 3932 | + web_scrape_params.WebScrapeParams, |
| 3933 | + ), |
| 3934 | + options=make_request_options( |
| 3935 | + extra_headers=extra_headers, extra_query=extra_query, extra_body=extra_body, timeout=timeout |
| 3936 | + ), |
| 3937 | + cast_to=WebScrapeResponse, |
| 3938 | + ) |
| 3939 | + |
3764 | 3940 | async def screenshot( |
3765 | 3941 | self, |
3766 | 3942 | *, |
@@ -6601,6 +6777,9 @@ def __init__(self, web: WebResource) -> None: |
6601 | 6777 | self.extract_styleguide = to_raw_response_wrapper( |
6602 | 6778 | web.extract_styleguide, |
6603 | 6779 | ) |
| 6780 | + self.scrape = to_raw_response_wrapper( |
| 6781 | + web.scrape, |
| 6782 | + ) |
6604 | 6783 | self.screenshot = to_raw_response_wrapper( |
6605 | 6784 | web.screenshot, |
6606 | 6785 | ) |
@@ -6649,6 +6828,9 @@ def __init__(self, web: AsyncWebResource) -> None: |
6649 | 6828 | self.extract_styleguide = async_to_raw_response_wrapper( |
6650 | 6829 | web.extract_styleguide, |
6651 | 6830 | ) |
| 6831 | + self.scrape = async_to_raw_response_wrapper( |
| 6832 | + web.scrape, |
| 6833 | + ) |
6652 | 6834 | self.screenshot = async_to_raw_response_wrapper( |
6653 | 6835 | web.screenshot, |
6654 | 6836 | ) |
@@ -6697,6 +6879,9 @@ def __init__(self, web: WebResource) -> None: |
6697 | 6879 | self.extract_styleguide = to_streamed_response_wrapper( |
6698 | 6880 | web.extract_styleguide, |
6699 | 6881 | ) |
| 6882 | + self.scrape = to_streamed_response_wrapper( |
| 6883 | + web.scrape, |
| 6884 | + ) |
6700 | 6885 | self.screenshot = to_streamed_response_wrapper( |
6701 | 6886 | web.screenshot, |
6702 | 6887 | ) |
@@ -6745,6 +6930,9 @@ def __init__(self, web: AsyncWebResource) -> None: |
6745 | 6930 | self.extract_styleguide = async_to_streamed_response_wrapper( |
6746 | 6931 | web.extract_styleguide, |
6747 | 6932 | ) |
| 6933 | + self.scrape = async_to_streamed_response_wrapper( |
| 6934 | + web.scrape, |
| 6935 | + ) |
6748 | 6936 | self.screenshot = async_to_streamed_response_wrapper( |
6749 | 6937 | web.screenshot, |
6750 | 6938 | ) |
|
0 commit comments