diff --git a/.release-please-manifest.json b/.release-please-manifest.json index e6c39ad..0746cbe 100644 --- a/.release-please-manifest.json +++ b/.release-please-manifest.json @@ -1,3 +1,3 @@ { - ".": "2.11.0" + ".": "2.12.0" } \ No newline at end of file diff --git a/.stats.yml b/.stats.yml index 9673209..ad6cd80 100644 --- a/.stats.yml +++ b/.stats.yml @@ -1,4 +1,4 @@ configured_endpoints: 41 -openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev/context.dev-b0c72d2fe5911c9ce155275b954198bf2152e915a7e5b9f719aad50e222b9ef8.yml -openapi_spec_hash: 5cf943ea5718059b7975417b012cee28 +openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev/context.dev-2c3f4e4b357ca78c019a3aacd006e2bb8b332d4172a300ee6c46c076a216ed3f.yml +openapi_spec_hash: b89be415cf6ee594b521885a9deb25ac config_hash: 0fb0ceca5946298c416cec0cca5260c7 diff --git a/CHANGELOG.md b/CHANGELOG.md index 54d9d18..ffa6a2e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,16 @@ # Changelog +## 2.12.0 (2026-08-23) + +Full Changelog: [v2.11.0...v2.12.0](https://github.com/context-dot-dev/context-python-sdk/compare/v2.11.0...v2.12.0) + +### Features + +* **api:** api update ([3c5c2c3](https://github.com/context-dot-dev/context-python-sdk/commit/3c5c2c39164d0be19c17949180c73937140687ea)) +* **api:** api update ([3a29950](https://github.com/context-dot-dev/context-python-sdk/commit/3a29950c03cfa9deab35b08061f295d5a2b2c092)) +* **api:** api update ([0a68836](https://github.com/context-dot-dev/context-python-sdk/commit/0a68836a31d685598a8f491f723970f8ea4e71f8)) +* **api:** api update ([288c743](https://github.com/context-dot-dev/context-python-sdk/commit/288c7436453a199728d8b9407c0a51e83e0c0718)) + ## 2.11.0 (2026-08-18) Full Changelog: [v2.10.0...v2.11.0](https://github.com/context-dot-dev/context-python-sdk/compare/v2.10.0...v2.11.0) diff --git a/pyproject.toml b/pyproject.toml index f87c5c2..46d9234 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "context.dev" -version = "2.11.0" +version = "2.12.0" description = "The official Python library for the context.dev API" dynamic = ["readme"] license = "Apache-2.0" diff --git a/src/context/dev/_version.py b/src/context/dev/_version.py index 0e1698d..aaa4b1d 100644 --- a/src/context/dev/_version.py +++ b/src/context/dev/_version.py @@ -1,4 +1,4 @@ # File generated from our OpenAPI spec by Stainless. See CONTRIBUTING.md for details. __title__ = "context.dev" -__version__ = "2.11.0" # x-release-please-version +__version__ = "2.12.0" # x-release-please-version diff --git a/src/context/dev/resources/web.py b/src/context/dev/resources/web.py index b24e638..bc0290a 100644 --- a/src/context/dev/resources/web.py +++ b/src/context/dev/resources/web.py @@ -71,6 +71,7 @@ def extract( *, schema: Dict[str, object], url: str, + actions: Iterable[web_extract_params.Action] | Omit = omit, fact_check: bool | Omit = omit, follow_subdomains: bool | Omit = omit, include_frames: bool | Omit = omit, @@ -96,13 +97,20 @@ def extract( relevant internal links, and extract structured data from the selected pages. Args: - schema: JSON Schema for the returned data object. TypeScript Zod users can pass a JSON - Schema generated from a Zod object; Python users can pass the equivalent JSON - Schema object. + schema: JSON Schema for the returned data object. Image fields such as `image_urls` or + `product_photos` automatically make page image references available to + extraction, so product data and photos can be returned in one call. TypeScript + Zod users can pass a JSON Schema generated from a Zod object; Python users can + pass the equivalent JSON Schema object. url: The starting website URL to crawl and extract from. Must include http:// or https://. + actions: Optional browser actions executed in order on the requested page after it loads, + before links are discovered or additional pages are crawled. Requires a paid + plan. When actions are provided and stopAfterMs is omitted, the crawl budget + defaults to 110000 ms. + fact_check: When true, every returned value must be grounded in facts stated on the page; fields that cannot be supported by the page are returned as null/empty. When false (default), the model may make reasonable inferences and derivations from @@ -130,7 +138,8 @@ def extract( exchange for more stable output on animated pages. stop_after_ms: Soft time budget for the crawl in milliseconds. Min: 10000 (10s). Max: 110000 - (110s). Default: 80000 (80s). + (110s). Defaults to 80000 (80s), or 110000 (110s) when browser actions are + provided. tags: Optional tags for tracking usage. Up to 20 tags, each 1 to 50 characters. @@ -155,6 +164,7 @@ def extract( { "schema": schema, "url": url, + "actions": actions, "fact_check": fact_check, "follow_subdomains": follow_subdomains, "include_frames": include_frames, @@ -1338,13 +1348,15 @@ def web_crawl_md( than this value, it will be aborted with a 408 status code. Maximum allowed value is 300000ms (5 minutes). - url_regex: Regex pattern. Only URLs matching this pattern will be followed and scraped. + url_regex: Regex pattern. Only URLs matching this pattern will be followed and scraped. An + automatic prefix scope in the form ^ follows a redirect of the + starting page. use_main_content_only: Extract only the main content, stripping headers, footers, sidebars, and navigation - wait_for_ms: Optional browser wait time in milliseconds after initial page load for each - crawled page. Min: 0. Max: 30000 (30 seconds). + wait_for_ms: Browser wait time in milliseconds after initial page load for each crawled page. + Defaults to 3500 (3.5 seconds). Min: 0. Max: 30000 (30 seconds). zdr: Set to enabled to bypass shared caches and omit request and response content from retained usage logs. Requires zero data retention to be enabled for your @@ -2309,6 +2321,7 @@ async def extract( *, schema: Dict[str, object], url: str, + actions: Iterable[web_extract_params.Action] | Omit = omit, fact_check: bool | Omit = omit, follow_subdomains: bool | Omit = omit, include_frames: bool | Omit = omit, @@ -2334,13 +2347,20 @@ async def extract( relevant internal links, and extract structured data from the selected pages. Args: - schema: JSON Schema for the returned data object. TypeScript Zod users can pass a JSON - Schema generated from a Zod object; Python users can pass the equivalent JSON - Schema object. + schema: JSON Schema for the returned data object. Image fields such as `image_urls` or + `product_photos` automatically make page image references available to + extraction, so product data and photos can be returned in one call. TypeScript + Zod users can pass a JSON Schema generated from a Zod object; Python users can + pass the equivalent JSON Schema object. url: The starting website URL to crawl and extract from. Must include http:// or https://. + actions: Optional browser actions executed in order on the requested page after it loads, + before links are discovered or additional pages are crawled. Requires a paid + plan. When actions are provided and stopAfterMs is omitted, the crawl budget + defaults to 110000 ms. + fact_check: When true, every returned value must be grounded in facts stated on the page; fields that cannot be supported by the page are returned as null/empty. When false (default), the model may make reasonable inferences and derivations from @@ -2368,7 +2388,8 @@ async def extract( exchange for more stable output on animated pages. stop_after_ms: Soft time budget for the crawl in milliseconds. Min: 10000 (10s). Max: 110000 - (110s). Default: 80000 (80s). + (110s). Defaults to 80000 (80s), or 110000 (110s) when browser actions are + provided. tags: Optional tags for tracking usage. Up to 20 tags, each 1 to 50 characters. @@ -2393,6 +2414,7 @@ async def extract( { "schema": schema, "url": url, + "actions": actions, "fact_check": fact_check, "follow_subdomains": follow_subdomains, "include_frames": include_frames, @@ -3576,13 +3598,15 @@ async def web_crawl_md( than this value, it will be aborted with a 408 status code. Maximum allowed value is 300000ms (5 minutes). - url_regex: Regex pattern. Only URLs matching this pattern will be followed and scraped. + url_regex: Regex pattern. Only URLs matching this pattern will be followed and scraped. An + automatic prefix scope in the form ^ follows a redirect of the + starting page. use_main_content_only: Extract only the main content, stripping headers, footers, sidebars, and navigation - wait_for_ms: Optional browser wait time in milliseconds after initial page load for each - crawled page. Min: 0. Max: 30000 (30 seconds). + wait_for_ms: Browser wait time in milliseconds after initial page load for each crawled page. + Defaults to 3500 (3.5 seconds). Min: 0. Max: 30000 (30 seconds). zdr: Set to enabled to bypass shared caches and omit request and response content from retained usage logs. Requires zero data retention to be enabled for your diff --git a/src/context/dev/types/ai_extract_product_response.py b/src/context/dev/types/ai_extract_product_response.py index d56e8d4..24c2ce1 100644 --- a/src/context/dev/types/ai_extract_product_response.py +++ b/src/context/dev/types/ai_extract_product_response.py @@ -45,6 +45,13 @@ class Product(BaseModel): target_audience: List[str] """Target audience for the product (array of strings)""" + availability: Optional[ + Literal[ + "in_stock", "out_of_stock", "limited_availability", "preorder", "backorder", "made_to_order", "discontinued" + ] + ] = None + """Normalized stock or ordering availability""" + billing_frequency: Optional[Literal["monthly", "yearly", "one_time", "usage_based"]] = None """Billing frequency for the product""" @@ -54,6 +61,11 @@ class Product(BaseModel): currency: Optional[str] = None """Currency code for the price (e.g., USD, EUR)""" + dimensions: Optional[List[str]] = None + """ + Dimension statements shown for the product, preserving labels, values, and units + """ + image_url: Optional[str] = None """URL to the product image""" @@ -63,6 +75,9 @@ class Product(BaseModel): pricing_model: Optional[Literal["per_seat", "flat", "tiered", "freemium", "custom"]] = None """Pricing model for the product""" + regular_price: Optional[float] = None + """Original or regular price before a displayed discount""" + url: Optional[str] = None """URL to the product page""" diff --git a/src/context/dev/types/ai_extract_products_response.py b/src/context/dev/types/ai_extract_products_response.py index c26bdb1..89d78f2 100644 --- a/src/context/dev/types/ai_extract_products_response.py +++ b/src/context/dev/types/ai_extract_products_response.py @@ -43,6 +43,13 @@ class Product(BaseModel): target_audience: List[str] """Target audience for the product (array of strings)""" + availability: Optional[ + Literal[ + "in_stock", "out_of_stock", "limited_availability", "preorder", "backorder", "made_to_order", "discontinued" + ] + ] = None + """Normalized stock or ordering availability""" + billing_frequency: Optional[Literal["monthly", "yearly", "one_time", "usage_based"]] = None """Billing frequency for the product""" @@ -52,6 +59,11 @@ class Product(BaseModel): currency: Optional[str] = None """Currency code for the price (e.g., USD, EUR)""" + dimensions: Optional[List[str]] = None + """ + Dimension statements shown for the product, preserving labels, values, and units + """ + image_url: Optional[str] = None """URL to the product image""" @@ -61,6 +73,9 @@ class Product(BaseModel): pricing_model: Optional[Literal["per_seat", "flat", "tiered", "freemium", "custom"]] = None """Pricing model for the product""" + regular_price: Optional[float] = None + """Original or regular price before a displayed discount""" + url: Optional[str] = None """URL to the product page""" diff --git a/src/context/dev/types/news_search_params.py b/src/context/dev/types/news_search_params.py index a7136e6..72a1a64 100644 --- a/src/context/dev/types/news_search_params.py +++ b/src/context/dev/types/news_search_params.py @@ -218,6 +218,7 @@ class FilterBy(TypedDict, total=False): "cg", "ch", "cl", + "cz", "de", "fi", "fr", diff --git a/src/context/dev/types/web_extract_params.py b/src/context/dev/types/web_extract_params.py index 16de5c2..6799f35 100644 --- a/src/context/dev/types/web_extract_params.py +++ b/src/context/dev/types/web_extract_params.py @@ -2,21 +2,30 @@ from __future__ import annotations -from typing import Dict -from typing_extensions import Required, Annotated, TypedDict +from typing import Dict, Union, Iterable +from typing_extensions import Literal, Required, Annotated, TypeAlias, TypedDict from .._types import SequenceNotStr from .._utils import PropertyInfo -__all__ = ["WebExtractParams", "Pdf"] +__all__ = [ + "WebExtractParams", + "Action", + "ActionWebScrapeWaitAction", + "ActionWebScrapePerformAction", + "ActionWebScrapeScrollAction", + "Pdf", +] class WebExtractParams(TypedDict, total=False): schema: Required[Dict[str, object]] """JSON Schema for the returned data object. - TypeScript Zod users can pass a JSON Schema generated from a Zod object; Python - users can pass the equivalent JSON Schema object. + Image fields such as `image_urls` or `product_photos` automatically make page + image references available to extraction, so product data and photos can be + returned in one call. TypeScript Zod users can pass a JSON Schema generated from + a Zod object; Python users can pass the equivalent JSON Schema object. """ url: Required[str] @@ -25,6 +34,14 @@ class WebExtractParams(TypedDict, total=False): Must include http:// or https://. """ + actions: Iterable[Action] + """ + Optional browser actions executed in order on the requested page after it loads, + before links are discovered or additional pages are crawled. Requires a paid + plan. When actions are provided and stopAfterMs is omitted, the crawl budget + defaults to 110000 ms. + """ + fact_check: Annotated[bool, PropertyInfo(alias="factCheck")] """ When true, every returned value must be grounded in facts stated on the page; @@ -74,7 +91,8 @@ class WebExtractParams(TypedDict, total=False): stop_after_ms: Annotated[int, PropertyInfo(alias="stopAfterMs")] """Soft time budget for the crawl in milliseconds. - Min: 10000 (10s). Max: 110000 (110s). Default: 80000 (80s). + Min: 10000 (10s). Max: 110000 (110s). Defaults to 80000 (80s), or 110000 (110s) + when browser actions are provided. """ tags: SequenceNotStr[str] @@ -94,6 +112,51 @@ class WebExtractParams(TypedDict, total=False): """ +class ActionWebScrapeWaitAction(TypedDict, total=False): + """Pause for a fixed number of milliseconds before continuing to the next action.""" + + do: Required[Literal["wait"]] + + time_ms: Required[Annotated[int, PropertyInfo(alias="timeMs")]] + + +class ActionWebScrapePerformAction(TypedDict, total=False): + """Resolve and perform one natural-language browser action.""" + + action: Required[str] + + do: Required[Literal["perform"]] + + +class ActionWebScrapeScrollAction(TypedDict, total=False): + """ + Scroll the page or a selected scrollable container, waiting adaptively for content and dimensions to settle after each iteration. + """ + + do: Required[Literal["scroll"]] + + amount: Union[int, Literal["viewport", "max"]] + """Pixels per scroll, one visible viewport, or the current scroll boundary. + + Defaults to viewport. + """ + + container: str + """CSS selector for the first matching scroll container. Defaults to the page.""" + + direction: Literal["up", "down", "left", "right"] + """Direction to scroll. Defaults to down.""" + + max_scrolls: Annotated[int, PropertyInfo(alias="maxScrolls")] + """Maximum scroll iterations. + + Stops early when scrolling and scrollable extent stop changing. Defaults to 1. + """ + + +Action: TypeAlias = Union[ActionWebScrapeWaitAction, ActionWebScrapePerformAction, ActionWebScrapeScrollAction] + + class Pdf(TypedDict, total=False): end: int """Last 1-based PDF page to parse. diff --git a/src/context/dev/types/web_extract_response.py b/src/context/dev/types/web_extract_response.py index 2230fa2..ddae93a 100644 --- a/src/context/dev/types/web_extract_response.py +++ b/src/context/dev/types/web_extract_response.py @@ -1,12 +1,34 @@ # File generated from our OpenAPI spec by Stainless. See CONTRIBUTING.md for details. from typing import Dict, List, Optional +from typing_extensions import Literal from pydantic import Field as FieldInfo from .._models import BaseModel -__all__ = ["WebExtractResponse", "Metadata", "KeyMetadata"] +__all__ = ["WebExtractResponse", "Metadata", "MetadataActionsApplied", "KeyMetadata"] + + +class MetadataActionsApplied(BaseModel): + instruction: str + + status: Literal["applied", "failed", "skipped"] + """Applied means the requested page state was visibly verified. + + Failed means it was not verified. Skipped means it was not attempted. + """ + + completion_evidence: Optional[str] = FieldInfo(alias="completionEvidence", default=None) + """Visible page evidence used to verify an applied action.""" + + duration_ms: Optional[float] = FieldInfo(alias="durationMs", default=None) + + error: Optional[str] = None + + method: Optional[str] = None + + target_description: Optional[str] = FieldInfo(alias="targetDescription", default=None) class Metadata(BaseModel): @@ -26,6 +48,9 @@ class Metadata(BaseModel): num_urls: int = FieldInfo(alias="numUrls") + actions_applied: Optional[List[MetadataActionsApplied]] = FieldInfo(alias="actionsApplied", default=None) + """One verified outcome per requested browser action, in request order.""" + class KeyMetadata(BaseModel): """Metadata about the API key used for the request. diff --git a/src/context/dev/types/web_web_crawl_md_params.py b/src/context/dev/types/web_web_crawl_md_params.py index 9bd49a6..77a9033 100644 --- a/src/context/dev/types/web_web_crawl_md_params.py +++ b/src/context/dev/types/web_web_crawl_md_params.py @@ -308,7 +308,12 @@ class WebWebCrawlMdParams(TypedDict, total=False): """ url_regex: Annotated[str, PropertyInfo(alias="urlRegex")] - """Regex pattern. Only URLs matching this pattern will be followed and scraped.""" + """Regex pattern. + + Only URLs matching this pattern will be followed and scraped. An automatic + prefix scope in the form ^ follows a redirect of the starting + page. + """ use_main_content_only: Annotated[bool, PropertyInfo(alias="useMainContentOnly")] """ @@ -317,9 +322,9 @@ class WebWebCrawlMdParams(TypedDict, total=False): """ wait_for_ms: Annotated[int, PropertyInfo(alias="waitForMs")] - """ - Optional browser wait time in milliseconds after initial page load for each - crawled page. Min: 0. Max: 30000 (30 seconds). + """Browser wait time in milliseconds after initial page load for each crawled page. + + Defaults to 3500 (3.5 seconds). Min: 0. Max: 30000 (30 seconds). """ zdr: Literal["enabled", "disabled"] diff --git a/src/context/dev/types/web_web_scrape_html_params.py b/src/context/dev/types/web_web_scrape_html_params.py index df85c86..05465a1 100644 --- a/src/context/dev/types/web_web_scrape_html_params.py +++ b/src/context/dev/types/web_web_scrape_html_params.py @@ -8,7 +8,14 @@ from .._types import SequenceNotStr from .._utils import PropertyInfo -__all__ = ["WebWebScrapeHTMLParams", "Action", "ActionWebScrapeWaitAction", "ActionWebScrapePerformAction", "Pdf"] +__all__ = [ + "WebWebScrapeHTMLParams", + "Action", + "ActionWebScrapeWaitAction", + "ActionWebScrapePerformAction", + "ActionWebScrapeScrollAction", + "Pdf", +] class WebWebScrapeHTMLParams(TypedDict, total=False): @@ -330,7 +337,33 @@ class ActionWebScrapePerformAction(TypedDict, total=False): do: Required[Literal["perform"]] -Action: TypeAlias = Union[ActionWebScrapeWaitAction, ActionWebScrapePerformAction] +class ActionWebScrapeScrollAction(TypedDict, total=False): + """ + Scroll the page or a selected scrollable container, waiting adaptively for content and dimensions to settle after each iteration. + """ + + do: Required[Literal["scroll"]] + + amount: Union[int, Literal["viewport", "max"]] + """Pixels per scroll, one visible viewport, or the current scroll boundary. + + Defaults to viewport. + """ + + container: str + """CSS selector for the first matching scroll container. Defaults to the page.""" + + direction: Literal["up", "down", "left", "right"] + """Direction to scroll. Defaults to down.""" + + max_scrolls: Annotated[int, PropertyInfo(alias="maxScrolls")] + """Maximum scroll iterations. + + Stops early when scrolling and scrollable extent stop changing. Defaults to 1. + """ + + +Action: TypeAlias = Union[ActionWebScrapeWaitAction, ActionWebScrapePerformAction, ActionWebScrapeScrollAction] class Pdf(TypedDict, total=False): diff --git a/src/context/dev/types/web_web_scrape_images_params.py b/src/context/dev/types/web_web_scrape_images_params.py index 952dc76..5127f70 100644 --- a/src/context/dev/types/web_web_scrape_images_params.py +++ b/src/context/dev/types/web_web_scrape_images_params.py @@ -13,6 +13,7 @@ "Action", "ActionWebScrapeWaitAction", "ActionWebScrapePerformAction", + "ActionWebScrapeScrollAction", "Enrichment", ] @@ -93,7 +94,33 @@ class ActionWebScrapePerformAction(TypedDict, total=False): do: Required[Literal["perform"]] -Action: TypeAlias = Union[ActionWebScrapeWaitAction, ActionWebScrapePerformAction] +class ActionWebScrapeScrollAction(TypedDict, total=False): + """ + Scroll the page or a selected scrollable container, waiting adaptively for content and dimensions to settle after each iteration. + """ + + do: Required[Literal["scroll"]] + + amount: Union[int, Literal["viewport", "max"]] + """Pixels per scroll, one visible viewport, or the current scroll boundary. + + Defaults to viewport. + """ + + container: str + """CSS selector for the first matching scroll container. Defaults to the page.""" + + direction: Literal["up", "down", "left", "right"] + """Direction to scroll. Defaults to down.""" + + max_scrolls: Annotated[int, PropertyInfo(alias="maxScrolls")] + """Maximum scroll iterations. + + Stops early when scrolling and scrollable extent stop changing. Defaults to 1. + """ + + +Action: TypeAlias = Union[ActionWebScrapeWaitAction, ActionWebScrapePerformAction, ActionWebScrapeScrollAction] class Enrichment(TypedDict, total=False): diff --git a/src/context/dev/types/web_web_scrape_images_response.py b/src/context/dev/types/web_web_scrape_images_response.py index abd62f2..3e432cf 100644 --- a/src/context/dev/types/web_web_scrape_images_response.py +++ b/src/context/dev/types/web_web_scrape_images_response.py @@ -3,9 +3,11 @@ from typing import List, Optional from typing_extensions import Literal +from pydantic import Field as FieldInfo + from .._models import BaseModel -__all__ = ["WebWebScrapeImagesResponse", "Image", "ImageEnrichment", "KeyMetadata"] +__all__ = ["WebWebScrapeImagesResponse", "Image", "ImageEnrichment", "ActionsApplied", "KeyMetadata"] class ImageEnrichment(BaseModel): @@ -46,6 +48,27 @@ class Image(BaseModel): """Requested metadata for images that could be processed.""" +class ActionsApplied(BaseModel): + instruction: str + + status: Literal["applied", "failed", "skipped"] + """Applied means the requested page state was visibly verified. + + Failed means it was not verified. Skipped means it was not attempted. + """ + + completion_evidence: Optional[str] = FieldInfo(alias="completionEvidence", default=None) + """Visible page evidence used to verify an applied action.""" + + duration_ms: Optional[float] = FieldInfo(alias="durationMs", default=None) + + error: Optional[str] = None + + method: Optional[str] = None + + target_description: Optional[str] = FieldInfo(alias="targetDescription", default=None) + + class KeyMetadata(BaseModel): """Metadata about the API key used for the request. @@ -69,6 +92,9 @@ class WebWebScrapeImagesResponse(BaseModel): url: str """Page URL that was scraped.""" + actions_applied: Optional[List[ActionsApplied]] = FieldInfo(alias="actionsApplied", default=None) + """One verified outcome per requested browser action, in request order.""" + key_metadata: Optional[KeyMetadata] = None """Metadata about the API key used for the request. diff --git a/src/context/dev/types/web_web_scrape_md_params.py b/src/context/dev/types/web_web_scrape_md_params.py index e655fd6..86a9b5b 100644 --- a/src/context/dev/types/web_web_scrape_md_params.py +++ b/src/context/dev/types/web_web_scrape_md_params.py @@ -8,7 +8,14 @@ from .._types import SequenceNotStr from .._utils import PropertyInfo -__all__ = ["WebWebScrapeMdParams", "Action", "ActionWebScrapeWaitAction", "ActionWebScrapePerformAction", "Pdf"] +__all__ = [ + "WebWebScrapeMdParams", + "Action", + "ActionWebScrapeWaitAction", + "ActionWebScrapePerformAction", + "ActionWebScrapeScrollAction", + "Pdf", +] class WebWebScrapeMdParams(TypedDict, total=False): @@ -349,7 +356,33 @@ class ActionWebScrapePerformAction(TypedDict, total=False): do: Required[Literal["perform"]] -Action: TypeAlias = Union[ActionWebScrapeWaitAction, ActionWebScrapePerformAction] +class ActionWebScrapeScrollAction(TypedDict, total=False): + """ + Scroll the page or a selected scrollable container, waiting adaptively for content and dimensions to settle after each iteration. + """ + + do: Required[Literal["scroll"]] + + amount: Union[int, Literal["viewport", "max"]] + """Pixels per scroll, one visible viewport, or the current scroll boundary. + + Defaults to viewport. + """ + + container: str + """CSS selector for the first matching scroll container. Defaults to the page.""" + + direction: Literal["up", "down", "left", "right"] + """Direction to scroll. Defaults to down.""" + + max_scrolls: Annotated[int, PropertyInfo(alias="maxScrolls")] + """Maximum scroll iterations. + + Stops early when scrolling and scrollable extent stop changing. Defaults to 1. + """ + + +Action: TypeAlias = Union[ActionWebScrapeWaitAction, ActionWebScrapePerformAction, ActionWebScrapeScrollAction] class Pdf(TypedDict, total=False): diff --git a/tests/api_resources/test_web.py b/tests/api_resources/test_web.py index 4805129..c28f7d2 100644 --- a/tests/api_resources/test_web.py +++ b/tests/api_resources/test_web.py @@ -54,6 +54,12 @@ def test_method_extract_with_all_params(self, client: ContextDev) -> None: "additionalProperties": "bar", }, url="https://example.com", + actions=[ + { + "do": "wait", + "time_ms": 0, + } + ], fact_check=True, follow_subdomains=True, include_frames=True, @@ -701,6 +707,12 @@ async def test_method_extract_with_all_params(self, async_client: AsyncContextDe "additionalProperties": "bar", }, url="https://example.com", + actions=[ + { + "do": "wait", + "time_ms": 0, + } + ], fact_check=True, follow_subdomains=True, include_frames=True,