Skip to content

Commit 4432665

Browse files
feat(api): api update
1 parent 94fa1f1 commit 4432665

6 files changed

Lines changed: 202 additions & 40 deletions

File tree

‎.stats.yml‎

Lines changed: 2 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -1,4 +1,4 @@
11
configured_endpoints: 32
2-
openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev/context.dev-48449a9e476c712dc65557934c8b97ac9708b06b7f96fa5c22d135037eeb9ce5.yml
3-
openapi_spec_hash: 6a0575b3dc2ea3e5d47c4961f0251213
2+
openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev/context.dev-1ad134d4ef2ddefee790dc464dd2da3e4c0614652620b25fbb1379d8f5676526.yml
3+
openapi_spec_hash: 24809b018c6869838112f064e5558509
44
config_hash: 70e7e80b5e87f94981bee396c6cd41e8

‎src/context/dev/resources/web.py‎

Lines changed: 71 additions & 29 deletions
Original file line numberDiff line numberDiff line change
@@ -2,7 +2,7 @@
22

33
from __future__ import annotations
44

5-
from typing import Dict, Union, Optional
5+
from typing import Dict, Union, Iterable, Optional
66
from typing_extensions import Literal
77

88
import httpx
@@ -1391,6 +1391,7 @@ def web_scrape_html(
13911391
self,
13921392
*,
13931393
url: str,
1394+
actions: Optional[Iterable[web_web_scrape_html_params.Action]] | Omit = omit,
13941395
country: Literal[
13951396
"ad",
13961397
"ae",
@@ -1617,12 +1618,18 @@ def web_scrape_html(
16171618
extra_body: Body | None = None,
16181619
timeout: float | httpx.Timeout | None | NotGiven = not_given,
16191620
) -> WebWebScrapeHTMLResponse:
1620-
"""
1621-
Scrapes the given URL and returns the raw HTML content of the page.
1621+
"""Scrapes the given URL and returns the raw HTML content of the page.
1622+
1623+
The base
1624+
request costs 1 credit; requests with browser actions cost 2 credits.
16221625
16231626
Args:
16241627
url: Full URL to scrape (must include http:// or https:// protocol)
16251628
1629+
actions: Optional browser actions executed in array order after the page loads and before
1630+
content is captured. Requires a paid plan. Send a JSON array in the query
1631+
parameter. Maximum: 5 actions.
1632+
16261633
country: Two-letter ISO 3166-1 alpha-2 country code identifying a supported Context.dev
16271634
residential proxy exit location. Must be one of Context.dev's supported
16281635
countries. When provided, Context.dev fetches the target page from that country.
@@ -1690,6 +1697,7 @@ def web_scrape_html(
16901697
query=maybe_transform(
16911698
{
16921699
"url": url,
1700+
"actions": actions,
16931701
"country": country,
16941702
"exclude_selectors": exclude_selectors,
16951703
"headers": headers,
@@ -1714,6 +1722,7 @@ def web_scrape_images(
17141722
self,
17151723
*,
17161724
url: str,
1725+
actions: Optional[Iterable[web_web_scrape_images_params.Action]] | Omit = omit,
17171726
dedupe: Union[bool, Literal["true", "false"]] | Omit = omit,
17181727
enrichment: Optional[web_web_scrape_images_params.Enrichment] | Omit = omit,
17191728
headers: Dict[str, str] | Omit = omit,
@@ -1731,12 +1740,17 @@ def web_scrape_images(
17311740
"""
17321741
Extract image assets from a web page, including standard URLs, inline SVGs, data
17331742
URIs, responsive image sources, metadata, CSS backgrounds, video posters, and
1734-
embeds. The base request costs 1 credit. When enrichment is enabled, the entire
1735-
call costs 5 credits.
1743+
embeds. The base request costs 1 credit, or 2 credits with browser actions. When
1744+
enrichment is enabled, the entire call costs 5 credits, including requests that
1745+
also use actions.
17361746
17371747
Args:
17381748
url: Page URL to inspect. Must include http:// or https://.
17391749
1750+
actions: Optional browser actions executed in array order after the page loads and before
1751+
content is captured. Requires a paid plan. Send a JSON array in the query
1752+
parameter. Maximum: 5 actions.
1753+
17401754
dedupe: When true, visually duplicate images are removed: every image is loaded and
17411755
perceptually hashed, and only the highest-resolution copy of each duplicate
17421756
group is kept. Images that cannot be downloaded or hashed are kept. Default:
@@ -1781,6 +1795,7 @@ def web_scrape_images(
17811795
query=maybe_transform(
17821796
{
17831797
"url": url,
1798+
"actions": actions,
17841799
"dedupe": dedupe,
17851800
"enrichment": enrichment,
17861801
"headers": headers,
@@ -1799,6 +1814,7 @@ def web_scrape_md(
17991814
self,
18001815
*,
18011816
url: str,
1817+
actions: Optional[Iterable[web_web_scrape_md_params.Action]] | Omit = omit,
18021818
country: Literal[
18031819
"ad",
18041820
"ae",
@@ -2036,21 +2052,25 @@ def web_scrape_md(
20362052
20372053
### Billing & errors
20382054
2039-
| HTTP status | Billed? | Meaning |
2040-
| ----------- | -------------- | ---------------------------------------------------------------------------------------- |
2041-
| 200 | Yes — 1 credit | Successful scrape, including a zero-length result when includeSelectors matched nothing |
2042-
| 400 | No | Invalid input, skipped PDF, or the page could not be scraped |
2043-
| 401 / 403 | No | Invalid/disabled key, insufficient permissions, or credits exhausted; inspect error_code |
2044-
| 404 | No | Target page returned or fingerprinted as not found |
2045-
| 408 | No | Request timed out |
2046-
| 415 | No | Unsupported content type |
2047-
| 429 | No | Per-minute rate limit exceeded; honor Retry-After |
2048-
| 500 | No | Internal error |
2055+
| HTTP status | Billed? | Meaning |
2056+
| ----------- | ----------------------------------------- | ---------------------------------------------------------------------------------------- |
2057+
| 200 | Yes — 1 credit, or 2 credits with actions | Successful scrape, including a zero-length result when includeSelectors matched nothing |
2058+
| 400 | No | Invalid input, skipped PDF, or the page could not be scraped |
2059+
| 401 / 403 | No | Invalid/disabled key, insufficient permissions, or credits exhausted; inspect error_code |
2060+
| 404 | No | Target page returned or fingerprinted as not found |
2061+
| 408 | No | Request timed out |
2062+
| 415 | No | Unsupported content type |
2063+
| 429 | No | Per-minute rate limit exceeded; honor Retry-After |
2064+
| 500 | No | Internal error |
20492065
20502066
Args:
20512067
url: Full URL to scrape into LLM usable Markdown (must include http:// or https://
20522068
protocol)
20532069
2070+
actions: Optional browser actions executed in array order after the page loads and before
2071+
content is captured. Requires a paid plan. Send a JSON array in the query
2072+
parameter. Maximum: 5 actions.
2073+
20542074
country: Two-letter ISO 3166-1 alpha-2 country code identifying a supported Context.dev
20552075
residential proxy exit location. Must be one of Context.dev's supported
20562076
countries. When provided, Context.dev fetches the target page from that country.
@@ -2123,6 +2143,7 @@ def web_scrape_md(
21232143
query=maybe_transform(
21242144
{
21252145
"url": url,
2146+
"actions": actions,
21262147
"country": country,
21272148
"exclude_selectors": exclude_selectors,
21282149
"headers": headers,
@@ -3574,6 +3595,7 @@ async def web_scrape_html(
35743595
self,
35753596
*,
35763597
url: str,
3598+
actions: Optional[Iterable[web_web_scrape_html_params.Action]] | Omit = omit,
35773599
country: Literal[
35783600
"ad",
35793601
"ae",
@@ -3800,12 +3822,18 @@ async def web_scrape_html(
38003822
extra_body: Body | None = None,
38013823
timeout: float | httpx.Timeout | None | NotGiven = not_given,
38023824
) -> WebWebScrapeHTMLResponse:
3803-
"""
3804-
Scrapes the given URL and returns the raw HTML content of the page.
3825+
"""Scrapes the given URL and returns the raw HTML content of the page.
3826+
3827+
The base
3828+
request costs 1 credit; requests with browser actions cost 2 credits.
38053829
38063830
Args:
38073831
url: Full URL to scrape (must include http:// or https:// protocol)
38083832
3833+
actions: Optional browser actions executed in array order after the page loads and before
3834+
content is captured. Requires a paid plan. Send a JSON array in the query
3835+
parameter. Maximum: 5 actions.
3836+
38093837
country: Two-letter ISO 3166-1 alpha-2 country code identifying a supported Context.dev
38103838
residential proxy exit location. Must be one of Context.dev's supported
38113839
countries. When provided, Context.dev fetches the target page from that country.
@@ -3873,6 +3901,7 @@ async def web_scrape_html(
38733901
query=await async_maybe_transform(
38743902
{
38753903
"url": url,
3904+
"actions": actions,
38763905
"country": country,
38773906
"exclude_selectors": exclude_selectors,
38783907
"headers": headers,
@@ -3897,6 +3926,7 @@ async def web_scrape_images(
38973926
self,
38983927
*,
38993928
url: str,
3929+
actions: Optional[Iterable[web_web_scrape_images_params.Action]] | Omit = omit,
39003930
dedupe: Union[bool, Literal["true", "false"]] | Omit = omit,
39013931
enrichment: Optional[web_web_scrape_images_params.Enrichment] | Omit = omit,
39023932
headers: Dict[str, str] | Omit = omit,
@@ -3914,12 +3944,17 @@ async def web_scrape_images(
39143944
"""
39153945
Extract image assets from a web page, including standard URLs, inline SVGs, data
39163946
URIs, responsive image sources, metadata, CSS backgrounds, video posters, and
3917-
embeds. The base request costs 1 credit. When enrichment is enabled, the entire
3918-
call costs 5 credits.
3947+
embeds. The base request costs 1 credit, or 2 credits with browser actions. When
3948+
enrichment is enabled, the entire call costs 5 credits, including requests that
3949+
also use actions.
39193950
39203951
Args:
39213952
url: Page URL to inspect. Must include http:// or https://.
39223953
3954+
actions: Optional browser actions executed in array order after the page loads and before
3955+
content is captured. Requires a paid plan. Send a JSON array in the query
3956+
parameter. Maximum: 5 actions.
3957+
39233958
dedupe: When true, visually duplicate images are removed: every image is loaded and
39243959
perceptually hashed, and only the highest-resolution copy of each duplicate
39253960
group is kept. Images that cannot be downloaded or hashed are kept. Default:
@@ -3964,6 +3999,7 @@ async def web_scrape_images(
39643999
query=await async_maybe_transform(
39654000
{
39664001
"url": url,
4002+
"actions": actions,
39674003
"dedupe": dedupe,
39684004
"enrichment": enrichment,
39694005
"headers": headers,
@@ -3982,6 +4018,7 @@ async def web_scrape_md(
39824018
self,
39834019
*,
39844020
url: str,
4021+
actions: Optional[Iterable[web_web_scrape_md_params.Action]] | Omit = omit,
39854022
country: Literal[
39864023
"ad",
39874024
"ae",
@@ -4219,21 +4256,25 @@ async def web_scrape_md(
42194256
42204257
### Billing & errors
42214258
4222-
| HTTP status | Billed? | Meaning |
4223-
| ----------- | -------------- | ---------------------------------------------------------------------------------------- |
4224-
| 200 | Yes — 1 credit | Successful scrape, including a zero-length result when includeSelectors matched nothing |
4225-
| 400 | No | Invalid input, skipped PDF, or the page could not be scraped |
4226-
| 401 / 403 | No | Invalid/disabled key, insufficient permissions, or credits exhausted; inspect error_code |
4227-
| 404 | No | Target page returned or fingerprinted as not found |
4228-
| 408 | No | Request timed out |
4229-
| 415 | No | Unsupported content type |
4230-
| 429 | No | Per-minute rate limit exceeded; honor Retry-After |
4231-
| 500 | No | Internal error |
4259+
| HTTP status | Billed? | Meaning |
4260+
| ----------- | ----------------------------------------- | ---------------------------------------------------------------------------------------- |
4261+
| 200 | Yes — 1 credit, or 2 credits with actions | Successful scrape, including a zero-length result when includeSelectors matched nothing |
4262+
| 400 | No | Invalid input, skipped PDF, or the page could not be scraped |
4263+
| 401 / 403 | No | Invalid/disabled key, insufficient permissions, or credits exhausted; inspect error_code |
4264+
| 404 | No | Target page returned or fingerprinted as not found |
4265+
| 408 | No | Request timed out |
4266+
| 415 | No | Unsupported content type |
4267+
| 429 | No | Per-minute rate limit exceeded; honor Retry-After |
4268+
| 500 | No | Internal error |
42324269
42334270
Args:
42344271
url: Full URL to scrape into LLM usable Markdown (must include http:// or https://
42354272
protocol)
42364273
4274+
actions: Optional browser actions executed in array order after the page loads and before
4275+
content is captured. Requires a paid plan. Send a JSON array in the query
4276+
parameter. Maximum: 5 actions.
4277+
42374278
country: Two-letter ISO 3166-1 alpha-2 country code identifying a supported Context.dev
42384279
residential proxy exit location. Must be one of Context.dev's supported
42394280
countries. When provided, Context.dev fetches the target page from that country.
@@ -4306,6 +4347,7 @@ async def web_scrape_md(
43064347
query=await async_maybe_transform(
43074348
{
43084349
"url": url,
4350+
"actions": actions,
43094351
"country": country,
43104352
"exclude_selectors": exclude_selectors,
43114353
"headers": headers,

‎src/context/dev/types/web_web_scrape_html_params.py‎

Lines changed: 29 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -2,19 +2,26 @@
22

33
from __future__ import annotations
44

5-
from typing import Dict, Union, Optional
6-
from typing_extensions import Literal, Required, Annotated, TypedDict
5+
from typing import Dict, Union, Iterable, Optional
6+
from typing_extensions import Literal, Required, Annotated, TypeAlias, TypedDict
77

88
from .._types import SequenceNotStr
99
from .._utils import PropertyInfo
1010

11-
__all__ = ["WebWebScrapeHTMLParams", "Pdf"]
11+
__all__ = ["WebWebScrapeHTMLParams", "Action", "ActionWebScrapeWaitAction", "ActionWebScrapePerformAction", "Pdf"]
1212

1313

1414
class WebWebScrapeHTMLParams(TypedDict, total=False):
1515
url: Required[str]
1616
"""Full URL to scrape (must include http:// or https:// protocol)"""
1717

18+
actions: Optional[Iterable[Action]]
19+
"""
20+
Optional browser actions executed in array order after the page loads and before
21+
content is captured. Requires a paid plan. Send a JSON array in the query
22+
parameter. Maximum: 5 actions.
23+
"""
24+
1825
country: Literal[
1926
"ad",
2027
"ae",
@@ -308,6 +315,25 @@ class WebWebScrapeHTMLParams(TypedDict, total=False):
308315
"""
309316

310317

318+
class ActionWebScrapeWaitAction(TypedDict, total=False):
319+
"""Pause for a fixed number of milliseconds before continuing to the next action."""
320+
321+
do: Required[Literal["wait"]]
322+
323+
time_ms: Required[Annotated[int, PropertyInfo(alias="timeMs")]]
324+
325+
326+
class ActionWebScrapePerformAction(TypedDict, total=False):
327+
"""Resolve and perform one natural-language browser action."""
328+
329+
action: Required[str]
330+
331+
do: Required[Literal["perform"]]
332+
333+
334+
Action: TypeAlias = Union[ActionWebScrapeWaitAction, ActionWebScrapePerformAction]
335+
336+
311337
class Pdf(TypedDict, total=False):
312338
"""PDF parsing controls.
313339

‎src/context/dev/types/web_web_scrape_images_params.py‎

Lines changed: 35 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -2,19 +2,32 @@
22

33
from __future__ import annotations
44

5-
from typing import Dict, Union, Optional
6-
from typing_extensions import Literal, Required, Annotated, TypedDict
5+
from typing import Dict, Union, Iterable, Optional
6+
from typing_extensions import Literal, Required, Annotated, TypeAlias, TypedDict
77

88
from .._types import SequenceNotStr
99
from .._utils import PropertyInfo
1010

11-
__all__ = ["WebWebScrapeImagesParams", "Enrichment"]
11+
__all__ = [
12+
"WebWebScrapeImagesParams",
13+
"Action",
14+
"ActionWebScrapeWaitAction",
15+
"ActionWebScrapePerformAction",
16+
"Enrichment",
17+
]
1218

1319

1420
class WebWebScrapeImagesParams(TypedDict, total=False):
1521
url: Required[str]
1622
"""Page URL to inspect. Must include http:// or https://."""
1723

24+
actions: Optional[Iterable[Action]]
25+
"""
26+
Optional browser actions executed in array order after the page loads and before
27+
content is captured. Requires a paid plan. Send a JSON array in the query
28+
parameter. Maximum: 5 actions.
29+
"""
30+
1831
dedupe: Union[bool, Literal["true", "false"]]
1932
"""
2033
When true, visually duplicate images are removed: every image is loaded and
@@ -64,6 +77,25 @@ class WebWebScrapeImagesParams(TypedDict, total=False):
6477
"""
6578

6679

80+
class ActionWebScrapeWaitAction(TypedDict, total=False):
81+
"""Pause for a fixed number of milliseconds before continuing to the next action."""
82+
83+
do: Required[Literal["wait"]]
84+
85+
time_ms: Required[Annotated[int, PropertyInfo(alias="timeMs")]]
86+
87+
88+
class ActionWebScrapePerformAction(TypedDict, total=False):
89+
"""Resolve and perform one natural-language browser action."""
90+
91+
action: Required[str]
92+
93+
do: Required[Literal["perform"]]
94+
95+
96+
Action: TypeAlias = Union[ActionWebScrapeWaitAction, ActionWebScrapePerformAction]
97+
98+
6799
class Enrichment(TypedDict, total=False):
68100
"""
69101
Optional per-image processing, sent as deep-object query params such as enrichment[resolution]=true.

0 commit comments

Comments
 (0)