Skip to content

Commit 53fa41c

Browse files
author
stlc-bot
committed
feat(scrape): add CSS extraction rules to HTML scraping (#1146)
Stainless-Generated-From: 9c6de15dc33c67452e013870363c056f393e3cfe
1 parent 2fdcf25 commit 53fa41c

4 files changed

Lines changed: 102 additions & 16 deletions

File tree

‎src/context/dev/resources/web.py‎

Lines changed: 44 additions & 16 deletions
Original file line numberDiff line numberDiff line change
@@ -1977,6 +1977,7 @@ def web_scrape_html(
19771977
]
19781978
| Omit = omit,
19791979
exclude_selectors: Optional[SequenceNotStr[str]] | Omit = omit,
1980+
extract_rules: Dict[str, web_web_scrape_html_params.ExtractRules] | Omit = omit,
19801981
headers: Dict[str, str] | Omit = omit,
19811982
include_frames: bool | Omit = omit,
19821983
include_selectors: Optional[SequenceNotStr[str]] | Omit = omit,
@@ -1995,14 +1996,18 @@ def web_scrape_html(
19951996
extra_body: Body | None = None,
19961997
timeout: float | httpx.Timeout | None | NotGiven = not_given,
19971998
) -> WebWebScrapeHTMLResponse:
1998-
"""Scrapes the given URL and returns the raw HTML content of the page.
1999-
2000-
The base
2001-
request costs 1 credit; requests with browser actions cost 2 credits. A request
2002-
that hits its timeoutOpts.milliseconds deadline fails with 408 and is not
2003-
billed, unless timeoutOpts.behavior=return-partial is set — then the page as
2004-
rendered so far is returned with `finalDOMState: "still-loading"` and billed at
2005-
the base cost of 1 credit.
1999+
"""Scrapes the given URL and returns the HTML content of the page.
2000+
2001+
Optional
2002+
extractRules return deterministic structured data in extracted using CSS
2003+
selectors, attributes, lists, and nested rules, without an LLM or additional
2004+
credits. Rules run on the returned HTML after selector and main-content
2005+
filtering. Send extractRules as a JSON-encoded query parameter. The base request
2006+
costs 1 credit; requests with browser actions cost 2 credits. A request that
2007+
hits its timeoutOpts.milliseconds deadline fails with 408 and is not billed,
2008+
unless timeoutOpts.behavior=return-partial is set — then the page as rendered so
2009+
far is returned with `finalDOMState: "still-loading"` and billed at the base
2010+
cost of 1 credit.
20062011
20072012
Args:
20082013
url: Full URL to scrape (must include http:// or https:// protocol)
@@ -2018,6 +2023,14 @@ def web_scrape_html(
20182023
Exclusion takes precedence: an element matching both is removed. Examples:
20192024
"nav", "footer", ".ad-banner", "[aria-hidden=true]".
20202025
2026+
extract_rules: Optional CSS extraction rules applied to the returned HTML after selector and
2027+
main-content filtering. Use selector strings ("h1", "a@href") or objects with
2028+
selector, type (item or list), and output (text, html, @attribute, or nested
2029+
rules). Text whitespace is normalized; html includes the matched element;
2030+
attributes are returned as written. Missing items are null and missing lists are
2031+
empty. CSS only; XPath is not supported. Maximum: 100 fields across 5 levels.
2032+
Send a JSON-encoded string in the extractRules query parameter.
2033+
20212034
headers: Optional outbound HTTP headers forwarded only to the target URL, sent as
20222035
deep-object query params such as headers[X-Custom]=value. When provided, caching
20232036
is bypassed: the result is neither read from nor written to cache.
@@ -2081,6 +2094,7 @@ def web_scrape_html(
20812094
"actions": actions,
20822095
"country": country,
20832096
"exclude_selectors": exclude_selectors,
2097+
"extract_rules": extract_rules,
20842098
"headers": headers,
20852099
"include_frames": include_frames,
20862100
"include_selectors": include_selectors,
@@ -4594,6 +4608,7 @@ async def web_scrape_html(
45944608
]
45954609
| Omit = omit,
45964610
exclude_selectors: Optional[SequenceNotStr[str]] | Omit = omit,
4611+
extract_rules: Dict[str, web_web_scrape_html_params.ExtractRules] | Omit = omit,
45974612
headers: Dict[str, str] | Omit = omit,
45984613
include_frames: bool | Omit = omit,
45994614
include_selectors: Optional[SequenceNotStr[str]] | Omit = omit,
@@ -4612,14 +4627,18 @@ async def web_scrape_html(
46124627
extra_body: Body | None = None,
46134628
timeout: float | httpx.Timeout | None | NotGiven = not_given,
46144629
) -> WebWebScrapeHTMLResponse:
4615-
"""Scrapes the given URL and returns the raw HTML content of the page.
4616-
4617-
The base
4618-
request costs 1 credit; requests with browser actions cost 2 credits. A request
4619-
that hits its timeoutOpts.milliseconds deadline fails with 408 and is not
4620-
billed, unless timeoutOpts.behavior=return-partial is set — then the page as
4621-
rendered so far is returned with `finalDOMState: "still-loading"` and billed at
4622-
the base cost of 1 credit.
4630+
"""Scrapes the given URL and returns the HTML content of the page.
4631+
4632+
Optional
4633+
extractRules return deterministic structured data in extracted using CSS
4634+
selectors, attributes, lists, and nested rules, without an LLM or additional
4635+
credits. Rules run on the returned HTML after selector and main-content
4636+
filtering. Send extractRules as a JSON-encoded query parameter. The base request
4637+
costs 1 credit; requests with browser actions cost 2 credits. A request that
4638+
hits its timeoutOpts.milliseconds deadline fails with 408 and is not billed,
4639+
unless timeoutOpts.behavior=return-partial is set — then the page as rendered so
4640+
far is returned with `finalDOMState: "still-loading"` and billed at the base
4641+
cost of 1 credit.
46234642
46244643
Args:
46254644
url: Full URL to scrape (must include http:// or https:// protocol)
@@ -4635,6 +4654,14 @@ async def web_scrape_html(
46354654
Exclusion takes precedence: an element matching both is removed. Examples:
46364655
"nav", "footer", ".ad-banner", "[aria-hidden=true]".
46374656
4657+
extract_rules: Optional CSS extraction rules applied to the returned HTML after selector and
4658+
main-content filtering. Use selector strings ("h1", "a@href") or objects with
4659+
selector, type (item or list), and output (text, html, @attribute, or nested
4660+
rules). Text whitespace is normalized; html includes the matched element;
4661+
attributes are returned as written. Missing items are null and missing lists are
4662+
empty. CSS only; XPath is not supported. Maximum: 100 fields across 5 levels.
4663+
Send a JSON-encoded string in the extractRules query parameter.
4664+
46384665
headers: Optional outbound HTTP headers forwarded only to the target URL, sent as
46394666
deep-object query params such as headers[X-Custom]=value. When provided, caching
46404667
is bypassed: the result is neither read from nor written to cache.
@@ -4698,6 +4725,7 @@ async def web_scrape_html(
46984725
"actions": actions,
46994726
"country": country,
47004727
"exclude_selectors": exclude_selectors,
4728+
"extract_rules": extract_rules,
47014729
"headers": headers,
47024730
"include_frames": include_frames,
47034731
"include_selectors": include_selectors,

‎src/context/dev/types/web_web_scrape_html_params.py‎

Lines changed: 48 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -14,6 +14,10 @@
1414
"ActionWebScrapeWaitAction",
1515
"ActionWebScrapePerformAction",
1616
"ActionWebScrapeScrollAction",
17+
"ExtractRules",
18+
"ExtractRulesUnionMember1",
19+
"ExtractRulesUnionMember1OutputHTMLExtractionRulesExtractRulesUnionMember1OutputHTMLExtractionRulesItem",
20+
"ExtractRulesUnionMember1OutputHTMLExtractionRulesExtractRulesUnionMember1OutputHTMLExtractionRulesItemUnionMember1",
1721
"Pdf",
1822
"TimeoutOpts",
1923
]
@@ -248,6 +252,17 @@ class WebWebScrapeHTMLParams(TypedDict, total=False):
248252
both is removed. Examples: "nav", "footer", ".ad-banner", "[aria-hidden=true]".
249253
"""
250254

255+
extract_rules: Annotated[Dict[str, ExtractRules], PropertyInfo(alias="extractRules")]
256+
"""
257+
Optional CSS extraction rules applied to the returned HTML after selector and
258+
main-content filtering. Use selector strings ("h1", "a@href") or objects with
259+
selector, type (item or list), and output (text, html, @attribute, or nested
260+
rules). Text whitespace is normalized; html includes the matched element;
261+
attributes are returned as written. Missing items are null and missing lists are
262+
empty. CSS only; XPath is not supported. Maximum: 100 fields across 5 levels.
263+
Send a JSON-encoded string in the extractRules query parameter.
264+
"""
265+
251266
headers: Dict[str, str]
252267
"""
253268
Optional outbound HTTP headers forwarded only to the target URL, sent as
@@ -368,6 +383,39 @@ class ActionWebScrapeScrollAction(TypedDict, total=False):
368383
Action: TypeAlias = Union[ActionWebScrapeWaitAction, ActionWebScrapePerformAction, ActionWebScrapeScrollAction]
369384

370385

386+
class ExtractRulesUnionMember1OutputHTMLExtractionRulesExtractRulesUnionMember1OutputHTMLExtractionRulesItemUnionMember1(
387+
TypedDict, total=False
388+
):
389+
selector: Required[str]
390+
391+
output: Union[Literal["text", "html"], str, object]
392+
393+
type: Literal["item", "list"]
394+
395+
396+
ExtractRulesUnionMember1OutputHTMLExtractionRulesExtractRulesUnionMember1OutputHTMLExtractionRulesItem: TypeAlias = Union[
397+
str,
398+
ExtractRulesUnionMember1OutputHTMLExtractionRulesExtractRulesUnionMember1OutputHTMLExtractionRulesItemUnionMember1,
399+
]
400+
401+
402+
class ExtractRulesUnionMember1(TypedDict, total=False):
403+
selector: Required[str]
404+
405+
output: Union[
406+
Literal["text", "html"],
407+
str,
408+
Dict[
409+
str, ExtractRulesUnionMember1OutputHTMLExtractionRulesExtractRulesUnionMember1OutputHTMLExtractionRulesItem
410+
],
411+
]
412+
413+
type: Literal["item", "list"]
414+
415+
416+
ExtractRules: TypeAlias = Union[str, ExtractRulesUnionMember1]
417+
418+
371419
class Pdf(TypedDict, total=False):
372420
"""PDF parsing controls.
373421

‎src/context/dev/types/web_web_scrape_html_response.py‎

Lines changed: 8 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -218,5 +218,13 @@ class WebWebScrapeHTMLResponse(BaseModel):
218218
afterward.
219219
"""
220220

221+
extracted: Optional[Dict[str, Union[str, List[Union[str, object, None]], object, None]]] = None
222+
"""Present only when extractRules is supplied.
223+
224+
Keys match the requested fields. Values are normalized text, raw attribute
225+
strings, outer HTML, nested objects, or lists. Missing items are null; lists
226+
with no matches are empty. Rules run on the returned HTML after filtering.
227+
"""
228+
221229
key_metadata: Optional[KeyMetadata] = None
222230
"""Credit usage, included whenever a valid API key is provided."""

‎tests/api_resources/test_web.py‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -579,6 +579,7 @@ def test_method_web_scrape_html_with_all_params(self, client: ContextDev) -> Non
579579
],
580580
country="de",
581581
exclude_selectors=["x"],
582+
extract_rules={"foo": "x"},
582583
headers={"foo": "J!"},
583584
include_frames=True,
584585
include_selectors=["x"],
@@ -1371,6 +1372,7 @@ async def test_method_web_scrape_html_with_all_params(self, async_client: AsyncC
13711372
],
13721373
country="de",
13731374
exclude_selectors=["x"],
1375+
extract_rules={"foo": "x"},
13741376
headers={"foo": "J!"},
13751377
include_frames=True,
13761378
include_selectors=["x"],

0 commit comments

Comments
 (0)