@@ -1977,6 +1977,7 @@ def web_scrape_html(
19771977 ]
19781978 | Omit = omit ,
19791979 exclude_selectors : Optional [SequenceNotStr [str ]] | Omit = omit ,
1980+ extract_rules : Dict [str , web_web_scrape_html_params .ExtractRules ] | Omit = omit ,
19801981 headers : Dict [str , str ] | Omit = omit ,
19811982 include_frames : bool | Omit = omit ,
19821983 include_selectors : Optional [SequenceNotStr [str ]] | Omit = omit ,
@@ -1995,14 +1996,18 @@ def web_scrape_html(
19951996 extra_body : Body | None = None ,
19961997 timeout : float | httpx .Timeout | None | NotGiven = not_given ,
19971998 ) -> WebWebScrapeHTMLResponse :
1998- """Scrapes the given URL and returns the raw HTML content of the page.
1999-
2000- The base
2001- request costs 1 credit; requests with browser actions cost 2 credits. A request
2002- that hits its timeoutOpts.milliseconds deadline fails with 408 and is not
2003- billed, unless timeoutOpts.behavior=return-partial is set — then the page as
2004- rendered so far is returned with `finalDOMState: "still-loading"` and billed at
2005- the base cost of 1 credit.
1999+ """Scrapes the given URL and returns the HTML content of the page.
2000+
2001+ Optional
2002+ extractRules return deterministic structured data in extracted using CSS
2003+ selectors, attributes, lists, and nested rules, without an LLM or additional
2004+ credits. Rules run on the returned HTML after selector and main-content
2005+ filtering. Send extractRules as a JSON-encoded query parameter. The base request
2006+ costs 1 credit; requests with browser actions cost 2 credits. A request that
2007+ hits its timeoutOpts.milliseconds deadline fails with 408 and is not billed,
2008+ unless timeoutOpts.behavior=return-partial is set — then the page as rendered so
2009+ far is returned with `finalDOMState: "still-loading"` and billed at the base
2010+ cost of 1 credit.
20062011
20072012 Args:
20082013 url: Full URL to scrape (must include http:// or https:// protocol)
@@ -2018,6 +2023,14 @@ def web_scrape_html(
20182023 Exclusion takes precedence: an element matching both is removed. Examples:
20192024 "nav", "footer", ".ad-banner", "[aria-hidden=true]".
20202025
2026+ extract_rules: Optional CSS extraction rules applied to the returned HTML after selector and
2027+ main-content filtering. Use selector strings ("h1", "a@href") or objects with
2028+ selector, type (item or list), and output (text, html, @attribute, or nested
2029+ rules). Text whitespace is normalized; html includes the matched element;
2030+ attributes are returned as written. Missing items are null and missing lists are
2031+ empty. CSS only; XPath is not supported. Maximum: 100 fields across 5 levels.
2032+ Send a JSON-encoded string in the extractRules query parameter.
2033+
20212034 headers: Optional outbound HTTP headers forwarded only to the target URL, sent as
20222035 deep-object query params such as headers[X-Custom]=value. When provided, caching
20232036 is bypassed: the result is neither read from nor written to cache.
@@ -2081,6 +2094,7 @@ def web_scrape_html(
20812094 "actions" : actions ,
20822095 "country" : country ,
20832096 "exclude_selectors" : exclude_selectors ,
2097+ "extract_rules" : extract_rules ,
20842098 "headers" : headers ,
20852099 "include_frames" : include_frames ,
20862100 "include_selectors" : include_selectors ,
@@ -4594,6 +4608,7 @@ async def web_scrape_html(
45944608 ]
45954609 | Omit = omit ,
45964610 exclude_selectors : Optional [SequenceNotStr [str ]] | Omit = omit ,
4611+ extract_rules : Dict [str , web_web_scrape_html_params .ExtractRules ] | Omit = omit ,
45974612 headers : Dict [str , str ] | Omit = omit ,
45984613 include_frames : bool | Omit = omit ,
45994614 include_selectors : Optional [SequenceNotStr [str ]] | Omit = omit ,
@@ -4612,14 +4627,18 @@ async def web_scrape_html(
46124627 extra_body : Body | None = None ,
46134628 timeout : float | httpx .Timeout | None | NotGiven = not_given ,
46144629 ) -> WebWebScrapeHTMLResponse :
4615- """Scrapes the given URL and returns the raw HTML content of the page.
4616-
4617- The base
4618- request costs 1 credit; requests with browser actions cost 2 credits. A request
4619- that hits its timeoutOpts.milliseconds deadline fails with 408 and is not
4620- billed, unless timeoutOpts.behavior=return-partial is set — then the page as
4621- rendered so far is returned with `finalDOMState: "still-loading"` and billed at
4622- the base cost of 1 credit.
4630+ """Scrapes the given URL and returns the HTML content of the page.
4631+
4632+ Optional
4633+ extractRules return deterministic structured data in extracted using CSS
4634+ selectors, attributes, lists, and nested rules, without an LLM or additional
4635+ credits. Rules run on the returned HTML after selector and main-content
4636+ filtering. Send extractRules as a JSON-encoded query parameter. The base request
4637+ costs 1 credit; requests with browser actions cost 2 credits. A request that
4638+ hits its timeoutOpts.milliseconds deadline fails with 408 and is not billed,
4639+ unless timeoutOpts.behavior=return-partial is set — then the page as rendered so
4640+ far is returned with `finalDOMState: "still-loading"` and billed at the base
4641+ cost of 1 credit.
46234642
46244643 Args:
46254644 url: Full URL to scrape (must include http:// or https:// protocol)
@@ -4635,6 +4654,14 @@ async def web_scrape_html(
46354654 Exclusion takes precedence: an element matching both is removed. Examples:
46364655 "nav", "footer", ".ad-banner", "[aria-hidden=true]".
46374656
4657+ extract_rules: Optional CSS extraction rules applied to the returned HTML after selector and
4658+ main-content filtering. Use selector strings ("h1", "a@href") or objects with
4659+ selector, type (item or list), and output (text, html, @attribute, or nested
4660+ rules). Text whitespace is normalized; html includes the matched element;
4661+ attributes are returned as written. Missing items are null and missing lists are
4662+ empty. CSS only; XPath is not supported. Maximum: 100 fields across 5 levels.
4663+ Send a JSON-encoded string in the extractRules query parameter.
4664+
46384665 headers: Optional outbound HTTP headers forwarded only to the target URL, sent as
46394666 deep-object query params such as headers[X-Custom]=value. When provided, caching
46404667 is bypassed: the result is neither read from nor written to cache.
@@ -4698,6 +4725,7 @@ async def web_scrape_html(
46984725 "actions" : actions ,
46994726 "country" : country ,
47004727 "exclude_selectors" : exclude_selectors ,
4728+ "extract_rules" : extract_rules ,
47014729 "headers" : headers ,
47024730 "include_frames" : include_frames ,
47034731 "include_selectors" : include_selectors ,
0 commit comments