Skip to content

Commit cdb570f

Browse files
author
stlc-bot
committed
feat(scrape): add json output format to POST /web/scrape (#1213)
Stainless-Generated-From: dc8dff8070b08bc866dc0f1c079d18b22e91822d
1 parent 9519004 commit cdb570f

4 files changed

Lines changed: 92 additions & 8 deletions

File tree

‎src/context/dev/resources/web.py‎

Lines changed: 20 additions & 8 deletions
Original file line numberDiff line numberDiff line change
@@ -392,6 +392,7 @@ def scrape(
392392
formats: web_scrape_params.Formats,
393393
url: str,
394394
image_params: web_scrape_params.ImageParams | Omit = omit,
395+
json_params: web_scrape_params.JsonParams | Omit = omit,
395396
markdown_params: web_scrape_params.MarkdownParams | Omit = omit,
396397
max_age_ms: int | Omit = omit,
397398
parse_params: web_scrape_params.ParseParams | Omit = omit,
@@ -411,10 +412,12 @@ def scrape(
411412
visit.
412413
413414
Each cache key includes only the settings that affect that output. HTML
414-
is shared with Markdown and parsed fields. Cached outputs can come from
415-
different visits within maxAgeMs; use 0 for a fresh capture. HTML-only requests
416-
use the existing fast acquisition path. One credit per request, including cache
417-
hits, or two with browser actions; PDF OCR adds one credit per recovered page on
415+
is shared with Markdown, parsed fields, and JSON extraction. Cached outputs can
416+
come from different visits within maxAgeMs; use 0 for a fresh capture. HTML-only
417+
requests use the existing fast acquisition path. One credit per request,
418+
including cache hits and missing pages, or two with browser actions; JSON
419+
extraction adds four credits and runs an LLM over the page Markdown on every
420+
request that has text to extract; PDF OCR adds one credit per recovered page on
418421
fresh extraction. Original response bytes and screenshots are limited to 20 MiB
419422
each, screenshots to 40 megapixels, and the combined browser capture to 60 MiB.
420423
@@ -425,6 +428,8 @@ def scrape(
425428
426429
image_params: Image options. Requires formats.images: true.
427430
431+
json_params: Required when formats.json is true.
432+
428433
markdown_params: Markdown options. Requires formats.markdown: true.
429434
430435
max_age_ms: Maximum age of each cached output. Defaults to 1 day; 0 fetches fresh and
@@ -467,6 +472,7 @@ def scrape(
467472
"formats": formats,
468473
"url": url,
469474
"image_params": image_params,
475+
"json_params": json_params,
470476
"markdown_params": markdown_params,
471477
"max_age_ms": max_age_ms,
472478
"parse_params": parse_params,
@@ -1869,6 +1875,7 @@ async def scrape(
18691875
formats: web_scrape_params.Formats,
18701876
url: str,
18711877
image_params: web_scrape_params.ImageParams | Omit = omit,
1878+
json_params: web_scrape_params.JsonParams | Omit = omit,
18721879
markdown_params: web_scrape_params.MarkdownParams | Omit = omit,
18731880
max_age_ms: int | Omit = omit,
18741881
parse_params: web_scrape_params.ParseParams | Omit = omit,
@@ -1888,10 +1895,12 @@ async def scrape(
18881895
visit.
18891896
18901897
Each cache key includes only the settings that affect that output. HTML
1891-
is shared with Markdown and parsed fields. Cached outputs can come from
1892-
different visits within maxAgeMs; use 0 for a fresh capture. HTML-only requests
1893-
use the existing fast acquisition path. One credit per request, including cache
1894-
hits, or two with browser actions; PDF OCR adds one credit per recovered page on
1898+
is shared with Markdown, parsed fields, and JSON extraction. Cached outputs can
1899+
come from different visits within maxAgeMs; use 0 for a fresh capture. HTML-only
1900+
requests use the existing fast acquisition path. One credit per request,
1901+
including cache hits and missing pages, or two with browser actions; JSON
1902+
extraction adds four credits and runs an LLM over the page Markdown on every
1903+
request that has text to extract; PDF OCR adds one credit per recovered page on
18951904
fresh extraction. Original response bytes and screenshots are limited to 20 MiB
18961905
each, screenshots to 40 megapixels, and the combined browser capture to 60 MiB.
18971906
@@ -1902,6 +1911,8 @@ async def scrape(
19021911
19031912
image_params: Image options. Requires formats.images: true.
19041913
1914+
json_params: Required when formats.json is true.
1915+
19051916
markdown_params: Markdown options. Requires formats.markdown: true.
19061917
19071918
max_age_ms: Maximum age of each cached output. Defaults to 1 day; 0 fetches fresh and
@@ -1944,6 +1955,7 @@ async def scrape(
19441955
"formats": formats,
19451956
"url": url,
19461957
"image_params": image_params,
1958+
"json_params": json_params,
19471959
"markdown_params": markdown_params,
19481960
"max_age_ms": max_age_ms,
19491961
"parse_params": parse_params,

‎src/context/dev/types/web_scrape_params.py‎

Lines changed: 32 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -12,6 +12,7 @@
1212
"WebScrapeParams",
1313
"Formats",
1414
"ImageParams",
15+
"JsonParams",
1516
"MarkdownParams",
1617
"ParseParams",
1718
"ParseParamsRules",
@@ -43,6 +44,9 @@ class WebScrapeParams(TypedDict, total=False):
4344
image_params: Annotated[ImageParams, PropertyInfo(alias="imageParams")]
4445
"""Image options. Requires formats.images: true."""
4546

47+
json_params: Annotated[JsonParams, PropertyInfo(alias="jsonParams")]
48+
"""Required when formats.json is true."""
49+
4650
markdown_params: Annotated[MarkdownParams, PropertyInfo(alias="markdownParams")]
4751
"""Markdown options. Requires formats.markdown: true."""
4852

@@ -100,6 +104,14 @@ class Formats(TypedDict, total=False):
100104
images: bool
101105
"""Images found on the page."""
102106

107+
json: bool
108+
"""
109+
Page data extracted by an LLM from the page Markdown into jsonParams.schema;
110+
values carried only in attributes or CSS classes need formats.parse instead.
111+
Adds four credits when the page has text to extract; when shared content filters
112+
leave no text the result is an empty object and only the base price applies.
113+
"""
114+
103115
markdown: bool
104116
"""Page content as Markdown."""
105117

@@ -124,6 +136,26 @@ class ImageParams(TypedDict, total=False):
124136
"""
125137

126138

139+
class JsonParams(TypedDict, total=False):
140+
"""Required when formats.json is true."""
141+
142+
schema: Required[Dict[str, object]]
143+
"""JSON Schema for the returned object.
144+
145+
Must describe a top-level object; at most 50 KB serialized. Optional fields the
146+
page does not state are omitted, or null when their type allows null, while
147+
required non-nullable fields always receive a best-effort value, so prefer
148+
nullable or optional fields for data a page may omit. Zod users can pass the
149+
output of z.toJSONSchema().
150+
"""
151+
152+
instructions: str
153+
"""
154+
Optional guidance on which facts to prioritize or how to interpret schema
155+
fields.
156+
"""
157+
158+
127159
class MarkdownParams(TypedDict, total=False):
128160
"""Markdown options. Requires formats.markdown: true."""
129161

‎src/context/dev/types/web_scrape_response.py‎

Lines changed: 20 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -15,6 +15,7 @@
1515
"HTML",
1616
"Images",
1717
"ImagesData",
18+
"Json",
1819
"Markdown",
1920
"Metadata",
2021
"MetadataAlternate",
@@ -100,6 +101,17 @@ class Images(BaseModel):
100101
requested: bool
101102

102103

104+
class Json(BaseModel):
105+
"""Page data extracted into jsonParams.schema, after shared content filters.
106+
107+
Values are grounded in the page; optional fields the page does not state are omitted, or null when their type allows null. An empty object when the filters leave no text.
108+
"""
109+
110+
data: Optional[Dict[str, object]] = None
111+
112+
requested: bool
113+
114+
103115
class Markdown(BaseModel):
104116
"""Markdown after content filters."""
105117

@@ -237,6 +249,14 @@ class WebScrapeResponse(BaseModel):
237249
images: Images
238250
"""Images after content filters. Empty when none are found."""
239251

252+
json_: Json = FieldInfo(alias="json")
253+
"""Page data extracted into jsonParams.schema, after shared content filters.
254+
255+
Values are grounded in the page; optional fields the page does not state are
256+
omitted, or null when their type allows null. An empty object when the filters
257+
leave no text.
258+
"""
259+
240260
markdown: Markdown
241261
"""Markdown after content filters."""
242262

‎tests/api_resources/test_web.py‎

Lines changed: 20 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -244,6 +244,7 @@ def test_method_scrape_with_all_params(self, client: ContextDev) -> None:
244244
"bytes": True,
245245
"html": True,
246246
"images": True,
247+
"json": True,
247248
"markdown": True,
248249
"parse": True,
249250
"screenshot": True,
@@ -253,6 +254,15 @@ def test_method_scrape_with_all_params(self, client: ContextDev) -> None:
253254
"dedupe": "none",
254255
"enrich": ["dimensions"],
255256
},
257+
json_params={
258+
"schema": {
259+
"type": "bar",
260+
"properties": "bar",
261+
"required": "bar",
262+
"additionalProperties": "bar",
263+
},
264+
"instructions": "instructions",
265+
},
256266
markdown_params={
257267
"include_images": True,
258268
"include_links": True,
@@ -764,6 +774,7 @@ async def test_method_scrape_with_all_params(self, async_client: AsyncContextDev
764774
"bytes": True,
765775
"html": True,
766776
"images": True,
777+
"json": True,
767778
"markdown": True,
768779
"parse": True,
769780
"screenshot": True,
@@ -773,6 +784,15 @@ async def test_method_scrape_with_all_params(self, async_client: AsyncContextDev
773784
"dedupe": "none",
774785
"enrich": ["dimensions"],
775786
},
787+
json_params={
788+
"schema": {
789+
"type": "bar",
790+
"properties": "bar",
791+
"required": "bar",
792+
"additionalProperties": "bar",
793+
},
794+
"instructions": "instructions",
795+
},
776796
markdown_params={
777797
"include_images": True,
778798
"include_links": True,

0 commit comments

Comments
 (0)