From 902897e9c547b404056d11a50ed11fca740d9077 Mon Sep 17 00:00:00 2001
From: "stainless-app[bot]"
<142633134+stainless-app[bot]@users.noreply.github.com>
Date: Sat, 4 Apr 2026 21:34:48 +0000
Subject: [PATCH 1/4] codegen metadata
---
.stats.yml | 4 ++--
1 file changed, 2 insertions(+), 2 deletions(-)
diff --git a/.stats.yml b/.stats.yml
index a0614e1..2b07393 100644
--- a/.stats.yml
+++ b/.stats.yml
@@ -1,4 +1,4 @@
configured_endpoints: 20
-openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev%2Fcontext.dev-56a21db16ac3a797f86daca01a2ea115a0365db4b2f9d8accdec4f4d3ee2eb83.yml
-openapi_spec_hash: bfcef090896da96023c4485a3d69e350
+openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev%2Fcontext.dev-97cdb78dc0d72e9df643a89660f2b0c9687f12c6e4d93f7767f6cfc1b4f2e4c7.yml
+openapi_spec_hash: 92fc94fd8865fabe78c2667490ca3884
config_hash: 38268bb88fc4dcbb8f2f94dd138b5910
From 8cd4279e72cdfc4cdad12d91466f672f4ef86cff Mon Sep 17 00:00:00 2001
From: "stainless-app[bot]"
<142633134+stainless-app[bot]@users.noreply.github.com>
Date: Sat, 4 Apr 2026 21:40:53 +0000
Subject: [PATCH 2/4] codegen metadata
---
.stats.yml | 2 +-
1 file changed, 1 insertion(+), 1 deletion(-)
diff --git a/.stats.yml b/.stats.yml
index 2b07393..3542b11 100644
--- a/.stats.yml
+++ b/.stats.yml
@@ -1,4 +1,4 @@
configured_endpoints: 20
openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev%2Fcontext.dev-97cdb78dc0d72e9df643a89660f2b0c9687f12c6e4d93f7767f6cfc1b4f2e4c7.yml
openapi_spec_hash: 92fc94fd8865fabe78c2667490ca3884
-config_hash: 38268bb88fc4dcbb8f2f94dd138b5910
+config_hash: 51eb368cba05800d9497df4fa318828e
From e40b8117067dece6aaef3ca122b655f1808be1a3 Mon Sep 17 00:00:00 2001
From: "stainless-app[bot]"
<142633134+stainless-app[bot]@users.noreply.github.com>
Date: Sat, 4 Apr 2026 21:42:15 +0000
Subject: [PATCH 3/4] feat(api): manual updates
---
.stats.yml | 4 +-
api.md | 2 +
src/context/dev/resources/web.py | 166 ++++++++++++++++++
src/context/dev/types/__init__.py | 2 +
.../dev/types/web_web_crawl_md_params.py | 45 +++++
.../dev/types/web_web_crawl_md_response.py | 53 ++++++
tests/api_resources/test_web.py | 101 +++++++++++
7 files changed, 371 insertions(+), 2 deletions(-)
create mode 100644 src/context/dev/types/web_web_crawl_md_params.py
create mode 100644 src/context/dev/types/web_web_crawl_md_response.py
diff --git a/.stats.yml b/.stats.yml
index 3542b11..a13a02a 100644
--- a/.stats.yml
+++ b/.stats.yml
@@ -1,4 +1,4 @@
-configured_endpoints: 20
+configured_endpoints: 21
openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev%2Fcontext.dev-97cdb78dc0d72e9df643a89660f2b0c9687f12c6e4d93f7767f6cfc1b4f2e4c7.yml
openapi_spec_hash: 92fc94fd8865fabe78c2667490ca3884
-config_hash: 51eb368cba05800d9497df4fa318828e
+config_hash: 682b89b02a20f5d1c13e2c91ecbcf5ce
diff --git a/api.md b/api.md
index 34b0ee0..4c7de31 100644
--- a/api.md
+++ b/api.md
@@ -5,6 +5,7 @@ Types:
```python
from context.dev.types import (
WebScreenshotResponse,
+ WebWebCrawlMdResponse,
WebWebScrapeHTMLResponse,
WebWebScrapeImagesResponse,
WebWebScrapeMdResponse,
@@ -15,6 +16,7 @@ from context.dev.types import (
Methods:
- client.web.screenshot(\*\*params) -> WebScreenshotResponse
+- client.web.web_crawl_md(\*\*params) -> WebWebCrawlMdResponse
- client.web.web_scrape_html(\*\*params) -> WebWebScrapeHTMLResponse
- client.web.web_scrape_images(\*\*params) -> WebWebScrapeImagesResponse
- client.web.web_scrape_md(\*\*params) -> WebWebScrapeMdResponse
diff --git a/src/context/dev/resources/web.py b/src/context/dev/resources/web.py
index 7e4203b..a32d4f1 100644
--- a/src/context/dev/resources/web.py
+++ b/src/context/dev/resources/web.py
@@ -8,6 +8,7 @@
from ..types import (
web_screenshot_params,
+ web_web_crawl_md_params,
web_web_scrape_md_params,
web_web_scrape_html_params,
web_web_scrape_images_params,
@@ -25,6 +26,7 @@
)
from .._base_client import make_request_options
from ..types.web_screenshot_response import WebScreenshotResponse
+from ..types.web_web_crawl_md_response import WebWebCrawlMdResponse
from ..types.web_web_scrape_md_response import WebWebScrapeMdResponse
from ..types.web_web_scrape_html_response import WebWebScrapeHTMLResponse
from ..types.web_web_scrape_images_response import WebWebScrapeImagesResponse
@@ -119,6 +121,82 @@ def screenshot(
cast_to=WebScreenshotResponse,
)
+ def web_crawl_md(
+ self,
+ *,
+ url: str,
+ follow_subdomains: bool | Omit = omit,
+ include_images: bool | Omit = omit,
+ include_links: bool | Omit = omit,
+ max_depth: int | Omit = omit,
+ max_pages: int | Omit = omit,
+ shorten_base64_images: bool | Omit = omit,
+ url_regex: str | Omit = omit,
+ use_main_content_only: bool | Omit = omit,
+ # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
+ # The extra values given here take precedence over values defined on the client or passed to this method.
+ extra_headers: Headers | None = None,
+ extra_query: Query | None = None,
+ extra_body: Body | None = None,
+ timeout: float | httpx.Timeout | None | NotGiven = not_given,
+ ) -> WebWebCrawlMdResponse:
+ """
+ Performs a crawl starting from a given URL, extracts page content as Markdown,
+ and returns results for all crawled pages. Only follows links within the same
+ domain as the starting URL. Costs 1 credit per successful page crawled.
+
+ Args:
+ url: The starting URL for the crawl (must include http:// or https:// protocol)
+
+ follow_subdomains: When true, follow links on subdomains of the starting URL's domain (e.g.
+ docs.example.com when starting from example.com). www and apex are always
+ treated as equivalent.
+
+ include_images: Include image references in the Markdown output
+
+ include_links: Preserve hyperlinks in the Markdown output
+
+ max_depth: Maximum link depth from the starting URL (0 = only the starting page)
+
+ max_pages: Maximum number of pages to crawl. Hard cap: 500.
+
+ shorten_base64_images: Truncate base64-encoded image data in the Markdown output
+
+ url_regex: Regex pattern. Only URLs matching this pattern will be followed and scraped.
+
+ use_main_content_only: Extract only the main content, stripping headers, footers, sidebars, and
+ navigation
+
+ extra_headers: Send extra headers
+
+ extra_query: Add additional query parameters to the request
+
+ extra_body: Add additional JSON properties to the request
+
+ timeout: Override the client-level default timeout for this request, in seconds
+ """
+ return self._post(
+ "/web/crawl",
+ body=maybe_transform(
+ {
+ "url": url,
+ "follow_subdomains": follow_subdomains,
+ "include_images": include_images,
+ "include_links": include_links,
+ "max_depth": max_depth,
+ "max_pages": max_pages,
+ "shorten_base64_images": shorten_base64_images,
+ "url_regex": url_regex,
+ "use_main_content_only": use_main_content_only,
+ },
+ web_web_crawl_md_params.WebWebCrawlMdParams,
+ ),
+ options=make_request_options(
+ extra_headers=extra_headers, extra_query=extra_query, extra_body=extra_body, timeout=timeout
+ ),
+ cast_to=WebWebCrawlMdResponse,
+ )
+
def web_scrape_html(
self,
*,
@@ -394,6 +472,82 @@ async def screenshot(
cast_to=WebScreenshotResponse,
)
+ async def web_crawl_md(
+ self,
+ *,
+ url: str,
+ follow_subdomains: bool | Omit = omit,
+ include_images: bool | Omit = omit,
+ include_links: bool | Omit = omit,
+ max_depth: int | Omit = omit,
+ max_pages: int | Omit = omit,
+ shorten_base64_images: bool | Omit = omit,
+ url_regex: str | Omit = omit,
+ use_main_content_only: bool | Omit = omit,
+ # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
+ # The extra values given here take precedence over values defined on the client or passed to this method.
+ extra_headers: Headers | None = None,
+ extra_query: Query | None = None,
+ extra_body: Body | None = None,
+ timeout: float | httpx.Timeout | None | NotGiven = not_given,
+ ) -> WebWebCrawlMdResponse:
+ """
+ Performs a crawl starting from a given URL, extracts page content as Markdown,
+ and returns results for all crawled pages. Only follows links within the same
+ domain as the starting URL. Costs 1 credit per successful page crawled.
+
+ Args:
+ url: The starting URL for the crawl (must include http:// or https:// protocol)
+
+ follow_subdomains: When true, follow links on subdomains of the starting URL's domain (e.g.
+ docs.example.com when starting from example.com). www and apex are always
+ treated as equivalent.
+
+ include_images: Include image references in the Markdown output
+
+ include_links: Preserve hyperlinks in the Markdown output
+
+ max_depth: Maximum link depth from the starting URL (0 = only the starting page)
+
+ max_pages: Maximum number of pages to crawl. Hard cap: 500.
+
+ shorten_base64_images: Truncate base64-encoded image data in the Markdown output
+
+ url_regex: Regex pattern. Only URLs matching this pattern will be followed and scraped.
+
+ use_main_content_only: Extract only the main content, stripping headers, footers, sidebars, and
+ navigation
+
+ extra_headers: Send extra headers
+
+ extra_query: Add additional query parameters to the request
+
+ extra_body: Add additional JSON properties to the request
+
+ timeout: Override the client-level default timeout for this request, in seconds
+ """
+ return await self._post(
+ "/web/crawl",
+ body=await async_maybe_transform(
+ {
+ "url": url,
+ "follow_subdomains": follow_subdomains,
+ "include_images": include_images,
+ "include_links": include_links,
+ "max_depth": max_depth,
+ "max_pages": max_pages,
+ "shorten_base64_images": shorten_base64_images,
+ "url_regex": url_regex,
+ "use_main_content_only": use_main_content_only,
+ },
+ web_web_crawl_md_params.WebWebCrawlMdParams,
+ ),
+ options=make_request_options(
+ extra_headers=extra_headers, extra_query=extra_query, extra_body=extra_body, timeout=timeout
+ ),
+ cast_to=WebWebCrawlMdResponse,
+ )
+
async def web_scrape_html(
self,
*,
@@ -590,6 +744,9 @@ def __init__(self, web: WebResource) -> None:
self.screenshot = to_raw_response_wrapper(
web.screenshot,
)
+ self.web_crawl_md = to_raw_response_wrapper(
+ web.web_crawl_md,
+ )
self.web_scrape_html = to_raw_response_wrapper(
web.web_scrape_html,
)
@@ -611,6 +768,9 @@ def __init__(self, web: AsyncWebResource) -> None:
self.screenshot = async_to_raw_response_wrapper(
web.screenshot,
)
+ self.web_crawl_md = async_to_raw_response_wrapper(
+ web.web_crawl_md,
+ )
self.web_scrape_html = async_to_raw_response_wrapper(
web.web_scrape_html,
)
@@ -632,6 +792,9 @@ def __init__(self, web: WebResource) -> None:
self.screenshot = to_streamed_response_wrapper(
web.screenshot,
)
+ self.web_crawl_md = to_streamed_response_wrapper(
+ web.web_crawl_md,
+ )
self.web_scrape_html = to_streamed_response_wrapper(
web.web_scrape_html,
)
@@ -653,6 +816,9 @@ def __init__(self, web: AsyncWebResource) -> None:
self.screenshot = async_to_streamed_response_wrapper(
web.screenshot,
)
+ self.web_crawl_md = async_to_streamed_response_wrapper(
+ web.web_crawl_md,
+ )
self.web_scrape_html = async_to_streamed_response_wrapper(
web.web_scrape_html,
)
diff --git a/src/context/dev/types/__init__.py b/src/context/dev/types/__init__.py
index c485ed5..4053426 100644
--- a/src/context/dev/types/__init__.py
+++ b/src/context/dev/types/__init__.py
@@ -9,9 +9,11 @@
from .brand_retrieve_response import BrandRetrieveResponse as BrandRetrieveResponse
from .utility_prefetch_params import UtilityPrefetchParams as UtilityPrefetchParams
from .web_screenshot_response import WebScreenshotResponse as WebScreenshotResponse
+from .web_web_crawl_md_params import WebWebCrawlMdParams as WebWebCrawlMdParams
from .web_web_scrape_md_params import WebWebScrapeMdParams as WebWebScrapeMdParams
from .ai_extract_product_params import AIExtractProductParams as AIExtractProductParams
from .utility_prefetch_response import UtilityPrefetchResponse as UtilityPrefetchResponse
+from .web_web_crawl_md_response import WebWebCrawlMdResponse as WebWebCrawlMdResponse
from .ai_extract_products_params import AIExtractProductsParams as AIExtractProductsParams
from .style_extract_fonts_params import StyleExtractFontsParams as StyleExtractFontsParams
from .web_web_scrape_html_params import WebWebScrapeHTMLParams as WebWebScrapeHTMLParams
diff --git a/src/context/dev/types/web_web_crawl_md_params.py b/src/context/dev/types/web_web_crawl_md_params.py
new file mode 100644
index 0000000..cbd849e
--- /dev/null
+++ b/src/context/dev/types/web_web_crawl_md_params.py
@@ -0,0 +1,45 @@
+# File generated from our OpenAPI spec by Stainless. See CONTRIBUTING.md for details.
+
+from __future__ import annotations
+
+from typing_extensions import Required, Annotated, TypedDict
+
+from .._utils import PropertyInfo
+
+__all__ = ["WebWebCrawlMdParams"]
+
+
+class WebWebCrawlMdParams(TypedDict, total=False):
+ url: Required[str]
+ """The starting URL for the crawl (must include http:// or https:// protocol)"""
+
+ follow_subdomains: Annotated[bool, PropertyInfo(alias="followSubdomains")]
+ """When true, follow links on subdomains of the starting URL's domain (e.g.
+
+ docs.example.com when starting from example.com). www and apex are always
+ treated as equivalent.
+ """
+
+ include_images: Annotated[bool, PropertyInfo(alias="includeImages")]
+ """Include image references in the Markdown output"""
+
+ include_links: Annotated[bool, PropertyInfo(alias="includeLinks")]
+ """Preserve hyperlinks in the Markdown output"""
+
+ max_depth: Annotated[int, PropertyInfo(alias="maxDepth")]
+ """Maximum link depth from the starting URL (0 = only the starting page)"""
+
+ max_pages: Annotated[int, PropertyInfo(alias="maxPages")]
+ """Maximum number of pages to crawl. Hard cap: 500."""
+
+ shorten_base64_images: Annotated[bool, PropertyInfo(alias="shortenBase64Images")]
+ """Truncate base64-encoded image data in the Markdown output"""
+
+ url_regex: Annotated[str, PropertyInfo(alias="urlRegex")]
+ """Regex pattern. Only URLs matching this pattern will be followed and scraped."""
+
+ use_main_content_only: Annotated[bool, PropertyInfo(alias="useMainContentOnly")]
+ """
+ Extract only the main content, stripping headers, footers, sidebars, and
+ navigation
+ """
diff --git a/src/context/dev/types/web_web_crawl_md_response.py b/src/context/dev/types/web_web_crawl_md_response.py
new file mode 100644
index 0000000..49ba2d8
--- /dev/null
+++ b/src/context/dev/types/web_web_crawl_md_response.py
@@ -0,0 +1,53 @@
+# File generated from our OpenAPI spec by Stainless. See CONTRIBUTING.md for details.
+
+from typing import List
+
+from pydantic import Field as FieldInfo
+
+from .._models import BaseModel
+
+__all__ = ["WebWebCrawlMdResponse", "Metadata", "Result", "ResultMetadata"]
+
+
+class Metadata(BaseModel):
+ max_crawl_depth: int = FieldInfo(alias="maxCrawlDepth")
+ """Maximum crawl depth reached during the crawl"""
+
+ num_failed: int = FieldInfo(alias="numFailed")
+ """Number of pages that failed to crawl"""
+
+ num_succeeded: int = FieldInfo(alias="numSucceeded")
+ """Number of pages successfully crawled"""
+
+ num_urls: int = FieldInfo(alias="numUrls")
+ """Total number of URLs crawled"""
+
+
+class ResultMetadata(BaseModel):
+ crawl_depth: int = FieldInfo(alias="crawlDepth")
+ """Depth relative to the start URL. 0 = start URL, 1 = one link away."""
+
+ status_code: int = FieldInfo(alias="statusCode")
+ """HTTP status code of the response"""
+
+ success: bool
+ """true if the page was fetched and parsed successfully"""
+
+ title: str
+ """The page's