From 902897e9c547b404056d11a50ed11fca740d9077 Mon Sep 17 00:00:00 2001 From: "stainless-app[bot]" <142633134+stainless-app[bot]@users.noreply.github.com> Date: Sat, 4 Apr 2026 21:34:48 +0000 Subject: [PATCH 1/4] codegen metadata --- .stats.yml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/.stats.yml b/.stats.yml index a0614e1..2b07393 100644 --- a/.stats.yml +++ b/.stats.yml @@ -1,4 +1,4 @@ configured_endpoints: 20 -openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev%2Fcontext.dev-56a21db16ac3a797f86daca01a2ea115a0365db4b2f9d8accdec4f4d3ee2eb83.yml -openapi_spec_hash: bfcef090896da96023c4485a3d69e350 +openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev%2Fcontext.dev-97cdb78dc0d72e9df643a89660f2b0c9687f12c6e4d93f7767f6cfc1b4f2e4c7.yml +openapi_spec_hash: 92fc94fd8865fabe78c2667490ca3884 config_hash: 38268bb88fc4dcbb8f2f94dd138b5910 From 8cd4279e72cdfc4cdad12d91466f672f4ef86cff Mon Sep 17 00:00:00 2001 From: "stainless-app[bot]" <142633134+stainless-app[bot]@users.noreply.github.com> Date: Sat, 4 Apr 2026 21:40:53 +0000 Subject: [PATCH 2/4] codegen metadata --- .stats.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.stats.yml b/.stats.yml index 2b07393..3542b11 100644 --- a/.stats.yml +++ b/.stats.yml @@ -1,4 +1,4 @@ configured_endpoints: 20 openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev%2Fcontext.dev-97cdb78dc0d72e9df643a89660f2b0c9687f12c6e4d93f7767f6cfc1b4f2e4c7.yml openapi_spec_hash: 92fc94fd8865fabe78c2667490ca3884 -config_hash: 38268bb88fc4dcbb8f2f94dd138b5910 +config_hash: 51eb368cba05800d9497df4fa318828e From e40b8117067dece6aaef3ca122b655f1808be1a3 Mon Sep 17 00:00:00 2001 From: "stainless-app[bot]" <142633134+stainless-app[bot]@users.noreply.github.com> Date: Sat, 4 Apr 2026 21:42:15 +0000 Subject: [PATCH 3/4] feat(api): manual updates --- .stats.yml | 4 +- api.md | 2 + src/context/dev/resources/web.py | 166 ++++++++++++++++++ src/context/dev/types/__init__.py | 2 + .../dev/types/web_web_crawl_md_params.py | 45 +++++ .../dev/types/web_web_crawl_md_response.py | 53 ++++++ tests/api_resources/test_web.py | 101 +++++++++++ 7 files changed, 371 insertions(+), 2 deletions(-) create mode 100644 src/context/dev/types/web_web_crawl_md_params.py create mode 100644 src/context/dev/types/web_web_crawl_md_response.py diff --git a/.stats.yml b/.stats.yml index 3542b11..a13a02a 100644 --- a/.stats.yml +++ b/.stats.yml @@ -1,4 +1,4 @@ -configured_endpoints: 20 +configured_endpoints: 21 openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev%2Fcontext.dev-97cdb78dc0d72e9df643a89660f2b0c9687f12c6e4d93f7767f6cfc1b4f2e4c7.yml openapi_spec_hash: 92fc94fd8865fabe78c2667490ca3884 -config_hash: 51eb368cba05800d9497df4fa318828e +config_hash: 682b89b02a20f5d1c13e2c91ecbcf5ce diff --git a/api.md b/api.md index 34b0ee0..4c7de31 100644 --- a/api.md +++ b/api.md @@ -5,6 +5,7 @@ Types: ```python from context.dev.types import ( WebScreenshotResponse, + WebWebCrawlMdResponse, WebWebScrapeHTMLResponse, WebWebScrapeImagesResponse, WebWebScrapeMdResponse, @@ -15,6 +16,7 @@ from context.dev.types import ( Methods: - client.web.screenshot(\*\*params) -> WebScreenshotResponse +- client.web.web_crawl_md(\*\*params) -> WebWebCrawlMdResponse - client.web.web_scrape_html(\*\*params) -> WebWebScrapeHTMLResponse - client.web.web_scrape_images(\*\*params) -> WebWebScrapeImagesResponse - client.web.web_scrape_md(\*\*params) -> WebWebScrapeMdResponse diff --git a/src/context/dev/resources/web.py b/src/context/dev/resources/web.py index 7e4203b..a32d4f1 100644 --- a/src/context/dev/resources/web.py +++ b/src/context/dev/resources/web.py @@ -8,6 +8,7 @@ from ..types import ( web_screenshot_params, + web_web_crawl_md_params, web_web_scrape_md_params, web_web_scrape_html_params, web_web_scrape_images_params, @@ -25,6 +26,7 @@ ) from .._base_client import make_request_options from ..types.web_screenshot_response import WebScreenshotResponse +from ..types.web_web_crawl_md_response import WebWebCrawlMdResponse from ..types.web_web_scrape_md_response import WebWebScrapeMdResponse from ..types.web_web_scrape_html_response import WebWebScrapeHTMLResponse from ..types.web_web_scrape_images_response import WebWebScrapeImagesResponse @@ -119,6 +121,82 @@ def screenshot( cast_to=WebScreenshotResponse, ) + def web_crawl_md( + self, + *, + url: str, + follow_subdomains: bool | Omit = omit, + include_images: bool | Omit = omit, + include_links: bool | Omit = omit, + max_depth: int | Omit = omit, + max_pages: int | Omit = omit, + shorten_base64_images: bool | Omit = omit, + url_regex: str | Omit = omit, + use_main_content_only: bool | Omit = omit, + # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs. + # The extra values given here take precedence over values defined on the client or passed to this method. + extra_headers: Headers | None = None, + extra_query: Query | None = None, + extra_body: Body | None = None, + timeout: float | httpx.Timeout | None | NotGiven = not_given, + ) -> WebWebCrawlMdResponse: + """ + Performs a crawl starting from a given URL, extracts page content as Markdown, + and returns results for all crawled pages. Only follows links within the same + domain as the starting URL. Costs 1 credit per successful page crawled. + + Args: + url: The starting URL for the crawl (must include http:// or https:// protocol) + + follow_subdomains: When true, follow links on subdomains of the starting URL's domain (e.g. + docs.example.com when starting from example.com). www and apex are always + treated as equivalent. + + include_images: Include image references in the Markdown output + + include_links: Preserve hyperlinks in the Markdown output + + max_depth: Maximum link depth from the starting URL (0 = only the starting page) + + max_pages: Maximum number of pages to crawl. Hard cap: 500. + + shorten_base64_images: Truncate base64-encoded image data in the Markdown output + + url_regex: Regex pattern. Only URLs matching this pattern will be followed and scraped. + + use_main_content_only: Extract only the main content, stripping headers, footers, sidebars, and + navigation + + extra_headers: Send extra headers + + extra_query: Add additional query parameters to the request + + extra_body: Add additional JSON properties to the request + + timeout: Override the client-level default timeout for this request, in seconds + """ + return self._post( + "/web/crawl", + body=maybe_transform( + { + "url": url, + "follow_subdomains": follow_subdomains, + "include_images": include_images, + "include_links": include_links, + "max_depth": max_depth, + "max_pages": max_pages, + "shorten_base64_images": shorten_base64_images, + "url_regex": url_regex, + "use_main_content_only": use_main_content_only, + }, + web_web_crawl_md_params.WebWebCrawlMdParams, + ), + options=make_request_options( + extra_headers=extra_headers, extra_query=extra_query, extra_body=extra_body, timeout=timeout + ), + cast_to=WebWebCrawlMdResponse, + ) + def web_scrape_html( self, *, @@ -394,6 +472,82 @@ async def screenshot( cast_to=WebScreenshotResponse, ) + async def web_crawl_md( + self, + *, + url: str, + follow_subdomains: bool | Omit = omit, + include_images: bool | Omit = omit, + include_links: bool | Omit = omit, + max_depth: int | Omit = omit, + max_pages: int | Omit = omit, + shorten_base64_images: bool | Omit = omit, + url_regex: str | Omit = omit, + use_main_content_only: bool | Omit = omit, + # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs. + # The extra values given here take precedence over values defined on the client or passed to this method. + extra_headers: Headers | None = None, + extra_query: Query | None = None, + extra_body: Body | None = None, + timeout: float | httpx.Timeout | None | NotGiven = not_given, + ) -> WebWebCrawlMdResponse: + """ + Performs a crawl starting from a given URL, extracts page content as Markdown, + and returns results for all crawled pages. Only follows links within the same + domain as the starting URL. Costs 1 credit per successful page crawled. + + Args: + url: The starting URL for the crawl (must include http:// or https:// protocol) + + follow_subdomains: When true, follow links on subdomains of the starting URL's domain (e.g. + docs.example.com when starting from example.com). www and apex are always + treated as equivalent. + + include_images: Include image references in the Markdown output + + include_links: Preserve hyperlinks in the Markdown output + + max_depth: Maximum link depth from the starting URL (0 = only the starting page) + + max_pages: Maximum number of pages to crawl. Hard cap: 500. + + shorten_base64_images: Truncate base64-encoded image data in the Markdown output + + url_regex: Regex pattern. Only URLs matching this pattern will be followed and scraped. + + use_main_content_only: Extract only the main content, stripping headers, footers, sidebars, and + navigation + + extra_headers: Send extra headers + + extra_query: Add additional query parameters to the request + + extra_body: Add additional JSON properties to the request + + timeout: Override the client-level default timeout for this request, in seconds + """ + return await self._post( + "/web/crawl", + body=await async_maybe_transform( + { + "url": url, + "follow_subdomains": follow_subdomains, + "include_images": include_images, + "include_links": include_links, + "max_depth": max_depth, + "max_pages": max_pages, + "shorten_base64_images": shorten_base64_images, + "url_regex": url_regex, + "use_main_content_only": use_main_content_only, + }, + web_web_crawl_md_params.WebWebCrawlMdParams, + ), + options=make_request_options( + extra_headers=extra_headers, extra_query=extra_query, extra_body=extra_body, timeout=timeout + ), + cast_to=WebWebCrawlMdResponse, + ) + async def web_scrape_html( self, *, @@ -590,6 +744,9 @@ def __init__(self, web: WebResource) -> None: self.screenshot = to_raw_response_wrapper( web.screenshot, ) + self.web_crawl_md = to_raw_response_wrapper( + web.web_crawl_md, + ) self.web_scrape_html = to_raw_response_wrapper( web.web_scrape_html, ) @@ -611,6 +768,9 @@ def __init__(self, web: AsyncWebResource) -> None: self.screenshot = async_to_raw_response_wrapper( web.screenshot, ) + self.web_crawl_md = async_to_raw_response_wrapper( + web.web_crawl_md, + ) self.web_scrape_html = async_to_raw_response_wrapper( web.web_scrape_html, ) @@ -632,6 +792,9 @@ def __init__(self, web: WebResource) -> None: self.screenshot = to_streamed_response_wrapper( web.screenshot, ) + self.web_crawl_md = to_streamed_response_wrapper( + web.web_crawl_md, + ) self.web_scrape_html = to_streamed_response_wrapper( web.web_scrape_html, ) @@ -653,6 +816,9 @@ def __init__(self, web: AsyncWebResource) -> None: self.screenshot = async_to_streamed_response_wrapper( web.screenshot, ) + self.web_crawl_md = async_to_streamed_response_wrapper( + web.web_crawl_md, + ) self.web_scrape_html = async_to_streamed_response_wrapper( web.web_scrape_html, ) diff --git a/src/context/dev/types/__init__.py b/src/context/dev/types/__init__.py index c485ed5..4053426 100644 --- a/src/context/dev/types/__init__.py +++ b/src/context/dev/types/__init__.py @@ -9,9 +9,11 @@ from .brand_retrieve_response import BrandRetrieveResponse as BrandRetrieveResponse from .utility_prefetch_params import UtilityPrefetchParams as UtilityPrefetchParams from .web_screenshot_response import WebScreenshotResponse as WebScreenshotResponse +from .web_web_crawl_md_params import WebWebCrawlMdParams as WebWebCrawlMdParams from .web_web_scrape_md_params import WebWebScrapeMdParams as WebWebScrapeMdParams from .ai_extract_product_params import AIExtractProductParams as AIExtractProductParams from .utility_prefetch_response import UtilityPrefetchResponse as UtilityPrefetchResponse +from .web_web_crawl_md_response import WebWebCrawlMdResponse as WebWebCrawlMdResponse from .ai_extract_products_params import AIExtractProductsParams as AIExtractProductsParams from .style_extract_fonts_params import StyleExtractFontsParams as StyleExtractFontsParams from .web_web_scrape_html_params import WebWebScrapeHTMLParams as WebWebScrapeHTMLParams diff --git a/src/context/dev/types/web_web_crawl_md_params.py b/src/context/dev/types/web_web_crawl_md_params.py new file mode 100644 index 0000000..cbd849e --- /dev/null +++ b/src/context/dev/types/web_web_crawl_md_params.py @@ -0,0 +1,45 @@ +# File generated from our OpenAPI spec by Stainless. See CONTRIBUTING.md for details. + +from __future__ import annotations + +from typing_extensions import Required, Annotated, TypedDict + +from .._utils import PropertyInfo + +__all__ = ["WebWebCrawlMdParams"] + + +class WebWebCrawlMdParams(TypedDict, total=False): + url: Required[str] + """The starting URL for the crawl (must include http:// or https:// protocol)""" + + follow_subdomains: Annotated[bool, PropertyInfo(alias="followSubdomains")] + """When true, follow links on subdomains of the starting URL's domain (e.g. + + docs.example.com when starting from example.com). www and apex are always + treated as equivalent. + """ + + include_images: Annotated[bool, PropertyInfo(alias="includeImages")] + """Include image references in the Markdown output""" + + include_links: Annotated[bool, PropertyInfo(alias="includeLinks")] + """Preserve hyperlinks in the Markdown output""" + + max_depth: Annotated[int, PropertyInfo(alias="maxDepth")] + """Maximum link depth from the starting URL (0 = only the starting page)""" + + max_pages: Annotated[int, PropertyInfo(alias="maxPages")] + """Maximum number of pages to crawl. Hard cap: 500.""" + + shorten_base64_images: Annotated[bool, PropertyInfo(alias="shortenBase64Images")] + """Truncate base64-encoded image data in the Markdown output""" + + url_regex: Annotated[str, PropertyInfo(alias="urlRegex")] + """Regex pattern. Only URLs matching this pattern will be followed and scraped.""" + + use_main_content_only: Annotated[bool, PropertyInfo(alias="useMainContentOnly")] + """ + Extract only the main content, stripping headers, footers, sidebars, and + navigation + """ diff --git a/src/context/dev/types/web_web_crawl_md_response.py b/src/context/dev/types/web_web_crawl_md_response.py new file mode 100644 index 0000000..49ba2d8 --- /dev/null +++ b/src/context/dev/types/web_web_crawl_md_response.py @@ -0,0 +1,53 @@ +# File generated from our OpenAPI spec by Stainless. See CONTRIBUTING.md for details. + +from typing import List + +from pydantic import Field as FieldInfo + +from .._models import BaseModel + +__all__ = ["WebWebCrawlMdResponse", "Metadata", "Result", "ResultMetadata"] + + +class Metadata(BaseModel): + max_crawl_depth: int = FieldInfo(alias="maxCrawlDepth") + """Maximum crawl depth reached during the crawl""" + + num_failed: int = FieldInfo(alias="numFailed") + """Number of pages that failed to crawl""" + + num_succeeded: int = FieldInfo(alias="numSucceeded") + """Number of pages successfully crawled""" + + num_urls: int = FieldInfo(alias="numUrls") + """Total number of URLs crawled""" + + +class ResultMetadata(BaseModel): + crawl_depth: int = FieldInfo(alias="crawlDepth") + """Depth relative to the start URL. 0 = start URL, 1 = one link away.""" + + status_code: int = FieldInfo(alias="statusCode") + """HTTP status code of the response""" + + success: bool + """true if the page was fetched and parsed successfully""" + + title: str + """The page's content (empty string if unavailable)""" + + url: str + """The URL that was fetched""" + + +class Result(BaseModel): + markdown: str + """Extracted page content as Markdown (empty string on failure)""" + + metadata: ResultMetadata + + +class WebWebCrawlMdResponse(BaseModel): + metadata: Metadata + + results: List[Result] diff --git a/tests/api_resources/test_web.py b/tests/api_resources/test_web.py index 417d5be..e170d5e 100644 --- a/tests/api_resources/test_web.py +++ b/tests/api_resources/test_web.py @@ -11,6 +11,7 @@ from tests.utils import assert_matches_type from context.dev.types import ( WebScreenshotResponse, + WebWebCrawlMdResponse, WebWebScrapeMdResponse, WebWebScrapeHTMLResponse, WebWebScrapeImagesResponse, @@ -68,6 +69,56 @@ def test_streaming_response_screenshot(self, client: ContextDev) -> None: assert cast(Any, response.is_closed) is True + @pytest.mark.skip(reason="Mock server tests are disabled") + @parametrize + def test_method_web_crawl_md(self, client: ContextDev) -> None: + web = client.web.web_crawl_md( + url="https://example.com", + ) + assert_matches_type(WebWebCrawlMdResponse, web, path=["response"]) + + @pytest.mark.skip(reason="Mock server tests are disabled") + @parametrize + def test_method_web_crawl_md_with_all_params(self, client: ContextDev) -> None: + web = client.web.web_crawl_md( + url="https://example.com", + follow_subdomains=True, + include_images=True, + include_links=True, + max_depth=0, + max_pages=1, + shorten_base64_images=True, + url_regex="urlRegex", + use_main_content_only=True, + ) + assert_matches_type(WebWebCrawlMdResponse, web, path=["response"]) + + @pytest.mark.skip(reason="Mock server tests are disabled") + @parametrize + def test_raw_response_web_crawl_md(self, client: ContextDev) -> None: + response = client.web.with_raw_response.web_crawl_md( + url="https://example.com", + ) + + assert response.is_closed is True + assert response.http_request.headers.get("X-Stainless-Lang") == "python" + web = response.parse() + assert_matches_type(WebWebCrawlMdResponse, web, path=["response"]) + + @pytest.mark.skip(reason="Mock server tests are disabled") + @parametrize + def test_streaming_response_web_crawl_md(self, client: ContextDev) -> None: + with client.web.with_streaming_response.web_crawl_md( + url="https://example.com", + ) as response: + assert not response.is_closed + assert response.http_request.headers.get("X-Stainless-Lang") == "python" + + web = response.parse() + assert_matches_type(WebWebCrawlMdResponse, web, path=["response"]) + + assert cast(Any, response.is_closed) is True + @pytest.mark.skip(reason="Mock server tests are disabled") @parametrize def test_method_web_scrape_html(self, client: ContextDev) -> None: @@ -276,6 +327,56 @@ async def test_streaming_response_screenshot(self, async_client: AsyncContextDev assert cast(Any, response.is_closed) is True + @pytest.mark.skip(reason="Mock server tests are disabled") + @parametrize + async def test_method_web_crawl_md(self, async_client: AsyncContextDev) -> None: + web = await async_client.web.web_crawl_md( + url="https://example.com", + ) + assert_matches_type(WebWebCrawlMdResponse, web, path=["response"]) + + @pytest.mark.skip(reason="Mock server tests are disabled") + @parametrize + async def test_method_web_crawl_md_with_all_params(self, async_client: AsyncContextDev) -> None: + web = await async_client.web.web_crawl_md( + url="https://example.com", + follow_subdomains=True, + include_images=True, + include_links=True, + max_depth=0, + max_pages=1, + shorten_base64_images=True, + url_regex="urlRegex", + use_main_content_only=True, + ) + assert_matches_type(WebWebCrawlMdResponse, web, path=["response"]) + + @pytest.mark.skip(reason="Mock server tests are disabled") + @parametrize + async def test_raw_response_web_crawl_md(self, async_client: AsyncContextDev) -> None: + response = await async_client.web.with_raw_response.web_crawl_md( + url="https://example.com", + ) + + assert response.is_closed is True + assert response.http_request.headers.get("X-Stainless-Lang") == "python" + web = await response.parse() + assert_matches_type(WebWebCrawlMdResponse, web, path=["response"]) + + @pytest.mark.skip(reason="Mock server tests are disabled") + @parametrize + async def test_streaming_response_web_crawl_md(self, async_client: AsyncContextDev) -> None: + async with async_client.web.with_streaming_response.web_crawl_md( + url="https://example.com", + ) as response: + assert not response.is_closed + assert response.http_request.headers.get("X-Stainless-Lang") == "python" + + web = await response.parse() + assert_matches_type(WebWebCrawlMdResponse, web, path=["response"]) + + assert cast(Any, response.is_closed) is True + @pytest.mark.skip(reason="Mock server tests are disabled") @parametrize async def test_method_web_scrape_html(self, async_client: AsyncContextDev) -> None: From effa0308e6907fa131910f12ed0d72dc000a825a Mon Sep 17 00:00:00 2001 From: "stainless-app[bot]" <142633134+stainless-app[bot]@users.noreply.github.com> Date: Sat, 4 Apr 2026 21:42:30 +0000 Subject: [PATCH 4/4] release: 0.6.0 --- .release-please-manifest.json | 2 +- CHANGELOG.md | 8 ++++++++ pyproject.toml | 2 +- src/context/dev/_version.py | 2 +- 4 files changed, 11 insertions(+), 3 deletions(-) diff --git a/.release-please-manifest.json b/.release-please-manifest.json index 2aca35a..4208b5c 100644 --- a/.release-please-manifest.json +++ b/.release-please-manifest.json @@ -1,3 +1,3 @@ { - ".": "0.5.0" + ".": "0.6.0" } \ No newline at end of file diff --git a/CHANGELOG.md b/CHANGELOG.md index 4f1e853..dcfed0a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,13 @@ # Changelog +## 0.6.0 (2026-04-04) + +Full Changelog: [v0.5.0...v0.6.0](https://github.com/context-dot-dev/context-python-sdk/compare/v0.5.0...v0.6.0) + +### Features + +* **api:** manual updates ([e40b811](https://github.com/context-dot-dev/context-python-sdk/commit/e40b8117067dece6aaef3ca122b655f1808be1a3)) + ## 0.5.0 (2026-04-03) Full Changelog: [v0.4.0...v0.5.0](https://github.com/context-dot-dev/context-python-sdk/compare/v0.4.0...v0.5.0) diff --git a/pyproject.toml b/pyproject.toml index 6f24b0c..875e33d 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "context.dev" -version = "0.5.0" +version = "0.6.0" description = "The official Python library for the context.dev API" dynamic = ["readme"] license = "Apache-2.0" diff --git a/src/context/dev/_version.py b/src/context/dev/_version.py index 6657ecc..6ae0421 100644 --- a/src/context/dev/_version.py +++ b/src/context/dev/_version.py @@ -1,4 +1,4 @@ # File generated from our OpenAPI spec by Stainless. See CONTRIBUTING.md for details. __title__ = "context.dev" -__version__ = "0.5.0" # x-release-please-version +__version__ = "0.6.0" # x-release-please-version