Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .release-please-manifest.json
Original file line number Diff line number Diff line change
@@ -1,3 +1,3 @@
{
".": "0.5.0"
".": "0.6.0"
}
8 changes: 4 additions & 4 deletions .stats.yml
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
configured_endpoints: 20
openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev%2Fcontext.dev-56a21db16ac3a797f86daca01a2ea115a0365db4b2f9d8accdec4f4d3ee2eb83.yml
openapi_spec_hash: bfcef090896da96023c4485a3d69e350
config_hash: 38268bb88fc4dcbb8f2f94dd138b5910
configured_endpoints: 21
openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev%2Fcontext.dev-97cdb78dc0d72e9df643a89660f2b0c9687f12c6e4d93f7767f6cfc1b4f2e4c7.yml
openapi_spec_hash: 92fc94fd8865fabe78c2667490ca3884
config_hash: 682b89b02a20f5d1c13e2c91ecbcf5ce
8 changes: 8 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
@@ -1,5 +1,13 @@
# Changelog

## 0.6.0 (2026-04-04)

Full Changelog: [v0.5.0...v0.6.0](https://github.com/context-dot-dev/context-python-sdk/compare/v0.5.0...v0.6.0)

### Features

* **api:** manual updates ([e40b811](https://github.com/context-dot-dev/context-python-sdk/commit/e40b8117067dece6aaef3ca122b655f1808be1a3))

## 0.5.0 (2026-04-03)

Full Changelog: [v0.4.0...v0.5.0](https://github.com/context-dot-dev/context-python-sdk/compare/v0.4.0...v0.5.0)
Expand Down
2 changes: 2 additions & 0 deletions api.md
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,7 @@ Types:
```python
from context.dev.types import (
WebScreenshotResponse,
WebWebCrawlMdResponse,
WebWebScrapeHTMLResponse,
WebWebScrapeImagesResponse,
WebWebScrapeMdResponse,
Expand All @@ -15,6 +16,7 @@ from context.dev.types import (
Methods:

- <code title="get /brand/screenshot">client.web.<a href="./src/context/dev/resources/web.py">screenshot</a>(\*\*<a href="src/context/dev/types/web_screenshot_params.py">params</a>) -> <a href="./src/context/dev/types/web_screenshot_response.py">WebScreenshotResponse</a></code>
- <code title="post /web/crawl">client.web.<a href="./src/context/dev/resources/web.py">web_crawl_md</a>(\*\*<a href="src/context/dev/types/web_web_crawl_md_params.py">params</a>) -> <a href="./src/context/dev/types/web_web_crawl_md_response.py">WebWebCrawlMdResponse</a></code>
- <code title="get /web/scrape/html">client.web.<a href="./src/context/dev/resources/web.py">web_scrape_html</a>(\*\*<a href="src/context/dev/types/web_web_scrape_html_params.py">params</a>) -> <a href="./src/context/dev/types/web_web_scrape_html_response.py">WebWebScrapeHTMLResponse</a></code>
- <code title="get /web/scrape/images">client.web.<a href="./src/context/dev/resources/web.py">web_scrape_images</a>(\*\*<a href="src/context/dev/types/web_web_scrape_images_params.py">params</a>) -> <a href="./src/context/dev/types/web_web_scrape_images_response.py">WebWebScrapeImagesResponse</a></code>
- <code title="get /web/scrape/markdown">client.web.<a href="./src/context/dev/resources/web.py">web_scrape_md</a>(\*\*<a href="src/context/dev/types/web_web_scrape_md_params.py">params</a>) -> <a href="./src/context/dev/types/web_web_scrape_md_response.py">WebWebScrapeMdResponse</a></code>
Expand Down
2 changes: 1 addition & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
[project]
name = "context.dev"
version = "0.5.0"
version = "0.6.0"
description = "The official Python library for the context.dev API"
dynamic = ["readme"]
license = "Apache-2.0"
Expand Down
2 changes: 1 addition & 1 deletion src/context/dev/_version.py
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
# File generated from our OpenAPI spec by Stainless. See CONTRIBUTING.md for details.

__title__ = "context.dev"
__version__ = "0.5.0" # x-release-please-version
__version__ = "0.6.0" # x-release-please-version
166 changes: 166 additions & 0 deletions src/context/dev/resources/web.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,7 @@

from ..types import (
web_screenshot_params,
web_web_crawl_md_params,
web_web_scrape_md_params,
web_web_scrape_html_params,
web_web_scrape_images_params,
Expand All @@ -25,6 +26,7 @@
)
from .._base_client import make_request_options
from ..types.web_screenshot_response import WebScreenshotResponse
from ..types.web_web_crawl_md_response import WebWebCrawlMdResponse
from ..types.web_web_scrape_md_response import WebWebScrapeMdResponse
from ..types.web_web_scrape_html_response import WebWebScrapeHTMLResponse
from ..types.web_web_scrape_images_response import WebWebScrapeImagesResponse
Expand Down Expand Up @@ -119,6 +121,82 @@ def screenshot(
cast_to=WebScreenshotResponse,
)

def web_crawl_md(
self,
*,
url: str,
follow_subdomains: bool | Omit = omit,
include_images: bool | Omit = omit,
include_links: bool | Omit = omit,
max_depth: int | Omit = omit,
max_pages: int | Omit = omit,
shorten_base64_images: bool | Omit = omit,
url_regex: str | Omit = omit,
use_main_content_only: bool | Omit = omit,
# Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
# The extra values given here take precedence over values defined on the client or passed to this method.
extra_headers: Headers | None = None,
extra_query: Query | None = None,
extra_body: Body | None = None,
timeout: float | httpx.Timeout | None | NotGiven = not_given,
) -> WebWebCrawlMdResponse:
"""
Performs a crawl starting from a given URL, extracts page content as Markdown,
and returns results for all crawled pages. Only follows links within the same
domain as the starting URL. Costs 1 credit per successful page crawled.

Args:
url: The starting URL for the crawl (must include http:// or https:// protocol)

follow_subdomains: When true, follow links on subdomains of the starting URL's domain (e.g.
docs.example.com when starting from example.com). www and apex are always
treated as equivalent.

include_images: Include image references in the Markdown output

include_links: Preserve hyperlinks in the Markdown output

max_depth: Maximum link depth from the starting URL (0 = only the starting page)

max_pages: Maximum number of pages to crawl. Hard cap: 500.

shorten_base64_images: Truncate base64-encoded image data in the Markdown output

url_regex: Regex pattern. Only URLs matching this pattern will be followed and scraped.

use_main_content_only: Extract only the main content, stripping headers, footers, sidebars, and
navigation

extra_headers: Send extra headers

extra_query: Add additional query parameters to the request

extra_body: Add additional JSON properties to the request

timeout: Override the client-level default timeout for this request, in seconds
"""
return self._post(
"/web/crawl",
body=maybe_transform(
{
"url": url,
"follow_subdomains": follow_subdomains,
"include_images": include_images,
"include_links": include_links,
"max_depth": max_depth,
"max_pages": max_pages,
"shorten_base64_images": shorten_base64_images,
"url_regex": url_regex,
"use_main_content_only": use_main_content_only,
},
web_web_crawl_md_params.WebWebCrawlMdParams,
),
options=make_request_options(
extra_headers=extra_headers, extra_query=extra_query, extra_body=extra_body, timeout=timeout
),
cast_to=WebWebCrawlMdResponse,
)

def web_scrape_html(
self,
*,
Expand Down Expand Up @@ -394,6 +472,82 @@ async def screenshot(
cast_to=WebScreenshotResponse,
)

async def web_crawl_md(
self,
*,
url: str,
follow_subdomains: bool | Omit = omit,
include_images: bool | Omit = omit,
include_links: bool | Omit = omit,
max_depth: int | Omit = omit,
max_pages: int | Omit = omit,
shorten_base64_images: bool | Omit = omit,
url_regex: str | Omit = omit,
use_main_content_only: bool | Omit = omit,
# Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
# The extra values given here take precedence over values defined on the client or passed to this method.
extra_headers: Headers | None = None,
extra_query: Query | None = None,
extra_body: Body | None = None,
timeout: float | httpx.Timeout | None | NotGiven = not_given,
) -> WebWebCrawlMdResponse:
"""
Performs a crawl starting from a given URL, extracts page content as Markdown,
and returns results for all crawled pages. Only follows links within the same
domain as the starting URL. Costs 1 credit per successful page crawled.

Args:
url: The starting URL for the crawl (must include http:// or https:// protocol)

follow_subdomains: When true, follow links on subdomains of the starting URL's domain (e.g.
docs.example.com when starting from example.com). www and apex are always
treated as equivalent.

include_images: Include image references in the Markdown output

include_links: Preserve hyperlinks in the Markdown output

max_depth: Maximum link depth from the starting URL (0 = only the starting page)

max_pages: Maximum number of pages to crawl. Hard cap: 500.

shorten_base64_images: Truncate base64-encoded image data in the Markdown output

url_regex: Regex pattern. Only URLs matching this pattern will be followed and scraped.

use_main_content_only: Extract only the main content, stripping headers, footers, sidebars, and
navigation

extra_headers: Send extra headers

extra_query: Add additional query parameters to the request

extra_body: Add additional JSON properties to the request

timeout: Override the client-level default timeout for this request, in seconds
"""
return await self._post(
"/web/crawl",
body=await async_maybe_transform(
{
"url": url,
"follow_subdomains": follow_subdomains,
"include_images": include_images,
"include_links": include_links,
"max_depth": max_depth,
"max_pages": max_pages,
"shorten_base64_images": shorten_base64_images,
"url_regex": url_regex,
"use_main_content_only": use_main_content_only,
},
web_web_crawl_md_params.WebWebCrawlMdParams,
),
options=make_request_options(
extra_headers=extra_headers, extra_query=extra_query, extra_body=extra_body, timeout=timeout
),
cast_to=WebWebCrawlMdResponse,
)

async def web_scrape_html(
self,
*,
Expand Down Expand Up @@ -590,6 +744,9 @@ def __init__(self, web: WebResource) -> None:
self.screenshot = to_raw_response_wrapper(
web.screenshot,
)
self.web_crawl_md = to_raw_response_wrapper(
web.web_crawl_md,
)
self.web_scrape_html = to_raw_response_wrapper(
web.web_scrape_html,
)
Expand All @@ -611,6 +768,9 @@ def __init__(self, web: AsyncWebResource) -> None:
self.screenshot = async_to_raw_response_wrapper(
web.screenshot,
)
self.web_crawl_md = async_to_raw_response_wrapper(
web.web_crawl_md,
)
self.web_scrape_html = async_to_raw_response_wrapper(
web.web_scrape_html,
)
Expand All @@ -632,6 +792,9 @@ def __init__(self, web: WebResource) -> None:
self.screenshot = to_streamed_response_wrapper(
web.screenshot,
)
self.web_crawl_md = to_streamed_response_wrapper(
web.web_crawl_md,
)
self.web_scrape_html = to_streamed_response_wrapper(
web.web_scrape_html,
)
Expand All @@ -653,6 +816,9 @@ def __init__(self, web: AsyncWebResource) -> None:
self.screenshot = async_to_streamed_response_wrapper(
web.screenshot,
)
self.web_crawl_md = async_to_streamed_response_wrapper(
web.web_crawl_md,
)
self.web_scrape_html = async_to_streamed_response_wrapper(
web.web_scrape_html,
)
Expand Down
2 changes: 2 additions & 0 deletions src/context/dev/types/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,9 +9,11 @@
from .brand_retrieve_response import BrandRetrieveResponse as BrandRetrieveResponse
from .utility_prefetch_params import UtilityPrefetchParams as UtilityPrefetchParams
from .web_screenshot_response import WebScreenshotResponse as WebScreenshotResponse
from .web_web_crawl_md_params import WebWebCrawlMdParams as WebWebCrawlMdParams
from .web_web_scrape_md_params import WebWebScrapeMdParams as WebWebScrapeMdParams
from .ai_extract_product_params import AIExtractProductParams as AIExtractProductParams
from .utility_prefetch_response import UtilityPrefetchResponse as UtilityPrefetchResponse
from .web_web_crawl_md_response import WebWebCrawlMdResponse as WebWebCrawlMdResponse
from .ai_extract_products_params import AIExtractProductsParams as AIExtractProductsParams
from .style_extract_fonts_params import StyleExtractFontsParams as StyleExtractFontsParams
from .web_web_scrape_html_params import WebWebScrapeHTMLParams as WebWebScrapeHTMLParams
Expand Down
45 changes: 45 additions & 0 deletions src/context/dev/types/web_web_crawl_md_params.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,45 @@
# File generated from our OpenAPI spec by Stainless. See CONTRIBUTING.md for details.

from __future__ import annotations

from typing_extensions import Required, Annotated, TypedDict

from .._utils import PropertyInfo

__all__ = ["WebWebCrawlMdParams"]


class WebWebCrawlMdParams(TypedDict, total=False):
url: Required[str]
"""The starting URL for the crawl (must include http:// or https:// protocol)"""

follow_subdomains: Annotated[bool, PropertyInfo(alias="followSubdomains")]
"""When true, follow links on subdomains of the starting URL's domain (e.g.

docs.example.com when starting from example.com). www and apex are always
treated as equivalent.
"""

include_images: Annotated[bool, PropertyInfo(alias="includeImages")]
"""Include image references in the Markdown output"""

include_links: Annotated[bool, PropertyInfo(alias="includeLinks")]
"""Preserve hyperlinks in the Markdown output"""

max_depth: Annotated[int, PropertyInfo(alias="maxDepth")]
"""Maximum link depth from the starting URL (0 = only the starting page)"""

max_pages: Annotated[int, PropertyInfo(alias="maxPages")]
"""Maximum number of pages to crawl. Hard cap: 500."""

shorten_base64_images: Annotated[bool, PropertyInfo(alias="shortenBase64Images")]
"""Truncate base64-encoded image data in the Markdown output"""

url_regex: Annotated[str, PropertyInfo(alias="urlRegex")]
"""Regex pattern. Only URLs matching this pattern will be followed and scraped."""

use_main_content_only: Annotated[bool, PropertyInfo(alias="useMainContentOnly")]
"""
Extract only the main content, stripping headers, footers, sidebars, and
navigation
"""
53 changes: 53 additions & 0 deletions src/context/dev/types/web_web_crawl_md_response.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,53 @@
# File generated from our OpenAPI spec by Stainless. See CONTRIBUTING.md for details.

from typing import List

from pydantic import Field as FieldInfo

from .._models import BaseModel

__all__ = ["WebWebCrawlMdResponse", "Metadata", "Result", "ResultMetadata"]


class Metadata(BaseModel):
max_crawl_depth: int = FieldInfo(alias="maxCrawlDepth")
"""Maximum crawl depth reached during the crawl"""

num_failed: int = FieldInfo(alias="numFailed")
"""Number of pages that failed to crawl"""

num_succeeded: int = FieldInfo(alias="numSucceeded")
"""Number of pages successfully crawled"""

num_urls: int = FieldInfo(alias="numUrls")
"""Total number of URLs crawled"""


class ResultMetadata(BaseModel):
crawl_depth: int = FieldInfo(alias="crawlDepth")
"""Depth relative to the start URL. 0 = start URL, 1 = one link away."""

status_code: int = FieldInfo(alias="statusCode")
"""HTTP status code of the response"""

success: bool
"""true if the page was fetched and parsed successfully"""

title: str
"""The page's <title> content (empty string if unavailable)"""

url: str
"""The URL that was fetched"""


class Result(BaseModel):
markdown: str
"""Extracted page content as Markdown (empty string on failure)"""

metadata: ResultMetadata


class WebWebCrawlMdResponse(BaseModel):
metadata: Metadata

results: List[Result]
Loading
Loading