Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .release-please-manifest.json
Original file line number Diff line number Diff line change
@@ -1,3 +1,3 @@
{
".": "0.9.0"
".": "0.10.0"
}
4 changes: 2 additions & 2 deletions .stats.yml
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
configured_endpoints: 21
openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev%2Fcontext.dev-ca8e38b0f28a9967dab631b8868f1d59bc035fd994a717e0125c77ca592bd105.yml
openapi_spec_hash: ef0a5df01201a032dcc41f9a25b733e6
openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev%2Fcontext.dev-ee23a181bfa364090a90254a433704adf4d3b413730fe75e5ad9a26273ec517a.yml
openapi_spec_hash: bc99d89cb1e7cdacc997214305bffee7
config_hash: 7d13dca2b2c6f71fc463cb6062efa5ea
8 changes: 8 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
@@ -1,5 +1,13 @@
# Changelog

## 0.10.0 (2026-04-24)

Full Changelog: [v0.9.0...v0.10.0](https://github.com/context-dot-dev/context-python-sdk/compare/v0.9.0...v0.10.0)

### Features

* **api:** api update ([494fae6](https://github.com/context-dot-dev/context-python-sdk/commit/494fae61c5dc13356848f8cd4c8c3f901e84a99c))

## 0.9.0 (2026-04-23)

Full Changelog: [v0.8.0...v0.9.0](https://github.com/context-dot-dev/context-python-sdk/compare/v0.8.0...v0.9.0)
Expand Down
2 changes: 1 addition & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
[project]
name = "context.dev"
version = "0.9.0"
version = "0.10.0"
description = "The official Python library for the context.dev API"
dynamic = ["readme"]
license = "Apache-2.0"
Expand Down
2 changes: 1 addition & 1 deletion src/context/dev/_version.py
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
# File generated from our OpenAPI spec by Stainless. See CONTRIBUTING.md for details.

__title__ = "context.dev"
__version__ = "0.9.0" # x-release-please-version
__version__ = "0.10.0" # x-release-please-version
36 changes: 36 additions & 0 deletions src/context/dev/resources/web.py
Original file line number Diff line number Diff line change
Expand Up @@ -254,6 +254,7 @@ def web_crawl_md(
max_age_ms: int | Omit = omit,
max_depth: int | Omit = omit,
max_pages: int | Omit = omit,
parse_pdf: bool | Omit = omit,
shorten_base64_images: bool | Omit = omit,
url_regex: str | Omit = omit,
use_main_content_only: bool | Omit = omit,
Expand Down Expand Up @@ -287,6 +288,10 @@ def web_crawl_md(

max_pages: Maximum number of pages to crawl. Hard cap: 500.

parse_pdf: When true (default), PDF pages are fetched and their text layer is extracted and
converted to Markdown alongside HTML pages. When false, PDF pages are skipped
entirely (not included in results and not counted as failures).

shorten_base64_images: Truncate base64-encoded image data in the Markdown output

url_regex: Regex pattern. Only URLs matching this pattern will be followed and scraped.
Expand All @@ -313,6 +318,7 @@ def web_crawl_md(
"max_age_ms": max_age_ms,
"max_depth": max_depth,
"max_pages": max_pages,
"parse_pdf": parse_pdf,
"shorten_base64_images": shorten_base64_images,
"url_regex": url_regex,
"use_main_content_only": use_main_content_only,
Expand All @@ -330,6 +336,7 @@ def web_scrape_html(
*,
url: str,
max_age_ms: int | Omit = omit,
parse_pdf: bool | Omit = omit,
# Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
# The extra values given here take precedence over values defined on the client or passed to this method.
extra_headers: Headers | None = None,
Expand All @@ -347,6 +354,10 @@ def web_scrape_html(
younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.

parse_pdf: When true (default), PDF URLs are fetched and their text layer is extracted and
returned wrapped in <html><pdf>…</pdf></html>. When false, PDF URLs are skipped
and a 400 WEBSITE_ACCESS_ERROR is returned.

extra_headers: Send extra headers

extra_query: Add additional query parameters to the request
Expand All @@ -366,6 +377,7 @@ def web_scrape_html(
{
"url": url,
"max_age_ms": max_age_ms,
"parse_pdf": parse_pdf,
},
web_web_scrape_html_params.WebWebScrapeHTMLParams,
),
Expand Down Expand Up @@ -420,6 +432,7 @@ def web_scrape_md(
include_images: bool | Omit = omit,
include_links: bool | Omit = omit,
max_age_ms: int | Omit = omit,
parse_pdf: bool | Omit = omit,
shorten_base64_images: bool | Omit = omit,
use_main_content_only: bool | Omit = omit,
# Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
Expand All @@ -444,6 +457,10 @@ def web_scrape_md(
younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.

parse_pdf: When true (default), PDF URLs are fetched and their text layer is extracted and
converted to Markdown. When false, PDF URLs are skipped and a 400
WEBSITE_ACCESS_ERROR is returned.

shorten_base64_images: Shorten base64-encoded image data in the Markdown output

use_main_content_only: Extract only the main content of the page, excluding headers, footers, sidebars,
Expand All @@ -470,6 +487,7 @@ def web_scrape_md(
"include_images": include_images,
"include_links": include_links,
"max_age_ms": max_age_ms,
"parse_pdf": parse_pdf,
"shorten_base64_images": shorten_base64_images,
"use_main_content_only": use_main_content_only,
},
Expand Down Expand Up @@ -747,6 +765,7 @@ async def web_crawl_md(
max_age_ms: int | Omit = omit,
max_depth: int | Omit = omit,
max_pages: int | Omit = omit,
parse_pdf: bool | Omit = omit,
shorten_base64_images: bool | Omit = omit,
url_regex: str | Omit = omit,
use_main_content_only: bool | Omit = omit,
Expand Down Expand Up @@ -780,6 +799,10 @@ async def web_crawl_md(

max_pages: Maximum number of pages to crawl. Hard cap: 500.

parse_pdf: When true (default), PDF pages are fetched and their text layer is extracted and
converted to Markdown alongside HTML pages. When false, PDF pages are skipped
entirely (not included in results and not counted as failures).

shorten_base64_images: Truncate base64-encoded image data in the Markdown output

url_regex: Regex pattern. Only URLs matching this pattern will be followed and scraped.
Expand All @@ -806,6 +829,7 @@ async def web_crawl_md(
"max_age_ms": max_age_ms,
"max_depth": max_depth,
"max_pages": max_pages,
"parse_pdf": parse_pdf,
"shorten_base64_images": shorten_base64_images,
"url_regex": url_regex,
"use_main_content_only": use_main_content_only,
Expand All @@ -823,6 +847,7 @@ async def web_scrape_html(
*,
url: str,
max_age_ms: int | Omit = omit,
parse_pdf: bool | Omit = omit,
# Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
# The extra values given here take precedence over values defined on the client or passed to this method.
extra_headers: Headers | None = None,
Expand All @@ -840,6 +865,10 @@ async def web_scrape_html(
younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.

parse_pdf: When true (default), PDF URLs are fetched and their text layer is extracted and
returned wrapped in <html><pdf>…</pdf></html>. When false, PDF URLs are skipped
and a 400 WEBSITE_ACCESS_ERROR is returned.

extra_headers: Send extra headers

extra_query: Add additional query parameters to the request
Expand All @@ -859,6 +888,7 @@ async def web_scrape_html(
{
"url": url,
"max_age_ms": max_age_ms,
"parse_pdf": parse_pdf,
},
web_web_scrape_html_params.WebWebScrapeHTMLParams,
),
Expand Down Expand Up @@ -913,6 +943,7 @@ async def web_scrape_md(
include_images: bool | Omit = omit,
include_links: bool | Omit = omit,
max_age_ms: int | Omit = omit,
parse_pdf: bool | Omit = omit,
shorten_base64_images: bool | Omit = omit,
use_main_content_only: bool | Omit = omit,
# Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
Expand All @@ -937,6 +968,10 @@ async def web_scrape_md(
younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.

parse_pdf: When true (default), PDF URLs are fetched and their text layer is extracted and
converted to Markdown. When false, PDF URLs are skipped and a 400
WEBSITE_ACCESS_ERROR is returned.

shorten_base64_images: Shorten base64-encoded image data in the Markdown output

use_main_content_only: Extract only the main content of the page, excluding headers, footers, sidebars,
Expand All @@ -963,6 +998,7 @@ async def web_scrape_md(
"include_images": include_images,
"include_links": include_links,
"max_age_ms": max_age_ms,
"parse_pdf": parse_pdf,
"shorten_base64_images": shorten_base64_images,
"use_main_content_only": use_main_content_only,
},
Expand Down
7 changes: 7 additions & 0 deletions src/context/dev/types/web_web_crawl_md_params.py
Original file line number Diff line number Diff line change
Expand Up @@ -39,6 +39,13 @@ class WebWebCrawlMdParams(TypedDict, total=False):
max_pages: Annotated[int, PropertyInfo(alias="maxPages")]
"""Maximum number of pages to crawl. Hard cap: 500."""

parse_pdf: Annotated[bool, PropertyInfo(alias="parsePDF")]
"""
When true (default), PDF pages are fetched and their text layer is extracted and
converted to Markdown alongside HTML pages. When false, PDF pages are skipped
entirely (not included in results and not counted as failures).
"""

shorten_base64_images: Annotated[bool, PropertyInfo(alias="shortenBase64Images")]
"""Truncate base64-encoded image data in the Markdown output"""

Expand Down
5 changes: 5 additions & 0 deletions src/context/dev/types/web_web_crawl_md_response.py
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,11 @@ class Metadata(BaseModel):
num_failed: int = FieldInfo(alias="numFailed")
"""Number of pages that failed to crawl"""

num_skipped: int = FieldInfo(alias="numSkipped")
"""
Number of URLs skipped (PDFs when parsePDF=false, or URLs not matching urlRegex)
"""

num_succeeded: int = FieldInfo(alias="numSucceeded")
"""Number of pages successfully crawled"""

Expand Down
7 changes: 7 additions & 0 deletions src/context/dev/types/web_web_scrape_html_params.py
Original file line number Diff line number Diff line change
Expand Up @@ -19,3 +19,10 @@ class WebWebScrapeHTMLParams(TypedDict, total=False):
younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
"""

parse_pdf: Annotated[bool, PropertyInfo(alias="parsePDF")]
"""
When true (default), PDF URLs are fetched and their text layer is extracted and
returned wrapped in <html><pdf>…</pdf></html>. When false, PDF URLs are skipped
and a 400 WEBSITE_ACCESS_ERROR is returned.
"""
7 changes: 7 additions & 0 deletions src/context/dev/types/web_web_scrape_md_params.py
Original file line number Diff line number Diff line change
Expand Up @@ -29,6 +29,13 @@ class WebWebScrapeMdParams(TypedDict, total=False):
omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
"""

parse_pdf: Annotated[bool, PropertyInfo(alias="parsePDF")]
"""
When true (default), PDF URLs are fetched and their text layer is extracted and
converted to Markdown. When false, PDF URLs are skipped and a 400
WEBSITE_ACCESS_ERROR is returned.
"""

shorten_base64_images: Annotated[bool, PropertyInfo(alias="shortenBase64Images")]
"""Shorten base64-encoded image data in the Markdown output"""

Expand Down
6 changes: 6 additions & 0 deletions tests/api_resources/test_web.py
Original file line number Diff line number Diff line change
Expand Up @@ -161,6 +161,7 @@ def test_method_web_crawl_md_with_all_params(self, client: ContextDev) -> None:
max_age_ms=0,
max_depth=0,
max_pages=1,
parse_pdf=True,
shorten_base64_images=True,
url_regex="^https?://[^/]+/blog/",
use_main_content_only=True,
Expand Down Expand Up @@ -207,6 +208,7 @@ def test_method_web_scrape_html_with_all_params(self, client: ContextDev) -> Non
web = client.web.web_scrape_html(
url="https://example.com",
max_age_ms=0,
parse_pdf=True,
)
assert_matches_type(WebWebScrapeHTMLResponse, web, path=["response"])

Expand Down Expand Up @@ -286,6 +288,7 @@ def test_method_web_scrape_md_with_all_params(self, client: ContextDev) -> None:
include_images=True,
include_links=True,
max_age_ms=0,
parse_pdf=True,
shorten_base64_images=True,
use_main_content_only=True,
)
Expand Down Expand Up @@ -502,6 +505,7 @@ async def test_method_web_crawl_md_with_all_params(self, async_client: AsyncCont
max_age_ms=0,
max_depth=0,
max_pages=1,
parse_pdf=True,
shorten_base64_images=True,
url_regex="^https?://[^/]+/blog/",
use_main_content_only=True,
Expand Down Expand Up @@ -548,6 +552,7 @@ async def test_method_web_scrape_html_with_all_params(self, async_client: AsyncC
web = await async_client.web.web_scrape_html(
url="https://example.com",
max_age_ms=0,
parse_pdf=True,
)
assert_matches_type(WebWebScrapeHTMLResponse, web, path=["response"])

Expand Down Expand Up @@ -627,6 +632,7 @@ async def test_method_web_scrape_md_with_all_params(self, async_client: AsyncCon
include_images=True,
include_links=True,
max_age_ms=0,
parse_pdf=True,
shorten_base64_images=True,
use_main_content_only=True,
)
Expand Down
Loading