From 494fae61c5dc13356848f8cd4c8c3f901e84a99c Mon Sep 17 00:00:00 2001 From: "stainless-app[bot]" <142633134+stainless-app[bot]@users.noreply.github.com> Date: Fri, 24 Apr 2026 01:02:20 +0000 Subject: [PATCH 1/2] feat(api): api update --- .stats.yml | 4 +-- src/context/dev/resources/web.py | 36 +++++++++++++++++++ .../dev/types/web_web_crawl_md_params.py | 7 ++++ .../dev/types/web_web_crawl_md_response.py | 5 +++ .../dev/types/web_web_scrape_html_params.py | 7 ++++ .../dev/types/web_web_scrape_md_params.py | 7 ++++ tests/api_resources/test_web.py | 6 ++++ 7 files changed, 70 insertions(+), 2 deletions(-) diff --git a/.stats.yml b/.stats.yml index 98e8c8a..592a4b2 100644 --- a/.stats.yml +++ b/.stats.yml @@ -1,4 +1,4 @@ configured_endpoints: 21 -openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev%2Fcontext.dev-ca8e38b0f28a9967dab631b8868f1d59bc035fd994a717e0125c77ca592bd105.yml -openapi_spec_hash: ef0a5df01201a032dcc41f9a25b733e6 +openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev%2Fcontext.dev-ee23a181bfa364090a90254a433704adf4d3b413730fe75e5ad9a26273ec517a.yml +openapi_spec_hash: bc99d89cb1e7cdacc997214305bffee7 config_hash: 7d13dca2b2c6f71fc463cb6062efa5ea diff --git a/src/context/dev/resources/web.py b/src/context/dev/resources/web.py index 5fca4db..59643e9 100644 --- a/src/context/dev/resources/web.py +++ b/src/context/dev/resources/web.py @@ -254,6 +254,7 @@ def web_crawl_md( max_age_ms: int | Omit = omit, max_depth: int | Omit = omit, max_pages: int | Omit = omit, + parse_pdf: bool | Omit = omit, shorten_base64_images: bool | Omit = omit, url_regex: str | Omit = omit, use_main_content_only: bool | Omit = omit, @@ -287,6 +288,10 @@ def web_crawl_md( max_pages: Maximum number of pages to crawl. Hard cap: 500. + parse_pdf: When true (default), PDF pages are fetched and their text layer is extracted and + converted to Markdown alongside HTML pages. When false, PDF pages are skipped + entirely (not included in results and not counted as failures). + shorten_base64_images: Truncate base64-encoded image data in the Markdown output url_regex: Regex pattern. Only URLs matching this pattern will be followed and scraped. @@ -313,6 +318,7 @@ def web_crawl_md( "max_age_ms": max_age_ms, "max_depth": max_depth, "max_pages": max_pages, + "parse_pdf": parse_pdf, "shorten_base64_images": shorten_base64_images, "url_regex": url_regex, "use_main_content_only": use_main_content_only, @@ -330,6 +336,7 @@ def web_scrape_html( *, url: str, max_age_ms: int | Omit = omit, + parse_pdf: bool | Omit = omit, # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs. # The extra values given here take precedence over values defined on the client or passed to this method. extra_headers: Headers | None = None, @@ -347,6 +354,10 @@ def web_scrape_html( younger than this many milliseconds. Defaults to 1 day (86400000 ms) when omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh. + parse_pdf: When true (default), PDF URLs are fetched and their text layer is extracted and + returned wrapped in …. When false, PDF URLs are skipped + and a 400 WEBSITE_ACCESS_ERROR is returned. + extra_headers: Send extra headers extra_query: Add additional query parameters to the request @@ -366,6 +377,7 @@ def web_scrape_html( { "url": url, "max_age_ms": max_age_ms, + "parse_pdf": parse_pdf, }, web_web_scrape_html_params.WebWebScrapeHTMLParams, ), @@ -420,6 +432,7 @@ def web_scrape_md( include_images: bool | Omit = omit, include_links: bool | Omit = omit, max_age_ms: int | Omit = omit, + parse_pdf: bool | Omit = omit, shorten_base64_images: bool | Omit = omit, use_main_content_only: bool | Omit = omit, # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs. @@ -444,6 +457,10 @@ def web_scrape_md( younger than this many milliseconds. Defaults to 1 day (86400000 ms) when omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh. + parse_pdf: When true (default), PDF URLs are fetched and their text layer is extracted and + converted to Markdown. When false, PDF URLs are skipped and a 400 + WEBSITE_ACCESS_ERROR is returned. + shorten_base64_images: Shorten base64-encoded image data in the Markdown output use_main_content_only: Extract only the main content of the page, excluding headers, footers, sidebars, @@ -470,6 +487,7 @@ def web_scrape_md( "include_images": include_images, "include_links": include_links, "max_age_ms": max_age_ms, + "parse_pdf": parse_pdf, "shorten_base64_images": shorten_base64_images, "use_main_content_only": use_main_content_only, }, @@ -747,6 +765,7 @@ async def web_crawl_md( max_age_ms: int | Omit = omit, max_depth: int | Omit = omit, max_pages: int | Omit = omit, + parse_pdf: bool | Omit = omit, shorten_base64_images: bool | Omit = omit, url_regex: str | Omit = omit, use_main_content_only: bool | Omit = omit, @@ -780,6 +799,10 @@ async def web_crawl_md( max_pages: Maximum number of pages to crawl. Hard cap: 500. + parse_pdf: When true (default), PDF pages are fetched and their text layer is extracted and + converted to Markdown alongside HTML pages. When false, PDF pages are skipped + entirely (not included in results and not counted as failures). + shorten_base64_images: Truncate base64-encoded image data in the Markdown output url_regex: Regex pattern. Only URLs matching this pattern will be followed and scraped. @@ -806,6 +829,7 @@ async def web_crawl_md( "max_age_ms": max_age_ms, "max_depth": max_depth, "max_pages": max_pages, + "parse_pdf": parse_pdf, "shorten_base64_images": shorten_base64_images, "url_regex": url_regex, "use_main_content_only": use_main_content_only, @@ -823,6 +847,7 @@ async def web_scrape_html( *, url: str, max_age_ms: int | Omit = omit, + parse_pdf: bool | Omit = omit, # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs. # The extra values given here take precedence over values defined on the client or passed to this method. extra_headers: Headers | None = None, @@ -840,6 +865,10 @@ async def web_scrape_html( younger than this many milliseconds. Defaults to 1 day (86400000 ms) when omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh. + parse_pdf: When true (default), PDF URLs are fetched and their text layer is extracted and + returned wrapped in …. When false, PDF URLs are skipped + and a 400 WEBSITE_ACCESS_ERROR is returned. + extra_headers: Send extra headers extra_query: Add additional query parameters to the request @@ -859,6 +888,7 @@ async def web_scrape_html( { "url": url, "max_age_ms": max_age_ms, + "parse_pdf": parse_pdf, }, web_web_scrape_html_params.WebWebScrapeHTMLParams, ), @@ -913,6 +943,7 @@ async def web_scrape_md( include_images: bool | Omit = omit, include_links: bool | Omit = omit, max_age_ms: int | Omit = omit, + parse_pdf: bool | Omit = omit, shorten_base64_images: bool | Omit = omit, use_main_content_only: bool | Omit = omit, # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs. @@ -937,6 +968,10 @@ async def web_scrape_md( younger than this many milliseconds. Defaults to 1 day (86400000 ms) when omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh. + parse_pdf: When true (default), PDF URLs are fetched and their text layer is extracted and + converted to Markdown. When false, PDF URLs are skipped and a 400 + WEBSITE_ACCESS_ERROR is returned. + shorten_base64_images: Shorten base64-encoded image data in the Markdown output use_main_content_only: Extract only the main content of the page, excluding headers, footers, sidebars, @@ -963,6 +998,7 @@ async def web_scrape_md( "include_images": include_images, "include_links": include_links, "max_age_ms": max_age_ms, + "parse_pdf": parse_pdf, "shorten_base64_images": shorten_base64_images, "use_main_content_only": use_main_content_only, }, diff --git a/src/context/dev/types/web_web_crawl_md_params.py b/src/context/dev/types/web_web_crawl_md_params.py index 7cf4bb9..69ee967 100644 --- a/src/context/dev/types/web_web_crawl_md_params.py +++ b/src/context/dev/types/web_web_crawl_md_params.py @@ -39,6 +39,13 @@ class WebWebCrawlMdParams(TypedDict, total=False): max_pages: Annotated[int, PropertyInfo(alias="maxPages")] """Maximum number of pages to crawl. Hard cap: 500.""" + parse_pdf: Annotated[bool, PropertyInfo(alias="parsePDF")] + """ + When true (default), PDF pages are fetched and their text layer is extracted and + converted to Markdown alongside HTML pages. When false, PDF pages are skipped + entirely (not included in results and not counted as failures). + """ + shorten_base64_images: Annotated[bool, PropertyInfo(alias="shortenBase64Images")] """Truncate base64-encoded image data in the Markdown output""" diff --git a/src/context/dev/types/web_web_crawl_md_response.py b/src/context/dev/types/web_web_crawl_md_response.py index 49ba2d8..9c6cda1 100644 --- a/src/context/dev/types/web_web_crawl_md_response.py +++ b/src/context/dev/types/web_web_crawl_md_response.py @@ -16,6 +16,11 @@ class Metadata(BaseModel): num_failed: int = FieldInfo(alias="numFailed") """Number of pages that failed to crawl""" + num_skipped: int = FieldInfo(alias="numSkipped") + """ + Number of URLs skipped (PDFs when parsePDF=false, or URLs not matching urlRegex) + """ + num_succeeded: int = FieldInfo(alias="numSucceeded") """Number of pages successfully crawled""" diff --git a/src/context/dev/types/web_web_scrape_html_params.py b/src/context/dev/types/web_web_scrape_html_params.py index d184801..f1eb86c 100644 --- a/src/context/dev/types/web_web_scrape_html_params.py +++ b/src/context/dev/types/web_web_scrape_html_params.py @@ -19,3 +19,10 @@ class WebWebScrapeHTMLParams(TypedDict, total=False): younger than this many milliseconds. Defaults to 1 day (86400000 ms) when omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh. """ + + parse_pdf: Annotated[bool, PropertyInfo(alias="parsePDF")] + """ + When true (default), PDF URLs are fetched and their text layer is extracted and + returned wrapped in …. When false, PDF URLs are skipped + and a 400 WEBSITE_ACCESS_ERROR is returned. + """ diff --git a/src/context/dev/types/web_web_scrape_md_params.py b/src/context/dev/types/web_web_scrape_md_params.py index 8cc9333..ab364f5 100644 --- a/src/context/dev/types/web_web_scrape_md_params.py +++ b/src/context/dev/types/web_web_scrape_md_params.py @@ -29,6 +29,13 @@ class WebWebScrapeMdParams(TypedDict, total=False): omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh. """ + parse_pdf: Annotated[bool, PropertyInfo(alias="parsePDF")] + """ + When true (default), PDF URLs are fetched and their text layer is extracted and + converted to Markdown. When false, PDF URLs are skipped and a 400 + WEBSITE_ACCESS_ERROR is returned. + """ + shorten_base64_images: Annotated[bool, PropertyInfo(alias="shortenBase64Images")] """Shorten base64-encoded image data in the Markdown output""" diff --git a/tests/api_resources/test_web.py b/tests/api_resources/test_web.py index 6474cb1..04082f7 100644 --- a/tests/api_resources/test_web.py +++ b/tests/api_resources/test_web.py @@ -161,6 +161,7 @@ def test_method_web_crawl_md_with_all_params(self, client: ContextDev) -> None: max_age_ms=0, max_depth=0, max_pages=1, + parse_pdf=True, shorten_base64_images=True, url_regex="^https?://[^/]+/blog/", use_main_content_only=True, @@ -207,6 +208,7 @@ def test_method_web_scrape_html_with_all_params(self, client: ContextDev) -> Non web = client.web.web_scrape_html( url="https://example.com", max_age_ms=0, + parse_pdf=True, ) assert_matches_type(WebWebScrapeHTMLResponse, web, path=["response"]) @@ -286,6 +288,7 @@ def test_method_web_scrape_md_with_all_params(self, client: ContextDev) -> None: include_images=True, include_links=True, max_age_ms=0, + parse_pdf=True, shorten_base64_images=True, use_main_content_only=True, ) @@ -502,6 +505,7 @@ async def test_method_web_crawl_md_with_all_params(self, async_client: AsyncCont max_age_ms=0, max_depth=0, max_pages=1, + parse_pdf=True, shorten_base64_images=True, url_regex="^https?://[^/]+/blog/", use_main_content_only=True, @@ -548,6 +552,7 @@ async def test_method_web_scrape_html_with_all_params(self, async_client: AsyncC web = await async_client.web.web_scrape_html( url="https://example.com", max_age_ms=0, + parse_pdf=True, ) assert_matches_type(WebWebScrapeHTMLResponse, web, path=["response"]) @@ -627,6 +632,7 @@ async def test_method_web_scrape_md_with_all_params(self, async_client: AsyncCon include_images=True, include_links=True, max_age_ms=0, + parse_pdf=True, shorten_base64_images=True, use_main_content_only=True, ) From 675a04f1dd5b05012d21fab80da397c5b8fc855d Mon Sep 17 00:00:00 2001 From: "stainless-app[bot]" <142633134+stainless-app[bot]@users.noreply.github.com> Date: Fri, 24 Apr 2026 01:02:41 +0000 Subject: [PATCH 2/2] release: 0.10.0 --- .release-please-manifest.json | 2 +- CHANGELOG.md | 8 ++++++++ pyproject.toml | 2 +- src/context/dev/_version.py | 2 +- 4 files changed, 11 insertions(+), 3 deletions(-) diff --git a/.release-please-manifest.json b/.release-please-manifest.json index 6d78745..091cfb1 100644 --- a/.release-please-manifest.json +++ b/.release-please-manifest.json @@ -1,3 +1,3 @@ { - ".": "0.9.0" + ".": "0.10.0" } \ No newline at end of file diff --git a/CHANGELOG.md b/CHANGELOG.md index 543fc7e..9a848de 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,13 @@ # Changelog +## 0.10.0 (2026-04-24) + +Full Changelog: [v0.9.0...v0.10.0](https://github.com/context-dot-dev/context-python-sdk/compare/v0.9.0...v0.10.0) + +### Features + +* **api:** api update ([494fae6](https://github.com/context-dot-dev/context-python-sdk/commit/494fae61c5dc13356848f8cd4c8c3f901e84a99c)) + ## 0.9.0 (2026-04-23) Full Changelog: [v0.8.0...v0.9.0](https://github.com/context-dot-dev/context-python-sdk/compare/v0.8.0...v0.9.0) diff --git a/pyproject.toml b/pyproject.toml index 370218e..21a2aa8 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "context.dev" -version = "0.9.0" +version = "0.10.0" description = "The official Python library for the context.dev API" dynamic = ["readme"] license = "Apache-2.0" diff --git a/src/context/dev/_version.py b/src/context/dev/_version.py index a5aba43..d074db9 100644 --- a/src/context/dev/_version.py +++ b/src/context/dev/_version.py @@ -1,4 +1,4 @@ # File generated from our OpenAPI spec by Stainless. See CONTRIBUTING.md for details. __title__ = "context.dev" -__version__ = "0.9.0" # x-release-please-version +__version__ = "0.10.0" # x-release-please-version