diff --git a/.release-please-manifest.json b/.release-please-manifest.json index 1b77f50..6538ca9 100644 --- a/.release-please-manifest.json +++ b/.release-please-manifest.json @@ -1,3 +1,3 @@ { - ".": "0.7.0" + ".": "0.8.0" } \ No newline at end of file diff --git a/.stats.yml b/.stats.yml index 998ed68..6d316e1 100644 --- a/.stats.yml +++ b/.stats.yml @@ -1,4 +1,4 @@ configured_endpoints: 21 -openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev%2Fcontext.dev-5ee7a321f8e01d2678d56277f011f71294c2b8ac4d0cff0bda412362c0d9cf73.yml -openapi_spec_hash: bf5d57e3bddb6975770cf85c267b5035 -config_hash: 682b89b02a20f5d1c13e2c91ecbcf5ce +openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev%2Fcontext.dev-5afee68d7308a77a9ef51fb53917080746f96327c6d745fc029363a7d1494c3c.yml +openapi_spec_hash: b2e32bb58d92a00f6ede04c4f7bf1a34 +config_hash: 7d13dca2b2c6f71fc463cb6062efa5ea diff --git a/CHANGELOG.md b/CHANGELOG.md index 6c00b74..cc6b274 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,36 @@ # Changelog +## 0.8.0 (2026-04-19) + +Full Changelog: [v0.7.0...v0.8.0](https://github.com/context-dot-dev/context-python-sdk/compare/v0.7.0...v0.8.0) + +### Features + +* **api:** api update ([512bf26](https://github.com/context-dot-dev/context-python-sdk/commit/512bf2610e59733e0e950531ca9e6018a5446672)) +* **api:** api update ([f98376c](https://github.com/context-dot-dev/context-python-sdk/commit/f98376cf22a331138f716f8abeb4458a8e1e38c2)) +* **api:** api update ([9ddc161](https://github.com/context-dot-dev/context-python-sdk/commit/9ddc161215cd12384152295997b95d5ac44bec08)) +* **api:** api update ([639075e](https://github.com/context-dot-dev/context-python-sdk/commit/639075e0a0aa6203375592b8b67438ae0719ca26)) +* **api:** api update ([707a6c3](https://github.com/context-dot-dev/context-python-sdk/commit/707a6c3e49096e2248ec9e4ae7e683138647c2d6)) +* **api:** api update ([bbc2458](https://github.com/context-dot-dev/context-python-sdk/commit/bbc2458c38933208b22e234d6bf1f1ffee7f883e)) +* **api:** api update ([241bacf](https://github.com/context-dot-dev/context-python-sdk/commit/241bacf196c1e316aef39429e3919fda1c5a78eb)) +* **api:** api update ([648e71b](https://github.com/context-dot-dev/context-python-sdk/commit/648e71bf7afb870dceafe6485bf47932c74ebe1b)) +* **api:** api update ([656e585](https://github.com/context-dot-dev/context-python-sdk/commit/656e585bcdf228d8651cdab5ee37f1119a51eafa)) +* **api:** api update ([3fc9944](https://github.com/context-dot-dev/context-python-sdk/commit/3fc9944ff014539c18e0216b93c936c1df2b0219)) +* **api:** api update ([b9374f5](https://github.com/context-dot-dev/context-python-sdk/commit/b9374f57a2a0ed207c39e80b6fbf710bbe170e11)) +* **api:** manual updates ([2af78a4](https://github.com/context-dot-dev/context-python-sdk/commit/2af78a43c1e9b5654668e82c009b679dc3ae8e73)) +* **api:** manual updates ([d6acbc6](https://github.com/context-dot-dev/context-python-sdk/commit/d6acbc6552d84e58a6ff7a9e99952a2a7a624647)) +* **api:** manual updates ([ec8ff2d](https://github.com/context-dot-dev/context-python-sdk/commit/ec8ff2d2a0d84fcb0c05791fa913441c60ed9c24)) + + +### Bug Fixes + +* ensure file data are only sent as 1 parameter ([7c314a2](https://github.com/context-dot-dev/context-python-sdk/commit/7c314a29f73103eb592686df101c86c6cfd26b3f)) + + +### Performance Improvements + +* **client:** optimize file structure copying in multipart requests ([573a70c](https://github.com/context-dot-dev/context-python-sdk/commit/573a70cee8cb9bed8377f816bfba419ecbe86109)) + ## 0.7.0 (2026-04-09) Full Changelog: [v0.6.0...v0.7.0](https://github.com/context-dot-dev/context-python-sdk/compare/v0.6.0...v0.7.0) diff --git a/api.md b/api.md index 4c7de31..963807e 100644 --- a/api.md +++ b/api.md @@ -4,6 +4,8 @@ Types: ```python from context.dev.types import ( + WebExtractFontsResponse, + WebExtractStyleguideResponse, WebScreenshotResponse, WebWebCrawlMdResponse, WebWebScrapeHTMLResponse, @@ -15,7 +17,9 @@ from context.dev.types import ( Methods: -- client.web.screenshot(\*\*params) -> WebScreenshotResponse +- client.web.extract_fonts(\*\*params) -> WebExtractFontsResponse +- client.web.extract_styleguide(\*\*params) -> WebExtractStyleguideResponse +- client.web.screenshot(\*\*params) -> WebScreenshotResponse - client.web.web_crawl_md(\*\*params) -> WebWebCrawlMdResponse - client.web.web_scrape_html(\*\*params) -> WebWebScrapeHTMLResponse - client.web.web_scrape_images(\*\*params) -> WebWebScrapeImagesResponse @@ -36,19 +40,6 @@ Methods: - client.ai.extract_product(\*\*params) -> AIExtractProductResponse - client.ai.extract_products(\*\*params) -> AIExtractProductsResponse -# Style - -Types: - -```python -from context.dev.types import StyleExtractFontsResponse, StyleExtractStyleguideResponse -``` - -Methods: - -- client.style.extract_fonts(\*\*params) -> StyleExtractFontsResponse -- client.style.extract_styleguide(\*\*params) -> StyleExtractStyleguideResponse - # Brand Types: @@ -85,7 +76,7 @@ from context.dev.types import IndustryRetrieveNaicsResponse Methods: -- client.industry.retrieve_naics(\*\*params) -> IndustryRetrieveNaicsResponse +- client.industry.retrieve_naics(\*\*params) -> IndustryRetrieveNaicsResponse # Utility diff --git a/pyproject.toml b/pyproject.toml index c5f1939..8a95885 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "context.dev" -version = "0.7.0" +version = "0.8.0" description = "The official Python library for the context.dev API" dynamic = ["readme"] license = "Apache-2.0" diff --git a/src/context/dev/_client.py b/src/context/dev/_client.py index 06cee7f..ace042b 100644 --- a/src/context/dev/_client.py +++ b/src/context/dev/_client.py @@ -31,11 +31,10 @@ ) if TYPE_CHECKING: - from .resources import ai, web, brand, style, utility, industry + from .resources import ai, web, brand, utility, industry from .resources.ai import AIResource, AsyncAIResource from .resources.web import WebResource, AsyncWebResource from .resources.brand import BrandResource, AsyncBrandResource - from .resources.style import StyleResource, AsyncStyleResource from .resources.utility import UtilityResource, AsyncUtilityResource from .resources.industry import IndustryResource, AsyncIndustryResource @@ -118,12 +117,6 @@ def ai(self) -> AIResource: return AIResource(self) - @cached_property - def style(self) -> StyleResource: - from .resources.style import StyleResource - - return StyleResource(self) - @cached_property def brand(self) -> BrandResource: from .resources.brand import BrandResource @@ -322,12 +315,6 @@ def ai(self) -> AsyncAIResource: return AsyncAIResource(self) - @cached_property - def style(self) -> AsyncStyleResource: - from .resources.style import AsyncStyleResource - - return AsyncStyleResource(self) - @cached_property def brand(self) -> AsyncBrandResource: from .resources.brand import AsyncBrandResource @@ -477,12 +464,6 @@ def ai(self) -> ai.AIResourceWithRawResponse: return AIResourceWithRawResponse(self._client.ai) - @cached_property - def style(self) -> style.StyleResourceWithRawResponse: - from .resources.style import StyleResourceWithRawResponse - - return StyleResourceWithRawResponse(self._client.style) - @cached_property def brand(self) -> brand.BrandResourceWithRawResponse: from .resources.brand import BrandResourceWithRawResponse @@ -520,12 +501,6 @@ def ai(self) -> ai.AsyncAIResourceWithRawResponse: return AsyncAIResourceWithRawResponse(self._client.ai) - @cached_property - def style(self) -> style.AsyncStyleResourceWithRawResponse: - from .resources.style import AsyncStyleResourceWithRawResponse - - return AsyncStyleResourceWithRawResponse(self._client.style) - @cached_property def brand(self) -> brand.AsyncBrandResourceWithRawResponse: from .resources.brand import AsyncBrandResourceWithRawResponse @@ -563,12 +538,6 @@ def ai(self) -> ai.AIResourceWithStreamingResponse: return AIResourceWithStreamingResponse(self._client.ai) - @cached_property - def style(self) -> style.StyleResourceWithStreamingResponse: - from .resources.style import StyleResourceWithStreamingResponse - - return StyleResourceWithStreamingResponse(self._client.style) - @cached_property def brand(self) -> brand.BrandResourceWithStreamingResponse: from .resources.brand import BrandResourceWithStreamingResponse @@ -606,12 +575,6 @@ def ai(self) -> ai.AsyncAIResourceWithStreamingResponse: return AsyncAIResourceWithStreamingResponse(self._client.ai) - @cached_property - def style(self) -> style.AsyncStyleResourceWithStreamingResponse: - from .resources.style import AsyncStyleResourceWithStreamingResponse - - return AsyncStyleResourceWithStreamingResponse(self._client.style) - @cached_property def brand(self) -> brand.AsyncBrandResourceWithStreamingResponse: from .resources.brand import AsyncBrandResourceWithStreamingResponse diff --git a/src/context/dev/_files.py b/src/context/dev/_files.py index cc14c14..0fdce17 100644 --- a/src/context/dev/_files.py +++ b/src/context/dev/_files.py @@ -3,8 +3,8 @@ import io import os import pathlib -from typing import overload -from typing_extensions import TypeGuard +from typing import Sequence, cast, overload +from typing_extensions import TypeVar, TypeGuard import anyio @@ -17,7 +17,9 @@ HttpxFileContent, HttpxRequestFiles, ) -from ._utils import is_tuple_t, is_mapping_t, is_sequence_t +from ._utils import is_list, is_mapping, is_tuple_t, is_mapping_t, is_sequence_t + +_T = TypeVar("_T") def is_base64_file_input(obj: object) -> TypeGuard[Base64FileInput]: @@ -121,3 +123,51 @@ async def async_read_file_content(file: FileContent) -> HttpxFileContent: return await anyio.Path(file).read_bytes() return file + + +def deepcopy_with_paths(item: _T, paths: Sequence[Sequence[str]]) -> _T: + """Copy only the containers along the given paths. + + Used to guard against mutation by extract_files without copying the entire structure. + Only dicts and lists that lie on a path are copied; everything else + is returned by reference. + + For example, given paths=[["foo", "files", "file"]] and the structure: + { + "foo": { + "bar": {"baz": {}}, + "files": {"file": } + } + } + The root dict, "foo", and "files" are copied (they lie on the path). + "bar" and "baz" are returned by reference (off the path). + """ + return _deepcopy_with_paths(item, paths, 0) + + +def _deepcopy_with_paths(item: _T, paths: Sequence[Sequence[str]], index: int) -> _T: + if not paths: + return item + if is_mapping(item): + key_to_paths: dict[str, list[Sequence[str]]] = {} + for path in paths: + if index < len(path): + key_to_paths.setdefault(path[index], []).append(path) + + # if no path continues through this mapping, it won't be mutated and copying it is redundant + if not key_to_paths: + return item + + result = dict(item) + for key, subpaths in key_to_paths.items(): + if key in result: + result[key] = _deepcopy_with_paths(result[key], subpaths, index + 1) + return cast(_T, result) + if is_list(item): + array_paths = [path for path in paths if index < len(path) and path[index] == ""] + + # if no path expects a list here, nothing will be mutated inside it - return by reference + if not array_paths: + return cast(_T, item) + return cast(_T, [_deepcopy_with_paths(entry, array_paths, index + 1) for entry in item]) + return item diff --git a/src/context/dev/_utils/__init__.py b/src/context/dev/_utils/__init__.py index 10cb66d..1c090e5 100644 --- a/src/context/dev/_utils/__init__.py +++ b/src/context/dev/_utils/__init__.py @@ -24,7 +24,6 @@ coerce_integer as coerce_integer, file_from_path as file_from_path, strip_not_given as strip_not_given, - deepcopy_minimal as deepcopy_minimal, get_async_library as get_async_library, maybe_coerce_float as maybe_coerce_float, get_required_header as get_required_header, diff --git a/src/context/dev/_utils/_utils.py b/src/context/dev/_utils/_utils.py index eec7f4a..771859f 100644 --- a/src/context/dev/_utils/_utils.py +++ b/src/context/dev/_utils/_utils.py @@ -86,8 +86,9 @@ def _extract_items( index += 1 if is_dict(obj): try: - # We are at the last entry in the path so we must remove the field - if (len(path)) == index: + # Remove the field if there are no more dict keys in the path, + # only "" traversal markers or end. + if all(p == "" for p in path[index:]): item = obj.pop(key) else: item = obj[key] @@ -176,21 +177,6 @@ def is_iterable(obj: object) -> TypeGuard[Iterable[object]]: return isinstance(obj, Iterable) -def deepcopy_minimal(item: _T) -> _T: - """Minimal reimplementation of copy.deepcopy() that will only copy certain object types: - - - mappings, e.g. `dict` - - list - - This is done for performance reasons. - """ - if is_mapping(item): - return cast(_T, {k: deepcopy_minimal(v) for k, v in item.items()}) - if is_list(item): - return cast(_T, [deepcopy_minimal(entry) for entry in item]) - return item - - # copied from https://github.com/Rapptz/RoboDanny def human_join(seq: Sequence[str], *, delim: str = ", ", final: str = "or") -> str: size = len(seq) diff --git a/src/context/dev/_version.py b/src/context/dev/_version.py index 7b88fdc..e505869 100644 --- a/src/context/dev/_version.py +++ b/src/context/dev/_version.py @@ -1,4 +1,4 @@ # File generated from our OpenAPI spec by Stainless. See CONTRIBUTING.md for details. __title__ = "context.dev" -__version__ = "0.7.0" # x-release-please-version +__version__ = "0.8.0" # x-release-please-version diff --git a/src/context/dev/resources/__init__.py b/src/context/dev/resources/__init__.py index 2bcb6db..edc5356 100644 --- a/src/context/dev/resources/__init__.py +++ b/src/context/dev/resources/__init__.py @@ -24,14 +24,6 @@ BrandResourceWithStreamingResponse, AsyncBrandResourceWithStreamingResponse, ) -from .style import ( - StyleResource, - AsyncStyleResource, - StyleResourceWithRawResponse, - AsyncStyleResourceWithRawResponse, - StyleResourceWithStreamingResponse, - AsyncStyleResourceWithStreamingResponse, -) from .utility import ( UtilityResource, AsyncUtilityResource, @@ -62,12 +54,6 @@ "AsyncAIResourceWithRawResponse", "AIResourceWithStreamingResponse", "AsyncAIResourceWithStreamingResponse", - "StyleResource", - "AsyncStyleResource", - "StyleResourceWithRawResponse", - "AsyncStyleResourceWithRawResponse", - "StyleResourceWithStreamingResponse", - "AsyncStyleResourceWithStreamingResponse", "BrandResource", "AsyncBrandResource", "BrandResourceWithRawResponse", diff --git a/src/context/dev/resources/brand.py b/src/context/dev/resources/brand.py index 24c4252..3373a27 100644 --- a/src/context/dev/resources/brand.py +++ b/src/context/dev/resources/brand.py @@ -637,7 +637,6 @@ def identify_from_transaction( high_confidence_only: When set to true, the API will perform an additional verification steps to ensure the identified brand matches the transaction with high confidence. - Defaults to false. max_speed: Optional parameter to optimize the API call for maximum speed. When set to true, the API will skip time-consuming operations for faster response at the cost of @@ -823,9 +822,8 @@ def retrieve_by_email( ) -> BrandRetrieveByEmailResponse: """ Retrieve brand information using an email address while detecting disposable and - free email addresses. This endpoint extracts the domain from the email address - and returns brand data for that domain. Disposable and free email addresses - (like gmail.com, yahoo.com) will throw a 422 error. + free email addresses. Disposable and free email addresses (like gmail.com, + yahoo.com) will throw a 422 error. Args: email: Email address to retrieve brand data for (e.g., 'contact@example.com'). The @@ -1008,8 +1006,7 @@ def retrieve_by_isin( ) -> BrandRetrieveByIsinResponse: """ Retrieve brand information using an ISIN (International Securities - Identification Number). This endpoint looks up the company associated with the - ISIN and returns its brand data. + Identification Number). Args: isin: ISIN (International Securities Identification Number) to retrieve brand data for @@ -1432,17 +1429,15 @@ def retrieve_by_name( extra_body: Body | None = None, timeout: float | httpx.Timeout | None | NotGiven = not_given, ) -> BrandRetrieveByNameResponse: - """Retrieve brand information using a company name. - - This endpoint searches for the - company by name and returns its brand data. + """ + Retrieve brand information using a company name. Args: name: Company name to retrieve brand data for (e.g., 'Apple Inc', 'Microsoft Corporation'). Must be 3-30 characters. - country_gl: Optional country code (GL parameter) to specify the country. This affects the - geographic location used for search queries. + country_gl: Optional country code hint (GL parameter) to specify the country for the company + name. force_language: Optional parameter to force the language of the retrieved brand data. @@ -1694,10 +1689,8 @@ def retrieve_by_ticker( extra_body: Body | None = None, timeout: float | httpx.Timeout | None | NotGiven = not_given, ) -> BrandRetrieveByTickerResponse: - """Retrieve brand information using a stock ticker symbol. - - This endpoint looks up - the company associated with the ticker and returns its brand data. + """ + Retrieve brand information using a stock ticker symbol. Args: ticker: Stock ticker symbol to retrieve brand data for (e.g., 'AAPL', 'GOOGL', 'BRK.A'). @@ -1758,8 +1751,8 @@ def retrieve_simplified( ) -> BrandRetrieveSimplifiedResponse: """ Returns a simplified version of brand data containing only essential - information: domain, title, colors, logos, and backdrops. This endpoint is - optimized for faster responses and reduced data transfer. + information: domain, title, colors, logos, and backdrops. Optimized for faster + responses and reduced data transfer. Args: domain: Domain name to retrieve simplified brand data for @@ -2395,7 +2388,6 @@ async def identify_from_transaction( high_confidence_only: When set to true, the API will perform an additional verification steps to ensure the identified brand matches the transaction with high confidence. - Defaults to false. max_speed: Optional parameter to optimize the API call for maximum speed. When set to true, the API will skip time-consuming operations for faster response at the cost of @@ -2581,9 +2573,8 @@ async def retrieve_by_email( ) -> BrandRetrieveByEmailResponse: """ Retrieve brand information using an email address while detecting disposable and - free email addresses. This endpoint extracts the domain from the email address - and returns brand data for that domain. Disposable and free email addresses - (like gmail.com, yahoo.com) will throw a 422 error. + free email addresses. Disposable and free email addresses (like gmail.com, + yahoo.com) will throw a 422 error. Args: email: Email address to retrieve brand data for (e.g., 'contact@example.com'). The @@ -2766,8 +2757,7 @@ async def retrieve_by_isin( ) -> BrandRetrieveByIsinResponse: """ Retrieve brand information using an ISIN (International Securities - Identification Number). This endpoint looks up the company associated with the - ISIN and returns its brand data. + Identification Number). Args: isin: ISIN (International Securities Identification Number) to retrieve brand data for @@ -3190,17 +3180,15 @@ async def retrieve_by_name( extra_body: Body | None = None, timeout: float | httpx.Timeout | None | NotGiven = not_given, ) -> BrandRetrieveByNameResponse: - """Retrieve brand information using a company name. - - This endpoint searches for the - company by name and returns its brand data. + """ + Retrieve brand information using a company name. Args: name: Company name to retrieve brand data for (e.g., 'Apple Inc', 'Microsoft Corporation'). Must be 3-30 characters. - country_gl: Optional country code (GL parameter) to specify the country. This affects the - geographic location used for search queries. + country_gl: Optional country code hint (GL parameter) to specify the country for the company + name. force_language: Optional parameter to force the language of the retrieved brand data. @@ -3452,10 +3440,8 @@ async def retrieve_by_ticker( extra_body: Body | None = None, timeout: float | httpx.Timeout | None | NotGiven = not_given, ) -> BrandRetrieveByTickerResponse: - """Retrieve brand information using a stock ticker symbol. - - This endpoint looks up - the company associated with the ticker and returns its brand data. + """ + Retrieve brand information using a stock ticker symbol. Args: ticker: Stock ticker symbol to retrieve brand data for (e.g., 'AAPL', 'GOOGL', 'BRK.A'). @@ -3516,8 +3502,8 @@ async def retrieve_simplified( ) -> BrandRetrieveSimplifiedResponse: """ Returns a simplified version of brand data containing only essential - information: domain, title, colors, logos, and backdrops. This endpoint is - optimized for faster responses and reduced data transfer. + information: domain, title, colors, logos, and backdrops. Optimized for faster + responses and reduced data transfer. Args: domain: Domain name to retrieve simplified brand data for diff --git a/src/context/dev/resources/industry.py b/src/context/dev/resources/industry.py index ae71953..73c96ac 100644 --- a/src/context/dev/resources/industry.py +++ b/src/context/dev/resources/industry.py @@ -56,12 +56,12 @@ def retrieve_naics( timeout: float | httpx.Timeout | None | NotGiven = not_given, ) -> IndustryRetrieveNaicsResponse: """ - Endpoint to classify any brand into a 2022 NAICS code. + Classify any brand into 2022 NAICS industry codes from its domain or name. Args: - input: Brand domain or title to retrieve NAICS code for. If a valid domain is provided - in `input`, it will be used for classification, otherwise, we will search for - the brand using the provided title. + input: Brand domain or title to retrieve NAICS code for. If a valid domain is provided, + it will be used for classification, otherwise, we will search for the brand + using the provided title. max_results: Maximum number of NAICS codes to return. Must be between 1 and 10. Defaults to 5. @@ -81,7 +81,7 @@ def retrieve_naics( timeout: Override the client-level default timeout for this request, in seconds """ return self._get( - "/brand/naics", + "/web/naics", options=make_request_options( extra_headers=extra_headers, extra_query=extra_query, @@ -136,12 +136,12 @@ async def retrieve_naics( timeout: float | httpx.Timeout | None | NotGiven = not_given, ) -> IndustryRetrieveNaicsResponse: """ - Endpoint to classify any brand into a 2022 NAICS code. + Classify any brand into 2022 NAICS industry codes from its domain or name. Args: - input: Brand domain or title to retrieve NAICS code for. If a valid domain is provided - in `input`, it will be used for classification, otherwise, we will search for - the brand using the provided title. + input: Brand domain or title to retrieve NAICS code for. If a valid domain is provided, + it will be used for classification, otherwise, we will search for the brand + using the provided title. max_results: Maximum number of NAICS codes to return. Must be between 1 and 10. Defaults to 5. @@ -161,7 +161,7 @@ async def retrieve_naics( timeout: Override the client-level default timeout for this request, in seconds """ return await self._get( - "/brand/naics", + "/web/naics", options=make_request_options( extra_headers=extra_headers, extra_query=extra_query, diff --git a/src/context/dev/resources/style.py b/src/context/dev/resources/style.py deleted file mode 100644 index 36ea477..0000000 --- a/src/context/dev/resources/style.py +++ /dev/null @@ -1,326 +0,0 @@ -# File generated from our OpenAPI spec by Stainless. See CONTRIBUTING.md for details. - -from __future__ import annotations - -import httpx - -from ..types import style_extract_fonts_params, style_extract_styleguide_params -from .._types import Body, Omit, Query, Headers, NotGiven, omit, not_given -from .._utils import maybe_transform, async_maybe_transform -from .._compat import cached_property -from .._resource import SyncAPIResource, AsyncAPIResource -from .._response import ( - to_raw_response_wrapper, - to_streamed_response_wrapper, - async_to_raw_response_wrapper, - async_to_streamed_response_wrapper, -) -from .._base_client import make_request_options -from ..types.style_extract_fonts_response import StyleExtractFontsResponse -from ..types.style_extract_styleguide_response import StyleExtractStyleguideResponse - -__all__ = ["StyleResource", "AsyncStyleResource"] - - -class StyleResource(SyncAPIResource): - @cached_property - def with_raw_response(self) -> StyleResourceWithRawResponse: - """ - This property can be used as a prefix for any HTTP method call to return - the raw response object instead of the parsed content. - - For more information, see https://www.github.com/context-dot-dev/context-python-sdk#accessing-raw-response-data-eg-headers - """ - return StyleResourceWithRawResponse(self) - - @cached_property - def with_streaming_response(self) -> StyleResourceWithStreamingResponse: - """ - An alternative to `.with_raw_response` that doesn't eagerly read the response body. - - For more information, see https://www.github.com/context-dot-dev/context-python-sdk#with_streaming_response - """ - return StyleResourceWithStreamingResponse(self) - - def extract_fonts( - self, - *, - domain: str, - timeout_ms: int | Omit = omit, - # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs. - # The extra values given here take precedence over values defined on the client or passed to this method. - extra_headers: Headers | None = None, - extra_query: Query | None = None, - extra_body: Body | None = None, - timeout: float | httpx.Timeout | None | NotGiven = not_given, - ) -> StyleExtractFontsResponse: - """ - Extract font information from a brand's website including font families, usage - statistics, fallbacks, and element/word counts. - - Args: - domain: Domain name to extract fonts from (e.g., 'example.com', 'google.com'). The - domain will be automatically normalized and validated. - - timeout_ms: Optional timeout in milliseconds for the request. If the request takes longer - than this value, it will be aborted with a 408 status code. Maximum allowed - value is 300000ms (5 minutes). - - extra_headers: Send extra headers - - extra_query: Add additional query parameters to the request - - extra_body: Add additional JSON properties to the request - - timeout: Override the client-level default timeout for this request, in seconds - """ - return self._get( - "/brand/fonts", - options=make_request_options( - extra_headers=extra_headers, - extra_query=extra_query, - extra_body=extra_body, - timeout=timeout, - query=maybe_transform( - { - "domain": domain, - "timeout_ms": timeout_ms, - }, - style_extract_fonts_params.StyleExtractFontsParams, - ), - ), - cast_to=StyleExtractFontsResponse, - ) - - def extract_styleguide( - self, - *, - direct_url: str | Omit = omit, - domain: str | Omit = omit, - timeout_ms: int | Omit = omit, - # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs. - # The extra values given here take precedence over values defined on the client or passed to this method. - extra_headers: Headers | None = None, - extra_query: Query | None = None, - extra_body: Body | None = None, - timeout: float | httpx.Timeout | None | NotGiven = not_given, - ) -> StyleExtractStyleguideResponse: - """ - Automatically extract comprehensive design system information from a brand's - website including colors, typography, spacing, shadows, and UI components. - Either 'domain' or 'directUrl' must be provided as a query parameter, but not - both. - - Args: - direct_url: A specific URL to fetch the styleguide from directly, bypassing domain - resolution (e.g., 'https://example.com/design-system'). - - domain: Domain name to extract styleguide from (e.g., 'example.com', 'google.com'). The - domain will be automatically normalized and validated. - - timeout_ms: Optional timeout in milliseconds for the request. If the request takes longer - than this value, it will be aborted with a 408 status code. Maximum allowed - value is 300000ms (5 minutes). - - extra_headers: Send extra headers - - extra_query: Add additional query parameters to the request - - extra_body: Add additional JSON properties to the request - - timeout: Override the client-level default timeout for this request, in seconds - """ - return self._get( - "/brand/styleguide", - options=make_request_options( - extra_headers=extra_headers, - extra_query=extra_query, - extra_body=extra_body, - timeout=timeout, - query=maybe_transform( - { - "direct_url": direct_url, - "domain": domain, - "timeout_ms": timeout_ms, - }, - style_extract_styleguide_params.StyleExtractStyleguideParams, - ), - ), - cast_to=StyleExtractStyleguideResponse, - ) - - -class AsyncStyleResource(AsyncAPIResource): - @cached_property - def with_raw_response(self) -> AsyncStyleResourceWithRawResponse: - """ - This property can be used as a prefix for any HTTP method call to return - the raw response object instead of the parsed content. - - For more information, see https://www.github.com/context-dot-dev/context-python-sdk#accessing-raw-response-data-eg-headers - """ - return AsyncStyleResourceWithRawResponse(self) - - @cached_property - def with_streaming_response(self) -> AsyncStyleResourceWithStreamingResponse: - """ - An alternative to `.with_raw_response` that doesn't eagerly read the response body. - - For more information, see https://www.github.com/context-dot-dev/context-python-sdk#with_streaming_response - """ - return AsyncStyleResourceWithStreamingResponse(self) - - async def extract_fonts( - self, - *, - domain: str, - timeout_ms: int | Omit = omit, - # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs. - # The extra values given here take precedence over values defined on the client or passed to this method. - extra_headers: Headers | None = None, - extra_query: Query | None = None, - extra_body: Body | None = None, - timeout: float | httpx.Timeout | None | NotGiven = not_given, - ) -> StyleExtractFontsResponse: - """ - Extract font information from a brand's website including font families, usage - statistics, fallbacks, and element/word counts. - - Args: - domain: Domain name to extract fonts from (e.g., 'example.com', 'google.com'). The - domain will be automatically normalized and validated. - - timeout_ms: Optional timeout in milliseconds for the request. If the request takes longer - than this value, it will be aborted with a 408 status code. Maximum allowed - value is 300000ms (5 minutes). - - extra_headers: Send extra headers - - extra_query: Add additional query parameters to the request - - extra_body: Add additional JSON properties to the request - - timeout: Override the client-level default timeout for this request, in seconds - """ - return await self._get( - "/brand/fonts", - options=make_request_options( - extra_headers=extra_headers, - extra_query=extra_query, - extra_body=extra_body, - timeout=timeout, - query=await async_maybe_transform( - { - "domain": domain, - "timeout_ms": timeout_ms, - }, - style_extract_fonts_params.StyleExtractFontsParams, - ), - ), - cast_to=StyleExtractFontsResponse, - ) - - async def extract_styleguide( - self, - *, - direct_url: str | Omit = omit, - domain: str | Omit = omit, - timeout_ms: int | Omit = omit, - # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs. - # The extra values given here take precedence over values defined on the client or passed to this method. - extra_headers: Headers | None = None, - extra_query: Query | None = None, - extra_body: Body | None = None, - timeout: float | httpx.Timeout | None | NotGiven = not_given, - ) -> StyleExtractStyleguideResponse: - """ - Automatically extract comprehensive design system information from a brand's - website including colors, typography, spacing, shadows, and UI components. - Either 'domain' or 'directUrl' must be provided as a query parameter, but not - both. - - Args: - direct_url: A specific URL to fetch the styleguide from directly, bypassing domain - resolution (e.g., 'https://example.com/design-system'). - - domain: Domain name to extract styleguide from (e.g., 'example.com', 'google.com'). The - domain will be automatically normalized and validated. - - timeout_ms: Optional timeout in milliseconds for the request. If the request takes longer - than this value, it will be aborted with a 408 status code. Maximum allowed - value is 300000ms (5 minutes). - - extra_headers: Send extra headers - - extra_query: Add additional query parameters to the request - - extra_body: Add additional JSON properties to the request - - timeout: Override the client-level default timeout for this request, in seconds - """ - return await self._get( - "/brand/styleguide", - options=make_request_options( - extra_headers=extra_headers, - extra_query=extra_query, - extra_body=extra_body, - timeout=timeout, - query=await async_maybe_transform( - { - "direct_url": direct_url, - "domain": domain, - "timeout_ms": timeout_ms, - }, - style_extract_styleguide_params.StyleExtractStyleguideParams, - ), - ), - cast_to=StyleExtractStyleguideResponse, - ) - - -class StyleResourceWithRawResponse: - def __init__(self, style: StyleResource) -> None: - self._style = style - - self.extract_fonts = to_raw_response_wrapper( - style.extract_fonts, - ) - self.extract_styleguide = to_raw_response_wrapper( - style.extract_styleguide, - ) - - -class AsyncStyleResourceWithRawResponse: - def __init__(self, style: AsyncStyleResource) -> None: - self._style = style - - self.extract_fonts = async_to_raw_response_wrapper( - style.extract_fonts, - ) - self.extract_styleguide = async_to_raw_response_wrapper( - style.extract_styleguide, - ) - - -class StyleResourceWithStreamingResponse: - def __init__(self, style: StyleResource) -> None: - self._style = style - - self.extract_fonts = to_streamed_response_wrapper( - style.extract_fonts, - ) - self.extract_styleguide = to_streamed_response_wrapper( - style.extract_styleguide, - ) - - -class AsyncStyleResourceWithStreamingResponse: - def __init__(self, style: AsyncStyleResource) -> None: - self._style = style - - self.extract_fonts = async_to_streamed_response_wrapper( - style.extract_fonts, - ) - self.extract_styleguide = async_to_streamed_response_wrapper( - style.extract_styleguide, - ) diff --git a/src/context/dev/resources/web.py b/src/context/dev/resources/web.py index 33c3d6f..2b68c4f 100644 --- a/src/context/dev/resources/web.py +++ b/src/context/dev/resources/web.py @@ -9,9 +9,11 @@ from ..types import ( web_screenshot_params, web_web_crawl_md_params, + web_extract_fonts_params, web_web_scrape_md_params, web_web_scrape_html_params, web_web_scrape_images_params, + web_extract_styleguide_params, web_web_scrape_sitemap_params, ) from .._types import Body, Omit, Query, Headers, NotGiven, omit, not_given @@ -27,9 +29,11 @@ from .._base_client import make_request_options from ..types.web_screenshot_response import WebScreenshotResponse from ..types.web_web_crawl_md_response import WebWebCrawlMdResponse +from ..types.web_extract_fonts_response import WebExtractFontsResponse from ..types.web_web_scrape_md_response import WebWebScrapeMdResponse from ..types.web_web_scrape_html_response import WebWebScrapeHTMLResponse from ..types.web_web_scrape_images_response import WebWebScrapeImagesResponse +from ..types.web_extract_styleguide_response import WebExtractStyleguideResponse from ..types.web_web_scrape_sitemap_response import WebWebScrapeSitemapResponse __all__ = ["WebResource", "AsyncWebResource"] @@ -55,6 +59,121 @@ def with_streaming_response(self) -> WebResourceWithStreamingResponse: """ return WebResourceWithStreamingResponse(self) + def extract_fonts( + self, + *, + direct_url: str | Omit = omit, + domain: str | Omit = omit, + timeout_ms: int | Omit = omit, + # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs. + # The extra values given here take precedence over values defined on the client or passed to this method. + extra_headers: Headers | None = None, + extra_query: Query | None = None, + extra_body: Body | None = None, + timeout: float | httpx.Timeout | None | NotGiven = not_given, + ) -> WebExtractFontsResponse: + """ + Scrape font information from a website including font families, usage + statistics, fallbacks, and element/word counts. + + Args: + direct_url: A specific URL to fetch fonts from directly, bypassing domain resolution (e.g., + 'https://example.com/design-system'). When provided, fonts are extracted from + this exact URL. You must provide either 'domain' or 'directUrl', but not both. + + domain: Domain name to extract fonts from (e.g., 'example.com', 'google.com'). The + domain will be automatically normalized and validated. You must provide either + 'domain' or 'directUrl', but not both. + + timeout_ms: Optional timeout in milliseconds for the request. If the request takes longer + than this value, it will be aborted with a 408 status code. Maximum allowed + value is 300000ms (5 minutes). + + extra_headers: Send extra headers + + extra_query: Add additional query parameters to the request + + extra_body: Add additional JSON properties to the request + + timeout: Override the client-level default timeout for this request, in seconds + """ + return self._get( + "/web/fonts", + options=make_request_options( + extra_headers=extra_headers, + extra_query=extra_query, + extra_body=extra_body, + timeout=timeout, + query=maybe_transform( + { + "direct_url": direct_url, + "domain": domain, + "timeout_ms": timeout_ms, + }, + web_extract_fonts_params.WebExtractFontsParams, + ), + ), + cast_to=WebExtractFontsResponse, + ) + + def extract_styleguide( + self, + *, + direct_url: str | Omit = omit, + domain: str | Omit = omit, + timeout_ms: int | Omit = omit, + # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs. + # The extra values given here take precedence over values defined on the client or passed to this method. + extra_headers: Headers | None = None, + extra_query: Query | None = None, + extra_body: Body | None = None, + timeout: float | httpx.Timeout | None | NotGiven = not_given, + ) -> WebExtractStyleguideResponse: + """ + Extract a comprehensive design system from a website including colors, + typography, spacing, shadows, and UI components. + + Args: + direct_url: A specific URL to fetch the styleguide from directly, bypassing domain + resolution (e.g., 'https://example.com/design-system'). When provided, the + styleguide is extracted from this exact URL. You must provide either 'domain' or + 'directUrl', but not both. + + domain: Domain name to extract styleguide from (e.g., 'example.com', 'google.com'). The + domain will be automatically normalized and validated. You must provide either + 'domain' or 'directUrl', but not both. + + timeout_ms: Optional timeout in milliseconds for the request. If the request takes longer + than this value, it will be aborted with a 408 status code. Maximum allowed + value is 300000ms (5 minutes). + + extra_headers: Send extra headers + + extra_query: Add additional query parameters to the request + + extra_body: Add additional JSON properties to the request + + timeout: Override the client-level default timeout for this request, in seconds + """ + return self._get( + "/web/styleguide", + options=make_request_options( + extra_headers=extra_headers, + extra_query=extra_query, + extra_body=extra_body, + timeout=timeout, + query=maybe_transform( + { + "direct_url": direct_url, + "domain": domain, + "timeout_ms": timeout_ms, + }, + web_extract_styleguide_params.WebExtractStyleguideParams, + ), + ), + cast_to=WebExtractStyleguideResponse, + ) + def screenshot( self, *, @@ -70,21 +189,17 @@ def screenshot( extra_body: Body | None = None, timeout: float | httpx.Timeout | None | NotGiven = not_given, ) -> WebScreenshotResponse: - """Capture a screenshot of a website. - - Supports both viewport (standard browser - view) and full-page screenshots. Can also screenshot specific page types (login, - pricing, etc.) by using heuristics to find the appropriate URL. Either 'domain' - or 'directUrl' must be provided as a query parameter, but not both. Returns a - URL to the uploaded screenshot image hosted on our CDN. + """ + Capture a screenshot of a website. Args: direct_url: A specific URL to screenshot directly, bypassing domain resolution (e.g., 'https://example.com/pricing'). When provided, the screenshot is taken of this - exact URL. + exact URL. You must provide either 'domain' or 'directUrl', but not both. domain: Domain name to take screenshot of (e.g., 'example.com', 'google.com'). The - domain will be automatically normalized and validated. + domain will be automatically normalized and validated. You must provide either + 'domain' or 'directUrl', but not both. full_screenshot: Optional parameter to determine screenshot type. If 'true', takes a full page screenshot capturing all content. If 'false' or not provided, takes a viewport @@ -109,7 +224,7 @@ def screenshot( timeout: Override the client-level default timeout for this request, in seconds """ return self._get( - "/brand/screenshot", + "/web/screenshot", options=make_request_options( extra_headers=extra_headers, extra_query=extra_query, @@ -150,8 +265,7 @@ def web_crawl_md( ) -> WebWebCrawlMdResponse: """ Performs a crawl starting from a given URL, extracts page content as Markdown, - and returns results for all crawled pages. Only follows links within the same - domain as the starting URL. Costs 1 credit per successful page crawled. + and returns results for all crawled pages. Args: url: The starting URL for the crawl (must include http:// or https:// protocol) @@ -209,6 +323,7 @@ def web_scrape_html( self, *, url: str, + max_age_ms: int | Omit = omit, # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs. # The extra values given here take precedence over values defined on the client or passed to this method. extra_headers: Headers | None = None, @@ -222,6 +337,10 @@ def web_scrape_html( Args: url: Full URL to scrape (must include http:// or https:// protocol) + max_age_ms: Return a cached result if a prior scrape for the same parameters exists and is + younger than this many milliseconds. Defaults to 1 day (86400000 ms) when + omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh. + extra_headers: Send extra headers extra_query: Add additional query parameters to the request @@ -237,7 +356,13 @@ def web_scrape_html( extra_query=extra_query, extra_body=extra_body, timeout=timeout, - query=maybe_transform({"url": url}, web_web_scrape_html_params.WebWebScrapeHTMLParams), + query=maybe_transform( + { + "url": url, + "max_age_ms": max_age_ms, + }, + web_web_scrape_html_params.WebWebScrapeHTMLParams, + ), ), cast_to=WebWebScrapeHTMLResponse, ) @@ -288,6 +413,7 @@ def web_scrape_md( url: str, include_images: bool | Omit = omit, include_links: bool | Omit = omit, + max_age_ms: int | Omit = omit, shorten_base64_images: bool | Omit = omit, use_main_content_only: bool | Omit = omit, # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs. @@ -298,17 +424,20 @@ def web_scrape_md( timeout: float | httpx.Timeout | None | NotGiven = not_given, ) -> WebWebScrapeMdResponse: """ - Scrapes the given URL, converts the HTML content to Markdown, and returns the - result. + Scrapes the given URL into LLM usable Markdown. Args: - url: Full URL to scrape and convert to markdown (must include http:// or https:// + url: Full URL to scrape into LLM usable Markdown (must include http:// or https:// protocol) include_images: Include image references in Markdown output include_links: Preserve hyperlinks in Markdown output + max_age_ms: Return a cached result if a prior scrape for the same parameters exists and is + younger than this many milliseconds. Defaults to 1 day (86400000 ms) when + omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh. + shorten_base64_images: Shorten base64-encoded image data in the Markdown output use_main_content_only: Extract only the main content of the page, excluding headers, footers, sidebars, @@ -334,6 +463,7 @@ def web_scrape_md( "url": url, "include_images": include_images, "include_links": include_links, + "max_age_ms": max_age_ms, "shorten_base64_images": shorten_base64_images, "use_main_content_only": use_main_content_only, }, @@ -356,13 +486,10 @@ def web_scrape_sitemap( timeout: float | httpx.Timeout | None | NotGiven = not_given, ) -> WebWebScrapeSitemapResponse: """ - Crawls the sitemap of the given domain and returns all discovered page URLs. - Supports sitemap index files (recursive), parallel fetching with concurrency - control, deduplication, and filters out non-page resources (images, PDFs, etc.). + Crawl an entire website's sitemap and return all discovered page URLs. Args: - domain: Domain name to crawl sitemaps for (e.g., 'example.com'). The domain will be - automatically normalized and validated. + domain: Domain to build a sitemap for max_links: Maximum number of links to return from the sitemap crawl. Defaults to 10,000. Minimum is 1, maximum is 100,000. @@ -414,6 +541,121 @@ def with_streaming_response(self) -> AsyncWebResourceWithStreamingResponse: """ return AsyncWebResourceWithStreamingResponse(self) + async def extract_fonts( + self, + *, + direct_url: str | Omit = omit, + domain: str | Omit = omit, + timeout_ms: int | Omit = omit, + # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs. + # The extra values given here take precedence over values defined on the client or passed to this method. + extra_headers: Headers | None = None, + extra_query: Query | None = None, + extra_body: Body | None = None, + timeout: float | httpx.Timeout | None | NotGiven = not_given, + ) -> WebExtractFontsResponse: + """ + Scrape font information from a website including font families, usage + statistics, fallbacks, and element/word counts. + + Args: + direct_url: A specific URL to fetch fonts from directly, bypassing domain resolution (e.g., + 'https://example.com/design-system'). When provided, fonts are extracted from + this exact URL. You must provide either 'domain' or 'directUrl', but not both. + + domain: Domain name to extract fonts from (e.g., 'example.com', 'google.com'). The + domain will be automatically normalized and validated. You must provide either + 'domain' or 'directUrl', but not both. + + timeout_ms: Optional timeout in milliseconds for the request. If the request takes longer + than this value, it will be aborted with a 408 status code. Maximum allowed + value is 300000ms (5 minutes). + + extra_headers: Send extra headers + + extra_query: Add additional query parameters to the request + + extra_body: Add additional JSON properties to the request + + timeout: Override the client-level default timeout for this request, in seconds + """ + return await self._get( + "/web/fonts", + options=make_request_options( + extra_headers=extra_headers, + extra_query=extra_query, + extra_body=extra_body, + timeout=timeout, + query=await async_maybe_transform( + { + "direct_url": direct_url, + "domain": domain, + "timeout_ms": timeout_ms, + }, + web_extract_fonts_params.WebExtractFontsParams, + ), + ), + cast_to=WebExtractFontsResponse, + ) + + async def extract_styleguide( + self, + *, + direct_url: str | Omit = omit, + domain: str | Omit = omit, + timeout_ms: int | Omit = omit, + # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs. + # The extra values given here take precedence over values defined on the client or passed to this method. + extra_headers: Headers | None = None, + extra_query: Query | None = None, + extra_body: Body | None = None, + timeout: float | httpx.Timeout | None | NotGiven = not_given, + ) -> WebExtractStyleguideResponse: + """ + Extract a comprehensive design system from a website including colors, + typography, spacing, shadows, and UI components. + + Args: + direct_url: A specific URL to fetch the styleguide from directly, bypassing domain + resolution (e.g., 'https://example.com/design-system'). When provided, the + styleguide is extracted from this exact URL. You must provide either 'domain' or + 'directUrl', but not both. + + domain: Domain name to extract styleguide from (e.g., 'example.com', 'google.com'). The + domain will be automatically normalized and validated. You must provide either + 'domain' or 'directUrl', but not both. + + timeout_ms: Optional timeout in milliseconds for the request. If the request takes longer + than this value, it will be aborted with a 408 status code. Maximum allowed + value is 300000ms (5 minutes). + + extra_headers: Send extra headers + + extra_query: Add additional query parameters to the request + + extra_body: Add additional JSON properties to the request + + timeout: Override the client-level default timeout for this request, in seconds + """ + return await self._get( + "/web/styleguide", + options=make_request_options( + extra_headers=extra_headers, + extra_query=extra_query, + extra_body=extra_body, + timeout=timeout, + query=await async_maybe_transform( + { + "direct_url": direct_url, + "domain": domain, + "timeout_ms": timeout_ms, + }, + web_extract_styleguide_params.WebExtractStyleguideParams, + ), + ), + cast_to=WebExtractStyleguideResponse, + ) + async def screenshot( self, *, @@ -429,21 +671,17 @@ async def screenshot( extra_body: Body | None = None, timeout: float | httpx.Timeout | None | NotGiven = not_given, ) -> WebScreenshotResponse: - """Capture a screenshot of a website. - - Supports both viewport (standard browser - view) and full-page screenshots. Can also screenshot specific page types (login, - pricing, etc.) by using heuristics to find the appropriate URL. Either 'domain' - or 'directUrl' must be provided as a query parameter, but not both. Returns a - URL to the uploaded screenshot image hosted on our CDN. + """ + Capture a screenshot of a website. Args: direct_url: A specific URL to screenshot directly, bypassing domain resolution (e.g., 'https://example.com/pricing'). When provided, the screenshot is taken of this - exact URL. + exact URL. You must provide either 'domain' or 'directUrl', but not both. domain: Domain name to take screenshot of (e.g., 'example.com', 'google.com'). The - domain will be automatically normalized and validated. + domain will be automatically normalized and validated. You must provide either + 'domain' or 'directUrl', but not both. full_screenshot: Optional parameter to determine screenshot type. If 'true', takes a full page screenshot capturing all content. If 'false' or not provided, takes a viewport @@ -468,7 +706,7 @@ async def screenshot( timeout: Override the client-level default timeout for this request, in seconds """ return await self._get( - "/brand/screenshot", + "/web/screenshot", options=make_request_options( extra_headers=extra_headers, extra_query=extra_query, @@ -509,8 +747,7 @@ async def web_crawl_md( ) -> WebWebCrawlMdResponse: """ Performs a crawl starting from a given URL, extracts page content as Markdown, - and returns results for all crawled pages. Only follows links within the same - domain as the starting URL. Costs 1 credit per successful page crawled. + and returns results for all crawled pages. Args: url: The starting URL for the crawl (must include http:// or https:// protocol) @@ -568,6 +805,7 @@ async def web_scrape_html( self, *, url: str, + max_age_ms: int | Omit = omit, # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs. # The extra values given here take precedence over values defined on the client or passed to this method. extra_headers: Headers | None = None, @@ -581,6 +819,10 @@ async def web_scrape_html( Args: url: Full URL to scrape (must include http:// or https:// protocol) + max_age_ms: Return a cached result if a prior scrape for the same parameters exists and is + younger than this many milliseconds. Defaults to 1 day (86400000 ms) when + omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh. + extra_headers: Send extra headers extra_query: Add additional query parameters to the request @@ -596,7 +838,13 @@ async def web_scrape_html( extra_query=extra_query, extra_body=extra_body, timeout=timeout, - query=await async_maybe_transform({"url": url}, web_web_scrape_html_params.WebWebScrapeHTMLParams), + query=await async_maybe_transform( + { + "url": url, + "max_age_ms": max_age_ms, + }, + web_web_scrape_html_params.WebWebScrapeHTMLParams, + ), ), cast_to=WebWebScrapeHTMLResponse, ) @@ -647,6 +895,7 @@ async def web_scrape_md( url: str, include_images: bool | Omit = omit, include_links: bool | Omit = omit, + max_age_ms: int | Omit = omit, shorten_base64_images: bool | Omit = omit, use_main_content_only: bool | Omit = omit, # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs. @@ -657,17 +906,20 @@ async def web_scrape_md( timeout: float | httpx.Timeout | None | NotGiven = not_given, ) -> WebWebScrapeMdResponse: """ - Scrapes the given URL, converts the HTML content to Markdown, and returns the - result. + Scrapes the given URL into LLM usable Markdown. Args: - url: Full URL to scrape and convert to markdown (must include http:// or https:// + url: Full URL to scrape into LLM usable Markdown (must include http:// or https:// protocol) include_images: Include image references in Markdown output include_links: Preserve hyperlinks in Markdown output + max_age_ms: Return a cached result if a prior scrape for the same parameters exists and is + younger than this many milliseconds. Defaults to 1 day (86400000 ms) when + omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh. + shorten_base64_images: Shorten base64-encoded image data in the Markdown output use_main_content_only: Extract only the main content of the page, excluding headers, footers, sidebars, @@ -693,6 +945,7 @@ async def web_scrape_md( "url": url, "include_images": include_images, "include_links": include_links, + "max_age_ms": max_age_ms, "shorten_base64_images": shorten_base64_images, "use_main_content_only": use_main_content_only, }, @@ -715,13 +968,10 @@ async def web_scrape_sitemap( timeout: float | httpx.Timeout | None | NotGiven = not_given, ) -> WebWebScrapeSitemapResponse: """ - Crawls the sitemap of the given domain and returns all discovered page URLs. - Supports sitemap index files (recursive), parallel fetching with concurrency - control, deduplication, and filters out non-page resources (images, PDFs, etc.). + Crawl an entire website's sitemap and return all discovered page URLs. Args: - domain: Domain name to crawl sitemaps for (e.g., 'example.com'). The domain will be - automatically normalized and validated. + domain: Domain to build a sitemap for max_links: Maximum number of links to return from the sitemap crawl. Defaults to 10,000. Minimum is 1, maximum is 100,000. @@ -757,6 +1007,12 @@ class WebResourceWithRawResponse: def __init__(self, web: WebResource) -> None: self._web = web + self.extract_fonts = to_raw_response_wrapper( + web.extract_fonts, + ) + self.extract_styleguide = to_raw_response_wrapper( + web.extract_styleguide, + ) self.screenshot = to_raw_response_wrapper( web.screenshot, ) @@ -781,6 +1037,12 @@ class AsyncWebResourceWithRawResponse: def __init__(self, web: AsyncWebResource) -> None: self._web = web + self.extract_fonts = async_to_raw_response_wrapper( + web.extract_fonts, + ) + self.extract_styleguide = async_to_raw_response_wrapper( + web.extract_styleguide, + ) self.screenshot = async_to_raw_response_wrapper( web.screenshot, ) @@ -805,6 +1067,12 @@ class WebResourceWithStreamingResponse: def __init__(self, web: WebResource) -> None: self._web = web + self.extract_fonts = to_streamed_response_wrapper( + web.extract_fonts, + ) + self.extract_styleguide = to_streamed_response_wrapper( + web.extract_styleguide, + ) self.screenshot = to_streamed_response_wrapper( web.screenshot, ) @@ -829,6 +1097,12 @@ class AsyncWebResourceWithStreamingResponse: def __init__(self, web: AsyncWebResource) -> None: self._web = web + self.extract_fonts = async_to_streamed_response_wrapper( + web.extract_fonts, + ) + self.extract_styleguide = async_to_streamed_response_wrapper( + web.extract_styleguide, + ) self.screenshot = async_to_streamed_response_wrapper( web.screenshot, ) diff --git a/src/context/dev/types/__init__.py b/src/context/dev/types/__init__.py index 4053426..232daf5 100644 --- a/src/context/dev/types/__init__.py +++ b/src/context/dev/types/__init__.py @@ -10,21 +10,22 @@ from .utility_prefetch_params import UtilityPrefetchParams as UtilityPrefetchParams from .web_screenshot_response import WebScreenshotResponse as WebScreenshotResponse from .web_web_crawl_md_params import WebWebCrawlMdParams as WebWebCrawlMdParams +from .web_extract_fonts_params import WebExtractFontsParams as WebExtractFontsParams from .web_web_scrape_md_params import WebWebScrapeMdParams as WebWebScrapeMdParams from .ai_extract_product_params import AIExtractProductParams as AIExtractProductParams from .utility_prefetch_response import UtilityPrefetchResponse as UtilityPrefetchResponse from .web_web_crawl_md_response import WebWebCrawlMdResponse as WebWebCrawlMdResponse from .ai_extract_products_params import AIExtractProductsParams as AIExtractProductsParams -from .style_extract_fonts_params import StyleExtractFontsParams as StyleExtractFontsParams +from .web_extract_fonts_response import WebExtractFontsResponse as WebExtractFontsResponse from .web_web_scrape_html_params import WebWebScrapeHTMLParams as WebWebScrapeHTMLParams from .web_web_scrape_md_response import WebWebScrapeMdResponse as WebWebScrapeMdResponse from .ai_extract_product_response import AIExtractProductResponse as AIExtractProductResponse from .ai_extract_products_response import AIExtractProductsResponse as AIExtractProductsResponse -from .style_extract_fonts_response import StyleExtractFontsResponse as StyleExtractFontsResponse from .web_web_scrape_html_response import WebWebScrapeHTMLResponse as WebWebScrapeHTMLResponse from .web_web_scrape_images_params import WebWebScrapeImagesParams as WebWebScrapeImagesParams from .brand_retrieve_by_isin_params import BrandRetrieveByIsinParams as BrandRetrieveByIsinParams from .brand_retrieve_by_name_params import BrandRetrieveByNameParams as BrandRetrieveByNameParams +from .web_extract_styleguide_params import WebExtractStyleguideParams as WebExtractStyleguideParams from .web_web_scrape_sitemap_params import WebWebScrapeSitemapParams as WebWebScrapeSitemapParams from .brand_retrieve_by_email_params import BrandRetrieveByEmailParams as BrandRetrieveByEmailParams from .industry_retrieve_naics_params import IndustryRetrieveNaicsParams as IndustryRetrieveNaicsParams @@ -32,14 +33,13 @@ from .brand_retrieve_by_isin_response import BrandRetrieveByIsinResponse as BrandRetrieveByIsinResponse from .brand_retrieve_by_name_response import BrandRetrieveByNameResponse as BrandRetrieveByNameResponse from .brand_retrieve_by_ticker_params import BrandRetrieveByTickerParams as BrandRetrieveByTickerParams -from .style_extract_styleguide_params import StyleExtractStyleguideParams as StyleExtractStyleguideParams +from .web_extract_styleguide_response import WebExtractStyleguideResponse as WebExtractStyleguideResponse from .web_web_scrape_sitemap_response import WebWebScrapeSitemapResponse as WebWebScrapeSitemapResponse from .brand_retrieve_by_email_response import BrandRetrieveByEmailResponse as BrandRetrieveByEmailResponse from .brand_retrieve_simplified_params import BrandRetrieveSimplifiedParams as BrandRetrieveSimplifiedParams from .industry_retrieve_naics_response import IndustryRetrieveNaicsResponse as IndustryRetrieveNaicsResponse from .utility_prefetch_by_email_params import UtilityPrefetchByEmailParams as UtilityPrefetchByEmailParams from .brand_retrieve_by_ticker_response import BrandRetrieveByTickerResponse as BrandRetrieveByTickerResponse -from .style_extract_styleguide_response import StyleExtractStyleguideResponse as StyleExtractStyleguideResponse from .brand_retrieve_simplified_response import BrandRetrieveSimplifiedResponse as BrandRetrieveSimplifiedResponse from .utility_prefetch_by_email_response import UtilityPrefetchByEmailResponse as UtilityPrefetchByEmailResponse from .brand_identify_from_transaction_params import ( diff --git a/src/context/dev/types/brand_identify_from_transaction_params.py b/src/context/dev/types/brand_identify_from_transaction_params.py index bd3f5a8..e6ecadd 100644 --- a/src/context/dev/types/brand_identify_from_transaction_params.py +++ b/src/context/dev/types/brand_identify_from_transaction_params.py @@ -390,7 +390,6 @@ class BrandIdentifyFromTransactionParams(TypedDict, total=False): """ When set to true, the API will perform an additional verification steps to ensure the identified brand matches the transaction with high confidence. - Defaults to false. """ max_speed: Annotated[bool, PropertyInfo(alias="maxSpeed")] diff --git a/src/context/dev/types/brand_retrieve_by_name_params.py b/src/context/dev/types/brand_retrieve_by_name_params.py index 31a9eef..d774223 100644 --- a/src/context/dev/types/brand_retrieve_by_name_params.py +++ b/src/context/dev/types/brand_retrieve_by_name_params.py @@ -257,9 +257,9 @@ class BrandRetrieveByNameParams(TypedDict, total=False): "zm", "zw", ] - """Optional country code (GL parameter) to specify the country. - - This affects the geographic location used for search queries. + """ + Optional country code hint (GL parameter) to specify the country for the company + name. """ force_language: Literal[ diff --git a/src/context/dev/types/industry_retrieve_naics_params.py b/src/context/dev/types/industry_retrieve_naics_params.py index cbaed87..03db484 100644 --- a/src/context/dev/types/industry_retrieve_naics_params.py +++ b/src/context/dev/types/industry_retrieve_naics_params.py @@ -13,8 +13,8 @@ class IndustryRetrieveNaicsParams(TypedDict, total=False): input: Required[str] """Brand domain or title to retrieve NAICS code for. - If a valid domain is provided in `input`, it will be used for classification, - otherwise, we will search for the brand using the provided title. + If a valid domain is provided, it will be used for classification, otherwise, we + will search for the brand using the provided title. """ max_results: Annotated[int, PropertyInfo(alias="maxResults")] diff --git a/src/context/dev/types/style_extract_fonts_params.py b/src/context/dev/types/style_extract_fonts_params.py deleted file mode 100644 index 178f6f3..0000000 --- a/src/context/dev/types/style_extract_fonts_params.py +++ /dev/null @@ -1,24 +0,0 @@ -# File generated from our OpenAPI spec by Stainless. See CONTRIBUTING.md for details. - -from __future__ import annotations - -from typing_extensions import Required, Annotated, TypedDict - -from .._utils import PropertyInfo - -__all__ = ["StyleExtractFontsParams"] - - -class StyleExtractFontsParams(TypedDict, total=False): - domain: Required[str] - """Domain name to extract fonts from (e.g., 'example.com', 'google.com'). - - The domain will be automatically normalized and validated. - """ - - timeout_ms: Annotated[int, PropertyInfo(alias="timeoutMS")] - """Optional timeout in milliseconds for the request. - - If the request takes longer than this value, it will be aborted with a 408 - status code. Maximum allowed value is 300000ms (5 minutes). - """ diff --git a/src/context/dev/types/web_extract_fonts_params.py b/src/context/dev/types/web_extract_fonts_params.py new file mode 100644 index 0000000..9bdc46e --- /dev/null +++ b/src/context/dev/types/web_extract_fonts_params.py @@ -0,0 +1,32 @@ +# File generated from our OpenAPI spec by Stainless. See CONTRIBUTING.md for details. + +from __future__ import annotations + +from typing_extensions import Annotated, TypedDict + +from .._utils import PropertyInfo + +__all__ = ["WebExtractFontsParams"] + + +class WebExtractFontsParams(TypedDict, total=False): + direct_url: Annotated[str, PropertyInfo(alias="directUrl")] + """ + A specific URL to fetch fonts from directly, bypassing domain resolution (e.g., + 'https://example.com/design-system'). When provided, fonts are extracted from + this exact URL. You must provide either 'domain' or 'directUrl', but not both. + """ + + domain: str + """Domain name to extract fonts from (e.g., 'example.com', 'google.com'). + + The domain will be automatically normalized and validated. You must provide + either 'domain' or 'directUrl', but not both. + """ + + timeout_ms: Annotated[int, PropertyInfo(alias="timeoutMS")] + """Optional timeout in milliseconds for the request. + + If the request takes longer than this value, it will be aborted with a 408 + status code. Maximum allowed value is 300000ms (5 minutes). + """ diff --git a/src/context/dev/types/style_extract_fonts_response.py b/src/context/dev/types/web_extract_fonts_response.py similarity index 90% rename from src/context/dev/types/style_extract_fonts_response.py rename to src/context/dev/types/web_extract_fonts_response.py index ab00e10..55fec61 100644 --- a/src/context/dev/types/style_extract_fonts_response.py +++ b/src/context/dev/types/web_extract_fonts_response.py @@ -4,7 +4,7 @@ from .._models import BaseModel -__all__ = ["StyleExtractFontsResponse", "Font"] +__all__ = ["WebExtractFontsResponse", "Font"] class Font(BaseModel): @@ -30,7 +30,7 @@ class Font(BaseModel): """Array of CSS selectors or element types where this font is used""" -class StyleExtractFontsResponse(BaseModel): +class WebExtractFontsResponse(BaseModel): code: int """HTTP status code, e.g., 200""" diff --git a/src/context/dev/types/style_extract_styleguide_params.py b/src/context/dev/types/web_extract_styleguide_params.py similarity index 63% rename from src/context/dev/types/style_extract_styleguide_params.py rename to src/context/dev/types/web_extract_styleguide_params.py index c07da4f..42b6d88 100644 --- a/src/context/dev/types/style_extract_styleguide_params.py +++ b/src/context/dev/types/web_extract_styleguide_params.py @@ -6,20 +6,23 @@ from .._utils import PropertyInfo -__all__ = ["StyleExtractStyleguideParams"] +__all__ = ["WebExtractStyleguideParams"] -class StyleExtractStyleguideParams(TypedDict, total=False): +class WebExtractStyleguideParams(TypedDict, total=False): direct_url: Annotated[str, PropertyInfo(alias="directUrl")] """ A specific URL to fetch the styleguide from directly, bypassing domain - resolution (e.g., 'https://example.com/design-system'). + resolution (e.g., 'https://example.com/design-system'). When provided, the + styleguide is extracted from this exact URL. You must provide either 'domain' or + 'directUrl', but not both. """ domain: str """Domain name to extract styleguide from (e.g., 'example.com', 'google.com'). - The domain will be automatically normalized and validated. + The domain will be automatically normalized and validated. You must provide + either 'domain' or 'directUrl', but not both. """ timeout_ms: Annotated[int, PropertyInfo(alias="timeoutMS")] diff --git a/src/context/dev/types/style_extract_styleguide_response.py b/src/context/dev/types/web_extract_styleguide_response.py similarity index 90% rename from src/context/dev/types/style_extract_styleguide_response.py rename to src/context/dev/types/web_extract_styleguide_response.py index 3dfe70d..77e2b55 100644 --- a/src/context/dev/types/style_extract_styleguide_response.py +++ b/src/context/dev/types/web_extract_styleguide_response.py @@ -1,6 +1,6 @@ # File generated from our OpenAPI spec by Stainless. See CONTRIBUTING.md for details. -from typing import List, Optional +from typing import Dict, List, Optional from typing_extensions import Literal from pydantic import Field as FieldInfo @@ -8,7 +8,7 @@ from .._models import BaseModel __all__ = [ - "StyleExtractStyleguideResponse", + "WebExtractStyleguideResponse", "Styleguide", "StyleguideColors", "StyleguideComponents", @@ -18,6 +18,7 @@ "StyleguideComponentsButtonSecondary", "StyleguideComponentsCard", "StyleguideElementSpacing", + "StyleguideFontLinks", "StyleguideShadows", "StyleguideTypography", "StyleguideTypographyHeadings", @@ -244,6 +245,30 @@ class StyleguideElementSpacing(BaseModel): xs: str +class StyleguideFontLinks(BaseModel): + files: Dict[str, str] + """Upright font files keyed by weight string (e.g. + + "400" for regular, "500", "700"). Values are absolute URLs. + """ + + type: Literal["google", "custom"] + + category: Optional[str] = None + """Google Fonts category when type is google (e.g. + + sans-serif, serif, monospace, display, handwriting). Omitted for custom fonts + when unknown. + """ + + display_name: Optional[str] = FieldInfo(alias="displayName", default=None) + """ + Present when type is custom: human-readable name derived from the fontLinks key + (strip build/hash suffixes, split camelCase / PascalCase, normalize separators). + Google entries omit this. + """ + + class StyleguideShadows(BaseModel): """Shadow styles used on the website""" @@ -371,6 +396,13 @@ class Styleguide(BaseModel): element_spacing: StyleguideElementSpacing = FieldInfo(alias="elementSpacing") """Spacing system used on the website""" + font_links: Dict[str, StyleguideFontLinks] = FieldInfo(alias="fontLinks") + """ + Font assets keyed by family name as it appears in fontFamily/fontFallbacks + (non-generic names only). Clients match typography.fontFamily / fontWeight or + button styles to pick a file URL from files. + """ + mode: Literal["light", "dark"] """The primary color mode of the website design""" @@ -381,7 +413,7 @@ class Styleguide(BaseModel): """Typography styles used on the website""" -class StyleExtractStyleguideResponse(BaseModel): +class WebExtractStyleguideResponse(BaseModel): code: Optional[int] = None """HTTP status code""" diff --git a/src/context/dev/types/web_screenshot_params.py b/src/context/dev/types/web_screenshot_params.py index d278e90..c6de68a 100644 --- a/src/context/dev/types/web_screenshot_params.py +++ b/src/context/dev/types/web_screenshot_params.py @@ -14,13 +14,14 @@ class WebScreenshotParams(TypedDict, total=False): """ A specific URL to screenshot directly, bypassing domain resolution (e.g., 'https://example.com/pricing'). When provided, the screenshot is taken of this - exact URL. + exact URL. You must provide either 'domain' or 'directUrl', but not both. """ domain: str """Domain name to take screenshot of (e.g., 'example.com', 'google.com'). - The domain will be automatically normalized and validated. + The domain will be automatically normalized and validated. You must provide + either 'domain' or 'directUrl', but not both. """ full_screenshot: Annotated[Literal["true", "false"], PropertyInfo(alias="fullScreenshot")] diff --git a/src/context/dev/types/web_web_scrape_html_params.py b/src/context/dev/types/web_web_scrape_html_params.py index 1847d07..d184801 100644 --- a/src/context/dev/types/web_web_scrape_html_params.py +++ b/src/context/dev/types/web_web_scrape_html_params.py @@ -2,7 +2,9 @@ from __future__ import annotations -from typing_extensions import Required, TypedDict +from typing_extensions import Required, Annotated, TypedDict + +from .._utils import PropertyInfo __all__ = ["WebWebScrapeHTMLParams"] @@ -10,3 +12,10 @@ class WebWebScrapeHTMLParams(TypedDict, total=False): url: Required[str] """Full URL to scrape (must include http:// or https:// protocol)""" + + max_age_ms: Annotated[int, PropertyInfo(alias="maxAgeMs")] + """ + Return a cached result if a prior scrape for the same parameters exists and is + younger than this many milliseconds. Defaults to 1 day (86400000 ms) when + omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh. + """ diff --git a/src/context/dev/types/web_web_scrape_md_params.py b/src/context/dev/types/web_web_scrape_md_params.py index d556da1..8cc9333 100644 --- a/src/context/dev/types/web_web_scrape_md_params.py +++ b/src/context/dev/types/web_web_scrape_md_params.py @@ -12,7 +12,7 @@ class WebWebScrapeMdParams(TypedDict, total=False): url: Required[str] """ - Full URL to scrape and convert to markdown (must include http:// or https:// + Full URL to scrape into LLM usable Markdown (must include http:// or https:// protocol) """ @@ -22,6 +22,13 @@ class WebWebScrapeMdParams(TypedDict, total=False): include_links: Annotated[bool, PropertyInfo(alias="includeLinks")] """Preserve hyperlinks in Markdown output""" + max_age_ms: Annotated[int, PropertyInfo(alias="maxAgeMs")] + """ + Return a cached result if a prior scrape for the same parameters exists and is + younger than this many milliseconds. Defaults to 1 day (86400000 ms) when + omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh. + """ + shorten_base64_images: Annotated[bool, PropertyInfo(alias="shortenBase64Images")] """Shorten base64-encoded image data in the Markdown output""" diff --git a/src/context/dev/types/web_web_scrape_sitemap_params.py b/src/context/dev/types/web_web_scrape_sitemap_params.py index f78f498..4a1c766 100644 --- a/src/context/dev/types/web_web_scrape_sitemap_params.py +++ b/src/context/dev/types/web_web_scrape_sitemap_params.py @@ -11,10 +11,7 @@ class WebWebScrapeSitemapParams(TypedDict, total=False): domain: Required[str] - """Domain name to crawl sitemaps for (e.g., 'example.com'). - - The domain will be automatically normalized and validated. - """ + """Domain to build a sitemap for""" max_links: Annotated[int, PropertyInfo(alias="maxLinks")] """Maximum number of links to return from the sitemap crawl. diff --git a/tests/api_resources/test_style.py b/tests/api_resources/test_style.py deleted file mode 100644 index 8c9f16a..0000000 --- a/tests/api_resources/test_style.py +++ /dev/null @@ -1,189 +0,0 @@ -# File generated from our OpenAPI spec by Stainless. See CONTRIBUTING.md for details. - -from __future__ import annotations - -import os -from typing import Any, cast - -import pytest - -from context.dev import ContextDev, AsyncContextDev -from tests.utils import assert_matches_type -from context.dev.types import ( - StyleExtractFontsResponse, - StyleExtractStyleguideResponse, -) - -base_url = os.environ.get("TEST_API_BASE_URL", "http://127.0.0.1:4010") - - -class TestStyle: - parametrize = pytest.mark.parametrize("client", [False, True], indirect=True, ids=["loose", "strict"]) - - @pytest.mark.skip(reason="Mock server tests are disabled") - @parametrize - def test_method_extract_fonts(self, client: ContextDev) -> None: - style = client.style.extract_fonts( - domain="domain", - ) - assert_matches_type(StyleExtractFontsResponse, style, path=["response"]) - - @pytest.mark.skip(reason="Mock server tests are disabled") - @parametrize - def test_method_extract_fonts_with_all_params(self, client: ContextDev) -> None: - style = client.style.extract_fonts( - domain="domain", - timeout_ms=1000, - ) - assert_matches_type(StyleExtractFontsResponse, style, path=["response"]) - - @pytest.mark.skip(reason="Mock server tests are disabled") - @parametrize - def test_raw_response_extract_fonts(self, client: ContextDev) -> None: - response = client.style.with_raw_response.extract_fonts( - domain="domain", - ) - - assert response.is_closed is True - assert response.http_request.headers.get("X-Stainless-Lang") == "python" - style = response.parse() - assert_matches_type(StyleExtractFontsResponse, style, path=["response"]) - - @pytest.mark.skip(reason="Mock server tests are disabled") - @parametrize - def test_streaming_response_extract_fonts(self, client: ContextDev) -> None: - with client.style.with_streaming_response.extract_fonts( - domain="domain", - ) as response: - assert not response.is_closed - assert response.http_request.headers.get("X-Stainless-Lang") == "python" - - style = response.parse() - assert_matches_type(StyleExtractFontsResponse, style, path=["response"]) - - assert cast(Any, response.is_closed) is True - - @pytest.mark.skip(reason="Mock server tests are disabled") - @parametrize - def test_method_extract_styleguide(self, client: ContextDev) -> None: - style = client.style.extract_styleguide() - assert_matches_type(StyleExtractStyleguideResponse, style, path=["response"]) - - @pytest.mark.skip(reason="Mock server tests are disabled") - @parametrize - def test_method_extract_styleguide_with_all_params(self, client: ContextDev) -> None: - style = client.style.extract_styleguide( - direct_url="https://example.com", - domain="domain", - timeout_ms=1000, - ) - assert_matches_type(StyleExtractStyleguideResponse, style, path=["response"]) - - @pytest.mark.skip(reason="Mock server tests are disabled") - @parametrize - def test_raw_response_extract_styleguide(self, client: ContextDev) -> None: - response = client.style.with_raw_response.extract_styleguide() - - assert response.is_closed is True - assert response.http_request.headers.get("X-Stainless-Lang") == "python" - style = response.parse() - assert_matches_type(StyleExtractStyleguideResponse, style, path=["response"]) - - @pytest.mark.skip(reason="Mock server tests are disabled") - @parametrize - def test_streaming_response_extract_styleguide(self, client: ContextDev) -> None: - with client.style.with_streaming_response.extract_styleguide() as response: - assert not response.is_closed - assert response.http_request.headers.get("X-Stainless-Lang") == "python" - - style = response.parse() - assert_matches_type(StyleExtractStyleguideResponse, style, path=["response"]) - - assert cast(Any, response.is_closed) is True - - -class TestAsyncStyle: - parametrize = pytest.mark.parametrize( - "async_client", [False, True, {"http_client": "aiohttp"}], indirect=True, ids=["loose", "strict", "aiohttp"] - ) - - @pytest.mark.skip(reason="Mock server tests are disabled") - @parametrize - async def test_method_extract_fonts(self, async_client: AsyncContextDev) -> None: - style = await async_client.style.extract_fonts( - domain="domain", - ) - assert_matches_type(StyleExtractFontsResponse, style, path=["response"]) - - @pytest.mark.skip(reason="Mock server tests are disabled") - @parametrize - async def test_method_extract_fonts_with_all_params(self, async_client: AsyncContextDev) -> None: - style = await async_client.style.extract_fonts( - domain="domain", - timeout_ms=1000, - ) - assert_matches_type(StyleExtractFontsResponse, style, path=["response"]) - - @pytest.mark.skip(reason="Mock server tests are disabled") - @parametrize - async def test_raw_response_extract_fonts(self, async_client: AsyncContextDev) -> None: - response = await async_client.style.with_raw_response.extract_fonts( - domain="domain", - ) - - assert response.is_closed is True - assert response.http_request.headers.get("X-Stainless-Lang") == "python" - style = await response.parse() - assert_matches_type(StyleExtractFontsResponse, style, path=["response"]) - - @pytest.mark.skip(reason="Mock server tests are disabled") - @parametrize - async def test_streaming_response_extract_fonts(self, async_client: AsyncContextDev) -> None: - async with async_client.style.with_streaming_response.extract_fonts( - domain="domain", - ) as response: - assert not response.is_closed - assert response.http_request.headers.get("X-Stainless-Lang") == "python" - - style = await response.parse() - assert_matches_type(StyleExtractFontsResponse, style, path=["response"]) - - assert cast(Any, response.is_closed) is True - - @pytest.mark.skip(reason="Mock server tests are disabled") - @parametrize - async def test_method_extract_styleguide(self, async_client: AsyncContextDev) -> None: - style = await async_client.style.extract_styleguide() - assert_matches_type(StyleExtractStyleguideResponse, style, path=["response"]) - - @pytest.mark.skip(reason="Mock server tests are disabled") - @parametrize - async def test_method_extract_styleguide_with_all_params(self, async_client: AsyncContextDev) -> None: - style = await async_client.style.extract_styleguide( - direct_url="https://example.com", - domain="domain", - timeout_ms=1000, - ) - assert_matches_type(StyleExtractStyleguideResponse, style, path=["response"]) - - @pytest.mark.skip(reason="Mock server tests are disabled") - @parametrize - async def test_raw_response_extract_styleguide(self, async_client: AsyncContextDev) -> None: - response = await async_client.style.with_raw_response.extract_styleguide() - - assert response.is_closed is True - assert response.http_request.headers.get("X-Stainless-Lang") == "python" - style = await response.parse() - assert_matches_type(StyleExtractStyleguideResponse, style, path=["response"]) - - @pytest.mark.skip(reason="Mock server tests are disabled") - @parametrize - async def test_streaming_response_extract_styleguide(self, async_client: AsyncContextDev) -> None: - async with async_client.style.with_streaming_response.extract_styleguide() as response: - assert not response.is_closed - assert response.http_request.headers.get("X-Stainless-Lang") == "python" - - style = await response.parse() - assert_matches_type(StyleExtractStyleguideResponse, style, path=["response"]) - - assert cast(Any, response.is_closed) is True diff --git a/tests/api_resources/test_web.py b/tests/api_resources/test_web.py index 2d8a511..6917533 100644 --- a/tests/api_resources/test_web.py +++ b/tests/api_resources/test_web.py @@ -13,9 +13,11 @@ WebScreenshotResponse, WebWebCrawlMdResponse, WebWebScrapeMdResponse, + WebExtractFontsResponse, WebWebScrapeHTMLResponse, WebWebScrapeImagesResponse, WebWebScrapeSitemapResponse, + WebExtractStyleguideResponse, ) base_url = os.environ.get("TEST_API_BASE_URL", "http://127.0.0.1:4010") @@ -24,6 +26,82 @@ class TestWeb: parametrize = pytest.mark.parametrize("client", [False, True], indirect=True, ids=["loose", "strict"]) + @pytest.mark.skip(reason="Mock server tests are disabled") + @parametrize + def test_method_extract_fonts(self, client: ContextDev) -> None: + web = client.web.extract_fonts() + assert_matches_type(WebExtractFontsResponse, web, path=["response"]) + + @pytest.mark.skip(reason="Mock server tests are disabled") + @parametrize + def test_method_extract_fonts_with_all_params(self, client: ContextDev) -> None: + web = client.web.extract_fonts( + direct_url="https://example.com", + domain="domain", + timeout_ms=1000, + ) + assert_matches_type(WebExtractFontsResponse, web, path=["response"]) + + @pytest.mark.skip(reason="Mock server tests are disabled") + @parametrize + def test_raw_response_extract_fonts(self, client: ContextDev) -> None: + response = client.web.with_raw_response.extract_fonts() + + assert response.is_closed is True + assert response.http_request.headers.get("X-Stainless-Lang") == "python" + web = response.parse() + assert_matches_type(WebExtractFontsResponse, web, path=["response"]) + + @pytest.mark.skip(reason="Mock server tests are disabled") + @parametrize + def test_streaming_response_extract_fonts(self, client: ContextDev) -> None: + with client.web.with_streaming_response.extract_fonts() as response: + assert not response.is_closed + assert response.http_request.headers.get("X-Stainless-Lang") == "python" + + web = response.parse() + assert_matches_type(WebExtractFontsResponse, web, path=["response"]) + + assert cast(Any, response.is_closed) is True + + @pytest.mark.skip(reason="Mock server tests are disabled") + @parametrize + def test_method_extract_styleguide(self, client: ContextDev) -> None: + web = client.web.extract_styleguide() + assert_matches_type(WebExtractStyleguideResponse, web, path=["response"]) + + @pytest.mark.skip(reason="Mock server tests are disabled") + @parametrize + def test_method_extract_styleguide_with_all_params(self, client: ContextDev) -> None: + web = client.web.extract_styleguide( + direct_url="https://example.com", + domain="domain", + timeout_ms=1000, + ) + assert_matches_type(WebExtractStyleguideResponse, web, path=["response"]) + + @pytest.mark.skip(reason="Mock server tests are disabled") + @parametrize + def test_raw_response_extract_styleguide(self, client: ContextDev) -> None: + response = client.web.with_raw_response.extract_styleguide() + + assert response.is_closed is True + assert response.http_request.headers.get("X-Stainless-Lang") == "python" + web = response.parse() + assert_matches_type(WebExtractStyleguideResponse, web, path=["response"]) + + @pytest.mark.skip(reason="Mock server tests are disabled") + @parametrize + def test_streaming_response_extract_styleguide(self, client: ContextDev) -> None: + with client.web.with_streaming_response.extract_styleguide() as response: + assert not response.is_closed + assert response.http_request.headers.get("X-Stainless-Lang") == "python" + + web = response.parse() + assert_matches_type(WebExtractStyleguideResponse, web, path=["response"]) + + assert cast(Any, response.is_closed) is True + @pytest.mark.skip(reason="Mock server tests are disabled") @parametrize def test_method_screenshot(self, client: ContextDev) -> None: @@ -122,6 +200,15 @@ def test_method_web_scrape_html(self, client: ContextDev) -> None: ) assert_matches_type(WebWebScrapeHTMLResponse, web, path=["response"]) + @pytest.mark.skip(reason="Mock server tests are disabled") + @parametrize + def test_method_web_scrape_html_with_all_params(self, client: ContextDev) -> None: + web = client.web.web_scrape_html( + url="https://example.com", + max_age_ms=0, + ) + assert_matches_type(WebWebScrapeHTMLResponse, web, path=["response"]) + @pytest.mark.skip(reason="Mock server tests are disabled") @parametrize def test_raw_response_web_scrape_html(self, client: ContextDev) -> None: @@ -197,6 +284,7 @@ def test_method_web_scrape_md_with_all_params(self, client: ContextDev) -> None: url="https://example.com", include_images=True, include_links=True, + max_age_ms=0, shorten_base64_images=True, use_main_content_only=True, ) @@ -277,6 +365,82 @@ class TestAsyncWeb: "async_client", [False, True, {"http_client": "aiohttp"}], indirect=True, ids=["loose", "strict", "aiohttp"] ) + @pytest.mark.skip(reason="Mock server tests are disabled") + @parametrize + async def test_method_extract_fonts(self, async_client: AsyncContextDev) -> None: + web = await async_client.web.extract_fonts() + assert_matches_type(WebExtractFontsResponse, web, path=["response"]) + + @pytest.mark.skip(reason="Mock server tests are disabled") + @parametrize + async def test_method_extract_fonts_with_all_params(self, async_client: AsyncContextDev) -> None: + web = await async_client.web.extract_fonts( + direct_url="https://example.com", + domain="domain", + timeout_ms=1000, + ) + assert_matches_type(WebExtractFontsResponse, web, path=["response"]) + + @pytest.mark.skip(reason="Mock server tests are disabled") + @parametrize + async def test_raw_response_extract_fonts(self, async_client: AsyncContextDev) -> None: + response = await async_client.web.with_raw_response.extract_fonts() + + assert response.is_closed is True + assert response.http_request.headers.get("X-Stainless-Lang") == "python" + web = await response.parse() + assert_matches_type(WebExtractFontsResponse, web, path=["response"]) + + @pytest.mark.skip(reason="Mock server tests are disabled") + @parametrize + async def test_streaming_response_extract_fonts(self, async_client: AsyncContextDev) -> None: + async with async_client.web.with_streaming_response.extract_fonts() as response: + assert not response.is_closed + assert response.http_request.headers.get("X-Stainless-Lang") == "python" + + web = await response.parse() + assert_matches_type(WebExtractFontsResponse, web, path=["response"]) + + assert cast(Any, response.is_closed) is True + + @pytest.mark.skip(reason="Mock server tests are disabled") + @parametrize + async def test_method_extract_styleguide(self, async_client: AsyncContextDev) -> None: + web = await async_client.web.extract_styleguide() + assert_matches_type(WebExtractStyleguideResponse, web, path=["response"]) + + @pytest.mark.skip(reason="Mock server tests are disabled") + @parametrize + async def test_method_extract_styleguide_with_all_params(self, async_client: AsyncContextDev) -> None: + web = await async_client.web.extract_styleguide( + direct_url="https://example.com", + domain="domain", + timeout_ms=1000, + ) + assert_matches_type(WebExtractStyleguideResponse, web, path=["response"]) + + @pytest.mark.skip(reason="Mock server tests are disabled") + @parametrize + async def test_raw_response_extract_styleguide(self, async_client: AsyncContextDev) -> None: + response = await async_client.web.with_raw_response.extract_styleguide() + + assert response.is_closed is True + assert response.http_request.headers.get("X-Stainless-Lang") == "python" + web = await response.parse() + assert_matches_type(WebExtractStyleguideResponse, web, path=["response"]) + + @pytest.mark.skip(reason="Mock server tests are disabled") + @parametrize + async def test_streaming_response_extract_styleguide(self, async_client: AsyncContextDev) -> None: + async with async_client.web.with_streaming_response.extract_styleguide() as response: + assert not response.is_closed + assert response.http_request.headers.get("X-Stainless-Lang") == "python" + + web = await response.parse() + assert_matches_type(WebExtractStyleguideResponse, web, path=["response"]) + + assert cast(Any, response.is_closed) is True + @pytest.mark.skip(reason="Mock server tests are disabled") @parametrize async def test_method_screenshot(self, async_client: AsyncContextDev) -> None: @@ -375,6 +539,15 @@ async def test_method_web_scrape_html(self, async_client: AsyncContextDev) -> No ) assert_matches_type(WebWebScrapeHTMLResponse, web, path=["response"]) + @pytest.mark.skip(reason="Mock server tests are disabled") + @parametrize + async def test_method_web_scrape_html_with_all_params(self, async_client: AsyncContextDev) -> None: + web = await async_client.web.web_scrape_html( + url="https://example.com", + max_age_ms=0, + ) + assert_matches_type(WebWebScrapeHTMLResponse, web, path=["response"]) + @pytest.mark.skip(reason="Mock server tests are disabled") @parametrize async def test_raw_response_web_scrape_html(self, async_client: AsyncContextDev) -> None: @@ -450,6 +623,7 @@ async def test_method_web_scrape_md_with_all_params(self, async_client: AsyncCon url="https://example.com", include_images=True, include_links=True, + max_age_ms=0, shorten_base64_images=True, use_main_content_only=True, ) diff --git a/tests/test_deepcopy.py b/tests/test_deepcopy.py deleted file mode 100644 index ec2cbfc..0000000 --- a/tests/test_deepcopy.py +++ /dev/null @@ -1,58 +0,0 @@ -from context.dev._utils import deepcopy_minimal - - -def assert_different_identities(obj1: object, obj2: object) -> None: - assert obj1 == obj2 - assert id(obj1) != id(obj2) - - -def test_simple_dict() -> None: - obj1 = {"foo": "bar"} - obj2 = deepcopy_minimal(obj1) - assert_different_identities(obj1, obj2) - - -def test_nested_dict() -> None: - obj1 = {"foo": {"bar": True}} - obj2 = deepcopy_minimal(obj1) - assert_different_identities(obj1, obj2) - assert_different_identities(obj1["foo"], obj2["foo"]) - - -def test_complex_nested_dict() -> None: - obj1 = {"foo": {"bar": [{"hello": "world"}]}} - obj2 = deepcopy_minimal(obj1) - assert_different_identities(obj1, obj2) - assert_different_identities(obj1["foo"], obj2["foo"]) - assert_different_identities(obj1["foo"]["bar"], obj2["foo"]["bar"]) - assert_different_identities(obj1["foo"]["bar"][0], obj2["foo"]["bar"][0]) - - -def test_simple_list() -> None: - obj1 = ["a", "b", "c"] - obj2 = deepcopy_minimal(obj1) - assert_different_identities(obj1, obj2) - - -def test_nested_list() -> None: - obj1 = ["a", [1, 2, 3]] - obj2 = deepcopy_minimal(obj1) - assert_different_identities(obj1, obj2) - assert_different_identities(obj1[1], obj2[1]) - - -class MyObject: ... - - -def test_ignores_other_types() -> None: - # custom classes - my_obj = MyObject() - obj1 = {"foo": my_obj} - obj2 = deepcopy_minimal(obj1) - assert_different_identities(obj1, obj2) - assert obj1["foo"] is my_obj - - # tuples - obj3 = ("a", "b") - obj4 = deepcopy_minimal(obj3) - assert obj3 is obj4 diff --git a/tests/test_extract_files.py b/tests/test_extract_files.py index e7585ee..34b6253 100644 --- a/tests/test_extract_files.py +++ b/tests/test_extract_files.py @@ -35,6 +35,15 @@ def test_multiple_files() -> None: assert query == {"documents": [{}, {}]} +def test_top_level_file_array() -> None: + query = {"files": [b"file one", b"file two"], "title": "hello"} + assert extract_files(query, paths=[["files", ""]]) == [ + ("files[]", b"file one"), + ("files[]", b"file two"), + ] + assert query == {"title": "hello"} + + @pytest.mark.parametrize( "query,paths,expected", [ diff --git a/tests/test_files.py b/tests/test_files.py index ede2488..e9e3a2d 100644 --- a/tests/test_files.py +++ b/tests/test_files.py @@ -4,7 +4,8 @@ import pytest from dirty_equals import IsDict, IsList, IsBytes, IsTuple -from context.dev._files import to_httpx_files, async_to_httpx_files +from context.dev._files import to_httpx_files, deepcopy_with_paths, async_to_httpx_files +from context.dev._utils import extract_files readme_path = Path(__file__).parent.parent.joinpath("README.md") @@ -49,3 +50,99 @@ def test_string_not_allowed() -> None: "file": "foo", # type: ignore } ) + + +def assert_different_identities(obj1: object, obj2: object) -> None: + assert obj1 == obj2 + assert obj1 is not obj2 + + +class TestDeepcopyWithPaths: + def test_copies_top_level_dict(self) -> None: + original = {"file": b"data", "other": "value"} + result = deepcopy_with_paths(original, [["file"]]) + assert_different_identities(result, original) + + def test_file_value_is_same_reference(self) -> None: + file_bytes = b"contents" + original = {"file": file_bytes} + result = deepcopy_with_paths(original, [["file"]]) + assert_different_identities(result, original) + assert result["file"] is file_bytes + + def test_list_popped_wholesale(self) -> None: + files = [b"f1", b"f2"] + original = {"files": files, "title": "t"} + result = deepcopy_with_paths(original, [["files", ""]]) + assert_different_identities(result, original) + result_files = result["files"] + assert isinstance(result_files, list) + assert_different_identities(result_files, files) + + def test_nested_array_path_copies_list_and_elements(self) -> None: + elem1 = {"file": b"f1", "extra": 1} + elem2 = {"file": b"f2", "extra": 2} + original = {"items": [elem1, elem2]} + result = deepcopy_with_paths(original, [["items", "", "file"]]) + assert_different_identities(result, original) + result_items = result["items"] + assert isinstance(result_items, list) + assert_different_identities(result_items, original["items"]) + assert_different_identities(result_items[0], elem1) + assert_different_identities(result_items[1], elem2) + + def test_empty_paths_returns_same_object(self) -> None: + original = {"foo": "bar"} + result = deepcopy_with_paths(original, []) + assert result is original + + def test_multiple_paths(self) -> None: + f1 = b"file1" + f2 = b"file2" + original = {"a": f1, "b": f2, "c": "unchanged"} + result = deepcopy_with_paths(original, [["a"], ["b"]]) + assert_different_identities(result, original) + assert result["a"] is f1 + assert result["b"] is f2 + assert result["c"] is original["c"] + + def test_extract_files_does_not_mutate_original_top_level(self) -> None: + file_bytes = b"contents" + original = {"file": file_bytes, "other": "value"} + + copied = deepcopy_with_paths(original, [["file"]]) + extracted = extract_files(copied, paths=[["file"]]) + + assert extracted == [("file", file_bytes)] + assert original == {"file": file_bytes, "other": "value"} + assert copied == {"other": "value"} + + def test_extract_files_does_not_mutate_original_nested_array_path(self) -> None: + file1 = b"f1" + file2 = b"f2" + original = { + "items": [ + {"file": file1, "extra": 1}, + {"file": file2, "extra": 2}, + ], + "title": "example", + } + + copied = deepcopy_with_paths(original, [["items", "", "file"]]) + extracted = extract_files(copied, paths=[["items", "", "file"]]) + + assert extracted == [("items[][file]", file1), ("items[][file]", file2)] + assert original == { + "items": [ + {"file": file1, "extra": 1}, + {"file": file2, "extra": 2}, + ], + "title": "example", + } + assert copied == { + "items": [ + {"extra": 1}, + {"extra": 2}, + ], + "title": "example", + }