diff --git a/.release-please-manifest.json b/.release-please-manifest.json
index 1b77f50..6538ca9 100644
--- a/.release-please-manifest.json
+++ b/.release-please-manifest.json
@@ -1,3 +1,3 @@
{
- ".": "0.7.0"
+ ".": "0.8.0"
}
\ No newline at end of file
diff --git a/.stats.yml b/.stats.yml
index 998ed68..6d316e1 100644
--- a/.stats.yml
+++ b/.stats.yml
@@ -1,4 +1,4 @@
configured_endpoints: 21
-openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev%2Fcontext.dev-5ee7a321f8e01d2678d56277f011f71294c2b8ac4d0cff0bda412362c0d9cf73.yml
-openapi_spec_hash: bf5d57e3bddb6975770cf85c267b5035
-config_hash: 682b89b02a20f5d1c13e2c91ecbcf5ce
+openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev%2Fcontext.dev-5afee68d7308a77a9ef51fb53917080746f96327c6d745fc029363a7d1494c3c.yml
+openapi_spec_hash: b2e32bb58d92a00f6ede04c4f7bf1a34
+config_hash: 7d13dca2b2c6f71fc463cb6062efa5ea
diff --git a/CHANGELOG.md b/CHANGELOG.md
index 6c00b74..cc6b274 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -1,5 +1,36 @@
# Changelog
+## 0.8.0 (2026-04-19)
+
+Full Changelog: [v0.7.0...v0.8.0](https://github.com/context-dot-dev/context-python-sdk/compare/v0.7.0...v0.8.0)
+
+### Features
+
+* **api:** api update ([512bf26](https://github.com/context-dot-dev/context-python-sdk/commit/512bf2610e59733e0e950531ca9e6018a5446672))
+* **api:** api update ([f98376c](https://github.com/context-dot-dev/context-python-sdk/commit/f98376cf22a331138f716f8abeb4458a8e1e38c2))
+* **api:** api update ([9ddc161](https://github.com/context-dot-dev/context-python-sdk/commit/9ddc161215cd12384152295997b95d5ac44bec08))
+* **api:** api update ([639075e](https://github.com/context-dot-dev/context-python-sdk/commit/639075e0a0aa6203375592b8b67438ae0719ca26))
+* **api:** api update ([707a6c3](https://github.com/context-dot-dev/context-python-sdk/commit/707a6c3e49096e2248ec9e4ae7e683138647c2d6))
+* **api:** api update ([bbc2458](https://github.com/context-dot-dev/context-python-sdk/commit/bbc2458c38933208b22e234d6bf1f1ffee7f883e))
+* **api:** api update ([241bacf](https://github.com/context-dot-dev/context-python-sdk/commit/241bacf196c1e316aef39429e3919fda1c5a78eb))
+* **api:** api update ([648e71b](https://github.com/context-dot-dev/context-python-sdk/commit/648e71bf7afb870dceafe6485bf47932c74ebe1b))
+* **api:** api update ([656e585](https://github.com/context-dot-dev/context-python-sdk/commit/656e585bcdf228d8651cdab5ee37f1119a51eafa))
+* **api:** api update ([3fc9944](https://github.com/context-dot-dev/context-python-sdk/commit/3fc9944ff014539c18e0216b93c936c1df2b0219))
+* **api:** api update ([b9374f5](https://github.com/context-dot-dev/context-python-sdk/commit/b9374f57a2a0ed207c39e80b6fbf710bbe170e11))
+* **api:** manual updates ([2af78a4](https://github.com/context-dot-dev/context-python-sdk/commit/2af78a43c1e9b5654668e82c009b679dc3ae8e73))
+* **api:** manual updates ([d6acbc6](https://github.com/context-dot-dev/context-python-sdk/commit/d6acbc6552d84e58a6ff7a9e99952a2a7a624647))
+* **api:** manual updates ([ec8ff2d](https://github.com/context-dot-dev/context-python-sdk/commit/ec8ff2d2a0d84fcb0c05791fa913441c60ed9c24))
+
+
+### Bug Fixes
+
+* ensure file data are only sent as 1 parameter ([7c314a2](https://github.com/context-dot-dev/context-python-sdk/commit/7c314a29f73103eb592686df101c86c6cfd26b3f))
+
+
+### Performance Improvements
+
+* **client:** optimize file structure copying in multipart requests ([573a70c](https://github.com/context-dot-dev/context-python-sdk/commit/573a70cee8cb9bed8377f816bfba419ecbe86109))
+
## 0.7.0 (2026-04-09)
Full Changelog: [v0.6.0...v0.7.0](https://github.com/context-dot-dev/context-python-sdk/compare/v0.6.0...v0.7.0)
diff --git a/api.md b/api.md
index 4c7de31..963807e 100644
--- a/api.md
+++ b/api.md
@@ -4,6 +4,8 @@ Types:
```python
from context.dev.types import (
+ WebExtractFontsResponse,
+ WebExtractStyleguideResponse,
WebScreenshotResponse,
WebWebCrawlMdResponse,
WebWebScrapeHTMLResponse,
@@ -15,7 +17,9 @@ from context.dev.types import (
Methods:
-- client.web.screenshot(\*\*params) -> WebScreenshotResponse
+- client.web.extract_fonts(\*\*params) -> WebExtractFontsResponse
+- client.web.extract_styleguide(\*\*params) -> WebExtractStyleguideResponse
+- client.web.screenshot(\*\*params) -> WebScreenshotResponse
- client.web.web_crawl_md(\*\*params) -> WebWebCrawlMdResponse
- client.web.web_scrape_html(\*\*params) -> WebWebScrapeHTMLResponse
- client.web.web_scrape_images(\*\*params) -> WebWebScrapeImagesResponse
@@ -36,19 +40,6 @@ Methods:
- client.ai.extract_product(\*\*params) -> AIExtractProductResponse
- client.ai.extract_products(\*\*params) -> AIExtractProductsResponse
-# Style
-
-Types:
-
-```python
-from context.dev.types import StyleExtractFontsResponse, StyleExtractStyleguideResponse
-```
-
-Methods:
-
-- client.style.extract_fonts(\*\*params) -> StyleExtractFontsResponse
-- client.style.extract_styleguide(\*\*params) -> StyleExtractStyleguideResponse
-
# Brand
Types:
@@ -85,7 +76,7 @@ from context.dev.types import IndustryRetrieveNaicsResponse
Methods:
-- client.industry.retrieve_naics(\*\*params) -> IndustryRetrieveNaicsResponse
+- client.industry.retrieve_naics(\*\*params) -> IndustryRetrieveNaicsResponse
# Utility
diff --git a/pyproject.toml b/pyproject.toml
index c5f1939..8a95885 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -1,6 +1,6 @@
[project]
name = "context.dev"
-version = "0.7.0"
+version = "0.8.0"
description = "The official Python library for the context.dev API"
dynamic = ["readme"]
license = "Apache-2.0"
diff --git a/src/context/dev/_client.py b/src/context/dev/_client.py
index 06cee7f..ace042b 100644
--- a/src/context/dev/_client.py
+++ b/src/context/dev/_client.py
@@ -31,11 +31,10 @@
)
if TYPE_CHECKING:
- from .resources import ai, web, brand, style, utility, industry
+ from .resources import ai, web, brand, utility, industry
from .resources.ai import AIResource, AsyncAIResource
from .resources.web import WebResource, AsyncWebResource
from .resources.brand import BrandResource, AsyncBrandResource
- from .resources.style import StyleResource, AsyncStyleResource
from .resources.utility import UtilityResource, AsyncUtilityResource
from .resources.industry import IndustryResource, AsyncIndustryResource
@@ -118,12 +117,6 @@ def ai(self) -> AIResource:
return AIResource(self)
- @cached_property
- def style(self) -> StyleResource:
- from .resources.style import StyleResource
-
- return StyleResource(self)
-
@cached_property
def brand(self) -> BrandResource:
from .resources.brand import BrandResource
@@ -322,12 +315,6 @@ def ai(self) -> AsyncAIResource:
return AsyncAIResource(self)
- @cached_property
- def style(self) -> AsyncStyleResource:
- from .resources.style import AsyncStyleResource
-
- return AsyncStyleResource(self)
-
@cached_property
def brand(self) -> AsyncBrandResource:
from .resources.brand import AsyncBrandResource
@@ -477,12 +464,6 @@ def ai(self) -> ai.AIResourceWithRawResponse:
return AIResourceWithRawResponse(self._client.ai)
- @cached_property
- def style(self) -> style.StyleResourceWithRawResponse:
- from .resources.style import StyleResourceWithRawResponse
-
- return StyleResourceWithRawResponse(self._client.style)
-
@cached_property
def brand(self) -> brand.BrandResourceWithRawResponse:
from .resources.brand import BrandResourceWithRawResponse
@@ -520,12 +501,6 @@ def ai(self) -> ai.AsyncAIResourceWithRawResponse:
return AsyncAIResourceWithRawResponse(self._client.ai)
- @cached_property
- def style(self) -> style.AsyncStyleResourceWithRawResponse:
- from .resources.style import AsyncStyleResourceWithRawResponse
-
- return AsyncStyleResourceWithRawResponse(self._client.style)
-
@cached_property
def brand(self) -> brand.AsyncBrandResourceWithRawResponse:
from .resources.brand import AsyncBrandResourceWithRawResponse
@@ -563,12 +538,6 @@ def ai(self) -> ai.AIResourceWithStreamingResponse:
return AIResourceWithStreamingResponse(self._client.ai)
- @cached_property
- def style(self) -> style.StyleResourceWithStreamingResponse:
- from .resources.style import StyleResourceWithStreamingResponse
-
- return StyleResourceWithStreamingResponse(self._client.style)
-
@cached_property
def brand(self) -> brand.BrandResourceWithStreamingResponse:
from .resources.brand import BrandResourceWithStreamingResponse
@@ -606,12 +575,6 @@ def ai(self) -> ai.AsyncAIResourceWithStreamingResponse:
return AsyncAIResourceWithStreamingResponse(self._client.ai)
- @cached_property
- def style(self) -> style.AsyncStyleResourceWithStreamingResponse:
- from .resources.style import AsyncStyleResourceWithStreamingResponse
-
- return AsyncStyleResourceWithStreamingResponse(self._client.style)
-
@cached_property
def brand(self) -> brand.AsyncBrandResourceWithStreamingResponse:
from .resources.brand import AsyncBrandResourceWithStreamingResponse
diff --git a/src/context/dev/_files.py b/src/context/dev/_files.py
index cc14c14..0fdce17 100644
--- a/src/context/dev/_files.py
+++ b/src/context/dev/_files.py
@@ -3,8 +3,8 @@
import io
import os
import pathlib
-from typing import overload
-from typing_extensions import TypeGuard
+from typing import Sequence, cast, overload
+from typing_extensions import TypeVar, TypeGuard
import anyio
@@ -17,7 +17,9 @@
HttpxFileContent,
HttpxRequestFiles,
)
-from ._utils import is_tuple_t, is_mapping_t, is_sequence_t
+from ._utils import is_list, is_mapping, is_tuple_t, is_mapping_t, is_sequence_t
+
+_T = TypeVar("_T")
def is_base64_file_input(obj: object) -> TypeGuard[Base64FileInput]:
@@ -121,3 +123,51 @@ async def async_read_file_content(file: FileContent) -> HttpxFileContent:
return await anyio.Path(file).read_bytes()
return file
+
+
+def deepcopy_with_paths(item: _T, paths: Sequence[Sequence[str]]) -> _T:
+ """Copy only the containers along the given paths.
+
+ Used to guard against mutation by extract_files without copying the entire structure.
+ Only dicts and lists that lie on a path are copied; everything else
+ is returned by reference.
+
+ For example, given paths=[["foo", "files", "file"]] and the structure:
+ {
+ "foo": {
+ "bar": {"baz": {}},
+ "files": {"file": }
+ }
+ }
+ The root dict, "foo", and "files" are copied (they lie on the path).
+ "bar" and "baz" are returned by reference (off the path).
+ """
+ return _deepcopy_with_paths(item, paths, 0)
+
+
+def _deepcopy_with_paths(item: _T, paths: Sequence[Sequence[str]], index: int) -> _T:
+ if not paths:
+ return item
+ if is_mapping(item):
+ key_to_paths: dict[str, list[Sequence[str]]] = {}
+ for path in paths:
+ if index < len(path):
+ key_to_paths.setdefault(path[index], []).append(path)
+
+ # if no path continues through this mapping, it won't be mutated and copying it is redundant
+ if not key_to_paths:
+ return item
+
+ result = dict(item)
+ for key, subpaths in key_to_paths.items():
+ if key in result:
+ result[key] = _deepcopy_with_paths(result[key], subpaths, index + 1)
+ return cast(_T, result)
+ if is_list(item):
+ array_paths = [path for path in paths if index < len(path) and path[index] == ""]
+
+ # if no path expects a list here, nothing will be mutated inside it - return by reference
+ if not array_paths:
+ return cast(_T, item)
+ return cast(_T, [_deepcopy_with_paths(entry, array_paths, index + 1) for entry in item])
+ return item
diff --git a/src/context/dev/_utils/__init__.py b/src/context/dev/_utils/__init__.py
index 10cb66d..1c090e5 100644
--- a/src/context/dev/_utils/__init__.py
+++ b/src/context/dev/_utils/__init__.py
@@ -24,7 +24,6 @@
coerce_integer as coerce_integer,
file_from_path as file_from_path,
strip_not_given as strip_not_given,
- deepcopy_minimal as deepcopy_minimal,
get_async_library as get_async_library,
maybe_coerce_float as maybe_coerce_float,
get_required_header as get_required_header,
diff --git a/src/context/dev/_utils/_utils.py b/src/context/dev/_utils/_utils.py
index eec7f4a..771859f 100644
--- a/src/context/dev/_utils/_utils.py
+++ b/src/context/dev/_utils/_utils.py
@@ -86,8 +86,9 @@ def _extract_items(
index += 1
if is_dict(obj):
try:
- # We are at the last entry in the path so we must remove the field
- if (len(path)) == index:
+ # Remove the field if there are no more dict keys in the path,
+ # only "" traversal markers or end.
+ if all(p == "" for p in path[index:]):
item = obj.pop(key)
else:
item = obj[key]
@@ -176,21 +177,6 @@ def is_iterable(obj: object) -> TypeGuard[Iterable[object]]:
return isinstance(obj, Iterable)
-def deepcopy_minimal(item: _T) -> _T:
- """Minimal reimplementation of copy.deepcopy() that will only copy certain object types:
-
- - mappings, e.g. `dict`
- - list
-
- This is done for performance reasons.
- """
- if is_mapping(item):
- return cast(_T, {k: deepcopy_minimal(v) for k, v in item.items()})
- if is_list(item):
- return cast(_T, [deepcopy_minimal(entry) for entry in item])
- return item
-
-
# copied from https://github.com/Rapptz/RoboDanny
def human_join(seq: Sequence[str], *, delim: str = ", ", final: str = "or") -> str:
size = len(seq)
diff --git a/src/context/dev/_version.py b/src/context/dev/_version.py
index 7b88fdc..e505869 100644
--- a/src/context/dev/_version.py
+++ b/src/context/dev/_version.py
@@ -1,4 +1,4 @@
# File generated from our OpenAPI spec by Stainless. See CONTRIBUTING.md for details.
__title__ = "context.dev"
-__version__ = "0.7.0" # x-release-please-version
+__version__ = "0.8.0" # x-release-please-version
diff --git a/src/context/dev/resources/__init__.py b/src/context/dev/resources/__init__.py
index 2bcb6db..edc5356 100644
--- a/src/context/dev/resources/__init__.py
+++ b/src/context/dev/resources/__init__.py
@@ -24,14 +24,6 @@
BrandResourceWithStreamingResponse,
AsyncBrandResourceWithStreamingResponse,
)
-from .style import (
- StyleResource,
- AsyncStyleResource,
- StyleResourceWithRawResponse,
- AsyncStyleResourceWithRawResponse,
- StyleResourceWithStreamingResponse,
- AsyncStyleResourceWithStreamingResponse,
-)
from .utility import (
UtilityResource,
AsyncUtilityResource,
@@ -62,12 +54,6 @@
"AsyncAIResourceWithRawResponse",
"AIResourceWithStreamingResponse",
"AsyncAIResourceWithStreamingResponse",
- "StyleResource",
- "AsyncStyleResource",
- "StyleResourceWithRawResponse",
- "AsyncStyleResourceWithRawResponse",
- "StyleResourceWithStreamingResponse",
- "AsyncStyleResourceWithStreamingResponse",
"BrandResource",
"AsyncBrandResource",
"BrandResourceWithRawResponse",
diff --git a/src/context/dev/resources/brand.py b/src/context/dev/resources/brand.py
index 24c4252..3373a27 100644
--- a/src/context/dev/resources/brand.py
+++ b/src/context/dev/resources/brand.py
@@ -637,7 +637,6 @@ def identify_from_transaction(
high_confidence_only: When set to true, the API will perform an additional verification steps to
ensure the identified brand matches the transaction with high confidence.
- Defaults to false.
max_speed: Optional parameter to optimize the API call for maximum speed. When set to true,
the API will skip time-consuming operations for faster response at the cost of
@@ -823,9 +822,8 @@ def retrieve_by_email(
) -> BrandRetrieveByEmailResponse:
"""
Retrieve brand information using an email address while detecting disposable and
- free email addresses. This endpoint extracts the domain from the email address
- and returns brand data for that domain. Disposable and free email addresses
- (like gmail.com, yahoo.com) will throw a 422 error.
+ free email addresses. Disposable and free email addresses (like gmail.com,
+ yahoo.com) will throw a 422 error.
Args:
email: Email address to retrieve brand data for (e.g., 'contact@example.com'). The
@@ -1008,8 +1006,7 @@ def retrieve_by_isin(
) -> BrandRetrieveByIsinResponse:
"""
Retrieve brand information using an ISIN (International Securities
- Identification Number). This endpoint looks up the company associated with the
- ISIN and returns its brand data.
+ Identification Number).
Args:
isin: ISIN (International Securities Identification Number) to retrieve brand data for
@@ -1432,17 +1429,15 @@ def retrieve_by_name(
extra_body: Body | None = None,
timeout: float | httpx.Timeout | None | NotGiven = not_given,
) -> BrandRetrieveByNameResponse:
- """Retrieve brand information using a company name.
-
- This endpoint searches for the
- company by name and returns its brand data.
+ """
+ Retrieve brand information using a company name.
Args:
name: Company name to retrieve brand data for (e.g., 'Apple Inc', 'Microsoft
Corporation'). Must be 3-30 characters.
- country_gl: Optional country code (GL parameter) to specify the country. This affects the
- geographic location used for search queries.
+ country_gl: Optional country code hint (GL parameter) to specify the country for the company
+ name.
force_language: Optional parameter to force the language of the retrieved brand data.
@@ -1694,10 +1689,8 @@ def retrieve_by_ticker(
extra_body: Body | None = None,
timeout: float | httpx.Timeout | None | NotGiven = not_given,
) -> BrandRetrieveByTickerResponse:
- """Retrieve brand information using a stock ticker symbol.
-
- This endpoint looks up
- the company associated with the ticker and returns its brand data.
+ """
+ Retrieve brand information using a stock ticker symbol.
Args:
ticker: Stock ticker symbol to retrieve brand data for (e.g., 'AAPL', 'GOOGL', 'BRK.A').
@@ -1758,8 +1751,8 @@ def retrieve_simplified(
) -> BrandRetrieveSimplifiedResponse:
"""
Returns a simplified version of brand data containing only essential
- information: domain, title, colors, logos, and backdrops. This endpoint is
- optimized for faster responses and reduced data transfer.
+ information: domain, title, colors, logos, and backdrops. Optimized for faster
+ responses and reduced data transfer.
Args:
domain: Domain name to retrieve simplified brand data for
@@ -2395,7 +2388,6 @@ async def identify_from_transaction(
high_confidence_only: When set to true, the API will perform an additional verification steps to
ensure the identified brand matches the transaction with high confidence.
- Defaults to false.
max_speed: Optional parameter to optimize the API call for maximum speed. When set to true,
the API will skip time-consuming operations for faster response at the cost of
@@ -2581,9 +2573,8 @@ async def retrieve_by_email(
) -> BrandRetrieveByEmailResponse:
"""
Retrieve brand information using an email address while detecting disposable and
- free email addresses. This endpoint extracts the domain from the email address
- and returns brand data for that domain. Disposable and free email addresses
- (like gmail.com, yahoo.com) will throw a 422 error.
+ free email addresses. Disposable and free email addresses (like gmail.com,
+ yahoo.com) will throw a 422 error.
Args:
email: Email address to retrieve brand data for (e.g., 'contact@example.com'). The
@@ -2766,8 +2757,7 @@ async def retrieve_by_isin(
) -> BrandRetrieveByIsinResponse:
"""
Retrieve brand information using an ISIN (International Securities
- Identification Number). This endpoint looks up the company associated with the
- ISIN and returns its brand data.
+ Identification Number).
Args:
isin: ISIN (International Securities Identification Number) to retrieve brand data for
@@ -3190,17 +3180,15 @@ async def retrieve_by_name(
extra_body: Body | None = None,
timeout: float | httpx.Timeout | None | NotGiven = not_given,
) -> BrandRetrieveByNameResponse:
- """Retrieve brand information using a company name.
-
- This endpoint searches for the
- company by name and returns its brand data.
+ """
+ Retrieve brand information using a company name.
Args:
name: Company name to retrieve brand data for (e.g., 'Apple Inc', 'Microsoft
Corporation'). Must be 3-30 characters.
- country_gl: Optional country code (GL parameter) to specify the country. This affects the
- geographic location used for search queries.
+ country_gl: Optional country code hint (GL parameter) to specify the country for the company
+ name.
force_language: Optional parameter to force the language of the retrieved brand data.
@@ -3452,10 +3440,8 @@ async def retrieve_by_ticker(
extra_body: Body | None = None,
timeout: float | httpx.Timeout | None | NotGiven = not_given,
) -> BrandRetrieveByTickerResponse:
- """Retrieve brand information using a stock ticker symbol.
-
- This endpoint looks up
- the company associated with the ticker and returns its brand data.
+ """
+ Retrieve brand information using a stock ticker symbol.
Args:
ticker: Stock ticker symbol to retrieve brand data for (e.g., 'AAPL', 'GOOGL', 'BRK.A').
@@ -3516,8 +3502,8 @@ async def retrieve_simplified(
) -> BrandRetrieveSimplifiedResponse:
"""
Returns a simplified version of brand data containing only essential
- information: domain, title, colors, logos, and backdrops. This endpoint is
- optimized for faster responses and reduced data transfer.
+ information: domain, title, colors, logos, and backdrops. Optimized for faster
+ responses and reduced data transfer.
Args:
domain: Domain name to retrieve simplified brand data for
diff --git a/src/context/dev/resources/industry.py b/src/context/dev/resources/industry.py
index ae71953..73c96ac 100644
--- a/src/context/dev/resources/industry.py
+++ b/src/context/dev/resources/industry.py
@@ -56,12 +56,12 @@ def retrieve_naics(
timeout: float | httpx.Timeout | None | NotGiven = not_given,
) -> IndustryRetrieveNaicsResponse:
"""
- Endpoint to classify any brand into a 2022 NAICS code.
+ Classify any brand into 2022 NAICS industry codes from its domain or name.
Args:
- input: Brand domain or title to retrieve NAICS code for. If a valid domain is provided
- in `input`, it will be used for classification, otherwise, we will search for
- the brand using the provided title.
+ input: Brand domain or title to retrieve NAICS code for. If a valid domain is provided,
+ it will be used for classification, otherwise, we will search for the brand
+ using the provided title.
max_results: Maximum number of NAICS codes to return. Must be between 1 and 10. Defaults
to 5.
@@ -81,7 +81,7 @@ def retrieve_naics(
timeout: Override the client-level default timeout for this request, in seconds
"""
return self._get(
- "/brand/naics",
+ "/web/naics",
options=make_request_options(
extra_headers=extra_headers,
extra_query=extra_query,
@@ -136,12 +136,12 @@ async def retrieve_naics(
timeout: float | httpx.Timeout | None | NotGiven = not_given,
) -> IndustryRetrieveNaicsResponse:
"""
- Endpoint to classify any brand into a 2022 NAICS code.
+ Classify any brand into 2022 NAICS industry codes from its domain or name.
Args:
- input: Brand domain or title to retrieve NAICS code for. If a valid domain is provided
- in `input`, it will be used for classification, otherwise, we will search for
- the brand using the provided title.
+ input: Brand domain or title to retrieve NAICS code for. If a valid domain is provided,
+ it will be used for classification, otherwise, we will search for the brand
+ using the provided title.
max_results: Maximum number of NAICS codes to return. Must be between 1 and 10. Defaults
to 5.
@@ -161,7 +161,7 @@ async def retrieve_naics(
timeout: Override the client-level default timeout for this request, in seconds
"""
return await self._get(
- "/brand/naics",
+ "/web/naics",
options=make_request_options(
extra_headers=extra_headers,
extra_query=extra_query,
diff --git a/src/context/dev/resources/style.py b/src/context/dev/resources/style.py
deleted file mode 100644
index 36ea477..0000000
--- a/src/context/dev/resources/style.py
+++ /dev/null
@@ -1,326 +0,0 @@
-# File generated from our OpenAPI spec by Stainless. See CONTRIBUTING.md for details.
-
-from __future__ import annotations
-
-import httpx
-
-from ..types import style_extract_fonts_params, style_extract_styleguide_params
-from .._types import Body, Omit, Query, Headers, NotGiven, omit, not_given
-from .._utils import maybe_transform, async_maybe_transform
-from .._compat import cached_property
-from .._resource import SyncAPIResource, AsyncAPIResource
-from .._response import (
- to_raw_response_wrapper,
- to_streamed_response_wrapper,
- async_to_raw_response_wrapper,
- async_to_streamed_response_wrapper,
-)
-from .._base_client import make_request_options
-from ..types.style_extract_fonts_response import StyleExtractFontsResponse
-from ..types.style_extract_styleguide_response import StyleExtractStyleguideResponse
-
-__all__ = ["StyleResource", "AsyncStyleResource"]
-
-
-class StyleResource(SyncAPIResource):
- @cached_property
- def with_raw_response(self) -> StyleResourceWithRawResponse:
- """
- This property can be used as a prefix for any HTTP method call to return
- the raw response object instead of the parsed content.
-
- For more information, see https://www.github.com/context-dot-dev/context-python-sdk#accessing-raw-response-data-eg-headers
- """
- return StyleResourceWithRawResponse(self)
-
- @cached_property
- def with_streaming_response(self) -> StyleResourceWithStreamingResponse:
- """
- An alternative to `.with_raw_response` that doesn't eagerly read the response body.
-
- For more information, see https://www.github.com/context-dot-dev/context-python-sdk#with_streaming_response
- """
- return StyleResourceWithStreamingResponse(self)
-
- def extract_fonts(
- self,
- *,
- domain: str,
- timeout_ms: int | Omit = omit,
- # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
- # The extra values given here take precedence over values defined on the client or passed to this method.
- extra_headers: Headers | None = None,
- extra_query: Query | None = None,
- extra_body: Body | None = None,
- timeout: float | httpx.Timeout | None | NotGiven = not_given,
- ) -> StyleExtractFontsResponse:
- """
- Extract font information from a brand's website including font families, usage
- statistics, fallbacks, and element/word counts.
-
- Args:
- domain: Domain name to extract fonts from (e.g., 'example.com', 'google.com'). The
- domain will be automatically normalized and validated.
-
- timeout_ms: Optional timeout in milliseconds for the request. If the request takes longer
- than this value, it will be aborted with a 408 status code. Maximum allowed
- value is 300000ms (5 minutes).
-
- extra_headers: Send extra headers
-
- extra_query: Add additional query parameters to the request
-
- extra_body: Add additional JSON properties to the request
-
- timeout: Override the client-level default timeout for this request, in seconds
- """
- return self._get(
- "/brand/fonts",
- options=make_request_options(
- extra_headers=extra_headers,
- extra_query=extra_query,
- extra_body=extra_body,
- timeout=timeout,
- query=maybe_transform(
- {
- "domain": domain,
- "timeout_ms": timeout_ms,
- },
- style_extract_fonts_params.StyleExtractFontsParams,
- ),
- ),
- cast_to=StyleExtractFontsResponse,
- )
-
- def extract_styleguide(
- self,
- *,
- direct_url: str | Omit = omit,
- domain: str | Omit = omit,
- timeout_ms: int | Omit = omit,
- # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
- # The extra values given here take precedence over values defined on the client or passed to this method.
- extra_headers: Headers | None = None,
- extra_query: Query | None = None,
- extra_body: Body | None = None,
- timeout: float | httpx.Timeout | None | NotGiven = not_given,
- ) -> StyleExtractStyleguideResponse:
- """
- Automatically extract comprehensive design system information from a brand's
- website including colors, typography, spacing, shadows, and UI components.
- Either 'domain' or 'directUrl' must be provided as a query parameter, but not
- both.
-
- Args:
- direct_url: A specific URL to fetch the styleguide from directly, bypassing domain
- resolution (e.g., 'https://example.com/design-system').
-
- domain: Domain name to extract styleguide from (e.g., 'example.com', 'google.com'). The
- domain will be automatically normalized and validated.
-
- timeout_ms: Optional timeout in milliseconds for the request. If the request takes longer
- than this value, it will be aborted with a 408 status code. Maximum allowed
- value is 300000ms (5 minutes).
-
- extra_headers: Send extra headers
-
- extra_query: Add additional query parameters to the request
-
- extra_body: Add additional JSON properties to the request
-
- timeout: Override the client-level default timeout for this request, in seconds
- """
- return self._get(
- "/brand/styleguide",
- options=make_request_options(
- extra_headers=extra_headers,
- extra_query=extra_query,
- extra_body=extra_body,
- timeout=timeout,
- query=maybe_transform(
- {
- "direct_url": direct_url,
- "domain": domain,
- "timeout_ms": timeout_ms,
- },
- style_extract_styleguide_params.StyleExtractStyleguideParams,
- ),
- ),
- cast_to=StyleExtractStyleguideResponse,
- )
-
-
-class AsyncStyleResource(AsyncAPIResource):
- @cached_property
- def with_raw_response(self) -> AsyncStyleResourceWithRawResponse:
- """
- This property can be used as a prefix for any HTTP method call to return
- the raw response object instead of the parsed content.
-
- For more information, see https://www.github.com/context-dot-dev/context-python-sdk#accessing-raw-response-data-eg-headers
- """
- return AsyncStyleResourceWithRawResponse(self)
-
- @cached_property
- def with_streaming_response(self) -> AsyncStyleResourceWithStreamingResponse:
- """
- An alternative to `.with_raw_response` that doesn't eagerly read the response body.
-
- For more information, see https://www.github.com/context-dot-dev/context-python-sdk#with_streaming_response
- """
- return AsyncStyleResourceWithStreamingResponse(self)
-
- async def extract_fonts(
- self,
- *,
- domain: str,
- timeout_ms: int | Omit = omit,
- # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
- # The extra values given here take precedence over values defined on the client or passed to this method.
- extra_headers: Headers | None = None,
- extra_query: Query | None = None,
- extra_body: Body | None = None,
- timeout: float | httpx.Timeout | None | NotGiven = not_given,
- ) -> StyleExtractFontsResponse:
- """
- Extract font information from a brand's website including font families, usage
- statistics, fallbacks, and element/word counts.
-
- Args:
- domain: Domain name to extract fonts from (e.g., 'example.com', 'google.com'). The
- domain will be automatically normalized and validated.
-
- timeout_ms: Optional timeout in milliseconds for the request. If the request takes longer
- than this value, it will be aborted with a 408 status code. Maximum allowed
- value is 300000ms (5 minutes).
-
- extra_headers: Send extra headers
-
- extra_query: Add additional query parameters to the request
-
- extra_body: Add additional JSON properties to the request
-
- timeout: Override the client-level default timeout for this request, in seconds
- """
- return await self._get(
- "/brand/fonts",
- options=make_request_options(
- extra_headers=extra_headers,
- extra_query=extra_query,
- extra_body=extra_body,
- timeout=timeout,
- query=await async_maybe_transform(
- {
- "domain": domain,
- "timeout_ms": timeout_ms,
- },
- style_extract_fonts_params.StyleExtractFontsParams,
- ),
- ),
- cast_to=StyleExtractFontsResponse,
- )
-
- async def extract_styleguide(
- self,
- *,
- direct_url: str | Omit = omit,
- domain: str | Omit = omit,
- timeout_ms: int | Omit = omit,
- # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
- # The extra values given here take precedence over values defined on the client or passed to this method.
- extra_headers: Headers | None = None,
- extra_query: Query | None = None,
- extra_body: Body | None = None,
- timeout: float | httpx.Timeout | None | NotGiven = not_given,
- ) -> StyleExtractStyleguideResponse:
- """
- Automatically extract comprehensive design system information from a brand's
- website including colors, typography, spacing, shadows, and UI components.
- Either 'domain' or 'directUrl' must be provided as a query parameter, but not
- both.
-
- Args:
- direct_url: A specific URL to fetch the styleguide from directly, bypassing domain
- resolution (e.g., 'https://example.com/design-system').
-
- domain: Domain name to extract styleguide from (e.g., 'example.com', 'google.com'). The
- domain will be automatically normalized and validated.
-
- timeout_ms: Optional timeout in milliseconds for the request. If the request takes longer
- than this value, it will be aborted with a 408 status code. Maximum allowed
- value is 300000ms (5 minutes).
-
- extra_headers: Send extra headers
-
- extra_query: Add additional query parameters to the request
-
- extra_body: Add additional JSON properties to the request
-
- timeout: Override the client-level default timeout for this request, in seconds
- """
- return await self._get(
- "/brand/styleguide",
- options=make_request_options(
- extra_headers=extra_headers,
- extra_query=extra_query,
- extra_body=extra_body,
- timeout=timeout,
- query=await async_maybe_transform(
- {
- "direct_url": direct_url,
- "domain": domain,
- "timeout_ms": timeout_ms,
- },
- style_extract_styleguide_params.StyleExtractStyleguideParams,
- ),
- ),
- cast_to=StyleExtractStyleguideResponse,
- )
-
-
-class StyleResourceWithRawResponse:
- def __init__(self, style: StyleResource) -> None:
- self._style = style
-
- self.extract_fonts = to_raw_response_wrapper(
- style.extract_fonts,
- )
- self.extract_styleguide = to_raw_response_wrapper(
- style.extract_styleguide,
- )
-
-
-class AsyncStyleResourceWithRawResponse:
- def __init__(self, style: AsyncStyleResource) -> None:
- self._style = style
-
- self.extract_fonts = async_to_raw_response_wrapper(
- style.extract_fonts,
- )
- self.extract_styleguide = async_to_raw_response_wrapper(
- style.extract_styleguide,
- )
-
-
-class StyleResourceWithStreamingResponse:
- def __init__(self, style: StyleResource) -> None:
- self._style = style
-
- self.extract_fonts = to_streamed_response_wrapper(
- style.extract_fonts,
- )
- self.extract_styleguide = to_streamed_response_wrapper(
- style.extract_styleguide,
- )
-
-
-class AsyncStyleResourceWithStreamingResponse:
- def __init__(self, style: AsyncStyleResource) -> None:
- self._style = style
-
- self.extract_fonts = async_to_streamed_response_wrapper(
- style.extract_fonts,
- )
- self.extract_styleguide = async_to_streamed_response_wrapper(
- style.extract_styleguide,
- )
diff --git a/src/context/dev/resources/web.py b/src/context/dev/resources/web.py
index 33c3d6f..2b68c4f 100644
--- a/src/context/dev/resources/web.py
+++ b/src/context/dev/resources/web.py
@@ -9,9 +9,11 @@
from ..types import (
web_screenshot_params,
web_web_crawl_md_params,
+ web_extract_fonts_params,
web_web_scrape_md_params,
web_web_scrape_html_params,
web_web_scrape_images_params,
+ web_extract_styleguide_params,
web_web_scrape_sitemap_params,
)
from .._types import Body, Omit, Query, Headers, NotGiven, omit, not_given
@@ -27,9 +29,11 @@
from .._base_client import make_request_options
from ..types.web_screenshot_response import WebScreenshotResponse
from ..types.web_web_crawl_md_response import WebWebCrawlMdResponse
+from ..types.web_extract_fonts_response import WebExtractFontsResponse
from ..types.web_web_scrape_md_response import WebWebScrapeMdResponse
from ..types.web_web_scrape_html_response import WebWebScrapeHTMLResponse
from ..types.web_web_scrape_images_response import WebWebScrapeImagesResponse
+from ..types.web_extract_styleguide_response import WebExtractStyleguideResponse
from ..types.web_web_scrape_sitemap_response import WebWebScrapeSitemapResponse
__all__ = ["WebResource", "AsyncWebResource"]
@@ -55,6 +59,121 @@ def with_streaming_response(self) -> WebResourceWithStreamingResponse:
"""
return WebResourceWithStreamingResponse(self)
+ def extract_fonts(
+ self,
+ *,
+ direct_url: str | Omit = omit,
+ domain: str | Omit = omit,
+ timeout_ms: int | Omit = omit,
+ # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
+ # The extra values given here take precedence over values defined on the client or passed to this method.
+ extra_headers: Headers | None = None,
+ extra_query: Query | None = None,
+ extra_body: Body | None = None,
+ timeout: float | httpx.Timeout | None | NotGiven = not_given,
+ ) -> WebExtractFontsResponse:
+ """
+ Scrape font information from a website including font families, usage
+ statistics, fallbacks, and element/word counts.
+
+ Args:
+ direct_url: A specific URL to fetch fonts from directly, bypassing domain resolution (e.g.,
+ 'https://example.com/design-system'). When provided, fonts are extracted from
+ this exact URL. You must provide either 'domain' or 'directUrl', but not both.
+
+ domain: Domain name to extract fonts from (e.g., 'example.com', 'google.com'). The
+ domain will be automatically normalized and validated. You must provide either
+ 'domain' or 'directUrl', but not both.
+
+ timeout_ms: Optional timeout in milliseconds for the request. If the request takes longer
+ than this value, it will be aborted with a 408 status code. Maximum allowed
+ value is 300000ms (5 minutes).
+
+ extra_headers: Send extra headers
+
+ extra_query: Add additional query parameters to the request
+
+ extra_body: Add additional JSON properties to the request
+
+ timeout: Override the client-level default timeout for this request, in seconds
+ """
+ return self._get(
+ "/web/fonts",
+ options=make_request_options(
+ extra_headers=extra_headers,
+ extra_query=extra_query,
+ extra_body=extra_body,
+ timeout=timeout,
+ query=maybe_transform(
+ {
+ "direct_url": direct_url,
+ "domain": domain,
+ "timeout_ms": timeout_ms,
+ },
+ web_extract_fonts_params.WebExtractFontsParams,
+ ),
+ ),
+ cast_to=WebExtractFontsResponse,
+ )
+
+ def extract_styleguide(
+ self,
+ *,
+ direct_url: str | Omit = omit,
+ domain: str | Omit = omit,
+ timeout_ms: int | Omit = omit,
+ # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
+ # The extra values given here take precedence over values defined on the client or passed to this method.
+ extra_headers: Headers | None = None,
+ extra_query: Query | None = None,
+ extra_body: Body | None = None,
+ timeout: float | httpx.Timeout | None | NotGiven = not_given,
+ ) -> WebExtractStyleguideResponse:
+ """
+ Extract a comprehensive design system from a website including colors,
+ typography, spacing, shadows, and UI components.
+
+ Args:
+ direct_url: A specific URL to fetch the styleguide from directly, bypassing domain
+ resolution (e.g., 'https://example.com/design-system'). When provided, the
+ styleguide is extracted from this exact URL. You must provide either 'domain' or
+ 'directUrl', but not both.
+
+ domain: Domain name to extract styleguide from (e.g., 'example.com', 'google.com'). The
+ domain will be automatically normalized and validated. You must provide either
+ 'domain' or 'directUrl', but not both.
+
+ timeout_ms: Optional timeout in milliseconds for the request. If the request takes longer
+ than this value, it will be aborted with a 408 status code. Maximum allowed
+ value is 300000ms (5 minutes).
+
+ extra_headers: Send extra headers
+
+ extra_query: Add additional query parameters to the request
+
+ extra_body: Add additional JSON properties to the request
+
+ timeout: Override the client-level default timeout for this request, in seconds
+ """
+ return self._get(
+ "/web/styleguide",
+ options=make_request_options(
+ extra_headers=extra_headers,
+ extra_query=extra_query,
+ extra_body=extra_body,
+ timeout=timeout,
+ query=maybe_transform(
+ {
+ "direct_url": direct_url,
+ "domain": domain,
+ "timeout_ms": timeout_ms,
+ },
+ web_extract_styleguide_params.WebExtractStyleguideParams,
+ ),
+ ),
+ cast_to=WebExtractStyleguideResponse,
+ )
+
def screenshot(
self,
*,
@@ -70,21 +189,17 @@ def screenshot(
extra_body: Body | None = None,
timeout: float | httpx.Timeout | None | NotGiven = not_given,
) -> WebScreenshotResponse:
- """Capture a screenshot of a website.
-
- Supports both viewport (standard browser
- view) and full-page screenshots. Can also screenshot specific page types (login,
- pricing, etc.) by using heuristics to find the appropriate URL. Either 'domain'
- or 'directUrl' must be provided as a query parameter, but not both. Returns a
- URL to the uploaded screenshot image hosted on our CDN.
+ """
+ Capture a screenshot of a website.
Args:
direct_url: A specific URL to screenshot directly, bypassing domain resolution (e.g.,
'https://example.com/pricing'). When provided, the screenshot is taken of this
- exact URL.
+ exact URL. You must provide either 'domain' or 'directUrl', but not both.
domain: Domain name to take screenshot of (e.g., 'example.com', 'google.com'). The
- domain will be automatically normalized and validated.
+ domain will be automatically normalized and validated. You must provide either
+ 'domain' or 'directUrl', but not both.
full_screenshot: Optional parameter to determine screenshot type. If 'true', takes a full page
screenshot capturing all content. If 'false' or not provided, takes a viewport
@@ -109,7 +224,7 @@ def screenshot(
timeout: Override the client-level default timeout for this request, in seconds
"""
return self._get(
- "/brand/screenshot",
+ "/web/screenshot",
options=make_request_options(
extra_headers=extra_headers,
extra_query=extra_query,
@@ -150,8 +265,7 @@ def web_crawl_md(
) -> WebWebCrawlMdResponse:
"""
Performs a crawl starting from a given URL, extracts page content as Markdown,
- and returns results for all crawled pages. Only follows links within the same
- domain as the starting URL. Costs 1 credit per successful page crawled.
+ and returns results for all crawled pages.
Args:
url: The starting URL for the crawl (must include http:// or https:// protocol)
@@ -209,6 +323,7 @@ def web_scrape_html(
self,
*,
url: str,
+ max_age_ms: int | Omit = omit,
# Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
# The extra values given here take precedence over values defined on the client or passed to this method.
extra_headers: Headers | None = None,
@@ -222,6 +337,10 @@ def web_scrape_html(
Args:
url: Full URL to scrape (must include http:// or https:// protocol)
+ max_age_ms: Return a cached result if a prior scrape for the same parameters exists and is
+ younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
+ omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
+
extra_headers: Send extra headers
extra_query: Add additional query parameters to the request
@@ -237,7 +356,13 @@ def web_scrape_html(
extra_query=extra_query,
extra_body=extra_body,
timeout=timeout,
- query=maybe_transform({"url": url}, web_web_scrape_html_params.WebWebScrapeHTMLParams),
+ query=maybe_transform(
+ {
+ "url": url,
+ "max_age_ms": max_age_ms,
+ },
+ web_web_scrape_html_params.WebWebScrapeHTMLParams,
+ ),
),
cast_to=WebWebScrapeHTMLResponse,
)
@@ -288,6 +413,7 @@ def web_scrape_md(
url: str,
include_images: bool | Omit = omit,
include_links: bool | Omit = omit,
+ max_age_ms: int | Omit = omit,
shorten_base64_images: bool | Omit = omit,
use_main_content_only: bool | Omit = omit,
# Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
@@ -298,17 +424,20 @@ def web_scrape_md(
timeout: float | httpx.Timeout | None | NotGiven = not_given,
) -> WebWebScrapeMdResponse:
"""
- Scrapes the given URL, converts the HTML content to Markdown, and returns the
- result.
+ Scrapes the given URL into LLM usable Markdown.
Args:
- url: Full URL to scrape and convert to markdown (must include http:// or https://
+ url: Full URL to scrape into LLM usable Markdown (must include http:// or https://
protocol)
include_images: Include image references in Markdown output
include_links: Preserve hyperlinks in Markdown output
+ max_age_ms: Return a cached result if a prior scrape for the same parameters exists and is
+ younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
+ omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
+
shorten_base64_images: Shorten base64-encoded image data in the Markdown output
use_main_content_only: Extract only the main content of the page, excluding headers, footers, sidebars,
@@ -334,6 +463,7 @@ def web_scrape_md(
"url": url,
"include_images": include_images,
"include_links": include_links,
+ "max_age_ms": max_age_ms,
"shorten_base64_images": shorten_base64_images,
"use_main_content_only": use_main_content_only,
},
@@ -356,13 +486,10 @@ def web_scrape_sitemap(
timeout: float | httpx.Timeout | None | NotGiven = not_given,
) -> WebWebScrapeSitemapResponse:
"""
- Crawls the sitemap of the given domain and returns all discovered page URLs.
- Supports sitemap index files (recursive), parallel fetching with concurrency
- control, deduplication, and filters out non-page resources (images, PDFs, etc.).
+ Crawl an entire website's sitemap and return all discovered page URLs.
Args:
- domain: Domain name to crawl sitemaps for (e.g., 'example.com'). The domain will be
- automatically normalized and validated.
+ domain: Domain to build a sitemap for
max_links: Maximum number of links to return from the sitemap crawl. Defaults to 10,000.
Minimum is 1, maximum is 100,000.
@@ -414,6 +541,121 @@ def with_streaming_response(self) -> AsyncWebResourceWithStreamingResponse:
"""
return AsyncWebResourceWithStreamingResponse(self)
+ async def extract_fonts(
+ self,
+ *,
+ direct_url: str | Omit = omit,
+ domain: str | Omit = omit,
+ timeout_ms: int | Omit = omit,
+ # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
+ # The extra values given here take precedence over values defined on the client or passed to this method.
+ extra_headers: Headers | None = None,
+ extra_query: Query | None = None,
+ extra_body: Body | None = None,
+ timeout: float | httpx.Timeout | None | NotGiven = not_given,
+ ) -> WebExtractFontsResponse:
+ """
+ Scrape font information from a website including font families, usage
+ statistics, fallbacks, and element/word counts.
+
+ Args:
+ direct_url: A specific URL to fetch fonts from directly, bypassing domain resolution (e.g.,
+ 'https://example.com/design-system'). When provided, fonts are extracted from
+ this exact URL. You must provide either 'domain' or 'directUrl', but not both.
+
+ domain: Domain name to extract fonts from (e.g., 'example.com', 'google.com'). The
+ domain will be automatically normalized and validated. You must provide either
+ 'domain' or 'directUrl', but not both.
+
+ timeout_ms: Optional timeout in milliseconds for the request. If the request takes longer
+ than this value, it will be aborted with a 408 status code. Maximum allowed
+ value is 300000ms (5 minutes).
+
+ extra_headers: Send extra headers
+
+ extra_query: Add additional query parameters to the request
+
+ extra_body: Add additional JSON properties to the request
+
+ timeout: Override the client-level default timeout for this request, in seconds
+ """
+ return await self._get(
+ "/web/fonts",
+ options=make_request_options(
+ extra_headers=extra_headers,
+ extra_query=extra_query,
+ extra_body=extra_body,
+ timeout=timeout,
+ query=await async_maybe_transform(
+ {
+ "direct_url": direct_url,
+ "domain": domain,
+ "timeout_ms": timeout_ms,
+ },
+ web_extract_fonts_params.WebExtractFontsParams,
+ ),
+ ),
+ cast_to=WebExtractFontsResponse,
+ )
+
+ async def extract_styleguide(
+ self,
+ *,
+ direct_url: str | Omit = omit,
+ domain: str | Omit = omit,
+ timeout_ms: int | Omit = omit,
+ # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
+ # The extra values given here take precedence over values defined on the client or passed to this method.
+ extra_headers: Headers | None = None,
+ extra_query: Query | None = None,
+ extra_body: Body | None = None,
+ timeout: float | httpx.Timeout | None | NotGiven = not_given,
+ ) -> WebExtractStyleguideResponse:
+ """
+ Extract a comprehensive design system from a website including colors,
+ typography, spacing, shadows, and UI components.
+
+ Args:
+ direct_url: A specific URL to fetch the styleguide from directly, bypassing domain
+ resolution (e.g., 'https://example.com/design-system'). When provided, the
+ styleguide is extracted from this exact URL. You must provide either 'domain' or
+ 'directUrl', but not both.
+
+ domain: Domain name to extract styleguide from (e.g., 'example.com', 'google.com'). The
+ domain will be automatically normalized and validated. You must provide either
+ 'domain' or 'directUrl', but not both.
+
+ timeout_ms: Optional timeout in milliseconds for the request. If the request takes longer
+ than this value, it will be aborted with a 408 status code. Maximum allowed
+ value is 300000ms (5 minutes).
+
+ extra_headers: Send extra headers
+
+ extra_query: Add additional query parameters to the request
+
+ extra_body: Add additional JSON properties to the request
+
+ timeout: Override the client-level default timeout for this request, in seconds
+ """
+ return await self._get(
+ "/web/styleguide",
+ options=make_request_options(
+ extra_headers=extra_headers,
+ extra_query=extra_query,
+ extra_body=extra_body,
+ timeout=timeout,
+ query=await async_maybe_transform(
+ {
+ "direct_url": direct_url,
+ "domain": domain,
+ "timeout_ms": timeout_ms,
+ },
+ web_extract_styleguide_params.WebExtractStyleguideParams,
+ ),
+ ),
+ cast_to=WebExtractStyleguideResponse,
+ )
+
async def screenshot(
self,
*,
@@ -429,21 +671,17 @@ async def screenshot(
extra_body: Body | None = None,
timeout: float | httpx.Timeout | None | NotGiven = not_given,
) -> WebScreenshotResponse:
- """Capture a screenshot of a website.
-
- Supports both viewport (standard browser
- view) and full-page screenshots. Can also screenshot specific page types (login,
- pricing, etc.) by using heuristics to find the appropriate URL. Either 'domain'
- or 'directUrl' must be provided as a query parameter, but not both. Returns a
- URL to the uploaded screenshot image hosted on our CDN.
+ """
+ Capture a screenshot of a website.
Args:
direct_url: A specific URL to screenshot directly, bypassing domain resolution (e.g.,
'https://example.com/pricing'). When provided, the screenshot is taken of this
- exact URL.
+ exact URL. You must provide either 'domain' or 'directUrl', but not both.
domain: Domain name to take screenshot of (e.g., 'example.com', 'google.com'). The
- domain will be automatically normalized and validated.
+ domain will be automatically normalized and validated. You must provide either
+ 'domain' or 'directUrl', but not both.
full_screenshot: Optional parameter to determine screenshot type. If 'true', takes a full page
screenshot capturing all content. If 'false' or not provided, takes a viewport
@@ -468,7 +706,7 @@ async def screenshot(
timeout: Override the client-level default timeout for this request, in seconds
"""
return await self._get(
- "/brand/screenshot",
+ "/web/screenshot",
options=make_request_options(
extra_headers=extra_headers,
extra_query=extra_query,
@@ -509,8 +747,7 @@ async def web_crawl_md(
) -> WebWebCrawlMdResponse:
"""
Performs a crawl starting from a given URL, extracts page content as Markdown,
- and returns results for all crawled pages. Only follows links within the same
- domain as the starting URL. Costs 1 credit per successful page crawled.
+ and returns results for all crawled pages.
Args:
url: The starting URL for the crawl (must include http:// or https:// protocol)
@@ -568,6 +805,7 @@ async def web_scrape_html(
self,
*,
url: str,
+ max_age_ms: int | Omit = omit,
# Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
# The extra values given here take precedence over values defined on the client or passed to this method.
extra_headers: Headers | None = None,
@@ -581,6 +819,10 @@ async def web_scrape_html(
Args:
url: Full URL to scrape (must include http:// or https:// protocol)
+ max_age_ms: Return a cached result if a prior scrape for the same parameters exists and is
+ younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
+ omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
+
extra_headers: Send extra headers
extra_query: Add additional query parameters to the request
@@ -596,7 +838,13 @@ async def web_scrape_html(
extra_query=extra_query,
extra_body=extra_body,
timeout=timeout,
- query=await async_maybe_transform({"url": url}, web_web_scrape_html_params.WebWebScrapeHTMLParams),
+ query=await async_maybe_transform(
+ {
+ "url": url,
+ "max_age_ms": max_age_ms,
+ },
+ web_web_scrape_html_params.WebWebScrapeHTMLParams,
+ ),
),
cast_to=WebWebScrapeHTMLResponse,
)
@@ -647,6 +895,7 @@ async def web_scrape_md(
url: str,
include_images: bool | Omit = omit,
include_links: bool | Omit = omit,
+ max_age_ms: int | Omit = omit,
shorten_base64_images: bool | Omit = omit,
use_main_content_only: bool | Omit = omit,
# Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
@@ -657,17 +906,20 @@ async def web_scrape_md(
timeout: float | httpx.Timeout | None | NotGiven = not_given,
) -> WebWebScrapeMdResponse:
"""
- Scrapes the given URL, converts the HTML content to Markdown, and returns the
- result.
+ Scrapes the given URL into LLM usable Markdown.
Args:
- url: Full URL to scrape and convert to markdown (must include http:// or https://
+ url: Full URL to scrape into LLM usable Markdown (must include http:// or https://
protocol)
include_images: Include image references in Markdown output
include_links: Preserve hyperlinks in Markdown output
+ max_age_ms: Return a cached result if a prior scrape for the same parameters exists and is
+ younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
+ omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
+
shorten_base64_images: Shorten base64-encoded image data in the Markdown output
use_main_content_only: Extract only the main content of the page, excluding headers, footers, sidebars,
@@ -693,6 +945,7 @@ async def web_scrape_md(
"url": url,
"include_images": include_images,
"include_links": include_links,
+ "max_age_ms": max_age_ms,
"shorten_base64_images": shorten_base64_images,
"use_main_content_only": use_main_content_only,
},
@@ -715,13 +968,10 @@ async def web_scrape_sitemap(
timeout: float | httpx.Timeout | None | NotGiven = not_given,
) -> WebWebScrapeSitemapResponse:
"""
- Crawls the sitemap of the given domain and returns all discovered page URLs.
- Supports sitemap index files (recursive), parallel fetching with concurrency
- control, deduplication, and filters out non-page resources (images, PDFs, etc.).
+ Crawl an entire website's sitemap and return all discovered page URLs.
Args:
- domain: Domain name to crawl sitemaps for (e.g., 'example.com'). The domain will be
- automatically normalized and validated.
+ domain: Domain to build a sitemap for
max_links: Maximum number of links to return from the sitemap crawl. Defaults to 10,000.
Minimum is 1, maximum is 100,000.
@@ -757,6 +1007,12 @@ class WebResourceWithRawResponse:
def __init__(self, web: WebResource) -> None:
self._web = web
+ self.extract_fonts = to_raw_response_wrapper(
+ web.extract_fonts,
+ )
+ self.extract_styleguide = to_raw_response_wrapper(
+ web.extract_styleguide,
+ )
self.screenshot = to_raw_response_wrapper(
web.screenshot,
)
@@ -781,6 +1037,12 @@ class AsyncWebResourceWithRawResponse:
def __init__(self, web: AsyncWebResource) -> None:
self._web = web
+ self.extract_fonts = async_to_raw_response_wrapper(
+ web.extract_fonts,
+ )
+ self.extract_styleguide = async_to_raw_response_wrapper(
+ web.extract_styleguide,
+ )
self.screenshot = async_to_raw_response_wrapper(
web.screenshot,
)
@@ -805,6 +1067,12 @@ class WebResourceWithStreamingResponse:
def __init__(self, web: WebResource) -> None:
self._web = web
+ self.extract_fonts = to_streamed_response_wrapper(
+ web.extract_fonts,
+ )
+ self.extract_styleguide = to_streamed_response_wrapper(
+ web.extract_styleguide,
+ )
self.screenshot = to_streamed_response_wrapper(
web.screenshot,
)
@@ -829,6 +1097,12 @@ class AsyncWebResourceWithStreamingResponse:
def __init__(self, web: AsyncWebResource) -> None:
self._web = web
+ self.extract_fonts = async_to_streamed_response_wrapper(
+ web.extract_fonts,
+ )
+ self.extract_styleguide = async_to_streamed_response_wrapper(
+ web.extract_styleguide,
+ )
self.screenshot = async_to_streamed_response_wrapper(
web.screenshot,
)
diff --git a/src/context/dev/types/__init__.py b/src/context/dev/types/__init__.py
index 4053426..232daf5 100644
--- a/src/context/dev/types/__init__.py
+++ b/src/context/dev/types/__init__.py
@@ -10,21 +10,22 @@
from .utility_prefetch_params import UtilityPrefetchParams as UtilityPrefetchParams
from .web_screenshot_response import WebScreenshotResponse as WebScreenshotResponse
from .web_web_crawl_md_params import WebWebCrawlMdParams as WebWebCrawlMdParams
+from .web_extract_fonts_params import WebExtractFontsParams as WebExtractFontsParams
from .web_web_scrape_md_params import WebWebScrapeMdParams as WebWebScrapeMdParams
from .ai_extract_product_params import AIExtractProductParams as AIExtractProductParams
from .utility_prefetch_response import UtilityPrefetchResponse as UtilityPrefetchResponse
from .web_web_crawl_md_response import WebWebCrawlMdResponse as WebWebCrawlMdResponse
from .ai_extract_products_params import AIExtractProductsParams as AIExtractProductsParams
-from .style_extract_fonts_params import StyleExtractFontsParams as StyleExtractFontsParams
+from .web_extract_fonts_response import WebExtractFontsResponse as WebExtractFontsResponse
from .web_web_scrape_html_params import WebWebScrapeHTMLParams as WebWebScrapeHTMLParams
from .web_web_scrape_md_response import WebWebScrapeMdResponse as WebWebScrapeMdResponse
from .ai_extract_product_response import AIExtractProductResponse as AIExtractProductResponse
from .ai_extract_products_response import AIExtractProductsResponse as AIExtractProductsResponse
-from .style_extract_fonts_response import StyleExtractFontsResponse as StyleExtractFontsResponse
from .web_web_scrape_html_response import WebWebScrapeHTMLResponse as WebWebScrapeHTMLResponse
from .web_web_scrape_images_params import WebWebScrapeImagesParams as WebWebScrapeImagesParams
from .brand_retrieve_by_isin_params import BrandRetrieveByIsinParams as BrandRetrieveByIsinParams
from .brand_retrieve_by_name_params import BrandRetrieveByNameParams as BrandRetrieveByNameParams
+from .web_extract_styleguide_params import WebExtractStyleguideParams as WebExtractStyleguideParams
from .web_web_scrape_sitemap_params import WebWebScrapeSitemapParams as WebWebScrapeSitemapParams
from .brand_retrieve_by_email_params import BrandRetrieveByEmailParams as BrandRetrieveByEmailParams
from .industry_retrieve_naics_params import IndustryRetrieveNaicsParams as IndustryRetrieveNaicsParams
@@ -32,14 +33,13 @@
from .brand_retrieve_by_isin_response import BrandRetrieveByIsinResponse as BrandRetrieveByIsinResponse
from .brand_retrieve_by_name_response import BrandRetrieveByNameResponse as BrandRetrieveByNameResponse
from .brand_retrieve_by_ticker_params import BrandRetrieveByTickerParams as BrandRetrieveByTickerParams
-from .style_extract_styleguide_params import StyleExtractStyleguideParams as StyleExtractStyleguideParams
+from .web_extract_styleguide_response import WebExtractStyleguideResponse as WebExtractStyleguideResponse
from .web_web_scrape_sitemap_response import WebWebScrapeSitemapResponse as WebWebScrapeSitemapResponse
from .brand_retrieve_by_email_response import BrandRetrieveByEmailResponse as BrandRetrieveByEmailResponse
from .brand_retrieve_simplified_params import BrandRetrieveSimplifiedParams as BrandRetrieveSimplifiedParams
from .industry_retrieve_naics_response import IndustryRetrieveNaicsResponse as IndustryRetrieveNaicsResponse
from .utility_prefetch_by_email_params import UtilityPrefetchByEmailParams as UtilityPrefetchByEmailParams
from .brand_retrieve_by_ticker_response import BrandRetrieveByTickerResponse as BrandRetrieveByTickerResponse
-from .style_extract_styleguide_response import StyleExtractStyleguideResponse as StyleExtractStyleguideResponse
from .brand_retrieve_simplified_response import BrandRetrieveSimplifiedResponse as BrandRetrieveSimplifiedResponse
from .utility_prefetch_by_email_response import UtilityPrefetchByEmailResponse as UtilityPrefetchByEmailResponse
from .brand_identify_from_transaction_params import (
diff --git a/src/context/dev/types/brand_identify_from_transaction_params.py b/src/context/dev/types/brand_identify_from_transaction_params.py
index bd3f5a8..e6ecadd 100644
--- a/src/context/dev/types/brand_identify_from_transaction_params.py
+++ b/src/context/dev/types/brand_identify_from_transaction_params.py
@@ -390,7 +390,6 @@ class BrandIdentifyFromTransactionParams(TypedDict, total=False):
"""
When set to true, the API will perform an additional verification steps to
ensure the identified brand matches the transaction with high confidence.
- Defaults to false.
"""
max_speed: Annotated[bool, PropertyInfo(alias="maxSpeed")]
diff --git a/src/context/dev/types/brand_retrieve_by_name_params.py b/src/context/dev/types/brand_retrieve_by_name_params.py
index 31a9eef..d774223 100644
--- a/src/context/dev/types/brand_retrieve_by_name_params.py
+++ b/src/context/dev/types/brand_retrieve_by_name_params.py
@@ -257,9 +257,9 @@ class BrandRetrieveByNameParams(TypedDict, total=False):
"zm",
"zw",
]
- """Optional country code (GL parameter) to specify the country.
-
- This affects the geographic location used for search queries.
+ """
+ Optional country code hint (GL parameter) to specify the country for the company
+ name.
"""
force_language: Literal[
diff --git a/src/context/dev/types/industry_retrieve_naics_params.py b/src/context/dev/types/industry_retrieve_naics_params.py
index cbaed87..03db484 100644
--- a/src/context/dev/types/industry_retrieve_naics_params.py
+++ b/src/context/dev/types/industry_retrieve_naics_params.py
@@ -13,8 +13,8 @@ class IndustryRetrieveNaicsParams(TypedDict, total=False):
input: Required[str]
"""Brand domain or title to retrieve NAICS code for.
- If a valid domain is provided in `input`, it will be used for classification,
- otherwise, we will search for the brand using the provided title.
+ If a valid domain is provided, it will be used for classification, otherwise, we
+ will search for the brand using the provided title.
"""
max_results: Annotated[int, PropertyInfo(alias="maxResults")]
diff --git a/src/context/dev/types/style_extract_fonts_params.py b/src/context/dev/types/style_extract_fonts_params.py
deleted file mode 100644
index 178f6f3..0000000
--- a/src/context/dev/types/style_extract_fonts_params.py
+++ /dev/null
@@ -1,24 +0,0 @@
-# File generated from our OpenAPI spec by Stainless. See CONTRIBUTING.md for details.
-
-from __future__ import annotations
-
-from typing_extensions import Required, Annotated, TypedDict
-
-from .._utils import PropertyInfo
-
-__all__ = ["StyleExtractFontsParams"]
-
-
-class StyleExtractFontsParams(TypedDict, total=False):
- domain: Required[str]
- """Domain name to extract fonts from (e.g., 'example.com', 'google.com').
-
- The domain will be automatically normalized and validated.
- """
-
- timeout_ms: Annotated[int, PropertyInfo(alias="timeoutMS")]
- """Optional timeout in milliseconds for the request.
-
- If the request takes longer than this value, it will be aborted with a 408
- status code. Maximum allowed value is 300000ms (5 minutes).
- """
diff --git a/src/context/dev/types/web_extract_fonts_params.py b/src/context/dev/types/web_extract_fonts_params.py
new file mode 100644
index 0000000..9bdc46e
--- /dev/null
+++ b/src/context/dev/types/web_extract_fonts_params.py
@@ -0,0 +1,32 @@
+# File generated from our OpenAPI spec by Stainless. See CONTRIBUTING.md for details.
+
+from __future__ import annotations
+
+from typing_extensions import Annotated, TypedDict
+
+from .._utils import PropertyInfo
+
+__all__ = ["WebExtractFontsParams"]
+
+
+class WebExtractFontsParams(TypedDict, total=False):
+ direct_url: Annotated[str, PropertyInfo(alias="directUrl")]
+ """
+ A specific URL to fetch fonts from directly, bypassing domain resolution (e.g.,
+ 'https://example.com/design-system'). When provided, fonts are extracted from
+ this exact URL. You must provide either 'domain' or 'directUrl', but not both.
+ """
+
+ domain: str
+ """Domain name to extract fonts from (e.g., 'example.com', 'google.com').
+
+ The domain will be automatically normalized and validated. You must provide
+ either 'domain' or 'directUrl', but not both.
+ """
+
+ timeout_ms: Annotated[int, PropertyInfo(alias="timeoutMS")]
+ """Optional timeout in milliseconds for the request.
+
+ If the request takes longer than this value, it will be aborted with a 408
+ status code. Maximum allowed value is 300000ms (5 minutes).
+ """
diff --git a/src/context/dev/types/style_extract_fonts_response.py b/src/context/dev/types/web_extract_fonts_response.py
similarity index 90%
rename from src/context/dev/types/style_extract_fonts_response.py
rename to src/context/dev/types/web_extract_fonts_response.py
index ab00e10..55fec61 100644
--- a/src/context/dev/types/style_extract_fonts_response.py
+++ b/src/context/dev/types/web_extract_fonts_response.py
@@ -4,7 +4,7 @@
from .._models import BaseModel
-__all__ = ["StyleExtractFontsResponse", "Font"]
+__all__ = ["WebExtractFontsResponse", "Font"]
class Font(BaseModel):
@@ -30,7 +30,7 @@ class Font(BaseModel):
"""Array of CSS selectors or element types where this font is used"""
-class StyleExtractFontsResponse(BaseModel):
+class WebExtractFontsResponse(BaseModel):
code: int
"""HTTP status code, e.g., 200"""
diff --git a/src/context/dev/types/style_extract_styleguide_params.py b/src/context/dev/types/web_extract_styleguide_params.py
similarity index 63%
rename from src/context/dev/types/style_extract_styleguide_params.py
rename to src/context/dev/types/web_extract_styleguide_params.py
index c07da4f..42b6d88 100644
--- a/src/context/dev/types/style_extract_styleguide_params.py
+++ b/src/context/dev/types/web_extract_styleguide_params.py
@@ -6,20 +6,23 @@
from .._utils import PropertyInfo
-__all__ = ["StyleExtractStyleguideParams"]
+__all__ = ["WebExtractStyleguideParams"]
-class StyleExtractStyleguideParams(TypedDict, total=False):
+class WebExtractStyleguideParams(TypedDict, total=False):
direct_url: Annotated[str, PropertyInfo(alias="directUrl")]
"""
A specific URL to fetch the styleguide from directly, bypassing domain
- resolution (e.g., 'https://example.com/design-system').
+ resolution (e.g., 'https://example.com/design-system'). When provided, the
+ styleguide is extracted from this exact URL. You must provide either 'domain' or
+ 'directUrl', but not both.
"""
domain: str
"""Domain name to extract styleguide from (e.g., 'example.com', 'google.com').
- The domain will be automatically normalized and validated.
+ The domain will be automatically normalized and validated. You must provide
+ either 'domain' or 'directUrl', but not both.
"""
timeout_ms: Annotated[int, PropertyInfo(alias="timeoutMS")]
diff --git a/src/context/dev/types/style_extract_styleguide_response.py b/src/context/dev/types/web_extract_styleguide_response.py
similarity index 90%
rename from src/context/dev/types/style_extract_styleguide_response.py
rename to src/context/dev/types/web_extract_styleguide_response.py
index 3dfe70d..77e2b55 100644
--- a/src/context/dev/types/style_extract_styleguide_response.py
+++ b/src/context/dev/types/web_extract_styleguide_response.py
@@ -1,6 +1,6 @@
# File generated from our OpenAPI spec by Stainless. See CONTRIBUTING.md for details.
-from typing import List, Optional
+from typing import Dict, List, Optional
from typing_extensions import Literal
from pydantic import Field as FieldInfo
@@ -8,7 +8,7 @@
from .._models import BaseModel
__all__ = [
- "StyleExtractStyleguideResponse",
+ "WebExtractStyleguideResponse",
"Styleguide",
"StyleguideColors",
"StyleguideComponents",
@@ -18,6 +18,7 @@
"StyleguideComponentsButtonSecondary",
"StyleguideComponentsCard",
"StyleguideElementSpacing",
+ "StyleguideFontLinks",
"StyleguideShadows",
"StyleguideTypography",
"StyleguideTypographyHeadings",
@@ -244,6 +245,30 @@ class StyleguideElementSpacing(BaseModel):
xs: str
+class StyleguideFontLinks(BaseModel):
+ files: Dict[str, str]
+ """Upright font files keyed by weight string (e.g.
+
+ "400" for regular, "500", "700"). Values are absolute URLs.
+ """
+
+ type: Literal["google", "custom"]
+
+ category: Optional[str] = None
+ """Google Fonts category when type is google (e.g.
+
+ sans-serif, serif, monospace, display, handwriting). Omitted for custom fonts
+ when unknown.
+ """
+
+ display_name: Optional[str] = FieldInfo(alias="displayName", default=None)
+ """
+ Present when type is custom: human-readable name derived from the fontLinks key
+ (strip build/hash suffixes, split camelCase / PascalCase, normalize separators).
+ Google entries omit this.
+ """
+
+
class StyleguideShadows(BaseModel):
"""Shadow styles used on the website"""
@@ -371,6 +396,13 @@ class Styleguide(BaseModel):
element_spacing: StyleguideElementSpacing = FieldInfo(alias="elementSpacing")
"""Spacing system used on the website"""
+ font_links: Dict[str, StyleguideFontLinks] = FieldInfo(alias="fontLinks")
+ """
+ Font assets keyed by family name as it appears in fontFamily/fontFallbacks
+ (non-generic names only). Clients match typography.fontFamily / fontWeight or
+ button styles to pick a file URL from files.
+ """
+
mode: Literal["light", "dark"]
"""The primary color mode of the website design"""
@@ -381,7 +413,7 @@ class Styleguide(BaseModel):
"""Typography styles used on the website"""
-class StyleExtractStyleguideResponse(BaseModel):
+class WebExtractStyleguideResponse(BaseModel):
code: Optional[int] = None
"""HTTP status code"""
diff --git a/src/context/dev/types/web_screenshot_params.py b/src/context/dev/types/web_screenshot_params.py
index d278e90..c6de68a 100644
--- a/src/context/dev/types/web_screenshot_params.py
+++ b/src/context/dev/types/web_screenshot_params.py
@@ -14,13 +14,14 @@ class WebScreenshotParams(TypedDict, total=False):
"""
A specific URL to screenshot directly, bypassing domain resolution (e.g.,
'https://example.com/pricing'). When provided, the screenshot is taken of this
- exact URL.
+ exact URL. You must provide either 'domain' or 'directUrl', but not both.
"""
domain: str
"""Domain name to take screenshot of (e.g., 'example.com', 'google.com').
- The domain will be automatically normalized and validated.
+ The domain will be automatically normalized and validated. You must provide
+ either 'domain' or 'directUrl', but not both.
"""
full_screenshot: Annotated[Literal["true", "false"], PropertyInfo(alias="fullScreenshot")]
diff --git a/src/context/dev/types/web_web_scrape_html_params.py b/src/context/dev/types/web_web_scrape_html_params.py
index 1847d07..d184801 100644
--- a/src/context/dev/types/web_web_scrape_html_params.py
+++ b/src/context/dev/types/web_web_scrape_html_params.py
@@ -2,7 +2,9 @@
from __future__ import annotations
-from typing_extensions import Required, TypedDict
+from typing_extensions import Required, Annotated, TypedDict
+
+from .._utils import PropertyInfo
__all__ = ["WebWebScrapeHTMLParams"]
@@ -10,3 +12,10 @@
class WebWebScrapeHTMLParams(TypedDict, total=False):
url: Required[str]
"""Full URL to scrape (must include http:// or https:// protocol)"""
+
+ max_age_ms: Annotated[int, PropertyInfo(alias="maxAgeMs")]
+ """
+ Return a cached result if a prior scrape for the same parameters exists and is
+ younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
+ omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
+ """
diff --git a/src/context/dev/types/web_web_scrape_md_params.py b/src/context/dev/types/web_web_scrape_md_params.py
index d556da1..8cc9333 100644
--- a/src/context/dev/types/web_web_scrape_md_params.py
+++ b/src/context/dev/types/web_web_scrape_md_params.py
@@ -12,7 +12,7 @@
class WebWebScrapeMdParams(TypedDict, total=False):
url: Required[str]
"""
- Full URL to scrape and convert to markdown (must include http:// or https://
+ Full URL to scrape into LLM usable Markdown (must include http:// or https://
protocol)
"""
@@ -22,6 +22,13 @@ class WebWebScrapeMdParams(TypedDict, total=False):
include_links: Annotated[bool, PropertyInfo(alias="includeLinks")]
"""Preserve hyperlinks in Markdown output"""
+ max_age_ms: Annotated[int, PropertyInfo(alias="maxAgeMs")]
+ """
+ Return a cached result if a prior scrape for the same parameters exists and is
+ younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
+ omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
+ """
+
shorten_base64_images: Annotated[bool, PropertyInfo(alias="shortenBase64Images")]
"""Shorten base64-encoded image data in the Markdown output"""
diff --git a/src/context/dev/types/web_web_scrape_sitemap_params.py b/src/context/dev/types/web_web_scrape_sitemap_params.py
index f78f498..4a1c766 100644
--- a/src/context/dev/types/web_web_scrape_sitemap_params.py
+++ b/src/context/dev/types/web_web_scrape_sitemap_params.py
@@ -11,10 +11,7 @@
class WebWebScrapeSitemapParams(TypedDict, total=False):
domain: Required[str]
- """Domain name to crawl sitemaps for (e.g., 'example.com').
-
- The domain will be automatically normalized and validated.
- """
+ """Domain to build a sitemap for"""
max_links: Annotated[int, PropertyInfo(alias="maxLinks")]
"""Maximum number of links to return from the sitemap crawl.
diff --git a/tests/api_resources/test_style.py b/tests/api_resources/test_style.py
deleted file mode 100644
index 8c9f16a..0000000
--- a/tests/api_resources/test_style.py
+++ /dev/null
@@ -1,189 +0,0 @@
-# File generated from our OpenAPI spec by Stainless. See CONTRIBUTING.md for details.
-
-from __future__ import annotations
-
-import os
-from typing import Any, cast
-
-import pytest
-
-from context.dev import ContextDev, AsyncContextDev
-from tests.utils import assert_matches_type
-from context.dev.types import (
- StyleExtractFontsResponse,
- StyleExtractStyleguideResponse,
-)
-
-base_url = os.environ.get("TEST_API_BASE_URL", "http://127.0.0.1:4010")
-
-
-class TestStyle:
- parametrize = pytest.mark.parametrize("client", [False, True], indirect=True, ids=["loose", "strict"])
-
- @pytest.mark.skip(reason="Mock server tests are disabled")
- @parametrize
- def test_method_extract_fonts(self, client: ContextDev) -> None:
- style = client.style.extract_fonts(
- domain="domain",
- )
- assert_matches_type(StyleExtractFontsResponse, style, path=["response"])
-
- @pytest.mark.skip(reason="Mock server tests are disabled")
- @parametrize
- def test_method_extract_fonts_with_all_params(self, client: ContextDev) -> None:
- style = client.style.extract_fonts(
- domain="domain",
- timeout_ms=1000,
- )
- assert_matches_type(StyleExtractFontsResponse, style, path=["response"])
-
- @pytest.mark.skip(reason="Mock server tests are disabled")
- @parametrize
- def test_raw_response_extract_fonts(self, client: ContextDev) -> None:
- response = client.style.with_raw_response.extract_fonts(
- domain="domain",
- )
-
- assert response.is_closed is True
- assert response.http_request.headers.get("X-Stainless-Lang") == "python"
- style = response.parse()
- assert_matches_type(StyleExtractFontsResponse, style, path=["response"])
-
- @pytest.mark.skip(reason="Mock server tests are disabled")
- @parametrize
- def test_streaming_response_extract_fonts(self, client: ContextDev) -> None:
- with client.style.with_streaming_response.extract_fonts(
- domain="domain",
- ) as response:
- assert not response.is_closed
- assert response.http_request.headers.get("X-Stainless-Lang") == "python"
-
- style = response.parse()
- assert_matches_type(StyleExtractFontsResponse, style, path=["response"])
-
- assert cast(Any, response.is_closed) is True
-
- @pytest.mark.skip(reason="Mock server tests are disabled")
- @parametrize
- def test_method_extract_styleguide(self, client: ContextDev) -> None:
- style = client.style.extract_styleguide()
- assert_matches_type(StyleExtractStyleguideResponse, style, path=["response"])
-
- @pytest.mark.skip(reason="Mock server tests are disabled")
- @parametrize
- def test_method_extract_styleguide_with_all_params(self, client: ContextDev) -> None:
- style = client.style.extract_styleguide(
- direct_url="https://example.com",
- domain="domain",
- timeout_ms=1000,
- )
- assert_matches_type(StyleExtractStyleguideResponse, style, path=["response"])
-
- @pytest.mark.skip(reason="Mock server tests are disabled")
- @parametrize
- def test_raw_response_extract_styleguide(self, client: ContextDev) -> None:
- response = client.style.with_raw_response.extract_styleguide()
-
- assert response.is_closed is True
- assert response.http_request.headers.get("X-Stainless-Lang") == "python"
- style = response.parse()
- assert_matches_type(StyleExtractStyleguideResponse, style, path=["response"])
-
- @pytest.mark.skip(reason="Mock server tests are disabled")
- @parametrize
- def test_streaming_response_extract_styleguide(self, client: ContextDev) -> None:
- with client.style.with_streaming_response.extract_styleguide() as response:
- assert not response.is_closed
- assert response.http_request.headers.get("X-Stainless-Lang") == "python"
-
- style = response.parse()
- assert_matches_type(StyleExtractStyleguideResponse, style, path=["response"])
-
- assert cast(Any, response.is_closed) is True
-
-
-class TestAsyncStyle:
- parametrize = pytest.mark.parametrize(
- "async_client", [False, True, {"http_client": "aiohttp"}], indirect=True, ids=["loose", "strict", "aiohttp"]
- )
-
- @pytest.mark.skip(reason="Mock server tests are disabled")
- @parametrize
- async def test_method_extract_fonts(self, async_client: AsyncContextDev) -> None:
- style = await async_client.style.extract_fonts(
- domain="domain",
- )
- assert_matches_type(StyleExtractFontsResponse, style, path=["response"])
-
- @pytest.mark.skip(reason="Mock server tests are disabled")
- @parametrize
- async def test_method_extract_fonts_with_all_params(self, async_client: AsyncContextDev) -> None:
- style = await async_client.style.extract_fonts(
- domain="domain",
- timeout_ms=1000,
- )
- assert_matches_type(StyleExtractFontsResponse, style, path=["response"])
-
- @pytest.mark.skip(reason="Mock server tests are disabled")
- @parametrize
- async def test_raw_response_extract_fonts(self, async_client: AsyncContextDev) -> None:
- response = await async_client.style.with_raw_response.extract_fonts(
- domain="domain",
- )
-
- assert response.is_closed is True
- assert response.http_request.headers.get("X-Stainless-Lang") == "python"
- style = await response.parse()
- assert_matches_type(StyleExtractFontsResponse, style, path=["response"])
-
- @pytest.mark.skip(reason="Mock server tests are disabled")
- @parametrize
- async def test_streaming_response_extract_fonts(self, async_client: AsyncContextDev) -> None:
- async with async_client.style.with_streaming_response.extract_fonts(
- domain="domain",
- ) as response:
- assert not response.is_closed
- assert response.http_request.headers.get("X-Stainless-Lang") == "python"
-
- style = await response.parse()
- assert_matches_type(StyleExtractFontsResponse, style, path=["response"])
-
- assert cast(Any, response.is_closed) is True
-
- @pytest.mark.skip(reason="Mock server tests are disabled")
- @parametrize
- async def test_method_extract_styleguide(self, async_client: AsyncContextDev) -> None:
- style = await async_client.style.extract_styleguide()
- assert_matches_type(StyleExtractStyleguideResponse, style, path=["response"])
-
- @pytest.mark.skip(reason="Mock server tests are disabled")
- @parametrize
- async def test_method_extract_styleguide_with_all_params(self, async_client: AsyncContextDev) -> None:
- style = await async_client.style.extract_styleguide(
- direct_url="https://example.com",
- domain="domain",
- timeout_ms=1000,
- )
- assert_matches_type(StyleExtractStyleguideResponse, style, path=["response"])
-
- @pytest.mark.skip(reason="Mock server tests are disabled")
- @parametrize
- async def test_raw_response_extract_styleguide(self, async_client: AsyncContextDev) -> None:
- response = await async_client.style.with_raw_response.extract_styleguide()
-
- assert response.is_closed is True
- assert response.http_request.headers.get("X-Stainless-Lang") == "python"
- style = await response.parse()
- assert_matches_type(StyleExtractStyleguideResponse, style, path=["response"])
-
- @pytest.mark.skip(reason="Mock server tests are disabled")
- @parametrize
- async def test_streaming_response_extract_styleguide(self, async_client: AsyncContextDev) -> None:
- async with async_client.style.with_streaming_response.extract_styleguide() as response:
- assert not response.is_closed
- assert response.http_request.headers.get("X-Stainless-Lang") == "python"
-
- style = await response.parse()
- assert_matches_type(StyleExtractStyleguideResponse, style, path=["response"])
-
- assert cast(Any, response.is_closed) is True
diff --git a/tests/api_resources/test_web.py b/tests/api_resources/test_web.py
index 2d8a511..6917533 100644
--- a/tests/api_resources/test_web.py
+++ b/tests/api_resources/test_web.py
@@ -13,9 +13,11 @@
WebScreenshotResponse,
WebWebCrawlMdResponse,
WebWebScrapeMdResponse,
+ WebExtractFontsResponse,
WebWebScrapeHTMLResponse,
WebWebScrapeImagesResponse,
WebWebScrapeSitemapResponse,
+ WebExtractStyleguideResponse,
)
base_url = os.environ.get("TEST_API_BASE_URL", "http://127.0.0.1:4010")
@@ -24,6 +26,82 @@
class TestWeb:
parametrize = pytest.mark.parametrize("client", [False, True], indirect=True, ids=["loose", "strict"])
+ @pytest.mark.skip(reason="Mock server tests are disabled")
+ @parametrize
+ def test_method_extract_fonts(self, client: ContextDev) -> None:
+ web = client.web.extract_fonts()
+ assert_matches_type(WebExtractFontsResponse, web, path=["response"])
+
+ @pytest.mark.skip(reason="Mock server tests are disabled")
+ @parametrize
+ def test_method_extract_fonts_with_all_params(self, client: ContextDev) -> None:
+ web = client.web.extract_fonts(
+ direct_url="https://example.com",
+ domain="domain",
+ timeout_ms=1000,
+ )
+ assert_matches_type(WebExtractFontsResponse, web, path=["response"])
+
+ @pytest.mark.skip(reason="Mock server tests are disabled")
+ @parametrize
+ def test_raw_response_extract_fonts(self, client: ContextDev) -> None:
+ response = client.web.with_raw_response.extract_fonts()
+
+ assert response.is_closed is True
+ assert response.http_request.headers.get("X-Stainless-Lang") == "python"
+ web = response.parse()
+ assert_matches_type(WebExtractFontsResponse, web, path=["response"])
+
+ @pytest.mark.skip(reason="Mock server tests are disabled")
+ @parametrize
+ def test_streaming_response_extract_fonts(self, client: ContextDev) -> None:
+ with client.web.with_streaming_response.extract_fonts() as response:
+ assert not response.is_closed
+ assert response.http_request.headers.get("X-Stainless-Lang") == "python"
+
+ web = response.parse()
+ assert_matches_type(WebExtractFontsResponse, web, path=["response"])
+
+ assert cast(Any, response.is_closed) is True
+
+ @pytest.mark.skip(reason="Mock server tests are disabled")
+ @parametrize
+ def test_method_extract_styleguide(self, client: ContextDev) -> None:
+ web = client.web.extract_styleguide()
+ assert_matches_type(WebExtractStyleguideResponse, web, path=["response"])
+
+ @pytest.mark.skip(reason="Mock server tests are disabled")
+ @parametrize
+ def test_method_extract_styleguide_with_all_params(self, client: ContextDev) -> None:
+ web = client.web.extract_styleguide(
+ direct_url="https://example.com",
+ domain="domain",
+ timeout_ms=1000,
+ )
+ assert_matches_type(WebExtractStyleguideResponse, web, path=["response"])
+
+ @pytest.mark.skip(reason="Mock server tests are disabled")
+ @parametrize
+ def test_raw_response_extract_styleguide(self, client: ContextDev) -> None:
+ response = client.web.with_raw_response.extract_styleguide()
+
+ assert response.is_closed is True
+ assert response.http_request.headers.get("X-Stainless-Lang") == "python"
+ web = response.parse()
+ assert_matches_type(WebExtractStyleguideResponse, web, path=["response"])
+
+ @pytest.mark.skip(reason="Mock server tests are disabled")
+ @parametrize
+ def test_streaming_response_extract_styleguide(self, client: ContextDev) -> None:
+ with client.web.with_streaming_response.extract_styleguide() as response:
+ assert not response.is_closed
+ assert response.http_request.headers.get("X-Stainless-Lang") == "python"
+
+ web = response.parse()
+ assert_matches_type(WebExtractStyleguideResponse, web, path=["response"])
+
+ assert cast(Any, response.is_closed) is True
+
@pytest.mark.skip(reason="Mock server tests are disabled")
@parametrize
def test_method_screenshot(self, client: ContextDev) -> None:
@@ -122,6 +200,15 @@ def test_method_web_scrape_html(self, client: ContextDev) -> None:
)
assert_matches_type(WebWebScrapeHTMLResponse, web, path=["response"])
+ @pytest.mark.skip(reason="Mock server tests are disabled")
+ @parametrize
+ def test_method_web_scrape_html_with_all_params(self, client: ContextDev) -> None:
+ web = client.web.web_scrape_html(
+ url="https://example.com",
+ max_age_ms=0,
+ )
+ assert_matches_type(WebWebScrapeHTMLResponse, web, path=["response"])
+
@pytest.mark.skip(reason="Mock server tests are disabled")
@parametrize
def test_raw_response_web_scrape_html(self, client: ContextDev) -> None:
@@ -197,6 +284,7 @@ def test_method_web_scrape_md_with_all_params(self, client: ContextDev) -> None:
url="https://example.com",
include_images=True,
include_links=True,
+ max_age_ms=0,
shorten_base64_images=True,
use_main_content_only=True,
)
@@ -277,6 +365,82 @@ class TestAsyncWeb:
"async_client", [False, True, {"http_client": "aiohttp"}], indirect=True, ids=["loose", "strict", "aiohttp"]
)
+ @pytest.mark.skip(reason="Mock server tests are disabled")
+ @parametrize
+ async def test_method_extract_fonts(self, async_client: AsyncContextDev) -> None:
+ web = await async_client.web.extract_fonts()
+ assert_matches_type(WebExtractFontsResponse, web, path=["response"])
+
+ @pytest.mark.skip(reason="Mock server tests are disabled")
+ @parametrize
+ async def test_method_extract_fonts_with_all_params(self, async_client: AsyncContextDev) -> None:
+ web = await async_client.web.extract_fonts(
+ direct_url="https://example.com",
+ domain="domain",
+ timeout_ms=1000,
+ )
+ assert_matches_type(WebExtractFontsResponse, web, path=["response"])
+
+ @pytest.mark.skip(reason="Mock server tests are disabled")
+ @parametrize
+ async def test_raw_response_extract_fonts(self, async_client: AsyncContextDev) -> None:
+ response = await async_client.web.with_raw_response.extract_fonts()
+
+ assert response.is_closed is True
+ assert response.http_request.headers.get("X-Stainless-Lang") == "python"
+ web = await response.parse()
+ assert_matches_type(WebExtractFontsResponse, web, path=["response"])
+
+ @pytest.mark.skip(reason="Mock server tests are disabled")
+ @parametrize
+ async def test_streaming_response_extract_fonts(self, async_client: AsyncContextDev) -> None:
+ async with async_client.web.with_streaming_response.extract_fonts() as response:
+ assert not response.is_closed
+ assert response.http_request.headers.get("X-Stainless-Lang") == "python"
+
+ web = await response.parse()
+ assert_matches_type(WebExtractFontsResponse, web, path=["response"])
+
+ assert cast(Any, response.is_closed) is True
+
+ @pytest.mark.skip(reason="Mock server tests are disabled")
+ @parametrize
+ async def test_method_extract_styleguide(self, async_client: AsyncContextDev) -> None:
+ web = await async_client.web.extract_styleguide()
+ assert_matches_type(WebExtractStyleguideResponse, web, path=["response"])
+
+ @pytest.mark.skip(reason="Mock server tests are disabled")
+ @parametrize
+ async def test_method_extract_styleguide_with_all_params(self, async_client: AsyncContextDev) -> None:
+ web = await async_client.web.extract_styleguide(
+ direct_url="https://example.com",
+ domain="domain",
+ timeout_ms=1000,
+ )
+ assert_matches_type(WebExtractStyleguideResponse, web, path=["response"])
+
+ @pytest.mark.skip(reason="Mock server tests are disabled")
+ @parametrize
+ async def test_raw_response_extract_styleguide(self, async_client: AsyncContextDev) -> None:
+ response = await async_client.web.with_raw_response.extract_styleguide()
+
+ assert response.is_closed is True
+ assert response.http_request.headers.get("X-Stainless-Lang") == "python"
+ web = await response.parse()
+ assert_matches_type(WebExtractStyleguideResponse, web, path=["response"])
+
+ @pytest.mark.skip(reason="Mock server tests are disabled")
+ @parametrize
+ async def test_streaming_response_extract_styleguide(self, async_client: AsyncContextDev) -> None:
+ async with async_client.web.with_streaming_response.extract_styleguide() as response:
+ assert not response.is_closed
+ assert response.http_request.headers.get("X-Stainless-Lang") == "python"
+
+ web = await response.parse()
+ assert_matches_type(WebExtractStyleguideResponse, web, path=["response"])
+
+ assert cast(Any, response.is_closed) is True
+
@pytest.mark.skip(reason="Mock server tests are disabled")
@parametrize
async def test_method_screenshot(self, async_client: AsyncContextDev) -> None:
@@ -375,6 +539,15 @@ async def test_method_web_scrape_html(self, async_client: AsyncContextDev) -> No
)
assert_matches_type(WebWebScrapeHTMLResponse, web, path=["response"])
+ @pytest.mark.skip(reason="Mock server tests are disabled")
+ @parametrize
+ async def test_method_web_scrape_html_with_all_params(self, async_client: AsyncContextDev) -> None:
+ web = await async_client.web.web_scrape_html(
+ url="https://example.com",
+ max_age_ms=0,
+ )
+ assert_matches_type(WebWebScrapeHTMLResponse, web, path=["response"])
+
@pytest.mark.skip(reason="Mock server tests are disabled")
@parametrize
async def test_raw_response_web_scrape_html(self, async_client: AsyncContextDev) -> None:
@@ -450,6 +623,7 @@ async def test_method_web_scrape_md_with_all_params(self, async_client: AsyncCon
url="https://example.com",
include_images=True,
include_links=True,
+ max_age_ms=0,
shorten_base64_images=True,
use_main_content_only=True,
)
diff --git a/tests/test_deepcopy.py b/tests/test_deepcopy.py
deleted file mode 100644
index ec2cbfc..0000000
--- a/tests/test_deepcopy.py
+++ /dev/null
@@ -1,58 +0,0 @@
-from context.dev._utils import deepcopy_minimal
-
-
-def assert_different_identities(obj1: object, obj2: object) -> None:
- assert obj1 == obj2
- assert id(obj1) != id(obj2)
-
-
-def test_simple_dict() -> None:
- obj1 = {"foo": "bar"}
- obj2 = deepcopy_minimal(obj1)
- assert_different_identities(obj1, obj2)
-
-
-def test_nested_dict() -> None:
- obj1 = {"foo": {"bar": True}}
- obj2 = deepcopy_minimal(obj1)
- assert_different_identities(obj1, obj2)
- assert_different_identities(obj1["foo"], obj2["foo"])
-
-
-def test_complex_nested_dict() -> None:
- obj1 = {"foo": {"bar": [{"hello": "world"}]}}
- obj2 = deepcopy_minimal(obj1)
- assert_different_identities(obj1, obj2)
- assert_different_identities(obj1["foo"], obj2["foo"])
- assert_different_identities(obj1["foo"]["bar"], obj2["foo"]["bar"])
- assert_different_identities(obj1["foo"]["bar"][0], obj2["foo"]["bar"][0])
-
-
-def test_simple_list() -> None:
- obj1 = ["a", "b", "c"]
- obj2 = deepcopy_minimal(obj1)
- assert_different_identities(obj1, obj2)
-
-
-def test_nested_list() -> None:
- obj1 = ["a", [1, 2, 3]]
- obj2 = deepcopy_minimal(obj1)
- assert_different_identities(obj1, obj2)
- assert_different_identities(obj1[1], obj2[1])
-
-
-class MyObject: ...
-
-
-def test_ignores_other_types() -> None:
- # custom classes
- my_obj = MyObject()
- obj1 = {"foo": my_obj}
- obj2 = deepcopy_minimal(obj1)
- assert_different_identities(obj1, obj2)
- assert obj1["foo"] is my_obj
-
- # tuples
- obj3 = ("a", "b")
- obj4 = deepcopy_minimal(obj3)
- assert obj3 is obj4
diff --git a/tests/test_extract_files.py b/tests/test_extract_files.py
index e7585ee..34b6253 100644
--- a/tests/test_extract_files.py
+++ b/tests/test_extract_files.py
@@ -35,6 +35,15 @@ def test_multiple_files() -> None:
assert query == {"documents": [{}, {}]}
+def test_top_level_file_array() -> None:
+ query = {"files": [b"file one", b"file two"], "title": "hello"}
+ assert extract_files(query, paths=[["files", ""]]) == [
+ ("files[]", b"file one"),
+ ("files[]", b"file two"),
+ ]
+ assert query == {"title": "hello"}
+
+
@pytest.mark.parametrize(
"query,paths,expected",
[
diff --git a/tests/test_files.py b/tests/test_files.py
index ede2488..e9e3a2d 100644
--- a/tests/test_files.py
+++ b/tests/test_files.py
@@ -4,7 +4,8 @@
import pytest
from dirty_equals import IsDict, IsList, IsBytes, IsTuple
-from context.dev._files import to_httpx_files, async_to_httpx_files
+from context.dev._files import to_httpx_files, deepcopy_with_paths, async_to_httpx_files
+from context.dev._utils import extract_files
readme_path = Path(__file__).parent.parent.joinpath("README.md")
@@ -49,3 +50,99 @@ def test_string_not_allowed() -> None:
"file": "foo", # type: ignore
}
)
+
+
+def assert_different_identities(obj1: object, obj2: object) -> None:
+ assert obj1 == obj2
+ assert obj1 is not obj2
+
+
+class TestDeepcopyWithPaths:
+ def test_copies_top_level_dict(self) -> None:
+ original = {"file": b"data", "other": "value"}
+ result = deepcopy_with_paths(original, [["file"]])
+ assert_different_identities(result, original)
+
+ def test_file_value_is_same_reference(self) -> None:
+ file_bytes = b"contents"
+ original = {"file": file_bytes}
+ result = deepcopy_with_paths(original, [["file"]])
+ assert_different_identities(result, original)
+ assert result["file"] is file_bytes
+
+ def test_list_popped_wholesale(self) -> None:
+ files = [b"f1", b"f2"]
+ original = {"files": files, "title": "t"}
+ result = deepcopy_with_paths(original, [["files", ""]])
+ assert_different_identities(result, original)
+ result_files = result["files"]
+ assert isinstance(result_files, list)
+ assert_different_identities(result_files, files)
+
+ def test_nested_array_path_copies_list_and_elements(self) -> None:
+ elem1 = {"file": b"f1", "extra": 1}
+ elem2 = {"file": b"f2", "extra": 2}
+ original = {"items": [elem1, elem2]}
+ result = deepcopy_with_paths(original, [["items", "", "file"]])
+ assert_different_identities(result, original)
+ result_items = result["items"]
+ assert isinstance(result_items, list)
+ assert_different_identities(result_items, original["items"])
+ assert_different_identities(result_items[0], elem1)
+ assert_different_identities(result_items[1], elem2)
+
+ def test_empty_paths_returns_same_object(self) -> None:
+ original = {"foo": "bar"}
+ result = deepcopy_with_paths(original, [])
+ assert result is original
+
+ def test_multiple_paths(self) -> None:
+ f1 = b"file1"
+ f2 = b"file2"
+ original = {"a": f1, "b": f2, "c": "unchanged"}
+ result = deepcopy_with_paths(original, [["a"], ["b"]])
+ assert_different_identities(result, original)
+ assert result["a"] is f1
+ assert result["b"] is f2
+ assert result["c"] is original["c"]
+
+ def test_extract_files_does_not_mutate_original_top_level(self) -> None:
+ file_bytes = b"contents"
+ original = {"file": file_bytes, "other": "value"}
+
+ copied = deepcopy_with_paths(original, [["file"]])
+ extracted = extract_files(copied, paths=[["file"]])
+
+ assert extracted == [("file", file_bytes)]
+ assert original == {"file": file_bytes, "other": "value"}
+ assert copied == {"other": "value"}
+
+ def test_extract_files_does_not_mutate_original_nested_array_path(self) -> None:
+ file1 = b"f1"
+ file2 = b"f2"
+ original = {
+ "items": [
+ {"file": file1, "extra": 1},
+ {"file": file2, "extra": 2},
+ ],
+ "title": "example",
+ }
+
+ copied = deepcopy_with_paths(original, [["items", "", "file"]])
+ extracted = extract_files(copied, paths=[["items", "", "file"]])
+
+ assert extracted == [("items[][file]", file1), ("items[][file]", file2)]
+ assert original == {
+ "items": [
+ {"file": file1, "extra": 1},
+ {"file": file2, "extra": 2},
+ ],
+ "title": "example",
+ }
+ assert copied == {
+ "items": [
+ {"extra": 1},
+ {"extra": 2},
+ ],
+ "title": "example",
+ }