Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .release-please-manifest.json
Original file line number Diff line number Diff line change
@@ -1,3 +1,3 @@
{
".": "2.3.0"
".": "2.4.0"
}
4 changes: 2 additions & 2 deletions .stats.yml
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
configured_endpoints: 30
openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev/context.dev-c90f13e3cf4b9ae6a1d8f22c0e7285ab12d412d3a659a51a247a97adc7f08121.yml
openapi_spec_hash: bf7879be38ebf3397a939c80b261a47a
openapi_spec_url: https://storage.googleapis.com/stainless-sdk-openapi-specs/context-dev/context.dev-fb56935a194e69348fecd985f7cf8b249795b46af2fb32f6c5c8ef648cf10c15.yml
openapi_spec_hash: 7260a560474283b7ad6ac5d426058ac9
config_hash: daabb160675d86b354711da1e77e5129
8 changes: 8 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
@@ -1,5 +1,13 @@
# Changelog

## 2.4.0 (2026-07-12)

Full Changelog: [v2.3.0...v2.4.0](https://github.com/context-dot-dev/context-python-sdk/compare/v2.3.0...v2.4.0)

### Features

* **api:** api update ([1aa99f7](https://github.com/context-dot-dev/context-python-sdk/commit/1aa99f71f83593bb9822fd5aa8d84f80f932971f))

## 2.3.0 (2026-07-12)

Full Changelog: [v2.2.0...v2.3.0](https://github.com/context-dot-dev/context-python-sdk/compare/v2.2.0...v2.3.0)
Expand Down
10 changes: 2 additions & 8 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -133,14 +133,8 @@ from context.dev import ContextDev

client = ContextDev()

response = client.web.extract(
schema={
"type": "bar",
"properties": "bar",
"required": "bar",
"additionalProperties": "bar",
},
url="https://example.com",
response = client.parse.handle(
body=b"Example data",
pdf={},
)
print(response.pdf)
Expand Down
2 changes: 1 addition & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
[project]
name = "context.dev"
version = "2.3.0"
version = "2.4.0"
description = "The official Python library for the context.dev API"
dynamic = ["readme"]
license = "Apache-2.0"
Expand Down
2 changes: 1 addition & 1 deletion src/context/dev/_version.py
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
# File generated from our OpenAPI spec by Stainless. See CONTRIBUTING.md for details.

__title__ = "context.dev"
__version__ = "2.3.0" # x-release-please-version
__version__ = "2.4.0" # x-release-please-version
229 changes: 179 additions & 50 deletions src/context/dev/resources/parse.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@
from __future__ import annotations

import os
from typing_extensions import Literal

import httpx

Expand Down Expand Up @@ -59,14 +60,83 @@ def handle(
self,
body: FileContent | BinaryTypes,
*,
base_url: str | Omit = omit,
extension: str | Omit = omit,
filename: str | Omit = omit,
extension: Literal[
"txt",
"text",
"md",
"markdown",
"html",
"htm",
"xhtml",
"xml",
"rss",
"atom",
"csv",
"tsv",
"yaml",
"yml",
"py",
"java",
"js",
"jsx",
"mjs",
"cjs",
"json",
"jsonl",
"ndjson",
"php",
"sh",
"bash",
"zsh",
"fish",
"rb",
"ts",
"tsx",
"rtf",
"srt",
"css",
"scss",
"less",
"styl",
"sass",
"svg",
"pdf",
"docx",
"doc",
"xlsx",
"xlsm",
"xlsb",
"xltx",
"xltm",
"xls",
"pptx",
"pptm",
"ppsx",
"ppsm",
"potx",
"potm",
"ppt",
"pps",
"pot",
"jpg",
"jpeg",
"jpe",
"png",
"gif",
"bmp",
"tiff",
"tif",
"webp",
"ppm",
"pbm",
"pgm",
"pnm",
]
| Omit = omit,
include_images: bool | Omit = omit,
include_links: bool | Omit = omit,
ocr: bool | Omit = omit,
pdf_end: int | Omit = omit,
pdf_start: int | Omit = omit,
pdf: parse_handle_params.Pdf | Omit = omit,
shorten_base64_images: bool | Omit = omit,
use_main_content_only: bool | Omit = omit,
# Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
Expand All @@ -78,30 +148,28 @@ def handle(
) -> ParseHandleResponse:
"""
Converts raw text, source code, web/data, PDF, Microsoft Office, and image bytes
into LLM-usable Markdown.
into LLM-usable Markdown. The base request costs 1 credit. When OCR runs
(requires ocr=true), the entire call costs 5 credits; ocr=true requests where no
OCR ends up running still cost 1 credit.

Args:
base_url: Optional HTTP(S) source document URL used to resolve relative links and image
references. Relative references remain relative when omitted.

extension: Optional file extension hint, such as pdf, docx, xlsx, pptx, html, json, csv,
md, py, rtf, jpg, png, or txt.

filename: Optional filename hint used to infer the extension when extension is omitted.
extension: Optional file extension hint. Case-insensitive; a leading dot is accepted (e.g.
".pdf").

include_images: Include image references in Markdown output

include_links: Preserve hyperlinks in Markdown output

ocr: When true for PDF inputs, detect and OCR images embedded in the selected pages,
inserting recognized text at each image's position in page reading order while
preserving the PDF text layer. pdfStart/pdfEnd limit the inclusive page range.
This is separate from automatic scanned-PDF OCR fallback.

pdf_end: Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
Must be greater than or equal to pdfStart when both are provided.
ocr: Gates all OCR. When true, PDFs get embedded-image OCR (recognized text inserted
at each image's position in page reading order, preserving the text layer;
pdf.start/pdf.end limit the page range), scanned PDFs with no text layer get
full-document OCR, and raster images get their visible text transcribed. When
false, no OCR runs: scanned PDFs may yield no content and images return only
format/dimension metadata. Calls where OCR actually runs cost 5 credits instead
of 1.

pdf_start: First 1-based PDF page to parse. When omitted, parsing starts at the first page.
pdf: PDF page-range controls. Use start/end to limit parsing (and OCR when ocr=true)
to an inclusive 1-based page range.

shorten_base64_images: Shorten base64-encoded image data in the Markdown output

Expand All @@ -126,14 +194,11 @@ def handle(
timeout=timeout,
query=maybe_transform(
{
"base_url": base_url,
"extension": extension,
"filename": filename,
"include_images": include_images,
"include_links": include_links,
"ocr": ocr,
"pdf_end": pdf_end,
"pdf_start": pdf_start,
"pdf": pdf,
"shorten_base64_images": shorten_base64_images,
"use_main_content_only": use_main_content_only,
},
Expand Down Expand Up @@ -168,14 +233,83 @@ async def handle(
self,
body: FileContent | AsyncBinaryTypes,
*,
base_url: str | Omit = omit,
extension: str | Omit = omit,
filename: str | Omit = omit,
extension: Literal[
"txt",
"text",
"md",
"markdown",
"html",
"htm",
"xhtml",
"xml",
"rss",
"atom",
"csv",
"tsv",
"yaml",
"yml",
"py",
"java",
"js",
"jsx",
"mjs",
"cjs",
"json",
"jsonl",
"ndjson",
"php",
"sh",
"bash",
"zsh",
"fish",
"rb",
"ts",
"tsx",
"rtf",
"srt",
"css",
"scss",
"less",
"styl",
"sass",
"svg",
"pdf",
"docx",
"doc",
"xlsx",
"xlsm",
"xlsb",
"xltx",
"xltm",
"xls",
"pptx",
"pptm",
"ppsx",
"ppsm",
"potx",
"potm",
"ppt",
"pps",
"pot",
"jpg",
"jpeg",
"jpe",
"png",
"gif",
"bmp",
"tiff",
"tif",
"webp",
"ppm",
"pbm",
"pgm",
"pnm",
]
| Omit = omit,
include_images: bool | Omit = omit,
include_links: bool | Omit = omit,
ocr: bool | Omit = omit,
pdf_end: int | Omit = omit,
pdf_start: int | Omit = omit,
pdf: parse_handle_params.Pdf | Omit = omit,
shorten_base64_images: bool | Omit = omit,
use_main_content_only: bool | Omit = omit,
# Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
Expand All @@ -187,30 +321,28 @@ async def handle(
) -> ParseHandleResponse:
"""
Converts raw text, source code, web/data, PDF, Microsoft Office, and image bytes
into LLM-usable Markdown.
into LLM-usable Markdown. The base request costs 1 credit. When OCR runs
(requires ocr=true), the entire call costs 5 credits; ocr=true requests where no
OCR ends up running still cost 1 credit.

Args:
base_url: Optional HTTP(S) source document URL used to resolve relative links and image
references. Relative references remain relative when omitted.

extension: Optional file extension hint, such as pdf, docx, xlsx, pptx, html, json, csv,
md, py, rtf, jpg, png, or txt.

filename: Optional filename hint used to infer the extension when extension is omitted.
extension: Optional file extension hint. Case-insensitive; a leading dot is accepted (e.g.
".pdf").

include_images: Include image references in Markdown output

include_links: Preserve hyperlinks in Markdown output

ocr: When true for PDF inputs, detect and OCR images embedded in the selected pages,
inserting recognized text at each image's position in page reading order while
preserving the PDF text layer. pdfStart/pdfEnd limit the inclusive page range.
This is separate from automatic scanned-PDF OCR fallback.

pdf_end: Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
Must be greater than or equal to pdfStart when both are provided.
ocr: Gates all OCR. When true, PDFs get embedded-image OCR (recognized text inserted
at each image's position in page reading order, preserving the text layer;
pdf.start/pdf.end limit the page range), scanned PDFs with no text layer get
full-document OCR, and raster images get their visible text transcribed. When
false, no OCR runs: scanned PDFs may yield no content and images return only
format/dimension metadata. Calls where OCR actually runs cost 5 credits instead
of 1.

pdf_start: First 1-based PDF page to parse. When omitted, parsing starts at the first page.
pdf: PDF page-range controls. Use start/end to limit parsing (and OCR when ocr=true)
to an inclusive 1-based page range.

shorten_base64_images: Shorten base64-encoded image data in the Markdown output

Expand All @@ -235,14 +367,11 @@ async def handle(
timeout=timeout,
query=await async_maybe_transform(
{
"base_url": base_url,
"extension": extension,
"filename": filename,
"include_images": include_images,
"include_links": include_links,
"ocr": ocr,
"pdf_end": pdf_end,
"pdf_start": pdf_start,
"pdf": pdf,
"shorten_base64_images": shorten_base64_images,
"use_main_content_only": use_main_content_only,
},
Expand Down
Loading
Loading