@@ -254,6 +254,7 @@ def web_crawl_md(
254254 max_age_ms : int | Omit = omit ,
255255 max_depth : int | Omit = omit ,
256256 max_pages : int | Omit = omit ,
257+ parse_pdf : bool | Omit = omit ,
257258 shorten_base64_images : bool | Omit = omit ,
258259 url_regex : str | Omit = omit ,
259260 use_main_content_only : bool | Omit = omit ,
@@ -287,6 +288,10 @@ def web_crawl_md(
287288
288289 max_pages: Maximum number of pages to crawl. Hard cap: 500.
289290
291+ parse_pdf: When true (default), PDF pages are fetched and their text layer is extracted and
292+ converted to Markdown alongside HTML pages. When false, PDF pages are skipped
293+ entirely (not included in results and not counted as failures).
294+
290295 shorten_base64_images: Truncate base64-encoded image data in the Markdown output
291296
292297 url_regex: Regex pattern. Only URLs matching this pattern will be followed and scraped.
@@ -313,6 +318,7 @@ def web_crawl_md(
313318 "max_age_ms" : max_age_ms ,
314319 "max_depth" : max_depth ,
315320 "max_pages" : max_pages ,
321+ "parse_pdf" : parse_pdf ,
316322 "shorten_base64_images" : shorten_base64_images ,
317323 "url_regex" : url_regex ,
318324 "use_main_content_only" : use_main_content_only ,
@@ -330,6 +336,7 @@ def web_scrape_html(
330336 * ,
331337 url : str ,
332338 max_age_ms : int | Omit = omit ,
339+ parse_pdf : bool | Omit = omit ,
333340 # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
334341 # The extra values given here take precedence over values defined on the client or passed to this method.
335342 extra_headers : Headers | None = None ,
@@ -347,6 +354,10 @@ def web_scrape_html(
347354 younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
348355 omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
349356
357+ parse_pdf: When true (default), PDF URLs are fetched and their text layer is extracted and
358+ returned wrapped in <html><pdf>…</pdf></html>. When false, PDF URLs are skipped
359+ and a 400 WEBSITE_ACCESS_ERROR is returned.
360+
350361 extra_headers: Send extra headers
351362
352363 extra_query: Add additional query parameters to the request
@@ -366,6 +377,7 @@ def web_scrape_html(
366377 {
367378 "url" : url ,
368379 "max_age_ms" : max_age_ms ,
380+ "parse_pdf" : parse_pdf ,
369381 },
370382 web_web_scrape_html_params .WebWebScrapeHTMLParams ,
371383 ),
@@ -420,6 +432,7 @@ def web_scrape_md(
420432 include_images : bool | Omit = omit ,
421433 include_links : bool | Omit = omit ,
422434 max_age_ms : int | Omit = omit ,
435+ parse_pdf : bool | Omit = omit ,
423436 shorten_base64_images : bool | Omit = omit ,
424437 use_main_content_only : bool | Omit = omit ,
425438 # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
@@ -444,6 +457,10 @@ def web_scrape_md(
444457 younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
445458 omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
446459
460+ parse_pdf: When true (default), PDF URLs are fetched and their text layer is extracted and
461+ converted to Markdown. When false, PDF URLs are skipped and a 400
462+ WEBSITE_ACCESS_ERROR is returned.
463+
447464 shorten_base64_images: Shorten base64-encoded image data in the Markdown output
448465
449466 use_main_content_only: Extract only the main content of the page, excluding headers, footers, sidebars,
@@ -470,6 +487,7 @@ def web_scrape_md(
470487 "include_images" : include_images ,
471488 "include_links" : include_links ,
472489 "max_age_ms" : max_age_ms ,
490+ "parse_pdf" : parse_pdf ,
473491 "shorten_base64_images" : shorten_base64_images ,
474492 "use_main_content_only" : use_main_content_only ,
475493 },
@@ -747,6 +765,7 @@ async def web_crawl_md(
747765 max_age_ms : int | Omit = omit ,
748766 max_depth : int | Omit = omit ,
749767 max_pages : int | Omit = omit ,
768+ parse_pdf : bool | Omit = omit ,
750769 shorten_base64_images : bool | Omit = omit ,
751770 url_regex : str | Omit = omit ,
752771 use_main_content_only : bool | Omit = omit ,
@@ -780,6 +799,10 @@ async def web_crawl_md(
780799
781800 max_pages: Maximum number of pages to crawl. Hard cap: 500.
782801
802+ parse_pdf: When true (default), PDF pages are fetched and their text layer is extracted and
803+ converted to Markdown alongside HTML pages. When false, PDF pages are skipped
804+ entirely (not included in results and not counted as failures).
805+
783806 shorten_base64_images: Truncate base64-encoded image data in the Markdown output
784807
785808 url_regex: Regex pattern. Only URLs matching this pattern will be followed and scraped.
@@ -806,6 +829,7 @@ async def web_crawl_md(
806829 "max_age_ms" : max_age_ms ,
807830 "max_depth" : max_depth ,
808831 "max_pages" : max_pages ,
832+ "parse_pdf" : parse_pdf ,
809833 "shorten_base64_images" : shorten_base64_images ,
810834 "url_regex" : url_regex ,
811835 "use_main_content_only" : use_main_content_only ,
@@ -823,6 +847,7 @@ async def web_scrape_html(
823847 * ,
824848 url : str ,
825849 max_age_ms : int | Omit = omit ,
850+ parse_pdf : bool | Omit = omit ,
826851 # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
827852 # The extra values given here take precedence over values defined on the client or passed to this method.
828853 extra_headers : Headers | None = None ,
@@ -840,6 +865,10 @@ async def web_scrape_html(
840865 younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
841866 omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
842867
868+ parse_pdf: When true (default), PDF URLs are fetched and their text layer is extracted and
869+ returned wrapped in <html><pdf>…</pdf></html>. When false, PDF URLs are skipped
870+ and a 400 WEBSITE_ACCESS_ERROR is returned.
871+
843872 extra_headers: Send extra headers
844873
845874 extra_query: Add additional query parameters to the request
@@ -859,6 +888,7 @@ async def web_scrape_html(
859888 {
860889 "url" : url ,
861890 "max_age_ms" : max_age_ms ,
891+ "parse_pdf" : parse_pdf ,
862892 },
863893 web_web_scrape_html_params .WebWebScrapeHTMLParams ,
864894 ),
@@ -913,6 +943,7 @@ async def web_scrape_md(
913943 include_images : bool | Omit = omit ,
914944 include_links : bool | Omit = omit ,
915945 max_age_ms : int | Omit = omit ,
946+ parse_pdf : bool | Omit = omit ,
916947 shorten_base64_images : bool | Omit = omit ,
917948 use_main_content_only : bool | Omit = omit ,
918949 # Use the following arguments if you need to pass additional parameters to the API that aren't available via kwargs.
@@ -937,6 +968,10 @@ async def web_scrape_md(
937968 younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
938969 omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
939970
971+ parse_pdf: When true (default), PDF URLs are fetched and their text layer is extracted and
972+ converted to Markdown. When false, PDF URLs are skipped and a 400
973+ WEBSITE_ACCESS_ERROR is returned.
974+
940975 shorten_base64_images: Shorten base64-encoded image data in the Markdown output
941976
942977 use_main_content_only: Extract only the main content of the page, excluding headers, footers, sidebars,
@@ -963,6 +998,7 @@ async def web_scrape_md(
963998 "include_images" : include_images ,
964999 "include_links" : include_links ,
9651000 "max_age_ms" : max_age_ms ,
1001+ "parse_pdf" : parse_pdf ,
9661002 "shorten_base64_images" : shorten_base64_images ,
9671003 "use_main_content_only" : use_main_content_only ,
9681004 },
0 commit comments