Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
97 changes: 76 additions & 21 deletions cycode/cli/apps/scan/code_scanner.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
import os
import time
import zipfile
from platform import platform
from typing import TYPE_CHECKING, Callable, Optional

Expand All @@ -23,6 +24,7 @@
from cycode.cli.files_collector.sca.sca_file_collector import add_sca_dependencies_tree_documents_if_needed
from cycode.cli.files_collector.zip_documents import zip_documents
from cycode.cli.models import CliError, Document, LocalScanResult
from cycode.cli.utils.host_info import is_64bit
from cycode.cli.utils.path_utils import get_absolute_path, get_path_by_os
from cycode.cli.utils.progress_bar import ScanProgressBarSection
from cycode.cli.utils.scan_batch import run_parallel_batched_scan
Expand Down Expand Up @@ -145,6 +147,7 @@ def _get_scan_documents_thread_func(
is_git_diff: bool,
is_commit_range: bool,
scan_parameters: dict,
prezipped: Optional['InMemoryZip'] = None,
) -> Callable[[list[Document]], tuple[str, CliError, LocalScanResult]]:
cycode_client = ctx.obj['client']
scan_type = ctx.obj['scan_type']
Expand All @@ -164,9 +167,14 @@ def _scan_batch_thread_func(batch: list[Document]) -> tuple[str, CliError, Local

should_use_sync_flow = _should_use_sync_flow(command_scan_type, scan_type, sync_option)

# the single ZIP flow already built the archive to check that it fits; don't build it twice
zipped_documents = prezipped

try:
logger.debug('Preparing local files, %s', {'batch_files_count': len(batch)})
zipped_documents = zip_documents(scan_type, batch)
if zipped_documents is None:
logger.debug('Preparing local files, %s', {'batch_files_count': len(batch)})
zipped_documents = zip_documents(scan_type, batch)

zip_file_size = zipped_documents.size
scan_result = _perform_scan(
cycode_client,
Expand All @@ -189,6 +197,9 @@ def _scan_batch_thread_func(batch: list[Document]) -> tuple[str, CliError, Local
except Exception as e:
error = handle_scan_exception(ctx, e, return_exception=True)
error_message = str(e)
finally:
if zipped_documents is not None:
zipped_documents.cleanup()

if local_scan_result:
detections_count = local_scan_result.detections_count
Expand Down Expand Up @@ -225,34 +236,77 @@ def _scan_batch_thread_func(batch: list[Document]) -> tuple[str, CliError, Local
return _scan_batch_thread_func


def _log_selected_upload_mode(mode: str, reason: str, documents_count: int) -> None:
logger.debug(
'Selected upload mode, %s',
{
'mode': mode,
'reason': reason,
'documents_count': documents_count,
'max_files_count': consts.ZIP_MAX_FILES_COUNT,
'zip64_enabled': is_64bit(),
},
)


def _exceeds_non_zip64_files_count(documents_to_scan: list[Document]) -> bool:
"""Whether a single ZIP can't hold all the documents because ZIP64 is unavailable.

Without ZIP64 (32-bit interpreter) the archive is capped at 65,535 entries.
"""
return not is_64bit() and len(documents_to_scan) > consts.ZIP_MAX_FILES_COUNT


def _run_presigned_upload_scan(
scan_batch_thread_func: Callable,
scan_type: str,
ctx: typer.Context,
is_git_diff: bool,
is_commit_range: bool,
scan_parameters: dict,
documents_to_scan: list[Document],
progress_bar: 'BaseProgressBar',
printer: 'ConsolePrinter',
) -> tuple:
try:
# Try to zip all documents as a single batch; ZipTooLargeError raised if it exceeds the scan type's limit
zip_documents(scan_type, documents_to_scan)
# It fits: skip batching and upload everything as one ZIP
scan_type = ctx.obj['scan_type']
documents_count = len(documents_to_scan)

def run_batched() -> tuple:
return run_parallel_batched_scan(
scan_batch_thread_func,
_get_scan_documents_thread_func(ctx, is_git_diff, is_commit_range, scan_parameters),
scan_type,
documents_to_scan,
progress_bar=progress_bar,
skip_batching=True,
)
except custom_exceptions.ZipTooLargeError:

if _exceeds_non_zip64_files_count(documents_to_scan):
# Don't waste time zipping documents we already know won't fit into a single ZIP
_log_selected_upload_mode('batched', 'files_count_exceeds_non_zip64_limit', documents_count)
return run_batched()

zipped_documents = None
try:
# Try to zip all documents as a single batch; ZipTooLargeError raised if it exceeds the scan type's limit
zipped_documents = zip_documents(scan_type, documents_to_scan)
except (custom_exceptions.ZipTooLargeError, zipfile.LargeZipFile):
# LargeZipFile is a safety net: the files count pre-check above should have caught it already
_log_selected_upload_mode('batched', 'zip_too_large', documents_count)
if zipped_documents is not None:
zipped_documents.cleanup()

printer.print_warning(
'The scan is too large to upload as a single file. This may result in corrupted scan results.'
)
return run_parallel_batched_scan(
scan_batch_thread_func,
scan_type,
documents_to_scan,
progress_bar=progress_bar,
)
return run_batched()

# It fits: skip batching and upload everything as one ZIP. The archive we just built is the one
# that gets uploaded, so the scan doesn't pay for compressing every document twice
_log_selected_upload_mode('single_zip', 'fits_single_zip', documents_count)
return run_parallel_batched_scan(
_get_scan_documents_thread_func(ctx, is_git_diff, is_commit_range, scan_parameters, zipped_documents),
scan_type,
documents_to_scan,
progress_bar=progress_bar,
skip_batching=True,
)


def scan_documents(
Expand All @@ -277,18 +331,19 @@ def scan_documents(
)
return

scan_batch_thread_func = _get_scan_documents_thread_func(ctx, is_git_diff, is_commit_range, scan_parameters)

# Presigned single-file upload is async-only; a --sync scan must stay on the batched inline path
# so it never builds one oversized zip to POST synchronously.
should_use_sync_flow = _should_use_sync_flow(ctx.info_name, scan_type, ctx.obj['sync'])
if should_use_presigned_upload(scan_type) and not should_use_sync_flow:
errors, local_scan_results = _run_presigned_upload_scan(
scan_batch_thread_func, scan_type, documents_to_scan, progress_bar, printer
ctx, is_git_diff, is_commit_range, scan_parameters, documents_to_scan, progress_bar, printer
)
else:
errors, local_scan_results = run_parallel_batched_scan(
scan_batch_thread_func, scan_type, documents_to_scan, progress_bar=progress_bar
_get_scan_documents_thread_func(ctx, is_git_diff, is_commit_range, scan_parameters),
scan_type,
documents_to_scan,
progress_bar=progress_bar,
)

try_set_aggregation_report_url_if_needed(ctx, scan_parameters, ctx.obj['client'], scan_type)
Expand Down
3 changes: 3 additions & 0 deletions cycode/cli/apps/scan/commit_range_scanner.py
Original file line number Diff line number Diff line change
Expand Up @@ -214,6 +214,9 @@ def _scan_commit_range_documents(

zip_file_size = from_commit_zipped_documents.size + to_commit_zipped_documents.size

from_commit_zipped_documents.cleanup()
to_commit_zipped_documents.cleanup()

detections_count = relevant_detections_count = 0
if local_scan_result:
detections_count = local_scan_result.detections_count
Expand Down
6 changes: 6 additions & 0 deletions cycode/cli/consts.py
Original file line number Diff line number Diff line change
Expand Up @@ -227,6 +227,12 @@
PRESIGNED_LINK_UPLOADED_ZIP_MAX_SIZE_LIMIT_IN_BYTES = 5 * 1024 * 1024 * 1024 # 5 GB (S3 presigned POST limit)
PRESIGNED_UPLOAD_SCAN_TYPES = {SAST_SCAN_TYPE, SECRET_SCAN_TYPE}

# the non-ZIP64 central directory stores the entry count in 16 bits; ZIP64 (64-bit interpreters) lifts it
ZIP_MAX_FILES_COUNT = 65_535

# the ZIP is built in memory up to this size, and spilled to a temp file beyond it
ZIP_SPOOL_MAX_SIZE_IN_BYTES = 64 * 1024 * 1024

DEFAULT_ZIP_MAX_SIZE_LIMIT_IN_BYTES = 20 * 1024 * 1024
ZIP_MAX_SIZE_LIMIT_IN_BYTES = {
SCA_SCAN_TYPE: 200 * 1024 * 1024,
Expand Down
8 changes: 8 additions & 0 deletions cycode/cli/exceptions/handle_scan_errors.py
Original file line number Diff line number Diff line change
@@ -1,3 +1,4 @@
import zipfile
from typing import Optional

import typer
Expand Down Expand Up @@ -26,6 +27,13 @@ def handle_scan_exception(ctx: typer.Context, err: Exception, *, return_exceptio
'Please try ignoring irrelevant paths using the `cycode ignore --by-path` command '
'and execute the scan again',
),
zipfile.LargeZipFile: CliError(
soft_fail=True,
code='zip_too_large_error',
message='The path you attempted to scan contains too many files to pack into a single archive. '
'Scanning such paths requires a 64-bit Python interpreter. '
'Please try ignoring irrelevant paths using a .cycodeignore file and execute the scan again',
),
custom_exceptions.FileCollectionError: CliError(
soft_fail=False,
code='file_collection_error',
Expand Down
83 changes: 74 additions & 9 deletions cycode/cli/files_collector/models/in_memory_zip.py
Original file line number Diff line number Diff line change
@@ -1,20 +1,48 @@
import shutil
import tempfile
from collections import defaultdict
from io import BytesIO
from os import SEEK_END
from pathlib import Path
from sys import getsizeof
from typing import Optional
from typing import IO, Optional
from zipfile import ZIP_DEFLATED, ZipFile

from cycode.cli import consts
from cycode.cli.user_settings.configuration_manager import ConfigurationManager
from cycode.cli.utils.host_info import is_64bit
from cycode.cli.utils.path_utils import concat_unique_id
from cycode.logger import get_logger

logger = get_logger('ZIP')

_SPOOL_DIRECTORY_NAME = 'tmp'


def _get_spool_directory(configuration_manager: ConfigurationManager) -> Optional[str]:
"""Directory to spill big ZIPs into. None falls back to the system temp directory."""
try:
directory = Path(configuration_manager.global_config_file_manager.get_config_directory_path())
spool_directory = directory / _SPOOL_DIRECTORY_NAME
spool_directory.mkdir(parents=True, exist_ok=True)
return str(spool_directory)
except OSError as e:
logger.debug('Failed to create the spool directory; falling back to the system one', exc_info=e)
return None


class InMemoryZip:
def __init__(self) -> None:
self.configuration_manager = ConfigurationManager()

self.in_memory_zip = BytesIO()
self.zip = ZipFile(self.in_memory_zip, mode='a', compression=ZIP_DEFLATED, allowZip64=False)
self._spool_max_size = consts.ZIP_SPOOL_MAX_SIZE_IN_BYTES
self._buffer = tempfile.SpooledTemporaryFile( # noqa: SIM115 # closed by cleanup(), lives past close()
max_size=self._spool_max_size,
dir=_get_spool_directory(self.configuration_manager),
)

# ZIP64 lifts the 65,535 entries and 4 GiB caps of the original ZIP format.
# It requires 64-bit offsets, so we only enable it on a 64-bit interpreter.
self._allow_zip64 = is_64bit()
self.zip = ZipFile(self._buffer, mode='a', compression=ZIP_DEFLATED, allowZip64=self._allow_zip64)

self._files_count = 0
self._extension_statistics = defaultdict(int)
Expand All @@ -35,17 +63,54 @@ def append(self, filename: str, unique_id: Optional[str], content: str) -> None:
def close(self) -> None:
self.zip.close()

def cleanup(self) -> None:
"""Release the buffer, deleting the spilled temp file if there is one."""
self._buffer.close()

def __enter__(self) -> 'InMemoryZip': # noqa: PYI034 # typing.Self needs Python 3.11
return self

def __exit__(self, *_: object) -> None:
self.cleanup()

def stream(self) -> IO[bytes]:
"""The whole archive as a file object, rewound. Doesn't copy it into memory.

Note: before Python 3.11 SpooledTemporaryFile isn't a real IOBase, so the returned object
has no seekable()/readable()/writable(). read/seek/tell work on every supported version.
"""
self._buffer.seek(0)
return self._buffer

def read(self) -> bytes:
self.in_memory_zip.seek(0)
return self.in_memory_zip.read()
self._buffer.seek(0)
return self._buffer.read()

def write_on_disk(self, path: 'Path') -> None:
with open(path, 'wb') as f:
f.write(self.read())
shutil.copyfileobj(self.stream(), f)

@property
def size(self) -> int:
return getsizeof(self.in_memory_zip)
position = self._buffer.tell()
try:
self._buffer.seek(0, SEEK_END)
return self._buffer.tell()
finally:
self._buffer.seek(position)

@property
def is_rolled_over(self) -> bool:
"""Whether the archive outgrew the threshold and moved from memory to the disk.

SpooledTemporaryFile spills on the write that crosses max_size, and the archive only grows,
so the size says it without reaching into the private _rolled flag.
"""
return self.size > self._spool_max_size

@property
def allow_zip64(self) -> bool:
return self._allow_zip64

@property
def files_count(self) -> int:
Expand Down
5 changes: 5 additions & 0 deletions cycode/cli/utils/host_info.py
Original file line number Diff line number Diff line change
Expand Up @@ -49,6 +49,11 @@ def _read_text_file(path: str) -> Optional[str]:
return None


def is_64bit() -> bool:
"""Whether the running Python interpreter is 64-bit (not the OS)."""
return sys.maxsize > 2**32


def get_hostname() -> Optional[str]:
try:
return socket.gethostname() or None
Expand Down
Loading
Loading