From f16128c5d7d2b88f7b116587efa37ad03fc678d9 Mon Sep 17 00:00:00 2001 From: "omer.roth" Date: Tue, 15 Sep 2026 16:56:59 +0300 Subject: [PATCH 1/2] CM-72724: Enable ZIP64 and spool the scan archive to disk Scans of repositories with more than 65,535 files crashed instead of falling back to the batched upload. The archive was built with allowZip64=False, so zipfile raised LargeZipFile - which the caller did not catch, because the only guard there was the archive's byte size. Enable ZIP64 on 64-bit interpreters, route to the batched upload when a single archive provably cannot hold the documents, and log which upload mode was selected and why. Also take the archive off the heap: it is now built into a SpooledTemporaryFile that stays in memory below 64 MB and spills into ~/.cycode/tmp beyond it, and the archive built to check that everything fits is reused for the upload instead of being thrown away and rebuilt. Co-Authored-By: Claude Opus 5 (1M context) --- cycode/cli/apps/scan/code_scanner.py | 97 +++++++--- cycode/cli/apps/scan/commit_range_scanner.py | 3 + cycode/cli/consts.py | 6 + cycode/cli/exceptions/handle_scan_errors.py | 8 + .../files_collector/models/in_memory_zip.py | 83 ++++++++- cycode/cli/utils/host_info.py | 5 + tests/cli/commands/scan/test_code_scanner.py | 107 ++++++++++- .../cli/exceptions/test_handle_scan_errors.py | 2 + .../cli/files_collector/test_in_memory_zip.py | 168 ++++++++++++++++++ 9 files changed, 448 insertions(+), 31 deletions(-) diff --git a/cycode/cli/apps/scan/code_scanner.py b/cycode/cli/apps/scan/code_scanner.py index 667138fa..77af5ea6 100644 --- a/cycode/cli/apps/scan/code_scanner.py +++ b/cycode/cli/apps/scan/code_scanner.py @@ -1,5 +1,6 @@ import os import time +import zipfile from platform import platform from typing import TYPE_CHECKING, Callable, Optional @@ -23,6 +24,7 @@ from cycode.cli.files_collector.sca.sca_file_collector import add_sca_dependencies_tree_documents_if_needed from cycode.cli.files_collector.zip_documents import zip_documents from cycode.cli.models import CliError, Document, LocalScanResult +from cycode.cli.utils.host_info import is_64bit from cycode.cli.utils.path_utils import get_absolute_path, get_path_by_os from cycode.cli.utils.progress_bar import ScanProgressBarSection from cycode.cli.utils.scan_batch import run_parallel_batched_scan @@ -145,6 +147,7 @@ def _get_scan_documents_thread_func( is_git_diff: bool, is_commit_range: bool, scan_parameters: dict, + prezipped: Optional['InMemoryZip'] = None, ) -> Callable[[list[Document]], tuple[str, CliError, LocalScanResult]]: cycode_client = ctx.obj['client'] scan_type = ctx.obj['scan_type'] @@ -164,9 +167,14 @@ def _scan_batch_thread_func(batch: list[Document]) -> tuple[str, CliError, Local should_use_sync_flow = _should_use_sync_flow(command_scan_type, scan_type, sync_option) + # the single ZIP flow already built the archive to check that it fits; don't build it twice + zipped_documents = prezipped + try: - logger.debug('Preparing local files, %s', {'batch_files_count': len(batch)}) - zipped_documents = zip_documents(scan_type, batch) + if zipped_documents is None: + logger.debug('Preparing local files, %s', {'batch_files_count': len(batch)}) + zipped_documents = zip_documents(scan_type, batch) + zip_file_size = zipped_documents.size scan_result = _perform_scan( cycode_client, @@ -189,6 +197,9 @@ def _scan_batch_thread_func(batch: list[Document]) -> tuple[str, CliError, Local except Exception as e: error = handle_scan_exception(ctx, e, return_exception=True) error_message = str(e) + finally: + if zipped_documents is not None: + zipped_documents.cleanup() if local_scan_result: detections_count = local_scan_result.detections_count @@ -225,34 +236,77 @@ def _scan_batch_thread_func(batch: list[Document]) -> tuple[str, CliError, Local return _scan_batch_thread_func +def _log_selected_upload_mode(mode: str, reason: str, documents_count: int) -> None: + logger.debug( + 'Selected upload mode, %s', + { + 'mode': mode, + 'reason': reason, + 'documents_count': documents_count, + 'max_files_count': consts.ZIP_MAX_FILES_COUNT, + 'zip64_enabled': is_64bit(), + }, + ) + + +def _exceeds_non_zip64_files_count(documents_to_scan: list[Document]) -> bool: + """Whether a single ZIP can't hold all the documents because ZIP64 is unavailable. + + Without ZIP64 (32-bit interpreter) the archive is capped at 65,535 entries. + """ + return not is_64bit() and len(documents_to_scan) > consts.ZIP_MAX_FILES_COUNT + + def _run_presigned_upload_scan( - scan_batch_thread_func: Callable, - scan_type: str, + ctx: typer.Context, + is_git_diff: bool, + is_commit_range: bool, + scan_parameters: dict, documents_to_scan: list[Document], progress_bar: 'BaseProgressBar', printer: 'ConsolePrinter', ) -> tuple: - try: - # Try to zip all documents as a single batch; ZipTooLargeError raised if it exceeds the scan type's limit - zip_documents(scan_type, documents_to_scan) - # It fits: skip batching and upload everything as one ZIP + scan_type = ctx.obj['scan_type'] + documents_count = len(documents_to_scan) + + def run_batched() -> tuple: return run_parallel_batched_scan( - scan_batch_thread_func, + _get_scan_documents_thread_func(ctx, is_git_diff, is_commit_range, scan_parameters), scan_type, documents_to_scan, progress_bar=progress_bar, - skip_batching=True, ) - except custom_exceptions.ZipTooLargeError: + + if _exceeds_non_zip64_files_count(documents_to_scan): + # Don't waste time zipping documents we already know won't fit into a single ZIP + _log_selected_upload_mode('batched', 'files_count_exceeds_non_zip64_limit', documents_count) + return run_batched() + + zipped_documents = None + try: + # Try to zip all documents as a single batch; ZipTooLargeError raised if it exceeds the scan type's limit + zipped_documents = zip_documents(scan_type, documents_to_scan) + except (custom_exceptions.ZipTooLargeError, zipfile.LargeZipFile): + # LargeZipFile is a safety net: the files count pre-check above should have caught it already + _log_selected_upload_mode('batched', 'zip_too_large', documents_count) + if zipped_documents is not None: + zipped_documents.cleanup() + printer.print_warning( 'The scan is too large to upload as a single file. This may result in corrupted scan results.' ) - return run_parallel_batched_scan( - scan_batch_thread_func, - scan_type, - documents_to_scan, - progress_bar=progress_bar, - ) + return run_batched() + + # It fits: skip batching and upload everything as one ZIP. The archive we just built is the one + # that gets uploaded, so the scan doesn't pay for compressing every document twice + _log_selected_upload_mode('single_zip', 'fits_single_zip', documents_count) + return run_parallel_batched_scan( + _get_scan_documents_thread_func(ctx, is_git_diff, is_commit_range, scan_parameters, zipped_documents), + scan_type, + documents_to_scan, + progress_bar=progress_bar, + skip_batching=True, + ) def scan_documents( @@ -277,18 +331,19 @@ def scan_documents( ) return - scan_batch_thread_func = _get_scan_documents_thread_func(ctx, is_git_diff, is_commit_range, scan_parameters) - # Presigned single-file upload is async-only; a --sync scan must stay on the batched inline path # so it never builds one oversized zip to POST synchronously. should_use_sync_flow = _should_use_sync_flow(ctx.info_name, scan_type, ctx.obj['sync']) if should_use_presigned_upload(scan_type) and not should_use_sync_flow: errors, local_scan_results = _run_presigned_upload_scan( - scan_batch_thread_func, scan_type, documents_to_scan, progress_bar, printer + ctx, is_git_diff, is_commit_range, scan_parameters, documents_to_scan, progress_bar, printer ) else: errors, local_scan_results = run_parallel_batched_scan( - scan_batch_thread_func, scan_type, documents_to_scan, progress_bar=progress_bar + _get_scan_documents_thread_func(ctx, is_git_diff, is_commit_range, scan_parameters), + scan_type, + documents_to_scan, + progress_bar=progress_bar, ) try_set_aggregation_report_url_if_needed(ctx, scan_parameters, ctx.obj['client'], scan_type) diff --git a/cycode/cli/apps/scan/commit_range_scanner.py b/cycode/cli/apps/scan/commit_range_scanner.py index 70b7e8e4..af7bf9bb 100644 --- a/cycode/cli/apps/scan/commit_range_scanner.py +++ b/cycode/cli/apps/scan/commit_range_scanner.py @@ -213,6 +213,9 @@ def _scan_commit_range_documents( error_message = str(e) zip_file_size = from_commit_zipped_documents.size + to_commit_zipped_documents.size + + from_commit_zipped_documents.cleanup() + to_commit_zipped_documents.cleanup() detections_count = relevant_detections_count = 0 if local_scan_result: diff --git a/cycode/cli/consts.py b/cycode/cli/consts.py index 104cfc9b..e2d09ce1 100644 --- a/cycode/cli/consts.py +++ b/cycode/cli/consts.py @@ -227,6 +227,12 @@ PRESIGNED_LINK_UPLOADED_ZIP_MAX_SIZE_LIMIT_IN_BYTES = 5 * 1024 * 1024 * 1024 # 5 GB (S3 presigned POST limit) PRESIGNED_UPLOAD_SCAN_TYPES = {SAST_SCAN_TYPE, SECRET_SCAN_TYPE} +# the non-ZIP64 central directory stores the entry count in 16 bits; ZIP64 (64-bit interpreters) lifts it +ZIP_MAX_FILES_COUNT = 65_535 + +# the ZIP is built in memory up to this size, and spilled to a temp file beyond it +ZIP_SPOOL_MAX_SIZE_IN_BYTES = 64 * 1024 * 1024 + DEFAULT_ZIP_MAX_SIZE_LIMIT_IN_BYTES = 20 * 1024 * 1024 ZIP_MAX_SIZE_LIMIT_IN_BYTES = { SCA_SCAN_TYPE: 200 * 1024 * 1024, diff --git a/cycode/cli/exceptions/handle_scan_errors.py b/cycode/cli/exceptions/handle_scan_errors.py index 56af186c..0b1e975b 100644 --- a/cycode/cli/exceptions/handle_scan_errors.py +++ b/cycode/cli/exceptions/handle_scan_errors.py @@ -1,3 +1,4 @@ +import zipfile from typing import Optional import typer @@ -26,6 +27,13 @@ def handle_scan_exception(ctx: typer.Context, err: Exception, *, return_exceptio 'Please try ignoring irrelevant paths using the `cycode ignore --by-path` command ' 'and execute the scan again', ), + zipfile.LargeZipFile: CliError( + soft_fail=True, + code='zip_too_large_error', + message='The path you attempted to scan contains too many files to pack into a single archive. ' + 'Scanning such paths requires a 64-bit Python interpreter. ' + 'Please try ignoring irrelevant paths using a .cycodeignore file and execute the scan again', + ), custom_exceptions.FileCollectionError: CliError( soft_fail=False, code='file_collection_error', diff --git a/cycode/cli/files_collector/models/in_memory_zip.py b/cycode/cli/files_collector/models/in_memory_zip.py index 8bb9bf9e..5da13113 100644 --- a/cycode/cli/files_collector/models/in_memory_zip.py +++ b/cycode/cli/files_collector/models/in_memory_zip.py @@ -1,20 +1,48 @@ +import shutil +import tempfile from collections import defaultdict -from io import BytesIO +from os import SEEK_END from pathlib import Path -from sys import getsizeof -from typing import Optional +from typing import IO, Optional from zipfile import ZIP_DEFLATED, ZipFile +from cycode.cli import consts from cycode.cli.user_settings.configuration_manager import ConfigurationManager +from cycode.cli.utils.host_info import is_64bit from cycode.cli.utils.path_utils import concat_unique_id +from cycode.logger import get_logger + +logger = get_logger('ZIP') + +_SPOOL_DIRECTORY_NAME = 'tmp' + + +def _get_spool_directory(configuration_manager: ConfigurationManager) -> Optional[str]: + """Directory to spill big ZIPs into. None falls back to the system temp directory.""" + try: + directory = Path(configuration_manager.global_config_file_manager.get_config_directory_path()) + spool_directory = directory / _SPOOL_DIRECTORY_NAME + spool_directory.mkdir(parents=True, exist_ok=True) + return str(spool_directory) + except OSError as e: + logger.debug('Failed to create the spool directory; falling back to the system one', exc_info=e) + return None class InMemoryZip: def __init__(self) -> None: self.configuration_manager = ConfigurationManager() - self.in_memory_zip = BytesIO() - self.zip = ZipFile(self.in_memory_zip, mode='a', compression=ZIP_DEFLATED, allowZip64=False) + self._spool_max_size = consts.ZIP_SPOOL_MAX_SIZE_IN_BYTES + self._buffer = tempfile.SpooledTemporaryFile( # noqa: SIM115 # closed by cleanup(), lives past close() + max_size=self._spool_max_size, + dir=_get_spool_directory(self.configuration_manager), + ) + + # ZIP64 lifts the 65,535 entries and 4 GiB caps of the original ZIP format. + # It requires 64-bit offsets, so we only enable it on a 64-bit interpreter. + self._allow_zip64 = is_64bit() + self.zip = ZipFile(self._buffer, mode='a', compression=ZIP_DEFLATED, allowZip64=self._allow_zip64) self._files_count = 0 self._extension_statistics = defaultdict(int) @@ -35,17 +63,54 @@ def append(self, filename: str, unique_id: Optional[str], content: str) -> None: def close(self) -> None: self.zip.close() + def cleanup(self) -> None: + """Release the buffer, deleting the spilled temp file if there is one.""" + self._buffer.close() + + def __enter__(self) -> 'InMemoryZip': # noqa: PYI034 # typing.Self needs Python 3.11 + return self + + def __exit__(self, *_: object) -> None: + self.cleanup() + + def stream(self) -> IO[bytes]: + """The whole archive as a file object, rewound. Doesn't copy it into memory. + + Note: before Python 3.11 SpooledTemporaryFile isn't a real IOBase, so the returned object + has no seekable()/readable()/writable(). read/seek/tell work on every supported version. + """ + self._buffer.seek(0) + return self._buffer + def read(self) -> bytes: - self.in_memory_zip.seek(0) - return self.in_memory_zip.read() + self._buffer.seek(0) + return self._buffer.read() def write_on_disk(self, path: 'Path') -> None: with open(path, 'wb') as f: - f.write(self.read()) + shutil.copyfileobj(self.stream(), f) @property def size(self) -> int: - return getsizeof(self.in_memory_zip) + position = self._buffer.tell() + try: + self._buffer.seek(0, SEEK_END) + return self._buffer.tell() + finally: + self._buffer.seek(position) + + @property + def is_rolled_over(self) -> bool: + """Whether the archive outgrew the threshold and moved from memory to the disk. + + SpooledTemporaryFile spills on the write that crosses max_size, and the archive only grows, + so the size says it without reaching into the private _rolled flag. + """ + return self.size > self._spool_max_size + + @property + def allow_zip64(self) -> bool: + return self._allow_zip64 @property def files_count(self) -> int: diff --git a/cycode/cli/utils/host_info.py b/cycode/cli/utils/host_info.py index 064f7823..5ea3e62d 100644 --- a/cycode/cli/utils/host_info.py +++ b/cycode/cli/utils/host_info.py @@ -49,6 +49,11 @@ def _read_text_file(path: str) -> Optional[str]: return None +def is_64bit() -> bool: + """Whether the running Python interpreter is 64-bit (not the OS).""" + return sys.maxsize > 2**32 + + def get_hostname() -> Optional[str]: try: return socket.gethostname() or None diff --git a/tests/cli/commands/scan/test_code_scanner.py b/tests/cli/commands/scan/test_code_scanner.py index 8b4a30b3..24f6a39f 100644 --- a/tests/cli/commands/scan/test_code_scanner.py +++ b/tests/cli/commands/scan/test_code_scanner.py @@ -1,11 +1,18 @@ import os +import zipfile from os.path import normpath from unittest.mock import MagicMock, Mock, patch import pytest from cycode.cli import consts -from cycode.cli.apps.scan.code_scanner import _perform_scan, scan_disk_files, scan_documents +from cycode.cli.apps.scan.code_scanner import ( + _get_scan_documents_thread_func, + _perform_scan, + _run_presigned_upload_scan, + scan_disk_files, + scan_documents, +) from cycode.cli.exceptions import custom_exceptions from cycode.cli.files_collector.file_excluder import _is_file_relevant_for_sca_scan from cycode.cli.files_collector.path_documents import _generate_document @@ -237,3 +244,101 @@ def test_perform_scan_falls_back_to_api_when_presigned_upload_raises_wrapped_err assert result is fallback_result mock_v4_async.assert_called_once() mock_async.assert_called_once() + + +def _presigned_scan_ctx() -> MagicMock: + ctx = MagicMock() + ctx.obj = { + 'client': MagicMock(), + 'scan_type': consts.SECRET_SCAN_TYPE, + 'severity_threshold': None, + 'sync': False, + 'progress_bar': MagicMock(), + } + return ctx + + +@pytest.mark.parametrize( + ('is_64bit', 'expected_skip_batching', 'expected_zip_calls'), + [ + # 64-bit: ZIP64 lifts the entries limit, so everything still goes up as a single ZIP + (True, True, 1), + # 32-bit: the archive can't hold that many entries; route to batches without zipping first + (False, None, 0), + ], +) +@patch('cycode.cli.apps.scan.code_scanner.run_parallel_batched_scan') +@patch('cycode.cli.apps.scan.code_scanner.zip_documents') +@patch('cycode.cli.apps.scan.code_scanner.is_64bit') +def test_run_presigned_upload_scan_routes_by_files_count( + mock_is_64bit: Mock, + mock_zip_documents: Mock, + mock_run_parallel_batched_scan: Mock, + is_64bit: bool, + expected_skip_batching: bool, + expected_zip_calls: int, +) -> None: + mock_is_64bit.return_value = is_64bit + documents = [Document(f'file_{index}.txt', 'content') for index in range(consts.ZIP_MAX_FILES_COUNT + 1)] + + _run_presigned_upload_scan(_presigned_scan_ctx(), False, False, {}, documents, MagicMock(), MagicMock()) + + assert mock_zip_documents.call_count == expected_zip_calls + assert mock_run_parallel_batched_scan.call_args.kwargs.get('skip_batching') is expected_skip_batching + + +@patch('cycode.cli.apps.scan.code_scanner.run_parallel_batched_scan') +@patch('cycode.cli.apps.scan.code_scanner.zip_documents') +def test_run_presigned_upload_scan_falls_back_to_batches_on_large_zip_file( + mock_zip_documents: Mock, mock_run_parallel_batched_scan: Mock +) -> None: + # safety net: zipfile raises LargeZipFile directly instead of our own ZipTooLargeError + mock_zip_documents.side_effect = zipfile.LargeZipFile('Files count would require ZIP64 extensions') + + _run_presigned_upload_scan( + _presigned_scan_ctx(), False, False, {}, [Document('file.txt', 'content')], MagicMock(), MagicMock() + ) + + assert mock_run_parallel_batched_scan.call_args.kwargs.get('skip_batching') is None + + +@patch('cycode.cli.apps.scan.code_scanner.run_parallel_batched_scan') +@patch('cycode.cli.apps.scan.code_scanner.zip_documents') +@patch('cycode.cli.apps.scan.code_scanner._get_scan_documents_thread_func') +def test_run_presigned_upload_scan_reuses_the_archive_it_built( + mock_get_thread_func: Mock, mock_zip_documents: Mock, mock_run_parallel_batched_scan: Mock +) -> None: + # the archive built to check that everything fits is the one we upload; don't compress it twice + zipped_documents = mock_zip_documents.return_value + + _run_presigned_upload_scan( + _presigned_scan_ctx(), False, False, {}, [Document('file.txt', 'content')], MagicMock(), MagicMock() + ) + + mock_zip_documents.assert_called_once() + assert mock_get_thread_func.call_args.args[-1] is zipped_documents + assert mock_run_parallel_batched_scan.call_args.kwargs.get('skip_batching') is True + + +@patch('cycode.cli.apps.scan.code_scanner._perform_scan') +@patch('cycode.cli.apps.scan.code_scanner.zip_documents') +def test_scan_batch_thread_func_does_not_rezip_a_prezipped_batch( + mock_zip_documents: Mock, mock_perform_scan: Mock +) -> None: + prezipped = MagicMock() + ctx = MagicMock() + ctx.obj = { + 'client': MagicMock(), + 'scan_type': consts.SECRET_SCAN_TYPE, + 'severity_threshold': None, + 'sync': False, + 'progress_bar': MagicMock(), + } + + scan_batch_thread_func = _get_scan_documents_thread_func(ctx, False, False, {}, prezipped) + scan_batch_thread_func([Document('file.txt', 'content')]) + + mock_zip_documents.assert_not_called() + assert mock_perform_scan.call_args.args[1] is prezipped + # the buffer is released once the batch is done with it + prezipped.cleanup.assert_called_once() diff --git a/tests/cli/exceptions/test_handle_scan_errors.py b/tests/cli/exceptions/test_handle_scan_errors.py index fb14bc8a..ed749f6a 100644 --- a/tests/cli/exceptions/test_handle_scan_errors.py +++ b/tests/cli/exceptions/test_handle_scan_errors.py @@ -1,3 +1,4 @@ +import zipfile from typing import TYPE_CHECKING, Any import click @@ -31,6 +32,7 @@ def ctx() -> typer.Context: (custom_exceptions.ScanAsyncError('msg'), True), (custom_exceptions.HttpUnauthorizedError('msg', Response()), True), (custom_exceptions.ZipTooLargeError(1000), True), + (zipfile.LargeZipFile('Files count would require ZIP64 extensions'), True), (custom_exceptions.TfplanKeyError('msg'), True), (custom_exceptions.FileCollectionError('Failed to generate dependencies tree for pom.xml'), None), (git_proxy.get_invalid_git_repository_error()(), None), diff --git a/tests/cli/files_collector/test_in_memory_zip.py b/tests/cli/files_collector/test_in_memory_zip.py index d1790c7c..afd807e5 100644 --- a/tests/cli/files_collector/test_in_memory_zip.py +++ b/tests/cli/files_collector/test_in_memory_zip.py @@ -2,9 +2,19 @@ import zipfile from io import BytesIO +from pathlib import Path +from typing import TYPE_CHECKING +from unittest.mock import Mock +from uuid import uuid4 +import pytest + +from cycode.cli import consts from cycode.cli.files_collector.models.in_memory_zip import InMemoryZip +if TYPE_CHECKING: + from _pytest.monkeypatch import MonkeyPatch + def test_append_with_surrogate_characters() -> None: """Test that surrogate characters are handled gracefully without raising encoding errors.""" @@ -27,3 +37,161 @@ def test_append_with_surrogate_characters() -> None: assert 'more text' in extracted # The surrogate should have been replaced with the replacement character assert '\udc96' not in extracted + + +@pytest.mark.parametrize(('is_64bit', 'expected_allow_zip64'), [(True, True), (False, False)]) +def test_allow_zip64_follows_interpreter_bitness( + monkeypatch: 'MonkeyPatch', is_64bit: bool, expected_allow_zip64: bool +) -> None: + """ZIP64 requires 64-bit offsets, so it's enabled only on a 64-bit interpreter.""" + monkeypatch.setattr('cycode.cli.files_collector.models.in_memory_zip.is_64bit', lambda: is_64bit) + + zip_file = InMemoryZip() + + assert zip_file.allow_zip64 is expected_allow_zip64 + assert zip_file.zip._allowZip64 is expected_allow_zip64 + + +def test_append_more_files_than_non_zip64_limit_on_64bit(monkeypatch: 'MonkeyPatch') -> None: + """With ZIP64 enabled, the archive holds more than 65,535 entries.""" + monkeypatch.setattr('cycode.cli.files_collector.models.in_memory_zip.is_64bit', lambda: True) + + files_count = consts.ZIP_MAX_FILES_COUNT + 1 + zip_file = InMemoryZip() + for index in range(files_count): + zip_file.append(f'file_{index}.txt', None, 'content') + zip_file.close() + + with zipfile.ZipFile(BytesIO(zip_file.read()), 'r') as zf: + assert len(zf.namelist()) == files_count + + assert zip_file.files_count == files_count + + +def test_append_more_files_than_non_zip64_limit_on_32bit(monkeypatch: 'MonkeyPatch') -> None: + """Without ZIP64, appending past 65,535 entries raises instead of silently truncating.""" + monkeypatch.setattr('cycode.cli.files_collector.models.in_memory_zip.is_64bit', lambda: False) + + zip_file = InMemoryZip() + with pytest.raises(zipfile.LargeZipFile): + for index in range(consts.ZIP_MAX_FILES_COUNT + 1): + zip_file.append(f'file_{index}.txt', None, 'content') + + +def test_size_is_the_real_archive_length() -> None: + """The size guard and the reported zip_size must reflect the actual bytes, not an approximation.""" + zip_file = InMemoryZip() + zip_file.append('test.txt', None, 'content' * 1000) + zip_file.close() + + assert zip_file.size == len(zip_file.read()) + + +def test_stays_in_memory_below_the_spool_threshold(monkeypatch: 'MonkeyPatch') -> None: + """Ordinary scans never touch the disk.""" + monkeypatch.setattr(consts, 'ZIP_SPOOL_MAX_SIZE_IN_BYTES', 1024 * 1024) + + zip_file = InMemoryZip() + zip_file.append('test.txt', None, 'content') + zip_file.close() + + assert zip_file.is_rolled_over is False + + +def test_rolls_over_to_disk_above_the_spool_threshold(monkeypatch: 'MonkeyPatch') -> None: + """A big archive spills into a temp file instead of growing the heap, and still reads back intact.""" + monkeypatch.setattr(consts, 'ZIP_SPOOL_MAX_SIZE_IN_BYTES', 1024) + + zip_file = InMemoryZip() + for index in range(100): + # random-ish content so deflate can't compress it away below the threshold + zip_file.append(f'file_{index}.txt', None, str(uuid4()) * 100) + zip_file.close() + + assert zip_file.is_rolled_over is True + with zipfile.ZipFile(BytesIO(zip_file.read()), 'r') as zf: + assert len(zf.namelist()) == 100 + + zip_file.cleanup() + + +def test_spills_into_the_cycode_configuration_directory(monkeypatch: 'MonkeyPatch', tmp_path: Path) -> None: + """The temp file lands under the Cycode configuration directory, not in an arbitrary location.""" + monkeypatch.setattr(Path, 'home', lambda: tmp_path) + monkeypatch.setattr(consts, 'ZIP_SPOOL_MAX_SIZE_IN_BYTES', 1) + + zip_file = InMemoryZip() + zip_file.append('test.txt', None, 'content') + zip_file.close() + + assert zip_file.is_rolled_over is True + assert (tmp_path / consts.CYCODE_CONFIGURATION_DIRECTORY / 'tmp').is_dir() + + zip_file.cleanup() + + +def test_falls_back_to_the_system_temp_directory(monkeypatch: 'MonkeyPatch') -> None: + """An unwritable configuration directory (read-only CI runner) must not fail the scan.""" + monkeypatch.setattr(consts, 'ZIP_SPOOL_MAX_SIZE_IN_BYTES', 1) + monkeypatch.setattr( + 'cycode.cli.files_collector.models.in_memory_zip.Path.mkdir', + Mock(side_effect=PermissionError('read-only file system')), + ) + + zip_file = InMemoryZip() + zip_file.append('test.txt', None, 'content') + zip_file.close() + + with zipfile.ZipFile(BytesIO(zip_file.read()), 'r') as zf: + assert zf.read('test.txt') == b'content' + + zip_file.cleanup() + + +def test_cleanup_is_idempotent() -> None: + """Callers release the buffer in a finally block; a double release must not raise.""" + zip_file = InMemoryZip() + zip_file.append('test.txt', None, 'content') + zip_file.close() + + zip_file.cleanup() + zip_file.cleanup() + + +def test_used_as_a_context_manager() -> None: + zip_file = InMemoryZip() + with zip_file as managed: + managed.append('test.txt', None, 'content') + managed.close() + content = managed.read() + + assert len(content) > 0 + + +def test_write_on_disk_matches_the_archive(tmp_path: Path) -> None: + """The --debug dump streams the archive out instead of copying it through memory.""" + zip_file = InMemoryZip() + zip_file.append('test.txt', None, 'content') + zip_file.close() + + zip_file_path = tmp_path / 'dump.zip' + zip_file.write_on_disk(zip_file_path) + + assert zip_file_path.read_bytes() == zip_file.read() + + +@pytest.mark.parametrize('spool_max_size', [1, 512, 1024 * 1024]) +def test_is_rolled_over_matches_the_interpreter(monkeypatch: 'MonkeyPatch', spool_max_size: int) -> None: + """Our size-based check must agree with CPython's own (private) rollover flag on every version.""" + monkeypatch.setattr(consts, 'ZIP_SPOOL_MAX_SIZE_IN_BYTES', spool_max_size) + + zip_file = InMemoryZip() + for index in range(20): + zip_file.append(f'file_{index}.txt', None, str(uuid4()) * 10) + zip_file.close() + + rolled_by_interpreter = getattr(zip_file._buffer, '_rolled', None) + if rolled_by_interpreter is not None: + assert zip_file.is_rolled_over is rolled_by_interpreter + + zip_file.cleanup() From 8b94fe257e6d44e2eda8d5262e8b0be98ebb594c Mon Sep 17 00:00:00 2001 From: "omer.roth" Date: Tue, 15 Sep 2026 17:04:06 +0300 Subject: [PATCH 2/2] CM-72724 ruff format --- cycode/cli/apps/scan/commit_range_scanner.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/cycode/cli/apps/scan/commit_range_scanner.py b/cycode/cli/apps/scan/commit_range_scanner.py index af7bf9bb..b1a58202 100644 --- a/cycode/cli/apps/scan/commit_range_scanner.py +++ b/cycode/cli/apps/scan/commit_range_scanner.py @@ -213,7 +213,7 @@ def _scan_commit_range_documents( error_message = str(e) zip_file_size = from_commit_zipped_documents.size + to_commit_zipped_documents.size - + from_commit_zipped_documents.cleanup() to_commit_zipped_documents.cleanup()