diff --git a/cycode/cli/apps/scan/code_scanner.py b/cycode/cli/apps/scan/code_scanner.py index 667138fa..77af5ea6 100644 --- a/cycode/cli/apps/scan/code_scanner.py +++ b/cycode/cli/apps/scan/code_scanner.py @@ -1,5 +1,6 @@ import os import time +import zipfile from platform import platform from typing import TYPE_CHECKING, Callable, Optional @@ -23,6 +24,7 @@ from cycode.cli.files_collector.sca.sca_file_collector import add_sca_dependencies_tree_documents_if_needed from cycode.cli.files_collector.zip_documents import zip_documents from cycode.cli.models import CliError, Document, LocalScanResult +from cycode.cli.utils.host_info import is_64bit from cycode.cli.utils.path_utils import get_absolute_path, get_path_by_os from cycode.cli.utils.progress_bar import ScanProgressBarSection from cycode.cli.utils.scan_batch import run_parallel_batched_scan @@ -145,6 +147,7 @@ def _get_scan_documents_thread_func( is_git_diff: bool, is_commit_range: bool, scan_parameters: dict, + prezipped: Optional['InMemoryZip'] = None, ) -> Callable[[list[Document]], tuple[str, CliError, LocalScanResult]]: cycode_client = ctx.obj['client'] scan_type = ctx.obj['scan_type'] @@ -164,9 +167,14 @@ def _scan_batch_thread_func(batch: list[Document]) -> tuple[str, CliError, Local should_use_sync_flow = _should_use_sync_flow(command_scan_type, scan_type, sync_option) + # the single ZIP flow already built the archive to check that it fits; don't build it twice + zipped_documents = prezipped + try: - logger.debug('Preparing local files, %s', {'batch_files_count': len(batch)}) - zipped_documents = zip_documents(scan_type, batch) + if zipped_documents is None: + logger.debug('Preparing local files, %s', {'batch_files_count': len(batch)}) + zipped_documents = zip_documents(scan_type, batch) + zip_file_size = zipped_documents.size scan_result = _perform_scan( cycode_client, @@ -189,6 +197,9 @@ def _scan_batch_thread_func(batch: list[Document]) -> tuple[str, CliError, Local except Exception as e: error = handle_scan_exception(ctx, e, return_exception=True) error_message = str(e) + finally: + if zipped_documents is not None: + zipped_documents.cleanup() if local_scan_result: detections_count = local_scan_result.detections_count @@ -225,34 +236,77 @@ def _scan_batch_thread_func(batch: list[Document]) -> tuple[str, CliError, Local return _scan_batch_thread_func +def _log_selected_upload_mode(mode: str, reason: str, documents_count: int) -> None: + logger.debug( + 'Selected upload mode, %s', + { + 'mode': mode, + 'reason': reason, + 'documents_count': documents_count, + 'max_files_count': consts.ZIP_MAX_FILES_COUNT, + 'zip64_enabled': is_64bit(), + }, + ) + + +def _exceeds_non_zip64_files_count(documents_to_scan: list[Document]) -> bool: + """Whether a single ZIP can't hold all the documents because ZIP64 is unavailable. + + Without ZIP64 (32-bit interpreter) the archive is capped at 65,535 entries. + """ + return not is_64bit() and len(documents_to_scan) > consts.ZIP_MAX_FILES_COUNT + + def _run_presigned_upload_scan( - scan_batch_thread_func: Callable, - scan_type: str, + ctx: typer.Context, + is_git_diff: bool, + is_commit_range: bool, + scan_parameters: dict, documents_to_scan: list[Document], progress_bar: 'BaseProgressBar', printer: 'ConsolePrinter', ) -> tuple: - try: - # Try to zip all documents as a single batch; ZipTooLargeError raised if it exceeds the scan type's limit - zip_documents(scan_type, documents_to_scan) - # It fits: skip batching and upload everything as one ZIP + scan_type = ctx.obj['scan_type'] + documents_count = len(documents_to_scan) + + def run_batched() -> tuple: return run_parallel_batched_scan( - scan_batch_thread_func, + _get_scan_documents_thread_func(ctx, is_git_diff, is_commit_range, scan_parameters), scan_type, documents_to_scan, progress_bar=progress_bar, - skip_batching=True, ) - except custom_exceptions.ZipTooLargeError: + + if _exceeds_non_zip64_files_count(documents_to_scan): + # Don't waste time zipping documents we already know won't fit into a single ZIP + _log_selected_upload_mode('batched', 'files_count_exceeds_non_zip64_limit', documents_count) + return run_batched() + + zipped_documents = None + try: + # Try to zip all documents as a single batch; ZipTooLargeError raised if it exceeds the scan type's limit + zipped_documents = zip_documents(scan_type, documents_to_scan) + except (custom_exceptions.ZipTooLargeError, zipfile.LargeZipFile): + # LargeZipFile is a safety net: the files count pre-check above should have caught it already + _log_selected_upload_mode('batched', 'zip_too_large', documents_count) + if zipped_documents is not None: + zipped_documents.cleanup() + printer.print_warning( 'The scan is too large to upload as a single file. This may result in corrupted scan results.' ) - return run_parallel_batched_scan( - scan_batch_thread_func, - scan_type, - documents_to_scan, - progress_bar=progress_bar, - ) + return run_batched() + + # It fits: skip batching and upload everything as one ZIP. The archive we just built is the one + # that gets uploaded, so the scan doesn't pay for compressing every document twice + _log_selected_upload_mode('single_zip', 'fits_single_zip', documents_count) + return run_parallel_batched_scan( + _get_scan_documents_thread_func(ctx, is_git_diff, is_commit_range, scan_parameters, zipped_documents), + scan_type, + documents_to_scan, + progress_bar=progress_bar, + skip_batching=True, + ) def scan_documents( @@ -277,18 +331,19 @@ def scan_documents( ) return - scan_batch_thread_func = _get_scan_documents_thread_func(ctx, is_git_diff, is_commit_range, scan_parameters) - # Presigned single-file upload is async-only; a --sync scan must stay on the batched inline path # so it never builds one oversized zip to POST synchronously. should_use_sync_flow = _should_use_sync_flow(ctx.info_name, scan_type, ctx.obj['sync']) if should_use_presigned_upload(scan_type) and not should_use_sync_flow: errors, local_scan_results = _run_presigned_upload_scan( - scan_batch_thread_func, scan_type, documents_to_scan, progress_bar, printer + ctx, is_git_diff, is_commit_range, scan_parameters, documents_to_scan, progress_bar, printer ) else: errors, local_scan_results = run_parallel_batched_scan( - scan_batch_thread_func, scan_type, documents_to_scan, progress_bar=progress_bar + _get_scan_documents_thread_func(ctx, is_git_diff, is_commit_range, scan_parameters), + scan_type, + documents_to_scan, + progress_bar=progress_bar, ) try_set_aggregation_report_url_if_needed(ctx, scan_parameters, ctx.obj['client'], scan_type) diff --git a/cycode/cli/apps/scan/commit_range_scanner.py b/cycode/cli/apps/scan/commit_range_scanner.py index 70b7e8e4..b1a58202 100644 --- a/cycode/cli/apps/scan/commit_range_scanner.py +++ b/cycode/cli/apps/scan/commit_range_scanner.py @@ -214,6 +214,9 @@ def _scan_commit_range_documents( zip_file_size = from_commit_zipped_documents.size + to_commit_zipped_documents.size + from_commit_zipped_documents.cleanup() + to_commit_zipped_documents.cleanup() + detections_count = relevant_detections_count = 0 if local_scan_result: detections_count = local_scan_result.detections_count diff --git a/cycode/cli/consts.py b/cycode/cli/consts.py index 104cfc9b..e2d09ce1 100644 --- a/cycode/cli/consts.py +++ b/cycode/cli/consts.py @@ -227,6 +227,12 @@ PRESIGNED_LINK_UPLOADED_ZIP_MAX_SIZE_LIMIT_IN_BYTES = 5 * 1024 * 1024 * 1024 # 5 GB (S3 presigned POST limit) PRESIGNED_UPLOAD_SCAN_TYPES = {SAST_SCAN_TYPE, SECRET_SCAN_TYPE} +# the non-ZIP64 central directory stores the entry count in 16 bits; ZIP64 (64-bit interpreters) lifts it +ZIP_MAX_FILES_COUNT = 65_535 + +# the ZIP is built in memory up to this size, and spilled to a temp file beyond it +ZIP_SPOOL_MAX_SIZE_IN_BYTES = 64 * 1024 * 1024 + DEFAULT_ZIP_MAX_SIZE_LIMIT_IN_BYTES = 20 * 1024 * 1024 ZIP_MAX_SIZE_LIMIT_IN_BYTES = { SCA_SCAN_TYPE: 200 * 1024 * 1024, diff --git a/cycode/cli/exceptions/handle_scan_errors.py b/cycode/cli/exceptions/handle_scan_errors.py index 56af186c..0b1e975b 100644 --- a/cycode/cli/exceptions/handle_scan_errors.py +++ b/cycode/cli/exceptions/handle_scan_errors.py @@ -1,3 +1,4 @@ +import zipfile from typing import Optional import typer @@ -26,6 +27,13 @@ def handle_scan_exception(ctx: typer.Context, err: Exception, *, return_exceptio 'Please try ignoring irrelevant paths using the `cycode ignore --by-path` command ' 'and execute the scan again', ), + zipfile.LargeZipFile: CliError( + soft_fail=True, + code='zip_too_large_error', + message='The path you attempted to scan contains too many files to pack into a single archive. ' + 'Scanning such paths requires a 64-bit Python interpreter. ' + 'Please try ignoring irrelevant paths using a .cycodeignore file and execute the scan again', + ), custom_exceptions.FileCollectionError: CliError( soft_fail=False, code='file_collection_error', diff --git a/cycode/cli/files_collector/models/in_memory_zip.py b/cycode/cli/files_collector/models/in_memory_zip.py index 8bb9bf9e..5da13113 100644 --- a/cycode/cli/files_collector/models/in_memory_zip.py +++ b/cycode/cli/files_collector/models/in_memory_zip.py @@ -1,20 +1,48 @@ +import shutil +import tempfile from collections import defaultdict -from io import BytesIO +from os import SEEK_END from pathlib import Path -from sys import getsizeof -from typing import Optional +from typing import IO, Optional from zipfile import ZIP_DEFLATED, ZipFile +from cycode.cli import consts from cycode.cli.user_settings.configuration_manager import ConfigurationManager +from cycode.cli.utils.host_info import is_64bit from cycode.cli.utils.path_utils import concat_unique_id +from cycode.logger import get_logger + +logger = get_logger('ZIP') + +_SPOOL_DIRECTORY_NAME = 'tmp' + + +def _get_spool_directory(configuration_manager: ConfigurationManager) -> Optional[str]: + """Directory to spill big ZIPs into. None falls back to the system temp directory.""" + try: + directory = Path(configuration_manager.global_config_file_manager.get_config_directory_path()) + spool_directory = directory / _SPOOL_DIRECTORY_NAME + spool_directory.mkdir(parents=True, exist_ok=True) + return str(spool_directory) + except OSError as e: + logger.debug('Failed to create the spool directory; falling back to the system one', exc_info=e) + return None class InMemoryZip: def __init__(self) -> None: self.configuration_manager = ConfigurationManager() - self.in_memory_zip = BytesIO() - self.zip = ZipFile(self.in_memory_zip, mode='a', compression=ZIP_DEFLATED, allowZip64=False) + self._spool_max_size = consts.ZIP_SPOOL_MAX_SIZE_IN_BYTES + self._buffer = tempfile.SpooledTemporaryFile( # noqa: SIM115 # closed by cleanup(), lives past close() + max_size=self._spool_max_size, + dir=_get_spool_directory(self.configuration_manager), + ) + + # ZIP64 lifts the 65,535 entries and 4 GiB caps of the original ZIP format. + # It requires 64-bit offsets, so we only enable it on a 64-bit interpreter. + self._allow_zip64 = is_64bit() + self.zip = ZipFile(self._buffer, mode='a', compression=ZIP_DEFLATED, allowZip64=self._allow_zip64) self._files_count = 0 self._extension_statistics = defaultdict(int) @@ -35,17 +63,54 @@ def append(self, filename: str, unique_id: Optional[str], content: str) -> None: def close(self) -> None: self.zip.close() + def cleanup(self) -> None: + """Release the buffer, deleting the spilled temp file if there is one.""" + self._buffer.close() + + def __enter__(self) -> 'InMemoryZip': # noqa: PYI034 # typing.Self needs Python 3.11 + return self + + def __exit__(self, *_: object) -> None: + self.cleanup() + + def stream(self) -> IO[bytes]: + """The whole archive as a file object, rewound. Doesn't copy it into memory. + + Note: before Python 3.11 SpooledTemporaryFile isn't a real IOBase, so the returned object + has no seekable()/readable()/writable(). read/seek/tell work on every supported version. + """ + self._buffer.seek(0) + return self._buffer + def read(self) -> bytes: - self.in_memory_zip.seek(0) - return self.in_memory_zip.read() + self._buffer.seek(0) + return self._buffer.read() def write_on_disk(self, path: 'Path') -> None: with open(path, 'wb') as f: - f.write(self.read()) + shutil.copyfileobj(self.stream(), f) @property def size(self) -> int: - return getsizeof(self.in_memory_zip) + position = self._buffer.tell() + try: + self._buffer.seek(0, SEEK_END) + return self._buffer.tell() + finally: + self._buffer.seek(position) + + @property + def is_rolled_over(self) -> bool: + """Whether the archive outgrew the threshold and moved from memory to the disk. + + SpooledTemporaryFile spills on the write that crosses max_size, and the archive only grows, + so the size says it without reaching into the private _rolled flag. + """ + return self.size > self._spool_max_size + + @property + def allow_zip64(self) -> bool: + return self._allow_zip64 @property def files_count(self) -> int: diff --git a/cycode/cli/utils/host_info.py b/cycode/cli/utils/host_info.py index 064f7823..5ea3e62d 100644 --- a/cycode/cli/utils/host_info.py +++ b/cycode/cli/utils/host_info.py @@ -49,6 +49,11 @@ def _read_text_file(path: str) -> Optional[str]: return None +def is_64bit() -> bool: + """Whether the running Python interpreter is 64-bit (not the OS).""" + return sys.maxsize > 2**32 + + def get_hostname() -> Optional[str]: try: return socket.gethostname() or None diff --git a/tests/cli/commands/scan/test_code_scanner.py b/tests/cli/commands/scan/test_code_scanner.py index 8b4a30b3..24f6a39f 100644 --- a/tests/cli/commands/scan/test_code_scanner.py +++ b/tests/cli/commands/scan/test_code_scanner.py @@ -1,11 +1,18 @@ import os +import zipfile from os.path import normpath from unittest.mock import MagicMock, Mock, patch import pytest from cycode.cli import consts -from cycode.cli.apps.scan.code_scanner import _perform_scan, scan_disk_files, scan_documents +from cycode.cli.apps.scan.code_scanner import ( + _get_scan_documents_thread_func, + _perform_scan, + _run_presigned_upload_scan, + scan_disk_files, + scan_documents, +) from cycode.cli.exceptions import custom_exceptions from cycode.cli.files_collector.file_excluder import _is_file_relevant_for_sca_scan from cycode.cli.files_collector.path_documents import _generate_document @@ -237,3 +244,101 @@ def test_perform_scan_falls_back_to_api_when_presigned_upload_raises_wrapped_err assert result is fallback_result mock_v4_async.assert_called_once() mock_async.assert_called_once() + + +def _presigned_scan_ctx() -> MagicMock: + ctx = MagicMock() + ctx.obj = { + 'client': MagicMock(), + 'scan_type': consts.SECRET_SCAN_TYPE, + 'severity_threshold': None, + 'sync': False, + 'progress_bar': MagicMock(), + } + return ctx + + +@pytest.mark.parametrize( + ('is_64bit', 'expected_skip_batching', 'expected_zip_calls'), + [ + # 64-bit: ZIP64 lifts the entries limit, so everything still goes up as a single ZIP + (True, True, 1), + # 32-bit: the archive can't hold that many entries; route to batches without zipping first + (False, None, 0), + ], +) +@patch('cycode.cli.apps.scan.code_scanner.run_parallel_batched_scan') +@patch('cycode.cli.apps.scan.code_scanner.zip_documents') +@patch('cycode.cli.apps.scan.code_scanner.is_64bit') +def test_run_presigned_upload_scan_routes_by_files_count( + mock_is_64bit: Mock, + mock_zip_documents: Mock, + mock_run_parallel_batched_scan: Mock, + is_64bit: bool, + expected_skip_batching: bool, + expected_zip_calls: int, +) -> None: + mock_is_64bit.return_value = is_64bit + documents = [Document(f'file_{index}.txt', 'content') for index in range(consts.ZIP_MAX_FILES_COUNT + 1)] + + _run_presigned_upload_scan(_presigned_scan_ctx(), False, False, {}, documents, MagicMock(), MagicMock()) + + assert mock_zip_documents.call_count == expected_zip_calls + assert mock_run_parallel_batched_scan.call_args.kwargs.get('skip_batching') is expected_skip_batching + + +@patch('cycode.cli.apps.scan.code_scanner.run_parallel_batched_scan') +@patch('cycode.cli.apps.scan.code_scanner.zip_documents') +def test_run_presigned_upload_scan_falls_back_to_batches_on_large_zip_file( + mock_zip_documents: Mock, mock_run_parallel_batched_scan: Mock +) -> None: + # safety net: zipfile raises LargeZipFile directly instead of our own ZipTooLargeError + mock_zip_documents.side_effect = zipfile.LargeZipFile('Files count would require ZIP64 extensions') + + _run_presigned_upload_scan( + _presigned_scan_ctx(), False, False, {}, [Document('file.txt', 'content')], MagicMock(), MagicMock() + ) + + assert mock_run_parallel_batched_scan.call_args.kwargs.get('skip_batching') is None + + +@patch('cycode.cli.apps.scan.code_scanner.run_parallel_batched_scan') +@patch('cycode.cli.apps.scan.code_scanner.zip_documents') +@patch('cycode.cli.apps.scan.code_scanner._get_scan_documents_thread_func') +def test_run_presigned_upload_scan_reuses_the_archive_it_built( + mock_get_thread_func: Mock, mock_zip_documents: Mock, mock_run_parallel_batched_scan: Mock +) -> None: + # the archive built to check that everything fits is the one we upload; don't compress it twice + zipped_documents = mock_zip_documents.return_value + + _run_presigned_upload_scan( + _presigned_scan_ctx(), False, False, {}, [Document('file.txt', 'content')], MagicMock(), MagicMock() + ) + + mock_zip_documents.assert_called_once() + assert mock_get_thread_func.call_args.args[-1] is zipped_documents + assert mock_run_parallel_batched_scan.call_args.kwargs.get('skip_batching') is True + + +@patch('cycode.cli.apps.scan.code_scanner._perform_scan') +@patch('cycode.cli.apps.scan.code_scanner.zip_documents') +def test_scan_batch_thread_func_does_not_rezip_a_prezipped_batch( + mock_zip_documents: Mock, mock_perform_scan: Mock +) -> None: + prezipped = MagicMock() + ctx = MagicMock() + ctx.obj = { + 'client': MagicMock(), + 'scan_type': consts.SECRET_SCAN_TYPE, + 'severity_threshold': None, + 'sync': False, + 'progress_bar': MagicMock(), + } + + scan_batch_thread_func = _get_scan_documents_thread_func(ctx, False, False, {}, prezipped) + scan_batch_thread_func([Document('file.txt', 'content')]) + + mock_zip_documents.assert_not_called() + assert mock_perform_scan.call_args.args[1] is prezipped + # the buffer is released once the batch is done with it + prezipped.cleanup.assert_called_once() diff --git a/tests/cli/exceptions/test_handle_scan_errors.py b/tests/cli/exceptions/test_handle_scan_errors.py index fb14bc8a..ed749f6a 100644 --- a/tests/cli/exceptions/test_handle_scan_errors.py +++ b/tests/cli/exceptions/test_handle_scan_errors.py @@ -1,3 +1,4 @@ +import zipfile from typing import TYPE_CHECKING, Any import click @@ -31,6 +32,7 @@ def ctx() -> typer.Context: (custom_exceptions.ScanAsyncError('msg'), True), (custom_exceptions.HttpUnauthorizedError('msg', Response()), True), (custom_exceptions.ZipTooLargeError(1000), True), + (zipfile.LargeZipFile('Files count would require ZIP64 extensions'), True), (custom_exceptions.TfplanKeyError('msg'), True), (custom_exceptions.FileCollectionError('Failed to generate dependencies tree for pom.xml'), None), (git_proxy.get_invalid_git_repository_error()(), None), diff --git a/tests/cli/files_collector/test_in_memory_zip.py b/tests/cli/files_collector/test_in_memory_zip.py index d1790c7c..afd807e5 100644 --- a/tests/cli/files_collector/test_in_memory_zip.py +++ b/tests/cli/files_collector/test_in_memory_zip.py @@ -2,9 +2,19 @@ import zipfile from io import BytesIO +from pathlib import Path +from typing import TYPE_CHECKING +from unittest.mock import Mock +from uuid import uuid4 +import pytest + +from cycode.cli import consts from cycode.cli.files_collector.models.in_memory_zip import InMemoryZip +if TYPE_CHECKING: + from _pytest.monkeypatch import MonkeyPatch + def test_append_with_surrogate_characters() -> None: """Test that surrogate characters are handled gracefully without raising encoding errors.""" @@ -27,3 +37,161 @@ def test_append_with_surrogate_characters() -> None: assert 'more text' in extracted # The surrogate should have been replaced with the replacement character assert '\udc96' not in extracted + + +@pytest.mark.parametrize(('is_64bit', 'expected_allow_zip64'), [(True, True), (False, False)]) +def test_allow_zip64_follows_interpreter_bitness( + monkeypatch: 'MonkeyPatch', is_64bit: bool, expected_allow_zip64: bool +) -> None: + """ZIP64 requires 64-bit offsets, so it's enabled only on a 64-bit interpreter.""" + monkeypatch.setattr('cycode.cli.files_collector.models.in_memory_zip.is_64bit', lambda: is_64bit) + + zip_file = InMemoryZip() + + assert zip_file.allow_zip64 is expected_allow_zip64 + assert zip_file.zip._allowZip64 is expected_allow_zip64 + + +def test_append_more_files_than_non_zip64_limit_on_64bit(monkeypatch: 'MonkeyPatch') -> None: + """With ZIP64 enabled, the archive holds more than 65,535 entries.""" + monkeypatch.setattr('cycode.cli.files_collector.models.in_memory_zip.is_64bit', lambda: True) + + files_count = consts.ZIP_MAX_FILES_COUNT + 1 + zip_file = InMemoryZip() + for index in range(files_count): + zip_file.append(f'file_{index}.txt', None, 'content') + zip_file.close() + + with zipfile.ZipFile(BytesIO(zip_file.read()), 'r') as zf: + assert len(zf.namelist()) == files_count + + assert zip_file.files_count == files_count + + +def test_append_more_files_than_non_zip64_limit_on_32bit(monkeypatch: 'MonkeyPatch') -> None: + """Without ZIP64, appending past 65,535 entries raises instead of silently truncating.""" + monkeypatch.setattr('cycode.cli.files_collector.models.in_memory_zip.is_64bit', lambda: False) + + zip_file = InMemoryZip() + with pytest.raises(zipfile.LargeZipFile): + for index in range(consts.ZIP_MAX_FILES_COUNT + 1): + zip_file.append(f'file_{index}.txt', None, 'content') + + +def test_size_is_the_real_archive_length() -> None: + """The size guard and the reported zip_size must reflect the actual bytes, not an approximation.""" + zip_file = InMemoryZip() + zip_file.append('test.txt', None, 'content' * 1000) + zip_file.close() + + assert zip_file.size == len(zip_file.read()) + + +def test_stays_in_memory_below_the_spool_threshold(monkeypatch: 'MonkeyPatch') -> None: + """Ordinary scans never touch the disk.""" + monkeypatch.setattr(consts, 'ZIP_SPOOL_MAX_SIZE_IN_BYTES', 1024 * 1024) + + zip_file = InMemoryZip() + zip_file.append('test.txt', None, 'content') + zip_file.close() + + assert zip_file.is_rolled_over is False + + +def test_rolls_over_to_disk_above_the_spool_threshold(monkeypatch: 'MonkeyPatch') -> None: + """A big archive spills into a temp file instead of growing the heap, and still reads back intact.""" + monkeypatch.setattr(consts, 'ZIP_SPOOL_MAX_SIZE_IN_BYTES', 1024) + + zip_file = InMemoryZip() + for index in range(100): + # random-ish content so deflate can't compress it away below the threshold + zip_file.append(f'file_{index}.txt', None, str(uuid4()) * 100) + zip_file.close() + + assert zip_file.is_rolled_over is True + with zipfile.ZipFile(BytesIO(zip_file.read()), 'r') as zf: + assert len(zf.namelist()) == 100 + + zip_file.cleanup() + + +def test_spills_into_the_cycode_configuration_directory(monkeypatch: 'MonkeyPatch', tmp_path: Path) -> None: + """The temp file lands under the Cycode configuration directory, not in an arbitrary location.""" + monkeypatch.setattr(Path, 'home', lambda: tmp_path) + monkeypatch.setattr(consts, 'ZIP_SPOOL_MAX_SIZE_IN_BYTES', 1) + + zip_file = InMemoryZip() + zip_file.append('test.txt', None, 'content') + zip_file.close() + + assert zip_file.is_rolled_over is True + assert (tmp_path / consts.CYCODE_CONFIGURATION_DIRECTORY / 'tmp').is_dir() + + zip_file.cleanup() + + +def test_falls_back_to_the_system_temp_directory(monkeypatch: 'MonkeyPatch') -> None: + """An unwritable configuration directory (read-only CI runner) must not fail the scan.""" + monkeypatch.setattr(consts, 'ZIP_SPOOL_MAX_SIZE_IN_BYTES', 1) + monkeypatch.setattr( + 'cycode.cli.files_collector.models.in_memory_zip.Path.mkdir', + Mock(side_effect=PermissionError('read-only file system')), + ) + + zip_file = InMemoryZip() + zip_file.append('test.txt', None, 'content') + zip_file.close() + + with zipfile.ZipFile(BytesIO(zip_file.read()), 'r') as zf: + assert zf.read('test.txt') == b'content' + + zip_file.cleanup() + + +def test_cleanup_is_idempotent() -> None: + """Callers release the buffer in a finally block; a double release must not raise.""" + zip_file = InMemoryZip() + zip_file.append('test.txt', None, 'content') + zip_file.close() + + zip_file.cleanup() + zip_file.cleanup() + + +def test_used_as_a_context_manager() -> None: + zip_file = InMemoryZip() + with zip_file as managed: + managed.append('test.txt', None, 'content') + managed.close() + content = managed.read() + + assert len(content) > 0 + + +def test_write_on_disk_matches_the_archive(tmp_path: Path) -> None: + """The --debug dump streams the archive out instead of copying it through memory.""" + zip_file = InMemoryZip() + zip_file.append('test.txt', None, 'content') + zip_file.close() + + zip_file_path = tmp_path / 'dump.zip' + zip_file.write_on_disk(zip_file_path) + + assert zip_file_path.read_bytes() == zip_file.read() + + +@pytest.mark.parametrize('spool_max_size', [1, 512, 1024 * 1024]) +def test_is_rolled_over_matches_the_interpreter(monkeypatch: 'MonkeyPatch', spool_max_size: int) -> None: + """Our size-based check must agree with CPython's own (private) rollover flag on every version.""" + monkeypatch.setattr(consts, 'ZIP_SPOOL_MAX_SIZE_IN_BYTES', spool_max_size) + + zip_file = InMemoryZip() + for index in range(20): + zip_file.append(f'file_{index}.txt', None, str(uuid4()) * 10) + zip_file.close() + + rolled_by_interpreter = getattr(zip_file._buffer, '_rolled', None) + if rolled_by_interpreter is not None: + assert zip_file.is_rolled_over is rolled_by_interpreter + + zip_file.cleanup()