diff --git a/.github/workflows/python-package.yml b/.github/workflows/python-package.yml index dc6c5ce..482e500 100644 --- a/.github/workflows/python-package.yml +++ b/.github/workflows/python-package.yml @@ -110,8 +110,29 @@ jobs: - name: Test with pytest env: QT_QPA_PLATFORM: offscreen + # --cov-fail-under is the real, in-repo coverage gate (deterministic and + # independent of any external service). It's a conservative floor: the + # full suite measures ~87% on a given OS today (coverage.py only counts + # the host backend + shared code — the two foreign backends aren't + # imported, so they don't drag the number down per-platform). 65 leaves + # headroom for a less-covered backend / Python-version variance while + # still catching any large regression. Ratchet it up once the per-OS + # numbers settle on Codecov. run: | - pytest tests -v -s -x --cov=PyMemoryEditor --cov-report=term + pytest tests -v -s -x --cov=PyMemoryEditor --cov-report=term --cov-report=xml --cov-fail-under=65 + - name: Upload coverage to Codecov + # Upload from every matrix cell so Codecov merges each OS's backend + # coverage into one combined view (each platform exercises a different + # win32/linux/macos backend). Informational only — see codecov.yml — so a + # flaky upload never blocks the merge; the hard gate is --cov-fail-under + # above. Runs even when tests fail so partial coverage stays visible. + if: always() + uses: codecov/codecov-action@v5 + with: + files: ./coverage.xml + flags: ${{ runner.os }}-py${{ matrix.python-version }} + token: ${{ secrets.CODECOV_TOKEN }} + fail_ci_if_error: false # Isolated job for the optional NumPy-accelerated scan fast path (the # `[speed]` extra). The matrix above deliberately runs WITHOUT NumPy so the @@ -158,5 +179,14 @@ jobs: - name: Test with pytest (NumPy fast path) env: QT_QPA_PLATFORM: offscreen + # Same conservative coverage gate as the main matrix (see that step's note). run: | - pytest tests -v -s -x --cov=PyMemoryEditor --cov-report=term + pytest tests -v -s -x --cov=PyMemoryEditor --cov-report=term --cov-report=xml --cov-fail-under=65 + - name: Upload coverage to Codecov + if: always() + uses: codecov/codecov-action@v5 + with: + files: ./coverage.xml + flags: speed-Linux-py3.12 + token: ${{ secrets.CODECOV_TOKEN }} + fail_ci_if_error: false diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index be87a86..5356628 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -10,7 +10,7 @@ source venv/bin/activate # On Windows: venv\Scripts\activate make install-dev # pip install -e ".[dev]" ``` -The `dev` extra includes `pytest`, `pytest-cov`, `flake8`, `mypy`, `build` and `twine`. +The `dev` extra installs the full toolchain and the GUI deps. The `Makefile` is the single source of truth for the dev commands below — run `make help` to see every target (docs, build, coverage, etc.). The raw command diff --git a/PyMemoryEditor/__init__.py b/PyMemoryEditor/__init__.py index 4e5ab72..635eb8b 100644 --- a/PyMemoryEditor/__init__.py +++ b/PyMemoryEditor/__init__.py @@ -18,6 +18,7 @@ from .process.abstract import AbstractProcess from .process.errors import ( AmbiguousProcessNameError, + BitnessDetectionError, ClosedProcess, ProcessIDNotExistsError, ProcessNotFoundError, @@ -85,6 +86,7 @@ __all__ = ( "AbstractProcess", "AmbiguousProcessNameError", + "BitnessDetectionError", "ClosedProcess", "MemoryRegion", "MemoryRegionSnapshot", diff --git a/PyMemoryEditor/app/_auto_refresh_dialog.py b/PyMemoryEditor/app/_auto_refresh_dialog.py new file mode 100644 index 0000000..d49ea76 --- /dev/null +++ b/PyMemoryEditor/app/_auto_refresh_dialog.py @@ -0,0 +1,131 @@ +# -*- coding: utf-8 -*- +""" +Shared base for the read-only, auto-refreshing table dialogs (Memory Map, +Modules, Threads). + +All three did the same dance — run a one-shot read off the UI thread, repopulate +a sortable table when it returns, poll on a timer, and tear the worker down +safely on close — each with its own near-identical ``_*Worker(QThread)`` and +lifecycle boilerplate. That triplicated the fiddly bits (the in-flight guard +that self-throttles the poll, the detach-on-shutdown safety) where a fix to one +copy wouldn't reach the others. This owns the lifecycle once; subclasses provide +only what differs: how to fetch the data, and how to render it. +""" +from typing import Callable, Optional + +from PySide6.QtCore import QThread, QTimer, Signal +from PySide6.QtWidgets import QDialog + +from ._widgets import shutdown_worker_thread + + +class _DataWorker(QThread): + """Runs a no-arg ``fetch`` callable off the UI thread, once.""" + + ready = Signal(object) + failed = Signal(str) + + def __init__(self, fetch: Callable[[], object], parent=None): + super().__init__(parent) + self._fetch = fetch + + def run(self) -> None: + try: + data = self._fetch() + except Exception as exc: # noqa: BLE001 — surfaced to the UI as a string + self.failed.emit(str(exc)) + return + self.ready.emit(data) + + +class AutoRefreshTableDialog(QDialog): + """Owns the fetch-worker lifecycle + auto-refresh timer for a table dialog. + + Subclasses must: + + * call ``super().__init__(process, refresh_interval_ms=..., parent=...)``, + then build their UI, then ``refresh()``, then ``_start_auto_refresh()``; + * implement :meth:`_fetch_data` (runs in the worker thread) and + :meth:`_on_data_ready` (runs on the UI thread); + * implement :meth:`_on_data_failed` and (optionally) :meth:`_set_loading_hint`. + """ + + def __init__(self, process, *, refresh_interval_ms: int, parent=None): + super().__init__(parent) + self._process = process + self._worker: Optional[_DataWorker] = None + self._has_data = False + self._refresh_interval_ms = refresh_interval_ms + self._auto_timer: Optional[QTimer] = None + + def _start_auto_refresh(self) -> None: + """Begin polling. Call once, after the first ``refresh()``. + + The ``refresh()`` in-flight guard self-throttles the cadence to however + long a fetch actually takes, so the timer can't stack workers on a slow + target. + """ + self._auto_timer = QTimer(self) + self._auto_timer.setInterval(self._refresh_interval_ms) + self._auto_timer.timeout.connect(self.refresh) + self._auto_timer.start() + + # ------------------------------------------------------------------ # + # Hooks for subclasses + # ------------------------------------------------------------------ # + + def _fetch_data(self): + """Read the data to display. Runs in the worker thread; may raise.""" + raise NotImplementedError + + def _on_data_ready(self, data) -> None: + """Render ``data`` into the table. Runs on the UI thread.""" + raise NotImplementedError + + def _on_data_failed(self, message: str) -> None: + """Report a failed fetch. Runs on the UI thread.""" + raise NotImplementedError + + def _set_loading_hint(self) -> None: + """Optionally show a one-time loading message before the first fetch.""" + + # ------------------------------------------------------------------ # + # Lifecycle (shared) + # ------------------------------------------------------------------ # + + def refresh(self) -> None: + # Skip if a fetch is already in flight — the timer would otherwise stack + # workers on a slow target. This self-throttles to the real fetch time. + if self._worker is not None and self._worker.isRunning(): + return + + # Loading hint only before the first successful fetch; the periodic + # refresh updates silently to avoid flicker. + if not self._has_data: + self._set_loading_hint() + + worker = _DataWorker(self._fetch_data, self) + worker.ready.connect(self._handle_ready) + worker.failed.connect(self._on_data_failed) + worker.finished.connect(self._on_worker_finished) + self._worker = worker + worker.start() + + def _handle_ready(self, data) -> None: + self._has_data = True + self._on_data_ready(data) + + def _on_worker_finished(self) -> None: + worker = self._worker + self._worker = None + if worker is not None: + worker.deleteLater() + + def closeEvent(self, event): # noqa: N802 — Qt naming + if self._auto_timer is not None: + self._auto_timer.stop() + # Unhook + join the worker; if it can't stop in time it's detached + # rather than destroyed under us. + shutdown_worker_thread(self._worker, wait_ms=1000) + self._worker = None + super().closeEvent(event) diff --git a/PyMemoryEditor/app/_widgets.py b/PyMemoryEditor/app/_widgets.py index a4ed742..8257ca4 100644 --- a/PyMemoryEditor/app/_widgets.py +++ b/PyMemoryEditor/app/_widgets.py @@ -6,7 +6,7 @@ previously appeared duplicated across several dialog modules. """ -from typing import List, Optional +from typing import Callable, Iterable, List, Optional, Tuple from PySide6.QtCore import Qt, QThread from PySide6.QtGui import QStandardItem @@ -90,3 +90,73 @@ def parse_hex_address(text: str) -> Optional[int]: return int(cleaned, 16) except (TypeError, ValueError): return None + + +def parse_offsets(texts: Iterable[str]) -> Optional[List[int]]: + """Parse pointer-chain offset tokens (in order) from raw field strings. + + Empty tokens are skipped; every remaining token is read as hex (with or + without ``0x``). Returns ``None`` if any non-empty token can't be parsed — + the caller treats that as "invalid chain, do nothing". Pure (no Qt) so it + can be unit-tested directly. + """ + offsets: List[int] = [] + for text in texts: + if not text: + continue + parsed = parse_hex_address(text) + if parsed is None: + # parse_hex_address handles the 0x form; fall back to a plain hex + # int for ambiguous tokens like "10" (the dialog treats offsets as + # hex throughout). + try: + parsed = int(text, 16) + except ValueError: + return None + offsets.append(parsed) + return offsets + + +def resolve_base_address( + text: str, module_lookup: Callable[[str], Optional[int]] +) -> Tuple[Optional[int], Optional[str]]: + """Resolve a pointer-chain base field into an absolute address. + + Accepts either a plain hex address (``0x14010F4F4``) or Cheat-Engine's + ``"module"+0xoffset`` form (``"libpython3.12.dylib"+0x4ED3D0``); for the + latter the module's current load base is looked up via ``module_lookup`` + (a ``name -> base | None`` callable) and the offset added, so a saved + pointer-scan path resolves correctly despite ASLR. + + Returns ``(address, None)`` on success or ``(None, error_message)`` on + failure — the caller renders ``error_message`` in a dialog. Pure (no Qt). + """ + text = text.strip() + + if "+" in text: + name_part, _, offset_part = text.partition("+") + module_name = name_part.strip().strip('"').strip("'").strip() + offset = parse_hex_address(offset_part) + if offset is None: + try: + offset = int(offset_part.strip(), 16) + except ValueError: + return None, ( + "The offset after '+' must be hex " + '(e.g. "game.exe"+0x10F4F4).' + ) + module_base = module_lookup(module_name) + if module_base is None: + return None, ( + f"Module {module_name!r} is not loaded in this process.\n\n" + "Open Tools → Modules to see the exact names available." + ) + return module_base + offset, None + + base = parse_hex_address(text) + if base is None: + return None, ( + 'Base must be hex (0x14010F4F4) or "module"+0xoffset ' + '(e.g. "game.exe"+0x10F4F4).' + ) + return base, None diff --git a/PyMemoryEditor/app/cheat_poll_worker.py b/PyMemoryEditor/app/cheat_poll_worker.py index ee0cfb3..cf4b84b 100644 --- a/PyMemoryEditor/app/cheat_poll_worker.py +++ b/PyMemoryEditor/app/cheat_poll_worker.py @@ -11,6 +11,7 @@ by row index means deletes/reorders between snapshot and signal can't apply a value to the wrong row. """ +import logging from typing import Any, Dict, List, Optional, Tuple from PySide6.QtCore import QMutex, QMutexLocker, QThread, Signal @@ -18,6 +19,16 @@ from PyMemoryEditor import AbstractProcess +# Child of the "PyMemoryEditor" logger, so the Log Console (which attaches a +# handler to "PyMemoryEditor") picks these up via propagation. +_LOG = logging.getLogger(__name__) + + +# Identity tuple for an entry: (address, pytype, length). Same key the cheat +# table uses to match worker results back to a row across reorders/deletes. +_EntryKey = Tuple[int, type, int] + + # Threshold above which the per-tick refresh collapses N read_process_memory # calls into one search_by_addresses batch. Below this the per-entry path is # simpler and roughly equivalent in syscalls (search_by_addresses still has @@ -38,9 +49,19 @@ class _CheatPollWorker(QThread): with ``(address, pytype, length, value)`` tuples for the UI to render. The worker also handles the freeze write itself, so the syscall never crosses thread boundaries. + + A frozen value is re-written every tick (~10 Hz). When that write *fails* + — a protected page, the target exiting — the worker can't pop a dialog per + tick, so it instead tracks the failing entries and emits ``freeze_failed`` + with the current ``{key: "ErrorType: message"}`` map. The signal fires only + when that set *changes* (a freeze starts or stops failing), so the UI gets a + persistent cue without 10 Hz spam, and the first failure of each entry is + logged once. Previously the exception was swallowed silently, so a freeze + that never landed looked identical to one that did. """ values_ready = Signal(object) # list[tuple[int, type, int, Any]] + freeze_failed = Signal(object) # dict[_EntryKey, str] — current failing freezes def __init__(self, process: AbstractProcess, parent=None): super().__init__(parent) @@ -48,6 +69,11 @@ def __init__(self, process: AbstractProcess, parent=None): self._mutex = QMutex() self._snapshot: List[Tuple[int, type, int, Any, bool]] = [] self._stop = False + # Entries whose freeze write is currently failing → last error string. + # Touched only by the worker thread (run() / _poll_once), so no lock. + self._freeze_failures: Dict[_EntryKey, str] = {} + # Last map handed to the UI, so run() can emit only on change. + self._last_emitted_failures: Dict[_EntryKey, str] = {} def update_snapshot( self, snapshot: List[Tuple[int, type, int, Any, bool]] @@ -77,6 +103,14 @@ def run(self) -> None: if results: self.values_ready.emit(results) + # Emit the freeze-failure state only when it changed since the last + # tick — a frozen page that keeps failing shouldn't fire the signal + # 10×/second, but the UI must learn the moment one starts or stops + # failing (including recovering back to an empty map). + if self._freeze_failures != self._last_emitted_failures: + self._last_emitted_failures = dict(self._freeze_failures) + self.freeze_failed.emit(dict(self._freeze_failures)) + QThread.msleep(TICK_INTERVAL_MS) def _poll_once( @@ -93,6 +127,14 @@ def _poll_once( freeze_by_addr[(*key, address)] = (frozen_value, is_frozen) results: List[Tuple[int, type, int, Any]] = [] + + # Recompute the failing-freeze set from scratch this tick. Rebuilding + # (rather than mutating in place) prunes entries that were deleted or + # unfrozen since the last tick for free; `previous` is kept only to log + # the *first* failure of each entry instead of once per tick. + previous = self._freeze_failures + new_failures: Dict[_EntryKey, str] = {} + for (pytype, length), addresses in groups.items(): values_by_address: Optional[Dict[int, Any]] = None if len(addresses) >= _BATCH_THRESHOLD: @@ -118,16 +160,30 @@ def _poll_once( current = None if is_frozen and frozen_value is not None: + key: _EntryKey = (address, pytype, length) try: self._process.write_process_memory( address, pytype, length, frozen_value ) current = frozen_value - except Exception: # noqa: BLE001 - pass + except Exception as exc: # noqa: BLE001 + # A freeze write that fails every tick must not be + # swallowed: record it (surfaced to the UI via + # freeze_failed) and log the first occurrence. `current` + # keeps the value we actually read, so the table shows + # the value drifting — the visible symptom of the freeze + # not taking hold. + message = "%s: %s" % (type(exc).__name__, exc) + new_failures[key] = message + if key not in previous: + _LOG.warning( + "Freeze write failed at 0x%X (%s, %dB): %s", + address, pytype.__name__, length, message, + ) results.append((address, pytype, length, current)) + self._freeze_failures = new_failures return results diff --git a/PyMemoryEditor/app/cheat_table.py b/PyMemoryEditor/app/cheat_table.py index 62f1fae..580dedb 100644 --- a/PyMemoryEditor/app/cheat_table.py +++ b/PyMemoryEditor/app/cheat_table.py @@ -21,7 +21,7 @@ from typing import Dict, List, Optional, Tuple from PySide6.QtCore import Qt, QTimer, Signal -from PySide6.QtGui import QAction, QKeySequence, QShortcut +from PySide6.QtGui import QAction, QBrush, QColor, QKeySequence, QShortcut from PySide6.QtWidgets import ( QAbstractItemView, QCheckBox, @@ -84,6 +84,7 @@ def __init__(self, process: AbstractProcess, parent=None): # the UI thread isn't blocked when the target is slow. self._poller = _CheatPollWorker(process, self) self._poller.values_ready.connect(self._on_values_ready) + self._poller.freeze_failed.connect(self._on_freeze_failed) self._poller.start() # A short cadence to push fresh entry snapshots into the worker. This @@ -253,7 +254,7 @@ def _write_row(self, row: int, entry: CheatEntry) -> None: self._table.setItem(row, self.COL_TYPE, type_item) value_item = QTableWidgetItem(self._value_text_for(entry)) - value_item.setToolTip("Double-click to write a new value into the process.") + value_item.setToolTip(self._VALUE_TOOLTIP) self._table.setItem(row, self.COL_VALUE, value_item) def _value_text_for(self, entry: CheatEntry) -> str: @@ -384,6 +385,45 @@ def _on_values_ready(self, results) -> None: finally: self._suspend_signals = False + # Foreground for a row whose frozen value can't be written back. A muted + # red that stays readable on both the light and dark table palettes. + _FREEZE_FAIL_COLOR = QColor("#d9534f") + _VALUE_TOOLTIP = "Double-click to write a new value into the process." + + def _on_freeze_failed(self, failures: dict) -> None: + """Flag rows whose freeze write is failing; clear those that recovered. + + The poller emits the full current failure map (identity key → + ``"ErrorType: message"``) only when it changes, so reconciling every row + against it here is cheap and keeps the cue in sync: a red value cell plus + an explanatory tooltip while a freeze keeps failing, restored to normal + the moment the write lands again. Without this, a frozen value that the + backend silently refuses to write looked identical to one that worked. + """ + rows_by_key: Dict[Tuple[int, type, int], int] = {} + for row, entry in enumerate(self._entries): + rows_by_key[(entry.address, entry.spec.pytype, entry.length)] = row + + # setForeground / setToolTip flip data roles on the value cell, which + # would re-enter _on_cell_changed and try to "write" the unchanged text. + self._suspend_signals = True + try: + for key, row in rows_by_key.items(): + item = self._table.item(row, self.COL_VALUE) + if item is None: + continue + message = failures.get(key) + if message is not None: + item.setForeground(QBrush(self._FREEZE_FAIL_COLOR)) + item.setToolTip("Freeze write is failing — %s" % message) + else: + # Clear the override (Qt falls back to the palette default) + # and restore the normal tooltip. + item.setData(Qt.ForegroundRole, None) + item.setToolTip(self._VALUE_TOOLTIP) + finally: + self._suspend_signals = False + def _editing_row(self) -> int: """Return the row currently being edited, or -1 if none.""" if self._table.state() != QAbstractItemView.EditingState: @@ -635,8 +675,17 @@ def _on_export(self) -> None: if not filename: return payload = {"entries": [entry.to_dict() for entry in self._entries]} - with open(filename, "w", encoding="utf-8") as handle: - json.dump(payload, handle, indent=2) + try: + with open(filename, "w", encoding="utf-8") as handle: + json.dump(payload, handle, indent=2) + except OSError as exc: + # A read-only target, a full disk or a permission error must not + # crash the app — surface it the way _on_import does for reads. + QMessageBox.critical(self, "Export", f"Could not write file:\n\n{exc}") + return + QMessageBox.information( + self, "Export", f"Saved {len(self._entries)} entr(y/ies) to:\n\n{filename}" + ) def _on_import(self) -> None: filename, _ = QFileDialog.getOpenFileName( diff --git a/PyMemoryEditor/app/main_window.py b/PyMemoryEditor/app/main_window.py index 78f28a2..6af63a7 100644 --- a/PyMemoryEditor/app/main_window.py +++ b/PyMemoryEditor/app/main_window.py @@ -89,6 +89,12 @@ def __init__(self, process: AbstractProcess): self._hex_viewers: List[MemoryViewerDialog] = [] self._proc_name = self._read_proc_name() + # Identity stamp captured at attach time. The heartbeat compares it so a + # recycled PID (the target dies and the OS hands its PID to an unrelated + # process) is treated as "exited" instead of silently retargeting a + # stranger's memory. None = couldn't read it (we then fall back to a + # plain pid_exists liveness check). + self._proc_create_time = self._read_proc_create_time() self.setWindowTitle(self._window_title()) self.setWindowIcon(app_icon()) self.resize(1280, 780) @@ -465,7 +471,11 @@ def _fill_initial_values(self, request: ScanRequest) -> None: # ones the pattern already located. Skip the auto-refresh and leave # the value column empty — the user can promote rows to the cheat # table for a live preview there. - if request.spec.is_pattern: + # + # A *regex* match is the exception: it has a real byte_length (the max + # match width) and reads as bytes, so we do fill the value column with + # the matched text (see _fmt_regex_match). + if request.spec.is_pattern and not request.spec.is_regex: return self._on_update_values(request) @@ -538,8 +548,8 @@ def _open_memory_map(self) -> None: self._memory_map.activateWindow() def _on_memory_map_closed(self, _result: int) -> None: - # Adopt the dialog's snapshot as the cached one — the user pressed - # Refresh in there, the data is fresh. + # Adopt the dialog's snapshot as the cached one — it auto-refreshes on a + # timer while open, so its last snapshot is fresh. if self._memory_map is not None: snap = self._memory_map.snapshot() if snap: @@ -760,8 +770,38 @@ def _read_proc_name(self) -> str: except (psutil.NoSuchProcess, psutil.AccessDenied, psutil.ZombieProcess): return "" - def _check_process_alive(self) -> None: + def _read_proc_create_time(self) -> Optional[float]: + """The target's process creation time, or None if it can't be read. + + Used as a cheap identity token: the PID alone is reused by the OS, but + (PID, create_time) together uniquely identify a live process. + """ + try: + return psutil.Process(self._process.pid).create_time() + except (psutil.NoSuchProcess, psutil.AccessDenied, psutil.ZombieProcess): + return None + + def _target_is_alive(self) -> bool: + """True only if the *same* process we attached to is still running. + + Guards against PID reuse: a bare ``pid_exists`` would see a recycled PID + as alive and let scans/writes hit an unrelated process. When we have a + creation-time stamp, require it to still match. + """ if not psutil.pid_exists(self._process.pid): + return False + if self._proc_create_time is None: + return True # no stamp to compare — best-effort liveness only + try: + return ( + psutil.Process(self._process.pid).create_time() + == self._proc_create_time + ) + except (psutil.NoSuchProcess, psutil.AccessDenied, psutil.ZombieProcess): + return False + + def _check_process_alive(self) -> None: + if not self._target_is_alive(): self._heartbeat.stop() self._scanner.set_busy(True) self._status.showMessage("Target process exited — operations disabled.") @@ -791,6 +831,7 @@ def _change_process(self) -> None: self._process = picker.process self._proc_name = self._read_proc_name() + self._proc_create_time = self._read_proc_create_time() self.setWindowTitle(self._window_title()) self._process_badge.setText(self._process_badge_text()) self._region_snapshot = None diff --git a/PyMemoryEditor/app/memory_map_dialog.py b/PyMemoryEditor/app/memory_map_dialog.py index 3309145..e4403d3 100644 --- a/PyMemoryEditor/app/memory_map_dialog.py +++ b/PyMemoryEditor/app/memory_map_dialog.py @@ -18,12 +18,11 @@ import sys from typing import List, Optional -from PySide6.QtCore import Qt, QThread, QTimer, Signal +from PySide6.QtCore import Qt, Signal from PySide6.QtGui import QFont, QGuiApplication, QStandardItem, QStandardItemModel from PySide6.QtWidgets import ( QAbstractItemView, QComboBox, - QDialog, QHBoxLayout, QHeaderView, QLabel, @@ -37,26 +36,8 @@ from PyMemoryEditor import AbstractProcess, MemoryRegion, MemoryRegionSnapshot -from ._widgets import NumericItem, shutdown_worker_thread - - -class _SnapshotWorker(QThread): - """Background thread that runs ``snapshot_memory_regions()`` off the UI.""" - - snapshot_ready = Signal(object) # List[MemoryRegion] - snapshot_failed = Signal(str) - - def __init__(self, process: AbstractProcess, parent=None): - super().__init__(parent) - self._process = process - - def run(self) -> None: - try: - snapshot = self._process.snapshot_memory_regions() - except Exception as exc: # noqa: BLE001 - self.snapshot_failed.emit(str(exc)) - return - self.snapshot_ready.emit(snapshot) +from ._auto_refresh_dialog import AutoRefreshTableDialog +from ._widgets import NumericItem def _format_size(size: int) -> str: @@ -176,17 +157,18 @@ def _region_shared(region) -> str: return "Shared" if region.is_shared else "Private" -class MemoryMapDialog(QDialog): +class MemoryMapDialog(AutoRefreshTableDialog): """Shows the output of ``get_memory_regions()`` in a sortable table.""" # qulonglong: 64-bit addresses overflow Qt's default int (C++ signed 32-bit). open_hex_viewer = Signal("qulonglong", "qulonglong") # (address, length) def __init__(self, process: AbstractProcess, parent=None): - super().__init__(parent) - self._process = process + # Auto-refresh the region list so allocations / frees (and any other + # mapping changes in the target) appear without a manual refresh; the + # refresh() guard self-throttles on a slow target. + super().__init__(process, refresh_interval_ms=1000, parent=parent) self._snapshot: List[MemoryRegion] = [] - self._worker: Optional[_SnapshotWorker] = None # allocate/free are a Windows/macOS capability (Linux raises # NotImplementedError); the controls are disabled there. @@ -200,14 +182,7 @@ def __init__(self, process: AbstractProcess, parent=None): self._build_ui() self.refresh() - - # Auto-refresh the region list so allocations / frees (and any other - # mapping changes in the target) appear without a manual refresh. The - # refresh() guard self-throttles if a snapshot takes longer than this. - self._auto_timer = QTimer(self) - self._auto_timer.setInterval(1000) - self._auto_timer.timeout.connect(self.refresh) - self._auto_timer.start() + self._start_auto_refresh() def _build_ui(self) -> None: layout = QVBoxLayout(self) @@ -366,27 +341,13 @@ def snapshot(self) -> MemoryRegionSnapshot: """ return MemoryRegionSnapshot(self._snapshot) - def refresh(self) -> None: - # Skip if a snapshot is already in flight — the 1000ms auto-refresh timer - # would otherwise stack workers on a slow (huge) target. This makes the - # refresh self-throttle to however long a snapshot actually takes. - if self._worker is not None and self._worker.isRunning(): - return - - # Only show the loading hint before the first snapshot; on the periodic - # refresh the count label updates silently to avoid flicker. The action - # controls stay enabled — disabling them every 1000ms would be unusable. - if not self._snapshot: - self._count_label.setText("Loading memory regions…") + def _set_loading_hint(self) -> None: + self._count_label.setText("Loading memory regions…") - worker = _SnapshotWorker(self._process, self) - worker.snapshot_ready.connect(self._on_snapshot_ready) - worker.snapshot_failed.connect(self._on_snapshot_failed) - worker.finished.connect(self._on_worker_finished) - self._worker = worker - worker.start() + def _fetch_data(self): + return self._process.snapshot_memory_regions() - def _on_snapshot_ready(self, snapshot) -> None: + def _on_data_ready(self, snapshot) -> None: # Preserve the MemoryRegionSnapshot tag — scans that reuse this cache # via the ``memory_regions=`` kwarg rely on the isinstance check to # skip the per-call ``sorted(...)`` step. @@ -478,26 +439,12 @@ def _select_address(self, address: int, *, scroll: bool = True) -> None: self._table.scrollTo(self._model.index(row, 0)) return - def _on_snapshot_failed(self, message: str) -> None: + def _on_data_failed(self, message: str) -> None: self._count_label.setText("Failed to read memory regions.") QMessageBox.critical( self, "Memory Map", f"Failed to read memory regions:\n\n{message}" ) - def _on_worker_finished(self) -> None: - worker = self._worker - self._worker = None - if worker is not None: - worker.deleteLater() - - def closeEvent(self, event): # noqa: N802 — Qt naming - self._auto_timer.stop() - # Unhook + join the snapshot worker; if it can't stop in time it's - # detached rather than destroyed under us. - shutdown_worker_thread(self._worker, wait_ms=1000) - self._worker = None - super().closeEvent(event) - def _selected_region(self) -> Optional[MemoryRegion]: rows = self._table.selectionModel().selectedRows() if not rows: diff --git a/PyMemoryEditor/app/memory_viewer_dialog.py b/PyMemoryEditor/app/memory_viewer_dialog.py index 89cf99b..2f2f646 100644 --- a/PyMemoryEditor/app/memory_viewer_dialog.py +++ b/PyMemoryEditor/app/memory_viewer_dialog.py @@ -24,7 +24,8 @@ from PyMemoryEditor import AbstractProcess -from ._widgets import parse_hex_address +from ._auto_refresh_dialog import _DataWorker +from ._widgets import parse_hex_address, shutdown_worker_thread # Child of the "PyMemoryEditor" logger, so the Log Console (which attaches a @@ -54,6 +55,8 @@ def __init__( ): super().__init__(parent) self._process = process + # In-flight read worker (None when idle). Reads run off the UI thread. + self._worker: Optional[_DataWorker] = None self.setWindowTitle(f"Memory Viewer — PID {process.pid}") self.resize(820, 560) @@ -148,19 +151,48 @@ def _parse_address(self) -> Optional[int]: return None def refresh(self) -> None: + # Skip if a read is still running. The auto-refresh timer fires this + # repeatedly, and a large length (up to 65536) on a slow target + # (notably macOS Mach-VM) takes long enough that running it on the UI + # thread froze input — so the read now runs on a worker, and this guard + # self-throttles the cadence to the real read time instead of stacking + # workers. + if self._worker is not None and self._worker.isRunning(): + return addr = self._parse_address() if addr is None: self._status.setText("Enter a hex address first.") return size = int(self._size_spin.value()) - try: - data = self._process.read_process_memory(addr, bytes, size) - # The conversion/format must stay inside the guard: a backend that - # returns a non-buffer object would make bytes(data) raise TypeError, - # which (outside the try) escapes this slot and crashes the app. - if not isinstance(data, (bytes, bytearray)): - data = bytes(data) - except Exception as exc: # noqa: BLE001 — surface every backend error + process = self._process + + def fetch(): + # Runs in the worker thread — never touches Qt widgets. addr/size + # travel back in the result so the rendered dump stays labelled with + # the range actually read even if the user edits the fields mid-read. + # Errors are returned (not raised) so the one result handler renders + # them with the same message/log the synchronous path produced. + try: + data = process.read_process_memory(addr, bytes, size) + # A backend returning a non-buffer object would make bytes(data) + # raise — keep that conversion inside the guard so it surfaces as + # a "Read failed" status rather than crashing the worker. + if not isinstance(data, (bytes, bytearray)): + data = bytes(data) + return addr, size, bytes(data), None + except Exception as exc: # noqa: BLE001 — surface every backend error + return addr, size, None, exc + + worker = _DataWorker(fetch, self) + worker.ready.connect(self._on_read_result) + worker.finished.connect(self._on_worker_finished) + self._worker = worker + worker.start() + + def _on_read_result(self, result) -> None: + """Render a finished read (UI thread). ``result`` is the fetch tuple.""" + addr, size, data, exc = result + if exc is not None: self._dump.setPlainText("") self._status.setText(f"Read failed: {type(exc).__name__}: {exc}") _LOG.warning( @@ -172,9 +204,15 @@ def refresh(self) -> None: ) return - self._dump.setPlainText(_format_hex_dump(addr, bytes(data))) + self._dump.setPlainText(_format_hex_dump(addr, data)) self._status.setText(f"Read {len(data):,} bytes from 0x{addr:X}") + def _on_worker_finished(self) -> None: + worker = self._worker + self._worker = None + if worker is not None: + worker.deleteLater() + def _toggle_auto(self, on: bool) -> None: self._auto_btn.setText("Auto-refresh: On" if on else "Auto-refresh: Off") if on: @@ -230,4 +268,9 @@ def _write_bytes(self) -> None: def closeEvent(self, event) -> None: self._timer.stop() + # Unhook + join the read worker; if it's still wedged in a backend call + # it's detached rather than destroyed under us (same safety the + # auto-refresh dialogs use). + shutdown_worker_thread(self._worker, wait_ms=1000) + self._worker = None super().closeEvent(event) diff --git a/PyMemoryEditor/app/modules_dialog.py b/PyMemoryEditor/app/modules_dialog.py index 3df9560..89670ae 100644 --- a/PyMemoryEditor/app/modules_dialog.py +++ b/PyMemoryEditor/app/modules_dialog.py @@ -20,11 +20,10 @@ """ from typing import List, Optional -from PySide6.QtCore import Qt, QThread, QTimer, Signal +from PySide6.QtCore import Qt, Signal from PySide6.QtGui import QGuiApplication, QStandardItem, QStandardItemModel from PySide6.QtWidgets import ( QAbstractItemView, - QDialog, QHBoxLayout, QHeaderView, QLabel, @@ -38,30 +37,12 @@ from PyMemoryEditor import AbstractProcess, ModuleInfo -from ._widgets import NumericItem, shutdown_worker_thread +from ._auto_refresh_dialog import AutoRefreshTableDialog +from ._widgets import NumericItem from .memory_map_dialog import _format_size -class _ModulesWorker(QThread): - """Background thread that runs ``process.get_modules()`` off the UI.""" - - modules_ready = Signal(object) # List[ModuleInfo] - modules_failed = Signal(str) - - def __init__(self, process: AbstractProcess, parent=None): - super().__init__(parent) - self._process = process - - def run(self) -> None: - try: - modules = list(self._process.get_modules()) - except Exception as exc: # noqa: BLE001 - self.modules_failed.emit(str(exc)) - return - self.modules_ready.emit(modules) - - -class ModulesDialog(QDialog): +class ModulesDialog(AutoRefreshTableDialog): """Shows the output of ``get_modules()`` in a sortable, filterable table.""" # qulonglong: 64-bit addresses overflow Qt's default (C++ signed 32-bit) int. @@ -69,24 +50,17 @@ class ModulesDialog(QDialog): resolve_pointer_chain = Signal("qulonglong") # module base address def __init__(self, process: AbstractProcess, parent=None): - super().__init__(parent) - self._process = process + # Auto-refresh so modules loaded/unloaded at runtime appear without a + # manual refresh; the refresh() guard self-throttles on a slow target. + super().__init__(process, refresh_interval_ms=1000, parent=parent) self._modules: List[ModuleInfo] = [] - self._worker: Optional[_ModulesWorker] = None self.setWindowTitle(f"Modules — PID {process.pid}") self.resize(820, 560) self._build_ui() self.refresh() - - # Auto-refresh so modules loaded/unloaded at runtime appear without a - # manual refresh. The refresh() guard self-throttles if an enumeration - # takes longer than this interval. - self._auto_timer = QTimer(self) - self._auto_timer.setInterval(1000) - self._auto_timer.timeout.connect(self.refresh) - self._auto_timer.start() + self._start_auto_refresh() def _build_ui(self) -> None: layout = QVBoxLayout(self) @@ -154,26 +128,13 @@ def _build_ui(self) -> None: self._table.customContextMenuRequested.connect(self._show_context_menu) layout.addWidget(self._table, 1) - def refresh(self) -> None: - # Skip if an enumeration is already in flight — the 1000ms auto-refresh - # timer would otherwise stack workers. This self-throttles to however - # long get_modules() actually takes. - if self._worker is not None and self._worker.isRunning(): - return - - # Loading hint only before the first list; on the periodic refresh the - # count updates silently to avoid flicker. - if not self._modules: - self._count_label.setText("Enumerating modules…") + def _set_loading_hint(self) -> None: + self._count_label.setText("Enumerating modules…") - worker = _ModulesWorker(self._process, self) - worker.modules_ready.connect(self._on_modules_ready) - worker.modules_failed.connect(self._on_modules_failed) - worker.finished.connect(self._on_worker_finished) - self._worker = worker - worker.start() + def _fetch_data(self): + return list(self._process.get_modules()) - def _on_modules_ready(self, modules) -> None: + def _on_data_ready(self, modules) -> None: self._modules = list(modules) self._apply_filter() @@ -244,18 +205,12 @@ def _select_address(self, address: int) -> None: self._table.selectRow(row) return - def _on_modules_failed(self, message: str) -> None: + def _on_data_failed(self, message: str) -> None: self._count_label.setText("Failed to enumerate modules.") QMessageBox.critical( self, "Modules", f"Failed to enumerate modules:\n\n{message}" ) - def _on_worker_finished(self) -> None: - worker = self._worker - self._worker = None - if worker is not None: - worker.deleteLater() - def _selected_module(self) -> Optional[dict]: rows = self._table.selectionModel().selectedRows() if not rows: @@ -325,11 +280,3 @@ def _emit_hex_viewer_request(self) -> None: # Cap the initial view to keep the hex widget responsive on big modules. size = min(module["size"] or 4096, 4096) self.open_hex_viewer.emit(module["base_address"], size) - - def closeEvent(self, event): # noqa: N802 — Qt naming - self._auto_timer.stop() - # Unhook + join the enumeration worker; if it can't stop in time it's - # detached rather than destroyed under us. - shutdown_worker_thread(self._worker, wait_ms=1000) - self._worker = None - super().closeEvent(event) diff --git a/PyMemoryEditor/app/pointer_chain_dialog.py b/PyMemoryEditor/app/pointer_chain_dialog.py index bf52c80..c261cfb 100644 --- a/PyMemoryEditor/app/pointer_chain_dialog.py +++ b/PyMemoryEditor/app/pointer_chain_dialog.py @@ -46,7 +46,7 @@ from PyMemoryEditor import AbstractProcess -from ._widgets import parse_hex_address +from ._widgets import parse_hex_address, parse_offsets, resolve_base_address from .value_types import VALUE_TYPES, ValueTypeSpec, find_spec @@ -319,23 +319,7 @@ def _update_remove_buttons(self) -> None: def _read_offsets(self) -> Optional[List[int]]: """Collect non-empty offsets in order; return None if any one is invalid.""" - offsets: List[int] = [] - for field in self._offset_fields: - text = field.text() - if not text: - continue - parsed = parse_hex_address(text) - if parsed is None: - # parse_hex_address only handles full hex addresses with or - # without ``0x``; try a plain hex int as a fallback for tokens - # like ``"10"`` that look ambiguous (decimal vs hex). The - # whole dialog treats offsets as hex. - try: - parsed = int(text, 16) - except ValueError: - return None - offsets.append(parsed) - return offsets + return parse_offsets(field.text() for field in self._offset_fields) def _on_value_type_changed(self, label: str) -> None: spec = find_spec(label) @@ -376,49 +360,15 @@ def _resolve_base(self, text: str) -> Optional[int]: """ Parse the Base field into an absolute address. - Accepts either a plain hex address (``0x14010F4F4``) or Cheat-Engine's - ``"module"+0xoffset`` form (``"libpython3.12.dylib"+0x4ED3D0``), looking - the module's current load base up via ``get_modules()`` and adding the - offset — so a saved pointer-scan path (module + module_offset) can be - pasted straight in and resolves correctly despite ASLR. Shows a specific - warning and returns ``None`` on failure. + Thin Qt wrapper over the pure :func:`resolve_base_address`: it owns the + module lookup and renders the error in a dialog; all the parsing rules + (hex vs ``"module"+0xoffset``) live in (and are unit-tested via) the + pure helper. """ - text = text.strip() - - if "+" in text: - name_part, _, offset_part = text.partition("+") - module_name = name_part.strip().strip('"').strip("'").strip() - offset = parse_hex_address(offset_part) - if offset is None: - try: - offset = int(offset_part.strip(), 16) - except ValueError: - QMessageBox.warning( - self, - "Resolve", - "The offset after '+' must be hex (e.g. \"game.exe\"+0x10F4F4).", - ) - return None - module_base = self._lookup_module_base(module_name) - if module_base is None: - QMessageBox.warning( - self, - "Resolve", - f"Module {module_name!r} is not loaded in this process.\n\n" - "Open Tools → Modules to see the exact names available.", - ) - return None - return module_base + offset - - base = parse_hex_address(text) - if base is None: - QMessageBox.warning( - self, - "Resolve", - 'Base must be hex (0x14010F4F4) or "module"+0xoffset ' - '(e.g. "game.exe"+0x10F4F4).', - ) - return base + address, error = resolve_base_address(text, self._lookup_module_base) + if error is not None: + QMessageBox.warning(self, "Resolve", error) + return address def _on_resolve(self) -> None: base_text = self._base_edit.text().strip() diff --git a/PyMemoryEditor/app/scan_worker.py b/PyMemoryEditor/app/scan_worker.py index bf3ad35..9906246 100644 --- a/PyMemoryEditor/app/scan_worker.py +++ b/PyMemoryEditor/app/scan_worker.py @@ -22,8 +22,8 @@ from PyMemoryEditor import AbstractProcess, MemoryRegion, ScanTypesEnum -from .scan_types import NextScanType, ScanType -from .value_types import ValueTypeSpec +from .scan_types import NextScanType, NO_VALUE_SCAN_TYPES, ScanType +from .value_types import parse_value, ValueTypeSpec _LOG = logging.getLogger(__name__) @@ -73,6 +73,86 @@ class ScanRequest: memory_regions: Optional[Sequence[MemoryRegion]] = None +def build_scan_request( + spec: ValueTypeSpec, + scan_type: ScanType, + *, + value_text: str, + second_value_text: str = "", + length_spin_value: Optional[int] = None, + writeable_only: bool = False, + with_value: bool = True, +) -> ScanRequest: + """ + Assemble a :class:`ScanRequest` from raw field values, with no Qt. + + This is the pure core of ``ScannerPanel._build_request`` lifted out of the + widget so the request-assembly rules (pattern short-circuit, the + str-ignores-length override, range parsing, the no-value scan types) can be + unit-tested without a ``QApplication``. The widget keeps only the bits that + are genuinely UI: reading the fields and showing a ``QMessageBox`` on the + ``ValueError`` raised here. + + :raises ValueError: if a value/pattern fails to parse (message is + user-facing — the caller picks the dialog title from ``spec.is_pattern``). + """ + # Pattern path — value is the pattern, scan_type is always EXACT. For an + # IDA pattern the length is irrelevant (derived from the pattern); for a + # regex it carries byte_length (the match width) from the Length field. + if spec.is_pattern: + value, length = parse_value( + spec, value_text, length_spin_value if spec.is_regex else None + ) + return ScanRequest( + spec=spec, + length=int(length), + scan_type=ScanTypesEnum.EXACT_VALUE, + value=None if not with_value else value, + writeable_only=writeable_only, + ) + + # String (UTF-8) ignores the length field: pass None so parse_value derives + # the buffer width from the typed text's UTF-8 byte length. Byte Array still + # honours the user-set override. + length_override = ( + length_spin_value + if spec.accepts_length_override and spec.pytype is not str + else None + ) + + # Increased/Decreased/Changed/Unchanged compare current vs previous and need + # no target value — just the value shape (type + length). + if scan_type in NO_VALUE_SCAN_TYPES: + length = length_override if length_override is not None else spec.length + return ScanRequest( + spec=spec, + length=int(length), + scan_type=scan_type, + value=None, + writeable_only=writeable_only, + ) + + value: Any + if scan_type in (ScanTypesEnum.VALUE_BETWEEN, ScanTypesEnum.NOT_VALUE_BETWEEN): + lo, lo_len = parse_value(spec, value_text, length_override) + hi, hi_len = parse_value(spec, second_value_text, length_override) + length = max(lo_len, hi_len) + value = (lo, hi) + else: + value, length = parse_value(spec, value_text, length_override) + + if not with_value: + value = None # Used by callers that only need spec/length/scan_type. + + return ScanRequest( + spec=spec, + length=int(length), + scan_type=scan_type, + value=value, + writeable_only=writeable_only, + ) + + class _BaseWorker(QThread): progress = Signal(float) # 0.0 … 100.0 status = Signal(str) # human status line @@ -99,14 +179,17 @@ def __init__(self, process: AbstractProcess, request: ScanRequest, parent=None): def run(self) -> None: req = self._request try: - # AOB pattern path: req.value is the IDA-style pattern string, - # routed through search_by_pattern. writeable_only doesn't apply + # Pattern path: req.value is the IDA-style string or a bytes regex, + # routed through search_by_pattern. req.length carries byte_length — + # ignored for IDA strings (inferred from the token count), required + # for a regex (its match width). writeable_only doesn't apply # (pattern scan filters by readability internally; restricting to # writable-only would silently miss code-section signatures, which # is the most common AOB use case). if req.spec.is_pattern: generator = self._process.search_by_pattern( req.value, + byte_length=req.length, progress_information=True, memory_regions=req.memory_regions, ) diff --git a/PyMemoryEditor/app/scanner_panel.py b/PyMemoryEditor/app/scanner_panel.py index fb06e08..8316fe2 100644 --- a/PyMemoryEditor/app/scanner_panel.py +++ b/PyMemoryEditor/app/scanner_panel.py @@ -16,7 +16,7 @@ * :pysig:`update_values_requested(ScanRequest)` — re-read values without filtering * :pysig:`cancel_requested()` """ -from typing import Any, Optional +from typing import Optional from PySide6.QtCore import Signal from PySide6.QtWidgets import ( @@ -43,8 +43,8 @@ NextScanType, is_next_scan_type, ) -from .scan_worker import ScanRequest -from .value_types import VALUE_TYPES, find_spec, parse_value +from .scan_worker import build_scan_request, ScanRequest +from .value_types import VALUE_TYPES, find_spec SCAN_TYPE_CHOICES = ( @@ -252,12 +252,14 @@ def _on_type_changed(self, label: str) -> None: return is_pattern = spec.is_pattern + is_regex = spec.is_regex is_string = spec.pytype is str and not is_pattern - # AOB pattern mode reuses the "Value" line for the pattern string and - # hides / forces the rest of the value-shape controls (length, - # second value, scan-type combo) because none of them apply to - # pattern matching. + # Pattern modes reuse the "Value" line for the pattern and force the + # scan-type combo to EXACT (Bigger/Smaller/Between don't apply). The + # IDA form also hides the length (its match width is inferred from the + # token count), but a *regex* has no inferable width, so its Length + # field stays enabled and supplies search_by_pattern's byte_length. # # String (UTF-8) also locks the length field: the buffer width is the # UTF-8 byte length of the typed text (multi-byte aware), so letting the @@ -265,17 +267,29 @@ def _on_type_changed(self, label: str) -> None: # value they entered. The field stays visible as a read-only readout # kept in sync by _sync_string_length / _on_value_text_changed. self._length_spin.setEnabled( - spec.accepts_length_override and not is_pattern and not is_string + (spec.accepts_length_override and not is_pattern and not is_string) + or is_regex ) - if is_pattern: + if is_regex: + self._value_edit.setPlaceholderText( + "e.g. Player[0-9]+ (text regex, matched against UTF-8 memory)" + ) + elif is_pattern: self._value_edit.setPlaceholderText( 'e.g. "48 8B ? ? 00 00" (IDA-style hex with ? wildcards)' ) else: self._value_edit.setPlaceholderText("e.g. 100 or 0x64 or Hello") - if is_pattern: + if is_regex: + # Length = the regex's max match width in bytes (byte_length); it + # drives the chunk overlap so a match straddling a chunk boundary is + # still found. Seed it with the spec's generous default. + self._length_spin.setMaximum(1024) + self._length_spin.setValue(spec.length) + self._length_spin.setSuffix(" bytes (max match width)") + elif is_pattern: # No meaningful length for an AOB pattern; the scanner derives it. self._length_spin.setMaximum(1024) self._length_spin.setValue(1) @@ -369,75 +383,24 @@ def _build_request(self, *, with_value: bool = True) -> Optional[ScanRequest]: _, scan_type = SCAN_TYPE_CHOICES[self._scan_combo.currentIndex()] - # AOB pattern path — value is the pattern string, scan_type is always - # EXACT (the combo was forced + disabled by _on_type_changed), and - # length is irrelevant (the scanner derives it from the pattern). - if spec.is_pattern: - try: - value, length = parse_value(spec, self._value_edit.text()) - except ValueError as exc: - QMessageBox.warning(self, "Invalid pattern", str(exc)) - return None - return ScanRequest( - spec=spec, - length=int(length), - scan_type=ScanTypesEnum.EXACT_VALUE, - value=None if not with_value else value, - writeable_only=self._writable_check.isChecked(), - ) - - # String (UTF-8) ignores the (disabled) length field: pass None so - # parse_value derives the buffer width from the typed text's UTF-8 - # byte length. Byte Array still honours the user-set override. - length_override = ( - self._length_spin.value() - if spec.accepts_length_override and spec.pytype is not str - else None - ) - - # Increased/Decreased/Changed/Unchanged compare current vs previous and - # need no target value — just the value shape (type + length). - if scan_type in NO_VALUE_SCAN_TYPES: - length = length_override if length_override is not None else spec.length - return ScanRequest( - spec=spec, - length=int(length), - scan_type=scan_type, - value=None, + # All the assembly rules live in the pure build_scan_request() (unit + # tested without Qt). The widget keeps only the genuinely-UI parts: + # reading the fields and turning the ValueError into a message box. + try: + return build_scan_request( + spec, + scan_type, + value_text=self._value_edit.text(), + second_value_text=self._second_value_edit.text(), + length_spin_value=self._length_spin.value(), writeable_only=self._writable_check.isChecked(), + with_value=with_value, ) - - value: Any - try: - if scan_type in ( - ScanTypesEnum.VALUE_BETWEEN, - ScanTypesEnum.NOT_VALUE_BETWEEN, - ): - lo, lo_len = parse_value(spec, self._value_edit.text(), length_override) - hi, hi_len = parse_value( - spec, self._second_value_edit.text(), length_override - ) - length = max(lo_len, hi_len) - value = (lo, hi) - else: - value, length = parse_value( - spec, self._value_edit.text(), length_override - ) except ValueError as exc: - QMessageBox.warning(self, "Invalid value", str(exc)) + title = "Invalid pattern" if spec.is_pattern else "Invalid value" + QMessageBox.warning(self, title, str(exc)) return None - if not with_value: - value = None # Used by callers that only need spec/length/scan_type. - - return ScanRequest( - spec=spec, - length=int(length), - scan_type=scan_type, - value=value, - writeable_only=self._writable_check.isChecked(), - ) - def _on_first_scan(self) -> None: _, scan_type = SCAN_TYPE_CHOICES[self._scan_combo.currentIndex()] if is_next_scan_type(scan_type): diff --git a/PyMemoryEditor/app/threads_dialog.py b/PyMemoryEditor/app/threads_dialog.py index f529b63..5c6694f 100644 --- a/PyMemoryEditor/app/threads_dialog.py +++ b/PyMemoryEditor/app/threads_dialog.py @@ -6,19 +6,19 @@ with optional auto-refresh. The intent mirrors Cheat Engine's "Process → Threads" window: you don't typically *act* on threads directly, but seeing them is useful for introspection (how many workers does this game have? -is the main thread alive?). The optional auto-refresh polls at ~1 Hz so -you can watch threads come and go. +is the main thread alive?). The auto-refresh polls every 300 ms so you can +watch threads come and go. -Lives alongside the existing Memory Map dialog — same shape, same patterns -(background worker, toolbar with Refresh, sortable table, Close button). +Lives alongside the Memory Map and Modules dialogs — same shape, same patterns +(shared ``AutoRefreshTableDialog`` base: background worker + auto-refresh timer, +sortable table, Close button). """ -from typing import List, Optional +from typing import List -from PySide6.QtCore import Qt, QThread, QTimer, Signal +from PySide6.QtCore import Qt from PySide6.QtGui import QStandardItem, QStandardItemModel from PySide6.QtWidgets import ( QAbstractItemView, - QDialog, QHBoxLayout, QHeaderView, QLabel, @@ -30,50 +30,26 @@ from PyMemoryEditor import AbstractProcess, ThreadInfo -from ._widgets import NumericItem, shutdown_worker_thread +from ._auto_refresh_dialog import AutoRefreshTableDialog +from ._widgets import NumericItem -class _ThreadsWorker(QThread): - """Background thread that runs ``process.get_threads()`` off the UI.""" - - threads_ready = Signal(object) # List[ThreadInfo] - threads_failed = Signal(str) - - def __init__(self, process: AbstractProcess, parent=None): - super().__init__(parent) - self._process = process - - def run(self) -> None: - try: - threads = list(self._process.get_threads()) - except Exception as exc: # noqa: BLE001 - self.threads_failed.emit(str(exc)) - return - self.threads_ready.emit(threads) - - -class ThreadsDialog(QDialog): +class ThreadsDialog(AutoRefreshTableDialog): """Lists the output of ``get_threads()`` in a sortable table.""" def __init__(self, process: AbstractProcess, parent=None): - super().__init__(parent) - self._process = process + # Auto-refresh at a brisk 300ms — threads spawn and exit often, so a + # quick cadence lets the user watch the churn live. The refresh() guard + # self-throttles if an enumeration takes longer than the interval. + super().__init__(process, refresh_interval_ms=300, parent=parent) self._threads: List[ThreadInfo] = [] - self._worker: Optional[_ThreadsWorker] = None self.setWindowTitle(f"Threads — PID {process.pid}") self.resize(640, 520) self._build_ui() self.refresh() - - # Auto-refresh at a fixed 300ms — threads spawn and exit often, so a - # brisk cadence lets the user watch the churn live. The refresh() guard - # self-throttles if an enumeration takes longer than this interval. - self._timer = QTimer(self) - self._timer.setInterval(300) - self._timer.timeout.connect(self.refresh) - self._timer.start() + self._start_auto_refresh() def _build_ui(self) -> None: layout = QVBoxLayout(self) @@ -126,25 +102,13 @@ def _build_ui(self) -> None: ) layout.addWidget(self._table, 1) - def refresh(self) -> None: - # Skip if an enumeration is in flight — the 300ms timer would otherwise - # stack workers; this self-throttles to however long get_threads() takes. - if self._worker is not None and self._worker.isRunning(): - return - - # Loading hint only before the first list; the periodic refresh updates - # the count silently to avoid flicker. - if not self._threads: - self._count_label.setText("Enumerating threads…") - - worker = _ThreadsWorker(self._process, self) - worker.threads_ready.connect(self._on_threads_ready) - worker.threads_failed.connect(self._on_threads_failed) - worker.finished.connect(self._on_worker_finished) - self._worker = worker - worker.start() - - def _on_threads_ready(self, threads) -> None: + def _set_loading_hint(self) -> None: + self._count_label.setText("Enumerating threads…") + + def _fetch_data(self): + return list(self._process.get_threads()) + + def _on_data_ready(self, threads) -> None: self._threads = list(threads) # Preserve selection + scroll across the rebuild (the list refreshes @@ -200,20 +164,8 @@ def _select_tid(self, tid: int) -> None: self._table.selectRow(row) return - def _on_threads_failed(self, message: str) -> None: + def _on_data_failed(self, message: str) -> None: self._count_label.setText("Failed to enumerate threads.") QMessageBox.critical( self, "Threads", f"Failed to enumerate threads:\n\n{message}" ) - - def _on_worker_finished(self) -> None: - worker = self._worker - self._worker = None - if worker is not None: - worker.deleteLater() - - def closeEvent(self, event): # noqa: N802 — Qt naming - self._timer.stop() - shutdown_worker_thread(self._worker, wait_ms=1000) - self._worker = None - super().closeEvent(event) diff --git a/PyMemoryEditor/app/value_types.py b/PyMemoryEditor/app/value_types.py index ca013b7..0e3d2f4 100644 --- a/PyMemoryEditor/app/value_types.py +++ b/PyMemoryEditor/app/value_types.py @@ -28,6 +28,13 @@ class ValueTypeSpec: # string and the scan-type / length controls are hidden because they # don't apply. is_pattern: bool = False + # When True the pattern is a raw *bytes regex* rather than an IDA-style hex + # string. Implies ``is_pattern``. Unlike the IDA form (whose match width is + # inferred from the token count), a regex's match width can't be inferred, + # so the panel keeps the Length field enabled and uses it as the + # ``byte_length`` (the number of bytes one match consumes) that + # ``search_by_pattern`` requires for regex input. + is_regex: bool = False def _parse_bool(text: str) -> bool: @@ -102,12 +109,60 @@ def _parse_pattern(text: str) -> str: return stripped +def _parse_regex(text: str) -> bytes: + """Validate a *text regex* and return it as the UTF-8 ``bytes`` pattern. + + The user types an ordinary string regex — e.g. ``Player[0-9]+`` — which is + UTF-8 encoded into the bytes pattern that ``search_by_pattern`` matches + against the target's memory (the library treats string memory as UTF-8). + Because the match runs against *bytes*, regex metacharacters operate on a + single **byte**: ``.`` matches one byte (any byte — the scan compiles with + ``re.DOTALL``) and ``\\d`` / ``[A-Z]`` are ASCII-only. A literal non-ASCII + character still matches its full UTF-8 byte sequence, but a metacharacter + like ``.`` only spans one byte of a multibyte character, so quantify those + with care (e.g. ``.+`` rather than ``.``). + + Like ``_parse_pattern`` this is an early "does this compile?" gate so a + malformed regex surfaces a clear ValueError in a dialog instead of blowing + up mid-scan. A regex has no inferable match width, so the scanner pairs this + value with the Length field as the ``byte_length`` (the max match width). + """ + if not text.strip(): + raise ValueError( + "Empty regex. Type a text regex such as 'Player[0-9]+'. " + "Set Length to the maximum match width in bytes." + ) + # Encode verbatim (not stripped) so whitespace inside the regex is honored. + pattern = text.encode("utf-8") + import re + + try: + re.compile(pattern, re.DOTALL) + except re.error as exc: + raise ValueError(f"Invalid regex: {exc}") + return pattern + + def _fmt_bytes(value: bytes) -> str: if value is None: return "" return " ".join(f"{b:02X}" for b in value) +def _fmt_regex_match(value) -> str: + """Format a regex result value (the bytes read at the match address). + + The scanner reads ``byte_length`` bytes at each hit; show them as text up + to the first NUL (C-string style) so a matched string reads cleanly, and + decode leniently so stray non-text bytes don't blow up the table. + """ + if value is None: + return "" + if isinstance(value, bytes): + return value.split(b"\x00", 1)[0].decode("utf-8", "replace") + return str(value) + + def _fmt_int(value): if value is None: return "" @@ -200,6 +255,21 @@ def _fmt_int(value): accepts_length_override=False, is_pattern=True, ), + # Text-regex scan — the "Value" input becomes a string regex (e.g. + # ``Player[0-9]+``) UTF-8 encoded into the bytes pattern, and the Length + # field supplies the ``byte_length`` (max match width) that + # ``search_by_pattern`` requires for regex. Default width is generous so + # typical string matches aren't clipped at a chunk boundary. + ValueTypeSpec( + "Regex (String)", + bytes, + 64, + _parse_regex, + _fmt_regex_match, + accepts_length_override=True, + is_pattern=True, + is_regex=True, + ), ) @@ -219,11 +289,15 @@ def parse_value( """ value = spec.parse(text) length = spec.length - # AOB patterns short-circuit: ``length`` isn't meaningful — the scanner - # derives the byte width from the pattern itself. Return early so the - # bytes/str length-inference rules below don't accidentally trip on the - # pattern string (whose len() counts characters, not target bytes). + # Pattern types short-circuit the bytes/str length-inference rules below + # (which would wrongly count the pattern *source* length). if spec.is_pattern: + # A regex's match width can't be inferred, so it comes from the Length + # field (``byte_length``); fall back to the spec default if unset. + # An IDA pattern derives its width from the pattern itself, so report 0. + if spec.is_regex: + bl = length_override if length_override is not None else spec.length + return value, max(1, int(bl)) return value, 0 if spec.accepts_length_override and length_override is not None: length = max(1, int(length_override)) diff --git a/PyMemoryEditor/linux/functions.py b/PyMemoryEditor/linux/functions.py index 31bab38..571bd0d 100644 --- a/PyMemoryEditor/linux/functions.py +++ b/PyMemoryEditor/linux/functions.py @@ -189,18 +189,29 @@ def get_memory_regions(pid: int) -> Generator["MemoryRegion", None, None]: for line in mapping_file: region_information = line.split() - addressing_range, privileges, offset, device, inode = region_information[ - 0:5 - ] - path = region_information[5] if len(region_information) >= 6 else "" - - start_address, end_address = [ - int(addr, 16) for addr in addressing_range.split("-") - ] - major_id, minor_id = [int(_id, 16) for _id in device.split(":")] - - offset = int(offset, 16) - inode = int(inode) # /proc//maps formats the inode as decimal. + try: + addressing_range, privileges, offset, device, inode = ( + region_information[0:5] + ) + path = region_information[5] if len(region_information) >= 6 else "" + + start_address, end_address = [ + int(addr, 16) for addr in addressing_range.split("-") + ] + major_id, minor_id = [int(_id, 16) for _id in device.split(":")] + + offset = int(offset, 16) + inode = int(inode) # /proc//maps formats the inode as decimal. + except (ValueError, IndexError) as exc: + # A single malformed line (kernel quirk, racing teardown) must + # not abort the whole region walk — skip it and keep going, the + # same log-and-continue contract the Windows/macOS walkers use. + _logger.debug( + "get_memory_regions: skipping unparseable maps line %r: %s", + line, + exc, + ) + continue size = end_address - start_address @@ -294,16 +305,18 @@ def _read_elf_class(path: str) -> Optional[int]: return ident[4] # e_ident[EI_CLASS]: 1 = ELFCLASS32, 2 = ELFCLASS64 -def is_process_64bit(pid: int) -> bool: +def _detect_process_64bit(pid: int) -> Optional[bool]: """ - Return ``True`` if the target process is 64-bit, ``False`` if 32-bit. + Return ``True``/``False`` from the ELF ``EI_CLASS`` byte of the target's + executable, or ``None`` when no header could be read. The raw *mechanism*: + no guessing and no warning — the caller decides what an unknown result + means (the public :func:`is_process_64bit` falls back to the host word size; + ``AbstractProcess.is_64bit`` honors ``strict_bitness``). - Reads the ``EI_CLASS`` byte of the process's ELF executable. The primary - source is ``/proc//exe``; if that symlink can't be read (a different - user without ``CAP_SYS_PTRACE``), it falls back to the first file-backed, - executable mapping in ``/proc//maps`` — the main image or a shared - library, which share the process's bitness. As a last resort it assumes the - host's word size. + The primary source is ``/proc//exe``; if that symlink can't be read (a + different user without ``CAP_SYS_PTRACE``), it falls back to the first + file-backed, executable mapping in ``/proc//maps`` — the main image or + a shared library, which share the process's bitness. """ ei_class = _read_elf_class("/proc/{}/exe".format(pid)) @@ -323,8 +336,28 @@ def is_process_64bit(pid: int) -> bool: return True if ei_class == 1: return False + return None + - # Couldn't determine it — assume the host's word size (the usual case). +def is_process_64bit(pid: int) -> bool: + """ + Return ``True`` if the target process is 64-bit, ``False`` if 32-bit. + + Thin *policy* wrapper over :func:`_detect_process_64bit`: when no ELF class + can be read it assumes the host's word size (the usual case) and warns so a + wrong pointer-width default (used by the pointer APIs) is traceable instead + of a silent mis-detection on a cross-bitness target. + """ + detected = _detect_process_64bit(pid) + if detected is not None: + return detected + + _logger.warning( + "is_process_64bit: could not read the ELF class for pid %d; assuming " + "the host word size. Pointer-width detection may be wrong for a " + "cross-bitness target.", + pid, + ) return ctypes.sizeof(ctypes.c_void_p) == 8 diff --git a/PyMemoryEditor/linux/process.py b/PyMemoryEditor/linux/process.py index c8bb3e0..44f7aa9 100644 --- a/PyMemoryEditor/linux/process.py +++ b/PyMemoryEditor/linux/process.py @@ -16,10 +16,10 @@ from ..process.region import MemoryRegion from ..process.thread_info import ThreadInfo from .functions import ( + _detect_process_64bit, get_memory_regions, get_modules, get_threads, - is_process_64bit, read_process_memory, search_addresses_by_pattern, search_addresses_by_value, @@ -44,6 +44,7 @@ def __init__( permission=None, case_sensitive: bool = True, exact_match: bool = True, + strict_bitness: bool = False, ): """ :param process_name: name of the target process. @@ -56,12 +57,16 @@ def __init__( :param case_sensitive: when False, process_name matching ignores case. :param exact_match: when False, ``process_name`` is matched as a substring (e.g. ``"chrome"`` finds ``"chromium-browser"``). + :param strict_bitness: raise ``BitnessDetectionError`` instead of + guessing the host word size when the target's ELF class can't be + read. See :class:`~PyMemoryEditor.AbstractProcess`. """ super().__init__( process_name=process_name, pid=pid, case_sensitive=case_sensitive, exact_match=exact_match, + strict_bitness=strict_bitness, ) self.__closed = False @@ -85,11 +90,11 @@ def close(self) -> bool: self.__closed = True return True - def _detect_is_64bit(self) -> bool: + def _detect_is_64bit(self) -> Optional[bool]: self.__require_open() - return is_process_64bit(self.pid) + return _detect_process_64bit(self.pid) - def get_memory_regions(self) -> Generator[dict, None, None]: + def get_memory_regions(self) -> Generator[MemoryRegion, None, None]: self.__require_open() return get_memory_regions(self.pid) diff --git a/PyMemoryEditor/macos/functions.py b/PyMemoryEditor/macos/functions.py index 5212ac4..e64893b 100644 --- a/PyMemoryEditor/macos/functions.py +++ b/PyMemoryEditor/macos/functions.py @@ -85,6 +85,34 @@ _MACH_WRITE_MAX_SIZE = 0xFFFFFFFF +# Host page size, used to align the protect-flip span in `_mach_write`. +# 4 KiB on Intel, 16 KiB on Apple Silicon — `mach_vm_protect` operates at page +# granularity, so we must align the address/size we hand it (see below). +try: + _PAGE_SIZE = os.sysconf("SC_PAGE_SIZE") +except (ValueError, OSError, AttributeError): # pragma: no cover - exotic hosts + _PAGE_SIZE = 0x4000 # conservative: the larger (Apple Silicon) page + + +def _page_aligned_span(address: int, size: int) -> Tuple[int, int]: + """ + Return ``(start, length)`` covering ``[address, address+size)`` expanded out + to whole page boundaries. + + ``mach_vm_protect`` works on page granularity: handing it an unaligned + address and a sub-page size lets the kernel round the affected span + *outward*, so a write that straddles a page boundary would have its + protection changed on a wider range than the literal byte range — and the + later restore, if it used the same unaligned ``[address, address+size)``, + could miss part of that range and leave a page slice permanently more + permissive. Aligning both the elevate and the restore to the exact same + page span keeps them symmetric. + """ + start = address & ~(_PAGE_SIZE - 1) + end = (address + size + _PAGE_SIZE - 1) & ~(_PAGE_SIZE - 1) + return start, end - start + + T = TypeVar("T") @@ -172,7 +200,7 @@ def _region_is_shared(task: int, address: int, basic_shared: int) -> bool: def get_memory_regions(task: int) -> Generator[MemoryRegion, None, None]: """ - Yield {address, size, struct} dicts describing each memory region of the task. + Yield a :class:`MemoryRegion` describing each memory region of the task. ``mach_vm_region`` returning :data:`KERN_INVALID_ADDRESS` is the documented way the kernel says "no more regions past this address" — the natural end @@ -335,8 +363,15 @@ def _mach_write(task: int, address: int, local_buffer_address: int, size: int) - original_protection = region.struct.Protection + # Align the protect span to whole pages: mach_vm_protect is page-granular, + # and the restore below must cover the exact same span as the elevate or it + # could leave a slice of a straddled page more permissive than it started. + prot_address, prot_size = _page_aligned_span(address, size) + new_protection = VM_PROT_READ | VM_PROT_WRITE | VM_PROT_COPY - protect_kr = libsystem.mach_vm_protect(task, address, size, 0, new_protection) + protect_kr = libsystem.mach_vm_protect( + task, prot_address, prot_size, 0, new_protection + ) if protect_kr != KERN_SUCCESS: raise OSError( "mach_vm_write failed (kr=%d) and mach_vm_protect could not elevate " @@ -357,7 +392,7 @@ def _mach_write(task: int, address: int, local_buffer_address: int, size: int) - # leaves the target page more permissive than it started, which is an # invisible side-effect the caller should know about. restore_kr = libsystem.mach_vm_protect( - task, address, size, 0, original_protection + task, prot_address, prot_size, 0, original_protection ) if restore_kr != KERN_SUCCESS: message = ( @@ -859,14 +894,14 @@ def read_u32(address: int) -> int: ) -def is_task_64bit(task: int) -> bool: +def _detect_task_64bit(task: int) -> Optional[bool]: """ - Return ``True`` if the target task is 64-bit, ``False`` if 32-bit. - - Reads the Mach-O header magic of a loaded image (every image in a process - shares its bitness): ``MH_MAGIC_64`` means 64-bit, ``MH_MAGIC`` means - 32-bit. macOS has shipped 64-bit only since Catalina (10.15), so if no - image header can be read this defaults to ``True``. + Return ``True``/``False`` from the Mach-O header magic of a loaded image + (every image in a process shares its bitness): ``MH_MAGIC_64`` means 64-bit, + ``MH_MAGIC`` means 32-bit. Returns ``None`` when no image header can be read. + The raw *mechanism*: no guessing and no warning — the caller decides what an + unknown result means (the public :func:`is_task_64bit` defaults to 64-bit; + ``AbstractProcess.is_64bit`` honors ``strict_bitness``). """ for module in get_modules(task): try: @@ -879,8 +914,27 @@ def is_task_64bit(task: int) -> bool: return True if magic == _MH_MAGIC_32: return False + return None + + +def is_task_64bit(task: int) -> bool: + """ + Return ``True`` if the target task is 64-bit, ``False`` if 32-bit. + + Thin *policy* wrapper over :func:`_detect_task_64bit`. macOS has shipped + 64-bit only since Catalina (10.15), so when no image header can be read this + defaults to ``True``; it warns so a wrong pointer-width default (used by the + pointer APIs) is traceable rather than silent. + """ + detected = _detect_task_64bit(task) + if detected is not None: + return detected - # No readable image header — macOS is 64-bit only on every supported release. + _logger.warning( + "is_task_64bit: no readable Mach-O header for task %d; assuming 64-bit " + "(macOS has been 64-bit only since Catalina).", + task, + ) return True diff --git a/PyMemoryEditor/macos/process.py b/PyMemoryEditor/macos/process.py index 70518cf..9e92879 100644 --- a/PyMemoryEditor/macos/process.py +++ b/PyMemoryEditor/macos/process.py @@ -24,13 +24,13 @@ get_modules, get_task_for_pid, get_threads, - is_task_64bit, read_process_memory, release_task, search_addresses_by_pattern, search_addresses_by_value, search_values_by_addresses, write_process_memory, + _detect_task_64bit, ) @@ -55,6 +55,7 @@ def __init__( permission=None, case_sensitive: bool = True, exact_match: bool = True, + strict_bitness: bool = False, ): """ :param process_name: name of the target process. @@ -67,12 +68,16 @@ def __init__( :param case_sensitive: when False, process_name matching ignores case. :param exact_match: when False, ``process_name`` is matched as a substring (e.g. ``"chrome"`` finds ``"Google Chrome"``). + :param strict_bitness: raise ``BitnessDetectionError`` instead of + defaulting to 64-bit when no Mach-O header can be read. See + :class:`~PyMemoryEditor.AbstractProcess`. """ super().__init__( process_name=process_name, pid=pid, case_sensitive=case_sensitive, exact_match=exact_match, + strict_bitness=strict_bitness, ) # `permission` is accepted for cross-platform parity but has no effect @@ -131,11 +136,11 @@ def __del__(self) -> None: # interpreter is shutting down. pass - def _detect_is_64bit(self) -> bool: + def _detect_is_64bit(self) -> Optional[bool]: self.__require_open() - return is_task_64bit(self.__task) + return _detect_task_64bit(self.__task) - def get_memory_regions(self) -> Generator[dict, None, None]: + def get_memory_regions(self) -> Generator[MemoryRegion, None, None]: self.__require_open() return get_memory_regions(self.__task) diff --git a/PyMemoryEditor/process/abstract.py b/PyMemoryEditor/process/abstract.py index a69ba1e..7343ca4 100644 --- a/PyMemoryEditor/process/abstract.py +++ b/PyMemoryEditor/process/abstract.py @@ -1,4 +1,6 @@ # -*- coding: utf-8 -*- +import ctypes +import logging import sys from abc import ABC, abstractmethod from typing import ( @@ -18,6 +20,7 @@ from ..enums import ScanTypesEnum from ..util import UNSET, _check_int_fits +from .errors import BitnessDetectionError from .info import ProcessInfo from .module_info import ModuleInfo from .region import MemoryRegion, MemoryRegionSnapshot @@ -28,6 +31,8 @@ from .remote_pointer import RemotePointer +_logger = logging.getLogger("PyMemoryEditor") + T = TypeVar("T") @@ -44,6 +49,7 @@ def __init__( pid: Optional[int] = None, case_sensitive: bool = True, exact_match: bool = True, + strict_bitness: bool = False, ): """ :param process_name: name of the target process. @@ -54,6 +60,12 @@ def __init__( substring — ``"chrome"`` matches ``"chrome.exe"`` / ``"Google Chrome"``. If more than one process matches, ``AmbiguousProcessNameError`` is raised so you can pick a PID from the list. + :param strict_bitness: when True, :attr:`is_64bit` raises + :class:`~PyMemoryEditor.BitnessDetectionError` if the target's + 32-/64-bit width can't be read from its headers, instead of falling + back to the host word size. Use it when a wrong pointer-width + default would be worse than a hard failure (the pointer APIs rely on + it). Check :attr:`is_bitness_certain` for the non-strict signal. """ self._process_info = ProcessInfo() @@ -75,6 +87,10 @@ def __init__( # Cache for the target's bitness — resolved lazily on first access of # `is_64bit` / `pointer_size` (a syscall per backend) and reused after. self._is_64bit_cache: Optional[bool] = None + # Whether that resolution read the target's headers (True/False) or fell + # back to a host-word-size guess (False). Set alongside the cache. + self._bitness_certain: Optional[bool] = None + self._strict_bitness = strict_bitness def __enter__(self): return self @@ -87,7 +103,7 @@ def pid(self) -> int: return self._process_info.pid @abstractmethod - def _detect_is_64bit(self) -> bool: + def _detect_is_64bit(self) -> Optional[bool]: """ Detect whether the target process is 64-bit. Backend-specific: @@ -98,6 +114,11 @@ def _detect_is_64bit(self) -> bool: * **macOS** — the Mach-O header magic of a loaded image (``MH_MAGIC_64`` vs ``MH_MAGIC``). + Returns ``True``/``False`` when the headers can be read, or ``None`` when + the bitness is *undeterminable* — :attr:`is_64bit` then either falls back + to the host word size or raises (see ``strict_bitness``). This is the raw + mechanism only: implementations must **not** guess or warn here. + Called once by :attr:`is_64bit`, which caches the result. Implementations may assume the process is still open. """ @@ -116,11 +137,54 @@ def is_64bit(self) -> bool: :meth:`resolve_pointer_chain`, :meth:`scan_pointer_paths`, :meth:`get_pointer` and :class:`~PyMemoryEditor.RemotePointer`: leave ``ptr_size`` as ``None`` and the right pointer width (4 or 8) is used. + + When the backend can't read the target's headers, the result depends on + ``strict_bitness`` (passed to the constructor): the default ``False`` + falls back to the **host** word size and logs a WARNING — convenient, + but possibly wrong for a cross-bitness target (a 32-bit process on a + 64-bit host) — while ``True`` raises + :class:`~PyMemoryEditor.BitnessDetectionError`. Either way, + :attr:`is_bitness_certain` reports whether the value was read or guessed. """ if self._is_64bit_cache is None: - self._is_64bit_cache = bool(self._detect_is_64bit()) + detected = self._detect_is_64bit() + if detected is not None: + self._is_64bit_cache = bool(detected) + self._bitness_certain = True + elif self._strict_bitness: + raise BitnessDetectionError(self.pid) + else: + host_is_64bit = ctypes.sizeof(ctypes.c_void_p) == 8 + _logger.warning( + "Could not determine the bitness of process %d from its " + "headers; assuming the host word size (%d-bit). The " + "pointer-width default used by resolve_pointer_chain / " + "RemotePointer / scan_pointer_paths may be wrong for a " + "cross-bitness target — pass ptr_size explicitly, or open " + "the process with strict_bitness=True to raise instead.", + self.pid, + 64 if host_is_64bit else 32, + ) + self._is_64bit_cache = host_is_64bit + self._bitness_certain = False return self._is_64bit_cache + @property + def is_bitness_certain(self) -> bool: + """ + ``True`` if :attr:`is_64bit` was read from the target's own headers, + ``False`` if it fell back to a guess of the host word size because they + couldn't be read. + + When ``False`` the automatic ``ptr_size`` default may be wrong for a + cross-bitness target — pass ``ptr_size`` explicitly to the pointer APIs, + or open the process with ``strict_bitness=True`` to turn the guess into + a :class:`~PyMemoryEditor.BitnessDetectionError` instead. Accessing this + resolves :attr:`is_64bit` if it hasn't been already. + """ + self.is_64bit # force detection (and the strict_bitness check) to run + return bool(self._bitness_certain) + @property def pointer_size(self) -> int: """ @@ -265,18 +329,35 @@ def search_by_value( for the provided value, returning the found addresses. :param pytype: type of value to be queried (bool, int, float, str or bytes). - :param bufflength: value size in bytes (1, 2, 4, 8). Optional — defaults + :param bufflength: value size in bytes — typically 1, 2, 4 or 8, though + any positive width is accepted (for ``int`` an unusual width such as + 3 or 6 is rounded up to the next C integer type). Optional — defaults to ``None``: numeric types (int, float, bool) use their default width (int→4, float→8, bool→1) and ``str`` / ``bytes`` infer it from the encoded length of ``value``. Since it is optional, pass ``value`` by keyword when omitting it: ``search_by_value(int, value=100)``. :param value: value to be queried (bool, int, float, str or bytes). Required. - :param scan_type: the way to compare the values. + :param scan_type: the way to compare the values. ``VALUE_BETWEEN`` and + ``NOT_VALUE_BETWEEN`` are not accepted here and raise ``ValueError``; + use ``search_by_value_between`` instead. :param progress_information: if True, a dictionary with the progress information will be returned. :param writeable_only: if True, search only at writeable memory regions. :param memory_regions: optional snapshot returned by `snapshot_memory_regions()`. Pass it to skip the region enumeration on hot iterative workflows. + :raises ValueError: if ``scan_type`` is ``VALUE_BETWEEN`` or + ``NOT_VALUE_BETWEEN``, or if an ``int`` ``value`` does not fit in + ``bufflength`` bytes. + + .. note:: + This scan always restricts itself to readable, **non-shared** + regions (``writeable_only`` narrows it further to writable ones). + Shared / file-backed mappings (libc text, memory-mapped files) are + always skipped — they're noise a value scan rarely wants, and + excluding them keeps results identical across platforms. The same + filter is applied to a caller-supplied ``memory_regions`` list too + (that argument only skips region *enumeration*, not the filtering), + so shared regions can't be opted back in. """ raise NotImplementedError() @@ -298,7 +379,10 @@ def search_by_pattern( hex string with ``?`` wildcards (``"48 8B ? ? 00"``), a raw bytes regex, or a pre-compiled ``re.Pattern[bytes]``. :param byte_length: required when ``pattern`` is a regex / pre-compiled - Pattern — the number of bytes one match consumes. Ignored for + Pattern — the **maximum** number of bytes one match can consume. It + drives the chunk overlap so a match straddling a chunk boundary is + still found (a variable-width regex like ``Player[0-9]+`` has no + fixed width, so give the largest match you expect). Ignored for IDA-style strings (inferred from the token count). :param progress_information: if True, yields ``(address, info)`` tuples (same shape as ``search_by_value``). @@ -345,7 +429,9 @@ def read_process_memory( :param address: target memory address (ex: 0x006A9EC0). :param pytype: type of the value to be received (bool, int, float, str or bytes). - :param bufflength: value size in bytes (1, 2, 4, 8). For numeric types + :param bufflength: value size in bytes — typically 1, 2, 4 or 8, though + any positive width is accepted (for ``int`` an unusual width such as + 3 or 6 is rounded up to the next C integer type). For numeric types (int, float, bool) you may omit this; defaults are int→4, float→8, bool→1. str and bytes require an explicit size. @@ -824,10 +910,11 @@ def scan_pointer_paths( read-only pointers (e.g. vtables), which is slower and noisier. :param static_ranges: explicit ``(start, size)`` ranges to treat as valid chain bases. Defaults to the image range of every loaded - module. **macOS note:** ``ModuleInfo.size`` there covers only the - ``__TEXT`` segment, so global pointers in ``__DATA`` may fall - outside the default static set — pass ``static_ranges`` explicitly - (or accept reduced static-base coverage) on macOS. + module. On macOS this default already spans **every** Mach-O segment + of each image (not just ``__TEXT``) — see :meth:`_static_image_ranges`, + which the macOS backend overrides — so global pointers in ``__DATA`` + are covered automatically; you do not need to pass ``static_ranges`` + for them. :param max_results: stop after yielding this many paths (``None`` = no cap). Recommended for shallow exploration of large targets. :param memory_regions: optional snapshot from diff --git a/PyMemoryEditor/process/errors.py b/PyMemoryEditor/process/errors.py index 24c67a7..59d1e9b 100644 --- a/PyMemoryEditor/process/errors.py +++ b/PyMemoryEditor/process/errors.py @@ -35,3 +35,24 @@ def __init__(self, process_name: str, pids: Iterable[int]): ) self.process_name = process_name self.pids = pid_list + + +class BitnessDetectionError(PyMemoryEditorError): + """ + Raised when ``strict_bitness=True`` and the target's 32-/64-bit width could + not be read from its own headers (the ELF class on Linux, the Mach-O magic + on macOS, ``IsWow64Process`` on Windows). + + Without strict mode the library would instead fall back to the host word + size — a guess that silently poisons the pointer-width default used by + ``resolve_pointer_chain`` / ``RemotePointer`` / ``scan_pointer_paths`` on a + cross-bitness target. Catch this to pass ``ptr_size`` explicitly instead. + """ + + def __init__(self, pid: int): + super().__init__( + "Could not determine the bitness of process %d from its headers. " + "Pass ptr_size explicitly to the pointer APIs, or open the process " + "without strict_bitness to fall back to the host word size." % pid + ) + self.pid = pid diff --git a/PyMemoryEditor/process/pointer_scan.py b/PyMemoryEditor/process/pointer_scan.py index 886ef64..6a7ab6a 100644 --- a/PyMemoryEditor/process/pointer_scan.py +++ b/PyMemoryEditor/process/pointer_scan.py @@ -191,8 +191,9 @@ def recipe(self) -> Tuple[Optional[str], Optional[int], Tuple[int, ...]]: every run and is deliberately left out. Used to intersect independent scans (see :func:`intersect_pointer_paths`). - Paths with no ``module`` have no portable recipe (only an absolute base - valid for one run) and compare equal only to themselves. + Paths with no ``module`` have no portable recipe — their base is only + valid for one run — so they return ``(None, None, offsets)`` and are + excluded from :func:`intersect_pointer_paths` entirely. """ return (self.module, self.module_offset, self.offsets) diff --git a/PyMemoryEditor/process/region.py b/PyMemoryEditor/process/region.py index b77e081..b4fa4bc 100644 --- a/PyMemoryEditor/process/region.py +++ b/PyMemoryEditor/process/region.py @@ -11,8 +11,9 @@ / ``is_shared``, - the backing file ``path`` (when the platform exposes it cheaply), - and the original platform descriptor in ``struct`` (a - ``MEMORY_BASIC_INFORMATION`` on Windows, the privileges-string struct on - Linux, the VM struct on macOS). + ``MEMORY_BASIC_INFORMATION_32`` / ``MEMORY_BASIC_INFORMATION_64`` on + Windows, the privileges-string ``MEMORY_BASIC_INFORMATION`` on Linux, the + VM struct on macOS). Portable client code never touches ``struct`` — the booleans cover every read/write/execute/shared question. Backends use :func:`make_region` to build @@ -53,9 +54,11 @@ class MemoryRegion: :param address: base address of the region. :param size: region size in bytes. - :param struct: platform-specific descriptor (``MEMORY_BASIC_INFORMATION`` - on Windows / Linux; the macOS VM struct). Portable code should rely on - the boolean fields below instead of poking at this directly. + :param struct: platform-specific descriptor + (``MEMORY_BASIC_INFORMATION_32`` / ``MEMORY_BASIC_INFORMATION_64`` on + Windows; ``MEMORY_BASIC_INFORMATION`` on Linux; the macOS VM struct). + Portable code should rely on the boolean fields below instead of poking + at this directly. :param is_readable: ``True`` when the region can be read. :param is_writable: ``True`` when the region can be written. :param is_executable: ``True`` when the region contains executable code. diff --git a/PyMemoryEditor/process/scanning.py b/PyMemoryEditor/process/scanning.py index 4505aa7..cd0a14f 100644 --- a/PyMemoryEditor/process/scanning.py +++ b/PyMemoryEditor/process/scanning.py @@ -283,8 +283,9 @@ def iter_search_results( # match begins. Read ``bufflength - 1`` extra bytes from the next # chunk so the scan can complete a straddling decode without ever # re-emitting an offset (the scanner only yields offsets in - # ``range(0, chunk_size - bufflength + 1, step)`` from the *augmented* - # size, which still maps to addresses inside the original chunk). + # ``range(0, read_size - bufflength + 1, step)`` over the *augmented* + # read_size, and the clamp below keeps every emitted offset inside the + # original chunk). str_overlap = bufflength - 1 if pytype is str else 0 for region in memory_regions: diff --git a/PyMemoryEditor/util/pattern.py b/PyMemoryEditor/util/pattern.py index 95d9b17..485a0f8 100644 --- a/PyMemoryEditor/util/pattern.py +++ b/PyMemoryEditor/util/pattern.py @@ -59,6 +59,8 @@ def compile_pattern( :raises ValueError: malformed IDA-style token, or ``byte_length`` omitted for a regex / pre-compiled pattern. + :raises TypeError: if ``pattern`` is not a ``str``, ``bytes`` or + ``re.Pattern[bytes]``. """ if isinstance(pattern, re.Pattern): if byte_length <= 0: diff --git a/PyMemoryEditor/util/scan.py b/PyMemoryEditor/util/scan.py index 9dbe0fc..a608698 100644 --- a/PyMemoryEditor/util/scan.py +++ b/PyMemoryEditor/util/scan.py @@ -4,7 +4,7 @@ import struct import sys from bisect import bisect_left -from typing import Generator, Iterable, Literal, Optional, Sequence, Tuple, Type, Union, cast +from typing import Callable, Generator, Iterable, Literal, Optional, Sequence, Tuple, Type, Union, cast from ..enums import ScanTypesEnum from . import scan_numpy @@ -61,10 +61,17 @@ def iter_region_chunks( overhead of a generator state machine in the hot path. Larger regions get a lazy generator that yields aligned chunks. - Chunk sizes are aligned to target_value_size so typed numeric scans don't - miss matches across boundaries. Strings (which can begin at any byte - offset) may miss matches that span chunk boundaries when the region - exceeds max_chunk — rare in practice and documented as a limitation. + Chunk sizes are aligned to target_value_size, so a typed **numeric** scan + (which steps by that size) can never have a value straddle a boundary — no + overlap is needed there. For scans that can match at *any* byte offset — + string value scans, pattern (AOB/regex) scans, and per-address reads — the + scan drivers in ``process.scanning`` (``iter_search_results``, + ``iter_pattern_results``, ``iter_values_for_addresses``) read + ``value_size - 1`` extra bytes from the next chunk and clamp matches to the + original chunk, so a match that straddles a boundary — including the + large-region splits produced here when ``region_size > max_chunk`` — is + still found. These chunks are non-overlapping by design; that overlap is the + consumer's responsibility, not this function's. """ if region_size <= max_chunk: return ((0, region_size),) @@ -207,6 +214,46 @@ def scan_memory_for_exact_value( yield offset +def _make_predicate( + scan_type: ScanTypesEnum, + target: Union[int, float], + start: Union[int, float], + end: Union[int, float], +) -> Callable[[Union[int, float]], bool]: + """ + Return a single-value predicate ``value -> bool`` implementing ``scan_type`` + against the decoded ``target`` (or the ``start`` / ``end`` pair for the + BETWEEN variants). + + This is the ONE place the eight comparison semantics live. All three scan + loops call it — the ``struct.iter_unpack`` numeric fast path, the + ``int.from_bytes`` fallback, and the ordered-string fast path — so a change + to a comparison can't silently diverge between them. (Previously each of + those branches inlined its own copy: sixteen near-identical loop bodies.) + The closure is built once per ``scan_memory`` call, outside the hot loop, so + the per-element cost is a single Python call rather than the re-evaluated + ``scan_type`` chain. The fast/slow equivalence is pinned by + ``tests/test_scan_properties.py``. + """ + if scan_type is ScanTypesEnum.EXACT_VALUE: + return lambda value: value == target + if scan_type is ScanTypesEnum.NOT_EXACT_VALUE: + return lambda value: value != target + if scan_type is ScanTypesEnum.BIGGER_THAN: + return lambda value: value > target + if scan_type is ScanTypesEnum.SMALLER_THAN: + return lambda value: value < target + if scan_type is ScanTypesEnum.BIGGER_THAN_OR_EXACT_VALUE: + return lambda value: value >= target + if scan_type is ScanTypesEnum.SMALLER_THAN_OR_EXACT_VALUE: + return lambda value: value <= target + if scan_type is ScanTypesEnum.VALUE_BETWEEN: + return lambda value: start <= value <= end + if scan_type is ScanTypesEnum.NOT_VALUE_BETWEEN: + return lambda value: not (start <= value <= end) + raise ValueError("Unsupported scan_type: %r" % (scan_type,)) + + def _scan_string_ordered( data: bytes, end: int, @@ -342,50 +389,20 @@ def scan_memory( yield from offsets return + # One predicate (built once, outside the loop) drives every scan_type, + # so the comparison semantics aren't duplicated per branch. The value + # production stays inlined (the C-level struct.iter_unpack), keeping the + # hot loop to a single Python call per element. + predicate = _make_predicate( + scan_type, target_value_decoded, start_target_value, end_target_value + ) unpacker = struct.iter_unpack(fmt, buffer[:total]) offset = 0 step = target_value_size - - if scan_type is ScanTypesEnum.EXACT_VALUE: - for (value,) in unpacker: - if value == target_value_decoded: - yield offset - offset += step - elif scan_type is ScanTypesEnum.NOT_EXACT_VALUE: - for (value,) in unpacker: - if value != target_value_decoded: - yield offset - offset += step - elif scan_type is ScanTypesEnum.BIGGER_THAN: - for (value,) in unpacker: - if value > target_value_decoded: - yield offset - offset += step - elif scan_type is ScanTypesEnum.SMALLER_THAN: - for (value,) in unpacker: - if value < target_value_decoded: - yield offset - offset += step - elif scan_type is ScanTypesEnum.BIGGER_THAN_OR_EXACT_VALUE: - for (value,) in unpacker: - if value >= target_value_decoded: - yield offset - offset += step - elif scan_type is ScanTypesEnum.SMALLER_THAN_OR_EXACT_VALUE: - for (value,) in unpacker: - if value <= target_value_decoded: - yield offset - offset += step - elif scan_type is ScanTypesEnum.VALUE_BETWEEN: - for (value,) in unpacker: - if start_target_value <= value <= end_target_value: - yield offset - offset += step - elif scan_type is ScanTypesEnum.NOT_VALUE_BETWEEN: - for (value,) in unpacker: - if not (start_target_value <= value <= end_target_value): - yield offset - offset += step + for (value,) in unpacker: + if predicate(value): + yield offset + offset += step return # Fallback: strings (byte-by-byte) or numeric with unusual sizes (3/6/7). @@ -402,85 +419,37 @@ def scan_memory( # byte-class prefilter finds those candidates in C, skipping the huge NUL # runs of reserved memory instead of stepping every byte in Python. Numerics # with unusual sizes (3/6/7) decode little-endian and fall through unchanged. + # Build the comparison predicate once — shared by the ordered-string fast + # path below and the byte-by-byte fallback loop, so neither re-inlines the + # eight scan_type branches. + predicate = _make_predicate( + scan_type, target_value_decoded, start_target_value, end_target_value + ) + if is_string: spec = None if first_byte is not None and scan_type is ScanTypesEnum.BIGGER_THAN: - spec = (first_byte, 0xFF, frozenset((first_byte,)), - lambda v: v > target_value_decoded) + spec = (first_byte, 0xFF, frozenset((first_byte,)), predicate) elif first_byte is not None and scan_type is ScanTypesEnum.BIGGER_THAN_OR_EXACT_VALUE: - spec = (first_byte, 0xFF, frozenset((first_byte,)), - lambda v: v >= target_value_decoded) + spec = (first_byte, 0xFF, frozenset((first_byte,)), predicate) elif first_byte is not None and scan_type is ScanTypesEnum.SMALLER_THAN: - spec = (0x00, first_byte, frozenset((first_byte,)), - lambda v: v < target_value_decoded) + spec = (0x00, first_byte, frozenset((first_byte,)), predicate) elif first_byte is not None and scan_type is ScanTypesEnum.SMALLER_THAN_OR_EXACT_VALUE: - spec = (0x00, first_byte, frozenset((first_byte,)), - lambda v: v <= target_value_decoded) + spec = (0x00, first_byte, frozenset((first_byte,)), predicate) elif ( scan_type is ScanTypesEnum.VALUE_BETWEEN and start_first_byte is not None and end_first_byte is not None ): spec = (start_first_byte, end_first_byte, - frozenset((start_first_byte, end_first_byte)), - lambda v: start_target_value <= v <= end_target_value) + frozenset((start_first_byte, end_first_byte)), predicate) if spec is not None: yield from _scan_string_ordered(data, end, target_value_size, *spec) return - if scan_type is ScanTypesEnum.EXACT_VALUE: - for offset in range(0, end, step): - value = int_from_bytes( - data[offset : offset + target_value_size], byte_order, signed=signed - ) - if value == target_value_decoded: - yield offset - elif scan_type is ScanTypesEnum.NOT_EXACT_VALUE: - for offset in range(0, end, step): - value = int_from_bytes( - data[offset : offset + target_value_size], byte_order, signed=signed - ) - if value != target_value_decoded: - yield offset - elif scan_type is ScanTypesEnum.BIGGER_THAN: - for offset in range(0, end, step): - value = int_from_bytes( - data[offset : offset + target_value_size], byte_order, signed=signed - ) - if value > target_value_decoded: - yield offset - elif scan_type is ScanTypesEnum.SMALLER_THAN: - for offset in range(0, end, step): - value = int_from_bytes( - data[offset : offset + target_value_size], byte_order, signed=signed - ) - if value < target_value_decoded: - yield offset - elif scan_type is ScanTypesEnum.BIGGER_THAN_OR_EXACT_VALUE: - for offset in range(0, end, step): - value = int_from_bytes( - data[offset : offset + target_value_size], byte_order, signed=signed - ) - if value >= target_value_decoded: - yield offset - elif scan_type is ScanTypesEnum.SMALLER_THAN_OR_EXACT_VALUE: - for offset in range(0, end, step): - value = int_from_bytes( - data[offset : offset + target_value_size], byte_order, signed=signed - ) - if value <= target_value_decoded: - yield offset - elif scan_type is ScanTypesEnum.VALUE_BETWEEN: - for offset in range(0, end, step): - value = int_from_bytes( - data[offset : offset + target_value_size], byte_order, signed=signed - ) - if start_target_value <= value <= end_target_value: - yield offset - elif scan_type is ScanTypesEnum.NOT_VALUE_BETWEEN: - for offset in range(0, end, step): - value = int_from_bytes( - data[offset : offset + target_value_size], byte_order, signed=signed - ) - if not (start_target_value <= value <= end_target_value): - yield offset + for offset in range(0, end, step): + value = int_from_bytes( + data[offset : offset + target_value_size], byte_order, signed=signed + ) + if predicate(value): + yield offset diff --git a/PyMemoryEditor/win32/functions.py b/PyMemoryEditor/win32/functions.py index adcb91f..381ae62 100644 --- a/PyMemoryEditor/win32/functions.py +++ b/PyMemoryEditor/win32/functions.py @@ -238,7 +238,16 @@ def mbi_class_for_handle(process_handle: int): ok = kernel32.IsWow64Process(process_handle, ctypes.byref(is_wow64)) if not ok: # Conservatively fall back to the host-bitness default rather than fail - # — the caller may not need region info at all. + # — the caller may not need region info at all. Warn, though: if the + # target really is a 32-bit (WOW64) process, the wrong MBI layout makes + # VirtualQueryEx return silently-corrupted region fields, which then + # poison every is_readable/is_writable filter and scan result. + _logger.warning( + "IsWow64Process failed (err=%d); assuming the target matches the " + "host bitness for region queries. Region fields may be wrong if the " + "target is actually a 32-bit (WOW64) process.", + ctypes.get_last_error(), + ) return MEMORY_BASIC_INFORMATION return ( @@ -246,9 +255,13 @@ def mbi_class_for_handle(process_handle: int): ) -def IsProcess64Bit(process_handle: int) -> bool: +def _detect_process_64bit(process_handle: int) -> Optional[bool]: """ - Return ``True`` if the target process is 64-bit, ``False`` if 32-bit. + Return ``True``/``False`` when the target's bitness can be determined, or + ``None`` when ``IsWow64Process`` fails so the answer is unknown. The raw + *mechanism*: no guessing and no warning — the caller decides what an unknown + result means (the public :func:`IsProcess64Bit` falls back to the host + bitness; ``AbstractProcess.is_64bit`` honors ``strict_bitness``). On a 32-bit OS every process is 32-bit. On a 64-bit OS a process is 32-bit exactly when it runs under WOW64 (``IsWow64Process`` returns True); a @@ -261,13 +274,34 @@ def IsProcess64Bit(process_handle: int) -> bool: is_wow64 = ctypes.wintypes.BOOL(0) ok = kernel32.IsWow64Process(process_handle, ctypes.byref(is_wow64)) if not ok: - # Couldn't query — fall back to the OS bitness (the most likely answer - # on a 64-bit host) rather than raise from a simple property access. - return True + return None return not bool(is_wow64.value) +def IsProcess64Bit(process_handle: int) -> bool: + """ + Return ``True`` if the target process is 64-bit, ``False`` if 32-bit. + + Thin *policy* wrapper over :func:`_detect_process_64bit`: when the bitness + can't be queried it falls back to the host bitness (the most likely answer + on a 64-bit host) rather than raise from a simple property access, and warns + so a wrong pointer-width default (used by the pointer APIs) is at least + traceable instead of silently mis-detected. + """ + detected = _detect_process_64bit(process_handle) + if detected is not None: + return detected + + _logger.warning( + "IsWow64Process failed (err=%d); assuming the target is 64-bit " + "(host bitness). Pointer-width detection may be wrong if the target " + "is actually a 32-bit (WOW64) process.", + ctypes.get_last_error(), + ) + return True + + T = TypeVar("T") @@ -301,11 +335,20 @@ def GetMemoryRegions(process_handle: int) -> Generator[MemoryRegion, None, None] consult ``GetLastError``: only fall through for the natural end-of-space case; for any other failure log it and bump the cursor by one page so the walk keeps making progress. + + The walk bounds come from ``GetNativeSystemInfo`` rather than + ``GetSystemInfo``: from a 32-bit (WOW64) Python attached to a 64-bit target, + ``GetSystemInfo`` reports the *caller's* 2/4 GB ceiling, which would stop the + walk early and silently drop every region above it in the target. The native + info reports the true address-space ceiling. For a 32-bit target the extra + range is empty — ``VirtualQueryEx`` returns ERROR_INVALID_PARAMETER past the + 32-bit boundary and the loop terminates there — so using the native ceiling + never over-enumerates. (From a 64-bit Python the two infos are identical.) """ mbi_class = mbi_class_for_handle(process_handle) - mem_region_begin = system_information.lpMinimumApplicationAddress - mem_region_end = system_information.lpMaximumApplicationAddress - page_size = system_information.dwPageSize or 0x1000 + mem_region_begin = _native_system_information.lpMinimumApplicationAddress + mem_region_end = _native_system_information.lpMaximumApplicationAddress + page_size = _native_system_information.dwPageSize or 0x1000 current_address = mem_region_begin diff --git a/PyMemoryEditor/win32/process.py b/PyMemoryEditor/win32/process.py index bf01c5f..748fb19 100644 --- a/PyMemoryEditor/win32/process.py +++ b/PyMemoryEditor/win32/process.py @@ -26,12 +26,12 @@ GetModules, GetProcessHandle, GetThreads, - IsProcess64Bit, ReadProcessMemory, SearchAddressesByPattern, SearchAddressesByValue, SearchValuesByAddresses, WriteProcessMemory, + _detect_process_64bit, ) @@ -93,6 +93,7 @@ def __init__( permission: Union[ProcessOperationsEnum, int] = DEFAULT_PERMISSION, case_sensitive: bool = False, exact_match: bool = True, + strict_bitness: bool = False, ): """ :param process_name: name of the target process. @@ -113,6 +114,7 @@ def __init__( pid=pid, case_sensitive=case_sensitive, exact_match=exact_match, + strict_bitness=strict_bitness, ) self.__closed = False @@ -177,11 +179,11 @@ def close(self) -> bool: raise OSError("CloseHandle failed.") return True - def _detect_is_64bit(self) -> bool: + def _detect_is_64bit(self) -> Optional[bool]: self.__require_open() - return IsProcess64Bit(self.__process_handle) + return _detect_process_64bit(self.__process_handle) - def get_memory_regions(self) -> Generator[dict, None, None]: + def get_memory_regions(self) -> Generator[MemoryRegion, None, None]: self.__require_open() return GetMemoryRegions(self.__process_handle) diff --git a/README.md b/README.md index 13ae79b..eb5f620 100644 --- a/README.md +++ b/README.md @@ -50,7 +50,7 @@ pymemoryeditor ``` For faster scans on large processes, add the `speed` extra. It pulls in NumPy -and automatically vectorizes the numeric scan comparison loop — 10–60× faster on +and automatically vectorizes the numeric scan comparison loop — ~10–30× faster on selective scans: ```bash diff --git a/assets/social-preview.html b/assets/social-preview.html new file mode 100644 index 0000000..0d122eb --- /dev/null +++ b/assets/social-preview.html @@ -0,0 +1,103 @@ + + + + + + + +
+
Pure-Python · ctypes · no native build
+

PyMemoryEditor

+
+ Read, write & scan the memory of any running process — straight from Python. +
+
+
+ + Windows +
+
+ + Linux +
+
+ + macOS +
+
+
+
+ + + + + + + + + + + + + + + + +
+ + diff --git a/assets/social-preview.png b/assets/social-preview.png new file mode 100644 index 0000000..854091a Binary files /dev/null and b/assets/social-preview.png differ diff --git a/codecov.yml b/codecov.yml new file mode 100644 index 0000000..1a63940 --- /dev/null +++ b/codecov.yml @@ -0,0 +1,23 @@ +# Codecov is used for *visibility* only — the enforced coverage gate lives in +# CI as pytest's `--cov-fail-under` (see .github/workflows/python-package.yml), +# which is deterministic and needs no external service. Marking Codecov's status +# checks informational keeps a flaky upload or a transient dip from blocking the +# merge button while still surfacing the trend and per-PR diff coverage. +# +# Coverage is uploaded from every OS/Python cell of the matrix; coverage.py only +# measures the host platform's backend (the other two aren't imported), so the +# merged Codecov view is what shows win32 + linux + macos coverage together. +coverage: + status: + project: + default: + informational: true + patch: + default: + informational: true + +# Group the per-cell uploads (flagged by OS + Python version) so the combined +# report reflects the whole matrix rather than the last upload to land. +comment: + layout: "reach, diff, flags, files" + require_changes: false diff --git a/docs/api/errors.md b/docs/api/errors.md index 4eeb93f..8f4a486 100644 --- a/docs/api/errors.md +++ b/docs/api/errors.md @@ -116,12 +116,12 @@ except AmbiguousProcessNameError as exc: - - + + - +
ExceptionWhen it's raised
TypeErrorNeither process_name nor pid provided to OpenProcess.
ValueErrorInvalid pytype, missing bufflength for str/bytes, invalid ptr_size, malformed pattern, etc.
TypeErrorNeither process_name nor pid provided to OpenProcess; or a scan pattern that is not str, bytes or a compiled re.Pattern.
ValueErrorInvalid pytype, missing bufflength for str/bytes, invalid ptr_size, malformed pattern, byte_length omitted for a regex pattern, etc.
PermissionErrorOS denied access to the target process or a specific region.
OSErrorLow-level read/write failure (e.g. page was freed between scan and read).
NotImplementedErrorallocate_memory / free_memory on Linux.
UserWarningpermission= passed on a non-Windows platform (silently ignored).
UserWarningpermission= passed on a non-Windows platform (ignored, with this warning emitted).
ResourceWarningmacOS: mach_vm_protect failed to restore a page's original protection after a write.
diff --git a/docs/api/openprocess.md b/docs/api/openprocess.md index c8bb597..5f44739 100644 --- a/docs/api/openprocess.md +++ b/docs/api/openprocess.md @@ -123,7 +123,9 @@ with OpenProcess( ``value`` by keyword when omitting it (``write_process_memory(addr, int, value=9999)``). :param value: the value to write. - :returns: the written value. + :returns: the original ``value`` you passed in — **not** the truncated/encoded + form actually written (a capped ``str``/``bytes`` write returns the full + original value). ``` ### Typed shortcuts @@ -262,7 +264,7 @@ identical on every platform. Build a :py:class:`RemotePointer` bound to this process — a live, re-resolving handle. See :doc:`../guide/pointers`. -.. py:method:: scan_pointer_paths(target_address, *, max_depth=5, max_offset=0x400, ptr_size=None, aligned=True, writable_only=True, static_ranges=None, max_results=None, memory_regions=None, progress_callback=None) +.. py:method:: scan_pointer_paths(target_address, *, max_depth=3, max_offset=0x400, ptr_size=None, aligned=True, writable_only=True, static_ranges=None, max_results=None, memory_regions=None, progress_callback=None) Reverse pointer scan — yield :py:class:`PointerPath` recipes that resolve to ``target_address``. See :doc:`../guide/pointer-scan`. diff --git a/docs/api/thread-info.md b/docs/api/thread-info.md index 80c7d3e..5c438ac 100644 --- a/docs/api/thread-info.md +++ b/docs/api/thread-info.md @@ -38,14 +38,17 @@ mean "this platform does not expose that attribute via the API we use". .. py:attribute:: state :type: Optional[str] - Short human-readable state — e.g. ``"R"`` / ``"S"`` on Linux. ``None`` - when not available. + Short human-readable state — e.g. ``"R"`` / ``"S"``. **Only Linux + populates this**; on Windows and macOS it is always ``None`` (those + backends don't fetch per-thread state cheaply). .. py:attribute:: priority :type: Optional[int] - Scheduling priority value as reported by the OS. The scale is - platform-specific; ``None`` when not available. + Scheduling priority value as reported by the OS (scale is + platform-specific). Populated on **Linux and Windows**; always ``None`` on + macOS, whose ``get_threads`` returns Mach thread ports without the extra + per-thread ``thread_info`` call that priority/state would require. .. py:attribute:: raw :type: Any diff --git a/docs/api/utilities.md b/docs/api/utilities.md index 1c8f802..c2064a2 100644 --- a/docs/api/utilities.md +++ b/docs/api/utilities.md @@ -8,6 +8,7 @@ arbitrary region for scanning. ```python from PyMemoryEditor.util import ( resolve_bufflength, + resolve_bufflength_for_value, convert_from_byte_array, value_to_bytes, values_to_bytes, @@ -18,6 +19,7 @@ from PyMemoryEditor.util import ( scan_memory_for_exact_value, PatternLike, DEFAULT_MAX_REGION_CHUNK, + NUMPY_AVAILABLE, ) ``` @@ -32,6 +34,17 @@ from PyMemoryEditor.util import ( :raises ValueError: ``bufflength`` is required for ``pytype=str`` / ``pytype=bytes``. +.. py:function:: resolve_bufflength_for_value(pytype, bufflength, *values) + + Like :py:func:`resolve_bufflength`, but for operations that already carry + the value(s) being matched (the ``search_by_value`` family). When + ``bufflength`` is ``None``: numeric / bool types fall back to the default + width (int→4, float→8, bool→1); ``str`` / ``bytes`` infer the width from the + longest encoded value instead of raising (``str`` encoded as UTF-8), so + ``search_by_value(str, value="hi")`` works without the caller counting + bytes. For a range search the shorter endpoint is NUL-padded up to this + width. + .. py:function:: convert_from_byte_array(byte_array, pytype, length) Convert a ctypes byte array to a Python value of type ``pytype``. String @@ -65,6 +78,8 @@ from PyMemoryEditor.util import ( number of bytes one match consumes. :raises ValueError: malformed IDA-style token, or ``byte_length`` omitted for a regex / pre-compiled pattern. + :raises TypeError: if ``pattern`` is not a ``str``, ``bytes`` or + ``re.Pattern[bytes]``. ``` ### Example @@ -73,7 +88,7 @@ from PyMemoryEditor.util import ( from PyMemoryEditor.util import compile_pattern regex, byte_length = compile_pattern("48 8B ? 00 00") -print(regex.pattern) # b'\\x48\\x8B.\\x00\\x00' +print(regex.pattern) # b'H\x8b.\x00\x00' (re.escape prints 0x48 as 'H') print(byte_length) # 5 ``` @@ -95,9 +110,11 @@ print(byte_length) # 5 .. py:function:: iter_region_chunks(region_size, target_value_size, max_chunk=DEFAULT_MAX_REGION_CHUNK) - Yield ``(offset, chunk_size)`` pairs that walk a single memory region in - bounded-size chunks. Regions up to ``max_chunk`` yield a single chunk; larger - ones are split into **contiguous, non-overlapping** chunks whose size is a + Return an iterable of ``(offset, chunk_size)`` pairs that walk a single + memory region in bounded-size chunks. Regions up to ``max_chunk`` return a + single-element tuple (avoiding generator overhead in the common, hot path); + larger ones return a lazy generator that yields **contiguous, + non-overlapping** chunks whose size is a multiple of ``target_value_size`` so a typed numeric scan never splits a value across a boundary. Boundary handling for *patterns* is done one level up by the scanner (it overlaps consecutive chunks by ``pattern_length - 1`` bytes); @@ -110,6 +127,13 @@ print(byte_length) # 5 Low-level scan kernels used by the backends. Public for advanced use only — the high-level :py:meth:`search_by_value` / :py:meth:`search_by_pattern` methods are the recommended API. + +.. py:data:: NUMPY_AVAILABLE + + ``True`` when NumPy is importable, in which case eligible numeric scans use + the vectorized fast path. Install it via the ``speed`` extra + (``pip install PyMemoryEditor[speed]``). Scan results are identical with or + without it. ``` ## Region predicates @@ -137,8 +161,10 @@ from PyMemoryEditor.process.region import ( .. py:function:: is_region_executable(struct) .. py:function:: is_region_shared(struct) - True/False from a platform descriptor (``MEMORY_BASIC_INFORMATION`` on - Windows/Linux; the VM struct on macOS). For a fully-populated region, + True/False from a platform descriptor + (``MEMORY_BASIC_INFORMATION_32`` / ``MEMORY_BASIC_INFORMATION_64`` on + Windows; ``MEMORY_BASIC_INFORMATION`` on Linux; the VM struct on macOS). For + a fully-populated region, prefer the boolean attributes on :py:class:`MemoryRegion` (``region.is_readable``, etc.). diff --git a/docs/app.md b/docs/app.md index d6395b9..156653f 100644 --- a/docs/app.md +++ b/docs/app.md @@ -40,15 +40,19 @@ name or PID. **🎯 Scanner** - Every `ScanTypesEnum` mode -- All five value types (`int`, `float`, `bool`, `str`, `bytes`) +- Int8 / Int16 / Int32 / Int64, Float / Double, Boolean, String (UTF-8) and + Byte Array value types - Range search -- AOB / byte signature search -- Regex search +- AOB / byte signature search (IDA-style) +- Regex (string) search — a text regex matched against UTF-8 memory. The + Length field sets the maximum match width; matching is byte-wise, so `.` + spans one byte (use `.+` for multibyte characters) **🔁 Refine workflow** - **First Scan → Next Scan** (Cheat Engine style) -- Eight Next Scan comparisons (increased, decreased, changed, unchanged, …) +- Six Next Scan comparisons (increased / decreased / changed / unchanged, plus + increased-by / decreased-by) - Live progress **📋 Cheat table** @@ -65,22 +69,22 @@ name or PID. - Same engine as `scan_pointer_paths` - Save scans to JSON - Rescan / compare scans to narrow them down -- Build live `RemotePointer` from a result +- Send a resolved address straight to the Cheat Table **🗺️ Memory map** - All regions with R/W/X flags -- Source file / module per region (where available) +- Backing file path per region (Linux; blank where the OS doesn't expose it) **🔬 Hex viewer** - Live dump with write-back -- Address goto, navigation +- Go to any address, with auto-refresh **🪵 Log console** - Same stream as `logging.getLogger("PyMemoryEditor")` -- Toggle DEBUG verbosity at runtime +- Pick the log level (DEBUG / INFO / WARNING / ERROR) at runtime @@ -89,14 +93,13 @@ name or PID. ```{admonition} Cross-platform dark theme :class: tip -The app ships with a dark theme that follows the system on macOS and Windows -11 and uses a manual toggle elsewhere. Themes live under -**View → Theme**. +The app ships with several built-in dark themes (Kali Teal by default). Pick one +from the **Theme** button on the toolbar; your choice is remembered between runs. ``` ## Typical workflow -1. **Open a process** from the dialog (or `File → Open Process`). +1. **Open a process** from the startup dialog (or later via `File → Change Process…`). 2. **Run a First Scan**: pick the value type, type the value you can see, hit *First Scan*. 3. **Refine** with Next Scan after the value changes — pick *Exact Value* with diff --git a/docs/guide/allocate-free.md b/docs/guide/allocate-free.md index de8c554..52f1fa2 100644 --- a/docs/guide/allocate-free.md +++ b/docs/guide/allocate-free.md @@ -25,13 +25,18 @@ allocation's size, so you don't have to: process.free_memory(address) ``` -Pass an explicit `size=` only when freeing a region that this object **did not -allocate** (e.g. one inherited from another script): +On **macOS**, pass an explicit `size=` when freeing a region that this object +**did not allocate** (e.g. one inherited from another script), since the size +cannot be looked up from the local allocation table: ```python process.free_memory(address, size=4096) ``` +On **Windows** the `size` argument is always ignored — `MEM_RELEASE` frees the +whole original allocation given only its base address, so a foreign region is +freed the same way (and only if `address` is a true allocation base). + ## Method signatures ```{eval-rst} @@ -135,6 +140,10 @@ Common values: Pass a `VM_PROT_*` bitmask (or leave `None` for the default of read+write): ```python +from PyMemoryEditor.macos.types import ( + VM_PROT_READ, VM_PROT_WRITE, VM_PROT_EXECUTE, +) + # read+write+execute (may fail under the hardened runtime). process.allocate_memory(4096, permission=VM_PROT_READ | VM_PROT_WRITE | VM_PROT_EXECUTE) ``` diff --git a/docs/guide/logging.md b/docs/guide/logging.md index 5132185..8249b59 100644 --- a/docs/guide/logging.md +++ b/docs/guide/logging.md @@ -28,7 +28,7 @@ WARNING PyMemoryEditor: mach_vm_protect could not restore protection at 0x14010 - +
LevelWhen it fires
DEBUGTransient skips during enumeration/scans (pages vanished mid-scan, unreadable chunks, a thread/module/image that couldn't be read).
WARNINGSurprising-but-recovered conditions — currently the macOS mach_vm_protect restore failure after a write to a read-only page.
WARNINGSurprising-but-recovered conditions. Two cases today: (1) the bitness of the target couldn't be detected and a fallback was assumed (a failed IsWow64Process on Windows, or no readable ELF/Mach-O header on Linux/macOS) — pointer-width-dependent APIs may be wrong if the guess is; (2) the macOS mach_vm_protect restore failure after a write to a read-only page.
```{note} @@ -56,11 +56,11 @@ logger.setLevel(logging.DEBUG) The GUI app exposes the same log stream in its **Log Console**: - +
MenuTools → Log Console
MenuTools → Log Console… (Ctrl+L)
-Toggling DEBUG verbosity in the console reveals the same messages the library -sends to the Python logger. +The console has a level selector (DEBUG / INFO / WARNING / ERROR); choosing +`DEBUG` reveals the same messages the library sends to the Python logger. ## macOS write-side-effect warning diff --git a/docs/guide/memory-regions.md b/docs/guide/memory-regions.md index efc17ec..e4da275 100644 --- a/docs/guide/memory-regions.md +++ b/docs/guide/memory-regions.md @@ -36,7 +36,7 @@ Each region is an instance of `MemoryRegion` — an immutable is_writableboolTrue if the region can be written. is_executableboolTrue if the region contains executable code. is_sharedboolTrue if the region is a shared/file-backed mapping. -pathstrFile backing the region (Linux only — empty on Windows/macOS). +pathstrBest-effort path of the file backing the region; populated on Linux (from /proc/<pid>/maps), "" when unknown. structplatform-specificRaw platform descriptor (see below). diff --git a/docs/guide/modules-threads.md b/docs/guide/modules-threads.md index bdfa718..3d5ecf2 100644 --- a/docs/guide/modules-threads.md +++ b/docs/guide/modules-threads.md @@ -128,15 +128,15 @@ with OpenProcess(process_name="game.exe") as process: :no-index: :type: Optional[str] - Short human-readable state (e.g. ``"R"``/``"S"`` on Linux). ``None`` on - platforms that don't surface it. + Short human-readable state (e.g. ``"R"``/``"S"``). **Linux only** — + always ``None`` on Windows and macOS. .. py:attribute:: priority :no-index: :type: Optional[int] - Scheduling priority as reported by the OS. Scale is platform-specific; - ``None`` when not exposed. + Scheduling priority as reported by the OS (scale is platform-specific). + Populated on **Linux and Windows**; always ``None`` on macOS. .. py:attribute:: raw :no-index: diff --git a/docs/guide/pattern-scan.md b/docs/guide/pattern-scan.md index 63b6330..3b9b271 100644 --- a/docs/guide/pattern-scan.md +++ b/docs/guide/pattern-scan.md @@ -44,7 +44,7 @@ email = rb"[A-Za-z0-9._%+\-]+@[A-Za-z0-9.\-]+\.[A-Za-z]{2,}" for address in process.search_by_pattern(email, byte_length=128): raw = process.read_process_memory(address, bytes, 128) - print(address, raw.split(b"\x00", 1)[0].decode("ascii", "replace")) + print(address, raw.split(b"\x00", 1)[0].decode("utf-8", "replace")) ``` ```{admonition} byte_length is required for regex @@ -114,7 +114,7 @@ ipv4 = re.compile(rb"(?SMALLER_THANvalue < target BIGGER_THAN_OR_EXACT_VALUEvalue ≥ target SMALLER_THAN_OR_EXACT_VALUEvalue ≤ target -VALUE_BETWEENmin ≤ value ≤ max (use search_by_value_between) -NOT_VALUE_BETWEENvalue < min or value > max +VALUE_BETWEENmin ≤ value ≤ max (rejected by search_by_value with ValueError — use search_by_value_between) +NOT_VALUE_BETWEENvalue < min or value > max (rejected by search_by_value with ValueError — use search_by_value_between(..., not_between=True)) ```python @@ -136,8 +148,12 @@ for address, value in process.search_by_addresses(int, 4, addresses): print(f"0x{address:X} -> {value}") ``` -If an address falls in an unmapped page, the value is `None` (unless -`raise_error=True`). +If an address isn't backed by any mapped region (it falls in a gap, or its +`[address, address+bufflength)` runs past the end of its region), the value is +**always** `None` — `raise_error` does not turn that into an exception, because +there is nothing there to read. `raise_error=True` only affects an address that +*is* inside a mapped region but whose read fails (e.g. the page vanished): then +it raises `OSError` instead of yielding `None`. ### Method signature @@ -151,7 +167,8 @@ If an address falls in an unmapped page, the value is `None` (unless addresses to read. Pass ``addresses`` by keyword when omitting it. :param Sequence[int] addresses: addresses to inspect. :param bool raise_error: when ``True``, raises ``OSError`` instead of yielding - ``None`` for an unreadable address. + ``None`` for an address that is inside a mapped region but fails to read. + Addresses with no backing region always yield ``None`` regardless. :param memory_regions: optional snapshot. :returns: a generator of ``(address, value)`` tuples. ``` @@ -200,7 +217,8 @@ missing. By default every scan runs in pure Python, with the hottest paths already delegated to C primitives: `bytes.find` for exact matches, `struct.iter_unpack` to decode a region, and a **regex byte-class prefilter** for ordered *string* -comparisons (`BIGGER_THAN` / `SMALLER_THAN` / `VALUE_BETWEEN` on `str`), which +comparisons (`BIGGER_THAN` / `SMALLER_THAN` / `BIGGER_THAN_OR_EXACT_VALUE` / +`SMALLER_THAN_OR_EXACT_VALUE` / `VALUE_BETWEEN` on `str`), which skips the long runs of non-matching bytes in C instead of stepping every offset. What stays in Python is the per-value **comparison loop** of the ordered *numeric* scans: for a multi-megabyte region it boxes and compares millions of @@ -242,9 +260,9 @@ emitting matches. - + - +
ScenarioTypical speedup
Selective scan of a large region (few matches — the usual first scan / refine step)10–60×
Selective scan of a large region (few matches — the usual first scan / refine step)~10–30×
Scan where most values match (e.g. > 0 on mostly-positive data)~2× (result building dominates)
str ordered scans (>, <, between)no NumPy fast path — instead C-accelerated by the regex byte-class prefilter (independent of the speed extra)
str ordered scans (>, <, ≥, ≤, between)no NumPy fast path — instead C-accelerated by the regex byte-class prefilter (independent of the speed extra)
bytes scans, or unusual widths (3/6/7 bytes)no change (no NumPy fast path; pure-Python loop)
EXACT_VALUE via search_by_valuealready bytes.find in C — NumPy not used
diff --git a/docs/index.md b/docs/index.md index f940d41..3a4f55e 100644 --- a/docs/index.md +++ b/docs/index.md @@ -63,7 +63,7 @@ it's the single easiest way to support the project and help others discover it. **🌍 Truly cross-platform** -One identical API on **Windows, Linux and macOS**, 32- and 64-bit. Write your +One identical API on **Windows, Linux and macOS**, 32-bit and 64-bit. Write your script once; it runs everywhere. **🪶 Zero dependencies** @@ -87,7 +87,7 @@ ASLR — save them once and reuse them every launch. **⚡ Optional NumPy acceleration** Add the [`speed`](installation.md#install-with-scan-acceleration-speed) extra -and selective scans get **10–60× faster** — a drop-in fast path, identical +and selective scans get **~10–30× faster** — a drop-in fast path, identical results. **🖥️ A GUI app, included** diff --git a/docs/installation.md b/docs/installation.md index e3882a9..9a57e0c 100644 --- a/docs/installation.md +++ b/docs/installation.md @@ -8,7 +8,7 @@ any platform. - +
Python3.10 or newer
Operating systems🪟 Windows · 🐧 Linux · 🍎 macOS (32-bit and 64-bit)
Operating systems🪟 Windows · 🐧 Linux · 🍎 macOS

@@ -53,7 +53,7 @@ pip install "PyMemoryEditor[speed]" That's the only change required — there is no new API and no flag to toggle. PyMemoryEditor detects NumPy at import time and switches the fast path on; if NumPy is absent it falls back to the pure-Python loop transparently. The results -are **identical** either way — only the speed changes (typically 10–60× faster +are **identical** either way — only the speed changes (typically ~10–30× faster on selective scans of large regions). See [Scan acceleration](guide/searching.md#scan-acceleration-the-speed-extra) for details and benchmarks. diff --git a/docs/quickstart.md b/docs/quickstart.md index d65f3da..632256f 100644 --- a/docs/quickstart.md +++ b/docs/quickstart.md @@ -56,7 +56,7 @@ There's a `read_*` / `write_*` pair for every common type — `read_float`, `read_bool`, `read_uint`, `read_string`, and more: ```python -name = process.read_string(address, 32) # up to 32 bytes, decoded as text +name = process.read_string(address, 32) # reads a 32-byte field, returned up to the first NUL ``` Prefer to spell out the type yourself? The generic `read_process_memory` / diff --git a/docs/why.md b/docs/why.md index 3ffc9cb..c9accf3 100644 --- a/docs/why.md +++ b/docs/why.md @@ -27,7 +27,7 @@ it helps others discover the library too. **🌍 Truly cross-platform** -One identical API on **Windows, Linux and macOS**, 32- and 64-bit. Write your +One identical API on **Windows, Linux and macOS**, 32-bit and 64-bit. Write your script once; it runs everywhere. **🪶 Zero dependencies** @@ -51,7 +51,7 @@ ASLR — save them once and reuse them every launch. **⚡ Optional NumPy acceleration** Add the [`speed`](installation.md#install-with-scan-acceleration-speed) extra -and selective scans get **10–60× faster** — a drop-in fast path, identical +and selective scans get **~10–30× faster** — a drop-in fast path, identical results. **🖥️ A GUI app, included** diff --git a/pyproject.toml b/pyproject.toml index 9c799e8..329cdf8 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -6,6 +6,7 @@ authors = [ { name = "Jean Loui Bernard Silva de Jesus", email = "contact@jeanloui.dev" }, ] license = "MIT" +license-files = ["LICENSE*"] readme = "README.md" keywords = [ "cheat-engine", @@ -38,6 +39,8 @@ classifiers = [ "Programming Language :: Python :: 3.13", "Topic :: Scientific/Engineering", "Topic :: Security", + "Topic :: Software Development :: Debuggers", + "Topic :: Software Development :: Libraries :: Python Modules", "Topic :: System :: Monitoring", "Typing :: Typed", ] @@ -133,7 +136,11 @@ packages = ["PyMemoryEditor"] # Maintainer-only tooling lives under scripts/ — keep it out of the source # distribution (the wheel already only ships the PyMemoryEditor package). [tool.hatch.build.targets.sdist] -exclude = ["scripts"] +exclude = [ + "/.github", + "/codecov.yml", + "/scripts", +] [build-system] requires = ["hatchling"] diff --git a/scripts/build_preview.py b/scripts/build_preview.py new file mode 100755 index 0000000..7c9973b --- /dev/null +++ b/scripts/build_preview.py @@ -0,0 +1,110 @@ +#!/usr/bin/env python3 +"""Render the GitHub social preview PNG from social-preview.html. + +Works on macOS, Linux and Windows by locating a Chromium-based browser +(Chrome, Chromium or Edge) and driving it in headless screenshot mode. + +This is a maintainer-only helper — it is intentionally kept out of the +published package (see the sdist/wheel excludes in ``pyproject.toml``). + +Usage: + python scripts/build_preview.py # -> assets/social-preview.png (2560x1280, retina 2x) + python scripts/build_preview.py --scale 1 # 1280x640 exact + BROWSER=/path/to/chrome python scripts/build_preview.py # force a specific binary +""" + +import argparse +import os +import shutil +import subprocess +import sys +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parent.parent +HTML = REPO_ROOT / "assets" / "social-preview.html" +OUT = REPO_ROOT / "assets" / "social-preview.png" +WIDTH, HEIGHT = 1280, 640 + +# Candidate binaries per platform. The first one found is used. +CANDIDATES = { + "darwin": [ + "/Applications/Google Chrome.app/Contents/MacOS/Google Chrome", + "/Applications/Chromium.app/Contents/MacOS/Chromium", + "/Applications/Microsoft Edge.app/Contents/MacOS/Microsoft Edge", + ], + "linux": [ + "google-chrome", "google-chrome-stable", "chromium", "chromium-browser", + "microsoft-edge", "microsoft-edge-stable", + ], + "win32": [ + r"C:\Program Files\Google\Chrome\Application\chrome.exe", + r"C:\Program Files (x86)\Google\Chrome\Application\chrome.exe", + r"C:\Program Files (x86)\Microsoft\Edge\Application\msedge.exe", + r"C:\Program Files\Microsoft\Edge\Application\msedge.exe", + ], +} + + +def find_browser() -> str: + # Explicit override wins. + override = os.environ.get("BROWSER") + if override: + if Path(override).exists() or shutil.which(override): + return override + sys.exit(f"BROWSER={override!r} not found.") + + platform = "win32" if sys.platform.startswith("win") else \ + "darwin" if sys.platform == "darwin" else "linux" + + for cand in CANDIDATES[platform]: + # Absolute path that exists, or a name resolvable on PATH. + if Path(cand).exists() or shutil.which(cand): + return cand + + # Last resort: try common command names on PATH regardless of platform. + for name in ("google-chrome", "chromium", "chrome", "msedge"): + found = shutil.which(name) + if found: + return found + + sys.exit( + "No Chrome/Chromium/Edge found. Install one, or set the BROWSER " + "env var to its full path." + ) + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--scale", type=int, default=2, + help="device scale factor: 2 = retina 2560x1280 (default), 1 = exact 1280x640", + ) + args = parser.parse_args() + + if not HTML.exists(): + sys.exit(f"Missing {HTML}") + + browser = find_browser() + print(f"Using browser: {browser}") + + cmd = [ + browser, + "--headless", + "--disable-gpu", + "--hide-scrollbars", + f"--force-device-scale-factor={args.scale}", + "--default-background-color=00000000", + f"--window-size={WIDTH},{HEIGHT}", + f"--screenshot={OUT}", + HTML.as_uri(), + ] + + result = subprocess.run(cmd) + if result.returncode != 0 or not OUT.exists(): + sys.exit(f"Screenshot failed (exit {result.returncode}).") + + print(f"Wrote {OUT} ({WIDTH * args.scale}x{HEIGHT * args.scale})") + + +if __name__ == "__main__": + main() diff --git a/tests/test_app_pointer_parsing.py b/tests/test_app_pointer_parsing.py new file mode 100644 index 0000000..2709533 --- /dev/null +++ b/tests/test_app_pointer_parsing.py @@ -0,0 +1,80 @@ +# -*- coding: utf-8 -*- + +""" +Unit tests for the pure pointer-chain field parsers in +PyMemoryEditor/app/_widgets.py (``parse_offsets`` / ``resolve_base_address``). + +These used to be trapped inside ``PointerChainDialog._read_offsets`` / +``_resolve_base`` — methods that called ``QMessageBox`` and read widget state, +so the parsing logic (hex offsets, the ``"module"+0xoffset`` base form, the +ASLR module-base lookup) could not be tested without a live dialog. They are +now pure functions; the dialog keeps only the message-box presentation. +""" + +import pytest + +pytest.importorskip("PySide6") + +from PyMemoryEditor.app._widgets import ( # noqa: E402 + parse_offsets, + resolve_base_address, +) + + +# --- parse_offsets --------------------------------------------------------- # + +def test_parse_offsets_reads_hex_with_and_without_prefix(): + assert parse_offsets(["0x10", "158", "0X0"]) == [0x10, 0x158, 0x0] + + +def test_parse_offsets_skips_empty_tokens_and_preserves_order(): + assert parse_offsets(["", "0x4", "", "0x8"]) == [0x4, 0x8] + + +def test_parse_offsets_returns_none_on_a_bad_token(): + assert parse_offsets(["0x4", "zzz"]) is None + + +def test_parse_offsets_empty_is_empty_list(): + assert parse_offsets([]) == [] + assert parse_offsets(["", ""]) == [] + + +# --- resolve_base_address -------------------------------------------------- # + +def _no_modules(_name): + return None + + +def test_resolve_plain_hex_base(): + addr, err = resolve_base_address("0x14010F4F4", _no_modules) + assert err is None + assert addr == 0x14010F4F4 + + +def test_resolve_module_plus_offset_uses_lookup_and_adds(): + def lookup(name): + assert name == "game.exe" + return 0x140000000 + + addr, err = resolve_base_address('"game.exe"+0x10F4F4', lookup) + assert err is None + assert addr == 0x140000000 + 0x10F4F4 + + +def test_resolve_unknown_module_reports_error(): + addr, err = resolve_base_address("missing.dll+0x10", _no_modules) + assert addr is None + assert "not loaded" in err + + +def test_resolve_bad_offset_reports_error(): + addr, err = resolve_base_address('"game.exe"+zzz', lambda _n: 0x1000) + assert addr is None + assert "must be hex" in err + + +def test_resolve_garbage_base_reports_error(): + addr, err = resolve_base_address("not-an-address", _no_modules) + assert addr is None + assert "Base must be hex" in err diff --git a/tests/test_app_scan_request.py b/tests/test_app_scan_request.py new file mode 100644 index 0000000..18723ac --- /dev/null +++ b/tests/test_app_scan_request.py @@ -0,0 +1,154 @@ +# -*- coding: utf-8 -*- + +""" +Unit tests for ``build_scan_request`` (PyMemoryEditor/app/scan_worker.py). + +This is the pure core of ``ScannerPanel._build_request`` — the rules that turn +the scanner panel's fields into a ``ScanRequest``: the AOB-pattern short +circuit, the "String ignores the length field / Byte Array honours it" split, +range parsing, and the no-value (Increased/Decreased/...) scan types. It used +to live inside a ``QWidget`` method and could only be exercised by driving the +live widget; lifting it out means these rules are now testable without a +``QApplication``. +""" + +import pytest + +pytest.importorskip("PySide6") + +from PyMemoryEditor import ScanTypesEnum # noqa: E402 +from PyMemoryEditor.app.scan_types import NextScanType # noqa: E402 +from PyMemoryEditor.app.scan_worker import build_scan_request # noqa: E402 +from PyMemoryEditor.app.value_types import VALUE_TYPES # noqa: E402 + + +def _spec(*, pytype=None, length=None, pattern=False): + for spec in VALUE_TYPES: + if spec.is_pattern != pattern: + continue + if pytype is not None and spec.pytype is not pytype: + continue + if length is not None and spec.length != length: + continue + return spec + raise AssertionError(f"no spec for pytype={pytype} length={length} pattern={pattern}") + + +def _regex_spec(): + for spec in VALUE_TYPES: + if spec.is_regex: + return spec + raise AssertionError("no regex spec") + + +INT4 = _spec(pytype=int, length=4) +STR = _spec(pytype=str) +BYTES = _spec(pytype=bytes, pattern=False) +AOB = _spec(pattern=True) +REGEX = _regex_spec() + + +def test_exact_int_uses_fixed_width_and_passes_flags(): + req = build_scan_request( + INT4, + ScanTypesEnum.EXACT_VALUE, + value_text="100", + length_spin_value=99, # ignored: int has a fixed width + writeable_only=True, + ) + assert req.scan_type is ScanTypesEnum.EXACT_VALUE + assert req.value == 100 + assert req.length == 4 + assert req.writeable_only is True + + +def test_pattern_forces_exact_and_ignores_scan_type(): + # Even if a non-exact scan type leaks in, the pattern path forces EXACT. + req = build_scan_request( + AOB, + ScanTypesEnum.BIGGER_THAN, + value_text="90 90 ? 00", + ) + assert req.scan_type is ScanTypesEnum.EXACT_VALUE + assert req.value == "90 90 ? 00" + + +def test_pattern_with_value_false_drops_value(): + req = build_scan_request( + AOB, + ScanTypesEnum.EXACT_VALUE, + value_text="90 90", + with_value=False, + ) + assert req.value is None + + +def test_regex_carries_byte_length_from_length_field_and_forces_exact(): + # Even a non-exact scan type is forced to EXACT; the Length field becomes + # the regex's byte_length (max match width) and the value is the UTF-8 + # bytes pattern. + req = build_scan_request( + REGEX, + ScanTypesEnum.BIGGER_THAN, + value_text=r"Player[0-9]+", + length_spin_value=32, + ) + assert req.scan_type is ScanTypesEnum.EXACT_VALUE + assert req.value == rb"Player[0-9]+" + assert req.length == 32 + + +def test_string_ignores_length_override_and_uses_utf8_byte_length(): + # A multibyte string must size by encoded bytes, not characters, and the + # spin value must be ignored for str. + req = build_scan_request( + STR, + ScanTypesEnum.EXACT_VALUE, + value_text="óó", # 2 chars, 4 UTF-8 bytes + length_spin_value=99, + ) + assert req.value == "óó" + assert req.length == 4 + + +def test_bytes_honours_length_override(): + req = build_scan_request( + BYTES, + ScanTypesEnum.EXACT_VALUE, + value_text="AA BB", # 2 bytes + length_spin_value=8, + ) + assert req.value == b"\xaa\xbb" + assert req.length == 8 + + +def test_no_value_scan_type_drops_value(): + req = build_scan_request( + INT4, + NextScanType.INCREASED_VALUE, + value_text="this is ignored", + ) + assert req.value is None + assert req.scan_type is NextScanType.INCREASED_VALUE + assert req.length == 4 + + +def test_value_between_packs_a_tuple_and_takes_the_wider_length(): + req = build_scan_request( + STR, + ScanTypesEnum.VALUE_BETWEEN, + value_text="a", # 1 byte + second_value_text="óó", # 4 bytes + ) + assert req.value == ("a", "óó") + assert req.length == 4 # max(1, 4) + + +def test_invalid_value_raises_valueerror(): + with pytest.raises(ValueError): + build_scan_request(INT4, ScanTypesEnum.EXACT_VALUE, value_text="not-an-int") + + +def test_invalid_pattern_raises_valueerror(): + with pytest.raises(ValueError): + build_scan_request(AOB, ScanTypesEnum.EXACT_VALUE, value_text="") diff --git a/tests/test_app_value_types.py b/tests/test_app_value_types.py index a2b8b68..7a56aa1 100644 --- a/tests/test_app_value_types.py +++ b/tests/test_app_value_types.py @@ -32,6 +32,13 @@ def _spec(*, pytype=None, length=None, pattern=False): raise AssertionError(f"no spec for pytype={pytype} length={length} pattern={pattern}") +def _regex_spec(): + for spec in VALUE_TYPES: + if spec.is_regex: + return spec + raise AssertionError("no regex spec") + + INT4 = _spec(pytype=int, length=4) INT1 = _spec(pytype=int, length=1) FLOAT = _spec(pytype=float, length=4) @@ -39,6 +46,7 @@ def _spec(*, pytype=None, length=None, pattern=False): STR = _spec(pytype=str) BYTES = _spec(pytype=bytes, pattern=False) AOB = _spec(pattern=True) +REGEX = _regex_spec() # --- find_spec ------------------------------------------------------------- # @@ -122,6 +130,26 @@ def test_parse_pattern_empty_and_malformed_raise(): AOB.parse("4G 8B") # invalid hex token +# --- string regex ---------------------------------------------------------- # + +def test_parse_regex_encodes_text_to_utf8_bytes_pattern(): + # A text regex is UTF-8 encoded into the bytes pattern; the re engine + # interprets the metacharacters (\d, [...], +) on the ASCII range. + assert REGEX.parse(r"Player[0-9]+") == rb"Player[0-9]+" + + +def test_parse_regex_literal_non_ascii_becomes_its_utf8_bytes(): + # A literal accented char matches its UTF-8 byte sequence (no rejection). + assert REGEX.parse("café") == "café".encode("utf-8") + + +def test_parse_regex_empty_and_invalid_raise(): + with pytest.raises(ValueError): + REGEX.parse(" ") # empty + with pytest.raises(ValueError): + REGEX.parse("(unclosed") # not a valid regex + + # --- parse_value: length inference (the part that decides scan width) ------ # def test_parse_value_int_passthrough_length(): @@ -155,6 +183,19 @@ def test_parse_value_pattern_reports_zero_length(): assert length == 0 +def test_parse_value_regex_uses_length_field_as_byte_length(): + # A regex has no inferable match width, so the Length field supplies it. + value, length = parse_value(REGEX, r"[A-Z]+", length_override=8) + assert value == rb"[A-Z]+" + assert length == 8 + + +def test_parse_value_regex_falls_back_to_default_width_when_unset(): + value, length = parse_value(REGEX, r"\d+") + assert value == rb"\d+" + assert length == REGEX.length + + # --- format round-trips ---------------------------------------------------- # def test_format_round_trips(): diff --git a/tests/test_bitness.py b/tests/test_bitness.py index 182c7ea..c89aa68 100644 --- a/tests/test_bitness.py +++ b/tests/test_bitness.py @@ -10,6 +10,7 @@ """ import ctypes +import logging import os import sys @@ -19,7 +20,7 @@ pytest.skip("Platform not supported by PyMemoryEditor", allow_module_level=True) -from PyMemoryEditor import OpenProcess # noqa: E402 +from PyMemoryEditor import BitnessDetectionError, OpenProcess # noqa: E402 HOST_IS_64BIT = ctypes.sizeof(ctypes.c_void_p) == 8 @@ -75,3 +76,55 @@ def test_remote_pointer_defaults_to_detected_ptr_size(process): assert pointer.address == ctypes.addressof(target) assert (pointer.value & 0xFFFFFFFF) == 0x0BADF00D + + +# --- strict_bitness / is_bitness_certain -------------------------------- # +# +# The self-process always has a readable header, so detection is certain; the +# undeterminable path (a cross-bitness target whose header can't be read) is +# forced by stubbing the backend's raw detector to return None — the contract +# `_detect_is_64bit` uses to signal "unknown". + + +def test_is_bitness_certain_true_for_readable_target(process): + """A target whose header is readable (here, ourselves) is certain.""" + assert process.is_bitness_certain is True + assert process.is_64bit is HOST_IS_64BIT + + +def test_undeterminable_falls_back_to_host_and_warns(process, monkeypatch, caplog): + """Non-strict (default): an unknown bitness guesses the host word size, + flags the result as uncertain, and logs a WARNING.""" + monkeypatch.setattr(process, "_detect_is_64bit", lambda: None) + process._is_64bit_cache = None # clear any cached detection + process._bitness_certain = None + + with caplog.at_level(logging.WARNING, logger="PyMemoryEditor"): + assert process.is_64bit is HOST_IS_64BIT + assert process.is_bitness_certain is False + assert any( + record.levelno == logging.WARNING and "bitness" in record.getMessage().lower() + for record in caplog.records + ) + + +def test_strict_bitness_raises_when_undeterminable(monkeypatch): + """strict_bitness=True turns the guess into a BitnessDetectionError.""" + handle = OpenProcess(pid=os.getpid(), strict_bitness=True) + try: + monkeypatch.setattr(handle, "_detect_is_64bit", lambda: None) + with pytest.raises(BitnessDetectionError) as excinfo: + _ = handle.is_64bit + assert excinfo.value.pid == os.getpid() + finally: + handle.close() + + +def test_strict_bitness_succeeds_when_determinable(): + """strict_bitness must not interfere when the header is readable.""" + handle = OpenProcess(pid=os.getpid(), strict_bitness=True) + try: + assert handle.is_64bit is HOST_IS_64BIT + assert handle.is_bitness_certain is True + finally: + handle.close() diff --git a/tests/test_bitness_fallback_warns.py b/tests/test_bitness_fallback_warns.py new file mode 100644 index 0000000..15d9bc8 --- /dev/null +++ b/tests/test_bitness_fallback_warns.py @@ -0,0 +1,76 @@ +# -*- coding: utf-8 -*- + +""" +Regression guard: the bitness detectors must *warn* when they can't read the +target's header and fall back to a guess, instead of mis-detecting silently. + +A wrong bitness silently poisons the pointer-width default used by the pointer +APIs (resolve_pointer_chain / RemotePointer / scan_pointer_paths). The Windows +path (``IsProcess64Bit`` / ``mbi_class_for_handle``) already warned on the +``IsWow64Process`` failure; these tests lock the same contract on the macOS and +Linux fallbacks so a refactor can't drop the warning unnoticed. + +Each test forces the no-header fallback via monkeypatch (no special process or +permission needed) and asserts a WARNING reaches the ``PyMemoryEditor`` logger. +""" + +import logging +import sys + +import pytest + + +@pytest.fixture +def warning_records(): + """Capture WARNING+ records emitted on the PyMemoryEditor logger.""" + records = [] + + class _ListHandler(logging.Handler): + def emit(self, record): + records.append(record) + + logger = logging.getLogger("PyMemoryEditor") + handler = _ListHandler(level=logging.WARNING) + previous_level = logger.level + logger.addHandler(handler) + logger.setLevel(logging.DEBUG) + try: + yield records + finally: + logger.removeHandler(handler) + logger.setLevel(previous_level) + + +@pytest.mark.skipif(sys.platform != "darwin", reason="macOS backend only") +def test_is_task_64bit_warns_when_no_header_readable(monkeypatch, warning_records): + from PyMemoryEditor.macos import functions as mac + + # No module yields a readable Mach-O header → the fallback fires. + monkeypatch.setattr(mac, "get_modules", lambda task: iter(())) + + result = mac.is_task_64bit(0) + + assert result is True # macOS is 64-bit only since Catalina + assert any( + r.levelno == logging.WARNING and "is_task_64bit" in r.getMessage() + for r in warning_records + ) + + +@pytest.mark.skipif( + not sys.platform.startswith("linux"), reason="Linux backend only" +) +def test_is_process_64bit_warns_when_elf_class_unknown(monkeypatch, warning_records): + from PyMemoryEditor.linux import functions as lin + + # Neither /proc//exe nor any file-backed mapping yields an EI_CLASS. + monkeypatch.setattr(lin, "_read_elf_class", lambda path: None) + monkeypatch.setattr(lin, "get_memory_regions", lambda pid: iter(())) + + result = lin.is_process_64bit(12345) + + assert isinstance(result, bool) + assert any( + r.levelno == logging.WARNING and "is_process_64bit" in r.getMessage() + for r in warning_records + ) diff --git a/tests/test_cheat_poll_worker.py b/tests/test_cheat_poll_worker.py index a112ba5..7dcff58 100644 --- a/tests/test_cheat_poll_worker.py +++ b/tests/test_cheat_poll_worker.py @@ -41,11 +41,18 @@ class _FakeProcess: frozen-write. """ - def __init__(self, values=None, raise_on_batch=False, raise_on_read=False): + def __init__( + self, + values=None, + raise_on_batch=False, + raise_on_read=False, + raise_on_write=False, + ): # Map (address, pytype, length) → value to return on read. self.values = values or {} self.raise_on_batch = raise_on_batch self.raise_on_read = raise_on_read + self.raise_on_write = raise_on_write self.read_calls = [] self.write_calls = [] self.batch_calls = [] @@ -65,6 +72,8 @@ def read_process_memory(self, address, pytype, length): def write_process_memory(self, address, pytype, length, value): self.write_calls.append((address, pytype, length, value)) + if self.raise_on_write: + raise OSError("simulated write failure") return value @@ -144,6 +153,57 @@ def test_frozen_entries_get_written_each_tick(qapp): assert process.write_calls == [(0x3000, int, 4, 42)] +def test_frozen_write_failure_is_recorded_not_swallowed(qapp): + """A failing freeze write must be recorded (so the UI can flag it), not + silently swallowed, and the read value must still be surfaced so the table + shows the value drifting away from the frozen target.""" + process = _FakeProcess(values={(0x3000, int, 4): 123}, raise_on_write=True) + worker = _make_worker(process) + + snapshot = [(0x3000, int, 4, 42, True)] # frozen with frozen_value=42 + results = worker._poll_once(snapshot) + + # The write was attempted... + assert process.write_calls == [(0x3000, int, 4, 42)] + # ...failed, so the surfaced value is the real (drifting) read, not 42. + assert results == [(0x3000, int, 4, 123)] + # ...and the failure is tracked for the UI cue. + assert (0x3000, int, 4) in worker._freeze_failures + assert "OSError" in worker._freeze_failures[(0x3000, int, 4)] + + +def test_frozen_write_failure_clears_when_write_recovers(qapp): + """Once the freeze write succeeds again, the entry drops out of the failing + set so the UI can clear its red cue.""" + process = _FakeProcess(values={(0x3000, int, 4): 123}, raise_on_write=True) + worker = _make_worker(process) + snapshot = [(0x3000, int, 4, 42, True)] + + worker._poll_once(snapshot) + assert (0x3000, int, 4) in worker._freeze_failures + + # Backend recovers; next tick's write lands. + process.raise_on_write = False + results = worker._poll_once(snapshot) + + assert worker._freeze_failures == {} + assert results == [(0x3000, int, 4, 42)] # frozen value applied again + + +def test_unfreezing_a_failing_entry_drops_it_from_failures(qapp): + """An entry removed/unfrozen between ticks must not linger in the failing + set (it's rebuilt from the live snapshot each tick).""" + process = _FakeProcess(values={(0x3000, int, 4): 123}, raise_on_write=True) + worker = _make_worker(process) + + worker._poll_once([(0x3000, int, 4, 42, True)]) + assert (0x3000, int, 4) in worker._freeze_failures + + # Same entry, but no longer frozen → no write attempt, no failure. + worker._poll_once([(0x3000, int, 4, 42, False)]) + assert worker._freeze_failures == {} + + def test_frozen_entry_with_none_value_does_not_write(qapp): """Freeze checkbox active but no frozen_value yet → don't write.""" process = _FakeProcess(values={(0x4000, int, 4): 5}) diff --git a/tests/test_macos_protect.py b/tests/test_macos_protect.py index f4ce638..1b9f99f 100644 --- a/tests/test_macos_protect.py +++ b/tests/test_macos_protect.py @@ -127,3 +127,61 @@ def test_write_to_readonly_page_via_protect_flip(): process.close() finally: _libsystem.munmap(address, size) + + +def test_page_aligned_span_covers_a_straddling_write(): + """``_page_aligned_span`` must expand to whole pages around the byte range. + + A write that crosses a page boundary has to have *both* pages protected + (and later restored) as one span — otherwise the restore can miss the + second page and leave it permanently more permissive. + """ + from PyMemoryEditor.macos.functions import _PAGE_SIZE, _page_aligned_span + + page = _PAGE_SIZE + + # Write of 8 bytes starting 4 bytes before a page boundary → touches two + # pages; the aligned span must start at the first page and cover both. + start, length = _page_aligned_span(page - 4, 8) + assert start == 0 + assert length == 2 * page + + # A fully-contained write rounds out to exactly one page. + start, length = _page_aligned_span(page + 16, 4) + assert start == page + assert length == page + + +def test_write_straddling_a_page_boundary_restores_both_pages(): + """A cross-page write must succeed and leave *both* pages read-only again. + + Regression guard for the protect-flip alignment fix: the elevate/restore + span is page-aligned, so the second page isn't left writable after the + write completes. + """ + from PyMemoryEditor.macos.functions import _PAGE_SIZE + + page = _PAGE_SIZE + size = 2 * page + address = _mmap_readonly(size) + + try: + process = OpenProcess(pid=os.getpid()) + try: + boundary = address + page + # 8 bytes centered on the page boundary (4 in each page). + payload = b"\x01\x02\x03\x04\x05\x06\x07\x08" + process.write_process_memory(boundary - 4, bytes, len(payload), payload) + assert process.read_process_memory(boundary - 4, bytes, len(payload)) == payload + + # Both pages must be back to read-only — the restore covered the + # whole aligned span, not just the literal byte range. + for r in process.get_memory_regions(): + if r.address <= address < r.address + r.size: + assert not r.is_writable, "first page left writable after write" + if r.address <= boundary < r.address + r.size: + assert not r.is_writable, "second page left writable after write" + finally: + process.close() + finally: + _libsystem.munmap(address, size)