Skip to content

Commit 6a28fcf

Browse files
committed
Set the Windows console to UTF-8 while a pipe or the pager runs
Pipes used the console's code page so that console programs such as more, sort and findstr would show them correctly. But UTF-8 tools, such as rg or Git for Windows, then lost every character the code page lacks, and piped output differed from redirected output. Pipes are UTF-8 again everywhere. On Windows with a console, cmd2 sets its input and output code pages to 65001 from before the pipe process or pager starts until it exits, as chcp 65001 would, so console programs decode that UTF-8 too. The code pages it replaced are restored once the last overlapping pipe ends, including when the pipe fails.
1 parent b13f515 commit 6a28fcf

6 files changed

Lines changed: 203 additions & 127 deletions

File tree

‎CHANGELOG.md‎

Lines changed: 4 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -11,11 +11,10 @@
1111
- A `shell` command piped to an interactive program, such as `shell git log | less`, now joins
1212
the pipeline's job, so both processes receive Ctrl-C and Ctrl-Z
1313
- On Windows, fixed piped output appearing garbled in console programs such as `more`
14-
(`help -v | more`), a regression in 4.2.4. Pipes were written as UTF-8, but console programs
15-
decode their input with the console's code page. Pipes now use that code page, including ones
16-
Python names otherwise, such as 20866 (KOI8-R) or 28591 (ISO-8859-1), and characters it cannot
17-
represent are replaced rather than failing the command. Without a console, with a code page
18-
Python has no codec for, and on other platforms, pipes still use UTF-8
14+
(`help -v | more`), a regression in 4.2.4. Pipes are written as UTF-8, but console programs
15+
decode their input with the console's code page. While a pipe or the pager runs, cmd2 now sets
16+
the console to UTF-8 (code page 65001, as `chcp 65001` does) and restores its code pages
17+
afterward, so console programs and UTF-8 tools such as `rg` both read the output correctly
1918
- When copying a command's output to the clipboard (`help >`) fails, later commands' output is
2019
no longer written to the temporary file that held it
2120

‎cmd2/cmd2.py‎

Lines changed: 21 additions & 21 deletions
Original file line numberDiff line numberDiff line change
@@ -1965,12 +1965,12 @@ def ppaged(
19651965
soft_wrap=soft_wrap,
19661966
**(rich_print_kwargs if rich_print_kwargs is not None else {}),
19671967
)
1968-
# As for a pipe: on Windows, the pager decodes with the console's code page.
1969-
output_bytes = capture.get().encode(utils._pipe_encoding(), "replace")
1968+
# As for a pipe: the pager gets UTF-8, and on Windows the console is UTF-8 while it runs.
1969+
output_bytes = capture.get().encode("utf-8", "replace")
19701970

19711971
# Prevent KeyboardInterrupts while in the pager. The pager application will
19721972
# still receive the SIGINT since it is in the same process group as us.
1973-
with self.sigint_protection:
1973+
with self.sigint_protection, utils._utf8_console():
19741974
import subprocess
19751975

19761976
pipe_proc = subprocess.Popen( # noqa: S602
@@ -3320,13 +3320,11 @@ def _redirect_output(self, statement: Statement) -> utils.RedirectionSavedState:
33203320
# Create a pipe with read and write sides
33213321
read_fd, write_fd = os.pipe()
33223322

3323-
# Open each side of the pipe. Both ends are given an explicit encoding: command output is rendered by Rich and
3324-
# routinely contains non-ASCII, which the locale encoding cannot always represent. On Windows, that is the
3325-
# console's code page, which console programs such as more decode with. It cannot represent everything either, so
3326-
# replace what it lacks rather than fail the command.
3327-
pipe_encoding = utils._pipe_encoding()
3328-
subproc_stdin = open(read_fd, encoding=pipe_encoding) # noqa: SIM115
3329-
new_stdout: TextIO = cast(TextIO, open(write_fd, "w", encoding=pipe_encoding, errors="replace")) # noqa: SIM115
3323+
# Open each side of the pipe. Both ends use UTF-8 rather than the locale's encoding, which cannot always represent
3324+
# the non-ASCII that Rich routinely renders. On Windows, the console is UTF-8 too while the pipe runs, for console
3325+
# programs such as more, which decode with its code page.
3326+
subproc_stdin = open(read_fd, encoding="utf-8") # noqa: SIM115
3327+
new_stdout: TextIO = cast(TextIO, open(write_fd, "w", encoding="utf-8")) # noqa: SIM115
33303328

33313329
# Isolate pipeline signals from cmd2. Terminal pipelines receive the foreground terminal; ProcReader relays their
33323330
# job-control stops.
@@ -3373,7 +3371,10 @@ def _redirect_output(self, statement: Statement) -> utils.RedirectionSavedState:
33733371
popen_command = f"read -r _ || exit 1; exec {user_shell} -c {shlex.quote(statement.redirect_to)}"
33743372
kwargs["executable"] = posix_shell
33753373

3376-
with contextlib.ExitStack() as terminal_stack, contextlib.ExitStack() as gate_stack:
3374+
with contextlib.ExitStack() as pipe_stack, contextlib.ExitStack() as gate_stack:
3375+
# The console stays UTF-8 from before the pipe process starts, since a program may read the code page once as
3376+
# it starts, until _restore_output() has reaped it.
3377+
pipe_stack.enter_context(utils._utf8_console())
33773378
if terminal_fd is not None:
33783379
# Should cmd2 fail before opening the gate, the held pipeline reads EOF and exits.
33793380
gate_stack.callback(new_stdout.close)
@@ -3391,7 +3392,7 @@ def _redirect_output(self, statement: Statement) -> utils.RedirectionSavedState:
33913392
subproc_stdin.close()
33923393
if terminal_fd is not None:
33933394
cmd_pipe_proc_reader = utils.ProcReader(proc, self.stdout, sys.stderr, terminal_fd=terminal_fd)
3394-
terminal_stack.enter_context(cmd_pipe_proc_reader._manage_terminal())
3395+
pipe_stack.enter_context(cmd_pipe_proc_reader._manage_terminal())
33953396

33963397
# Popen was called with shell=True so the user can chain pipe commands and redirect their output
33973398
# like: !ls -l | grep user | wc -l > out.txt. But this makes it difficult to know if the pipe process started
@@ -3429,14 +3430,13 @@ def _redirect_output(self, statement: Statement) -> utils.RedirectionSavedState:
34293430
pipe_fd, cmd_pipe_proc_reader, interruptible=lambda: not self.sigint_protection
34303431
)
34313432
),
3432-
encoding=pipe_encoding,
3433-
errors="replace",
3433+
encoding="utf-8",
34343434
)
34353435

34363436
self.stdout = new_stdout
34373437

3438-
# Keep the pipeline's job control until _restore_output() reaps the pipe process.
3439-
redir_saved_state.pipeline_job = terminal_stack.pop_all()
3438+
# Keep the UTF-8 console and the pipeline's job control until _restore_output() reaps the pipe process.
3439+
redir_saved_state.pipe_context = pipe_stack.pop_all()
34403440

34413441
elif statement.redirector in (constants.REDIRECTION_OVERWRITE, constants.REDIRECTION_APPEND):
34423442
if statement.redirect_to:
@@ -3495,11 +3495,11 @@ def _restore_output(self, statement: Statement, saved_redir_state: utils.Redirec
34953495
:param statement: Statement object which contains the parsed input from the user
34963496
:param saved_redir_state: contains information needed to restore state data
34973497
"""
3498-
# The pipeline's job control ends once its pipe process has been reaped.
3499-
with contextlib.ExitStack() as terminal_stack:
3500-
if saved_redir_state.pipeline_job is not None:
3501-
terminal_stack.callback(saved_redir_state.pipeline_job.close)
3502-
saved_redir_state.pipeline_job = None
3498+
# The UTF-8 console and the pipeline's job control end once its pipe process has been reaped.
3499+
with contextlib.ExitStack() as pipe_stack:
3500+
if saved_redir_state.pipe_context is not None:
3501+
pipe_stack.callback(saved_redir_state.pipe_context.close)
3502+
saved_redir_state.pipe_context = None
35033503

35043504
try:
35053505
if saved_redir_state.redirecting:

‎cmd2/utils.py‎

Lines changed: 53 additions & 48 deletions
Original file line numberDiff line numberDiff line change
@@ -536,58 +536,62 @@ def write(self, b: bytes) -> None:
536536
self.std_sim_instance.flush()
537537

538538

539-
# Windows code pages Python names other than cpNNNNN. Before Python 3.14, which covers every code page Windows supports, a
540-
# console set to one of these would otherwise get UTF-8.
541-
_CODE_PAGE_CODECS = {
542-
20127: "ascii",
543-
20866: "koi8_r",
544-
21866: "koi8_u",
545-
28591: "latin_1",
546-
28592: "iso8859_2",
547-
28593: "iso8859_3",
548-
28594: "iso8859_4",
549-
28595: "iso8859_5",
550-
28596: "iso8859_6",
551-
28597: "iso8859_7",
552-
28598: "iso8859_8",
553-
28599: "iso8859_9",
554-
28603: "iso8859_13",
555-
28605: "iso8859_15",
556-
51932: "euc_jp",
557-
51949: "euc_kr",
558-
54936: "gb18030",
559-
}
560-
561-
562-
def _code_page_encoding(code_page: int) -> str | None:
563-
"""Return the name of the Python codec for a Windows code page, or None if there is none.
564-
565-
:param code_page: a Windows code page identifier, such as 437
539+
class _Utf8Console:
540+
"""Set the Windows console to UTF-8 while cmd2 pipes output to a shell command or pager.
541+
542+
cmd2 writes pipes as UTF-8. Windows console programs such as more, sort, and findstr decode
543+
piped input with the console's output code page, PowerShell with its input code page, and
544+
tools such as rg with UTF-8. The UTF-8 code page, 65001, suits them all, as `chcp 65001`
545+
would. It belongs to the whole console, so the code pages it replaced are restored once the
546+
last pipe that uses it has ended. Without a console, and on other platforms, it does nothing.
566547
"""
567-
import codecs
568548

569-
for name in (f"cp{code_page}", _CODE_PAGE_CODECS.get(code_page)):
570-
if name is not None:
571-
with contextlib.suppress(LookupError):
572-
# Normalized, so that code page 65001 is reported as utf-8
573-
return codecs.lookup(name).name
574-
return None
549+
UTF8 = 65001
575550

551+
def __init__(self, kernel32: Any = None) -> None:
552+
"""Initialize for the console of a Windows API.
576553
577-
def _pipe_encoding() -> str:
578-
"""Return the encoding for output cmd2 pipes to a shell command.
554+
:param kernel32: the Windows API to use, or None for ctypes.windll.kernel32 on Windows
555+
"""
556+
self._kernel32 = kernel32
557+
self._lock = threading.Lock()
558+
# How many pipes use the console, and its input and output code pages to restore after the last
559+
self._users = 0
560+
self._saved: tuple[int, int] | None = None
561+
562+
@contextlib.contextmanager
563+
def __call__(self) -> Iterator[None]:
564+
"""Keep the console at UTF-8 for as long as the context lasts."""
565+
kernel32 = self._kernel32
566+
if kernel32 is None and sys.platform == "win32":
567+
import ctypes
568+
569+
kernel32 = ctypes.windll.kernel32
570+
if kernel32 is None:
571+
yield
572+
return
573+
with self._lock:
574+
if not self._users:
575+
# Without a console, the code pages are 0.
576+
code_pages = (kernel32.GetConsoleCP(), kernel32.GetConsoleOutputCP())
577+
if code_pages[1] and code_pages != (self.UTF8, self.UTF8):
578+
self._saved = code_pages
579+
kernel32.SetConsoleCP(self.UTF8)
580+
kernel32.SetConsoleOutputCP(self.UTF8)
581+
self._users += 1
582+
try:
583+
yield
584+
finally:
585+
with self._lock:
586+
self._users -= 1
587+
if not self._users and self._saved is not None:
588+
input_code_page, output_code_page = self._saved
589+
self._saved = None
590+
kernel32.SetConsoleCP(input_code_page)
591+
kernel32.SetConsoleOutputCP(output_code_page)
579592

580-
Windows console programs such as more, sort, and findstr decode piped input with the
581-
console's output code page, and would show UTF-8 as mojibake. Elsewhere, and on Windows
582-
without a console or with a code page Python cannot encode, use UTF-8.
583-
"""
584-
if sys.platform == "win32":
585-
import ctypes
586593

587-
code_page = ctypes.windll.kernel32.GetConsoleOutputCP()
588-
if code_page:
589-
return _code_page_encoding(code_page) or "utf-8"
590-
return "utf-8"
594+
_utf8_console = _Utf8Console()
591595

592596

593597
@contextlib.contextmanager
@@ -1460,8 +1464,9 @@ def __init__(
14601464
self.saved_pipe_proc_reader = pipe_proc_reader
14611465
self.saved_redirecting = saved_redirecting
14621466

1463-
# Holds a terminal pipeline's job control until its pipe process has been reaped
1464-
self.pipeline_job: contextlib.ExitStack | None = None
1467+
# Holds what a pipe needs until its pipe process has been reaped: on Windows a UTF-8 console, and for a terminal
1468+
# pipeline its job control
1469+
self.pipe_context: contextlib.ExitStack | None = None
14651470

14661471

14671472
def categorize(func: Callable[..., Any] | Iterable[Callable[..., Any]], category: str) -> None:

‎tests/test_cmd2.py‎

Lines changed: 20 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -3795,8 +3795,11 @@ def test_ppaged_with_pager(outsim_app, monkeypatch, chop) -> None:
37953795
assert expected_cmd == popen_mock.call_args_list[0].args[0]
37963796

37973797

3798-
def test_ppaged_encodes_for_the_console(outsim_app, monkeypatch) -> None:
3799-
"""As for a pipe: on Windows, the pager decodes with the console's code page, and what it lacks is replaced."""
3798+
def test_ppaged_pages_utf8_in_a_utf8_console(outsim_app, monkeypatch) -> None:
3799+
"""As for a pipe: the pager gets UTF-8, and on Windows the console is UTF-8 until the pager has exited.
3800+
3801+
The pager there is more, which decodes with the console's code page.
3802+
"""
38003803
stdin_mock = mock.MagicMock()
38013804
stdin_mock.isatty.return_value = True
38023805
monkeypatch.setattr(outsim_app, "stdin", stdin_mock)
@@ -3805,12 +3808,25 @@ def test_ppaged_encodes_for_the_console(outsim_app, monkeypatch) -> None:
38053808
monkeypatch.setattr(outsim_app, "stdout", stdout_mock)
38063809
if not sys.platform.startswith("win") and os.environ.get("TERM") is None:
38073810
monkeypatch.setenv("TERM", "simulated")
3808-
monkeypatch.setattr("cmd2.utils._pipe_encoding", lambda: "cp437")
3811+
events = []
3812+
3813+
@contextlib.contextmanager
3814+
def utf8_console():
3815+
events.append("utf-8 console")
3816+
try:
3817+
yield
3818+
finally:
3819+
events.append("restored")
3820+
3821+
monkeypatch.setattr("cmd2.utils._utf8_console", utf8_console)
38093822
popen_mock = mock.MagicMock(name="Popen")
3823+
popen_mock.side_effect = lambda *args, **kwargs: events.append("pager started") or popen_mock.return_value
3824+
popen_mock.return_value.communicate.side_effect = lambda *args: events.append("pager exited")
38103825
monkeypatch.setattr("subprocess.Popen", popen_mock)
38113826
outsim_app.ppaged("box ─ smile \U0001f642")
38123827
paged = popen_mock.return_value.communicate.call_args.args[0]
3813-
assert "box ─ smile ?".encode("cp437") in paged
3828+
assert "box ─ smile \U0001f642".encode() in paged
3829+
assert events == ["utf-8 console", "pager started", "pager exited", "restored"]
38143830

38153831

38163832
def test_ppaged_no_pager(outsim_app) -> None:

‎tests/test_suite_environment.py‎

Lines changed: 39 additions & 10 deletions
Original file line numberDiff line numberDiff line change
@@ -67,27 +67,56 @@ def test_redirection_to_a_file_uses_utf8(tmp_path) -> None:
6767
PASS_THROUGH = "import sys; sys.stdin.reconfigure(encoding='utf-8'); sys.stdout.write(sys.stdin.read())"
6868

6969

70-
def test_piping_uses_pipe_encoding(tmp_path, running_pipe_process) -> None:
71-
"""The pipe the subprocess reads from uses the encoding its consumer expects, not the locale's.
70+
def test_piping_uses_utf8(tmp_path, running_pipe_process) -> None:
71+
"""The pipe the subprocess reads from uses UTF-8, not the locale's encoding.
7272
73-
That is UTF-8, except on Windows, where console programs such as more decode with the
74-
console's code page.
73+
On Windows, console programs such as more decode with the console's code page, so cmd2
74+
sets the console to UTF-8 while the pipe runs.
7575
"""
7676
app = EncodingProbe(allow_cli_args=False)
7777
target = tmp_path / "piped.txt"
7878
app.onecmd_plus_hooks(f'show_encoding | "{sys.executable}" -c "{PASS_THROUGH}" > "{target}"')
79-
assert f"ENCODING={cmd2.utils._pipe_encoding()}" in target.read_text(encoding="utf-8")
79+
assert "ENCODING=utf-8" in target.read_text(encoding="utf-8")
80+
81+
82+
def test_piping_sets_the_console_to_utf8_while_the_pipe_runs(tmp_path, monkeypatch, running_pipe_process) -> None:
83+
"""The console is UTF-8 before the pipe process starts, and stays so until it has exited."""
84+
import contextlib
85+
import subprocess
86+
87+
events = []
88+
processes = []
89+
real_popen = subprocess.Popen
90+
91+
@contextlib.contextmanager
92+
def utf8_console():
93+
events.append("utf-8 console")
94+
try:
95+
yield
96+
finally:
97+
events.append(("restored", [process.returncode for process in processes]))
98+
99+
def popen(*args, **kwargs):
100+
events.append("pipe started")
101+
processes.append(real_popen(*args, **kwargs))
102+
return processes[-1]
103+
104+
monkeypatch.setattr(cmd2.utils, "_utf8_console", utf8_console)
105+
monkeypatch.setattr(subprocess, "Popen", popen)
106+
app = EncodingProbe(allow_cli_args=False)
107+
target = tmp_path / "piped.txt"
108+
app.onecmd_plus_hooks(f'say hello | "{sys.executable}" -c "{PASS_THROUGH}" > "{target}"')
109+
assert events == ["utf-8 console", "pipe started", ("restored", [0])]
110+
assert "hello" in target.read_text(encoding="utf-8")
80111

81112

82113
#: A pass-through filter that copies bytes, so the test sees exactly what cmd2 wrote to the pipe.
83114
BYTES_PASS_THROUGH = "import sys; sys.stdout.buffer.write(sys.stdin.buffer.read())"
84115

85116

86-
def test_piping_replaces_what_the_pipe_encoding_lacks(tmp_path, monkeypatch, running_pipe_process) -> None:
87-
"""A console code page cannot represent all output, which must not fail the command."""
88-
monkeypatch.setattr(cmd2.utils, "_pipe_encoding", lambda: "cp437")
117+
def test_piping_keeps_all_output(tmp_path, running_pipe_process) -> None:
118+
"""UTF-8 represents all output, such as box drawing and emoji, which a console code page could not."""
89119
app = EncodingProbe(allow_cli_args=False)
90120
target = tmp_path / "piped.bin"
91121
app.onecmd_plus_hooks(f'say ─\U0001f607 | "{sys.executable}" -c "{BYTES_PASS_THROUGH}" > "{target}"')
92-
# Box drawing exists in cp437, the emoji does not
93-
assert target.read_bytes().startswith(b"\xc4?")
122+
assert target.read_bytes().startswith("─\U0001f607".encode())

0 commit comments

Comments
 (0)