Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
36 changes: 22 additions & 14 deletions Include/cpython/unicodeobject.h
Original file line number Diff line number Diff line change
Expand Up @@ -557,25 +557,33 @@ typedef struct {
_Py_DEPRECATED_EXTERNALLY(3.14) PyAPI_FUNC(void) _PyUnicodeWriter_Init(
_PyUnicodeWriter *writer);

/* Prepare the buffer to write 'length' characters
with the specified maximum character.

Return 0 on success, raise an exception and return -1 on error. */
#define _PyUnicodeWriter_Prepare(WRITER, LENGTH, MAXCHAR) \
(((MAXCHAR) <= (WRITER)->maxchar \
&& (LENGTH) <= (WRITER)->size - (WRITER)->pos) \
? 0 \
: (((LENGTH) == 0) \
? 0 \
: _PyUnicodeWriter_PrepareInternal((WRITER), (LENGTH), (MAXCHAR))))

/* Don't call this function directly, use the _PyUnicodeWriter_Prepare() macro
instead. */
// Don't call this function directly, use _PyUnicodeWriter_Prepare() instead.
_Py_DEPRECATED_EXTERNALLY(3.14) PyAPI_FUNC(int) _PyUnicodeWriter_PrepareInternal(
_PyUnicodeWriter *writer,
Py_ssize_t length,
Py_UCS4 maxchar);

// Prepare the buffer to write 'length' characters
// with the specified maximum character.
//
// Return 0 on success. Set an exception and return -1 on error.
_Py_DEPRECATED_EXTERNALLY(3.14) static inline int
_PyUnicodeWriter_Prepare(_PyUnicodeWriter *writer,
Py_ssize_t length, Py_UCS4 maxchar)
{
assert(0 <= length);
if (maxchar <= writer->maxchar && length <= (writer->size - writer->pos)) {
return 0;
}
if (length == 0) {
return 0;
}
_Py_COMP_DIAG_PUSH
_Py_COMP_DIAG_IGNORE_DEPR_DECLS
return _PyUnicodeWriter_PrepareInternal(writer, length, maxchar);
_Py_COMP_DIAG_POP
}

/* Prepare the buffer to have at least the kind KIND.
For example, kind=PyUnicode_2BYTE_KIND ensures that the writer will
support characters in range U+000-U+FFFF.
Expand Down
55 changes: 36 additions & 19 deletions Include/internal/pycore_unicodeobject.h
Original file line number Diff line number Diff line change
Expand Up @@ -130,38 +130,51 @@ _PyUnicodeWriter_CanWrite(_PyUnicodeWriter *writer)
#endif

static inline void
_PyUnicodeWriter_Update(_PyUnicodeWriter *writer)
_PyUnicodeWriter_SetBuffer(_PyUnicodeWriter *writer, PyObject *buffer)
{
PyObject *buffer = writer->buffer;
writer->maxchar = PyUnicode_MAX_CHAR_VALUE(buffer);
assert(writer->pos <= PyUnicode_GET_LENGTH(buffer));

// Py_DECREF() the previous buffer (if any)
Py_XSETREF(writer->buffer, buffer);
writer->data = PyUnicode_DATA(buffer);
writer->kind = PyUnicode_KIND(buffer);
writer->maxchar = PyUnicode_MAX_CHAR_VALUE(buffer);
writer->size = PyUnicode_GET_LENGTH(buffer);
writer->readonly = 0;
}

if (!writer->readonly) {
writer->size = PyUnicode_GET_LENGTH(buffer);
}
else {
/* Copy-on-write mode: set buffer size to 0 so
* _PyUnicodeWriter_Prepare() will copy (and enlarge) the buffer on
* next write. */
writer->size = 0;
}
static inline void
_PyUnicodeWriter_SetReadOnly(_PyUnicodeWriter *writer, PyObject *obj,
Py_ssize_t length)
{
assert(writer->buffer == NULL);
assert(writer->pos == 0);
// Micro-optimization: pass length as a parameter, as it's usually known
// by the caller
assert(length == PyUnicode_GET_LENGTH(obj));

writer->buffer = obj;
writer->data = NULL;
/* Set kind and size to 0 to make sure that the next
* _PyUnicodeWriter_Prepare() call allocates a new buffer and copies
* characters. */
writer->kind = 0;
writer->maxchar = PyUnicode_MAX_CHAR_VALUE(obj);
writer->size = 0;
writer->pos = length;
writer->readonly = 1;
}

static inline int
_PyUnicodeWriter_WriteCharInline(_PyUnicodeWriter *writer, Py_UCS4 ch)
{
if (ch > writer->maxchar || 1 > writer->size - writer->pos) {
if (ch > writer->maxchar || 1 > (writer->size - writer->pos)) {
if (writer->buffer == NULL && ch <= 255) {
// If the first write is a Latin1 character, use the singleton
// as a read-only object
PyObject *obj = _Py_LATIN1_CHR(ch);
writer->readonly = 1;
writer->buffer = obj; // Py_NewRef() is not need on immortal object
_PyUnicodeWriter_Update(writer);
assert(writer->pos == 0);
writer->pos = 1;
// The next write will create a new buffer and copy the string
// Py_NewRef() is not needed on immortal object
_PyUnicodeWriter_SetReadOnly(writer, obj, 1);
return 0;
}

Expand All @@ -176,6 +189,10 @@ _PyUnicodeWriter_WriteCharInline(_PyUnicodeWriter *writer, Py_UCS4 ch)
return 0;
}

extern PyObject* _PyUnicodeWriter_FinishWithSize(
_PyUnicodeWriter *writer,
Py_ssize_t size);

/* --- Unicode API -------------------------------------------------------- */

// Export for '_json' shared extension
Expand Down
26 changes: 26 additions & 0 deletions Lib/test/test_capi/test_unicode.py
Original file line number Diff line number Diff line change
Expand Up @@ -461,6 +461,14 @@ def check_format(expected, format, *args):
check_format('%abc',
b'%%%s', b'abc')

# test "%s" with empty string
check_format('x=',
b'x=%s', b'')
check_format('x=',
b'x=%0s', b'')
check_format('x=',
b'x=%.3s', b'')

# truncated string
check_format('abc',
b'%.3s', b'abcdef')
Expand Down Expand Up @@ -1970,6 +1978,24 @@ def test_ascii(self):
writer.write_ascii(b"Python! <truncated>", 6)
self.assertEqual(writer.finish(), "Hello Python")

def test_write_latin1(self):
# Test _PyUnicodeWriter_WriteLatin1String()
writer = self.create_writer(0)
# Start with ASCII buffer
writer.write_latin1(b"abc IGNORED", 3)
writer.write_latin1(b"IGNORED", 0)
# Change buffer kind to UCS-1
writer.write_latin1(b"\xe9", 1)
# Change buffer kind to UCS-2
writer.write_str('[\u20ac]')
writer.write_latin1(b"def\xa0", 4)
# Change buffer kind to UCS-4
writer.write_str('[\U0010ffff]')
writer.write_latin1(b"ghi\xff.", 5)
writer.write_latin1(b"IGNORED", 0)
self.assertEqual(writer.finish(),
"abc\xe9[\u20ac]def\xa0[\U0010ffff]ghi\xff.")

def test_invalid_utf8(self):
writer = self.create_writer(0)
with self.assertRaises(UnicodeDecodeError):
Expand Down
37 changes: 37 additions & 0 deletions Lib/test/test_format.py
Original file line number Diff line number Diff line change
Expand Up @@ -240,6 +240,8 @@ def test_common_format(self):
testcommon("%d", 42, "42")
testcommon("%d", -42, "-42")
testcommon("%d", 42.0, "42")
testcommon("%#d", 42, "42")
testcommon("%#d", -42, "-42")
testcommon("%#x", 1, "0x1")
testcommon("%#X", 1, "0X1")
testcommon("%#o", 1, "0o1")
Expand All @@ -250,8 +252,12 @@ def test_common_format(self):
testcommon("%#X", 0, "0X0")
testcommon("%x", 0x42, "42")
testcommon("%x", -0x42, "-42")
testcommon("%#x", 0x42, "0x42")
testcommon("%#x", -0x42, "-0x42")
testcommon("%o", 0o42, "42")
testcommon("%o", -0o42, "-42")
testcommon("%#o", 0o42, "0o42")
testcommon("%#o", -0o42, "-0o42")
# alternate float formatting
testcommon('%g', 1.1, '1.1')
testcommon('%#g', 1.1, '1.10000')
Expand Down Expand Up @@ -344,12 +350,21 @@ def test_common_format(self):
"format argument 1: %g requires a real number, not str")

def test_str_format(self):
testformat("%s", "abc", "abc")
testformat("%r", "\u0378", "'\\u0378'") # non printable
testformat("%a", "\u0378", "'\\u0378'") # non printable
testformat("%r", "\u0374", "'\u0374'") # printable
testformat("%a", "\u0374", "'\\u0374'") # printable
testformat('%(x)r', {'x': 1}, '1')

# Some small ints
for fmt in ('s', 'r', 'a'):
with self.subTest(fmt=fmt):
testformat("%" + fmt, 42, "42")
testformat("%#" + fmt, 42, "42")
testformat("%" + fmt, -42, "-42")
testformat("%#" + fmt, -42, "-42")

# Test exception for unknown format characters, etc.
if verbose:
print('Testing exceptions')
Expand Down Expand Up @@ -639,6 +654,28 @@ def test_precision_c_limits(self):
with self.assertRaises(ValueError) as cm:
format(c, ".%sf" % (INT_MAX + 1))

@support.cpython_only
def test_precision_near_int_max(self):
# gh-158446: Precisions just below INT_MAX are rejected before any
# output buffer size is computed from them.
_testcapi = import_module("_testcapi")
INT_MAX = _testcapi.INT_MAX

f = 1e300
c = complex(f)
for prec in (INT_MAX, INT_MAX - 1023):
for code in "feg":
spec = ".%d%s" % (prec, code)
with self.subTest(spec=spec):
with self.assertRaises(ValueError):
format(f, spec)
with self.assertRaises(ValueError):
format(c, spec)
with self.assertRaises(ValueError):
("%" + spec) % f
with self.assertRaises(ValueError):
("%" + spec).encode() % f

def test_g_format_has_no_trailing_zeros(self):
# regression test for bugs.python.org/issue40780
self.assertEqual("%.3g" % 1505.0, "1.5e+03")
Expand Down
65 changes: 65 additions & 0 deletions Lib/test/test_sys.py
Original file line number Diff line number Diff line change
Expand Up @@ -563,6 +563,38 @@ def g456():
leave_g.set()
t.join()

@support.cpython_only
@requires_subinterpreters
@threading_helper.requires_working_threading()
def test_current_frames_other_interpreters(self):
# gh-158364: sys._current_frames() would access frames of another
# interpreter and crash
import threading

entered = threading.Event()
left = threading.Event()

def park():
entered.set()
left.wait()

t = threading.Thread(target=park)
with threading_helper.start_threads([t], unlock=left.set):
entered.wait()
interp = interpreters.create()
try:
interp.exec(f"""if True:
import sys
import threading

frames = sys._current_frames()
assert threading.get_ident() in frames, frames
assert frames[threading.get_ident()].f_globals is globals()
assert {t.ident} not in frames, frames
""")
finally:
interp.close()

@threading_helper.reap_threads
@threading_helper.requires_working_threading()
def test_current_exceptions(self):
Expand Down Expand Up @@ -629,6 +661,39 @@ def g456():
leave_g.set()
t.join()

@support.cpython_only
@requires_subinterpreters
@threading_helper.requires_working_threading()
def test_current_exceptions_other_interpreters(self):
# gh-158364: sys._current_exceptions() would hand out exceptions of
# another interpreter and crash
import threading

entered = threading.Event()
left = threading.Event()

def hold():
# The thread has to be handling an exception, otherwise
# sys._current_exceptions() has nothing to report for it.
try:
raise ValueError
except ValueError:
entered.set()
left.wait()

t = threading.Thread(target=hold)
with threading_helper.start_threads([t], unlock=left.set):
entered.wait()
interp = interpreters.create()
try:
interp.exec(f"""if True:
import sys

assert {t.ident} not in sys._current_exceptions()
""")
finally:
interp.close()

def test_attributes(self):
self.assertIsInstance(sys.api_version, int)
self.assertIsInstance(sys.argv, list)
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,2 @@
Fix :c:func:`!_PyUnicodeWriter_WriteLatin1String` when the writer buffer
kind is UCS-2 or UCS-4. Patch by Victor Stinner.
Original file line number Diff line number Diff line change
@@ -0,0 +1,3 @@
Fix crash when :func:`sys._current_frames` or
:func:`sys._current_exceptions` is called while another interpreter is
running.
Original file line number Diff line number Diff line change
@@ -0,0 +1,5 @@
Fix a crash or incorrect output that could occur when formatting a
:class:`float` or :class:`complex` with a precision close to the platform's
``INT_MAX``. :c:func:`PyOS_double_to_string` now raises :exc:`ValueError` for
any precision of that magnitude, regardless of presentation type or value, as
the format string parsers already did for precisions above ``INT_MAX``.
27 changes: 27 additions & 0 deletions Modules/_testcapi/unicode.c
Original file line number Diff line number Diff line change
Expand Up @@ -685,6 +685,32 @@ writer_write_substring(PyObject *self_raw, PyObject *args)
}


static PyObject*
writer_write_latin1(PyObject *self_raw, PyObject *args)
{
WriterObject *self = (WriterObject *)self_raw;
if (writer_check(self) < 0) {
return NULL;
}

const char *str;
Py_ssize_t bsize, size;
if (!PyArg_ParseTuple(args, "z#n", &str, &bsize, &size)) {
return NULL;
}

_PyUnicodeWriter *writer = (_PyUnicodeWriter*)self->writer;
_Py_COMP_DIAG_PUSH
_Py_COMP_DIAG_IGNORE_DEPR_DECLS
if (_PyUnicodeWriter_WriteLatin1String(writer, str, size) < 0) {
return NULL;
}
_Py_COMP_DIAG_POP

Py_RETURN_NONE;
}


static PyObject*
writer_decodeutf8stateful(PyObject *self_raw, PyObject *args)
{
Expand Down Expand Up @@ -778,6 +804,7 @@ static PyMethodDef writer_methods[] = {
{"write_str", _PyCFunction_CAST(writer_write_str), METH_O},
{"write_repr", _PyCFunction_CAST(writer_write_repr), METH_O},
{"write_substring", _PyCFunction_CAST(writer_write_substring), METH_VARARGS},
{"write_latin1", _PyCFunction_CAST(writer_write_latin1), METH_VARARGS},
{"decodeutf8stateful", _PyCFunction_CAST(writer_decodeutf8stateful), METH_VARARGS},
{"get_pointer", _PyCFunction_CAST(writer_get_pointer), METH_VARARGS},
{"get_buffer", _PyCFunction_CAST(writer_get_buffer), METH_VARARGS},
Expand Down
Loading
Loading