Skip to content

Commit 1942935

Browse files
committed
Merge branch 'main' into unicode_overflow
2 parents 85966a9 + 8542958 commit 1942935

6 files changed

Lines changed: 61 additions & 7 deletions

File tree

‎Doc/c-api/unicode.rst‎

Lines changed: 3 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -1797,6 +1797,9 @@ object.
17971797
The instance must be destroyed by :c:func:`PyUnicodeWriter_Finish` on
17981798
success, or :c:func:`PyUnicodeWriter_Discard` on error.
17991799
1800+
The API is **not thread safe**. To share a writer with multiple threads, a
1801+
critical section or a lock is needed.
1802+
18001803
.. c:function:: PyUnicodeWriter* PyUnicodeWriter_Create(Py_ssize_t length)
18011804
18021805
Create a Unicode writer instance.

‎Include/internal/pycore_unicodeobject.h‎

Lines changed: 22 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -16,7 +16,9 @@ extern "C" {
1616
#define _Py_MAX_UNICODE 0x10ffff
1717

1818

19-
extern int _PyUnicode_IsModifiable(PyObject *unicode);
19+
// Export for '_multibytecodec' shared extension. _PyUnicodeWriter_CanWrite()
20+
// calls this function when assertions are enabled.
21+
PyAPI_FUNC(int) _PyUnicode_IsModifiable(PyObject *unicode);
2022
extern void _PyUnicodeWriter_InitWithBuffer(
2123
_PyUnicodeWriter *writer,
2224
PyObject *buffer);
@@ -105,12 +107,31 @@ _PyUnicode_EnsureUnicode(PyObject *obj)
105107
return 0;
106108
}
107109

110+
#ifndef NDEBUG
111+
static inline int
112+
_PyUnicodeWriter_CanWrite(_PyUnicodeWriter *writer)
113+
{
114+
// Code adapted from _PyUnicode_IsModifiable()
115+
assert(!writer->readonly);
116+
PyObject *buffer = writer->buffer;
117+
assert(buffer != NULL);
118+
// Do not use _PyObject_IsUniquelyReferenced(): the caller can have its own
119+
// lock to prevent a writer being used by two theads at the same time.
120+
assert(Py_REFCNT(buffer) == 1);
121+
assert(PyUnstable_Unicode_GET_CACHED_HASH(buffer) == -1);
122+
assert(!PyUnicode_CHECK_INTERNED(buffer));
123+
assert(!_Py_IsImmortal(buffer));
124+
return 1;
125+
}
126+
#endif
127+
108128
static inline int
109129
_PyUnicodeWriter_WriteCharInline(_PyUnicodeWriter *writer, Py_UCS4 ch)
110130
{
111131
assert(ch <= _Py_MAX_UNICODE);
112132
if (_PyUnicodeWriter_Prepare(writer, 1, ch) < 0)
113133
return -1;
134+
assert(_PyUnicodeWriter_CanWrite(writer));
114135
PyUnicode_WRITE(writer->kind, writer->data, writer->pos, ch);
115136
writer->pos++;
116137
return 0;

‎Lib/test/test_capi/test_unicode.py‎

Lines changed: 11 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -1983,6 +1983,17 @@ def test_substring_empty(self):
19831983
writer.write_substring("abc", 1, 1)
19841984
self.assertEqual(writer.finish(), '')
19851985

1986+
def test_singletons(self):
1987+
writer = self.create_writer(5)
1988+
self.assertIs(writer.finish(), '')
1989+
1990+
for ch in range(256):
1991+
with self.subTest(ch=ch):
1992+
ch = chr(ch)
1993+
writer = self.create_writer(0)
1994+
writer.write_substring(ch + 'xxx', 0, 1)
1995+
self.assertIs(writer.finish(), ch)
1996+
19861997
@unittest.skipUnless(support.Py_DEBUG, 'need debug build (Py_DEBUG)')
19871998
def test_detect_overflow(self):
19881999
# Test detection of buffer overflow

‎Objects/longobject.c‎

Lines changed: 4 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -2220,6 +2220,7 @@ long_to_decimal_string_internal(PyObject *aa,
22202220
Py_DECREF(scratch);
22212221
return -1;
22222222
}
2223+
assert(_PyUnicodeWriter_CanWrite(writer));
22232224
}
22242225
else if (bytes_writer) {
22252226
*bytes_str = PyBytesWriter_GrowAndUpdatePointer(bytes_writer, strlen,
@@ -2390,8 +2391,10 @@ long_format_binary(PyObject *aa, int base, int alternate,
23902391
}
23912392

23922393
if (writer) {
2393-
if (_PyUnicodeWriter_Prepare(writer, sz, 'x') == -1)
2394+
if (_PyUnicodeWriter_Prepare(writer, sz, 'x') == -1) {
23942395
return -1;
2396+
}
2397+
assert(_PyUnicodeWriter_CanWrite(writer));
23952398
}
23962399
else if (bytes_writer) {
23972400
*bytes_str = PyBytesWriter_GrowAndUpdatePointer(bytes_writer, sz,

‎Objects/unicode_writer.c‎

Lines changed: 7 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -350,6 +350,8 @@ _PyUnicodeWriter_WriteStr(_PyUnicodeWriter *writer, PyObject *str)
350350
if (_PyUnicodeWriter_PrepareInternal(writer, len, maxchar) == -1)
351351
return -1;
352352
}
353+
354+
assert(_PyUnicodeWriter_CanWrite(writer));
353355
_PyUnicode_FastCopyCharacters(writer->buffer, writer->pos,
354356
str, 0, len);
355357
writer->pos += len;
@@ -428,6 +430,7 @@ _PyUnicodeWriter_WriteSubstring(_PyUnicodeWriter *writer, PyObject *str,
428430
if (_PyUnicodeWriter_Prepare(writer, len, maxchar) < 0) {
429431
return -1;
430432
}
433+
assert(_PyUnicodeWriter_CanWrite(writer));
431434

432435
_PyUnicode_FastCopyCharacters(writer->buffer, writer->pos,
433436
str, start, len);
@@ -485,8 +488,10 @@ _PyUnicodeWriter_WriteASCIIString(_PyUnicodeWriter *writer,
485488
return 0;
486489
}
487490

488-
if (_PyUnicodeWriter_Prepare(writer, len, 127) == -1)
491+
if (_PyUnicodeWriter_Prepare(writer, len, 127) == -1) {
489492
return -1;
493+
}
494+
assert(_PyUnicodeWriter_CanWrite(writer));
490495

491496
switch (writer->kind)
492497
{
@@ -591,6 +596,7 @@ _PyUnicodeWriter_WriteLatin1String(_PyUnicodeWriter *writer,
591596
maxchar = ucs1lib_find_max_char((const Py_UCS1*)str, (const Py_UCS1*)str + len);
592597
if (_PyUnicodeWriter_Prepare(writer, len, maxchar) == -1)
593598
return -1;
599+
assert(_PyUnicodeWriter_CanWrite(writer));
594600
unicode_write_cstr(writer->buffer, writer->pos, str, len);
595601
writer->pos += len;
596602
return 0;

‎Objects/unicodeobject.c‎

Lines changed: 14 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -1744,18 +1744,21 @@ unicode_is_singleton(PyObject *unicode)
17441744
}
17451745
#endif
17461746

1747+
// If this function is updated, update also _PyUnicodeWriter_CanWrite().
17471748
int
17481749
_PyUnicode_IsModifiable(PyObject *unicode)
17491750
{
17501751
assert(_PyUnicode_CHECK(unicode));
1752+
if (!PyUnicode_CheckExact(unicode))
1753+
return 0;
1754+
// On Free Threading, this test fails if called from a thread other
1755+
// than the one which created the str object.
17511756
if (!_PyObject_IsUniquelyReferenced(unicode))
17521757
return 0;
17531758
if (PyUnicode_HASH(unicode) != -1)
17541759
return 0;
17551760
if (PyUnicode_CHECK_INTERNED(unicode))
17561761
return 0;
1757-
if (!PyUnicode_CheckExact(unicode))
1758-
return 0;
17591762
#ifdef Py_DEBUG
17601763
/* singleton refcount is greater than 1 */
17611764
assert(!unicode_is_singleton(unicode));
@@ -2009,6 +2012,7 @@ PyUnicodeWriter_WriteWideChar(PyUnicodeWriter *pub_writer,
20092012
if (_PyUnicodeWriter_Prepare(writer, size - num_surrogates, maxchar) < 0) {
20102013
return -1;
20112014
}
2015+
assert(_PyUnicodeWriter_CanWrite(writer));
20122016

20132017
int kind = writer->kind;
20142018
void *data = (Py_UCS1*)writer->data + writer->pos * kind;
@@ -2267,6 +2271,7 @@ PyUnicodeWriter_WriteUCS4(PyUnicodeWriter *pub_writer,
22672271
if (_PyUnicodeWriter_Prepare(writer, size, max_char) < 0) {
22682272
return -1;
22692273
}
2274+
assert(_PyUnicodeWriter_CanWrite(writer));
22702275

22712276
int kind = writer->kind;
22722277
void *data = (Py_UCS1*)writer->data + writer->pos * kind;
@@ -2553,8 +2558,10 @@ unicode_fromformat_write_str(_PyUnicodeWriter *writer, PyObject *str,
25532558
else
25542559
maxchar = writer->maxchar;
25552560

2556-
if (_PyUnicodeWriter_Prepare(writer, arglen, maxchar) == -1)
2561+
if (_PyUnicodeWriter_Prepare(writer, arglen, maxchar) == -1) {
25572562
return -1;
2563+
}
2564+
assert(_PyUnicodeWriter_CanWrite(writer));
25582565

25592566
fill = Py_MAX(width - length, 0);
25602567
if (fill && !(flags & F_LJUST)) {
@@ -2844,8 +2851,10 @@ unicode_fromformat_arg(_PyUnicodeWriter *writer,
28442851
Py_ssize_t spacepad = Py_MAX(width - precision - sign, 0);
28452852
Py_ssize_t zeropad = Py_MAX(precision - len, 0);
28462853

2847-
if (_PyUnicodeWriter_Prepare(writer, width, 127) == -1)
2854+
if (_PyUnicodeWriter_Prepare(writer, width, 127) == -1) {
28482855
return NULL;
2856+
}
2857+
assert(_PyUnicodeWriter_CanWrite(writer));
28492858

28502859
if (spacepad && !(flags & F_LJUST)) {
28512860
if (PyUnicode_Fill(writer->buffer, writer->pos, spacepad, ' ') == -1)
@@ -5372,6 +5381,7 @@ _PyUnicode_DecodeUTF8Writer(_PyUnicodeWriter *writer,
53725381
if (_PyUnicodeWriter_Prepare(writer, size, 127) < 0) {
53735382
return -1;
53745383
}
5384+
assert(_PyUnicodeWriter_CanWrite(writer));
53755385

53765386
const char *starts = s;
53775387
const char *end = s + size;

0 commit comments

Comments
 (0)