diff --git a/Lib/test/test_gdb/test_pretty_print.py b/Lib/test/test_gdb/test_pretty_print.py index edb79ef86d31db..ea047a12b802e8 100644 --- a/Lib/test/test_gdb/test_pretty_print.py +++ b/Lib/test/test_gdb/test_pretty_print.py @@ -103,13 +103,16 @@ def test_bytes(self): self.assertGdbRepr(b'And now for something hopefully the same') self.assertGdbRepr(b'string with embedded NUL here \0 and then some more text') self.assertGdbRepr(b'this is a tab:\t' + b' this is a slash:\\' b' this is a slash-N:\n' b' this is a slash-R:\r' ) + # Test double quotes (") + self.assertGdbRepr(b"it's a quote") self.assertGdbRepr(b'this is byte 255:\xff and byte 128:\x80') - self.assertGdbRepr(bytes([b for b in range(255)])) + self.assertGdbRepr(bytes([b for b in range(256)])) @support.requires_resource('cpu') def test_strings(self): diff --git a/Misc/NEWS.d/next/Tools-Demos/2026-09-30-06-55-37.gh-issue-158451.pZE56z.rst b/Misc/NEWS.d/next/Tools-Demos/2026-09-30-06-55-37.gh-issue-158451.pZE56z.rst new file mode 100644 index 00000000000000..2d84d6f0a1393f --- /dev/null +++ b/Misc/NEWS.d/next/Tools-Demos/2026-09-30-06-55-37.gh-issue-158451.pZE56z.rst @@ -0,0 +1,3 @@ +``python-gdb.py`` is now able to format an invalid Unicode string: render +invalid characters as ``\Uhhhhhhhh`` instead of raising an exception. Patch by +Victor Stinner. diff --git a/Tools/gdb/libpython.py b/Tools/gdb/libpython.py index 422e4f605920a0..a7f89442dd21db 100755 --- a/Tools/gdb/libpython.py +++ b/Tools/gdb/libpython.py @@ -46,6 +46,9 @@ import sys +MAX_UNICODE = 0x10_ffff + + # Look up the gdb.Type for some standard types: # Those need to be refreshed as types (pointer sizes) may change when # gdb loads different executables @@ -1378,11 +1381,15 @@ def write_repr(self, out, visited): class PyBytesObjectPtr(PyObjectPtr): _typename = 'PyBytesObject' - def __str__(self): + def get_bytes(self): field_ob_size = self.field('ob_size') field_ob_sval = self.field('ob_sval') char_ptr = field_ob_sval.address.cast(_type_unsigned_char_ptr()) - return ''.join([chr(char_ptr[i]) for i in safe_range(field_ob_size)]) + return [char_ptr[i] for i in safe_range(field_ob_size)] + + def __str__(self): + as_bytes = self.get_bytes() + return ''.join([chr(byte) for byte in as_bytes]) def proxyval(self, visited): return str(self) @@ -1390,17 +1397,17 @@ def proxyval(self, visited): def write_repr(self, out, visited): # Write this out as a Python bytes literal, i.e. with a "b" prefix - # Get a PyStringObject* within the Python gdb process: - proxy = self.proxyval(visited) + as_bytes = self.get_bytes() # Transliteration of Python's Objects/bytesobject.c:PyBytes_Repr # to Python code: quote = "'" - if "'" in proxy and not '"' in proxy: + if ord("'") in as_bytes and ord('"') not in as_bytes: quote = '"' out.write('b') out.write(quote) - for byte in proxy: + for value in as_bytes: + byte = chr(value) if byte == quote or byte == '\\': out.write('\\') out.write(byte) @@ -1410,10 +1417,10 @@ def write_repr(self, out, visited): out.write('\\n') elif byte == '\r': out.write('\\r') - elif byte < ' ' or ord(byte) >= 0x7f: + elif value < ord(' ') or value >= 0x7f: out.write('\\x') - out.write(hexdigits[(ord(byte) & 0xf0) >> 4]) - out.write(hexdigits[ord(byte) & 0xf]) + out.write(hexdigits[(value & 0xf0) >> 4]) + out.write(hexdigits[value & 0xf]) else: out.write(byte) out.write(quote) @@ -1466,10 +1473,17 @@ def _unichr_is_printable(char): return unicodedata.category(char) not in ("C", "Z") +def safe_chr(i): + if i <= MAX_UNICODE: + return chr(i) + else: + return f'\\U{i:08x}' + + class PyUnicodeObjectPtr(PyObjectPtr): _typename = 'PyUnicodeObject' - def proxyval(self, visited): + def get_code_points(self): compact = self.field('_base') ascii = compact['_base'] state = ascii['state'] @@ -1491,11 +1505,13 @@ def proxyval(self, visited): # Gather a list of ints from the code point array; these are either # UCS-1, UCS-2 or UCS-4 code points: - code_points = [int(field_str[i]) for i in safe_range(field_length)] + return [int(field_str[i]) for i in safe_range(field_length)] + def proxyval(self, visited): + code_points = self.get_code_points() # Convert the int code points to unicode characters, and generate a # local unicode instance. - result = ''.join(map(chr, code_points)) + result = ''.join(map(safe_chr, code_points)) return result def write_repr(self, out, visited): @@ -1506,20 +1522,18 @@ def write_repr(self, out, visited): encoding = gdb.host_charset() # Get a PyUnicodeObject* within the Python gdb process: - proxy = self.proxyval(visited) + code_points = self.get_code_points() # Transliteration of Python's Object/unicodeobject.c:unicode_repr # to Python: - if "'" in proxy and '"' not in proxy: + if ord("'") in code_points and ord('"') not in code_points: quote = '"' else: quote = "'" out.write(quote) - i = 0 - while i < len(proxy): - ch = proxy[i] - i += 1 + for code_point in code_points: + ch = safe_chr(code_point) # Escape quotes and backslashes if ch == quote or ch == '\\': @@ -1535,24 +1549,24 @@ def write_repr(self, out, visited): out.write('\\r') # Map non-printable US ASCII to '\xhh' */ - elif ch < ' ' or ord(ch) == 0x7F: + elif ch < ' ' or code_point == 0x7F: out.write('\\x') - out.write(hexdigits[(ord(ch) >> 4) & 0x000F]) - out.write(hexdigits[ord(ch) & 0x000F]) + out.write(hexdigits[(code_point >> 4) & 0x000F]) + out.write(hexdigits[code_point & 0x000F]) # Copy ASCII characters as-is - elif ord(ch) < 0x7F: + elif code_point < 0x7F: out.write(ch) # Non-ASCII characters else: - ucs = ch - ch2 = None - - printable = ucs.isprintable() + if code_point <= MAX_UNICODE: + printable = ch.isprintable() + else: + printable = False if printable: try: - ucs.encode(encoding) + ch.encode(encoding) # LookupError or ValueError if the host charset is unknown # or invalid. except (UnicodeEncodeError, LookupError, ValueError): @@ -1561,14 +1575,7 @@ def write_repr(self, out, visited): # Map Unicode whitespace and control characters # (categories Z* and C* except ASCII space) if not printable: - if ch2 is not None: - # Match Python's representation of non-printable - # wide characters. - code = (ord(ch) & 0x03FF) << 10 - code |= ord(ch2) & 0x03FF - code += 0x00010000 - else: - code = ord(ucs) + code = code_point # Map 8-bit characters to '\\xhh' if code <= 0xff: @@ -1596,8 +1603,6 @@ def write_repr(self, out, visited): else: # Copy characters as-is out.write(ch) - if ch2 is not None: - out.write(ch2) out.write(quote)