Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions Include/internal/pycore_bytesobject.h
Original file line number Diff line number Diff line change
Expand Up @@ -19,6 +19,10 @@ extern PyObject* _PyBytes_FormatEx(
* reference rather than modifying its first argument in place. */
extern PyObject* _PyBytes_Concat(PyObject *a, PyObject *b);

/* Clear the hash so it will be lazily recomputed. Use when going from an
immutable bytes object to a mutable one. */
extern void _PyBytes_ClearHash(PyObject *op);

extern PyObject* _PyBytes_FromHex(
PyObject *string,
int use_bytearray);
Expand Down
26 changes: 26 additions & 0 deletions Lib/test/test_bytes.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@
import re
import sys
import copy
lazy import codecs
import functools
import pickle
import tempfile
Expand Down Expand Up @@ -1706,6 +1707,31 @@ def test_take_bytes_optimization(self):
bytes_header_size = sys.getsizeof(b'')
self.assertEqual(ba.__alloc__(), 499 + bytes_header_size)

def test_take_bytes_hash(self):
# gh-158219: Ensure `bytearray` resets hash when adopting an encoded
# bytes.

def encode(string, errors='strict'):
encoded = string.encode('utf-8')
hash(encoded) # a codec may hash its own output
return encoded, len(string)

def hashing_codec(name):
if name != 'test_take_bytes_hash':
return None
return codecs.CodecInfo(encode, None, name=name)

codecs.register(hashing_codec)
self.addCleanup(codecs.unregister, hashing_codec)

ba = bytearray('hello', 'test_take_bytes_hash')

# Object should be same bytes and hash should include mutation.
ba[0] = ord('H')
taken = ba.take_bytes()
self.assertEqual(taken, b'Hello')
self.assertEqual(hash(taken), hash(b'Hello'))

def test_take_bytes_reentrant_resize(self):
# gh-153570: n.__index__() can resize the bytearray, so take_bytes()
# must re-read the size afterwards. It cached the size before the
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,2 @@
Fix :meth:`bytearray.take_bytes` returning a :class:`bytes` object with an
incorrect hash when the data was modified but not resized.
6 changes: 6 additions & 0 deletions Objects/bytearrayobject.c
Original file line number Diff line number Diff line change
Expand Up @@ -61,6 +61,12 @@ bytearray_reinit_from_bytes(PyByteArrayObject *self, Py_ssize_t size)
/* Only the empty bytes may be immortal. */
assert((alloc == 0) == _Py_IsImmortal(self->ob_bytes_object));

/* Bytes may be uniquely referenced with a hash set. Clear the hash so after
mutation it will be recomputed (gh-158219). */
if (!_Py_IsImmortal(self->ob_bytes_object)) {
_PyBytes_ClearHash(self->ob_bytes_object);
}

self->ob_bytes = self->ob_start = PyBytes_AS_STRING(self->ob_bytes_object);
Py_SET_SIZE(self, size);
FT_ATOMIC_STORE_SSIZE_RELAXED(self->ob_alloc, alloc);
Expand Down
18 changes: 18 additions & 0 deletions Objects/bytesobject.c
Original file line number Diff line number Diff line change
Expand Up @@ -76,6 +76,20 @@ _Py_COMP_DIAG_IGNORE_DEPR_DECLS
_Py_COMP_DIAG_POP
}

void
_PyBytes_ClearHash(PyObject *op)
{
// Safety check.
assert(PyBytes_Check(op));
assert(!_Py_IsImmortal(op));

// Clear before checking mutability.
set_ob_shash(_PyBytes_CAST(op), -1);

// Comprehensive checks: These require the hash is `-1`.
assert(_PyBytes_IsMutable(op));
}


/*
For PyBytes_FromString(), the parameter 'str' points to a null-terminated
Expand Down Expand Up @@ -3346,6 +3360,10 @@ _PyBytes_IsMutable(PyObject *self)
unsigned char ch = PyBytes_AS_STRING(self)[0];
assert(self != (PyObject*)CHARACTER(ch));
}

// There should not be a computed hash. Mutations to the bytes mean the hash
// needs to be recalculated (gh-158219).
assert(get_ob_shash((PyBytesObject *)self) == -1);
return 1;
}
#endif
Expand Down
Loading