Skip to content

Commit dcdd406

Browse files
miss-islingtonencukoutonghuarootrasmusfaber
authored
[3.13] gh-156002: Bound zipfile decompression for bzip2/LZMA (GH-156003) (GH-156362) (#156738)
* [3.15] gh-156002: Bound zipfile decompression for bzip2/LZMA/Zstandard (GH-156003) (GH-156362) Patch by @tonghuaroot. zipfile.ZipExtFile._read1() bounds the output of each decompress() call for DEFLATE members by passing a max_length to zlib, but for bzip2, LZMA, and Zstandard members it called decompress() with no bound. A whole compressed chunk was therefore expanded into a single allocation before the data[:self._left] clip ran, so a consumer that deliberately reads in small chunks to limit memory (for example zf.open(name).read(8192)) was silently unprotected for non-DEFLATE members. A small, spec-conformant archive member declaring a large uncompressed size could drive multi-GB peak memory. _read1() now passes a per-call bound to the non-DEFLATE decompress() (mirroring the DEFLATE branch) and drains the decompressor's internal buffer across calls by checking needs_input before reading more compressed input. zipfile's LZMADecompressor wrapper forwards max_length and exposes needs_input so the bound also holds for LZMA members. (cherry picked from commit f897dbf) (cherry picked from commit 1b424c0) * gh-156002: Keep reading through monkey-patched zipfile decompressors (GH-157180) (GH-157557) (cherry picked from commit f507e69) Co-authored-by: Petr Viktorin <encukou@gmail.com> Co-authored-by: tonghuaroot <tonghuaroot@gmail.com> Co-authored-by: rasmusfaber <rfaber@gmail.com>
1 parent fc9b015 commit dcdd406

4 files changed

Lines changed: 143 additions & 5 deletions

File tree

‎Lib/test/test_zipfile/test_core.py‎

Lines changed: 104 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -2619,6 +2619,110 @@ def tearDown(self):
26192619
unlink(TESTFN2)
26202620

26212621

2622+
class AbstractBoundedDecompressTests:
2623+
# ZipExtFile._read1() bounds the output of each decompress() call so that a
2624+
# small member declaring a large uncompressed size cannot expand into one
2625+
# unbounded read.
2626+
def test_read1_output_is_bounded(self):
2627+
buf = io.BytesIO()
2628+
with zipfile.ZipFile(buf, "w", compression=self.compression) as zf:
2629+
zf.writestr("big", b"\0" * (4 * 1024 * 1024))
2630+
with zipfile.ZipFile(io.BytesIO(buf.getvalue())) as zf:
2631+
with zf.open("big") as f:
2632+
self.assertLessEqual(len(f._read1(100)), f.MIN_READ_SIZE)
2633+
2634+
2635+
class StoredBoundedDecompressTests(AbstractBoundedDecompressTests,
2636+
unittest.TestCase):
2637+
compression = zipfile.ZIP_STORED
2638+
2639+
2640+
@requires_zlib()
2641+
class DeflateBoundedDecompressTests(AbstractBoundedDecompressTests,
2642+
unittest.TestCase):
2643+
compression = zipfile.ZIP_DEFLATED
2644+
2645+
2646+
@requires_bz2()
2647+
class Bzip2BoundedDecompressTests(AbstractBoundedDecompressTests,
2648+
unittest.TestCase):
2649+
compression = zipfile.ZIP_BZIP2
2650+
2651+
2652+
@requires_lzma()
2653+
class LzmaBoundedDecompressTests(AbstractBoundedDecompressTests,
2654+
unittest.TestCase):
2655+
compression = zipfile.ZIP_LZMA
2656+
2657+
2658+
2659+
class MonkeypatchedDecompressorTests(unittest.TestCase):
2660+
# Some third-party projects monkey-patch _get_decompressor() to add
2661+
# additional compression schemes. This can break at any time as the
2662+
# internal compressor objects change.
2663+
# To protect users, we try to keep this case working.
2664+
# See also: GH-156002 and GH-113767.
2665+
COMPRESSION = 99
2666+
2667+
class Compressor:
2668+
"""Compressor with only the original BZ2Compressor API"""
2669+
def compress(self, data):
2670+
return data.swapcase()
2671+
2672+
def flush(self):
2673+
return b''
2674+
2675+
class Decompressor:
2676+
"""Decompressor with only the 3.3+ BZ2Decompressor API"""
2677+
eof = False
2678+
2679+
def decompress(self, data):
2680+
return data.swapcase()
2681+
2682+
def setUp(self):
2683+
orig_check_compression = zipfile._check_compression
2684+
orig_get_compressor = zipfile._get_compressor
2685+
orig_get_decompressor = zipfile._get_decompressor
2686+
2687+
def check_compression(compression):
2688+
if compression != self.COMPRESSION:
2689+
orig_check_compression(compression)
2690+
2691+
def get_compressor(compress_type, compresslevel=None):
2692+
if compress_type == self.COMPRESSION:
2693+
return self.Compressor()
2694+
return orig_get_compressor(compress_type, compresslevel)
2695+
2696+
def get_decompressor(compress_type):
2697+
if compress_type == self.COMPRESSION:
2698+
return self.Decompressor()
2699+
return orig_get_decompressor(compress_type)
2700+
2701+
self.enterContext(mock.patch.object(
2702+
zipfile, '_check_compression', check_compression))
2703+
self.enterContext(mock.patch.object(
2704+
zipfile, '_get_compressor', get_compressor))
2705+
self.enterContext(mock.patch.object(
2706+
zipfile, '_get_decompressor', get_decompressor))
2707+
2708+
def test_roundtrip_monkeypatched_decompressor(self):
2709+
data = bytes(range(256)) * 8
2710+
buf = io.BytesIO()
2711+
with zipfile.ZipFile(buf, "w", compression=self.COMPRESSION) as zf:
2712+
zf.writestr("member", data)
2713+
self.assertIn(data.swapcase(), buf.getvalue())
2714+
with zipfile.ZipFile(io.BytesIO(buf.getvalue())) as zf:
2715+
self.assertEqual(zf.read("member"), data)
2716+
with zf.open("member") as f:
2717+
self.assertEqual(f.read(100), data[:100])
2718+
self.assertEqual(f.read1(100), data[100:200])
2719+
f.seek(-100, os.SEEK_END)
2720+
self.assertEqual(f.read(), data[-100:])
2721+
# Rewinding past the read buffer re-creates the decompressor.
2722+
f.seek(0)
2723+
self.assertEqual(f.read(), data)
2724+
2725+
26222726
class AbstractBadCrcTests:
26232727
def test_testzip_with_bad_crc(self):
26242728
"""Tests that files with bad CRCs return their name from testzip."""

‎Lib/zipfile/__init__.py‎

Lines changed: 30 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -726,7 +726,16 @@ def __init__(self):
726726
self._unconsumed = b''
727727
self.eof = False
728728

729-
def decompress(self, data):
729+
@property
730+
def needs_input(self):
731+
# While the LZMA properties header is still being buffered, more input
732+
# is required; afterwards defer to the wrapped decompressor so a bounded
733+
# decompress() call can be drained across reads.
734+
if self._decomp is None:
735+
return True
736+
return self._decomp.needs_input
737+
738+
def decompress(self, data, max_length=-1):
730739
if self._decomp is None:
731740
self._unconsumed += data
732741
if len(self._unconsumed) <= 4:
@@ -742,7 +751,7 @@ def decompress(self, data):
742751
data = self._unconsumed[4 + psize:]
743752
del self._unconsumed
744753

745-
result = self._decomp.decompress(data)
754+
result = self._decomp.decompress(data, max_length)
746755
self.eof = self._decomp.eof
747756
return result
748757

@@ -1102,8 +1111,15 @@ def _read1(self, n):
11021111
data = self._decompressor.unconsumed_tail
11031112
if n > len(data):
11041113
data += self._read2(n - len(data))
1105-
else:
1114+
elif self._compress_type == ZIP_STORED:
11061115
data = self._read2(n)
1116+
else:
1117+
# bzip2/lzma/zstd: a bounded decompress() call may leave input
1118+
# buffered inside the decompressor; drain that before reading more.
1119+
if getattr(self._decompressor, "needs_input", True):
1120+
data = self._read2(n)
1121+
else:
1122+
data = b''
11071123

11081124
if self._compress_type == ZIP_STORED:
11091125
self._eof = self._compress_left <= 0
@@ -1116,8 +1132,17 @@ def _read1(self, n):
11161132
if self._eof:
11171133
data += self._decompressor.flush()
11181134
else:
1119-
data = self._decompressor.decompress(data)
1120-
self._eof = self._decompressor.eof or self._compress_left <= 0
1135+
# Bound the output of a single decompress() call (mirroring the
1136+
# DEFLATE path above) so that a small compressed member cannot
1137+
# expand into one unbounded read.
1138+
try:
1139+
data = self._decompressor.decompress(data, max(n, self.MIN_READ_SIZE))
1140+
except TypeError:
1141+
# See MonkeypatchedDecompressorTests in test_core.py
1142+
data = self._decompressor.decompress(data)
1143+
self._eof = (self._decompressor.eof or
1144+
self._compress_left <= 0 and
1145+
getattr(self._decompressor, "needs_input", True))
11211146

11221147
data = data[:self._left]
11231148
self._left -= len(data)
Lines changed: 5 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,5 @@
1+
:mod:`zipfile` again reads members through a third-party decompressor
2+
installed by monkey-patching the private ``_get_decompressor()`` to return an
3+
object that only implements old BZ2Decompressor API from Python 3.3.
4+
Note that decompressors without ``needs_input`` and two-argument
5+
``decompress()`` are vulnerable to :cve:`2026-15310`.
Lines changed: 4 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,4 @@
1+
Bound the amount of data :mod:`zipfile` decompresses per read for members
2+
compressed with bzip2, LZMA, or Zstandard, matching the existing limit for
3+
deflate. A small archive member could previously expand into an unbounded
4+
allocation even when read in small chunks.

0 commit comments

Comments
 (0)