Skip to content

Commit 09a2e7e

Browse files
miss-islingtonencukoutonghuarootrasmusfaber
authored
[3.12] gh-156002: Bound zipfile decompression for bzip2/LZMA/Zstandard (GH-156003) (GH-156362) (#156739)
* gh-156002: Bound zipfile decompression for bzip2/LZMA/Zstandard (GH-156003) (GH-156362) Patch by @tonghuaroot. zipfile.ZipExtFile._read1() bounds the output of each decompress() call for DEFLATE members by passing a max_length to zlib, but for bzip2, LZMA, and Zstandard members it called decompress() with no bound. A whole compressed chunk was therefore expanded into a single allocation before the data[:self._left] clip ran, so a consumer that deliberately reads in small chunks to limit memory (for example zf.open(name).read(8192)) was silently unprotected for non-DEFLATE members. A small, spec-conformant archive member declaring a large uncompressed size could drive multi-GB peak memory. _read1() now passes a per-call bound to the non-DEFLATE decompress() (mirroring the DEFLATE branch) and drains the decompressor's internal buffer across calls by checking needs_input before reading more compressed input. zipfile's LZMADecompressor wrapper forwards max_length and exposes needs_input so the bound also holds for LZMA members. (cherry picked from commit f897dbf) (cherry picked from commit 1b424c0) * gh-156002: Keep reading through monkey-patched zipfile decompressors (GH-157180) (GH-157557) (cherry picked from commit f507e69) Co-authored-by: Petr Viktorin <encukou@gmail.com> Co-authored-by: tonghuaroot <tonghuaroot@gmail.com> Co-authored-by: rasmusfaber <rfaber@gmail.com>
1 parent 764fd0a commit 09a2e7e

4 files changed

Lines changed: 143 additions & 5 deletions

File tree

‎Lib/test/test_zipfile/test_core.py‎

Lines changed: 104 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -2590,6 +2590,110 @@ def tearDown(self):
25902590
unlink(TESTFN2)
25912591

25922592

2593+
class AbstractBoundedDecompressTests:
2594+
# ZipExtFile._read1() bounds the output of each decompress() call so that a
2595+
# small member declaring a large uncompressed size cannot expand into one
2596+
# unbounded read.
2597+
def test_read1_output_is_bounded(self):
2598+
buf = io.BytesIO()
2599+
with zipfile.ZipFile(buf, "w", compression=self.compression) as zf:
2600+
zf.writestr("big", b"\0" * (4 * 1024 * 1024))
2601+
with zipfile.ZipFile(io.BytesIO(buf.getvalue())) as zf:
2602+
with zf.open("big") as f:
2603+
self.assertLessEqual(len(f._read1(100)), f.MIN_READ_SIZE)
2604+
2605+
2606+
class StoredBoundedDecompressTests(AbstractBoundedDecompressTests,
2607+
unittest.TestCase):
2608+
compression = zipfile.ZIP_STORED
2609+
2610+
2611+
@requires_zlib()
2612+
class DeflateBoundedDecompressTests(AbstractBoundedDecompressTests,
2613+
unittest.TestCase):
2614+
compression = zipfile.ZIP_DEFLATED
2615+
2616+
2617+
@requires_bz2()
2618+
class Bzip2BoundedDecompressTests(AbstractBoundedDecompressTests,
2619+
unittest.TestCase):
2620+
compression = zipfile.ZIP_BZIP2
2621+
2622+
2623+
@requires_lzma()
2624+
class LzmaBoundedDecompressTests(AbstractBoundedDecompressTests,
2625+
unittest.TestCase):
2626+
compression = zipfile.ZIP_LZMA
2627+
2628+
2629+
2630+
class MonkeypatchedDecompressorTests(unittest.TestCase):
2631+
# Some third-party projects monkey-patch _get_decompressor() to add
2632+
# additional compression schemes. This can break at any time as the
2633+
# internal compressor objects change.
2634+
# To protect users, we try to keep this case working.
2635+
# See also: GH-156002 and GH-113767.
2636+
COMPRESSION = 99
2637+
2638+
class Compressor:
2639+
"""Compressor with only the original BZ2Compressor API"""
2640+
def compress(self, data):
2641+
return data.swapcase()
2642+
2643+
def flush(self):
2644+
return b''
2645+
2646+
class Decompressor:
2647+
"""Decompressor with only the 3.3+ BZ2Decompressor API"""
2648+
eof = False
2649+
2650+
def decompress(self, data):
2651+
return data.swapcase()
2652+
2653+
def setUp(self):
2654+
orig_check_compression = zipfile._check_compression
2655+
orig_get_compressor = zipfile._get_compressor
2656+
orig_get_decompressor = zipfile._get_decompressor
2657+
2658+
def check_compression(compression):
2659+
if compression != self.COMPRESSION:
2660+
orig_check_compression(compression)
2661+
2662+
def get_compressor(compress_type, compresslevel=None):
2663+
if compress_type == self.COMPRESSION:
2664+
return self.Compressor()
2665+
return orig_get_compressor(compress_type, compresslevel)
2666+
2667+
def get_decompressor(compress_type):
2668+
if compress_type == self.COMPRESSION:
2669+
return self.Decompressor()
2670+
return orig_get_decompressor(compress_type)
2671+
2672+
self.enterContext(mock.patch.object(
2673+
zipfile, '_check_compression', check_compression))
2674+
self.enterContext(mock.patch.object(
2675+
zipfile, '_get_compressor', get_compressor))
2676+
self.enterContext(mock.patch.object(
2677+
zipfile, '_get_decompressor', get_decompressor))
2678+
2679+
def test_roundtrip_monkeypatched_decompressor(self):
2680+
data = bytes(range(256)) * 8
2681+
buf = io.BytesIO()
2682+
with zipfile.ZipFile(buf, "w", compression=self.COMPRESSION) as zf:
2683+
zf.writestr("member", data)
2684+
self.assertIn(data.swapcase(), buf.getvalue())
2685+
with zipfile.ZipFile(io.BytesIO(buf.getvalue())) as zf:
2686+
self.assertEqual(zf.read("member"), data)
2687+
with zf.open("member") as f:
2688+
self.assertEqual(f.read(100), data[:100])
2689+
self.assertEqual(f.read1(100), data[100:200])
2690+
f.seek(-100, os.SEEK_END)
2691+
self.assertEqual(f.read(), data[-100:])
2692+
# Rewinding past the read buffer re-creates the decompressor.
2693+
f.seek(0)
2694+
self.assertEqual(f.read(), data)
2695+
2696+
25932697
class AbstractBadCrcTests:
25942698
def test_testzip_with_bad_crc(self):
25952699
"""Tests that files with bad CRCs return their name from testzip."""

‎Lib/zipfile/__init__.py‎

Lines changed: 30 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -698,7 +698,16 @@ def __init__(self):
698698
self._unconsumed = b''
699699
self.eof = False
700700

701-
def decompress(self, data):
701+
@property
702+
def needs_input(self):
703+
# While the LZMA properties header is still being buffered, more input
704+
# is required; afterwards defer to the wrapped decompressor so a bounded
705+
# decompress() call can be drained across reads.
706+
if self._decomp is None:
707+
return True
708+
return self._decomp.needs_input
709+
710+
def decompress(self, data, max_length=-1):
702711
if self._decomp is None:
703712
self._unconsumed += data
704713
if len(self._unconsumed) <= 4:
@@ -714,7 +723,7 @@ def decompress(self, data):
714723
data = self._unconsumed[4 + psize:]
715724
del self._unconsumed
716725

717-
result = self._decomp.decompress(data)
726+
result = self._decomp.decompress(data, max_length)
718727
self.eof = self._decomp.eof
719728
return result
720729

@@ -1074,8 +1083,15 @@ def _read1(self, n):
10741083
data = self._decompressor.unconsumed_tail
10751084
if n > len(data):
10761085
data += self._read2(n - len(data))
1077-
else:
1086+
elif self._compress_type == ZIP_STORED:
10781087
data = self._read2(n)
1088+
else:
1089+
# bzip2/lzma/zstd: a bounded decompress() call may leave input
1090+
# buffered inside the decompressor; drain that before reading more.
1091+
if getattr(self._decompressor, "needs_input", True):
1092+
data = self._read2(n)
1093+
else:
1094+
data = b''
10791095

10801096
if self._compress_type == ZIP_STORED:
10811097
self._eof = self._compress_left <= 0
@@ -1088,8 +1104,17 @@ def _read1(self, n):
10881104
if self._eof:
10891105
data += self._decompressor.flush()
10901106
else:
1091-
data = self._decompressor.decompress(data)
1092-
self._eof = self._decompressor.eof or self._compress_left <= 0
1107+
# Bound the output of a single decompress() call (mirroring the
1108+
# DEFLATE path above) so that a small compressed member cannot
1109+
# expand into one unbounded read.
1110+
try:
1111+
data = self._decompressor.decompress(data, max(n, self.MIN_READ_SIZE))
1112+
except TypeError:
1113+
# See MonkeypatchedDecompressorTests in test_core.py
1114+
data = self._decompressor.decompress(data)
1115+
self._eof = (self._decompressor.eof or
1116+
self._compress_left <= 0 and
1117+
getattr(self._decompressor, "needs_input", True))
10931118

10941119
data = data[:self._left]
10951120
self._left -= len(data)
Lines changed: 5 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,5 @@
1+
:mod:`zipfile` again reads members through a third-party decompressor
2+
installed by monkey-patching the private ``_get_decompressor()`` to return an
3+
object that only implements old BZ2Decompressor API from Python 3.3.
4+
Note that decompressors without ``needs_input`` and two-argument
5+
``decompress()`` are vulnerable to :cve:`2026-15310`.
Lines changed: 4 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,4 @@
1+
Bound the amount of data :mod:`zipfile` decompresses per read for members
2+
compressed with bzip2, LZMA, or Zstandard, matching the existing limit for
3+
deflate. A small archive member could previously expand into an unbounded
4+
allocation even when read in small chunks.

0 commit comments

Comments
 (0)