Skip to content

Commit 48881a8

Browse files
encukoutonghuaroot
authored andcommitted
[3.15] gh-156002: Bound zipfile decompression for bzip2/LZMA/Zstandard (GH-156003) (GH-156362)
Patch by @tonghuaroot. zipfile.ZipExtFile._read1() bounds the output of each decompress() call for DEFLATE members by passing a max_length to zlib, but for bzip2, LZMA, and Zstandard members it called decompress() with no bound. A whole compressed chunk was therefore expanded into a single allocation before the data[:self._left] clip ran, so a consumer that deliberately reads in small chunks to limit memory (for example zf.open(name).read(8192)) was silently unprotected for non-DEFLATE members. A small, spec-conformant archive member declaring a large uncompressed size could drive multi-GB peak memory. _read1() now passes a per-call bound to the non-DEFLATE decompress() (mirroring the DEFLATE branch) and drains the decompressor's internal buffer across calls by checking needs_input before reading more compressed input. zipfile's LZMADecompressor wrapper forwards max_length and exposes needs_input so the bound also holds for LZMA members. (cherry picked from commit f897dbf) (cherry picked from commit 1b424c0) Co-authored-by: Petr Viktorin <encukou@gmail.com> Co-authored-by: tonghuaroot <tonghuaroot@gmail.com>
1 parent 9eaf48a commit 48881a8

3 files changed

Lines changed: 79 additions & 5 deletions

File tree

‎Lib/test/test_zipfile.py‎

Lines changed: 42 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -2418,6 +2418,48 @@ def tearDown(self):
24182418
unlink(TESTFN2)
24192419

24202420

2421+
class AbstractBoundedDecompressTests:
2422+
# ZipExtFile._read1() bounds the output of each decompress() call so that a
2423+
# small member declaring a large uncompressed size cannot expand into one
2424+
# unbounded read.
2425+
def test_read1_output_is_bounded(self):
2426+
buf = io.BytesIO()
2427+
with zipfile.ZipFile(buf, "w", compression=self.compression) as zf:
2428+
zf.writestr("big", b"\0" * (4 * 1024 * 1024))
2429+
with zipfile.ZipFile(io.BytesIO(buf.getvalue())) as zf:
2430+
with zf.open("big") as f:
2431+
self.assertLessEqual(len(f._read1(100)), f.MIN_READ_SIZE)
2432+
2433+
2434+
class StoredBoundedDecompressTests(AbstractBoundedDecompressTests,
2435+
unittest.TestCase):
2436+
compression = zipfile.ZIP_STORED
2437+
2438+
2439+
@requires_zlib()
2440+
class DeflateBoundedDecompressTests(AbstractBoundedDecompressTests,
2441+
unittest.TestCase):
2442+
compression = zipfile.ZIP_DEFLATED
2443+
2444+
2445+
@requires_bz2()
2446+
class Bzip2BoundedDecompressTests(AbstractBoundedDecompressTests,
2447+
unittest.TestCase):
2448+
compression = zipfile.ZIP_BZIP2
2449+
2450+
2451+
@requires_lzma()
2452+
class LzmaBoundedDecompressTests(AbstractBoundedDecompressTests,
2453+
unittest.TestCase):
2454+
compression = zipfile.ZIP_LZMA
2455+
2456+
2457+
@requires_zstd()
2458+
class ZstdBoundedDecompressTests(AbstractBoundedDecompressTests,
2459+
unittest.TestCase):
2460+
compression = zipfile.ZIP_ZSTANDARD
2461+
2462+
24212463
class AbstractBadCrcTests:
24222464
def test_testzip_with_bad_crc(self):
24232465
"""Tests that files with bad CRCs return their name from testzip."""

‎Lib/zipfile.py‎

Lines changed: 33 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -676,7 +676,16 @@ def __init__(self):
676676
self._unconsumed = b''
677677
self.eof = False
678678

679-
def decompress(self, data):
679+
@property
680+
def _needs_input(self):
681+
# While the LZMA properties header is still being buffered, more input
682+
# is required; afterwards defer to the wrapped decompressor so a bounded
683+
# decompress() call can be drained across reads.
684+
if self._decomp is None:
685+
return True
686+
return self._decomp.needs_input
687+
688+
def decompress(self, data, max_length=-1):
680689
if self._decomp is None:
681690
self._unconsumed += data
682691
if len(self._unconsumed) <= 4:
@@ -692,7 +701,7 @@ def decompress(self, data):
692701
data = self._unconsumed[4 + psize:]
693702
del self._unconsumed
694703

695-
result = self._decomp.decompress(data)
704+
result = self._decomp.decompress(data, max_length)
696705
self.eof = self._decomp.eof
697706
return result
698707

@@ -752,6 +761,13 @@ def _get_compressor(compress_type, compresslevel=None):
752761
return None
753762

754763

764+
def _decompressor_needs_input(decompressor):
765+
# bz2/zstd expose the stdlib decompressor's public needs_input; the LZMA
766+
# wrapper keeps it private (_needs_input) to avoid adding public API.
767+
needs_input = getattr(decompressor, "needs_input", None)
768+
return decompressor._needs_input if needs_input is None else needs_input
769+
770+
755771
def _get_decompressor(compress_type):
756772
_check_compression(compress_type)
757773
if compress_type == ZIP_STORED:
@@ -1048,8 +1064,15 @@ def _read1(self, n):
10481064
data = self._decompressor.unconsumed_tail
10491065
if n > len(data):
10501066
data += self._read2(n - len(data))
1051-
else:
1067+
elif self._compress_type == ZIP_STORED:
10521068
data = self._read2(n)
1069+
else:
1070+
# bzip2/lzma/zstd: a bounded decompress() call may leave input
1071+
# buffered inside the decompressor; drain that before reading more.
1072+
if _decompressor_needs_input(self._decompressor):
1073+
data = self._read2(n)
1074+
else:
1075+
data = b''
10531076

10541077
if self._compress_type == ZIP_STORED:
10551078
self._eof = self._compress_left <= 0
@@ -1062,8 +1085,13 @@ def _read1(self, n):
10621085
if self._eof:
10631086
data += self._decompressor.flush()
10641087
else:
1065-
data = self._decompressor.decompress(data)
1066-
self._eof = self._decompressor.eof or self._compress_left <= 0
1088+
# Bound the output of a single decompress() call (mirroring the
1089+
# DEFLATE path above) so that a small compressed member cannot
1090+
# expand into one unbounded read.
1091+
data = self._decompressor.decompress(data, max(n, self.MIN_READ_SIZE))
1092+
self._eof = (self._decompressor.eof or
1093+
self._compress_left <= 0 and
1094+
_decompressor_needs_input(self._decompressor))
10671095

10681096
data = data[:self._left]
10691097
self._left -= len(data)
Lines changed: 4 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,4 @@
1+
Bound the amount of data :mod:`zipfile` decompresses per read for members
2+
compressed with bzip2, LZMA, or Zstandard, matching the existing limit for
3+
deflate. A small archive member could previously expand into an unbounded
4+
allocation even when read in small chunks.

0 commit comments

Comments
 (0)