From 5a2612763a9e201878787b2afc368815fcdf5725 Mon Sep 17 00:00:00 2001 From: Jeremy Howard Date: Tue, 25 Aug 2026 19:38:17 +1000 Subject: [PATCH] Add parallel compression and archive creation --- Cargo.toml | 4 +- DEV.md | 70 ++++- README.md | 92 +++++- pyproject.toml | 2 +- python/fbz/__init__.py | 8 +- src/bin/fbz.rs | 252 ++++++++++++++-- src/bin/fbz/archive_create.rs | 34 +++ src/bin/fbz/tar_create.rs | 43 +++ src/bz2_encode/LICENSE-MIT | 21 ++ src/bz2_encode/bitwriter.rs | 165 +++++++++++ src/bz2_encode/bwt.rs | 381 ++++++++++++++++++++++++ src/bz2_encode/huffman.rs | 283 ++++++++++++++++++ src/bz2_encode/mod.rs | 237 +++++++++++++++ src/bz2_encode/mtf.rs | 214 ++++++++++++++ src/bz2_encode/rle1.rs | 225 ++++++++++++++ src/crc.rs | 40 ++- src/deflate.rs | 23 +- src/deflate_encode.rs | 535 ++++++++++++++++++++++++++++++++++ src/encode.rs | 165 +++++++++++ src/error.rs | 9 + src/gzip.rs | 71 ++++- src/lib.rs | 24 +- src/lz4.rs | 4 +- src/lz4_encode.rs | 298 +++++++++++++++++++ src/matchfinder.rs | 117 ++++++++ src/pipeline.rs | 107 ++++++- src/zip.rs | 453 ++++++++++++++++++++++++++++ tests/archive_perf.rs | 1 - tests/bz2_encode.rs | 45 +++ tests/cli.rs | 145 ++++++++- tests/common/process.rs | 3 +- tests/compression_cli_perf.rs | 218 ++++++++++++++ tests/compression_perf.rs | 83 ++++++ tests/encode.rs | 31 ++ tests/gzip_encode.rs | 49 ++++ tests/lz4_encode.rs | 40 +++ tests/lz4_perf.rs | 1 - tests/reader_perf.rs | 1 - tests/support/mod.rs | 14 + tests/test_api.py | 9 +- tests/zip_encode.rs | 56 ++++ 41 files changed, 4497 insertions(+), 76 deletions(-) create mode 100644 src/bin/fbz/archive_create.rs create mode 100644 src/bin/fbz/tar_create.rs create mode 100644 src/bz2_encode/LICENSE-MIT create mode 100644 src/bz2_encode/bitwriter.rs create mode 100644 src/bz2_encode/bwt.rs create mode 100644 src/bz2_encode/huffman.rs create mode 100644 src/bz2_encode/mod.rs create mode 100644 src/bz2_encode/mtf.rs create mode 100644 src/bz2_encode/rle1.rs create mode 100644 src/deflate_encode.rs create mode 100644 src/encode.rs create mode 100644 src/lz4_encode.rs create mode 100644 src/matchfinder.rs create mode 100644 src/zip.rs create mode 100644 tests/bz2_encode.rs create mode 100644 tests/compression_cli_perf.rs create mode 100644 tests/compression_perf.rs create mode 100644 tests/encode.rs create mode 100644 tests/gzip_encode.rs create mode 100644 tests/lz4_encode.rs create mode 100644 tests/zip_encode.rs diff --git a/Cargo.toml b/Cargo.toml index 92f15f7..b010b36 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -4,11 +4,11 @@ version = "0.1.9" edition = "2024" rust-version = "1.91" license = "Apache-2.0" -description = "Compression-format research workbench with fast bzip2, gzip, LZ4, and ZIP decompression" +description = "Fast parallel compression and decompression for bzip2, gzip, LZ4, tar, and ZIP" repository = "https://github.com/AnswerDotAI/fbz" homepage = "https://github.com/AnswerDotAI/fbz" documentation = "https://docs.rs/fbz" -include = ["/src/**", "/Cargo.toml", "/README.md", "/LICENSE"] +include = ["/src/**", "/Cargo.toml", "/README.md", "/DEV.md", "/LICENSE"] [lib] name = "fbz" diff --git a/DEV.md b/DEV.md index 8b46af6..0d8e9df 100644 --- a/DEV.md +++ b/DEV.md @@ -7,13 +7,18 @@ ```text src/bitreader.rs bounded MSB-first in-memory bit reads src/block.rs independently decodable block construction and validation -src/crc.rs bzip2 block and combined-stream CRC primitives +src/crc.rs incremental bzip2 block and combined-stream CRC primitives src/decode.rs serial/parallel decode scheduling and index construction src/decoder.rs bzip2 block machinery and 12-bit Huffman fast tables +src/bz2_encode/ block-parallel BWT/MTF/Huffman bzip2 encoder +src/encode.rs unified stream compression API and progress adapter src/format.rs cheap structural scan for header and marker candidates src/gzip.rs gzip framing, LSB-first DEFLATE implementation, CRC32, and reports src/deflate.rs format-neutral raw-DEFLATE API shared by gzip and ZIP +src/deflate_encode.rs segmented raw-DEFLATE and gzip encoder machinery src/lz4.rs safe LZ4 frame/block decoder and independent-block scheduling +src/lz4_encode.rs independent-block LZ4 frame encoder +src/matchfinder.rs shared latest-match and hash-chain LZ match finders src/history.rs shared overlapping LZ back-reference expansion src/output.rs owned/borrowed decoded-output sink abstraction src/pipeline.rs shared ordered, byte-budgeted, staged worker scheduler @@ -21,7 +26,10 @@ src/index.rs stable persistent index format src/indexed.rs seekable decoded view and block cache src/lib.rs public Rust API and private PyO3 binding src/source.rs owned and memory-mapped compressed sources +src/zip.rs streaming ZIP/Zip64 creation over raw DEFLATE +src/bin/fbz/archive_create.rs safe input-to-archive name derivation src/bin/fbz/archive_extract.rs shared same-filesystem staging and atomic commit +src/bin/fbz/tar_create.rs streaming tar composition over each encoder src/bin/fbz/tar_extract.rs bounded decode-to-tar bridge src/bin/fbz/zip_extract.rs ZIP parsing policy and adaptive entry extraction python/fbz/ thin Python I/O wrapper over fbz._core @@ -35,19 +43,27 @@ The current bzip2 scanner deliberately does not treat 48-bit marker matches or l The gzip decoder is an in-repo RFC 1952/RFC 1951 implementation rather than a wrapper around a production codec. It parses optional headers and concatenated members, decodes stored/fixed/dynamic blocks, maintains the 32 KiB LZ77 history, and validates FHCRC, CRC32, and ISIZE. For large inputs, independently discovered dynamic-block boundaries seed unknown history with compact markers. Primary jobs compute the CRC of each known clean suffix before ordered resolution. Resolution workers resolve the marker prefix, hash that prefix, and combine the two CRCs without rescanning the clean bytes. A marker-free history switches the same decoder to byte output. Reports retain member boundaries, DEFLATE block ranges, and accepted/fallback chunk counts. `crc32fast` is the sole production helper; `flate2` is dev-only. +The gzip encoder creates one interoperable member rather than concatenating independently compressed members. It carries each segment's preceding 32 KiB dictionary into a 1 MiB raw-DEFLATE job, emits a non-final byte-aligned sync boundary, then commits segments in order before one final block and CRC32/ISIZE trailer. Fixed, dynamic, and stored encodings are built for each segment and the shortest is retained. This exposes within-stream parallelism while preserving normal gzip semantics and compression ratio. The raw encoder is also the ZIP creation codec. + +The bzip2 encoder incrementally forms RLE1 blocks and schedules BWT, MTF/RLE2, and grouped Huffman coding independently. Completed bit strings are concatenated exactly rather than padded to bytes, and block CRCs are combined in stream order. The safe SA-IS BWT, MTF/RLE2, and Huffman pieces are adapted from crabz2 0.4.0 under the bundled MIT license; fbz supplies the streaming orchestration, byte-budgeted scheduler, and shared CRC. Automatic mode caps at 12 BWT workers because 12→18 improved the SimpleWiki measurement only 7% while increasing RSS from about 445 to 581 MiB. Explicit `-P` remains exact. + The decoder remains independent of files, Python, and the CLI. Parallel scanning/decoding and indexed seeking are layered over it. Native workers never call Python. Large offsets use explicit 64-bit bit/byte types, and speculative block-marker hits are accepted only when they form an exact stream chain with valid block and combined stream CRCs. Core decode APIs report completed compressed and decoded byte counts without knowing anything about terminals. The CLI selects bzip2, gzip, LZ4, or ZIP by a recognised extension and falls back to magic for stdin or unknown names. It layers delayed, rate-limited TTY progress rendering over the shared callbacks; redirected stderr and `--quiet` produce no progress output. Decoded files use same-directory temporary files and atomic persistence, then inherit the compressed input's modification time and permissions. `--rm` removes an input only after decode, persistence, and metadata copying all succeed. An `OutputSink` wrapper enforces output-size limits, so each decoder has one code path for files, stdout, validation, listing, and archive extraction. The LZ4 decoder parses current frames, concatenation, and skippable frames itself. It validates descriptor bits and XXH32 header, block, and content checksums, and bounds every literal and match before writing. A frame header creates an incremental block cursor rather than a complete layout. Independent blocks are gathered into at most 64-entry batches—only until there is enough work to amortize the pool—and become ordinary `pipeline::Job`s. A parse failure discovered during bounded look-ahead is held until every earlier valid block has decoded and emitted, preserving stream error order. Compressed jobs reserve the frame's declared maximum decoded block size; stored blocks borrow their source bytes and reserve no decoded allocation. Frames containing only stored blocks without block checksums bypass the worker pool because their only remaining work is ordered output and optional content hashing. Retained results remain charged at their allocation capacity rather than logical length, so highly compressible blocks cannot understate memory use. If fewer than two natural blocks fit the speculative budget, decoding proceeds incrementally on the coordinator instead of rejecting the frame. The pool is created lazily and reused across concatenated frames. Linked frames use the same parser, block decoder, output sink, progress, and report path, but decode serially with a rolling 64 KiB history. This is one code path with a scheduling branch, not separate Reader and writer implementations. -Tar format semantics use the mature `tar` crate, pinned from 0.4.46 and built without its optional xattr feature. It handles streaming GNU/PAX/long-name/link entries and confines extracted paths to the destination. A zero-capacity rendezvous channel transfers each owned decoder chunk and its live suffix offset to `tar::Archive`. The channel queues no chunks and applies backpressure. `tar::Archive` pulls data through `Read`, which copies once from the current chunk into its request buffer. Extraction writes immediately into a same-filesystem temporary directory, drains all trailing tar padding so codec validation completes, then preflights every destination conflict and moves entries into place with renames. Multiple inputs remain sequential so their per-codec worker pools cannot oversubscribe the global thread budget. +The LZ4 encoder emits standard independent-block frames with a content checksum. Its block size adapts from 4 MiB down to 64 KiB when the memory budget is small. Levels 1–6 use a latest-position table and LZ4-style adaptive skip; levels 7–9 use the shared bounded hash chain. Word-at-a-time common-prefix comparison is shared by both matchers. Each job chooses compressed or stored representation, and the coordinator emits completed blocks in order without retaining a complete frame. + +Tar format semantics use the mature `tar` crate, pinned from 0.4.46 and built without its optional xattr feature. Creation feeds `tar::Builder` directly into the selected streaming encoder, so the first tar bytes can be compressed immediately and there is no intermediate tar. Extraction uses a zero-capacity rendezvous channel to transfer each owned decoder chunk and its live suffix offset to `tar::Archive`; the channel queues no chunks and applies backpressure. Extraction writes immediately into a same-filesystem temporary directory, drains all trailing tar padding so codec validation completes, then preflights every destination conflict and moves entries into place with renames. Multiple archive inputs remain sequential so their per-codec worker pools cannot oversubscribe the global thread budget. ZIP structure semantics use `zip` 8.6.0 with all codec features disabled. The crate parses the central directory, Zip64 fields, data descriptors, names, modes, symlink kinds, and timestamp extra fields; fbz reads each raw stored/DEFLATE range and sends it through its own validated codec path. Because the crate intentionally collapses duplicate raw names into its index map, a small bounds-checked central-record count detects and rejects that ambiguity before using its metadata. Parsing also rejects encryption, unsupported methods, escaping or equivalent paths, non-directory ancestors, overlapping data ranges, and ranges crossing the central directory. An aggregate declared-size check runs before extraction, and a per-entry sink prevents output exceeding its declaration before size and CRC32 are checked. ZIP and tar share the same staging/preflight/rename implementation. ZIP scheduling deliberately uses one parallelism level at a time. A sole DEFLATE entry at least 16 MiB compressed, or an entry at least 64 MiB in a multi-entry archive, uses all requested workers inside the raw DEFLATE decoder. Remaining entries run serial inner decoders concurrently on one Rayon pool. This keeps thread ownership and memory behaviour obvious, avoids nested oversubscription, and lets many ordinary entries naturally absorb stragglers through work stealing. -The shared `pipeline.rs` scheduler provides ordered results, byte-budgeted admission, cancellation, and a staged priority queue. Bzip2 uses the rolling candidate path: workers reserve the maximum possible decoded block size, then shrink that reservation to actual retained output until ordered validation consumes or rejects it. Gzip uses the staged path: native workers alternate speculative DEFLATE decoding with higher-priority marker resolution, while the coordinator advances only the 32 KiB dependency windows and emits resolved chunks in order. LZ4 independent blocks use the simpler ordered-job path and charge each retained allocation against the budget. Decode results and outstanding resolution results have bounded horizons, preventing dependency stalls from causing unbounded memory. +ZIP creation follows the same policy over uncompressed sizes. One file of at least 16 MiB uses the segmented raw-DEFLATE engine. In multi-file archives, entries at least 64 MiB use that path sequentially, while ordinary entries are compressed concurrently with serial inner DEFLATE. A custom structural writer is smaller and more direct here than contorting the `zip` crate to accept externally segmented raw bitstreams; it writes descriptors, central records, Zip64 structures, Unix modes/symlinks, and extended timestamps while retaining only central-directory metadata. The `zip` crate remains the maintained parser for extraction. + +The shared `pipeline.rs` scheduler provides ordered results, byte-budgeted admission, cancellation, and a staged priority queue. Bzip2 uses the rolling candidate path: workers reserve the maximum possible decoded block size, then shrink that reservation to actual retained output until ordered validation consumes or rejects it. Gzip decoding uses the staged path for speculative DEFLATE and priority marker resolution. LZ4 decoding and multi-entry ZIP use ordinary ordered jobs. Compression's long-lived `StreamingOrdered` pool accepts work as bytes arrive, retains output in key order, catches worker panics, and cancels queued work when dropped. Gzip, LZ4, and bzip2 encoders all use it rather than maintaining codec-specific thread/channel machinery. Reservations conservatively cover owned input, temporary working state, and retained output until the coordinator consumes it. The shared `OutputSink` boundary accepts owned decoder chunks. Direct writer APIs adapt those chunks to `Write::write_all` without a channel or allocation copy. `Reader` and streaming tar extraction instead use the same zero-capacity owned-chunk pipe, so completed bzip2 blocks, resolved parallel-gzip segments, and decoded LZ4 blocks move into the consumer rather than being copied into an intermediate pipe buffer. The pipe adds at most the consumer's current chunk and the producer's next blocked chunk beyond the scheduler budget. Its receiver owns a cancellation flag; bzip2 scanning checks that flag between bounded waves, while LZ4 checks it before parsing each serial block or parallel batch. @@ -75,7 +91,9 @@ Run `cargo fmt --check` after Rust edits and `chkstyle` after Python edits once User-facing comparisons between installable CLIs are recorded only in [README Performance](README.md#performance). This file documents fixture reproduction, regression gates, and implementation-oriented diagnostics. -The normal release test path decodes selected valid and corrupt cases from the maintained upstream `bzip2-testfiles` collection. Generated byte distributions add differential coverage. Valid bzip2 outputs are compared byte-for-byte with `libbz2-rs-sys`. Gzip and raw-DEFLATE tests cover stored, fixed-Huffman, and dynamic-Huffman blocks; optional headers and FHCRC; concatenated members; exact end-of-stream boundaries; truncation; and trailer corruption across varied inputs and compression levels generated by `flate2`. LZ4 tests use `lz4_flex` to generate a matrix spanning empty, repetitive, byte-distribution, and pseudorandom inputs; all four standard block sizes; independent and linked blocks; and every checksum/content-size combination. The production decoders do not depend on any oracle. CLI tests additionally cover LZ4 extension/magic dispatch, reports, limits, corruption, and `.tar.lz4` streaming extraction alongside the existing bzip2, gzip, tar, and ZIP policy cases. +The normal release test path decodes selected valid and corrupt cases from the maintained upstream `bzip2-testfiles` collection. Generated byte distributions add differential coverage. Valid bzip2 outputs are compared byte-for-byte with `libbz2-rs-sys`. Gzip and raw-DEFLATE tests cover stored, fixed-Huffman, and dynamic-Huffman blocks; optional headers and FHCRC; concatenated members; exact end-of-stream boundaries; truncation; and trailer corruption across varied inputs and compression levels generated by `flate2`. LZ4 tests use `lz4_flex` to generate a matrix spanning empty, repetitive, byte-distribution, and pseudorandom inputs; all four standard block sizes; independent and linked blocks; and every checksum/content-size combination. The production decoders do not depend on any oracle. + +Encoder tests cover empty, repetitive, patterned, multiblock, and incremental-write inputs. fbz and an independent implementation both decode every generated stream: system bzip2 for bzip2, `flate2` for gzip, and `lz4_flex` for LZ4. ZIP output is validated and extracted with Info-ZIP, including the large-entry segmented-DEFLATE path; CLI tests cover `.tar.bz2`, `.tar.gz`, `.tar.lz4`, and ZIP creation/extraction composition. The unified Rust and Python APIs have separate integration tests. The normal release path contains warmed end-to-end performance regression gates capped at 1.3 times each oracle, allowing for noise on shared runners. The gzip gates independently exercise a highly compressible LZ77-heavy shape and an incompressible literal-heavy shape against `flate2`; the bzip2 gate uses `libbz2-rs-sys`. Representative local acceptance remains 1.2 times the corresponding oracle. The ignored full-wiki gzip test applies that threshold to rapidgzip-rust. Keep the whole release test suite below five seconds on the primary development laptop; individual timed workloads should normally be about 0.1 seconds or less. @@ -90,6 +108,48 @@ cargo test --release --test stream_cli_perf gzip_cli_comparison -- --ignored --e Regenerate both compressed inputs with the shared instructions in [Local Wikipedia benchmarks](#local-wikipedia-benchmarks). Run the tests separately so their parallel decoders do not compete for the same cores. User-facing results belong only in the README; this section records reproduction rather than a duplicate table. +### Local compression comparisons + +`tests/compression_cli_perf.rs` decodes `meta/simplewiki-first-5pct.xml.bz2` before timing and writes the resulting 84,423,012-byte payload into its temporary fixture. Standalone codecs write to a sink; archive tests write temporary files and report their sizes. Fixture creation and one warm-up per CLI are outside the measured interval, followed by exactly one child-process measurement. Run tests separately so encoders do not compete for cores: + +```bash +cargo test --release --test compression_cli_perf bzip2_compression_cli_comparison -- --ignored --exact --nocapture +cargo test --release --test compression_cli_perf gzip_compression_cli_comparison -- --ignored --exact --nocapture +cargo test --release --test compression_cli_perf lz4_compression_cli_comparison -- --ignored --exact --nocapture +cargo test --release --test compression_cli_perf tar_bzip2_compression_cli_comparison -- --ignored --exact --nocapture +cargo test --release --test compression_cli_perf tar_gzip_compression_cli_comparison -- --ignored --exact --nocapture +cargo test --release --test compression_cli_perf zip_single_compression_cli_comparison -- --ignored --exact --nocapture +cargo test --release --test compression_cli_perf zip_many_compression_cli_comparison -- --ignored --exact --nocapture +``` + +Regenerate the underlying payload using [Regenerating the fixtures](#regenerating-the-fixtures). Benchmark scheduling does not use hard-coded source or compressed lengths; the shared fixture loader asserts the independently established decoded acceptance length so truncation cannot look fast. The familiar CLI timing results belong only in README. The same single runs recorded peak RSS for implementation work: fbz used 436 MiB for bzip2, 161 MiB for gzip, 97 MiB for LZ4, 162 MiB for one-entry ZIP, and 332 MiB for 18-entry ZIP. Reference CLIs used 8 MiB or less except multithreaded `lz4` at 47 MiB. These are observed process peaks, not the configurable scheduler reservation limit. + +The bzip2 one-run-per-count diagnostic explains the 12-worker automatic cap: + +```bash +cargo test --release --test compression_cli_perf bzip2_compression_thread_sweep -- --ignored --exact --nocapture +``` + +| Workers | Compression | Peak RSS | +|---:|---:|---:| +| 1 | 4.80 s | 54 MiB | +| 2 | 2.54 s | 98 MiB | +| 4 | 1.49 s | 175 MiB | +| 6 | 1.20 s | 240 MiB | +| 8 | 959 ms | 338 MiB | +| 12 | 746 ms | 445 MiB | +| 18 | 695 ms | 581 MiB | + +ZIP's across-entry scheduler can be checked independently at 8, 12, and 18 workers: + +```bash +cargo test --release --test compression_cli_perf zip_compression_thread_sweep -- --ignored --exact --nocapture +``` + +A single local sweep measured 220 ms / 226 MiB at 12 workers and 150 ms / 336 MiB at 18. That clear throughput gain does not support an automatic ZIP worker cap; the shared memory budget still bounds scheduled entries, and explicit `-P` remains available when lower memory is preferable. + +`tests/compression_perf.rs` contains dev-only in-process comparisons against `flate2` and `lz4_flex`, plus an LZ4 worker sweep. They are useful for isolating codec work from CLI startup and I/O but stay out of README because they are not comparisons users can run as tools. + ### Local Reader benchmark `tests/reader_perf.rs` compares the public `Read` adapter with the direct writer path on 84.4 MiB SimpleWiki inputs. Each row is one release-mode run, and both timings include opening the file. The Reader path necessarily copies into the caller's buffer but transfers decoder-owned chunks into its rendezvous pipe without another copy. The LZ4 test builds its frame before timing: @@ -247,7 +307,7 @@ cargo test --release --test wiki_perf rapidgzip_rust_process_metrics -- --ignore cargo test --release --test wiki_perf system_gzip_process_metrics -- --ignored --exact --nocapture ``` -The metrics helper uses `wait4` and, on macOS, `proc_pid_rusage`'s `ri_phys_footprint` field on its own child; it needs no task-inspection permission. Treat its wall time as diagnostic and use `gzip_reference_ratio` for the speed acceptance ratio. +The metrics helper uses `wait4` and, on macOS, `proc_pid_rusage`'s `ri_phys_footprint` field on its own child; it needs no task-inspection permission. Wall time stops as soon as `wait4` reports child exit, before joining the 50 ms footprint sampler, so short-process timings do not include sampler shutdown latency. The 1,000-stream enwiki comparison has a separate ignored test for each implementation and mode so a changed decoder can be measured without rerunning unchanged baselines. Each test reads the compressed fixture before starting its single timed decode, with no warmups or repeats: ```bash diff --git a/README.md b/README.md index 118aea8..43c9ac2 100644 --- a/README.md +++ b/README.md @@ -1,14 +1,18 @@ # fbz -**Fast, reliable parallel decompression.** +**Fast, reliable parallel compression and decompression.** -`fbz` is one decompression CLI for bzip2, gzip, LZ4, ZIP, and compressed tar archives. It selects the format automatically, uses the available CPU cores, validates every stream, and safely extracts archives. The same engine is available as a Rust crate and Python module. +`fbz` is one CLI for compressing and decompressing bzip2, gzip, LZ4, ZIP, and compressed tar archives. It selects formats from filenames or magic, uses the available CPU cores, validates every decoded stream, and safely creates and extracts archives. The stream codecs are also available through Rust and Python APIs. `fbz` was created because we found existing tools tended to be too slow (as the benchmarks below show) or too unreliable (e.g `pbzip2` 1.1.13 fails to decompress the full English Wikipedia archive). And we wanted a single tool we could use for all common formats with a single CLI interface. ## Performance -The same 80.5 MiB SimpleWiki XML payload is used in every row with automatic thread selection. Stream formats are fully decoded and validated without writing output; archives are extracted. Each CLI is warmed once, then measured once on the primary Apple Silicon development machine. +The same 80.5 MiB SimpleWiki XML payload is used in every row with automatic thread selection. Each CLI is warmed once, then measured once on the primary Apple Silicon development machine. + +### Decompression + +Stream formats are fully decoded and validated without writing output; archives are extracted. | Format | `fbz` | Familiar tool | Speedup | |---|---:|---:|---:| @@ -19,6 +23,22 @@ The same 80.5 MiB SimpleWiki XML payload is used in every row with automatic thr | `.tar.gz` | 57 ms | `tar`: 118 ms | 2.1x | | `.lz4` | 55 ms | `lz4`: 56 ms | 1.0x | +### Compression + +The standalone rows write compressed bytes to a sink. Archive rows create a real archive; the ZIP rows also show that the scheduler handles one large file and many ordinary files without nested parallelism. + +| Format | `fbz` | Familiar tool | Relative speed | +|---|---:|---:|---:| +| `.bz2` | 736 ms | `bzip2`: 3.30 s | 4.5x | +| `.tar.bz2` | 729 ms | `tar`: 3.34 s | 4.6x | +| `.gz` | 147 ms | `gzip`: 1.40 s | 9.5x | +| `.tar.gz` | 144 ms | `tar`: 1.44 s | 10x | +| `.zip` (1 file) | 142 ms | `zip`: 1.47 s | 10x | +| `.zip` (18 files) | 138 ms | `zip`: 1.47 s | 11x | +| `.lz4` | 35 ms | `lz4`: 29 ms | 0.83x | + +The fbz LZ4 output was about 7% smaller than the reference output on this payload. Other compressed sizes were within 1% of their familiar tools. + See [Benchmarking details](#benchmarking-details) for the details. ## Install @@ -49,13 +69,25 @@ fbz events.json.lz4 # write events.json fbz source.tar.gz # extract into the current directory fbz source.tar.lz4 -C unpacked # stream-decode and extract fbz source.tbz2 -C unpacked # extract into unpacked/ -fbz dataset.zip -C unpacked # extract ZIP entries adaptively in parallel +fbz dataset.zip -C unpacked # extract ZIP entries adaptively in parallel fbz --extract -C unpacked - # extract tar or ZIP data from stdin fbz source.tgz -o source.tar # decode without extracting fbz dump.xml.bz2 -o result.xml # choose the decoded output path fbz dump.xml.bz2 -o - # write decoded bytes to stdout ``` +`-z/--compress` reverses the operation. An output suffix selects the format; when `-o` is omitted, `--format` selects it and fbz appends the conventional suffix. Standalone streams accept stdin and can write stdout. Tar and ZIP creation accept multiple filesystem inputs and stream output without an intermediate tar or plaintext file. + +```bash +fbz -z --format bzip2 dump.xml # write dump.xml.bz2 +fbz -z events.json -o events.json.gz # infer gzip from the output +fbz -z data -o data.lz4 # independent-block LZ4 frame +fbz -z src docs -o source.tar.gz # stream tar directly into gzip +fbz -z src docs -o source.tar.bz2 # stream tar directly into bzip2 +fbz -z src docs -o source.zip # adaptive parallel ZIP creation +fbz -z --format gzip -o - < events # write one gzip member to stdout +``` + Multiple inputs are processed in order, with parallelism applied inside each compressed stream. `-C/--output-dir` collects decoded files and is the extraction root for archives: ```bash @@ -80,7 +112,17 @@ fbz --list --json dump.xml.bz2 # emit the complete layout as JSON ## Python -The Python API currently exposes the bzip2 backend. Unified Python dispatch will follow the CLI workbench rather than being designed ahead of it. +The Python API exposes one-shot compression for all three stream formats. Decompression, validation, scanning, and indexed seeking currently expose the bzip2 backend. + +### One-shot compression + +```python +import fbz + +compressed = fbz.compress(plain_bytes, "gzip", level=6) +``` + +The format is `"bzip2"`, `"gzip"`, or `"lz4"`. `threads=0` selects automatically, `memory_limit` bounds scheduled work, and the format-specific default level is used when `level` is omitted. ### One-shot decompression and validation @@ -141,6 +183,23 @@ fn main() -> fbz::Result<()> { `decompress` returns a `Vec`. `decode_to_writer` returns a validated `Index` while streaming output, `build_index` validates into a sink, and their `*_with_progress` variants report completed compressed and decoded byte counts. `IndexedReader` implements `Read` and `Seek`; it can build an index itself or load a persisted one with `open_with_index`. +The unified compression API covers bzip2, gzip, and LZ4 and writes incrementally: + +```rust +use fbz::{EncodeFormat, EncodeOptions, compress_to_writer}; + +fn main() -> fbz::Result<()> { + let mut input = std::fs::File::open("events.json")?; + let mut output = std::fs::File::create("events.json.gz")?; + compress_to_writer(&mut input, &mut output, EncodeFormat::Gzip, EncodeOptions::default())?; + Ok(()) +} +``` + +`compress` returns a `Vec`, while `Encoder` implements `Write` for producers that generate data incrementally. `EncodeOptions` controls worker count, memory budget, and compression level. Gzip produces one standard member, LZ4 produces a standard independent-block frame, and bzip2 produces an ordinary `BZh1`–`BZh9` stream; none requires an fbz decoder. + +`zip::create_to_writer` creates stored/DEFLATE ZIP and Zip64 archives from `zip::PathInput` values. For tar composition, feed a `gzip::Encoder`, `lz4::Encoder`, or `Bzip2Encoder` directly to a streaming `tar::Builder`, which is the same composition used by the CLI. + The in-repo gzip decoder is available separately so callers can choose explicitly: ```rust @@ -182,22 +241,22 @@ Checksum errors discovered after output has begun are returned by a later `read( ## CLI output safety -- Existing decoded files and archive entries are rejected by default. `--force` replaces them; `--skip-existing` applies to decoded files rather than archives. +- Existing generated files and archive entries are rejected by default. `--force` replaces them; `--skip-existing` applies to standalone outputs and newly created archives rather than extraction into an existing tree. - Decoded-file outputs use a same-directory temporary file and become visible atomically only after successful checksum validation. - Tar entries stream into a same-filesystem staging directory through a bounded pipe. ZIP entries decode directly into the same staging scheme. Entries are preflighted and moved into the destination only after every relevant compression stream and archive structure validates, so a late CRC failure leaves no extracted files. - Tar and ZIP paths and link targets are confined to the destination. ZIP rejects unsafe or duplicate paths; tar safely skips unsafe entries. New entries use the archive's permissions and modification times where provided. Standalone decoded files inherit those values from the compressed input. -- `--rm` removes each compressed input only after its decoded file or all archive entries have been committed successfully. +- `--rm` removes an input only after its compressed or decoded output has been committed successfully. Archive creation deliberately does not remove its source tree. - `--max-output SIZE` limits decoded bytes per input, including tar framing and padding. Sizes accept binary suffixes such as `K`, `MiB`, and `G`. -Long interactive operations report completion, decoded throughput, compression ratio, and ETA on stderr. Progress is disabled automatically when stderr is redirected; `-q/--quiet` also suppresses progress and skip notices. +Long interactive standalone operations report completion, throughput, compression ratio, and ETA on stderr. Progress is disabled automatically when stderr is redirected; `-q/--quiet` also suppresses progress and skip notices. -`-P/--threads 0`, the default, uses the machine's available parallelism; an explicit positive value is honoured by every decoder. `--memory-limit` bounds speculative output in the shared scheduler and defaults to `1G`. Gzip uses parallel dynamic-block discovery only when the input and memory budget can amortize it. LZ4 frames with independent blocks decode those blocks concurrently and commit them in order; automatic LZ4 decoding caps this memory-bandwidth-bound work at four workers, while explicit `-P` values remain unchanged. Linked-block frames decode serially because each block depends on the preceding 64 KiB history. ZIP uses one level of parallelism at a time: large entries use the parallel DEFLATE engine, while archives of ordinary entries decode files concurrently without nested worker pools. +`-P/--threads 0`, the default, selects parallelism automatically; an explicit positive value is honoured by every codec. `--memory-limit` is the byte budget for in-flight scheduler reservations and defaults to `1G`; it bounds queued input, working state, and retained results rather than promising an exact process-RSS ceiling. Automatic bzip2 compression stops at 12 workers because its BWT working sets reach a clear throughput plateau there. Automatic LZ4 decoding stops at four workers because it is memory-bandwidth bound; explicit `-P` values remain unchanged. Gzip and standalone LZ4 compression divide one standard stream into independently encoded ordered segments or blocks. ZIP uses one level of parallelism at a time: large entries use the parallel DEFLATE engine, while archives of ordinary entries process files concurrently without nested worker pools. ## Benchmarking details -The headline benchmark uses the first 84,423,012 decoded bytes of SimpleWiki for every format. The standalone bzip2 and gzip rows run each tool's validation mode. The ZIP row extracts 18 equal-sized files, while the compressed-tar rows extract one file; each archive comparison performs the same work on both sides. LZ4 uses `fbz`'s automatic four-worker limit. All results are single local release-mode observations after one untimed warm-up, not statistical aggregates. +The headline benchmarks use the first 84,423,012 decoded bytes of SimpleWiki for every format. In the decompression table, standalone bzip2 and gzip use each tool's validation mode; ZIP extracts 18 equal files, while compressed tar extracts one. In the compression table, standalone codecs write to a sink, tar wraps one file, and ZIP creates either one large entry or 18 equal entries. Each comparison performs the same work on both sides. All results are single local release-mode observations after one untimed warm-up, not statistical aggregates. -The familiar reference CLIs are the system `bzip2`, `gzip`, and `tar`; Apple Info-ZIP `unzip` 6.00; and Homebrew `lz4` 1.10.0. The detailed fixture-generation and single-run commands live in [DEV.md](DEV.md), alongside in-process codec comparisons and separate memory diagnostics. +The familiar reference CLIs are the system `bzip2`, `gzip`, and `tar`; Apple Info-ZIP `zip`/`unzip` 3.0/6.00; and Homebrew `lz4` 1.10.0. The detailed fixture-generation and single-run commands live in [DEV.md](DEV.md), alongside in-process codec comparisons and separate memory diagnostics. RSS is deliberately omitted from the headline tables: it is bounded and configurable, but the parallel encoders trade memory for throughput and the exact figures are implementation diagnostics rather than user-visible work. On the complete 1.57 GiB SimpleWiki XML recompressed with system `gzip -6`, `fbz` validated the stream in 0.33 seconds, compared with 0.36 seconds for a local `rapidgzip-rust` checkout and 1.37 seconds for Apple `gzip`. This larger result is kept here because it exercises sustained parallel gzip decoding; it is not mixed into the common-payload headline table. @@ -209,7 +268,13 @@ Homebrew `pbzip2` 1.1.13 could not safely decompress the complete 26,668,484,995 The production codec logic is portable Rust. The bzip2 decoder uses a tuned 4096-entry Huffman lookup table for codes up to 12 bits and canonical fallback for longer codes. A structural scan finds possible non-byte-aligned block markers; these remain speculative until ordered decoding establishes the exact stream chain and validates all block and combined-stream CRCs. A rolling scheduler keeps workers busy across concatenated streams while bounding decoded results awaiting validation. -The gzip backend implements RFC 1952 framing and DEFLATE directly in this repository. For sufficiently large dynamic-Huffman inputs it discovers independently decodable boundaries, decodes speculative chunks through the shared byte-budgeted scheduler, and represents unknown predecessor bytes as compact markers. The ordered coordinator resolves only the suffix needed to derive the next 32 KiB history window; full marker resolution and per-chunk CRC run as priority work on the same staged worker queue, and CRCs are combined in order. Once a chunk has a marker-free window, the same decoder switches its remaining output from `u16` markers to ordinary bytes. Small, stored-heavy, fixed-heavy, one-thread, and low-memory inputs use the serial path; concatenated members may independently choose either path. FHCRC, CRC32, and ISIZE are always validated. LZ4 framing and block decoding are likewise implemented in safe Rust. It parses one frame header at a time and schedules independent blocks in bounded batches, so output can begin without a whole-frame layout pass. Independent blocks use the same ordered, byte-budgeted scheduler as bzip2; linked blocks retain only the preceding 64 KiB window. LZ4 and DEFLATE share one optimized overlapping back-reference expansion primitive. Header, block, and content XXH32 checksums are validated where present. ZIP reuses the raw DEFLATE core and uses the mature `zip` crate only for container structure and metadata. It supports stored and DEFLATE entries, Zip64, streaming data descriptors, Unix symlinks/modes, and Unix/NTFS modification-time fields; encryption and uncommon legacy compression methods are intentionally unsupported. `crc32fast` and `twox-hash` are the production checksum helpers; `flate2` and `lz4_flex` are dev-only differential oracles. +The gzip backend implements RFC 1952 framing and DEFLATE directly in this repository. For sufficiently large dynamic-Huffman inputs its decoder discovers independently decodable boundaries, represents unknown predecessor bytes as compact markers, and resolves only the suffix needed for the next 32 KiB history window. Its encoder schedules 1 MiB raw-DEFLATE segments with the preceding 32 KiB dictionary, joins their byte-aligned boundaries in order, and writes one ordinary gzip member and trailer. The same raw-DEFLATE encoder creates ZIP entries. Fixed, dynamic, and stored blocks are selected by encoded size. + +LZ4 framing and blocks are likewise implemented in safe Rust. Decoding schedules independent blocks in bounded batches and retains only 64 KiB for linked history. Compression emits independent blocks so they can be encoded and later decoded in parallel; compressible blocks use a fast latest-match table, with the shared hash-chain matcher available at higher levels, and incompressible blocks are stored. Header, block, and content XXH32 checksums are handled where present. + +The bzip2 encoder splits ordinary `BZh1`–`BZh9` streams at their natural block boundaries. RLE1 runs are formed incrementally, BWT/MTF/RLE2/Huffman work runs independently per block, and exact bit strings plus combined CRCs are committed in order. The BWT uses a safe SA-IS suffix array. Decoder and encoder share the bzip2 CRC implementation. + +ZIP extraction uses the mature `zip` crate with codec features disabled for container structure and metadata, then feeds raw entry ranges through fbz's decoder. ZIP creation writes the small amount of required structure directly so externally produced raw-DEFLATE segments can stream without being copied through a second codec. It supports stored and DEFLATE entries, Zip64, data descriptors, Unix symlinks/modes, and extended timestamps. Tar creation and extraction use the mature `tar` crate as a streaming structural layer. Encryption and uncommon legacy ZIP methods are intentionally unsupported. `crc32fast` and `twox-hash` are the production checksum helpers; `libbz2-rs-sys`, `flate2`, and `lz4_flex` are dev-only differential oracles. Legacy randomized blocks generated by bzip2 releases before 0.9.5 are intentionally unsupported. Normal `BZh1` through `BZh9` streams and concatenated streams are supported. @@ -227,10 +292,11 @@ The open-source implementations and codebases consulted were: - [`LZ4`](https://github.com/lz4/lz4), the reference format and Homebrew CLI performance baseline. - [`lz4_flex`](https://github.com/pseitz/lz4_flex), used dev-only to generate a broad interoperability matrix and benchmark frames. - [`lz4-rs`](https://github.com/bozaro/lz4-rs), consulted as a second local implementation reference. +- [`crabz2`](https://github.com/jwmurray/crabz2), whose MIT-licensed BWT, MTF/RLE2, and grouped-Huffman encoder machinery was adapted for fbz's block-parallel bzip2 compressor. - Rob Landley's 0BSD [`bzcat` implementation in Toybox](https://github.com/landley/toybox), from which fbz's specialised bzip2 decoder is derived. ## Development [DEV.md](DEV.md) documents the architecture, test strategy, benchmark fixture generation, build commands, and release process. -`fbz` is licensed under the [Apache License 2.0](LICENSE). +`fbz` is licensed under the [Apache License 2.0](LICENSE). The adapted crabz2 encoder files retain their bundled MIT license and attribution. diff --git a/pyproject.toml b/pyproject.toml index 0f03612..2bccb3a 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -6,7 +6,7 @@ backend-path = ["."] [project] name = "fbz" dynamic = ["version"] -description = "Compression-format research workbench with fast bzip2, gzip, LZ4, and ZIP decompression" +description = "Fast parallel compression and decompression for bzip2, gzip, LZ4, tar, and ZIP" license = {text = "Apache-2.0"} requires-python = ">=3.10" readme = "README.md" diff --git a/python/fbz/__init__.py b/python/fbz/__init__.py index e8a12df..d4bbf5c 100644 --- a/python/fbz/__init__.py +++ b/python/fbz/__init__.py @@ -2,7 +2,7 @@ import io, os from pathlib import Path -from ._core import BadBzip2File, _IndexedReader, __version__, _build_index, _decompress, _scan, _test, bz2_crc32 +from ._core import BadBzip2File, _IndexedReader, __version__, _build_index, _compress, _decompress, _scan, _test, bz2_crc32 DEFAULT_MEMORY_LIMIT = 1024 * 1024 * 1024 DEFAULT_CACHE_LIMIT = 64 * 1024 * 1024 @@ -28,6 +28,10 @@ def decompress(data: bytes, *, threads=0, memory_limit=DEFAULT_MEMORY_LIMIT) -> "Decompress and fully CRC-validate one or more concatenated bzip2 streams." return _decompress(data, threads, memory_limit) +def compress(data: bytes, format: str, *, threads=0, memory_limit=DEFAULT_MEMORY_LIMIT, level=None) -> bytes: + "Compress *data* as bzip2, gzip, or LZ4." + return _compress(data, format, threads, memory_limit, level) + class IndexedBzip2File(io.RawIOBase): "Seekable binary reader backed by a validated bzip2 block index." @@ -96,5 +100,5 @@ def test(source, *, threads=0, memory_limit=DEFAULT_MEMORY_LIMIT): __all__ = [ "__version__", "BadBzip2File", "BlockCandidate", "DEFAULT_CACHE_LIMIT", "DEFAULT_MEMORY_LIMIT", "EndCandidate", - "IndexedBzip2File", "ScanResult", "StreamHeaderCandidate", "build_index", "bz2_crc32", "decompress", "open", "scan", "test" + "IndexedBzip2File", "ScanResult", "StreamHeaderCandidate", "build_index", "bz2_crc32", "compress", "decompress", "open", "scan", "test" ] diff --git a/src/bin/fbz.rs b/src/bin/fbz.rs index a8f908e..d9d7cdc 100644 --- a/src/bin/fbz.rs +++ b/src/bin/fbz.rs @@ -1,5 +1,9 @@ +#[path = "fbz/archive_create.rs"] +mod archive_create; #[path = "fbz/archive_extract.rs"] mod archive_extract; +#[path = "fbz/tar_create.rs"] +mod tar_create; #[path = "fbz/tar_extract.rs"] mod tar_extract; #[path = "fbz/zip_extract.rs"] @@ -13,10 +17,10 @@ use std::{ time::{Duration, Instant}, }; -use clap::{ArgGroup, Parser}; +use clap::{ArgGroup, Parser, ValueEnum}; use fbz::{ - DecodeOptions, DecodeProgress, Error, Format, Index, OutputSink, Source, WriterSink, build_index_with_progress, decode_stream_to_sink_with_progress, - decode_to_writer_with_progress, gzip, lz4, + DecodeOptions, DecodeProgress, EncodeOptions, EncodeProgress, Error, Format, Index, OutputSink, Source, WriterSink, build_index_with_progress, + decode_stream_to_sink_with_progress, decode_to_writer_with_progress, gzip, lz4, }; use serde_json::{Value, json}; use tempfile::NamedTempFile; @@ -24,8 +28,8 @@ use tempfile::NamedTempFile; #[derive(Parser)] #[command( version, - about = "Parallel bzip2, gzip, LZ4, and ZIP decompression with safe archive extraction", - long_about = "Parallel bzip2, gzip, and LZ4 frame decompression, streaming tar extraction, and adaptive parallel ZIP extraction. Decoding is the default operation. Recognised codec suffixes are removed from normal output names. ZIP and compressed tar archives extract automatically unless -o is given; -o is not valid for ZIP. Independent LZ4 blocks decode in parallel; linked blocks decode serially.", + about = "Fast parallel compression and decompression with safe archive handling", + long_about = "Parallel compression and decompression for bzip2, gzip, and LZ4, plus streaming compressed-tar and adaptive ZIP creation/extraction. Decoding is the default operation; -z enables compression. Compression format is inferred from the output suffix or selected with --format.", after_help = r#"Examples: fbz dump.xml.bz2 Write dump.xml fbz events.json.gz -o - Write decoded bytes to stdout @@ -33,12 +37,17 @@ use tempfile::NamedTempFile; fbz backup.tgz -C restored Extract into restored/ fbz backup.tar.lz4 -C restored Extract a tar-wrapped LZ4 frame fbz dataset.zip -C restored Extract ZIP entries in parallel + fbz -z data -o data.bz2 Compress as parallel bzip2 + fbz -z data -o data.gz Compress as one standard gzip member + fbz -z data -o data.lz4 Compress as independent LZ4 blocks + fbz -z src -o src.tar.gz Stream tar directly into gzip + fbz -z src -o src.zip Create ZIP with adaptive parallelism fbz --test archive.tar.bz2 Validate without writing output fbz --list --json data.gz Show the validated layout as JSON"#, - group(ArgGroup::new("mode").args(["test", "index", "list", "extract"])) + group(ArgGroup::new("mode").args(["test", "index", "list", "extract", "compress"])) )] struct Cli { - /// Input files, or - for stdin except with --index or --list. + /// Input files, or - where the selected operation accepts a byte stream. #[arg(required = true, num_args = 1..)] inputs: Vec, /// Fully decode and validate checksums without writing output. @@ -53,16 +62,25 @@ struct Cli { /// Extract a tar or ZIP archive; automatic for recognised archive suffixes. #[arg(short = 'x', long)] extract: bool, - /// Write decoded bytes to PATH, or - for stdout; requires one input and disables automatic extraction. + /// Compress rather than decompress. + #[arg(short = 'z', long)] + compress: bool, + /// Compression format; inferred from -o when possible. + #[arg(long, value_enum, requires = "compress")] + format: Option, + /// Compression level from 1 (fastest) through 9 (smallest); format-specific default. + #[arg(short = 'l', long, requires = "compress")] + level: Option, + /// Write to PATH or stdout; tar and ZIP creation accept multiple inputs. #[arg(short, long, conflicts_with_all = ["test", "list", "extract", "output_dir"])] output: Option, - /// Put decoded files or extracted archive entries in DIRECTORY. + /// Put generated files or extracted archive entries in DIRECTORY. #[arg(short = 'C', long = "output-dir", conflicts_with_all = ["test", "index", "list", "output"])] output_dir: Option, - /// Decoder worker threads; 0 uses all available CPUs. + /// Compression or decompression workers; 0 selects automatically. #[arg(short = 'P', long, default_value_t = 0)] threads: usize, - /// Maximum speculative decoder output; accepts binary size suffixes. + /// Maximum in-flight codec memory budget; accepts binary size suffixes. #[arg(long, default_value = "1G", value_parser = parse_size)] memory_limit: usize, /// Maximum decoded bytes per input; accepts binary size suffixes. @@ -71,10 +89,10 @@ struct Cli { /// Replace existing output files or archive entries. #[arg(short, long, conflicts_with_all = ["test", "list", "skip_existing"])] force: bool, - /// Skip existing decoded output files. + /// Skip existing output files; unavailable during extraction. #[arg(long, conflicts_with_all = ["test", "list", "extract", "force"])] skip_existing: bool, - /// Remove compressed inputs after successful decoding or extraction. + /// Remove inputs after standalone compression, decoding, or extraction; not archive creation. #[arg(long = "rm", conflicts_with_all = ["test", "index", "list"])] remove_input: bool, /// Suppress progress and skip notices. @@ -85,6 +103,17 @@ struct Cli { json: bool, } +#[derive(Clone, Copy, Debug, PartialEq, Eq, ValueEnum)] +enum CompressionFormat { + Bzip2, + Gzip, + Lz4, + TarBzip2, + TarGzip, + TarLz4, + Zip, +} + fn main() -> ExitCode { match run(Cli::parse()) { Ok(()) => ExitCode::SUCCESS, @@ -97,6 +126,9 @@ fn main() -> ExitCode { fn run(cli: Cli) -> fbz::Result<()> { validate_cli(&cli)?; + if cli.compress { + return run_compress(&cli); + } let options = DecodeOptions { threads: cli.threads, memory_limit: cli.memory_limit }; if cli.test { for input in &cli.inputs { @@ -114,10 +146,13 @@ fn run(cli: Cli) -> fbz::Result<()> { } fn validate_cli(cli: &Cli) -> fbz::Result<()> { - if cli.output.is_some() && cli.inputs.len() != 1 { + let selected_format = if cli.compress { Some(compression_format(cli)?) } else { None }; + let archive_compression = + matches!(selected_format, Some(CompressionFormat::TarBzip2 | CompressionFormat::TarGzip | CompressionFormat::TarLz4 | CompressionFormat::Zip)); + if cli.output.is_some() && cli.inputs.len() != 1 && !archive_compression { return Err(invalid("--output requires exactly one input")); } - if cli.output.is_some() && cli.inputs.iter().any(|input| is_zip_archive(input)) { + if !cli.compress && cli.output.is_some() && cli.inputs.iter().any(|input| is_zip_archive(input)) { return Err(invalid("--output is not supported for ZIP archives")); } if cli.inputs.iter().any(|input| input == "-") && cli.inputs.len() != 1 { @@ -126,12 +161,171 @@ fn validate_cli(cli: &Cli) -> fbz::Result<()> { if (cli.index || cli.list) && cli.inputs.iter().any(|input| input == "-") { return Err(invalid("stdin is supported only for decoding and --test")); } - if cli.skip_existing && cli.inputs.iter().any(|input| should_extract(cli, input)) { + if !cli.compress && cli.skip_existing && cli.inputs.iter().any(|input| should_extract(cli, input)) { return Err(invalid("--skip-existing is not supported when extracting archives")); } + if cli.compress && cli.max_output.is_some() { + return Err(invalid("--max-output applies only to decompression")); + } + if archive_compression && cli.remove_input { + return Err(invalid("--rm is not supported when creating archives")); + } Ok(()) } +fn inferred_compression_format(path: &Path) -> Option { + let name = path.to_string_lossy().to_ascii_lowercase(); + if name.ends_with(".tar.bz2") || name.ends_with(".tar.bzip2") || name.ends_with(".tbz") || name.ends_with(".tbz2") { + Some(CompressionFormat::TarBzip2) + } else if name.ends_with(".tar.lz4") { + Some(CompressionFormat::TarLz4) + } else if name.ends_with(".tar.gz") || name.ends_with(".tar.gzip") || name.ends_with(".tgz") { + Some(CompressionFormat::TarGzip) + } else if name.ends_with(".zip") { + Some(CompressionFormat::Zip) + } else if name.ends_with(".gz") || name.ends_with(".gzip") { + Some(CompressionFormat::Gzip) + } else if name.ends_with(".bz2") || name.ends_with(".bzip2") { + Some(CompressionFormat::Bzip2) + } else if name.ends_with(".lz4") { + Some(CompressionFormat::Lz4) + } else { + None + } +} + +fn compression_format(cli: &Cli) -> fbz::Result { + let inferred = cli.output.as_deref().filter(|path| *path != Path::new("-")).and_then(inferred_compression_format); + match (cli.format, inferred) { + (Some(explicit), Some(inferred)) if explicit != inferred => Err(invalid("--format conflicts with the --output filename")), + (Some(explicit), _) | (None, Some(explicit)) => Ok(explicit), + (None, None) => Err(invalid("compression format must be selected by --format or the --output filename")), + } +} + +fn compressed_output(input: &Path, format: CompressionFormat, directory: Option<&Path>) -> PathBuf { + let suffix = match format { + CompressionFormat::Bzip2 => ".bz2", + CompressionFormat::Gzip => ".gz", + CompressionFormat::Lz4 => ".lz4", + CompressionFormat::TarBzip2 => ".tar.bz2", + CompressionFormat::TarGzip => ".tar.gz", + CompressionFormat::TarLz4 => ".tar.lz4", + CompressionFormat::Zip => ".zip", + }; + let name = input.file_name().unwrap_or(input.as_os_str()); + let output = PathBuf::from(format!("{}{suffix}", name.to_string_lossy())); + directory.map_or_else(|| PathBuf::from(format!("{}{suffix}", input.display())), |directory| directory.join(output)) +} + +fn run_compress(cli: &Cli) -> fbz::Result<()> { + let format = compression_format(cli)?; + let options = EncodeOptions { threads: cli.threads, memory_limit: cli.memory_limit, level: cli.level }; + if let Some(directory) = &cli.output_dir { + fs::create_dir_all(directory)?; + } + match format { + CompressionFormat::TarBzip2 | CompressionFormat::TarGzip | CompressionFormat::TarLz4 => compress_tar(cli, format, options), + CompressionFormat::Zip => compress_zip(cli, options), + CompressionFormat::Bzip2 | CompressionFormat::Gzip | CompressionFormat::Lz4 => compress_streams(cli, format, options), + } +} + +fn archive_output(cli: &Cli, format: CompressionFormat, kind: &str) -> fbz::Result { + cli.output + .clone() + .or_else(|| (cli.inputs.len() == 1 && cli.inputs[0] != "-").then(|| compressed_output(Path::new(&cli.inputs[0]), format, cli.output_dir.as_deref()))) + .ok_or_else(|| invalid(format!("{kind} compression with multiple inputs requires --output"))) +} + +fn write_archive_output(cli: &Cli, output: &Path, write: impl FnOnce(&mut dyn Write) -> fbz::Result<()>) -> fbz::Result<()> { + if output == Path::new("-") { + return write(&mut io::stdout().lock()); + } + if should_skip(output, cli.skip_existing, cli.quiet) { + return Ok(()); + } + atomic_write(output, cli.force, |writer| write(writer)) +} + +fn compress_tar(cli: &Cli, format: CompressionFormat, options: EncodeOptions) -> fbz::Result<()> { + let output = archive_output(cli, format, "tar")?; + write_archive_output(cli, &output, |writer| match format { + CompressionFormat::TarBzip2 => tar_create::pack_bzip2(&cli.inputs, writer, options).map(|_| ()), + CompressionFormat::TarGzip => tar_create::pack_gzip(&cli.inputs, writer, options).map(|_| ()), + CompressionFormat::TarLz4 => tar_create::pack_lz4(&cli.inputs, writer, options).map(|_| ()), + _ => unreachable!(), + }) +} + +fn compress_zip(cli: &Cli, options: EncodeOptions) -> fbz::Result<()> { + if cli.inputs.iter().any(|input| input == "-") { + return Err(invalid("stdin cannot be used as a ZIP archive entry")); + } + let output = archive_output(cli, CompressionFormat::Zip, "ZIP")?; + let inputs = cli + .inputs + .iter() + .map(|input| { + let source = PathBuf::from(input); + Ok(fbz::zip::PathInput { archive_path: archive_create::archive_name(&source)?, source }) + }) + .collect::>>()?; + write_archive_output(cli, &output, |writer| fbz::zip::create_to_writer(&inputs, writer, options).map(|_| ())) +} + +fn compress_streams(cli: &Cli, format: CompressionFormat, options: EncodeOptions) -> fbz::Result<()> { + for input in &cli.inputs { + let output = cli + .output + .clone() + .unwrap_or_else(|| if input == "-" { PathBuf::from("-") } else { compressed_output(Path::new(input), format, cli.output_dir.as_deref()) }); + if output == Path::new("-") { + let mut source: Box = if input == "-" { Box::new(io::stdin().lock()) } else { Box::new(fs::File::open(input)?) }; + let total = if input == "-" { 0 } else { fs::metadata(input)?.len() }; + compress_stream(format, &mut source, &mut io::stdout().lock(), options, input, total, cli.quiet)?; + if cli.remove_input && input != "-" { + fs::remove_file(input)?; + } + } else { + if should_skip(&output, cli.skip_existing, cli.quiet) { + continue; + } + let input_path = Path::new(input); + if input_path == output { + return Err(invalid(format!("input and output are both {}", input_path.display()))); + } + let mut source = fs::File::open(input_path)?; + let total = source.metadata()?.len(); + atomic_write(&output, cli.force, |writer| compress_stream(format, &mut source, writer, options, input, total, cli.quiet))?; + preserve_metadata(input_path, &output)?; + if cli.remove_input { + fs::remove_file(input_path)?; + } + } + } + Ok(()) +} + +fn compress_stream( + format: CompressionFormat, + input: &mut impl Read, + output: &mut impl Write, + options: EncodeOptions, + label: &str, + total: u64, + quiet: bool, +) -> fbz::Result<()> { + let format = match format { + CompressionFormat::Bzip2 => fbz::EncodeFormat::Bzip2, + CompressionFormat::Gzip => fbz::EncodeFormat::Gzip, + CompressionFormat::Lz4 => fbz::EncodeFormat::Lz4, + _ => return Err(invalid("selected format is not a standalone stream codec")), + }; + let mut display = ProgressDisplay::new(label, total, quiet || total == 0); + fbz::compress_to_writer_with_progress(input, output, format, options, |progress| display.update_encode(progress)).map(|_| ()) +} + fn should_extract(cli: &Cli, input: &str) -> bool { cli.extract || (cli.output.is_none() && is_archive(input)) } @@ -416,27 +610,37 @@ impl ProgressDisplay { } fn update(&mut self, progress: DecodeProgress) { + let compressed = progress.compressed_bytes.min(self.total); + let ratio = if compressed == 0 { 0.0 } else { progress.decoded_bytes as f64 / compressed as f64 }; + self.render(compressed, compressed, progress.decoded_bytes, progress.decoded_bytes, ratio); + } + + fn update_encode(&mut self, progress: EncodeProgress) { + let input = progress.input_bytes.min(self.total); + let ratio = if progress.output_bytes == 0 { 0.0 } else { progress.input_bytes as f64 / progress.output_bytes as f64 }; + self.render(input, progress.input_bytes, progress.output_bytes, progress.input_bytes, ratio); + } + + fn render(&mut self, completed: u64, left: u64, right: u64, throughput: u64, ratio: f64) { if !self.enabled { return; } let elapsed = self.started.elapsed(); - let finished = progress.compressed_bytes >= self.total; + let finished = completed >= self.total; if (!self.drawn && elapsed < Duration::from_millis(200)) || (!finished && self.last_draw.elapsed() < Duration::from_millis(100)) { return; } - let compressed = progress.compressed_bytes.min(self.total); - let percent = if self.total == 0 { 100.0 } else { compressed as f64 * 100.0 / self.total as f64 }; + let percent = if self.total == 0 { 100.0 } else { completed as f64 * 100.0 / self.total as f64 }; let seconds = elapsed.as_secs_f64().max(0.001); - let rate = progress.decoded_bytes as f64 / seconds; - let ratio = if compressed == 0 { 0.0 } else { progress.decoded_bytes as f64 / compressed as f64 }; - let eta = if compressed == 0 || finished { 0.0 } else { seconds * (self.total - compressed) as f64 / compressed as f64 }; + let rate = throughput as f64 / seconds; + let eta = if completed == 0 || finished { 0.0 } else { seconds * (self.total - completed) as f64 / completed as f64 }; let _ = write!( self.stderr, "\r\x1b[2K{}: {:5.1}% {} → {} {}/s {:.1}× ETA {}", self.label, percent, - format_bytes(compressed), - format_bytes(progress.decoded_bytes), + format_bytes(left), + format_bytes(right), format_bytes(rate as u64), ratio, format_duration(eta), diff --git a/src/bin/fbz/archive_create.rs b/src/bin/fbz/archive_create.rs new file mode 100644 index 0000000..d15320c --- /dev/null +++ b/src/bin/fbz/archive_create.rs @@ -0,0 +1,34 @@ +use std::{ + env, + path::{Component, Path, PathBuf}, +}; + +use fbz::{Error, Result}; + +fn invalid(message: impl Into) -> Error { + Error::InvalidConfiguration(message.into()) +} + +pub(super) fn archive_name(path: &Path) -> Result { + let relative = if path.is_absolute() { + let current = env::current_dir()?; + path.strip_prefix(¤t).ok().map(Path::to_path_buf).or_else(|| path.file_name().map(PathBuf::from)) + } else { + Some(path.to_path_buf()) + } + .ok_or_else(|| invalid(format!("cannot derive an archive name for {}", path.display())))?; + let mut clean = PathBuf::new(); + for component in relative.components() { + match component { + Component::Normal(name) => clean.push(name), + Component::CurDir => {} + Component::ParentDir | Component::RootDir | Component::Prefix(_) => { + return Err(invalid(format!("refusing unsafe archive input name {}", relative.display()))); + } + } + } + if clean.as_os_str().is_empty() { + return Err(invalid(format!("cannot derive an archive name for {}", path.display()))); + } + Ok(clean) +} diff --git a/src/bin/fbz/tar_create.rs b/src/bin/fbz/tar_create.rs new file mode 100644 index 0000000..2ccd51e --- /dev/null +++ b/src/bin/fbz/tar_create.rs @@ -0,0 +1,43 @@ +use std::{collections::HashSet, fs, io::Write, path::Path}; + +use fbz::{Bzip2EncodeReport, Bzip2Encoder, EncodeOptions, Error, Result, gzip, lz4}; + +use super::archive_create::archive_name; + +fn invalid(message: impl Into) -> Error { + Error::InvalidConfiguration(message.into()) +} + +fn append_inputs(inputs: &[String], encoder: W) -> Result { + let mut archive = tar::Builder::new(encoder); + let mut names = HashSet::new(); + for input in inputs { + if input == "-" { + return Err(invalid("stdin cannot be used as a tar archive entry")); + } + let path = Path::new(input); + let name = archive_name(path)?; + if !names.insert(name.clone()) { + return Err(invalid(format!("duplicate archive root {}", name.display()))); + } + let metadata = fs::symlink_metadata(path)?; + if metadata.is_dir() { + archive.append_dir_all(&name, path)?; + } else { + archive.append_path_with_name(path, &name)?; + } + } + archive.into_inner().map_err(Error::from) +} + +pub(super) fn pack_gzip(inputs: &[String], output: &mut W, options: EncodeOptions) -> Result { + append_inputs(inputs, gzip::Encoder::new(output, options)?)?.finish().map(|(_, report)| report) +} + +pub(super) fn pack_lz4(inputs: &[String], output: &mut W, options: EncodeOptions) -> Result { + append_inputs(inputs, lz4::Encoder::new(output, options)?)?.finish().map(|(_, report)| report) +} + +pub(super) fn pack_bzip2(inputs: &[String], output: &mut W, options: EncodeOptions) -> Result { + append_inputs(inputs, Bzip2Encoder::new(output, options)?)?.finish().map(|(_, report)| report) +} diff --git a/src/bz2_encode/LICENSE-MIT b/src/bz2_encode/LICENSE-MIT new file mode 100644 index 0000000..001f74b --- /dev/null +++ b/src/bz2_encode/LICENSE-MIT @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2026 John Murray + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/src/bz2_encode/bitwriter.rs b/src/bz2_encode/bitwriter.rs new file mode 100644 index 0000000..74fcfee --- /dev/null +++ b/src/bz2_encode/bitwriter.rs @@ -0,0 +1,165 @@ +//! MSB-first bit writer — the mirror image of the decoder's `BitReader`. +//! +//! Sans-io by construction: everything lands in a `Vec` the caller owns. + +use std::vec::Vec; + +/// Accumulates bits most-significant-first into a byte buffer. +pub struct BitWriter { + out: Vec, + acc: u64, + nbits: u32, +} + +impl BitWriter { + pub fn new() -> Self { + BitWriter { out: Vec::new(), acc: 0, nbits: 0 } + } + + pub fn with_capacity(cap: usize) -> Self { + BitWriter { out: Vec::with_capacity(cap), acc: 0, nbits: 0 } + } + + /// Write the low `n` bits of `val`, most significant first. `n <= 32`. + #[inline] + pub fn write_bits(&mut self, n: u32, val: u32) { + debug_assert!(n <= 32); + let val = if n == 32 { val as u64 } else { (val as u64) & ((1u64 << n) - 1) }; + self.acc = (self.acc << n) | val; + self.nbits += n; + while self.nbits >= 8 { + self.nbits -= 8; + self.out.push((self.acc >> self.nbits) as u8); + } + self.acc &= (1u64 << self.nbits) - 1; + } + + #[inline] + pub fn write_bit(&mut self, bit: u32) { + self.write_bits(1, bit); + } + + /// Write a 48-bit block or end-of-stream magic. + pub fn write_magic(&mut self, magic: u64) { + self.write_bits(24, (magic >> 24) as u32); + self.write_bits(24, (magic & 0xff_ffff) as u32); + } + + pub fn write_u8(&mut self, byte: u8) { + self.write_bits(8, byte as u32); + } + + pub fn write_u32(&mut self, val: u32) { + self.write_bits(32, val); + } + + /// Take the whole bytes emitted so far, leaving any partial byte behind. + /// + /// This is what keeps a streaming writer's memory bounded: output can be + /// handed off as it is produced instead of accumulating until `finish`. + pub fn drain(&mut self) -> Vec { + std::mem::take(&mut self.out) + } + + /// Append an exact MSB-first bit buffer whose final byte may be padded. + pub fn write_buffer(&mut self, bytes: &[u8], bit_len: usize) { + debug_assert!(bit_len <= bytes.len() * 8); + let full = bit_len / 8; + for &byte in &bytes[..full] { + self.write_u8(byte); + } + let trailing = bit_len % 8; + if trailing != 0 { + self.write_bits(trailing as u32, u32::from(bytes[full] >> (8 - trailing))); + } + } + + /// Pad a copy of the final byte while retaining the exact meaningful length. + pub fn finish_bits(mut self) -> (Vec, usize) { + let bit_len = self.out.len() * 8 + self.nbits as usize; + if self.nbits > 0 { + self.out.push((self.acc << (8 - self.nbits)) as u8); + } + (self.out, bit_len) + } + + /// Pad the final partial byte with zero bits and take the buffer. + pub fn finish(mut self) -> Vec { + if self.nbits > 0 { + let pad = 8 - self.nbits; + self.write_bits(pad, 0); + } + debug_assert_eq!(self.nbits, 0); + self.out + } +} + +impl Default for BitWriter { + fn default() -> Self { + BitWriter::new() + } +} + +#[cfg(test)] +mod tests { + use std::vec; + + use super::*; + + #[test] + fn packs_msb_first() { + let mut w = BitWriter::new(); + w.write_bits(4, 0b1010); + w.write_bits(4, 0b0011); + assert_eq!(w.finish(), vec![0b1010_0011]); + } + + #[test] + fn pads_trailing_byte_with_zeros() { + let mut w = BitWriter::new(); + w.write_bits(3, 0b101); + assert_eq!(w.finish(), vec![0b1010_0000]); + } + + #[test] + fn writes_wide_fields() { + let mut w = BitWriter::new(); + w.write_u32(0xdead_beef); + assert_eq!(w.finish(), vec![0xde, 0xad, 0xbe, 0xef]); + } + + #[test] + fn writes_stream_header_shape() { + // "BZh9" then the 48-bit block magic, exactly as the decoder expects. + let mut w = BitWriter::new(); + w.write_u8(b'B'); + w.write_u8(b'Z'); + w.write_u8(b'h'); + w.write_u8(b'9'); + w.write_magic(0x3141_5926_5359); + assert_eq!(w.finish(), vec![0x42, 0x5a, 0x68, 0x39, 0x31, 0x41, 0x59, 0x26, 0x53, 0x59]); + } + + #[test] + fn drain_hands_back_whole_bytes_only() { + let mut w = BitWriter::new(); + w.write_bits(12, 0xabc); + assert_eq!(w.drain(), vec![0xab]); + assert_eq!(w.drain(), Vec::::new()); + w.write_bits(4, 0xd); + assert_eq!(w.finish(), vec![0xcd]); + } + + #[test] + fn straddles_accumulator_boundaries() { + let mut w = BitWriter::new(); + for _ in 0..10 { + w.write_bits(24, 0xab_cdef); + } + let out = w.finish(); + assert_eq!(out.len(), 30); + for chunk in out.chunks(3) { + assert_eq!(chunk, &[0xab, 0xcd, 0xef]); + } + } +} diff --git a/src/bz2_encode/bwt.rs b/src/bz2_encode/bwt.rs new file mode 100644 index 0000000..d980b5d --- /dev/null +++ b/src/bz2_encode/bwt.rs @@ -0,0 +1,381 @@ +//! Burrows–Wheeler transform via a from-scratch SA-IS suffix array. +//! +//! bzip2 sorts the *rotations* of the block, not sentinel-terminated suffixes. +//! We get rotation order out of a suffix sorter by running SA-IS over the +//! doubled block `block ‖ block` and keeping the suffixes that start below `n`: +//! the first `n` characters of such a suffix are exactly the corresponding +//! rotation, so suffix order and rotation order agree wherever the rotations +//! differ. Where two rotations are *equal* the doubled-string order is decided +//! by the trailing remainder, but equal rotations necessarily share the same +//! preceding byte, so the last column — and therefore the transform — is +//! unaffected. + +use std::vec; +use std::vec::Vec; + +/// Sentinel for "no entry yet" inside the suffix array under construction. +const EMPTY: u32 = u32::MAX; + +/// Burrows–Wheeler transform of `block`. +/// +/// Returns the last column of the sorted rotation matrix and the row index of +/// the unrotated block (bzip2's `origPtr`). `block` must not be empty. +pub fn transform(block: &[u8]) -> (Vec, usize) { + let n = block.len(); + assert!(n > 0, "BWT of an empty block"); + + if n == 1 { + return (vec![block[0]], 0); + } + + // Doubled block, shifted up by one so 0 can serve as the unique sentinel. + let mut doubled = Vec::with_capacity(2 * n + 1); + for _ in 0..2 { + doubled.extend(block.iter().map(|&b| b as u32 + 1)); + } + doubled.push(0); + + let sa = sais(&doubled, 257); + + let mut last = Vec::with_capacity(n); + let mut orig_ptr = 0usize; + for &suffix in &sa { + let p = suffix as usize; + if p >= n { + continue; + } + if p == 0 { + orig_ptr = last.len(); + } + last.push(block[(p + n - 1) % n]); + } + debug_assert_eq!(last.len(), n); + + (last, orig_ptr) +} + +/// Suffix array of `s` by SA-IS. `s` must end with a unique smallest value `0`, +/// and every value must be below `k`. +fn sais(s: &[u32], k: usize) -> Vec { + let n = s.len(); + let mut sa = vec![EMPTY; n]; + if n == 1 { + sa[0] = 0; + return sa; + } + + let types = classify(s); + let counts = counts(s, k); + + // Pass 1: seed the LMS suffixes in arbitrary order and induce, which sorts + // the LMS *substrings* even though the suffixes themselves are not yet. + let mut bucket = bucket_ends(&counts); + for i in (1..n).rev() { + if is_lms(&types, i) { + let c = s[i] as usize; + bucket[c] -= 1; + sa[bucket[c] as usize] = i as u32; + } + } + induce(s, &mut sa, &types, &counts); + + // Name the sorted LMS substrings. + let lms_sorted: Vec = sa.iter().copied().filter(|&p| p != EMPTY && p > 0 && is_lms(&types, p as usize)).collect(); + let n1 = lms_sorted.len(); + + let mut names = vec![EMPTY; n / 2 + 1]; + let mut name = 0u32; + let mut prev: Option = None; + for &p in &lms_sorted { + let p = p as usize; + let fresh = match prev { + None => true, + Some(q) => !lms_substr_eq(s, &types, p, q), + }; + if fresh { + name += 1; + prev = Some(p); + } + names[p / 2] = name - 1; + } + + // The reduced string: LMS names in order of position. Its last symbol is + // the sentinel's name, which is 0 and unique, so the recursion's + // precondition holds. + let lms_pos: Vec = (1..n).filter(|&i| is_lms(&types, i)).map(|i| i as u32).collect(); + debug_assert_eq!(lms_pos.len(), n1); + let reduced: Vec = lms_pos.iter().map(|&p| names[p as usize / 2]).collect(); + + let sub_sa = if (name as usize) < n1 { + sais(&reduced, name as usize) + } else { + // All names distinct: the suffix array is just the inverse permutation. + let mut sub = vec![0u32; n1]; + for (i, &c) in reduced.iter().enumerate() { + sub[c as usize] = i as u32; + } + sub + }; + + // Pass 2: seed the LMS suffixes in their true order, induce the rest. + for slot in sa.iter_mut() { + *slot = EMPTY; + } + let mut bucket = bucket_ends(&counts); + for i in (0..n1).rev() { + let p = lms_pos[sub_sa[i] as usize] as usize; + let c = s[p] as usize; + bucket[c] -= 1; + sa[bucket[c] as usize] = p as u32; + } + induce(s, &mut sa, &types, &counts); + + sa +} + +/// `true` marks an S-type position (its suffix is smaller than the next one's). +fn classify(s: &[u32]) -> Vec { + let n = s.len(); + let mut types = vec![false; n]; + types[n - 1] = true; + for i in (0..n - 1).rev() { + types[i] = match s[i].cmp(&s[i + 1]) { + core::cmp::Ordering::Less => true, + core::cmp::Ordering::Greater => false, + core::cmp::Ordering::Equal => types[i + 1], + }; + } + types +} + +#[inline] +fn is_lms(types: &[bool], i: usize) -> bool { + i > 0 && types[i] && !types[i - 1] +} + +fn counts(s: &[u32], k: usize) -> Vec { + let mut counts = vec![0u32; k]; + for &c in s { + counts[c as usize] += 1; + } + counts +} + +fn bucket_starts(counts: &[u32]) -> Vec { + let mut starts = Vec::with_capacity(counts.len()); + let mut sum = 0u32; + for &c in counts { + starts.push(sum); + sum += c; + } + starts +} + +fn bucket_ends(counts: &[u32]) -> Vec { + let mut ends = Vec::with_capacity(counts.len()); + let mut sum = 0u32; + for &c in counts { + sum += c; + ends.push(sum); + } + ends +} + +/// Induced sorting: L-type suffixes left to right, then S-type right to left. +fn induce(s: &[u32], sa: &mut [u32], types: &[bool], counts: &[u32]) { + let n = s.len(); + + let mut bucket = bucket_starts(counts); + for i in 0..n { + let p = sa[i]; + if p != EMPTY && p > 0 { + let j = (p - 1) as usize; + if !types[j] { + let c = s[j] as usize; + sa[bucket[c] as usize] = j as u32; + bucket[c] += 1; + } + } + } + + let mut bucket = bucket_ends(counts); + for i in (0..n).rev() { + let p = sa[i]; + if p != EMPTY && p > 0 { + let j = (p - 1) as usize; + if types[j] { + let c = s[j] as usize; + bucket[c] -= 1; + sa[bucket[c] as usize] = j as u32; + } + } + } +} + +/// Compare the LMS substrings starting at `p` and `q` (each runs to and +/// including the next LMS position). +fn lms_substr_eq(s: &[u32], types: &[bool], p: usize, q: usize) -> bool { + let n = s.len(); + if p == n - 1 || q == n - 1 { + return p == q; + } + let mut d = 0usize; + loop { + if p + d >= n || q + d >= n { + return false; + } + let p_lms = d > 0 && is_lms(types, p + d); + let q_lms = d > 0 && is_lms(types, q + d); + if p_lms && q_lms { + return true; + } + if p_lms != q_lms { + return false; + } + if s[p + d] != s[q + d] || types[p + d] != types[q + d] { + return false; + } + d += 1; + } +} + +#[cfg(test)] +mod tests { + use super::*; + + /// Reference transform: sort every rotation outright. O(n^2 log n) but + /// obviously correct, which is the point. + fn naive(block: &[u8]) -> (Vec, usize) { + let n = block.len(); + let rotation = |i: usize| -> Vec { (0..n).map(|k| block[(i + k) % n]).collect() }; + let mut order: Vec = (0..n).collect(); + order.sort_by_key(|&i| rotation(i)); + let last = order.iter().map(|&i| block[(i + n - 1) % n]).collect::>(); + let orig = order.iter().position(|&i| i == 0).unwrap(); + (last, orig) + } + + /// The decoder's inverse BWT, so the tests close the loop on our own oracle. + fn inverse(last: &[u8], orig_ptr: usize) -> Vec { + let n = last.len(); + let mut cftab = [0u32; 257]; + for &b in last { + cftab[b as usize + 1] += 1; + } + for i in 1..=256 { + cftab[i] += cftab[i - 1]; + } + let mut tt: Vec = last.iter().map(|&b| b as u32).collect(); + for i in 0..n { + let b = (tt[i] & 0xff) as usize; + let idx = cftab[b] as usize; + tt[idx] |= (i as u32) << 8; + cftab[b] += 1; + } + let mut out = Vec::with_capacity(n); + let mut t_pos = tt[orig_ptr] >> 8; + for _ in 0..n { + t_pos = tt[t_pos as usize]; + out.push((t_pos & 0xff) as u8); + t_pos >>= 8; + } + out + } + + #[test] + fn transforms_banana() { + // Rotations of "banana" sort to abanan, anaban, ananab, banana, + // nabana, nanaba — last column "nnbaaa", with "banana" itself at row 3. + let (last, orig) = transform(b"banana"); + assert_eq!(last, b"nnbaaa"); + assert_eq!(orig, 3); + assert_eq!(inverse(&last, orig), b"banana"); + } + + #[test] + fn transforms_the_classic_bananaaa() { + let (last, orig) = transform(b"^BANANA|"); + assert_eq!(naive(b"^BANANA|"), (last.clone(), orig)); + assert_eq!(inverse(&last, orig), b"^BANANA|"); + } + + #[test] + fn transforms_single_byte() { + let (last, orig) = transform(b"a"); + assert_eq!(last, b"a"); + assert_eq!(orig, 0); + assert_eq!(inverse(&last, orig), b"a"); + } + + #[test] + fn transforms_all_identical_bytes() { + for n in 1..40usize { + let block = vec![7u8; n]; + let (last, orig) = transform(&block); + assert_eq!(last, block); + assert_eq!(inverse(&last, orig), block); + } + } + + #[test] + fn matches_the_naive_transform_on_small_inputs() { + // Deterministic xorshift so failures reproduce. + let mut state = 0x1234_5678u32; + let mut next = move || { + state ^= state << 13; + state ^= state >> 17; + state ^= state << 5; + state + }; + for case in 0..300 { + let n = 1 + (case % 40); + let alphabet = 1 + (case % 5) as u8; + let block: Vec = (0..n).map(|_| (next() % alphabet as u32) as u8).collect(); + let (last, orig) = transform(&block); + let (want_last, _) = naive(&block); + assert_eq!(last, want_last, "block {:?}", block); + assert_eq!(inverse(&last, orig), block, "block {:?}", block); + } + } + + #[test] + fn handles_periodic_blocks() { + // Equal rotations are the case where suffix order and rotation order + // can disagree; the last column must still come out right. + for period in 1..8usize { + for reps in 1..8usize { + let block: Vec = (0..period * reps).map(|i| (i % period) as u8).collect(); + let (last, orig) = transform(&block); + let (want_last, _) = naive(&block); + assert_eq!(last, want_last); + assert_eq!(inverse(&last, orig), block); + } + } + } + + #[test] + fn round_trips_a_larger_text_block() { + let mut block = Vec::new(); + while block.len() < 60_000 { + block.extend_from_slice(b"the quick brown fox jumps over the lazy dog. "); + block.extend_from_slice(b"aaaaaaaaaaaaaaaaaaaa"); + } + let (last, orig) = transform(&block); + assert_eq!(inverse(&last, orig), block); + } + + #[test] + fn round_trips_high_entropy_data() { + let mut state = 0x9e37_79b9u32; + let block: Vec = (0..30_000) + .map(|_| { + state ^= state << 13; + state ^= state >> 17; + state ^= state << 5; + (state >> 24) as u8 + }) + .collect(); + let (last, orig) = transform(&block); + assert_eq!(inverse(&last, orig), block); + } +} diff --git a/src/bz2_encode/huffman.rs b/src/bz2_encode/huffman.rs new file mode 100644 index 0000000..b7769f9 --- /dev/null +++ b/src/bz2_encode/huffman.rs @@ -0,0 +1,283 @@ +//! Grouped Huffman coding: 2–6 tables, one selector per 50 symbols. +//! +//! ## Code-length cap +//! +//! The decoder in `lib.rs` reads code lengths as a delta walk and rejects any +//! value outside `1..=20` (`invalid bzip2 Huffman delta length`), while its +//! table builder tolerates up to `MAX_CODE_LEN` = 23. So 20 is the hard ceiling +//! that our own decoder — and the format's delta encoding — will accept. +//! +//! We cap at 17 rather than 20 because that is what libbz2's encoder emits, and +//! staying inside the reference encoder's window keeps us safely interoperable +//! with third-party decoders that assume the narrower range. 17 bits is far +//! past the point where the length limit costs anything measurable: the +//! alphabet is at most 258 symbols, so an unconstrained Huffman code only +//! reaches 17 bits on extremely skewed distributions. + +use std::vec; +use std::vec::Vec; + +use super::bitwriter::BitWriter; + +/// Longest code we will emit. See the module note — the decoder accepts 20. +pub const MAX_CODE_LEN: u8 = 17; + +/// The decoder's delta reader rejects any length outside `1..=20`, so our cap +/// has to stay inside that window. Enforced at compile time. +const _: () = assert!(MAX_CODE_LEN >= 1 && MAX_CODE_LEN <= 20); + +/// Symbols per selector group. +pub const GROUP_SIZE: usize = 50; + +/// Refinement passes over the group→table assignment, as in libbz2. +const N_ITERS: usize = 4; + +/// The Huffman coding chosen for one block. +pub struct Coding { + /// Number of tables actually used (2..=6). + pub n_groups: usize, + /// Table index for each group of `GROUP_SIZE` symbols. + pub selectors: Vec, + /// Code lengths per table, each `alpha_size` long. + pub lens: Vec>, + /// Canonical codes per table. + pub codes: Vec>, +} + +/// Choose the table count the way bzip2 does, from the symbol count. +fn group_count(n_syms: usize) -> usize { + match n_syms { + 0..=199 => 2, + 200..=599 => 3, + 600..=1199 => 4, + 1200..=2399 => 5, + _ => 6, + } +} + +/// Build the grouped Huffman coding for one block's symbol stream. +pub fn build(syms: &[u16], alpha_size: usize) -> Coding { + assert!(!syms.is_empty(), "empty symbol stream"); + assert!(alpha_size >= 3); + + let n_groups = group_count(syms.len()); + let mut lens = initial_lengths(syms, alpha_size, n_groups); + + for _ in 0..N_ITERS { + let mut rfreq = vec![vec![0u32; alpha_size]; n_groups]; + let selectors = assign_selectors(syms, &lens); + for (g, &t) in selectors.iter().enumerate() { + let start = g * GROUP_SIZE; + let end = (start + GROUP_SIZE).min(syms.len()); + for &s in &syms[start..end] { + rfreq[t as usize][s as usize] += 1; + } + } + for (t, len) in lens.iter_mut().enumerate() { + // A zero-frequency symbol still needs a code: the delta encoding + // has no way to say "unused", so every symbol gets weight >= 1. + let weights: Vec = rfreq[t].iter().map(|&f| f.max(1) as u64).collect(); + *len = package_merge(&weights, MAX_CODE_LEN); + } + } + + // One last assignment against the final code lengths. For fixed lengths the + // per-group argmin is optimal, so this can only shrink the output. + let selectors = assign_selectors(syms, &lens); + + let codes = lens.iter().map(|l| canonical_codes(l)).collect(); + + Coding { n_groups, selectors, lens, codes } +} + +/// Pick, for each group of 50 symbols, the table that codes it in fewest bits. +fn assign_selectors(syms: &[u16], lens: &[Vec]) -> Vec { + let mut selectors = Vec::with_capacity(syms.len() / GROUP_SIZE + 1); + let mut start = 0usize; + while start < syms.len() { + let end = (start + GROUP_SIZE).min(syms.len()); + let mut best = 0usize; + let mut best_cost = u64::MAX; + for (t, len) in lens.iter().enumerate() { + let cost: u64 = syms[start..end].iter().map(|&s| len[s as usize] as u64).sum(); + if cost < best_cost { + best_cost = cost; + best = t; + } + } + selectors.push(best as u8); + start = end; + } + selectors +} + +/// Seed the tables by splitting the alphabet into runs of roughly equal total +/// frequency — libbz2's starting point, which the refinement passes then move. +fn initial_lengths(syms: &[u16], alpha_size: usize, n_groups: usize) -> Vec> { + let mut freq = vec![0u32; alpha_size]; + for &s in syms { + freq[s as usize] += 1; + } + + let mut lens = vec![vec![0u8; alpha_size]; n_groups]; + let mut remaining: u32 = syms.len() as u32; + let mut gs: usize = 0; + let mut n_part = n_groups; + + while n_part > 0 { + let target = remaining / n_part as u32; + // `ge` is an inclusive upper bound that starts just below `gs`, so an + // empty slice is representable. + let mut ge: isize = gs as isize - 1; + let mut acc: u32 = 0; + while acc < target && ge < alpha_size as isize - 1 { + ge += 1; + acc += freq[ge as usize]; + } + if ge > gs as isize && n_part != n_groups && n_part != 1 && (n_groups - n_part) % 2 == 1 { + acc -= freq[ge as usize]; + ge -= 1; + } + for (v, slot) in lens[n_part - 1].iter_mut().enumerate() { + *slot = if v as isize >= gs as isize && v as isize <= ge { 1 } else { 15 }; + } + n_part -= 1; + gs = (ge + 1) as usize; + remaining -= acc; + } + + lens +} + +/// Length-limited optimal prefix code lengths by package-merge. +/// +/// Every weight must be at least 1, so every symbol gets a code. `limit` must +/// satisfy `2^limit >= weights.len()`. +pub fn package_merge(weights: &[u64], limit: u8) -> Vec { + let m = weights.len(); + assert!(m >= 2, "package-merge needs at least two symbols"); + assert!(limit >= 1 && (limit >= 32 || (1u64 << limit) >= m as u64), "code-length limit too small for the alphabet"); + + const NONE: u32 = u32::MAX; + // Parallel arrays instead of a struct so the arena stays compact. + let mut weight: Vec = Vec::with_capacity(m * 2); + let mut left: Vec = Vec::with_capacity(m * 2); + let mut right: Vec = Vec::with_capacity(m * 2); + + let mut order: Vec = (0..m).collect(); + order.sort_by_key(|&i| (weights[i], i)); + + // The original coins, ascending by weight; the same list is re-merged at + // every denomination. + let mut leaves: Vec = Vec::with_capacity(m); + for &sym in &order { + weight.push(weights[sym]); + left.push(NONE); + right.push(sym as u32); + leaves.push((weight.len() - 1) as u32); + } + + let mut level = leaves.clone(); + for _ in 1..limit { + let mut packaged: Vec = Vec::with_capacity(level.len() / 2); + let mut i = 0; + while i + 1 < level.len() { + let (a, b) = (level[i], level[i + 1]); + weight.push(weight[a as usize] + weight[b as usize]); + left.push(a); + right.push(b); + packaged.push((weight.len() - 1) as u32); + i += 2; + } + + // Merge the fresh packages back in with the original coins. + let mut merged = Vec::with_capacity(leaves.len() + packaged.len()); + let (mut x, mut y) = (0usize, 0usize); + while x < leaves.len() || y < packaged.len() { + let take_leaf = y >= packaged.len() || (x < leaves.len() && weight[leaves[x] as usize] <= weight[packaged[y] as usize]); + if take_leaf { + merged.push(leaves[x]); + x += 1; + } else { + merged.push(packaged[y]); + y += 1; + } + } + level = merged; + } + + // The cheapest 2m-2 coins form the solution; a symbol's code length is the + // number of selected coins it appears in. + let mut lens = vec![0u8; m]; + let mut stack: Vec = Vec::new(); + for &node in level.iter().take(2 * m - 2) { + stack.push(node); + while let Some(nd) = stack.pop() { + if left[nd as usize] == NONE { + lens[right[nd as usize] as usize] += 1; + } else { + stack.push(left[nd as usize]); + stack.push(right[nd as usize]); + } + } + } + + debug_assert!(lens.iter().all(|&l| l >= 1 && l <= limit)); + lens +} + +/// Canonical code assignment, matching the decoder's `limit`/`base`/`perm` +/// construction: codes are handed out in ascending length, and within a length +/// in ascending symbol order. +pub fn canonical_codes(lens: &[u8]) -> Vec { + let min = *lens.iter().min().unwrap(); + let max = *lens.iter().max().unwrap(); + let mut codes = vec![0u32; lens.len()]; + let mut next = 0u32; + for l in min..=max { + for (i, &ln) in lens.iter().enumerate() { + if ln == l { + codes[i] = next; + next += 1; + } + } + next <<= 1; + } + codes +} + +/// Emit the selector list, move-to-front coded over the table indices and then +/// written in unary. +pub fn write_selectors(w: &mut BitWriter, selectors: &[u8], n_groups: usize) { + let mut pos: Vec = (0..n_groups as u8).collect(); + for &sel in selectors { + let j = pos.iter().position(|&p| p == sel).expect("unknown selector"); + for _ in 0..j { + w.write_bit(1); + } + w.write_bit(0); + let v = pos[j]; + pos.copy_within(0..j, 1); + pos[0] = v; + } +} + +/// Emit each table's code lengths as a delta walk from the previous length. +pub fn write_tables(w: &mut BitWriter, lens: &[Vec]) { + for table in lens { + let mut curr = table[0] as i32; + w.write_bits(5, curr as u32); + for &len in table { + let target = len as i32; + while curr < target { + w.write_bits(2, 0b10); // "1 then 0" -> increment + curr += 1; + } + while curr > target { + w.write_bits(2, 0b11); // "1 then 1" -> decrement + curr -= 1; + } + w.write_bit(0); + } + } +} diff --git a/src/bz2_encode/mod.rs b/src/bz2_encode/mod.rs new file mode 100644 index 0000000..c756bbf --- /dev/null +++ b/src/bz2_encode/mod.rs @@ -0,0 +1,237 @@ +//! Parallel bzip2 compression. +//! +//! The BWT, MTF/RLE2, and grouped-Huffman implementation is adapted from +//! crabz2 0.4.0 by John Murray under the MIT license included in this folder. + +mod bitwriter; +mod bwt; +mod huffman; +mod mtf; +mod rle1; + +use std::io::{self, Read, Write}; + +use bitwriter::BitWriter; + +use crate::{EncodeOptions, Error, Result, crc::Bz2Crc, pipeline::StreamingOrdered}; + +const BLOCK_MAGIC: u64 = 0x3141_5926_5359; +const EOS_MAGIC: u64 = 0x1772_4538_5090; +const AUTO_WORKERS: usize = 12; + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct EncodeReport { + pub input_len: u64, + pub output_len: u64, + pub blocks: u64, +} + +struct BlockJob { + rle: Vec, + crc: u32, + input_len: usize, +} + +struct EncodedBlock { + bytes: Vec, + bit_len: usize, + crc: u32, + input_len: usize, +} + +fn write_symbol_map(output: &mut BitWriter, in_use: &[u8]) { + let mut used = [false; 256]; + for &byte in in_use { + used[byte as usize] = true; + } + let mut groups = 0_u32; + for (index, chunk) in used.chunks(16).enumerate() { + if chunk.iter().any(|&value| value) { + groups |= 1 << (15 - index); + } + } + output.write_bits(16, groups); + for (index, chunk) in used.chunks(16).enumerate() { + if groups & (1 << (15 - index)) == 0 { + continue; + } + let mut bits = 0_u32; + for (bit, &value) in chunk.iter().enumerate() { + if value { + bits |= 1 << (15 - bit); + } + } + output.write_bits(16, bits); + } +} + +fn encode_block(job: BlockJob) -> EncodedBlock { + let (last, origin) = bwt::transform(&job.rle); + let symbols = mtf::encode(&last); + let coding = huffman::build(&symbols.syms, symbols.alpha_size()); + let mut output = BitWriter::with_capacity(job.rle.len() / 2); + output.write_magic(BLOCK_MAGIC); + output.write_u32(job.crc); + output.write_bit(0); + output.write_bits(24, origin as u32); + write_symbol_map(&mut output, &symbols.in_use); + output.write_bits(3, coding.n_groups as u32); + output.write_bits(15, coding.selectors.len() as u32); + huffman::write_selectors(&mut output, &coding.selectors, coding.n_groups); + huffman::write_tables(&mut output, &coding.lens); + for (group, &table) in coding.selectors.iter().enumerate() { + let start = group * huffman::GROUP_SIZE; + let end = (start + huffman::GROUP_SIZE).min(symbols.syms.len()); + let lengths = &coding.lens[table as usize]; + let codes = &coding.codes[table as usize]; + for &symbol in &symbols.syms[start..end] { + output.write_bits(lengths[symbol as usize] as u32, codes[symbol as usize]); + } + } + let (bytes, bit_len) = output.finish_bits(); + EncodedBlock { bytes, bit_len, crc: job.crc, input_len: job.input_len } +} + +fn reservation(block_limit: usize) -> usize { + block_limit.saturating_mul(32) +} + +pub struct Encoder { + output: Option, + bits: BitWriter, + pipeline: StreamingOrdered, + block_limit: usize, + reservation: usize, + block: Vec, + block_crc: Bz2Crc, + block_input_len: usize, + runs: rle1::Splitter, + combined_crc: u32, + input_len: u64, + output_len: u64, + blocks: u64, +} + +impl Encoder { + pub fn new(output: W, options: EncodeOptions) -> Result { + let options = options.validate()?; + let level = options.level_or(9); + let block_limit = level as usize * 100_000 - 19; + let reservation = reservation(block_limit); + if reservation > options.memory_limit { + return Err(Error::InvalidConfiguration(format!( + "bzip2 level {level} compression requires a memory limit of at least {reservation} bytes; choose a lower level or raise --memory-limit" + ))); + } + let requested = options.resolved_threads(); + let requested = if options.threads == 0 { requested.min(AUTO_WORKERS) } else { requested }; + let workers = requested.min((options.memory_limit / reservation).max(1)); + let mut encoder = Self { + output: Some(output), + bits: BitWriter::new(), + pipeline: StreamingOrdered::new(workers, options.memory_limit, "fbz-bzip2-encode")?, + block_limit, + reservation, + block: Vec::with_capacity(block_limit), + block_crc: Bz2Crc::new(), + block_input_len: 0, + runs: rle1::Splitter::new(), + combined_crc: 0, + input_len: 0, + output_len: 0, + blocks: 0, + }; + for byte in [b'B', b'Z', b'h', b'0' + level] { + encoder.bits.write_u8(byte); + } + encoder.drain_bits()?; + Ok(encoder) + } + + fn drain_bits(&mut self) -> Result<()> { + let bytes = self.bits.drain(); + self.output.as_mut().unwrap().write_all(&bytes)?; + self.output_len += bytes.len() as u64; + Ok(()) + } + + fn commit_next(&mut self) -> Result<()> { + let block = self.pipeline.take_next()?; + self.bits.write_buffer(&block.bytes, block.bit_len); + self.drain_bits()?; + self.combined_crc = self.combined_crc.rotate_left(1) ^ block.crc; + self.input_len += block.input_len as u64; + self.blocks += 1; + Ok(()) + } + + fn submit_block(&mut self) -> Result<()> { + if self.block.is_empty() { + return Ok(()); + } + while !self.pipeline.can_submit(self.reservation) { + self.commit_next()?; + } + let rle = std::mem::replace(&mut self.block, Vec::with_capacity(self.block_limit)); + let crc = std::mem::replace(&mut self.block_crc, Bz2Crc::new()).finish(); + let input_len = std::mem::take(&mut self.block_input_len); + self.pipeline.submit(self.reservation, move || encode_block(BlockJob { rle, crc, input_len })) + } + + fn commit_group(&mut self, group: rle1::Group) -> Result<()> { + if self.block.len() + group.encoded_len() > self.block_limit { + self.submit_block()?; + } + group.write_into(&mut self.block); + self.block_crc.push_repeat(group.byte, group.raw_len); + self.block_input_len += group.raw_len; + Ok(()) + } + + pub fn finish(mut self) -> Result<(W, EncodeReport)> { + if let Some(group) = self.runs.finish() { + self.commit_group(group)?; + } + self.submit_block()?; + while self.pipeline.has_pending() { + self.commit_next()?; + } + self.bits.write_magic(EOS_MAGIC); + self.bits.write_u32(self.combined_crc); + let trailing = std::mem::take(&mut self.bits).finish(); + self.output.as_mut().unwrap().write_all(&trailing)?; + self.output.as_mut().unwrap().flush()?; + self.output_len += trailing.len() as u64; + let report = EncodeReport { input_len: self.input_len, output_len: self.output_len, blocks: self.blocks }; + Ok((self.output.take().unwrap(), report)) + } +} + +impl Write for Encoder { + fn write(&mut self, bytes: &[u8]) -> io::Result { + for &byte in bytes { + if let Some(group) = self.runs.push(byte) { + self.commit_group(group).map_err(Error::into_io)?; + } + } + Ok(bytes.len()) + } + + fn flush(&mut self) -> io::Result<()> { + self.drain_bits().map_err(Error::into_io)?; + self.output.as_mut().unwrap().flush() + } +} + +pub fn compress(data: &[u8], options: EncodeOptions) -> Result> { + let mut output = Vec::new(); + compress_to_writer(&mut io::Cursor::new(data), &mut output, options)?; + Ok(output) +} + +pub fn compress_to_writer(input: &mut impl Read, output: &mut impl Write, options: EncodeOptions) -> Result { + let mut encoder = Encoder::new(output, options)?; + io::copy(input, &mut encoder)?; + let (_, report) = encoder.finish()?; + Ok(report) +} diff --git a/src/bz2_encode/mtf.rs b/src/bz2_encode/mtf.rs new file mode 100644 index 0000000..17f80a1 --- /dev/null +++ b/src/bz2_encode/mtf.rs @@ -0,0 +1,214 @@ +//! Move-to-front plus RLE2 — the stage that turns the BWT output into the +//! symbol alphabet the Huffman coder sees. +//! +//! Symbol space, matching the decoder exactly: +//! +//! * `0` = RUNA, `1` = RUNB — a bijective base-2 encoding of a run of MTF +//! index zero, least significant digit first, RUNA worth `1 << k` and RUNB +//! worth `2 << k`. +//! * `2 ..= n_in_use` — MTF index `sym - 1`, i.e. indices 1 and up. +//! * `n_in_use + 1` = EOB. + +use std::vec::Vec; + +pub const RUNA: u16 = 0; +pub const RUNB: u16 = 1; + +/// The MTF/RLE2 symbol stream for one block. +pub struct Symbols { + /// The byte values present in the block, ascending — bzip2's `seqToUnseq`. + pub in_use: Vec, + /// The symbol stream, terminated by EOB. + pub syms: Vec, +} + +impl Symbols { + /// Size of the Huffman alphabet: `n_in_use + 2`. + pub fn alpha_size(&self) -> usize { + self.in_use.len() + 2 + } +} + +/// Which of the 256 byte values occur in `block`. +pub fn in_use(block: &[u8]) -> Vec { + let mut seen = [false; 256]; + for &b in block { + seen[b as usize] = true; + } + (0..256usize).filter(|&b| seen[b]).map(|b| b as u8).collect() +} + +/// Move-to-front and RLE2 encode the BWT output. +pub fn encode(block: &[u8]) -> Symbols { + let in_use = in_use(block); + let mut syms = Vec::with_capacity(block.len() / 2 + 8); + + // MTF list starts as the in-use bytes in ascending order. A linear scan is + // what libbz2 does too: after the BWT the hit is almost always near the + // front, so the average probe is very short. + let mut mtf = in_use.clone(); + let mut zeros: u32 = 0; + + for &b in block { + let idx = mtf.iter().position(|&m| m == b).expect("byte not in use"); + if idx == 0 { + zeros += 1; + continue; + } + flush_zero_run(&mut syms, &mut zeros); + mtf.copy_within(0..idx, 1); + mtf[0] = b; + syms.push(idx as u16 + 1); + } + flush_zero_run(&mut syms, &mut zeros); + + let eob = (in_use.len() + 1) as u16; + syms.push(eob); + + Symbols { in_use, syms } +} + +/// Emit a pending run of MTF index zero as RUNA/RUNB digits. +fn flush_zero_run(syms: &mut Vec, zeros: &mut u32) { + let mut n = *zeros; + *zeros = 0; + while n > 0 { + if n % 2 == 1 { + syms.push(RUNA); + n = (n - 1) / 2; + } else { + syms.push(RUNB); + n = (n - 2) / 2; + } + } +} + +#[cfg(test)] +mod tests { + use std::vec; + + use super::*; + + /// The decoder's MTF/RLE2 loop, so we test against our own oracle. + fn decode(sym: &Symbols) -> Vec { + let mut mtf = sym.in_use.clone(); + let eob = (sym.in_use.len() + 1) as u16; + let mut out = Vec::new(); + let mut run: u64 = 0; + let mut run_bit: u32 = 0; + for &s in &sym.syms { + if s <= 1 { + run += ((s as u64) + 1) << run_bit; + run_bit += 1; + continue; + } + if run > 0 { + let b = mtf[0]; + for _ in 0..run { + out.push(b); + } + run = 0; + run_bit = 0; + } + if s == eob { + break; + } + let nn = (s - 1) as usize; + let b = mtf[nn]; + mtf.copy_within(0..nn, 1); + mtf[0] = b; + out.push(b); + } + out + } + + #[test] + fn encodes_a_zero_run_bijectively() { + // "aaaa" -> MTF indices 0,0,0,0 -> run of 4 -> RUNB, RUNA. + let sym = encode(b"aaaa"); + assert_eq!(sym.in_use, vec![b'a']); + assert_eq!(sym.syms, vec![RUNB, RUNA, 2]); // 2 == eob for one byte value + assert_eq!(sym.syms.last(), Some(&2)); // EOB == n_in_use + 1 + assert_eq!(sym.alpha_size(), 3); + assert_eq!(decode(&sym), b"aaaa"); + } + + #[test] + fn run_lengths_one_through_ten_use_the_right_digits() { + let expected: [&[u16]; 10] = [ + &[RUNA], + &[RUNB], + &[RUNA, RUNA], + &[RUNB, RUNA], + &[RUNA, RUNB], + &[RUNB, RUNB], + &[RUNA, RUNA, RUNA], + &[RUNB, RUNA, RUNA], + &[RUNA, RUNB, RUNA], + &[RUNB, RUNB, RUNA], + ]; + for (i, want) in expected.iter().enumerate() { + let n = i + 1; + let mut syms = Vec::new(); + let mut zeros = n as u32; + flush_zero_run(&mut syms, &mut zeros); + assert_eq!(&syms[..], *want, "run of {}", n); + // And the digits sum back to n the way the decoder adds them up. + let total: u64 = syms.iter().enumerate().map(|(k, &s)| ((s as u64) + 1) << k).sum(); + assert_eq!(total, n as u64); + } + } + + #[test] + fn moves_bytes_to_the_front() { + // in_use = [a, b, c]; "abc" -> indices 0, 1, 2. + let sym = encode(b"abc"); + assert_eq!(sym.in_use, vec![b'a', b'b', b'c']); + // zero run of 1 (RUNA), then index 1 -> sym 2, then index 2 -> sym 3, EOB 4. + assert_eq!(sym.syms, vec![RUNA, 2, 3, 4]); + assert_eq!(decode(&sym), b"abc"); + } + + #[test] + fn alternating_bytes_never_hit_index_zero_twice() { + let sym = encode(b"ababab"); + // a -> 0 (RUNA), b -> 1, a -> 1, b -> 1, a -> 1, b -> 1 + assert_eq!(sym.syms, vec![RUNA, 2, 2, 2, 2, 2, 3]); + assert_eq!(decode(&sym), b"ababab"); + } + + #[test] + fn round_trips_varied_blocks() { + let mut state = 0x2545_f491u32; + let mut next = move || { + state ^= state << 13; + state ^= state >> 17; + state ^= state << 5; + state + }; + for case in 1..200usize { + let n = case * 7; + let alphabet = 1 + (case % 200) as u32; + let block: Vec = (0..n).map(|_| (next() % alphabet) as u8).collect(); + let sym = encode(&block); + assert_eq!(decode(&sym), block, "case {}", case); + } + } + + #[test] + fn round_trips_a_long_single_byte_run() { + let block = vec![0xffu8; 100_000]; + let sym = encode(&block); + assert_eq!(decode(&sym), block); + } + + #[test] + fn round_trips_all_256_byte_values() { + let block: Vec = (0..=255u8).chain((0..=255u8).rev()).collect(); + let sym = encode(&block); + assert_eq!(sym.in_use.len(), 256); + assert_eq!(sym.alpha_size(), 258); + assert_eq!(sym.syms.last(), Some(&257)); // EOB == n_in_use + 1 + assert_eq!(decode(&sym), block); + } +} diff --git a/src/bz2_encode/rle1.rs b/src/bz2_encode/rle1.rs new file mode 100644 index 0000000..73baf97 --- /dev/null +++ b/src/bz2_encode/rle1.rs @@ -0,0 +1,225 @@ +//! RLE1 — the first run-length stage, applied to the raw plaintext. +//! +//! A run of four identical bytes is followed by a single byte giving the number +//! of *extra* repeats (0..=255), so the longest run one group can express is +//! 259 bytes. The decoder resets its run state after consuming that count byte, +//! which is what lets a block boundary fall between any two groups — the +//! property the block builder relies on. + +use std::vec::Vec; + +/// The longest run one group can express: four literals plus 255 extra. +pub const MAX_RUN: usize = 4 + 255; + +/// A maximal run of one byte value, capped at [`MAX_RUN`]. This is the atom the +/// block builder places: a group is never split across blocks. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct Group { + pub byte: u8, + pub raw_len: usize, +} + +impl Group { + /// How many bytes this group occupies in the RLE1 output. + pub fn encoded_len(self) -> usize { + if self.raw_len >= 4 { 5 } else { self.raw_len } + } + + /// Append the group's encoded form to `out`. + pub fn write_into(self, out: &mut Vec) { + let b = self.byte; + if self.raw_len >= 4 { + out.extend_from_slice(&[b, b, b, b]); + out.push((self.raw_len - 4) as u8); + } else { + for _ in 0..self.raw_len { + out.push(b); + } + } + } +} + +/// Splits a byte stream into RLE1 groups as it arrives. +/// +/// Incremental by design: a run may span any number of calls, so the caller +/// never has to buffer plaintext just to find run boundaries. +pub struct Splitter { + byte: u8, + len: usize, +} + +impl Splitter { + pub fn new() -> Splitter { + Splitter { byte: 0, len: 0 } + } + + /// Feed one byte. Returns the group that just closed, if any. + #[inline] + pub fn push(&mut self, b: u8) -> Option { + if self.len > 0 && b == self.byte && self.len < MAX_RUN { + self.len += 1; + return None; + } + let done = self.take(); + self.byte = b; + self.len = 1; + done + } + + /// Close the stream, returning the final partial group. + pub fn finish(&mut self) -> Option { + self.take() + } + + fn take(&mut self) -> Option { + if self.len == 0 { + return None; + } + let group = Group { byte: self.byte, raw_len: self.len }; + self.len = 0; + Some(group) + } +} + +impl Default for Splitter { + fn default() -> Self { + Splitter::new() + } +} + +#[cfg(test)] +mod tests { + use std::vec; + + use super::*; + + /// Drive the splitter over a whole buffer, as the block builder does. + fn encode(input: &[u8]) -> Vec { + let mut out = Vec::new(); + let mut split = Splitter::new(); + let mut raw = 0usize; + for &b in input { + if let Some(g) = split.push(b) { + g.write_into(&mut out); + raw += g.raw_len; + } + } + if let Some(g) = split.finish() { + g.write_into(&mut out); + raw += g.raw_len; + } + assert_eq!(raw, input.len(), "groups must account for every input byte"); + out + } + + /// The inverse transform, lifted from the decoder's tail loop, so the tests + /// check the two halves against each other rather than against a guess. + fn decode(enc: &[u8]) -> Vec { + let mut out = Vec::new(); + let mut prev: i32 = -1; + let mut count = 0u32; + for &b in enc { + if count == 4 { + for _ in 0..b { + out.push(prev as u8); + } + count = 0; + prev = -1; + } else { + out.push(b); + if b as i32 == prev { + count += 1; + } else { + prev = b as i32; + count = 1; + } + } + } + out + } + + #[test] + fn passes_short_runs_through() { + assert_eq!(encode(b"abcaaabbb"), b"abcaaabbb"); + assert_eq!(decode(&encode(b"abcaaabbb")), b"abcaaabbb"); + } + + #[test] + fn encodes_a_run_of_exactly_four() { + assert_eq!(encode(b"aaaa"), b"aaaa\x00"); + assert_eq!(decode(&encode(b"aaaa")), b"aaaa"); + } + + #[test] + fn encodes_a_run_of_five() { + assert_eq!(encode(b"aaaaa"), b"aaaa\x01"); + assert_eq!(decode(&encode(b"aaaaa")), b"aaaaa"); + } + + #[test] + fn splits_runs_longer_than_the_cap() { + let input = vec![b'z'; 300]; + let enc = encode(&input); + // 259 in the first group, then 41 in a second. + assert_eq!(&enc[..5], b"zzzz\xff"); + assert_eq!(&enc[5..10], &[b'z', b'z', b'z', b'z', 41 - 4]); + assert_eq!(enc.len(), 10); + assert_eq!(decode(&enc), input); + } + + #[test] + fn splits_a_run_of_exactly_262() { + // 259 plus a 3-byte tail that has to come out as literals. + let input = vec![b'q'; 262]; + assert_eq!(encode(&input), b"qqqq\xffqqq"); + assert_eq!(decode(&encode(&input)), input); + } + + #[test] + fn group_sizes_are_reported_correctly() { + for raw_len in 1..=MAX_RUN { + let g = Group { byte: 7, raw_len }; + let mut out = Vec::new(); + g.write_into(&mut out); + assert_eq!(out.len(), g.encoded_len(), "raw_len {raw_len}"); + assert_eq!(decode(&out), vec![7u8; raw_len]); + } + } + + #[test] + fn handles_empty_input() { + assert!(encode(b"").is_empty()); + } + + #[test] + fn a_run_never_exceeds_the_cap() { + let input = vec![b'k'; 5000]; + let mut split = Splitter::new(); + let mut groups = Vec::new(); + for &b in &input { + if let Some(g) = split.push(b) { + groups.push(g); + } + } + if let Some(g) = split.finish() { + groups.push(g); + } + assert!(groups.iter().all(|g| g.raw_len <= MAX_RUN)); + assert_eq!(groups.iter().map(|g| g.raw_len).sum::(), 5000); + } + + #[test] + fn round_trips_a_gnarly_input() { + let mut input = Vec::new(); + for i in 0..2000u32 { + let b = (i / 7 % 5) as u8; + for _ in 0..(i % 9) { + input.push(b); + } + } + for cut in [0, 1, 2, 3, 4, 5, 100, 1000, input.len()] { + let slice = &input[..cut.min(input.len())]; + assert_eq!(decode(&encode(slice)), slice); + } + } +} diff --git a/src/crc.rs b/src/crc.rs index 67a462f..ec6decd 100644 --- a/src/crc.rs +++ b/src/crc.rs @@ -18,14 +18,42 @@ const fn make_table() -> [u32; 256] { const TABLE: [u32; 256] = make_table(); +#[derive(Clone, Copy)] +pub(crate) struct Bz2Crc { + state: u32, +} + +impl Bz2Crc { + pub fn new() -> Self { + Self { state: u32::MAX } + } + + #[inline] + pub fn push_repeat(&mut self, byte: u8, count: usize) { + for _ in 0..count { + let index = ((self.state >> 24) as u8 ^ byte) as usize; + self.state = (self.state << 8) ^ TABLE[index]; + } + } + + #[inline] + pub fn update(&mut self, data: &[u8]) { + for &byte in data { + let index = ((self.state >> 24) as u8 ^ byte) as usize; + self.state = (self.state << 8) ^ TABLE[index]; + } + } + + pub fn finish(self) -> u32 { + !self.state + } +} + /// Compute the CRC used for an uncompressed bzip2 block. pub fn bz2_crc32(data: &[u8]) -> u32 { - let mut crc = u32::MAX; - for &byte in data { - let index = ((crc >> 24) as u8 ^ byte) as usize; - crc = (crc << 8) ^ TABLE[index]; - } - !crc + let mut crc = Bz2Crc::new(); + crc.update(data); + crc.finish() } /// Add a block CRC to bzip2's combined stream CRC. diff --git a/src/deflate.rs b/src/deflate.rs index 4b29d79..3e9646e 100644 --- a/src/deflate.rs +++ b/src/deflate.rs @@ -1,9 +1,28 @@ -//! Raw DEFLATE decompression shared by gzip and ZIP framing. +//! Raw DEFLATE compression and decompression shared by gzip and ZIP framing. -use crate::{DecodeOptions, DecodeProgress, OutputSink, Result, gzip}; +use std::io::{Read, Write}; +use crate::{DecodeOptions, DecodeProgress, EncodeOptions, OutputSink, Result, deflate_encode, gzip}; + +pub(crate) const LENGTH_BASE: [usize; 29] = [3, 4, 5, 6, 7, 8, 9, 10, 11, 13, 15, 17, 19, 23, 27, 31, 35, 43, 51, 59, 67, 83, 99, 115, 131, 163, 195, 227, 258]; +pub(crate) const LENGTH_EXTRA: [u8; 29] = [0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 2, 2, 2, 2, 3, 3, 3, 3, 4, 4, 4, 4, 5, 5, 5, 5, 0]; +pub(crate) const DISTANCE_BASE: [usize; 30] = + [1, 2, 3, 4, 5, 7, 9, 13, 17, 25, 33, 49, 65, 97, 129, 193, 257, 385, 513, 769, 1025, 1537, 2049, 3073, 4097, 6145, 8193, 12289, 16385, 24577]; +pub(crate) const DISTANCE_EXTRA: [u8; 30] = [0, 0, 0, 0, 1, 1, 2, 2, 3, 3, 4, 4, 5, 5, 6, 6, 7, 7, 8, 8, 9, 9, 10, 10, 11, 11, 12, 12, 13, 13]; + +pub use deflate_encode::{EncodeReport, Encoder}; pub use gzip::DeflateReport as Report; +pub fn compress_to_writer(input: &mut impl Read, output: &mut impl Write, options: EncodeOptions) -> Result { + deflate_encode::compress_to_writer(input, output, options) +} + +#[doc(hidden)] +pub fn compress_bytes_serial(data: &[u8], level: u8) -> Result<(Vec, EncodeReport)> { + EncodeOptions { level: Some(level), ..EncodeOptions::default() }.validate()?; + Ok(deflate_encode::compress_bytes_serial(data, level)) +} + /// Decode one raw DEFLATE stream into an output sink. #[doc(hidden)] pub fn decompress_to_sink_with_options_and_progress( diff --git a/src/deflate_encode.rs b/src/deflate_encode.rs new file mode 100644 index 0000000..f2100b5 --- /dev/null +++ b/src/deflate_encode.rs @@ -0,0 +1,535 @@ +use std::{ + cmp::Reverse, + collections::BinaryHeap, + io::{self, Read, Write}, +}; + +use crc32fast::Hasher; + +use crate::{ + EncodeOptions, Error, Result, + deflate::{DISTANCE_BASE, DISTANCE_EXTRA, LENGTH_BASE, LENGTH_EXTRA}, + matchfinder::HashChain, + pipeline::StreamingOrdered, +}; + +const WINDOW_SIZE: usize = 32 * 1024; +const TARGET_SEGMENT_SIZE: usize = 1024 * 1024; +const MIN_SEGMENT_SIZE: usize = 64 * 1024; +const MAX_MATCH: usize = 258; + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct EncodeReport { + pub input_len: u64, + pub output_len: u64, + pub crc: u32, +} + +struct BitWriter { + bytes: Vec, + pending: u64, + bits: u8, +} + +impl BitWriter { + fn new(capacity: usize) -> Self { + Self { bytes: Vec::with_capacity(capacity), pending: 0, bits: 0 } + } + + fn write_bits(&mut self, value: u32, count: u8) { + debug_assert!(count <= 24); + self.pending |= u64::from(value) << self.bits; + self.bits += count; + while self.bits >= 8 { + self.bytes.push(self.pending as u8); + self.pending >>= 8; + self.bits -= 8; + } + } + + fn align_zero(&mut self) { + if self.bits != 0 { + self.bytes.push(self.pending as u8); + self.pending = 0; + self.bits = 0; + } + } + + fn write_aligned(&mut self, bytes: &[u8]) { + debug_assert_eq!(self.bits, 0); + self.bytes.extend_from_slice(bytes); + } + + fn finish_aligned(mut self) -> Vec { + self.align_zero(); + self.bytes + } +} + +fn reverse_code(code: u16, bits: u8) -> u16 { + code.reverse_bits() >> (16 - bits) +} + +fn fixed_code(symbol: usize) -> (u16, u8) { + match symbol { + 0..=143 => (reverse_code(0x30 + symbol as u16, 8), 8), + 144..=255 => (reverse_code(0x190 + (symbol - 144) as u16, 9), 9), + 256..=279 => (reverse_code((symbol - 256) as u16, 7), 7), + 280..=287 => (reverse_code(0xc0 + (symbol - 280) as u16, 8), 8), + _ => unreachable!(), + } +} + +fn write_fixed_symbol(output: &mut BitWriter, symbol: usize) { + let (code, bits) = fixed_code(symbol); + output.write_bits(u32::from(code), bits); +} + +fn symbol_for(value: usize, bases: &[usize], extras: &[u8]) -> (usize, u32, u8) { + for index in (0..bases.len()).rev() { + if value >= bases[index] { + return (index, (value - bases[index]) as u32, extras[index]); + } + } + unreachable!() +} + +fn write_match(output: &mut BitWriter, length: usize, distance: usize) { + let (length_index, length_extra, length_bits) = symbol_for(length, &LENGTH_BASE, &LENGTH_EXTRA); + write_fixed_symbol(output, 257 + length_index); + output.write_bits(length_extra, length_bits); + let (distance_symbol, distance_extra, distance_bits) = symbol_for(distance, &DISTANCE_BASE, &DISTANCE_EXTRA); + output.write_bits(u32::from(reverse_code(distance_symbol as u16, 5)), 5); + output.write_bits(distance_extra, distance_bits); +} + +#[derive(Clone, Copy)] +enum Token { + Literal(u8), + Match { length: u16, distance: u16 }, +} + +fn chain_depth(level: u8) -> usize { + match level { + 1 => 4, + 2 => 8, + 3 => 16, + 4 => 32, + 5 => 48, + 6 => 64, + 7 => 96, + 8 => 192, + 9 => 384, + _ => unreachable!(), + } +} + +fn sync_boundary(output: &mut BitWriter) { + output.write_bits(0, 3); + output.align_zero(); + output.write_aligned(&[0, 0, 0xff, 0xff]); +} + +fn tokenize(bytes: &[u8], prefix_len: usize, level: u8) -> Vec { + let mut tokens = Vec::with_capacity(bytes.len() - prefix_len); + let max_chain = chain_depth(level); + let mut finder = HashChain::new(bytes.len(), WINDOW_SIZE, max_chain); + let dictionary_start = prefix_len.saturating_sub(WINDOW_SIZE); + for position in dictionary_start..prefix_len { + finder.insert(bytes, position); + } + let mut position = prefix_len; + while position < bytes.len() { + let (length, distance) = finder.best_match(bytes, position, MAX_MATCH, 3, max_chain); + if length == 0 { + tokens.push(Token::Literal(bytes[position])); + finder.insert(bytes, position); + position += 1; + } else { + finder.insert(bytes, position); + if level >= 4 && position + 1 < bytes.len() { + let (next_length, _) = finder.best_match(bytes, position + 1, MAX_MATCH, 3, max_chain / 2); + if next_length > length + 1 { + tokens.push(Token::Literal(bytes[position])); + position += 1; + continue; + } + } + tokens.push(Token::Match { length: length as u16, distance: distance as u16 }); + let end = position + length; + position += 1; + while position < end { + finder.insert(bytes, position); + position += 1; + } + } + } + tokens +} + +fn encode_fixed(tokens: &[Token], input_len: usize) -> Vec { + let mut output = BitWriter::new(input_len / 2); + output.write_bits(0b010, 3); + for token in tokens { + match *token { + Token::Literal(byte) => write_fixed_symbol(&mut output, byte as usize), + Token::Match { length, distance } => write_match(&mut output, length as usize, distance as usize), + } + } + write_fixed_symbol(&mut output, 256); + sync_boundary(&mut output); + output.finish_aligned() +} + +fn huffman_lengths(frequencies: &[u32], max_bits: u8) -> Vec { + let mut scaled = frequencies.to_vec(); + loop { + let mut heap = BinaryHeap::new(); + let mut children = vec![None; scaled.len()]; + for (symbol, &frequency) in scaled.iter().enumerate() { + if frequency != 0 { + heap.push(Reverse((u64::from(frequency), symbol))); + } + } + debug_assert!(heap.len() >= 2); + while heap.len() > 1 { + let Reverse((left_frequency, left)) = heap.pop().unwrap(); + let Reverse((right_frequency, right)) = heap.pop().unwrap(); + let node = children.len(); + children.push(Some((left, right))); + heap.push(Reverse((left_frequency + right_frequency, node))); + } + let root = heap.pop().unwrap().0.1; + let mut lengths = vec![0_u8; scaled.len()]; + let mut stack = vec![(root, 0_u8)]; + while let Some((node, depth)) = stack.pop() { + if node < scaled.len() { + lengths[node] = depth.max(1); + } else { + let (left, right) = children[node].unwrap(); + stack.push((left, depth + 1)); + stack.push((right, depth + 1)); + } + } + if lengths.iter().copied().max().unwrap_or(0) <= max_bits { + return lengths; + } + for frequency in &mut scaled { + if *frequency != 0 { + *frequency = frequency.div_ceil(2); + } + } + } +} + +fn ensure_two(frequencies: &mut [u32]) { + let active: Vec<_> = frequencies.iter().enumerate().filter_map(|(symbol, &frequency)| (frequency != 0).then_some(symbol)).collect(); + if active.len() >= 2 { + return; + } + if active.is_empty() { + frequencies[0] = 1; + frequencies[1] = 1; + return; + } + let other = if active.first() == Some(&0) { 1 } else { 0 }; + frequencies[other] = 1; +} + +fn huffman_codes(lengths: &[u8], max_bits: u8) -> Vec { + let mut counts = vec![0_u16; max_bits as usize + 1]; + for &length in lengths { + if length != 0 { + counts[length as usize] += 1; + } + } + let mut next = vec![0_u16; counts.len()]; + let mut code = 0_u16; + for bits in 1..counts.len() { + code = (code + counts[bits - 1]) << 1; + next[bits] = code; + } + lengths + .iter() + .map(|&length| { + if length == 0 { + 0 + } else { + let code = next[length as usize]; + next[length as usize] += 1; + reverse_code(code, length) + } + }) + .collect() +} + +fn frequencies(tokens: &[Token]) -> ([u32; 286], [u32; 30]) { + let mut literals = [0_u32; 286]; + let mut distances = [0_u32; 30]; + for token in tokens { + match *token { + Token::Literal(byte) => literals[byte as usize] += 1, + Token::Match { length, distance } => { + let (length_symbol, _, _) = symbol_for(length as usize, &LENGTH_BASE, &LENGTH_EXTRA); + let (distance_symbol, _, _) = symbol_for(distance as usize, &DISTANCE_BASE, &DISTANCE_EXTRA); + literals[257 + length_symbol] += 1; + distances[distance_symbol] += 1; + } + } + } + literals[256] += 1; + ensure_two(&mut literals); + ensure_two(&mut distances); + (literals, distances) +} + +const CODE_LENGTH_ORDER: [usize; 19] = [16, 17, 18, 0, 8, 7, 9, 6, 10, 5, 11, 4, 12, 3, 13, 2, 14, 1, 15]; + +fn write_code(output: &mut BitWriter, symbol: usize, codes: &[u16], lengths: &[u8]) { + output.write_bits(u32::from(codes[symbol]), lengths[symbol]); +} + +fn encode_dynamic(tokens: &[Token], input_len: usize) -> Vec { + let (literal_frequencies, distance_frequencies) = frequencies(tokens); + let literal_lengths = huffman_lengths(&literal_frequencies, 15); + let distance_lengths = huffman_lengths(&distance_frequencies, 15); + let literal_count = literal_lengths.iter().rposition(|&length| length != 0).unwrap().max(256) + 1; + let distance_count = distance_lengths.iter().rposition(|&length| length != 0).unwrap() + 1; + let all_lengths: Vec<_> = literal_lengths[..literal_count].iter().chain(&distance_lengths[..distance_count]).copied().collect(); + let mut code_length_frequencies = [0_u32; 19]; + for &length in &all_lengths { + code_length_frequencies[length as usize] += 1; + } + ensure_two(&mut code_length_frequencies); + let code_length_lengths = huffman_lengths(&code_length_frequencies, 7); + let code_length_count = CODE_LENGTH_ORDER.iter().rposition(|&symbol| code_length_lengths[symbol] != 0).unwrap().max(3) + 1; + let literal_codes = huffman_codes(&literal_lengths, 15); + let distance_codes = huffman_codes(&distance_lengths, 15); + let code_length_codes = huffman_codes(&code_length_lengths, 7); + + let mut output = BitWriter::new(input_len / 2); + output.write_bits(0b100, 3); + output.write_bits((literal_count - 257) as u32, 5); + output.write_bits((distance_count - 1) as u32, 5); + output.write_bits((code_length_count - 4) as u32, 4); + for &symbol in &CODE_LENGTH_ORDER[..code_length_count] { + output.write_bits(u32::from(code_length_lengths[symbol]), 3); + } + for &length in &all_lengths { + write_code(&mut output, length as usize, &code_length_codes, &code_length_lengths); + } + for token in tokens { + match *token { + Token::Literal(byte) => write_code(&mut output, byte as usize, &literal_codes, &literal_lengths), + Token::Match { length, distance } => { + let (length_symbol, length_extra, length_bits) = symbol_for(length as usize, &LENGTH_BASE, &LENGTH_EXTRA); + write_code(&mut output, 257 + length_symbol, &literal_codes, &literal_lengths); + output.write_bits(length_extra, length_bits); + let (distance_symbol, distance_extra, distance_bits) = symbol_for(distance as usize, &DISTANCE_BASE, &DISTANCE_EXTRA); + write_code(&mut output, distance_symbol, &distance_codes, &distance_lengths); + output.write_bits(distance_extra, distance_bits); + } + } + } + write_code(&mut output, 256, &literal_codes, &literal_lengths); + sync_boundary(&mut output); + output.finish_aligned() +} + +fn encode_stored(bytes: &[u8]) -> Vec { + let mut output = BitWriter::new(bytes.len() + bytes.len().div_ceil(u16::MAX as usize) * 5); + for chunk in bytes.chunks(u16::MAX as usize) { + output.write_bits(0, 3); + output.align_zero(); + let length = chunk.len() as u16; + output.write_aligned(&length.to_le_bytes()); + output.write_aligned(&(!length).to_le_bytes()); + output.write_aligned(chunk); + } + output.finish_aligned() +} + +struct Segment { + encoded: Vec, + input_len: usize, + crc: Hasher, +} + +fn encode_segment(buffer: Vec, prefix_len: usize, level: u8) -> Segment { + let plain = &buffer[prefix_len..]; + let tokens = tokenize(&buffer, prefix_len, level); + let fixed = encode_fixed(&tokens, plain.len()); + let dynamic = encode_dynamic(&tokens, plain.len()); + let stored = encode_stored(plain); + let encoded = if dynamic.len() < fixed.len() { dynamic } else { fixed }; + let encoded = if encoded.len() < stored.len() { encoded } else { stored }; + let mut crc = Hasher::new(); + crc.update(plain); + Segment { encoded, input_len: plain.len(), crc } +} + +fn reservation(segment_size: usize) -> usize { + (segment_size + WINDOW_SIZE).saturating_mul(6).saturating_add((1 << 16) * size_of::()) +} + +fn segment_size(memory_limit: usize) -> Result { + let mut size = TARGET_SEGMENT_SIZE; + while size > MIN_SEGMENT_SIZE && reservation(size) > memory_limit { + size /= 2; + } + if reservation(size) > memory_limit { + return Err(Error::InvalidConfiguration(format!("compression memory limit must be at least {} bytes", reservation(size)))); + } + Ok(size) +} + +pub(crate) fn validate_options(options: EncodeOptions) -> Result { + let options = options.validate()?; + segment_size(options.memory_limit)?; + Ok(options) +} + +fn final_block() -> Vec { + let mut output = BitWriter::new(5); + output.write_bits(1, 3); + output.align_zero(); + output.write_aligned(&[0, 0, 0xff, 0xff]); + output.finish_aligned() +} + +pub(crate) fn compress_bytes_serial(data: &[u8], level: u8) -> (Vec, EncodeReport) { + let mut output = Vec::with_capacity(data.len() / 2); + let mut crc = Hasher::new(); + let mut previous = Vec::new(); + for chunk in data.chunks(TARGET_SEGMENT_SIZE) { + let mut buffer = Vec::with_capacity(previous.len() + chunk.len()); + buffer.extend_from_slice(&previous); + buffer.extend_from_slice(chunk); + let segment = encode_segment(buffer, previous.len(), level); + output.extend_from_slice(&segment.encoded); + crc.combine(&segment.crc); + let keep = chunk.len().min(WINDOW_SIZE); + if chunk.len() >= WINDOW_SIZE { + previous.clear(); + previous.extend_from_slice(&chunk[chunk.len() - keep..]); + } else { + previous.extend_from_slice(chunk); + if previous.len() > WINDOW_SIZE { + let discard = previous.len() - WINDOW_SIZE; + previous.copy_within(discard.., 0); + previous.truncate(WINDOW_SIZE); + } + } + } + output.extend_from_slice(&final_block()); + let report = EncodeReport { input_len: data.len() as u64, output_len: output.len() as u64, crc: crc.finalize() }; + (output, report) +} + +pub struct Encoder { + output: Option, + options: EncodeOptions, + pipeline: StreamingOrdered>, + buffer: Vec, + prefix_len: usize, + segment_size: usize, + reservation: usize, + crc: Hasher, + input_len: u64, + output_len: u64, +} + +impl Encoder { + pub fn new(output: W, options: EncodeOptions) -> Result { + let options = validate_options(options)?; + let options = EncodeOptions { level: Some(options.level_or(6)), ..options }; + let segment_size = segment_size(options.memory_limit)?; + let reservation = reservation(segment_size); + let workers = options.resolved_threads().min((options.memory_limit / reservation).max(1)); + Ok(Self { + output: Some(output), + options, + pipeline: StreamingOrdered::new(workers, options.memory_limit, "fbz-deflate")?, + buffer: Vec::with_capacity(segment_size + WINDOW_SIZE), + prefix_len: 0, + segment_size, + reservation, + crc: Hasher::new(), + input_len: 0, + output_len: 0, + }) + } + + fn commit_next(&mut self) -> Result<()> { + let segment = self.pipeline.take_next()??; + self.output.as_mut().unwrap().write_all(&segment.encoded)?; + self.crc.combine(&segment.crc); + self.input_len += segment.input_len as u64; + self.output_len += segment.encoded.len() as u64; + Ok(()) + } + + fn submit_buffer(&mut self) -> Result<()> { + if self.buffer.len() == self.prefix_len { + return Ok(()); + } + while !self.pipeline.can_submit(self.reservation) { + self.commit_next()?; + } + let mut next = Vec::with_capacity(self.segment_size + WINDOW_SIZE); + let keep = self.buffer.len().min(WINDOW_SIZE); + next.extend_from_slice(&self.buffer[self.buffer.len() - keep..]); + let buffer = std::mem::replace(&mut self.buffer, next); + let prefix_len = self.prefix_len; + let level = self.options.level.unwrap(); + self.prefix_len = keep; + self.pipeline.submit(self.reservation, move || Ok(encode_segment(buffer, prefix_len, level))) + } + + fn flush_segments(&mut self) -> Result<()> { + self.submit_buffer()?; + while self.pipeline.has_pending() { + self.commit_next()?; + } + Ok(()) + } + + pub fn finish(mut self) -> Result<(W, EncodeReport)> { + self.flush_segments()?; + let final_block = final_block(); + let output = self.output.as_mut().unwrap(); + output.write_all(&final_block)?; + output.flush()?; + self.output_len += final_block.len() as u64; + let report = EncodeReport { input_len: self.input_len, output_len: self.output_len, crc: self.crc.finalize() }; + Ok((self.output.take().unwrap(), report)) + } +} + +impl Write for Encoder { + fn write(&mut self, mut bytes: &[u8]) -> io::Result { + let total = bytes.len(); + while !bytes.is_empty() { + let used = self.buffer.len() - self.prefix_len; + let take = bytes.len().min(self.segment_size - used); + self.buffer.extend_from_slice(&bytes[..take]); + bytes = &bytes[take..]; + if self.buffer.len() - self.prefix_len == self.segment_size { + self.submit_buffer().map_err(Error::into_io)?; + } + } + Ok(total) + } + + fn flush(&mut self) -> io::Result<()> { + self.flush_segments().map_err(Error::into_io)?; + self.output.as_mut().unwrap().flush() + } +} + +pub fn compress_to_writer(input: &mut impl Read, output: &mut impl Write, options: EncodeOptions) -> Result { + let mut encoder = Encoder::new(output, options)?; + io::copy(input, &mut encoder)?; + let (_, report) = encoder.finish()?; + Ok(report) +} diff --git a/src/encode.rs b/src/encode.rs new file mode 100644 index 0000000..dd509d4 --- /dev/null +++ b/src/encode.rs @@ -0,0 +1,165 @@ +use std::{ + cell::Cell, + io::{self, Read, Write}, + rc::Rc, + thread, +}; + +use crate::{Bzip2Encoder, Error, Result, decode::DEFAULT_MEMORY_LIMIT, gzip, lz4}; + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum EncodeFormat { + Bzip2, + Gzip, + Lz4, +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct EncodeReport { + pub format: EncodeFormat, + pub input_len: u64, + pub output_len: u64, +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct EncodeProgress { + pub input_bytes: u64, + pub output_bytes: u64, +} + +#[derive(Clone, Copy, Debug)] +pub struct EncodeOptions { + /// Zero selects the process's available parallelism. + pub threads: usize, + /// Maximum bytes reserved for in-flight input, working state, and encoded output. + pub memory_limit: usize, + /// Format-specific compression level from 1 through 9, or the codec default. + pub level: Option, +} + +impl Default for EncodeOptions { + fn default() -> Self { + Self { threads: 0, memory_limit: DEFAULT_MEMORY_LIMIT, level: None } + } +} + +impl EncodeOptions { + pub fn resolved_threads(self) -> usize { + if self.threads != 0 { self.threads } else { thread::available_parallelism().map(usize::from).unwrap_or(1) } + } + + pub(crate) fn validate(self) -> Result { + if self.level.is_some_and(|level| !(1..=9).contains(&level)) { + return Err(Error::InvalidConfiguration("compression level must be between 1 and 9".into())); + } + if self.memory_limit == 0 { + return Err(Error::InvalidConfiguration("compression memory limit must be greater than zero".into())); + } + Ok(self) + } + + pub(crate) fn level_or(self, default: u8) -> u8 { + self.level.unwrap_or(default) + } +} + +pub enum Encoder { + Bzip2(Bzip2Encoder), + Gzip(gzip::Encoder), + Lz4(lz4::Encoder), +} + +impl Encoder { + pub fn new(output: W, format: EncodeFormat, options: EncodeOptions) -> Result { + match format { + EncodeFormat::Bzip2 => Bzip2Encoder::new(output, options).map(Self::Bzip2), + EncodeFormat::Gzip => gzip::Encoder::new(output, options).map(Self::Gzip), + EncodeFormat::Lz4 => lz4::Encoder::new(output, options).map(Self::Lz4), + } + } + + pub fn finish(self) -> Result<(W, EncodeReport)> { + match self { + Self::Bzip2(encoder) => encoder + .finish() + .map(|(output, report)| (output, EncodeReport { format: EncodeFormat::Bzip2, input_len: report.input_len, output_len: report.output_len })), + Self::Gzip(encoder) => encoder + .finish() + .map(|(output, report)| (output, EncodeReport { format: EncodeFormat::Gzip, input_len: report.input_len, output_len: report.output_len })), + Self::Lz4(encoder) => encoder + .finish() + .map(|(output, report)| (output, EncodeReport { format: EncodeFormat::Lz4, input_len: report.input_len, output_len: report.output_len })), + } + } +} + +impl Write for Encoder { + fn write(&mut self, bytes: &[u8]) -> io::Result { + match self { + Self::Bzip2(encoder) => encoder.write(bytes), + Self::Gzip(encoder) => encoder.write(bytes), + Self::Lz4(encoder) => encoder.write(bytes), + } + } + + fn flush(&mut self) -> io::Result<()> { + match self { + Self::Bzip2(encoder) => encoder.flush(), + Self::Gzip(encoder) => encoder.flush(), + Self::Lz4(encoder) => encoder.flush(), + } + } +} + +pub fn compress(data: &[u8], format: EncodeFormat, options: EncodeOptions) -> Result> { + let mut output = Vec::new(); + compress_to_writer(&mut io::Cursor::new(data), &mut output, format, options)?; + Ok(output) +} + +pub fn compress_to_writer(input: &mut impl Read, output: &mut impl Write, format: EncodeFormat, options: EncodeOptions) -> Result { + compress_to_writer_with_progress(input, output, format, options, |_| {}) +} + +struct CountingWriter<'a, W> { + inner: &'a mut W, + written: Rc>, +} + +impl Write for CountingWriter<'_, W> { + fn write(&mut self, bytes: &[u8]) -> io::Result { + let written = self.inner.write(bytes)?; + self.written.set(self.written.get() + written as u64); + Ok(written) + } + + fn flush(&mut self) -> io::Result<()> { + self.inner.flush() + } +} + +pub fn compress_to_writer_with_progress( + input: &mut impl Read, + output: &mut impl Write, + format: EncodeFormat, + options: EncodeOptions, + mut progress: impl FnMut(EncodeProgress), +) -> Result { + let written = Rc::new(Cell::new(0)); + let mut counted = CountingWriter { inner: output, written: Rc::clone(&written) }; + let mut encoder = Encoder::new(&mut counted, format, options)?; + let mut buffer = vec![0_u8; 128 * 1024]; + let mut input_bytes = 0_u64; + loop { + let read = input.read(&mut buffer)?; + if read == 0 { + break; + } + encoder.write_all(&buffer[..read])?; + input_bytes += read as u64; + progress(EncodeProgress { input_bytes, output_bytes: written.get() }); + } + let (_, report) = encoder.finish()?; + progress(EncodeProgress { input_bytes: report.input_len, output_bytes: report.output_len }); + Ok(report) +} diff --git a/src/error.rs b/src/error.rs index 2704fee..2d574d1 100644 --- a/src/error.rs +++ b/src/error.rs @@ -85,3 +85,12 @@ impl From for Error { Self::Io(source) } } + +impl Error { + pub(crate) fn into_io(self) -> io::Error { + match self { + Self::Io(error) => error, + error => io::Error::other(error), + } + } +} diff --git a/src/gzip.rs b/src/gzip.rs index 0ad22f9..80d7cfc 100644 --- a/src/gzip.rs +++ b/src/gzip.rs @@ -1,9 +1,11 @@ -//! Gzip framing and DEFLATE decompression implemented in safe Rust. +//! Gzip framing and DEFLATE compression/decompression implemented in safe Rust. use std::{io::Write, sync::OnceLock}; use crate::{ - DecodeOptions, DecodeProgress, Error, OutputSink, Result, WriterSink, + DecodeOptions, DecodeProgress, EncodeOptions, Error, OutputSink, Result, WriterSink, + deflate::{DISTANCE_BASE, DISTANCE_EXTRA, LENGTH_BASE, LENGTH_EXTRA}, + deflate_encode, history::extend_match, pipeline::{Job, PipelineLimits, run_staged_ordered}, }; @@ -17,12 +19,6 @@ const MIN_PARALLEL_INPUT: usize = 16 * 1024 * 1024; const PARALLEL_OUTPUT_LIMIT: usize = 8 * 1024 * 1024; const PARALLEL_JOB_MEMORY: usize = 2 * PARALLEL_OUTPUT_LIMIT + 64 * 1024; -const LENGTH_BASE: [usize; 29] = [3, 4, 5, 6, 7, 8, 9, 10, 11, 13, 15, 17, 19, 23, 27, 31, 35, 43, 51, 59, 67, 83, 99, 115, 131, 163, 195, 227, 258]; -const LENGTH_EXTRA: [u8; 29] = [0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 2, 2, 2, 2, 3, 3, 3, 3, 4, 4, 4, 4, 5, 5, 5, 5, 0]; -const DISTANCE_BASE: [usize; 30] = - [1, 2, 3, 4, 5, 7, 9, 13, 17, 25, 33, 49, 65, 97, 129, 193, 257, 385, 513, 769, 1025, 1537, 2049, 3073, 4097, 6145, 8193, 12289, 16385, 24577]; -const DISTANCE_EXTRA: [u8; 30] = [0, 0, 0, 0, 1, 1, 2, 2, 3, 3, 4, 4, 5, 5, 6, 6, 7, 7, 8, 8, 9, 9, 10, 10, 11, 11, 12, 12, 13, 13]; - #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub enum BlockKind { Stored, @@ -87,6 +83,65 @@ pub struct DeflateReport { pub fallback_chunks: u64, } +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct EncodeReport { + pub input_len: u64, + pub output_len: u64, + pub crc: u32, +} + +pub struct Encoder { + inner: deflate_encode::Encoder, +} + +impl Encoder { + pub fn new(mut output: W, options: EncodeOptions) -> Result { + let options = deflate_encode::validate_options(options)?; + let extra_flags = match options.level_or(6) { + 1..=2 => 4, + 8..=9 => 2, + _ => 0, + }; + output.write_all(&[0x1f, 0x8b, 8, 0, 0, 0, 0, 0, extra_flags, 255])?; + Ok(Self { inner: deflate_encode::Encoder::new(output, options)? }) + } + + pub fn finish(self) -> Result<(W, EncodeReport)> { + let (mut output, deflate) = self.inner.finish()?; + output.write_all(&deflate.crc.to_le_bytes())?; + output.write_all(&(deflate.input_len as u32).to_le_bytes())?; + output.flush()?; + Ok((output, EncodeReport { input_len: deflate.input_len, output_len: deflate.output_len + 18, crc: deflate.crc })) + } +} + +impl Write for Encoder { + fn write(&mut self, bytes: &[u8]) -> std::io::Result { + self.inner.write(bytes) + } + + fn flush(&mut self) -> std::io::Result<()> { + self.inner.flush() + } +} + +pub fn compress(data: &[u8]) -> Result> { + compress_with_options(data, EncodeOptions::default()) +} + +pub fn compress_with_options(data: &[u8], options: EncodeOptions) -> Result> { + let mut output = Vec::new(); + compress_to_writer(&mut std::io::Cursor::new(data), &mut output, options)?; + Ok(output) +} + +pub fn compress_to_writer(input: &mut impl std::io::Read, output: &mut impl Write, options: EncodeOptions) -> Result { + let mut encoder = Encoder::new(output, options)?; + std::io::copy(input, &mut encoder)?; + let (_, report) = encoder.finish()?; + Ok(report) +} + #[derive(Clone, Debug)] struct Header { deflate_start: usize, diff --git a/src/lib.rs b/src/lib.rs index d01737f..f852562 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -1,11 +1,14 @@ -//! Portable bzip2 primitives and format inspection. +//! Fast parallel compression and decompression for bzip2, gzip, LZ4, and ZIP. mod bitreader; mod block; +mod bz2_encode; mod crc; mod decode; mod decoder; pub mod deflate; +mod deflate_encode; +mod encode; mod error; mod format; pub mod gzip; @@ -13,19 +16,24 @@ mod history; mod index; mod indexed; pub mod lz4; +mod lz4_encode; +mod matchfinder; mod output; mod pipeline; mod reader; mod source; mod stream; +pub mod zip; pub use bitreader::BitReader; pub use block::{MAX_DECODED_BLOCK, MAX_ENCODED_BLOCK, decode_block}; +pub use bz2_encode::{EncodeReport as Bzip2EncodeReport, Encoder as Bzip2Encoder, compress as compress_bzip2, compress_to_writer as compress_bzip2_to_writer}; pub use crc::{bz2_crc32, combine_stream_crc}; pub use decode::{ DEFAULT_MEMORY_LIMIT, DecodeOptions, DecodeProgress, build_index, build_index_with_progress, decode_to_writer, decode_to_writer_with_progress, decompress, decompress_to_sink_with_progress, decompress_to_writer, decompress_to_writer_with_progress, }; +pub use encode::{EncodeFormat, EncodeOptions, EncodeProgress, EncodeReport, Encoder, compress, compress_to_writer, compress_to_writer_with_progress}; pub use error::{DecodeError, Error, Result}; pub use format::{BLOCK_MAGIC, BlockCandidate, END_MAGIC, EndCandidate, ScanResult, StreamHeaderCandidate, scan}; pub use index::{BlockIndex, Index, StreamIndex}; @@ -81,6 +89,19 @@ mod python { Ok(PyBytes::new(py, &output).unbind()) } + #[pyfunction(name = "_compress", signature = (data, format, threads=0, memory_limit=crate::DEFAULT_MEMORY_LIMIT, level=None))] + fn py_compress(py: Python<'_>, data: &[u8], format: &str, threads: usize, memory_limit: usize, level: Option) -> PyResult> { + let format = match format { + "bzip2" => crate::EncodeFormat::Bzip2, + "gzip" => crate::EncodeFormat::Gzip, + "lz4" => crate::EncodeFormat::Lz4, + _ => return Err(PyValueError::new_err("format must be 'bzip2', 'gzip', or 'lz4'")), + }; + let data = data.to_vec(); + let output = py.detach(move || crate::compress(&data, format, crate::EncodeOptions { threads, memory_limit, level })).map_err(python_error)?; + Ok(PyBytes::new(py, &output).unbind()) + } + #[pyfunction(name = "_build_index", signature = (path, threads=0, memory_limit=crate::DEFAULT_MEMORY_LIMIT))] fn py_build_index(py: Python<'_>, path: String, threads: usize, memory_limit: usize) -> PyResult> { let encoded = py @@ -188,6 +209,7 @@ mod python { m.add_function(wrap_pyfunction!(py_scan, m)?)?; m.add_function(wrap_pyfunction!(py_bz2_crc32, m)?)?; m.add_function(wrap_pyfunction!(py_decompress, m)?)?; + m.add_function(wrap_pyfunction!(py_compress, m)?)?; m.add_function(wrap_pyfunction!(py_build_index, m)?)?; m.add_function(wrap_pyfunction!(py_test, m)?)?; m.add_class::()?; diff --git a/src/lz4.rs b/src/lz4.rs index 0d6ae18..e33ca6a 100644 --- a/src/lz4.rs +++ b/src/lz4.rs @@ -1,4 +1,4 @@ -//! LZ4 frame and block decompression implemented in safe Rust. +//! LZ4 frame and block compression/decompression implemented in safe Rust. use std::{hash::Hasher, io::Write}; @@ -9,6 +9,8 @@ use crate::history::extend_match; use crate::pipeline::{Job, PipelineLimits, run_ordered}; use crate::{DecodeOptions, DecodeProgress, Error, OutputSink, Result, WriterSink}; +pub use crate::lz4_encode::{EncodeReport, Encoder, compress, compress_to_writer}; + const FRAME_MAGIC: u32 = 0x184d_2204; const LEGACY_MAGIC: u32 = 0x184c_2102; const SKIPPABLE_MAGIC_START: u32 = 0x184d_2a50; diff --git a/src/lz4_encode.rs b/src/lz4_encode.rs new file mode 100644 index 0000000..535e64e --- /dev/null +++ b/src/lz4_encode.rs @@ -0,0 +1,298 @@ +use std::{ + hash::Hasher as _, + io::{self, Read, Write}, +}; + +use twox_hash::XxHash32; + +use crate::{ + EncodeOptions, Error, Result, + matchfinder::{HashChain, LatestMatch, match_length}, + pipeline::StreamingOrdered, +}; + +const MAGIC: u32 = 0x184d_2204; +const WINDOW_SIZE: usize = u16::MAX as usize; +const TARGET_BLOCK_SIZE: usize = 4 * 1024 * 1024; +const MIN_BLOCK_SIZE: usize = 64 * 1024; + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct EncodeReport { + pub input_len: u64, + pub output_len: u64, + pub blocks: u64, +} + +fn invalid(message: impl Into) -> Error { + Error::InvalidConfiguration(message.into()) +} + +fn xxhash32(data: &[u8]) -> u32 { + let mut hasher = XxHash32::with_seed(0); + hasher.write(data); + hasher.finish() as u32 +} + +fn chain_depth(level: u8) -> usize { + match level { + 1..=6 => 1, + 7 => 4, + 8 => 16, + 9 => 64, + _ => unreachable!(), + } +} + +fn write_length(output: &mut Vec, mut length: usize) { + while length >= 255 { + output.push(255); + length -= 255; + } + output.push(length as u8); +} + +fn write_sequence(output: &mut Vec, literals: &[u8], distance: usize, match_length: usize) { + let literal_nibble = literals.len().min(15); + let match_base = match_length - 4; + let match_nibble = match_base.min(15); + output.push(((literal_nibble << 4) | match_nibble) as u8); + if literals.len() >= 15 { + write_length(output, literals.len() - 15); + } + output.extend_from_slice(literals); + output.extend_from_slice(&(distance as u16).to_le_bytes()); + if match_base >= 15 { + write_length(output, match_base - 15); + } +} + +fn write_last_literals(output: &mut Vec, literals: &[u8]) { + let literal_nibble = literals.len().min(15); + output.push((literal_nibble << 4) as u8); + if literals.len() >= 15 { + write_length(output, literals.len() - 15); + } + output.extend_from_slice(literals); +} + +fn compress_block_fast(input: &[u8]) -> Vec { + let mut output = Vec::with_capacity(input.len() / 2); + let mut finder = LatestMatch::new(WINDOW_SIZE); + let mut anchor = 0; + let mut position = 0; + while position + 12 <= input.len() { + let max_length = input.len() - position - 5; + let Some(mut candidate) = finder.insert_and_find(input, position) else { + position += 1 + ((position - anchor) >> 6); + continue; + }; + let distance = position - candidate; + let mut length = match_length(input, position, candidate, max_length); + while position > anchor && candidate > 0 && input[position - 1] == input[candidate - 1] { + position -= 1; + candidate -= 1; + length += 1; + } + write_sequence(&mut output, &input[anchor..position], distance, length); + let end = position + length; + if end >= 2 { + finder.insert(input, end - 2); + } + position = end; + anchor = end; + } + write_last_literals(&mut output, &input[anchor..]); + output +} + +fn compress_block_high(input: &[u8], level: u8) -> Vec { + let mut output = Vec::with_capacity(input.len() / 2); + let max_chain = chain_depth(level); + let mut finder = HashChain::new(input.len(), WINDOW_SIZE, max_chain); + let mut anchor = 0; + let mut position = 0; + while position + 12 <= input.len() { + let max_length = input.len() - position - 5; + let (mut length, distance) = finder.best_match(input, position, max_length, 4, max_chain); + if length == 0 { + finder.insert(input, position); + position += 1; + continue; + } + let mut candidate = position - distance; + while position > anchor && candidate > 0 && input[position - 1] == input[candidate - 1] { + position -= 1; + candidate -= 1; + length += 1; + } + write_sequence(&mut output, &input[anchor..position], distance, length); + let end = position + length; + if end >= 2 { + finder.insert(input, end - 2); + } + position = end; + anchor = end; + } + write_last_literals(&mut output, &input[anchor..]); + output +} + +fn compress_block(input: &[u8], level: u8) -> Vec { + if level <= 6 { compress_block_fast(input) } else { compress_block_high(input, level) } +} + +struct Block { + bytes: Vec, + input_len: usize, + stored: bool, +} + +fn encode_block(input: Vec, level: u8) -> Block { + let input_len = input.len(); + let encoded = compress_block(&input, level); + if encoded.len() < input.len() { Block { bytes: encoded, input_len, stored: false } } else { Block { bytes: input, input_len, stored: true } } +} + +fn reservation(block_size: usize) -> usize { + block_size.saturating_mul(6).saturating_add((1 << 16) * size_of::()) +} + +fn selected_block_size(memory_limit: usize) -> Result { + let mut size = TARGET_BLOCK_SIZE; + while size > MIN_BLOCK_SIZE && reservation(size) > memory_limit { + size /= 4; + } + if reservation(size) > memory_limit { + return Err(invalid(format!("compression memory limit must be at least {} bytes", reservation(size)))); + } + Ok(size) +} + +fn descriptor_code(block_size: usize) -> u8 { + match block_size { + 64_000..=65_536 => 4, + 65_537..=262_144 => 5, + 262_145..=1_048_576 => 6, + _ => 7, + } +} + +pub struct Encoder { + output: Option, + pipeline: StreamingOrdered, + options: EncodeOptions, + buffer: Vec, + block_size: usize, + reservation: usize, + content_hash: XxHash32, + input_len: u64, + output_len: u64, + blocks: u64, +} + +impl Encoder { + pub fn new(mut output: W, options: EncodeOptions) -> Result { + let options = options.validate()?; + let options = EncodeOptions { level: Some(options.level_or(1)), ..options }; + let block_size = selected_block_size(options.memory_limit)?; + let reservation = reservation(block_size); + let workers = options.resolved_threads().min((options.memory_limit / reservation).max(1)); + let flg = 0x64_u8; + let bd = descriptor_code(block_size) << 4; + let checksum = (xxhash32(&[flg, bd]) >> 8) as u8; + output.write_all(&MAGIC.to_le_bytes())?; + output.write_all(&[flg, bd, checksum])?; + Ok(Self { + output: Some(output), + pipeline: StreamingOrdered::new(workers, options.memory_limit, "fbz-lz4-encode")?, + options, + buffer: Vec::with_capacity(block_size), + block_size, + reservation, + content_hash: XxHash32::with_seed(0), + input_len: 0, + output_len: 7, + blocks: 0, + }) + } + + fn commit_next(&mut self) -> Result<()> { + let block = self.pipeline.take_next()?; + let mut size = block.bytes.len() as u32; + if block.stored { + size |= 1 << 31; + } + let output = self.output.as_mut().unwrap(); + output.write_all(&size.to_le_bytes())?; + output.write_all(&block.bytes)?; + self.input_len += block.input_len as u64; + self.output_len += 4 + block.bytes.len() as u64; + self.blocks += 1; + Ok(()) + } + + fn submit_buffer(&mut self) -> Result<()> { + if self.buffer.is_empty() { + return Ok(()); + } + while !self.pipeline.can_submit(self.reservation) { + self.commit_next()?; + } + let input = std::mem::replace(&mut self.buffer, Vec::with_capacity(self.block_size)); + let level = self.options.level.unwrap(); + self.pipeline.submit(self.reservation, move || encode_block(input, level)) + } + + fn flush_blocks(&mut self) -> Result<()> { + self.submit_buffer()?; + while self.pipeline.has_pending() { + self.commit_next()?; + } + Ok(()) + } + + pub fn finish(mut self) -> Result<(W, EncodeReport)> { + self.flush_blocks()?; + let output = self.output.as_mut().unwrap(); + output.write_all(&0_u32.to_le_bytes())?; + output.write_all(&(self.content_hash.finish() as u32).to_le_bytes())?; + output.flush()?; + self.output_len += 8; + let report = EncodeReport { input_len: self.input_len, output_len: self.output_len, blocks: self.blocks }; + Ok((self.output.take().unwrap(), report)) + } +} + +impl Write for Encoder { + fn write(&mut self, mut bytes: &[u8]) -> io::Result { + let total = bytes.len(); + self.content_hash.write(bytes); + while !bytes.is_empty() { + let take = bytes.len().min(self.block_size - self.buffer.len()); + self.buffer.extend_from_slice(&bytes[..take]); + bytes = &bytes[take..]; + if self.buffer.len() == self.block_size { + self.submit_buffer().map_err(Error::into_io)?; + } + } + Ok(total) + } + + fn flush(&mut self) -> io::Result<()> { + self.flush_blocks().map_err(Error::into_io)?; + self.output.as_mut().unwrap().flush() + } +} + +pub fn compress(data: &[u8], options: EncodeOptions) -> Result> { + let mut output = Vec::new(); + compress_to_writer(&mut io::Cursor::new(data), &mut output, options)?; + Ok(output) +} + +pub fn compress_to_writer(input: &mut impl Read, output: &mut impl Write, options: EncodeOptions) -> Result { + let mut encoder = Encoder::new(output, options)?; + io::copy(input, &mut encoder)?; + let (_, report) = encoder.finish()?; + Ok(report) +} diff --git a/src/matchfinder.rs b/src/matchfinder.rs new file mode 100644 index 0000000..7505436 --- /dev/null +++ b/src/matchfinder.rs @@ -0,0 +1,117 @@ +const HASH_SIZE: usize = 1 << 16; +const NONE: u32 = u32::MAX; + +#[inline] +pub(crate) fn match_length(bytes: &[u8], left: usize, right: usize, limit: usize) -> usize { + let mut length = 0; + while length + 8 <= limit { + let a = u64::from_le_bytes(bytes[left + length..left + length + 8].try_into().unwrap()); + let b = u64::from_le_bytes(bytes[right + length..right + length + 8].try_into().unwrap()); + let different = a ^ b; + if different != 0 { + return length + (different.trailing_zeros() as usize / 8); + } + length += 8; + } + while length < limit && bytes[left + length] == bytes[right + length] { + length += 1; + } + length +} + +pub(crate) struct LatestMatch { + head: Vec, + window: usize, +} + +impl LatestMatch { + pub fn new(window: usize) -> Self { + Self { head: vec![NONE; HASH_SIZE], window } + } + + #[inline] + fn hash(bytes: &[u8], position: usize) -> usize { + let value = u32::from_le_bytes(bytes[position..position + 4].try_into().unwrap()); + ((value.wrapping_mul(0x9e37_79b1)) >> 16) as usize + } + + #[inline] + pub fn insert_and_find(&mut self, bytes: &[u8], position: usize) -> Option { + if position + 4 > bytes.len() { + return None; + } + let slot = Self::hash(bytes, position); + let candidate = self.head[slot]; + self.head[slot] = position as u32; + let candidate = (candidate != NONE).then_some(candidate as usize)?; + (position - candidate <= self.window && bytes[candidate..candidate + 4] == bytes[position..position + 4]).then_some(candidate) + } + + #[inline] + pub fn insert(&mut self, bytes: &[u8], position: usize) { + if position + 4 <= bytes.len() { + self.head[Self::hash(bytes, position)] = position as u32; + } + } +} + +pub(crate) struct HashChain { + head: Vec, + previous: Option>, + window: usize, +} + +impl HashChain { + pub fn new(input_len: usize, window: usize, max_chain: usize) -> Self { + assert!(input_len < u32::MAX as usize); + Self { head: vec![NONE; HASH_SIZE], previous: (max_chain > 1).then(|| vec![NONE; input_len]), window } + } + + fn hash(bytes: &[u8], position: usize) -> usize { + let value = u32::from(bytes[position]) << 16 | u32::from(bytes[position + 1]) << 8 | u32::from(bytes[position + 2]); + ((value.wrapping_mul(0x1e35_a7bd)) >> 16) as usize + } + + pub fn insert(&mut self, bytes: &[u8], position: usize) { + if position + 2 >= bytes.len() { + return; + } + let slot = Self::hash(bytes, position); + if let Some(previous) = &mut self.previous { + previous[position] = self.head[slot]; + } + self.head[slot] = position as u32; + } + + pub fn best_match(&self, bytes: &[u8], position: usize, max_length: usize, min_length: usize, max_chain: usize) -> (usize, usize) { + if position + min_length > bytes.len() || min_length < 3 { + return (0, 0); + } + let mut candidate = self.head[Self::hash(bytes, position)]; + let minimum = position.saturating_sub(self.window); + let limit = (bytes.len() - position).min(max_length); + let mut best_length = min_length - 1; + let mut best_distance = 0; + let mut searched = 0; + while candidate != NONE && candidate as usize >= minimum && searched < max_chain { + let candidate_position = candidate as usize; + let distance = position - candidate_position; + if bytes[candidate_position] == bytes[position] && bytes.get(candidate_position + best_length) == bytes.get(position + best_length) { + let length = match_length(bytes, candidate_position, position, limit); + if length > best_length { + best_length = length; + best_distance = distance; + if length == limit { + break; + } + } + } + searched += 1; + if searched >= max_chain { + break; + } + candidate = self.previous.as_ref().map_or(NONE, |previous| previous[candidate_position]); + } + if best_length >= min_length { (best_length, best_distance) } else { (0, 0) } + } +} diff --git a/src/pipeline.rs b/src/pipeline.rs index fd30622..606ca5d 100644 --- a/src/pipeline.rs +++ b/src/pipeline.rs @@ -1,10 +1,15 @@ use std::{ collections::HashMap, - sync::{Condvar, Mutex, mpsc}, + panic::{AssertUnwindSafe, catch_unwind}, + sync::{ + Arc, Condvar, Mutex, + atomic::{AtomicBool, Ordering}, + mpsc, + }, thread, }; -use rayon::ThreadPool; +use rayon::{ThreadPool, ThreadPoolBuilder}; use crate::{Error, Result}; @@ -308,3 +313,101 @@ where result }) } + +type StreamingMessage = (usize, usize, std::thread::Result); + +/// A bounded, long-lived ordered worker pool for streaming producers. +/// +/// Reservations cover both a job's owned input and its retained result until +/// the coordinator consumes that result. This is deliberately conservative: +/// codecs can account for temporary worker allocations in the reservation too. +pub(crate) struct StreamingOrdered { + pool: ThreadPool, + sender: mpsc::Sender>, + receiver: mpsc::Receiver>, + ready: HashMap)>, + next_submit: usize, + next_take: usize, + reserved: usize, + active: usize, + max_active: usize, + memory_limit: usize, + cancelled: Arc, +} + +impl StreamingOrdered { + pub fn new(threads: usize, memory_limit: usize, thread_name: &'static str) -> Result { + let max_active = threads.max(1); + let pool = ThreadPoolBuilder::new() + .num_threads(max_active) + .thread_name(move |index| format!("{thread_name}-{index}")) + .build() + .map_err(|error| Error::InvalidConfiguration(error.to_string()))?; + let (sender, receiver) = mpsc::channel(); + Ok(Self { + pool, + sender, + receiver, + ready: HashMap::new(), + next_submit: 0, + next_take: 0, + reserved: 0, + active: 0, + max_active, + memory_limit, + cancelled: Arc::new(AtomicBool::new(false)), + }) + } + + pub fn can_submit(&self, reservation: usize) -> bool { + self.active < self.max_active && reservation <= self.memory_limit.saturating_sub(self.reserved) + } + + pub fn submit(&mut self, reservation: usize, job: impl FnOnce() -> T + Send + 'static) -> Result<()> { + if reservation > self.memory_limit { + return Err(Error::InvalidConfiguration("a streaming job reservation exceeds the memory limit".into())); + } + if !self.can_submit(reservation) { + return Err(Error::InvalidConfiguration("streaming pipeline is full".into())); + } + let key = self.next_submit; + self.next_submit += 1; + self.active += 1; + self.reserved += reservation; + let sender = self.sender.clone(); + let cancelled = Arc::clone(&self.cancelled); + self.pool.spawn(move || { + if cancelled.load(Ordering::Relaxed) { + return; + } + let result = catch_unwind(AssertUnwindSafe(job)); + let _ = sender.send((key, reservation, result)); + }); + Ok(()) + } + + pub fn take_next(&mut self) -> Result { + if self.next_take >= self.next_submit { + return Err(Error::InvalidConfiguration("streaming pipeline has no pending result".into())); + } + while !self.ready.contains_key(&self.next_take) { + let message = self.receiver.recv().map_err(|_| Error::InvalidConfiguration("streaming worker stopped early".into()))?; + self.ready.insert(message.0, (message.1, message.2)); + } + let (reservation, result) = self.ready.remove(&self.next_take).unwrap(); + self.next_take += 1; + self.active -= 1; + self.reserved -= reservation; + result.map_err(|_| Error::InvalidConfiguration("streaming worker panicked".into())) + } + + pub fn has_pending(&self) -> bool { + self.next_take < self.next_submit + } +} + +impl Drop for StreamingOrdered { + fn drop(&mut self) { + self.cancelled.store(true, Ordering::Relaxed); + } +} diff --git a/src/zip.rs b/src/zip.rs new file mode 100644 index 0000000..a85c6b3 --- /dev/null +++ b/src/zip.rs @@ -0,0 +1,453 @@ +//! ZIP archive creation using fbz's raw-DEFLATE encoder. + +use std::{ + fs, + io::{self, Write}, + path::{Path, PathBuf}, + time::UNIX_EPOCH, +}; + +use rayon::ThreadPoolBuilder; + +use crate::{ + EncodeOptions, Error, Result, deflate, + pipeline::{Job, PipelineLimits, run_ordered}, +}; + +const UTF8_DATA_DESCRIPTOR: u16 = 0x0808; +const METHOD_STORED: u16 = 0; +const METHOD_DEFLATE: u16 = 8; +const SMALL_WORKING_MEMORY: usize = 8 * 1024 * 1024; +const SINGLE_ENTRY_PARALLEL: u64 = 16 * 1024 * 1024; +const MULTI_ENTRY_PARALLEL: u64 = 64 * 1024 * 1024; + +#[derive(Clone, Debug)] +pub struct PathInput { + pub source: PathBuf, + pub archive_path: PathBuf, +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +enum Kind { + File, + Directory, + Symlink, +} + +#[derive(Clone, Debug)] +struct Entry { + source: PathBuf, + name: String, + kind: Kind, + size: u64, + mode: u32, + modified: Option, +} + +#[derive(Clone, Debug)] +struct CentralEntry { + name: String, + method: u16, + crc: u32, + compressed_size: u64, + uncompressed_size: u64, + local_offset: u64, + mode: u32, + modified: Option, + zip64: bool, +} + +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct EncodeReport { + pub entries: usize, + pub input_len: u64, + pub output_len: u64, +} + +fn invalid(message: impl Into) -> Error { + Error::InvalidZip(message.into()) +} + +#[cfg(unix)] +fn mode(metadata: &fs::Metadata) -> u32 { + use std::os::unix::fs::{FileTypeExt, PermissionsExt}; + let kind = if metadata.file_type().is_symlink() { + 0o120000 + } else if metadata.is_dir() { + 0o040000 + } else if metadata.file_type().is_file() { + 0o100000 + } else if metadata.file_type().is_fifo() { + 0o010000 + } else { + 0 + }; + kind | metadata.permissions().mode() +} + +#[cfg(not(unix))] +fn mode(metadata: &fs::Metadata) -> u32 { + if metadata.is_dir() { 0o040755 } else { 0o100644 } +} + +fn modified(metadata: &fs::Metadata) -> Option { + metadata.modified().ok()?.duration_since(UNIX_EPOCH).ok()?.as_secs().try_into().ok() +} + +#[cfg(unix)] +fn symlink_bytes(path: &Path) -> Result> { + use std::os::unix::ffi::OsStrExt; + Ok(fs::read_link(path)?.as_os_str().as_bytes().to_vec()) +} + +#[cfg(not(unix))] +fn symlink_bytes(path: &Path) -> Result> { + Ok(fs::read_link(path)?.as_os_str().to_string_lossy().into_owned().into_bytes()) +} + +fn zip_name(path: &Path, directory: bool) -> Result { + let mut parts = Vec::new(); + for component in path.components() { + match component { + std::path::Component::Normal(part) => parts.push(part.to_str().ok_or_else(|| invalid(format!("ZIP path {} is not UTF-8", path.display())))?), + std::path::Component::CurDir => {} + _ => return Err(invalid(format!("unsafe ZIP path {}", path.display()))), + } + } + if parts.is_empty() { + return Err(invalid("empty ZIP path")); + } + let mut name = parts.join("/"); + if directory { + name.push('/'); + } + Ok(name) +} + +fn collect(source: &Path, archive_path: &Path, entries: &mut Vec) -> Result<()> { + let metadata = fs::symlink_metadata(source)?; + let kind = if metadata.file_type().is_symlink() { + Kind::Symlink + } else if metadata.is_dir() { + Kind::Directory + } else if metadata.is_file() { + Kind::File + } else { + return Err(invalid(format!("unsupported filesystem entry {}", source.display()))); + }; + let size = match kind { + Kind::File => metadata.len(), + Kind::Symlink => symlink_bytes(source)?.len() as u64, + Kind::Directory => 0, + }; + entries.push(Entry { + source: source.to_path_buf(), + name: zip_name(archive_path, kind == Kind::Directory)?, + kind, + size, + mode: mode(&metadata), + modified: modified(&metadata), + }); + if kind == Kind::Directory { + let mut children = fs::read_dir(source)?.map(|entry| entry.map(|entry| entry.path())).collect::>>()?; + children.sort_unstable(); + for child in children { + collect(&child, &archive_path.join(child.file_name().unwrap()), entries)?; + } + } + Ok(()) +} + +struct CountingWriter { + inner: W, + position: u64, +} + +impl Write for CountingWriter { + fn write(&mut self, bytes: &[u8]) -> io::Result { + let written = self.inner.write(bytes)?; + self.position += written as u64; + Ok(written) + } + + fn flush(&mut self) -> io::Result<()> { + self.inner.flush() + } +} + +fn push_u16(output: &mut impl Write, value: u16) -> io::Result<()> { + output.write_all(&value.to_le_bytes()) +} +fn push_u32(output: &mut impl Write, value: u32) -> io::Result<()> { + output.write_all(&value.to_le_bytes()) +} +fn push_u64(output: &mut impl Write, value: u64) -> io::Result<()> { + output.write_all(&value.to_le_bytes()) +} + +fn timestamp_extra(modified: Option) -> Vec { + let Some(modified) = modified else { return Vec::new() }; + let mut extra = Vec::with_capacity(9); + extra.extend_from_slice(&0x5455_u16.to_le_bytes()); + extra.extend_from_slice(&5_u16.to_le_bytes()); + extra.push(1); + extra.extend_from_slice(&modified.to_le_bytes()); + extra +} + +fn may_need_zip64(size: u64) -> bool { + size >= 0xff00_0000 +} + +fn write_local_header(output: &mut impl Write, entry: &Entry, method: u16, zip64: bool) -> Result<()> { + let name = entry.name.as_bytes(); + let mut extra = Vec::new(); + if zip64 { + extra.extend_from_slice(&1_u16.to_le_bytes()); + extra.extend_from_slice(&16_u16.to_le_bytes()); + extra.extend_from_slice(&entry.size.to_le_bytes()); + extra.extend_from_slice(&0_u64.to_le_bytes()); + } + extra.extend_from_slice(×tamp_extra(entry.modified)); + push_u32(output, 0x0403_4b50)?; + push_u16(output, if zip64 { 45 } else { 20 })?; + push_u16(output, UTF8_DATA_DESCRIPTOR)?; + push_u16(output, method)?; + push_u16(output, 0)?; + push_u16(output, 0)?; + push_u32(output, 0)?; + push_u32(output, if zip64 { u32::MAX } else { 0 })?; + push_u32(output, if zip64 { u32::MAX } else { 0 })?; + push_u16(output, name.len().try_into().map_err(|_| invalid("ZIP path exceeds 65535 bytes"))?)?; + push_u16(output, extra.len().try_into().map_err(|_| invalid("ZIP local extra data is too large"))?)?; + output.write_all(name)?; + output.write_all(&extra)?; + Ok(()) +} + +fn write_descriptor(output: &mut impl Write, crc: u32, compressed: u64, uncompressed: u64, zip64: bool) -> Result<()> { + push_u32(output, 0x0807_4b50)?; + push_u32(output, crc)?; + if zip64 { + push_u64(output, compressed)?; + push_u64(output, uncompressed)?; + } else { + push_u32(output, compressed.try_into().map_err(|_| invalid("compressed ZIP entry unexpectedly requires Zip64"))?)?; + push_u32(output, uncompressed.try_into().map_err(|_| invalid("ZIP entry unexpectedly requires Zip64"))?)?; + } + Ok(()) +} + +fn write_central(output: &mut impl Write, entry: &CentralEntry) -> Result<()> { + let name = entry.name.as_bytes(); + let size64 = entry.uncompressed_size > u32::MAX as u64; + let compressed64 = entry.compressed_size > u32::MAX as u64; + let offset64 = entry.local_offset > u32::MAX as u64; + let mut zip64_values = Vec::new(); + if size64 { + zip64_values.extend_from_slice(&entry.uncompressed_size.to_le_bytes()) + } + if compressed64 { + zip64_values.extend_from_slice(&entry.compressed_size.to_le_bytes()) + } + if offset64 { + zip64_values.extend_from_slice(&entry.local_offset.to_le_bytes()) + } + let mut extra = Vec::new(); + if !zip64_values.is_empty() { + extra.extend_from_slice(&1_u16.to_le_bytes()); + extra.extend_from_slice(&(zip64_values.len() as u16).to_le_bytes()); + extra.extend_from_slice(&zip64_values); + } + extra.extend_from_slice(×tamp_extra(entry.modified)); + let needed = if entry.zip64 || !zip64_values.is_empty() { 45 } else { 20 }; + push_u32(output, 0x0201_4b50)?; + push_u16(output, (3 << 8) | needed)?; + push_u16(output, needed)?; + push_u16(output, UTF8_DATA_DESCRIPTOR)?; + push_u16(output, entry.method)?; + push_u16(output, 0)?; + push_u16(output, 0)?; + push_u32(output, entry.crc)?; + push_u32(output, if compressed64 { u32::MAX } else { entry.compressed_size as u32 })?; + push_u32(output, if size64 { u32::MAX } else { entry.uncompressed_size as u32 })?; + push_u16(output, name.len().try_into().map_err(|_| invalid("ZIP path exceeds 65535 bytes"))?)?; + push_u16(output, extra.len().try_into().map_err(|_| invalid("ZIP central extra data is too large"))?)?; + push_u16(output, 0)?; + push_u16(output, 0)?; + push_u16(output, 0)?; + push_u32(output, entry.mode << 16)?; + push_u32(output, if offset64 { u32::MAX } else { entry.local_offset as u32 })?; + output.write_all(name)?; + output.write_all(&extra)?; + Ok(()) +} + +fn stored_bytes(entry: &Entry) -> Result> { + match entry.kind { + Kind::Directory => Ok(Vec::new()), + Kind::Symlink => symlink_bytes(&entry.source), + Kind::File => fs::read(&entry.source).map_err(Error::from), + } +} + +struct Prepared { + entry: Entry, + bytes: Vec, + method: u16, + crc: u32, + uncompressed_size: u64, +} + +fn prepare(entry: Entry, level: u8) -> Result { + let plain = stored_bytes(&entry)?; + let crc = crc32fast::hash(&plain); + if entry.kind != Kind::File { + let uncompressed_size = plain.len() as u64; + return Ok(Prepared { entry, bytes: plain, method: METHOD_STORED, crc, uncompressed_size }); + } + let (encoded, report) = deflate::compress_bytes_serial(&plain, level)?; + if encoded.len() < plain.len() { + Ok(Prepared { entry, bytes: encoded, method: METHOD_DEFLATE, crc: report.crc, uncompressed_size: report.input_len }) + } else { + let uncompressed_size = plain.len() as u64; + Ok(Prepared { entry, bytes: plain, method: METHOD_STORED, crc, uncompressed_size }) + } +} + +fn write_prepared(output: &mut CountingWriter, prepared: Prepared, central: &mut Vec) -> Result<()> { + let zip64 = may_need_zip64(prepared.uncompressed_size); + let local_offset = output.position; + write_local_header(output, &prepared.entry, prepared.method, zip64)?; + output.write_all(&prepared.bytes)?; + write_descriptor(output, prepared.crc, prepared.bytes.len() as u64, prepared.uncompressed_size, zip64)?; + central.push(CentralEntry { + name: prepared.entry.name, + method: prepared.method, + crc: prepared.crc, + compressed_size: prepared.bytes.len() as u64, + uncompressed_size: prepared.uncompressed_size, + local_offset, + mode: prepared.entry.mode, + modified: prepared.entry.modified, + zip64, + }); + Ok(()) +} + +fn write_large(output: &mut CountingWriter, entry: Entry, options: EncodeOptions, central: &mut Vec) -> Result<()> { + let zip64 = may_need_zip64(entry.size); + let local_offset = output.position; + write_local_header(output, &entry, METHOD_DEFLATE, zip64)?; + let compressed_start = output.position; + let mut source = fs::File::open(&entry.source)?; + let report = deflate::compress_to_writer(&mut source, output, options)?; + let compressed_size = output.position - compressed_start; + write_descriptor(output, report.crc, compressed_size, report.input_len, zip64)?; + central.push(CentralEntry { + name: entry.name, + method: METHOD_DEFLATE, + crc: report.crc, + compressed_size, + uncompressed_size: report.input_len, + local_offset, + mode: entry.mode, + modified: entry.modified, + zip64, + }); + Ok(()) +} + +fn finish_archive(output: &mut CountingWriter, central: &[CentralEntry]) -> Result<()> { + let central_start = output.position; + for entry in central { + write_central(output, entry)?; + } + let central_size = output.position - central_start; + let zip64 = + central.iter().any(|entry| entry.zip64) || central.len() > u16::MAX as usize || central_start > u32::MAX as u64 || central_size > u32::MAX as u64; + if zip64 { + let zip64_start = output.position; + push_u32(output, 0x0606_4b50)?; + push_u64(output, 44)?; + push_u16(output, (3 << 8) | 45)?; + push_u16(output, 45)?; + push_u32(output, 0)?; + push_u32(output, 0)?; + push_u64(output, central.len() as u64)?; + push_u64(output, central.len() as u64)?; + push_u64(output, central_size)?; + push_u64(output, central_start)?; + push_u32(output, 0x0706_4b50)?; + push_u32(output, 0)?; + push_u64(output, zip64_start)?; + push_u32(output, 1)?; + } + push_u32(output, 0x0605_4b50)?; + push_u16(output, 0)?; + push_u16(output, 0)?; + push_u16(output, central.len().min(u16::MAX as usize) as u16)?; + push_u16(output, central.len().min(u16::MAX as usize) as u16)?; + push_u32(output, central_size.min(u32::MAX as u64) as u32)?; + push_u32(output, central_start.min(u32::MAX as u64) as u32)?; + push_u16(output, 0)?; + output.flush()?; + Ok(()) +} + +pub fn create_to_writer(inputs: &[PathInput], output: &mut W, options: EncodeOptions) -> Result { + let options = options.validate()?; + let mut entries = Vec::new(); + for input in inputs { + collect(&input.source, &input.archive_path, &mut entries)?; + } + entries.sort_unstable_by(|left, right| left.name.cmp(&right.name)); + for pair in entries.windows(2) { + if pair[0].name == pair[1].name { + return Err(invalid(format!("duplicate ZIP path {}", pair[0].name))); + } + } + let input_len = entries.iter().map(|entry| entry.size).sum(); + let file_count = entries.iter().filter(|entry| entry.kind == Kind::File).count(); + let threshold = if file_count == 1 { SINGLE_ENTRY_PARALLEL } else { MULTI_ENTRY_PARALLEL }; + let small_reservation = |entry: &Entry| match entry.kind { + Kind::File => (entry.size as usize).saturating_mul(2).saturating_add(SMALL_WORKING_MEMORY), + Kind::Directory | Kind::Symlink => entry.size as usize + 1024, + }; + let (large, small): (Vec<_>, Vec<_>) = entries.into_iter().partition(|entry| { + entry.kind == Kind::File && (entry.size >= threshold || entry.size > usize::MAX as u64 || small_reservation(entry) > options.memory_limit) + }); + + let mut output = CountingWriter { inner: output, position: 0 }; + let mut central = Vec::new(); + for entry in large { + write_large(&mut output, entry, options, &mut central)?; + } + if options.resolved_threads() == 1 || small.len() <= 1 { + for entry in small { + write_prepared(&mut output, prepare(entry, options.level_or(6))?, &mut central)?; + } + } else { + let pool = ThreadPoolBuilder::new() + .num_threads(options.resolved_threads()) + .thread_name(|index| format!("fbz-zip-encode-{index}")) + .build() + .map_err(|error| invalid(error.to_string()))?; + let jobs: Vec<_> = small.into_iter().enumerate().map(|(key, entry)| Job { key, reservation: small_reservation(&entry), payload: entry }).collect(); + run_ordered( + &pool, + &jobs, + PipelineLimits { memory: options.memory_limit, active: options.resolved_threads() }, + |entry| prepare(entry.clone(), options.level_or(6)), + |result| result.as_ref().map_or(0, |prepared| prepared.bytes.capacity()), + |results| { + for key in 0..jobs.len() { + write_prepared(&mut output, results.take(key)??, &mut central)?; + } + Ok(()) + }, + )?; + } + finish_archive(&mut output, ¢ral)?; + Ok(EncodeReport { entries: central.len(), input_len, output_len: output.position }) +} diff --git a/tests/archive_perf.rs b/tests/archive_perf.rs index 43a9e7b..8c5148a 100644 --- a/tests/archive_perf.rs +++ b/tests/archive_perf.rs @@ -9,7 +9,6 @@ use crabz2::{Level, compress}; use fbz::{DecodeOptions, OutputSink, gzip as gzip_decoder}; use flate2::{Compression, write::GzEncoder}; -#[allow(dead_code)] mod support; use support::simplewiki_prefix; diff --git a/tests/bz2_encode.rs b/tests/bz2_encode.rs new file mode 100644 index 0000000..bfa45ff --- /dev/null +++ b/tests/bz2_encode.rs @@ -0,0 +1,45 @@ +mod support; + +use std::{fs, io::Write, process::Command}; + +use fbz::{Bzip2Encoder, DecodeOptions, EncodeFormat, EncodeOptions, compress, decompress}; + +#[test] +fn bzip2_encoder_roundtrips_levels_and_system_decoder() { + let inputs = support::compression_inputs(b"hello bzip2", 30_000); + for level in [1, 6, 9] { + for input in &inputs { + let options = EncodeOptions { threads: 3, memory_limit: 256 * 1024 * 1024, level: Some(level) }; + let encoded = compress(input, EncodeFormat::Bzip2, options).unwrap(); + assert_eq!(decompress(&encoded, DecodeOptions::default()).unwrap(), *input); + let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join("data.bz2"); + fs::write(&path, encoded).unwrap(); + assert!(Command::new("bzip2").args(["-t", path.to_str().unwrap()]).status().unwrap().success()); + } + } +} + +#[test] +fn bzip2_encoder_parallel_blocks_and_incremental_writes() { + let input = support::patterned_bytes_with(220_000, 17, 101); + let mut encoded = Vec::new(); + let mut encoder = Bzip2Encoder::new(&mut encoded, EncodeOptions { threads: 4, memory_limit: 64 * 1024 * 1024, level: Some(1) }).unwrap(); + for chunk in input.chunks(7777) { + encoder.write_all(chunk).unwrap(); + } + encoder.flush().unwrap(); + let (_, report) = encoder.finish().unwrap(); + assert_eq!(report.input_len, input.len() as u64); + assert!(report.blocks >= 2); + assert_eq!(decompress(&encoded, DecodeOptions::default()).unwrap(), input); +} + +#[test] +fn bzip2_encoder_default_is_level_nine_and_memory_is_bounded() { + let encoded = compress(b"default level", EncodeFormat::Bzip2, EncodeOptions::default()).unwrap(); + assert_eq!(&encoded[..4], b"BZh9"); + let error = compress(b"too little memory", EncodeFormat::Bzip2, EncodeOptions { memory_limit: 1024 * 1024, level: Some(9), ..EncodeOptions::default() }) + .unwrap_err(); + assert!(error.to_string().contains("memory limit")); +} diff --git a/tests/cli.rs b/tests/cli.rs index f2fae15..ac9aea3 100644 --- a/tests/cli.rs +++ b/tests/cli.rs @@ -11,7 +11,6 @@ use flate2::{Compression, write::GzEncoder}; use lz4_flex::frame::{BlockMode as Lz4BlockMode, BlockSize as Lz4BlockSize, FrameEncoder as Lz4Encoder, FrameInfo as Lz4FrameInfo}; use zip::{CompressionMethod as ZipCompression, ZipArchive, ZipWriter, write::FullFileOptions}; -#[allow(dead_code)] mod support; use support::{ZipMethod, zip_bytes, zip_with_modes}; @@ -85,6 +84,144 @@ fn write_tgz(path: &Path, entries: &[(&str, &[u8])]) { write_gzip(path, &tar_bytes(entries)); } +#[test] +fn gzip_compression_infers_format_and_interoperates() { + let directory = tempfile::tempdir().unwrap(); + let input = directory.path().join("sample.txt"); + let output = directory.path().join("sample.txt.gz"); + let plain = support::patterned_bytes(2_000_000); + fs::write(&input, &plain).unwrap(); + + let compressed = binary().args(["-z", input.to_str().unwrap(), "-o", output.to_str().unwrap(), "-P", "3"]).output().unwrap(); + assert!(compressed.status.success(), "{}", String::from_utf8_lossy(&compressed.stderr)); + assert!(Command::new("gzip").args(["-t", output.to_str().unwrap()]).status().unwrap().success()); + + let decoded = binary().args([output.to_str().unwrap(), "-o", "-"]).output().unwrap(); + assert!(decoded.status.success()); + assert_eq!(decoded.stdout, plain); + + let missing_format = binary().args(["-z", input.to_str().unwrap()]).output().unwrap(); + assert_eq!(missing_format.status.code(), Some(2)); +} + +#[test] +fn gzip_compression_supports_stdin_stdout_and_default_names() { + let plain = b"stream compression through the public CLI".repeat(10_000); + let mut child = binary().args(["-z", "--format", "gzip", "-", "-o", "-"]).stdin(Stdio::piped()).stdout(Stdio::piped()).spawn().unwrap(); + child.stdin.take().unwrap().write_all(&plain).unwrap(); + let compressed = child.wait_with_output().unwrap(); + assert!(compressed.status.success()); + assert_eq!(fbz::gzip::decompress(&compressed.stdout).unwrap(), plain); + + let directory = tempfile::tempdir().unwrap(); + let input = directory.path().join("named"); + fs::write(&input, &plain).unwrap(); + let result = binary().args(["-z", "--format", "gzip", input.to_str().unwrap()]).output().unwrap(); + assert!(result.status.success(), "{}", String::from_utf8_lossy(&result.stderr)); + assert!(directory.path().join("named.gz").exists()); + + let removed = directory.path().join("removed-after-stdout"); + fs::write(&removed, &plain).unwrap(); + let result = binary().args(["-z", "--format", "gzip", "--rm", removed.to_str().unwrap(), "-o", "-"]).output().unwrap(); + assert!(result.status.success()); + assert_eq!(fbz::gzip::decompress(&result.stdout).unwrap(), plain); + assert!(!removed.exists()); +} + +#[test] +fn tar_gzip_compression_streams_multiple_inputs() { + let directory = tempfile::tempdir().unwrap(); + let first = directory.path().join("first.txt"); + let tree = directory.path().join("tree"); + fs::create_dir(&tree).unwrap(); + fs::write(&first, b"first contents").unwrap(); + fs::write(tree.join("second.txt"), b"second contents").unwrap(); + let archive = directory.path().join("bundle.tar.gz"); + + let packed = binary().args(["-z", first.to_str().unwrap(), tree.to_str().unwrap(), "-o", archive.to_str().unwrap(), "-P", "3"]).output().unwrap(); + assert!(packed.status.success(), "{}", String::from_utf8_lossy(&packed.stderr)); + assert!(Command::new("gzip").args(["-t", archive.to_str().unwrap()]).status().unwrap().success()); + let skipped = binary().args(["-z", "--skip-existing", first.to_str().unwrap(), tree.to_str().unwrap(), "-o", archive.to_str().unwrap()]).output().unwrap(); + assert!(skipped.status.success()); + + let destination = directory.path().join("unpacked"); + let unpacked = binary().args([archive.to_str().unwrap(), "-C", destination.to_str().unwrap()]).output().unwrap(); + assert!(unpacked.status.success(), "{}", String::from_utf8_lossy(&unpacked.stderr)); + assert_eq!(fs::read(destination.join("first.txt")).unwrap(), b"first contents"); + assert_eq!(fs::read(destination.join("tree/second.txt")).unwrap(), b"second contents"); +} + +#[test] +fn zip_compression_reuses_fbz_deflate_and_extracts_safely() { + let directory = tempfile::tempdir().unwrap(); + let first = directory.path().join("first.txt"); + let tree = directory.path().join("tree"); + fs::create_dir(&tree).unwrap(); + fs::write(&first, b"first zip contents").unwrap(); + fs::write(tree.join("second.txt"), b"second zip contents".repeat(20_000)).unwrap(); + let archive = directory.path().join("bundle.zip"); + + let packed = binary().args(["-z", first.to_str().unwrap(), tree.to_str().unwrap(), "-o", archive.to_str().unwrap(), "-P", "3"]).output().unwrap(); + assert!(packed.status.success(), "{}", String::from_utf8_lossy(&packed.stderr)); + assert!(Command::new("unzip").args(["-t", archive.to_str().unwrap()]).status().unwrap().success()); + let skipped = binary().args(["-z", "--skip-existing", first.to_str().unwrap(), tree.to_str().unwrap(), "-o", archive.to_str().unwrap()]).output().unwrap(); + assert!(skipped.status.success()); + + let destination = directory.path().join("unzipped"); + let unpacked = binary().args([archive.to_str().unwrap(), "-C", destination.to_str().unwrap()]).output().unwrap(); + assert!(unpacked.status.success(), "{}", String::from_utf8_lossy(&unpacked.stderr)); + assert_eq!(fs::read(destination.join("first.txt")).unwrap(), b"first zip contents"); + assert_eq!(fs::read(destination.join("tree/second.txt")).unwrap(), b"second zip contents".repeat(20_000)); +} + +#[test] +fn lz4_compression_and_tar_composition_interoperate() { + let directory = tempfile::tempdir().unwrap(); + let input = directory.path().join("sample.txt"); + let frame = directory.path().join("sample.txt.lz4"); + let plain = b"LZ4 compression contents".repeat(10_000); + fs::write(&input, &plain).unwrap(); + + let compressed = binary().args(["-z", input.to_str().unwrap(), "-o", frame.to_str().unwrap(), "-P", "3"]).output().unwrap(); + assert!(compressed.status.success(), "{}", String::from_utf8_lossy(&compressed.stderr)); + assert!(Command::new("lz4").args(["-t", frame.to_str().unwrap()]).status().unwrap().success()); + let decoded = binary().args([frame.to_str().unwrap(), "-o", "-"]).output().unwrap(); + assert!(decoded.status.success()); + assert_eq!(decoded.stdout, plain); + + let archive = directory.path().join("bundle.tar.lz4"); + let packed = binary().args(["-z", input.to_str().unwrap(), "-o", archive.to_str().unwrap(), "-P", "3"]).output().unwrap(); + assert!(packed.status.success(), "{}", String::from_utf8_lossy(&packed.stderr)); + let destination = directory.path().join("unpacked-lz4"); + let unpacked = binary().args([archive.to_str().unwrap(), "-C", destination.to_str().unwrap()]).output().unwrap(); + assert!(unpacked.status.success(), "{}", String::from_utf8_lossy(&unpacked.stderr)); + assert_eq!(fs::read(destination.join("sample.txt")).unwrap(), plain); +} + +#[test] +fn bzip2_compression_and_tar_composition_interoperate() { + let directory = tempfile::tempdir().unwrap(); + let input = directory.path().join("sample.txt"); + let stream = directory.path().join("sample.txt.bz2"); + let plain = b"bzip2 compression contents".repeat(10_000); + fs::write(&input, &plain).unwrap(); + + let compressed = binary().args(["-z", input.to_str().unwrap(), "-o", stream.to_str().unwrap(), "-P", "3"]).output().unwrap(); + assert!(compressed.status.success(), "{}", String::from_utf8_lossy(&compressed.stderr)); + assert!(Command::new("bzip2").args(["-t", stream.to_str().unwrap()]).status().unwrap().success()); + let decoded = binary().args([stream.to_str().unwrap(), "-o", "-"]).output().unwrap(); + assert!(decoded.status.success()); + assert_eq!(decoded.stdout, plain); + + let archive = directory.path().join("bundle.tar.bz2"); + let packed = binary().args(["-z", input.to_str().unwrap(), "-o", archive.to_str().unwrap(), "-P", "3"]).output().unwrap(); + assert!(packed.status.success(), "{}", String::from_utf8_lossy(&packed.stderr)); + let destination = directory.path().join("unpacked-bzip2"); + let unpacked = binary().args([archive.to_str().unwrap(), "-C", destination.to_str().unwrap()]).output().unwrap(); + assert!(unpacked.status.success(), "{}", String::from_utf8_lossy(&unpacked.stderr)); + assert_eq!(fs::read(destination.join("sample.txt")).unwrap(), plain); +} + fn traversal_tar() -> Vec { let contents = b"must stay inside destination"; let mut header = tar::Header::new_gnu(); @@ -138,7 +275,7 @@ fn decode_test_index_and_list() { let input = directory.path().join("sample.bz2"); let output = directory.path().join("sample"); let index = directory.path().join("sample.fbz2i"); - let plain: Vec<_> = (0..250_000).map(|i| ((i * 31 + i / 97) & 255) as u8).collect(); + let plain = support::patterned_bytes(250_000); write_compressed(&input, &plain); let decoded = binary().args([input.to_str().unwrap(), "-P", "2"]).status().unwrap(); @@ -318,7 +455,7 @@ fn gzip_extension_selects_decoder_across_cli_modes() { let directory = tempfile::tempdir().unwrap(); let input = directory.path().join("sample.gz"); let output = directory.path().join("sample"); - let plain: Vec<_> = (0..250_000).map(|i| ((i * 31 + i / 97) & 255) as u8).collect(); + let plain = support::patterned_bytes(250_000); write_gzip(&input, &plain); let decoded = binary().arg(input.to_str().unwrap()).output().unwrap(); @@ -379,7 +516,7 @@ fn lz4_extension_magic_reporting_limits_and_corruption_work() { let directory = tempfile::tempdir().unwrap(); let input = directory.path().join("sample.lz4"); let output = directory.path().join("sample"); - let plain: Vec<_> = (0..2_000_000).map(|i| ((i * 31 + i / 97) & 255) as u8).collect(); + let plain = support::patterned_bytes(2_000_000); let encoded = lz4_bytes(&plain, Lz4BlockMode::Independent); fs::write(&input, &encoded).unwrap(); diff --git a/tests/common/process.rs b/tests/common/process.rs index d20150c..4af8996 100644 --- a/tests/common/process.rs +++ b/tests/common/process.rs @@ -106,6 +106,7 @@ pub fn measure(command: &mut Command) -> io::Result { } // SAFETY: a successful `wait4` initialized the complete `rusage` value. let usage = unsafe { usage.assume_init() }; + let wall = started.elapsed(); let duration = |value: libc::timeval| Duration::new(value.tv_sec as u64, (value.tv_usec as u32) * 1_000); #[cfg(target_os = "macos")] let peak_rss_bytes = usage.ru_maxrss as u64; @@ -115,7 +116,7 @@ pub fn measure(command: &mut Command) -> io::Result { let peak_phys_footprint_bytes = sampler.worker.join().unwrap_or(None); Ok(ProcessMetrics { status: ExitStatus::from_raw(status), - wall: started.elapsed(), + wall, user: duration(usage.ru_utime), system: duration(usage.ru_stime), peak_rss_bytes, diff --git a/tests/compression_cli_perf.rs b/tests/compression_cli_perf.rs new file mode 100644 index 0000000..03f79cc --- /dev/null +++ b/tests/compression_cli_perf.rs @@ -0,0 +1,218 @@ +#[allow(dead_code, unused_imports)] +mod common; +mod support; + +use std::{ + fs, + path::{Path, PathBuf}, + process::{Command, Stdio}, +}; + +fn input_file(directory: &tempfile::TempDir) -> std::path::PathBuf { + let path = directory.path().join("simplewiki-prefix.xml"); + fs::write(&path, support::simplewiki_prefix()).unwrap(); + path +} + +fn fbz(format: &str, input: &Path) -> Command { + fbz_with_threads(format, input, 0) +} + +fn fbz_with_threads(format: &str, input: &Path, threads: usize) -> Command { + let mut command = Command::new(env!("CARGO_BIN_EXE_fbz")); + command.args(["-z", "--format", format, "-q", "-P", &threads.to_string(), "-o", "-"]).arg(input).stdout(Stdio::null()).stderr(Stdio::null()); + command +} + +fn reference(program: &str, arguments: &[&str], input: &Path) -> Command { + let mut command = Command::new(program); + command.args(arguments).arg(input).stdout(Stdio::null()).stderr(Stdio::null()); + command +} + +fn compare(format: &str, program: &str, arguments: &[&str]) { + let directory = tempfile::tempdir().unwrap(); + let input = input_file(&directory); + assert!(fbz(format, &input).status().unwrap().success()); + assert!(reference(program, arguments, &input).status().unwrap().success()); + let ours = common::measure(&mut fbz(format, &input)).unwrap(); + let theirs = common::measure(&mut reference(program, arguments, &input)).unwrap(); + assert!(ours.status.success()); + assert!(theirs.status.success()); + eprintln!( + "{format} compression: fbz {:.1} ms / {:.1} MiB RSS, {program} {:.1} ms / {:.1} MiB RSS, {:.2}x speed", + ours.wall.as_secs_f64() * 1_000.0, + ours.peak_rss_bytes as f64 / 1_048_576.0, + theirs.wall.as_secs_f64() * 1_000.0, + theirs.peak_rss_bytes as f64 / 1_048_576.0, + theirs.wall.as_secs_f64() / ours.wall.as_secs_f64(), + ); +} + +#[test] +#[ignore = "local single-run bzip2 compression CLI comparison"] +fn bzip2_compression_cli_comparison() { + compare("bzip2", "bzip2", &["-9", "-c"]); +} + +#[test] +#[ignore = "local single-run gzip compression CLI comparison"] +fn gzip_compression_cli_comparison() { + compare("gzip", "gzip", &["-6", "-c"]); +} + +#[test] +#[ignore = "local single-run LZ4 compression CLI comparison"] +fn lz4_compression_cli_comparison() { + compare("lz4", "lz4", &["-c"]); +} + +#[derive(Clone, Copy)] +enum ZipShape { + Single, + Many, +} + +fn zip_inputs(directory: &Path, shape: ZipShape) -> Vec { + let contents = support::simplewiki_prefix(); + match shape { + ZipShape::Single => { + fs::write(directory.join("payload.xml"), contents).unwrap(); + vec![PathBuf::from("payload.xml")] + } + ZipShape::Many => contents + .chunks(contents.len().div_ceil(18)) + .enumerate() + .map(|(index, chunk)| { + let name = PathBuf::from(format!("part-{index:02}.xml")); + fs::write(directory.join(&name), chunk).unwrap(); + name + }) + .collect(), + } +} + +fn fbz_zip(directory: &Path, output: &str, inputs: &[PathBuf]) -> Command { + fbz_zip_with_threads(directory, output, inputs, 0) +} + +fn fbz_zip_with_threads(directory: &Path, output: &str, inputs: &[PathBuf], threads: usize) -> Command { + let mut command = Command::new(env!("CARGO_BIN_EXE_fbz")); + command.current_dir(directory).args(["-z", "--format", "zip", "-q", "-P", &threads.to_string(), "-o", output]); + command.args(inputs); + command +} + +fn info_zip(directory: &Path, output: &str, inputs: &[PathBuf]) -> Command { + let mut command = Command::new("zip"); + command.current_dir(directory).args(["-q", "-6", output]); + command.args(inputs); + command +} + +fn compare_zip(shape: ZipShape) { + let directory = tempfile::tempdir().unwrap(); + let inputs = zip_inputs(directory.path(), shape); + assert!(fbz_zip(directory.path(), "fbz-warm.zip", &inputs).status().unwrap().success()); + assert!(info_zip(directory.path(), "zip-warm.zip", &inputs).status().unwrap().success()); + let ours = common::measure(&mut fbz_zip(directory.path(), "fbz.zip", &inputs)).unwrap(); + let theirs = common::measure(&mut info_zip(directory.path(), "zip.zip", &inputs)).unwrap(); + assert!(ours.status.success()); + assert!(theirs.status.success()); + let our_size = fs::metadata(directory.path().join("fbz.zip")).unwrap().len(); + let their_size = fs::metadata(directory.path().join("zip.zip")).unwrap().len(); + eprintln!( + "ZIP {}: fbz {:.1} ms / {:.1} MiB RSS / {:.1} MiB, zip {:.1} ms / {:.1} MiB RSS / {:.1} MiB, {:.2}x speed", + match shape { + ZipShape::Single => "one entry", + ZipShape::Many => "18 entries", + }, + ours.wall.as_secs_f64() * 1_000.0, + ours.peak_rss_bytes as f64 / 1_048_576.0, + our_size as f64 / 1_048_576.0, + theirs.wall.as_secs_f64() * 1_000.0, + theirs.peak_rss_bytes as f64 / 1_048_576.0, + their_size as f64 / 1_048_576.0, + theirs.wall.as_secs_f64() / ours.wall.as_secs_f64(), + ); +} + +fn fbz_tar(directory: &Path, format: &str, output: &str) -> Command { + let mut command = Command::new(env!("CARGO_BIN_EXE_fbz")); + command.current_dir(directory).args(["-z", "--format", format, "-q", "-o", output, "payload.xml"]); + command +} + +fn system_tar(directory: &Path, flag: &str, output: &str) -> Command { + let mut command = Command::new("tar"); + command.current_dir(directory).args([flag, output, "payload.xml"]); + command +} + +fn compare_tar(format: &str, flag: &str, suffix: &str) { + let directory = tempfile::tempdir().unwrap(); + fs::write(directory.path().join("payload.xml"), support::simplewiki_prefix()).unwrap(); + assert!(fbz_tar(directory.path(), format, &format!("fbz-warm.{suffix}")).status().unwrap().success()); + assert!(system_tar(directory.path(), flag, &format!("tar-warm.{suffix}")).status().unwrap().success()); + let ours = common::measure(&mut fbz_tar(directory.path(), format, &format!("fbz.{suffix}"))).unwrap(); + let theirs = common::measure(&mut system_tar(directory.path(), flag, &format!("tar.{suffix}"))).unwrap(); + assert!(ours.status.success()); + assert!(theirs.status.success()); + eprintln!( + ".{suffix} creation: fbz {:.1} ms / {:.1} MiB RSS, tar {:.1} ms / {:.1} MiB RSS, {:.2}x speed", + ours.wall.as_secs_f64() * 1_000.0, + ours.peak_rss_bytes as f64 / 1_048_576.0, + theirs.wall.as_secs_f64() * 1_000.0, + theirs.peak_rss_bytes as f64 / 1_048_576.0, + theirs.wall.as_secs_f64() / ours.wall.as_secs_f64(), + ); +} + +#[test] +#[ignore = "local single-run tar.bz2 creation comparison"] +fn tar_bzip2_compression_cli_comparison() { + compare_tar("tar-bzip2", "-cjf", "tar.bz2"); +} + +#[test] +#[ignore = "local single-run tar.gz creation comparison"] +fn tar_gzip_compression_cli_comparison() { + compare_tar("tar-gzip", "-czf", "tar.gz"); +} + +#[test] +#[ignore = "local single-run one-entry ZIP creation comparison"] +fn zip_single_compression_cli_comparison() { + compare_zip(ZipShape::Single); +} + +#[test] +#[ignore = "local single-run many-entry ZIP creation comparison"] +fn zip_many_compression_cli_comparison() { + compare_zip(ZipShape::Many); +} + +#[test] +#[ignore = "local one-run-per-count ZIP compression scaling diagnostic"] +fn zip_compression_thread_sweep() { + let directory = tempfile::tempdir().unwrap(); + let inputs = zip_inputs(directory.path(), ZipShape::Many); + for threads in [8, 12, 18] { + let output = format!("fbz-{threads}.zip"); + let metrics = common::measure(&mut fbz_zip_with_threads(directory.path(), &output, &inputs, threads)).unwrap(); + assert!(metrics.status.success()); + eprintln!("{threads:>2} threads: {:>7.1} ms, peak RSS {:>5.1} MiB", metrics.wall.as_secs_f64() * 1_000.0, metrics.peak_rss_bytes as f64 / 1_048_576.0,); + } +} + +#[test] +#[ignore = "local one-run-per-count bzip2 compression scaling diagnostic"] +fn bzip2_compression_thread_sweep() { + let directory = tempfile::tempdir().unwrap(); + let input = input_file(&directory); + for threads in [1, 2, 4, 6, 8, 12, 18] { + let metrics = common::measure(&mut fbz_with_threads("bzip2", &input, threads)).unwrap(); + assert!(metrics.status.success()); + eprintln!("{threads:>2} threads: {:>7.1} ms, peak RSS {:>5.1} MiB", metrics.wall.as_secs_f64() * 1_000.0, metrics.peak_rss_bytes as f64 / 1_048_576.0,); + } +} diff --git a/tests/compression_perf.rs b/tests/compression_perf.rs new file mode 100644 index 0000000..f62bc98 --- /dev/null +++ b/tests/compression_perf.rs @@ -0,0 +1,83 @@ +mod support; + +use std::{io::Write, time::Instant}; + +use fbz::{EncodeOptions, gzip}; +use flate2::{Compression, write::GzEncoder}; +use lz4_flex::frame::FrameEncoder as Lz4Encoder; + +fn flate2_gzip(input: &[u8]) -> Vec { + let mut encoder = GzEncoder::new(Vec::new(), Compression::new(6)); + encoder.write_all(input).unwrap(); + encoder.finish().unwrap() +} + +fn lz4_flex(input: &[u8]) -> Vec { + let mut encoder = Lz4Encoder::new(Vec::new()); + encoder.write_all(input).unwrap(); + encoder.finish().unwrap() +} + +#[test] +#[ignore = "local single-run gzip compression comparison"] +fn gzip_compression_comparison() { + let input = support::simplewiki_prefix(); + let options = EncodeOptions { threads: 0, memory_limit: 1024 * 1024 * 1024, level: Some(6) }; + + let start = Instant::now(); + let ours = gzip::compress_with_options(&input, options).unwrap(); + let ours_time = start.elapsed(); + + let start = Instant::now(); + let oracle = flate2_gzip(&input); + let oracle_time = start.elapsed(); + + assert_eq!(gzip::decompress(&ours).unwrap(), input); + eprintln!( + "gzip compression: fbz {:.3?} {} bytes; flate2 {:.3?} {} bytes; speed {:.2}x, size {:.3}x", + ours_time, + ours.len(), + oracle_time, + oracle.len(), + oracle_time.as_secs_f64() / ours_time.as_secs_f64(), + ours.len() as f64 / oracle.len() as f64, + ); +} + +#[test] +#[ignore = "local single-run LZ4 compression comparison"] +fn lz4_compression_comparison() { + let input = support::simplewiki_prefix(); + let options = EncodeOptions { threads: 0, memory_limit: 1024 * 1024 * 1024, level: Some(6) }; + + let start = Instant::now(); + let ours = fbz::lz4::compress(&input, options).unwrap(); + let ours_time = start.elapsed(); + + let start = Instant::now(); + let oracle = lz4_flex(&input); + let oracle_time = start.elapsed(); + + assert_eq!(fbz::lz4::decompress(&ours).unwrap(), input); + eprintln!( + "LZ4 compression: fbz {:.3?} {} bytes; lz4_flex {:.3?} {} bytes; speed {:.2}x, size {:.3}x", + ours_time, + ours.len(), + oracle_time, + oracle.len(), + oracle_time.as_secs_f64() / ours_time.as_secs_f64(), + ours.len() as f64 / oracle.len() as f64, + ); +} + +#[test] +#[ignore = "local single-run LZ4 compression thread diagnostic"] +fn lz4_compression_thread_sweep() { + let input = support::simplewiki_prefix(); + for threads in [1, 2, 4, 8, 12, 18] { + let options = EncodeOptions { threads, memory_limit: 1024 * 1024 * 1024, level: None }; + let start = Instant::now(); + let encoded = fbz::lz4::compress(&input, options).unwrap(); + eprintln!("LZ4 compression {threads:>2} workers: {:.3?}, {} bytes", start.elapsed(), encoded.len()); + } +} diff --git a/tests/encode.rs b/tests/encode.rs new file mode 100644 index 0000000..de5b17b --- /dev/null +++ b/tests/encode.rs @@ -0,0 +1,31 @@ +use std::io::Write; + +use fbz::{DecodeOptions, EncodeFormat, EncodeOptions, Encoder, Format, compress, decompress, gzip, lz4}; + +fn decode(format: EncodeFormat, encoded: &[u8]) -> Vec { + match format { + EncodeFormat::Bzip2 => decompress(encoded, DecodeOptions::default()).unwrap(), + EncodeFormat::Gzip => gzip::decompress(encoded).unwrap(), + EncodeFormat::Lz4 => lz4::decompress(encoded).unwrap(), + } +} + +#[test] +fn unified_encoder_covers_every_stream_format() { + let input = b"one streaming compression API".repeat(10_000); + for (format, magic) in [(EncodeFormat::Bzip2, Format::Bzip2), (EncodeFormat::Gzip, Format::Gzip), (EncodeFormat::Lz4, Format::Lz4)] { + let encoded = compress(&input, format, EncodeOptions { threads: 3, memory_limit: 128 * 1024 * 1024, level: None }).unwrap(); + assert_eq!(Format::from_magic(&encoded), Some(magic)); + assert_eq!(decode(format, &encoded), input); + + let mut incremental = Vec::new(); + let mut encoder = Encoder::new(&mut incremental, format, EncodeOptions { threads: 2, memory_limit: 128 * 1024 * 1024, level: None }).unwrap(); + for chunk in input.chunks(7777) { + encoder.write_all(chunk).unwrap(); + } + let (_, report) = encoder.finish().unwrap(); + assert_eq!(report.format, format); + assert_eq!(report.input_len, input.len() as u64); + assert_eq!(decode(format, &incremental), input); + } +} diff --git a/tests/gzip_encode.rs b/tests/gzip_encode.rs new file mode 100644 index 0000000..809dffd --- /dev/null +++ b/tests/gzip_encode.rs @@ -0,0 +1,49 @@ +mod support; + +use std::io::Write; + +use fbz::{EncodeOptions, gzip}; +use flate2::read::GzDecoder; + +#[test] +fn gzip_encoder_roundtrips_with_fbz_and_flate2() { + let inputs = support::compression_inputs(b"hello gzip", 2_500_000); + for (index, input) in inputs.into_iter().enumerate() { + let threads = if index < 2 { 1 } else { 4 }; + let options = EncodeOptions { threads, memory_limit: 64 * 1024 * 1024, level: Some(6) }; + let encoded = gzip::compress_with_options(&input, options).unwrap(); + assert_eq!(gzip::decompress(&encoded).unwrap(), input); + + let mut decoded = Vec::new(); + std::io::copy(&mut GzDecoder::new(&encoded[..]), &mut decoded).unwrap(); + assert_eq!(decoded, input); + + let report = gzip::decompress_to_writer(&encoded, &mut std::io::sink()).unwrap(); + assert_eq!(report.members.len(), 1); + } +} + +#[test] +fn gzip_encoder_accepts_incremental_writes_and_flushes() { + let input: Vec<_> = (0..400_000).map(|index| (index % 251) as u8).collect(); + let mut encoded = Vec::new(); + let mut encoder = gzip::Encoder::new(&mut encoded, EncodeOptions { threads: 3, memory_limit: 32 * 1024 * 1024, level: Some(3) }).unwrap(); + for chunk in input.chunks(7777) { + encoder.write_all(chunk).unwrap(); + } + encoder.flush().unwrap(); + let (_, report) = encoder.finish().unwrap(); + assert_eq!(report.input_len, input.len() as u64); + assert_eq!(gzip::decompress(&encoded).unwrap(), input); +} + +#[test] +fn gzip_encoder_rejects_invalid_options() { + let error = gzip::compress_with_options(b"data", EncodeOptions { level: Some(0), ..EncodeOptions::default() }).unwrap_err(); + assert!(error.to_string().contains("level")); + let error = gzip::compress_with_options(b"data", EncodeOptions { memory_limit: 1, ..EncodeOptions::default() }).unwrap_err(); + assert!(error.to_string().contains("memory limit")); + let mut output = b"unchanged".to_vec(); + assert!(gzip::Encoder::new(&mut output, EncodeOptions { memory_limit: 1, ..EncodeOptions::default() }).is_err()); + assert_eq!(output, b"unchanged"); +} diff --git a/tests/lz4_encode.rs b/tests/lz4_encode.rs new file mode 100644 index 0000000..1bc6bde --- /dev/null +++ b/tests/lz4_encode.rs @@ -0,0 +1,40 @@ +mod support; + +use std::io::{Read, Write}; + +use fbz::{EncodeOptions, lz4}; +use lz4_flex::frame::FrameDecoder; + +#[test] +fn lz4_encoder_roundtrips_with_fbz_and_lz4_flex() { + let inputs = support::compression_inputs(b"hello LZ4", 10_000_000); + for (index, input) in inputs.into_iter().enumerate() { + let threads = if index < 2 { 1 } else { 4 }; + let options = EncodeOptions { threads, memory_limit: 128 * 1024 * 1024, level: Some(6) }; + let encoded = lz4::compress(&input, options).unwrap(); + assert_eq!(lz4::decompress(&encoded).unwrap(), input); + + let mut decoded = Vec::new(); + FrameDecoder::new(&encoded[..]).read_to_end(&mut decoded).unwrap(); + assert_eq!(decoded, input); + + let report = lz4::decompress_to_writer(&encoded, &mut std::io::sink()).unwrap(); + assert_eq!(report.frames.len(), 1); + assert_eq!(report.frames[0].block_mode, lz4::BlockMode::Independent); + assert!(report.frames[0].content_checksum); + } +} + +#[test] +fn lz4_encoder_accepts_incremental_writes_and_low_memory() { + let input: Vec<_> = (0..1_000_000).map(|index| (index % 251) as u8).collect(); + let mut encoded = Vec::new(); + let mut encoder = lz4::Encoder::new(&mut encoded, EncodeOptions { threads: 4, memory_limit: 2 * 1024 * 1024, level: Some(3) }).unwrap(); + for chunk in input.chunks(7777) { + encoder.write_all(chunk).unwrap(); + } + encoder.flush().unwrap(); + let (_, report) = encoder.finish().unwrap(); + assert_eq!(report.input_len, input.len() as u64); + assert_eq!(lz4::decompress(&encoded).unwrap(), input); +} diff --git a/tests/lz4_perf.rs b/tests/lz4_perf.rs index 51969c3..4c50c56 100644 --- a/tests/lz4_perf.rs +++ b/tests/lz4_perf.rs @@ -1,6 +1,5 @@ #[allow(dead_code, unused_imports)] mod common; -#[allow(dead_code)] mod support; use std::{fs, hint::black_box, io::Write, path::PathBuf, process::Command, time::Instant}; diff --git a/tests/reader_perf.rs b/tests/reader_perf.rs index 9898b7c..508822c 100644 --- a/tests/reader_perf.rs +++ b/tests/reader_perf.rs @@ -1,4 +1,3 @@ -#[allow(dead_code)] mod support; use std::{ diff --git a/tests/support/mod.rs b/tests/support/mod.rs index a2b30d3..e42da22 100644 --- a/tests/support/mod.rs +++ b/tests/support/mod.rs @@ -1,3 +1,5 @@ +#![allow(dead_code)] + use std::{fs, io::Write, path::Path}; use fbz::{DecodeOptions, decompress}; @@ -18,6 +20,18 @@ pub(crate) fn simplewiki_prefix() -> Vec { contents } +pub(crate) fn patterned_bytes(size: usize) -> Vec { + patterned_bytes_with(size, 31, 97) +} + +pub(crate) fn patterned_bytes_with(size: usize, stride: usize, period: usize) -> Vec { + (0..size).map(|index| ((index * stride + index / period) & 255) as u8).collect() +} + +pub(crate) fn compression_inputs(greeting: &[u8], size: usize) -> Vec> { + vec![Vec::new(), greeting.to_vec(), vec![b'a'; size], patterned_bytes(size)] +} + fn push_u16(output: &mut Vec, value: u16) { output.extend_from_slice(&value.to_le_bytes()); } diff --git a/tests/test_api.py b/tests/test_api.py index c69ec8a..d506449 100644 --- a/tests/test_api.py +++ b/tests/test_api.py @@ -1,4 +1,4 @@ -import bz2, io +import bz2, gzip, io from threading import Event, Thread import pytest @@ -12,6 +12,13 @@ def test_decompress_matches_libbz2_at_every_level(level): plain = patterned(20_000) assert fbz.decompress(bz2.compress(plain, compresslevel=level), threads=2) == plain +def test_compress_stream_formats_and_options(): + plain = patterned(300_000) + assert bz2.decompress(fbz.compress(plain, "bzip2", threads=2, level=3)) == plain + assert gzip.decompress(fbz.compress(plain, "gzip", threads=2, level=6)) == plain + assert fbz.compress(plain, "lz4", threads=2)[:4] == b"\x04\x22\x4d\x18" + with pytest.raises(ValueError, match="format"): fbz.compress(plain, "zip") + def test_parallel_multiblock_and_concatenated_are_deterministic(): first, second = patterned(350_000), patterned(75_000) compressed = bz2.compress(first, compresslevel=1) + bz2.compress(second, compresslevel=9) diff --git a/tests/zip_encode.rs b/tests/zip_encode.rs new file mode 100644 index 0000000..abcaecb --- /dev/null +++ b/tests/zip_encode.rs @@ -0,0 +1,56 @@ +mod support; + +use std::{fs, path::PathBuf, process::Command}; + +use fbz::{ + EncodeOptions, + zip::{PathInput, create_to_writer}, +}; + +#[test] +fn zip_encoder_roundtrips_files_and_directories() { + let directory = tempfile::tempdir().unwrap(); + let source = directory.path().join("source"); + fs::create_dir(&source).unwrap(); + fs::write(source.join("small.txt"), b"small contents").unwrap(); + fs::write(source.join("repeated.bin"), b"repeat me".repeat(200_000)).unwrap(); + + let mut encoded = Vec::new(); + let report = create_to_writer( + &[PathInput { source: source.clone(), archive_path: PathBuf::from("bundle") }], + &mut encoded, + EncodeOptions { threads: 4, memory_limit: 64 * 1024 * 1024, level: Some(6) }, + ) + .unwrap(); + assert_eq!(report.entries, 3); + + let path = directory.path().join("archive.zip"); + fs::write(&path, encoded).unwrap(); + assert!(Command::new("unzip").args(["-t", path.to_str().unwrap()]).status().unwrap().success()); + let small = Command::new("unzip").args(["-p", path.to_str().unwrap(), "bundle/small.txt"]).output().unwrap(); + assert!(small.status.success()); + assert_eq!(small.stdout, b"small contents"); + let repeated = Command::new("unzip").args(["-p", path.to_str().unwrap(), "bundle/repeated.bin"]).output().unwrap(); + assert!(repeated.status.success()); + assert_eq!(repeated.stdout, b"repeat me".repeat(200_000)); +} + +#[test] +fn zip_encoder_uses_parallel_deflate_for_a_large_entry() { + let directory = tempfile::tempdir().unwrap(); + let source = directory.path().join("large.bin"); + let plain = support::patterned_bytes(20 * 1024 * 1024); + fs::write(&source, &plain).unwrap(); + let mut encoded = Vec::new(); + create_to_writer( + &[PathInput { source, archive_path: PathBuf::from("large.bin") }], + &mut encoded, + EncodeOptions { threads: 4, memory_limit: 64 * 1024 * 1024, level: Some(3) }, + ) + .unwrap(); + let path = directory.path().join("large.zip"); + fs::write(&path, encoded).unwrap(); + let decoded = Command::new("unzip").args(["-p", path.to_str().unwrap(), "large.bin"]).output().unwrap(); + assert!(decoded.status.success()); + assert_eq!(decoded.stdout, plain); +}