Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions DOCS/CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,10 @@ section under a version heading.
- PDF to text (`src/io/pdf/content.rs`, `font.rs`, `text.rs`): every page, or `--page N`, laid out as pdftotext lays it out; ToUnicode, the standard encodings with `/Differences`, Type0 Identity fonts, Type3, and the Core 14 widths (`scripts/gen-pdf-tables.py`). Matches pdftotext line for line on 14 fixtures, dehyphenation included.
- PDF to Markdown, HTML, and Word (`src/io/pdf/markdown.rs`): headings from type size and weight (three levels by size and bold lines below them), bullet and numbered lists, paragraphs with hyphenated words joined, and page numbers dropped.

### Performance

- Small one-shot inflates (a PDF page's content streams, small ZIP parts) start from a slab a few times their size instead of a megabyte, build the packed literal table only for inputs of 4 KB and up, and no longer build packed copies of the distance and code-length tables, which nothing reads: a 300-page PDF of 14,400 small streams reads its text in 0.13 s instead of 0.61.

## [0.23.1] - 2026-09-27

### Performance
Expand Down
112 changes: 88 additions & 24 deletions src/io/deflate/inflate.rs
Original file line number Diff line number Diff line change
Expand Up @@ -138,6 +138,9 @@ const MAX_SYMBOL_BITS: u32 = 48;
const MAX_MATCH: usize = 258;
/// The fast loop sizes its output this much at a time.
const SLAB: usize = 1 << 20;
/// A one-shot input shorter than this decodes without the packed literal
/// table: building it costs more than it saves on so few symbols.
const PACK_FROM: usize = 4096;

/// Entry layout of the full table. Bits 0 to 3: the code length. Bits
/// 4 to 7: the same (kept so the packed layout can tell entries apart).
Expand Down Expand Up @@ -196,7 +199,14 @@ impl Huffman {
}
}

fn build(lengths: &[u8], offset: usize, alphabet: Alphabet) -> Result<Huffman, InflateError> {
/// `pack` asks for the literal table's packed copy, which only the
/// fast loop reads; other alphabets never need one.
fn build(
lengths: &[u8],
offset: usize,
alphabet: Alphabet,
pack: bool,
) -> Result<Huffman, InflateError> {
let max_length = u32::from(lengths.iter().copied().max().unwrap_or(0));
if max_length == 0 {
return Ok(Huffman::empty());
Expand Down Expand Up @@ -235,20 +245,23 @@ impl Huffman {
}
let mut table = vec![0u32; PRIMARY_SIZE];
// Long codes: each primary prefix gets a second-level table wide
// enough for the longest code behind it.
let mut sub_bits = vec![0u32; PRIMARY_SIZE];
for (symbol, length) in lengths.iter().enumerate() {
let length = u32::from(*length);
if length > PRIMARY_BITS {
let prefix = (reversed_codes[symbol] as usize) & (PRIMARY_SIZE - 1);
sub_bits[prefix] = sub_bits[prefix].max(length - PRIMARY_BITS);
// enough for the longest code behind it. Most codes fit the
// primary bits, and then there is nothing to lay out.
if max_length > PRIMARY_BITS {
let mut sub_bits = vec![0u32; PRIMARY_SIZE];
for (symbol, length) in lengths.iter().enumerate() {
let length = u32::from(*length);
if length > PRIMARY_BITS {
let prefix = (reversed_codes[symbol] as usize) & (PRIMARY_SIZE - 1);
sub_bits[prefix] = sub_bits[prefix].max(length - PRIMARY_BITS);
}
}
}
for prefix in 0..PRIMARY_SIZE {
if sub_bits[prefix] > 0 {
let start = table.len() as u32;
table.resize(table.len() + (1usize << sub_bits[prefix]), 0);
table[prefix] = POINTER | (start << 8) | sub_bits[prefix];
for prefix in 0..PRIMARY_SIZE {
if sub_bits[prefix] > 0 {
let start = table.len() as u32;
table.resize(table.len() + (1usize << sub_bits[prefix]), 0);
table[prefix] = POINTER | (start << 8) | sub_bits[prefix];
}
}
}
for (symbol, length) in lengths.iter().enumerate() {
Expand Down Expand Up @@ -279,6 +292,12 @@ impl Huffman {
}
}
}
if alphabet != Alphabet::Literals || !pack {
return Ok(Huffman {
table,
packed: Vec::new(),
});
}
// Fold literals: where a literal's code leaves room in the
// primary bits for whole further literal codes, the packed entry
// holds up to three and the fast path consumes them in one step.
Expand Down Expand Up @@ -461,6 +480,9 @@ pub struct Inflater {
/// Input a block header straddled: kept here and read before the
/// next piece, so a caller never re-sends bytes.
pending: Vec<u8>,
/// Build the literal table's packed copy for the fast loop: worth it
/// on long streams, not on a few hundred bytes.
packs: bool,
}

impl Default for Inflater {
Expand All @@ -484,6 +506,7 @@ impl Inflater {
compacted: 0,
total_in: 0,
pending: Vec::new(),
packs: true,
}
}

Expand Down Expand Up @@ -685,13 +708,13 @@ impl Inflater {
self.state = State::Stored(length);
}
1 => {
let (literals, distances) = fixed_tables()?;
let (literals, distances) = fixed_tables(self.packs)?;
self.literals = literals;
self.distances = distances;
self.state = State::Huffman;
}
2 => {
let (literals, distances) = dynamic_tables(reader, offset)?;
let (literals, distances) = dynamic_tables(reader, offset, self.packs)?;
self.literals = literals;
self.distances = distances;
self.state = State::Huffman;
Expand Down Expand Up @@ -1037,12 +1060,21 @@ pub fn inflate(input: &[u8], out: &mut Vec<u8>, limit: usize) -> Result<usize, I
// and a match, so it never doubles (a doubling held the old slab and
// a new one twice its size at once) and is moved out, not copied.
let expected = limit.min(input.len().saturating_mul(1032));
// A small input starts from a slab a few times its size and asks for
// output that much at a time, doubling: the fast loop's megabyte per
// call cost a fresh map, its faults, and an unmap for every small
// stream (a PDF page's content streams, small ZIP entries).
let mut want = usize::MAX;
inflater.packs = input.len() >= PACK_FROM;
if expected > SLAB {
inflater.out = vec![0u8; expected + SLAB + MAX_MATCH + 16];
} else {
want = input.len().saturating_mul(4).clamp(4096, SLAB);
inflater.out = vec![0u8; want + MAX_MATCH + 16];
}
let mut at = 0usize;
loop {
let (consumed, progress) = inflater.push(&input[at..], usize::MAX)?;
let (consumed, progress) = inflater.push(&input[at..], want)?;
at += consumed;
if inflater.output().len() > limit {
return Err(InflateError::TooLarge);
Expand All @@ -1054,7 +1086,7 @@ pub fn inflate(input: &[u8], out: &mut Vec<u8>, limit: usize) -> Result<usize, I
return Err(InflateError::Truncated);
}
}
Progress::OutputFull => {}
Progress::OutputFull => want = want.saturating_mul(2),
}
}
let consumed = inflater.total_in() - inflater.leftover().len();
Expand All @@ -1066,7 +1098,7 @@ pub fn inflate(input: &[u8], out: &mut Vec<u8>, limit: usize) -> Result<usize, I
Ok(consumed)
}

fn fixed_tables() -> Result<(Huffman, Huffman), InflateError> {
fn fixed_tables(pack: bool) -> Result<(Huffman, Huffman), InflateError> {
let mut lengths = [0u8; 288];
for (symbol, length) in lengths.iter_mut().enumerate() {
*length = match symbol {
Expand All @@ -1076,14 +1108,15 @@ fn fixed_tables() -> Result<(Huffman, Huffman), InflateError> {
_ => 8,
};
}
let literals = Huffman::build(&lengths, 0, Alphabet::Literals)?;
let distances = Huffman::build(&[5u8; 30], 0, Alphabet::Other)?;
let literals = Huffman::build(&lengths, 0, Alphabet::Literals, pack)?;
let distances = Huffman::build(&[5u8; 30], 0, Alphabet::Other, false)?;
Ok((literals, distances))
}

fn dynamic_tables(
reader: &mut BitReader<'_>,
offset: usize,
pack: bool,
) -> Result<(Huffman, Huffman), InflateError> {
let literal_count = reader.bits(5)? as usize + 257;
let distance_count = reader.bits(5)? as usize + 1;
Expand All @@ -1098,7 +1131,7 @@ fn dynamic_tables(
for index in CODE_LENGTH_ORDER.iter().take(code_length_count) {
code_lengths[*index] = reader.bits(3)? as u8;
}
let code_length_code = Huffman::build(&code_lengths, offset, Alphabet::Other)?;
let code_length_code = Huffman::build(&code_lengths, offset, Alphabet::Other, false)?;
let total = literal_count + distance_count;
let mut lengths = vec![0u8; total];
let mut index = 0usize;
Expand Down Expand Up @@ -1136,8 +1169,8 @@ fn dynamic_tables(
what: "no end-of-block code",
});
}
let literals = Huffman::build(&lengths[..literal_count], offset, Alphabet::Literals)?;
let distances = Huffman::build(&lengths[literal_count..], offset, Alphabet::Other)?;
let literals = Huffman::build(&lengths[..literal_count], offset, Alphabet::Literals, pack)?;
let distances = Huffman::build(&lengths[literal_count..], offset, Alphabet::Other, false)?;
Ok((literals, distances))
}

Expand Down Expand Up @@ -1177,6 +1210,37 @@ mod tests {
assert_eq!(run(&input).unwrap(), b"hello");
}

/// A small input that inflates far past its first slab (4x the input,
/// at least 4 KB) grows by doubling, and a small dynamic-block input
/// decodes without the packed literal table.
#[test]
fn small_inputs_grow_and_decode_unpacked() {
let mut expanded = Vec::new();
for index in 0..300_000u32 {
expanded.push(if index % 1000 < 900 {
b'a'
} else {
(index % 251) as u8
});
}
let mut compressed = Vec::new();
crate::io::deflate::deflate(&expanded, &mut compressed);
assert!(
compressed.len() < PACK_FROM,
"the input is small: {}",
compressed.len()
);
let mut out = Vec::new();
inflate(&compressed, &mut out, usize::MAX).expect("inflates");
assert!(out == expanded);
let text = b"The quick brown fox jumps over the lazy dog, twice: the quick brown fox.";
let mut compressed = Vec::new();
crate::io::deflate::deflate(text, &mut compressed);
let mut out = Vec::new();
inflate(&compressed, &mut out, usize::MAX).expect("inflates");
assert_eq!(out, text);
}

#[test]
fn fixed_huffman_block() {
let input = [75, 76, 74, 78, 68, 66, 0];
Expand Down
Loading