Skip to content

Commit 7220adb

Browse files
Jun025claude
andcommitted
classfile: decode CONSTANT_Utf8 as modified UTF-8
CONSTANT_Utf8 is modified UTF-8 (JVMS 4.4.7): NUL is C0 80 and a supplementary character is a surrogate pair of two 3-byte sequences. String::from_utf8 rejects both, so a class with "\0" or e.g. U+1F40D in any constant fails to parse. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
1 parent a8bc80e commit 7220adb

1 file changed

Lines changed: 93 additions & 2 deletions

File tree

‎classfile/src/constant_pool.rs‎

Lines changed: 93 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -1,4 +1,5 @@
1-
use alloc::{collections::BTreeMap, string::String, sync::Arc};
1+
use alloc::{collections::BTreeMap, string::String, sync::Arc, vec::Vec};
2+
use core::char::REPLACEMENT_CHARACTER;
23

34
use nom::{
45
IResult, Parser,
@@ -10,7 +11,52 @@ use nom::{
1011

1112
fn parse_utf8(data: &[u8]) -> IResult<&[u8], Arc<String>> {
1213
let (data, length) = be_u16(data)?;
13-
map_res(take(length as usize), |utf8: &[u8]| String::from_utf8(utf8.to_vec()).map(Arc::new)).parse(data)
14+
map_res(take(length as usize), |utf8: &[u8]| decode_modified_utf8(utf8).map(Arc::new)).parse(data)
15+
}
16+
17+
// CONSTANT_Utf8 is *modified* UTF-8 (JVMS §4.4.7), not UTF-8: NUL is the two bytes C0 80, and a
18+
// supplementary character is its UTF-16 surrogate pair with each half encoded as three bytes
19+
// (ED A0..AF xx ED B0..BF xx). `String::from_utf8` rejects both, so any class holding "\0" or a
20+
// supplementary character in a constant failed to load with ClassFormatError. Decoding goes
21+
// through UTF-16 units, which is what the Java string is anyway.
22+
//
23+
// Strictly wider than before: input that is valid UTF-8 takes the old path unchanged. On the slow
24+
// path the standard 4-byte form is accepted too (javac never emits it, but the old parser did
25+
// accept it, so rejecting it now would be a regression), and a lone surrogate — legal in a Java
26+
// string, unrepresentable in a Rust `String` — becomes U+FFFD instead of failing the whole class.
27+
fn decode_modified_utf8(bytes: &[u8]) -> Result<String, ()> {
28+
if let Ok(x) = core::str::from_utf8(bytes) {
29+
return Ok(x.into());
30+
}
31+
32+
let continuation = |i: usize| match bytes.get(i) {
33+
Some(&b) if b & 0xC0 == 0x80 => Ok((b & 0x3F) as u32),
34+
_ => Err(()),
35+
};
36+
37+
let mut units = Vec::with_capacity(bytes.len());
38+
let mut i = 0;
39+
while i < bytes.len() {
40+
let b = bytes[i] as u32;
41+
let (code_point, length) = match bytes[i] {
42+
0x00..=0x7F => (b, 1),
43+
0xC0..=0xDF => (((b & 0x1F) << 6) | continuation(i + 1)?, 2),
44+
0xE0..=0xEF => (((b & 0x0F) << 12) | (continuation(i + 1)? << 6) | continuation(i + 2)?, 3),
45+
0xF0..=0xF4 => (
46+
((b & 0x07) << 18) | (continuation(i + 1)? << 12) | (continuation(i + 2)? << 6) | continuation(i + 3)?,
47+
4,
48+
),
49+
_ => return Err(()),
50+
};
51+
match char::from_u32(code_point) {
52+
Some(c) if code_point > 0xFFFF => units.extend_from_slice(c.encode_utf16(&mut [0; 2])),
53+
None if code_point > 0x10FFFF => return Err(()),
54+
_ => units.push(code_point as u16), // BMP, or one surrogate half (paired up below)
55+
}
56+
i += length;
57+
}
58+
59+
Ok(char::decode_utf16(units).map(|x| x.unwrap_or(REPLACEMENT_CHARACTER)).collect())
1460
}
1561

1662
#[derive(Debug)]
@@ -301,3 +347,48 @@ mod tests {
301347
assert!(ConstantPoolItem::parse_all(&[0x00, 0x02, 0x05, 0, 0, 0, 0, 0, 0, 0, 0]).is_err());
302348
}
303349
}
350+
351+
#[cfg(test)]
352+
mod modified_utf8_tests {
353+
use super::parse_utf8;
354+
355+
fn constant(bytes: &[u8]) -> Option<alloc::string::String> {
356+
let mut data = (bytes.len() as u16).to_be_bytes().to_vec();
357+
data.extend_from_slice(bytes);
358+
parse_utf8(&data).ok().map(|(rest, x)| {
359+
assert!(rest.is_empty());
360+
(*x).clone()
361+
})
362+
}
363+
364+
#[test]
365+
fn nul_is_c0_80() {
366+
// javac's encoding of "MTR\0"
367+
assert_eq!(constant(b"MTR\xC0\x80").as_deref(), Some("MTR\0"));
368+
}
369+
370+
#[test]
371+
fn supplementary_is_a_surrogate_pair() {
372+
// javac's encoding of "\uD83D\uDC0D" (U+1F40D) — six bytes, two 3-byte halves
373+
assert_eq!(constant(b"a\xED\xA0\xBD\xED\xB0\x8Db").as_deref(), Some("a\u{1F40D}b"));
374+
}
375+
376+
#[test]
377+
fn ascii_and_hangul_are_unchanged() {
378+
assert_eq!(constant(b"java/lang/Object").as_deref(), Some("java/lang/Object"));
379+
assert_eq!(constant("한글 5개".as_bytes()).as_deref(), Some("한글 5개"));
380+
assert_eq!(constant(b"").as_deref(), Some(""));
381+
}
382+
383+
#[test]
384+
fn edges_of_the_wider_decoder() {
385+
// standard 4-byte UTF-8 next to C0 80: the old parser accepted the first alone
386+
assert_eq!(constant(b"\xF0\x9F\x90\x8D\xC0\x80").as_deref(), Some("\u{1F40D}\0"));
387+
// a lone surrogate half cannot live in a Rust String
388+
assert_eq!(constant(b"x\xED\xA0\xBD").as_deref(), Some("x\u{FFFD}"));
389+
// still malformed: truncated sequence, stray continuation byte, invalid lead byte
390+
assert_eq!(constant(b"\xC0"), None);
391+
assert_eq!(constant(b"\x80"), None);
392+
assert_eq!(constant(b"\xFF"), None);
393+
}
394+
}

0 commit comments

Comments
 (0)