1- use alloc:: { collections:: BTreeMap , string:: String , sync:: Arc } ;
1+ use alloc:: { collections:: BTreeMap , string:: String , sync:: Arc , vec:: Vec } ;
2+ use core:: char:: REPLACEMENT_CHARACTER ;
23
34use nom:: {
45 IResult , Parser ,
@@ -10,7 +11,52 @@ use nom::{
1011
1112fn parse_utf8 ( data : & [ u8 ] ) -> IResult < & [ u8 ] , Arc < String > > {
1213 let ( data, length) = be_u16 ( data) ?;
13- map_res ( take ( length as usize ) , |utf8 : & [ u8 ] | String :: from_utf8 ( utf8. to_vec ( ) ) . map ( Arc :: new) ) . parse ( data)
14+ map_res ( take ( length as usize ) , |utf8 : & [ u8 ] | decode_modified_utf8 ( utf8) . map ( Arc :: new) ) . parse ( data)
15+ }
16+
17+ // CONSTANT_Utf8 is *modified* UTF-8 (JVMS §4.4.7), not UTF-8: NUL is the two bytes C0 80, and a
18+ // supplementary character is its UTF-16 surrogate pair with each half encoded as three bytes
19+ // (ED A0..AF xx ED B0..BF xx). `String::from_utf8` rejects both, so any class holding "\0" or a
20+ // supplementary character in a constant failed to load with ClassFormatError. Decoding goes
21+ // through UTF-16 units, which is what the Java string is anyway.
22+ //
23+ // Strictly wider than before: input that is valid UTF-8 takes the old path unchanged. On the slow
24+ // path the standard 4-byte form is accepted too (javac never emits it, but the old parser did
25+ // accept it, so rejecting it now would be a regression), and a lone surrogate — legal in a Java
26+ // string, unrepresentable in a Rust `String` — becomes U+FFFD instead of failing the whole class.
27+ fn decode_modified_utf8 ( bytes : & [ u8 ] ) -> Result < String , ( ) > {
28+ if let Ok ( x) = core:: str:: from_utf8 ( bytes) {
29+ return Ok ( x. into ( ) ) ;
30+ }
31+
32+ let continuation = |i : usize | match bytes. get ( i) {
33+ Some ( & b) if b & 0xC0 == 0x80 => Ok ( ( b & 0x3F ) as u32 ) ,
34+ _ => Err ( ( ) ) ,
35+ } ;
36+
37+ let mut units = Vec :: with_capacity ( bytes. len ( ) ) ;
38+ let mut i = 0 ;
39+ while i < bytes. len ( ) {
40+ let b = bytes[ i] as u32 ;
41+ let ( code_point, length) = match bytes[ i] {
42+ 0x00 ..=0x7F => ( b, 1 ) ,
43+ 0xC0 ..=0xDF => ( ( ( b & 0x1F ) << 6 ) | continuation ( i + 1 ) ?, 2 ) ,
44+ 0xE0 ..=0xEF => ( ( ( b & 0x0F ) << 12 ) | ( continuation ( i + 1 ) ? << 6 ) | continuation ( i + 2 ) ?, 3 ) ,
45+ 0xF0 ..=0xF4 => (
46+ ( ( b & 0x07 ) << 18 ) | ( continuation ( i + 1 ) ? << 12 ) | ( continuation ( i + 2 ) ? << 6 ) | continuation ( i + 3 ) ?,
47+ 4 ,
48+ ) ,
49+ _ => return Err ( ( ) ) ,
50+ } ;
51+ match char:: from_u32 ( code_point) {
52+ Some ( c) if code_point > 0xFFFF => units. extend_from_slice ( c. encode_utf16 ( & mut [ 0 ; 2 ] ) ) ,
53+ None if code_point > 0x10FFFF => return Err ( ( ) ) ,
54+ _ => units. push ( code_point as u16 ) , // BMP, or one surrogate half (paired up below)
55+ }
56+ i += length;
57+ }
58+
59+ Ok ( char:: decode_utf16 ( units) . map ( |x| x. unwrap_or ( REPLACEMENT_CHARACTER ) ) . collect ( ) )
1460}
1561
1662#[ derive( Debug ) ]
@@ -301,3 +347,48 @@ mod tests {
301347 assert ! ( ConstantPoolItem :: parse_all( & [ 0x00 , 0x02 , 0x05 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 ] ) . is_err( ) ) ;
302348 }
303349}
350+
351+ #[ cfg( test) ]
352+ mod modified_utf8_tests {
353+ use super :: parse_utf8;
354+
355+ fn constant ( bytes : & [ u8 ] ) -> Option < alloc:: string:: String > {
356+ let mut data = ( bytes. len ( ) as u16 ) . to_be_bytes ( ) . to_vec ( ) ;
357+ data. extend_from_slice ( bytes) ;
358+ parse_utf8 ( & data) . ok ( ) . map ( |( rest, x) | {
359+ assert ! ( rest. is_empty( ) ) ;
360+ ( * x) . clone ( )
361+ } )
362+ }
363+
364+ #[ test]
365+ fn nul_is_c0_80 ( ) {
366+ // javac's encoding of "MTR\0"
367+ assert_eq ! ( constant( b"MTR\xC0 \x80 " ) . as_deref( ) , Some ( "MTR\0 " ) ) ;
368+ }
369+
370+ #[ test]
371+ fn supplementary_is_a_surrogate_pair ( ) {
372+ // javac's encoding of "\uD83D\uDC0D" (U+1F40D) — six bytes, two 3-byte halves
373+ assert_eq ! ( constant( b"a\xED \xA0 \xBD \xED \xB0 \x8D b" ) . as_deref( ) , Some ( "a\u{1F40D} b" ) ) ;
374+ }
375+
376+ #[ test]
377+ fn ascii_and_hangul_are_unchanged ( ) {
378+ assert_eq ! ( constant( b"java/lang/Object" ) . as_deref( ) , Some ( "java/lang/Object" ) ) ;
379+ assert_eq ! ( constant( "한글 5개" . as_bytes( ) ) . as_deref( ) , Some ( "한글 5개" ) ) ;
380+ assert_eq ! ( constant( b"" ) . as_deref( ) , Some ( "" ) ) ;
381+ }
382+
383+ #[ test]
384+ fn edges_of_the_wider_decoder ( ) {
385+ // standard 4-byte UTF-8 next to C0 80: the old parser accepted the first alone
386+ assert_eq ! ( constant( b"\xF0 \x9F \x90 \x8D \xC0 \x80 " ) . as_deref( ) , Some ( "\u{1F40D} \0 " ) ) ;
387+ // a lone surrogate half cannot live in a Rust String
388+ assert_eq ! ( constant( b"x\xED \xA0 \xBD " ) . as_deref( ) , Some ( "x\u{FFFD} " ) ) ;
389+ // still malformed: truncated sequence, stray continuation byte, invalid lead byte
390+ assert_eq ! ( constant( b"\xC0 " ) , None ) ;
391+ assert_eq ! ( constant( b"\x80 " ) , None ) ;
392+ assert_eq ! ( constant( b"\xFF " ) , None ) ;
393+ }
394+ }
0 commit comments