From a7a68077be257887ca3caed289f38b43f51e971e Mon Sep 17 00:00:00 2001 From: Matt Turner Date: Mon, 10 Aug 2026 00:49:28 -0400 Subject: [PATCH] Scan character classes with SIMD in the JIT on x86-64 fast_forward_start_bits() tests the start bitmap one code unit at a time on every architecture, so a pattern whose first character is a class scans far more slowly than one starting with a literal, even though the JIT already has a vectorized scan for the literal case. Add a hook for it, alongside the three that already exist, and implement it for x86-64 with SSE2. The bitmap is reduced to a list of ranges at compile time; the scan then tests each range with the usual unsigned range idiom, where subtracting the low bound makes the range start at zero and a saturating subtract of the span leaves zero exactly for the bytes inside it, since anything below the low bound wraps to a value larger than the span. That is PSUBB, PSUBUSB and PCMPEQB per range, OR'd together, with PCMPEQB alone for a single character. Bitmaps needing more than four ranges keep the existing scan. Vectorizing only pays for a sparse class. The byte-at-a-time loop stops at the first code unit in the class, so where the class is dense it stops almost immediately, while a vector loop has already tested a whole block. Measured on \b\w{12,}\b, whose class accepts 63 code units, the vector scan was 38.6% slower. Classes accepting more than 48 code units are therefore left alone. Measured over a 4MB subject on an i7-1370P, scanning for a class that does not occur: [QXZ] goes from 1445 to 25961 MB/s, and [0-9]{6} from 3337 to 33440 MB/s. Every other pattern measured is unchanged, including \b\w{12,}\b at 220 MB/s. RunTest and pcre2_jit_test pass. SSE4.2's PCMPESTRI was measured as an alternative. It handles up to eight ranges in one instruction and so is flat in the number of ranges, but it is slower than this sequence for one or two ranges, which is what real classes mostly are: 11.5 GB/s against 28.5 GB/s for a single range. It also has no 256-bit form, so it would foreclose widening this loop to AVX2 later. PCMPISTRI, the faster of the two, cannot be used at all, because it treats a zero byte as the end of the subject. --- src/pcre2_jit_compile.c | 10 ++ src/pcre2_jit_simd_inc.h | 234 +++++++++++++++++++++++++++++++++++++++ 2 files changed, 244 insertions(+) diff --git a/src/pcre2_jit_compile.c b/src/pcre2_jit_compile.c index 9013b6864..c591d4def 100644 --- a/src/pcre2_jit_compile.c +++ b/src/pcre2_jit_compile.c @@ -6967,6 +6967,16 @@ if (common->match_end_ptr != 0) SELECT(SLJIT_GREATER, STR_END, TMP1, 0, STR_END); } +#ifdef JIT_HAS_FAST_FORWARD_START_BITS_SIMD +if (JIT_HAS_FAST_FORWARD_START_BITS_SIMD && common->mode == PCRE2_JIT_COMPLETE + && fast_forward_start_bits_simd(common, start_bits)) + { + if (common->match_end_ptr != 0) + OP1(SLJIT_MOV, STR_END, 0, RETURN_ADDR, 0); + return; + } +#endif + start = LABEL(); partial_quit = CMP(SLJIT_GREATER_EQUAL, STR_PTR, 0, STR_END, 0); diff --git a/src/pcre2_jit_simd_inc.h b/src/pcre2_jit_simd_inc.h index 0a24cdad8..b8fb98681 100644 --- a/src/pcre2_jit_simd_inc.h +++ b/src/pcre2_jit_simd_inc.h @@ -52,6 +52,52 @@ typedef enum { vector_compare_match2, } vector_compare_type; +#if defined(SLJIT_CONFIG_X86_64) && SLJIT_CONFIG_X86_64 && PCRE2_CODE_UNIT_WIDTH == 8 + +/* Largest number of ranges a start bitmap may be split into before the + vectorized scans give up on it. */ +#define MAX_START_BITS_RANGES 8 + +typedef struct { + sljit_u32 low; + sljit_u32 high; +} start_bits_range; + +/* Split a start bitmap into inclusive ranges of set bits, for the vectorized + start-bits scans below. Returns the number of ranges, or -1 if there are more + than max_ranges of them, in which case the caller must fall back to testing + the bitmap one code unit at a time. */ + +static int extract_start_bits_ranges(const sljit_u8 *bits, start_bits_range *ranges, int max_ranges) +{ +int count = 0; +int i = 0; + +SLJIT_ASSERT(max_ranges <= MAX_START_BITS_RANGES); + +while (i < 256) + { + if ((bits[i >> 3] & (1 << (i & 0x7))) == 0) + { + i++; + continue; + } + + if (count >= max_ranges) + return -1; + + ranges[count].low = (sljit_u32)i; + while (i < 256 && (bits[i >> 3] & (1 << (i & 0x7))) != 0) + i++; + ranges[count].high = (sljit_u32)(i - 1); + count++; + } + +return count; +} + +#endif /* SLJIT_CONFIG_X86_64 && PCRE2_CODE_UNIT_WIDTH == 8 */ + #if (defined SLJIT_CONFIG_X86 && SLJIT_CONFIG_X86) static SLJIT_INLINE sljit_s32 max_fast_forward_char_pair_offset(void) { @@ -373,6 +419,194 @@ if (common->utf && offset > 0) #endif } +#if defined(SLJIT_CONFIG_X86_64) && SLJIT_CONFIG_X86_64 && PCRE2_CODE_UNIT_WIDTH == 8 + +#define JIT_HAS_FAST_FORWARD_START_BITS_SIMD 1 + +/* Number of ranges the vectorized start-bits scan will handle. Each one costs + two vector registers for its bounds; beyond this the byte-at-a-time bitmap + scan is likely to be the better option anyway, since every extra range adds + three instructions to each loop iteration. */ +#define X86_START_BITS_MAX_RANGES 4 + +/* Largest number of accepted code units before the class counts as dense and + the byte-at-a-time scan is left to do the job. */ +#define X86_START_BITS_MAX_COVERED 48 + +/* Emit an SSE2 instruction operating on two vector registers. A REX prefix is + needed for xmm8 and above, which the start-bits scan reaches as soon as it + holds the bounds of more than two ranges. */ +static SLJIT_INLINE void emit_sse2_op(struct sljit_compiler *compiler, sljit_u8 opcode, + sljit_s32 dst_ind, sljit_s32 src_ind) +{ +sljit_u8 instruction[5]; +int size = 0; + +instruction[size++] = 0x66; + +if (dst_ind >= 8 || src_ind >= 8) + instruction[size++] = (sljit_u8)(0x40 | ((dst_ind >= 8) ? 0x4 : 0) | ((src_ind >= 8) ? 0x1 : 0)); + +instruction[size++] = 0x0f; +instruction[size++] = opcode; +instruction[size++] = (sljit_u8)(0xc0 | ((dst_ind & 0x7) << 3) | (src_ind & 0x7)); +sljit_emit_op_custom(compiler, instruction, size); +} + +/* Emit the byte mask for one range into dst, given the data in src. + For a single character this is one PCMPEQB. For a range it is the standard + unsigned-range idiom: subtract the low bound so the range starts at zero, + then a saturating subtract of the span leaves zero exactly for the bytes + inside it. Values below the low bound wrap to something larger than the span, + so they saturate to non-zero and are correctly excluded. */ +static void emit_x86_class_range(struct sljit_compiler *compiler, sljit_s32 dst_ind, + sljit_s32 src_ind, sljit_s32 zero_ind, sljit_s32 low_ind, sljit_s32 span_ind, BOOL single) +{ +/* MOVDQA dst, src */ +emit_sse2_op(compiler, 0x6f, dst_ind, src_ind); + +if (single) + { + /* PCMPEQB dst, low */ + emit_sse2_op(compiler, 0x74, dst_ind, low_ind); + return; + } + +/* PSUBB dst, low */ +emit_sse2_op(compiler, 0xf8, dst_ind, low_ind); +/* PSUBUSB dst, span */ +emit_sse2_op(compiler, 0xd8, dst_ind, span_ind); +/* PCMPEQB dst, zero */ +emit_sse2_op(compiler, 0x74, dst_ind, zero_ind); +} + +/* Scan for the first code unit whose start bit is set, 16 bytes at a time. + Returns FALSE without emitting anything if the bitmap does not reduce to few + enough ranges, in which case the caller emits its own scan. */ +static BOOL fast_forward_start_bits_simd(compiler_common *common, const sljit_u8 *start_bits) +{ +DEFINE_COMPILER; +sljit_u8 instruction[4]; +sljit_s32 reg_type = SLJIT_SIMD_REG_128; +struct sljit_label *start; +struct sljit_jump *quit; +start_bits_range ranges[X86_START_BITS_MAX_RANGES]; +sljit_s32 tmp1_reg_ind = sljit_get_register_index(SLJIT_GP_REGISTER, TMP1); +sljit_s32 data_ind = sljit_get_register_index(reg_type, SLJIT_VR0); +sljit_s32 zero_ind = sljit_get_register_index(reg_type, SLJIT_VR1); +sljit_s32 acc_ind = sljit_get_register_index(reg_type, SLJIT_VR2); +sljit_s32 tmp_ind = sljit_get_register_index(reg_type, SLJIT_VR3); +sljit_s32 low_ind[X86_START_BITS_MAX_RANGES]; +sljit_s32 span_ind[X86_START_BITS_MAX_RANGES]; +BOOL single[X86_START_BITS_MAX_RANGES]; +sljit_s32 value; +int count, covered, k; + +count = extract_start_bits_ranges(start_bits, ranges, X86_START_BITS_MAX_RANGES); +if (count <= 0) + return FALSE; + +/* A vector scan only pays when it gets to scan. The byte-at-a-time loop stops + at the first code unit in the class, so for a dense class it usually stops + immediately, while this loop has already tested a whole block. Count the + code units the class accepts and leave the dense ones alone: \w covers 63 of + them and is common enough in real subjects to matter. */ +covered = 0; +for (k = 0; k < count; k++) + covered += (int)(ranges[k].high - ranges[k].low) + 1; + +if (covered > X86_START_BITS_MAX_COVERED) + return FALSE; + +/* Load the bounds. A single character needs only one register, so pack the + pairs from VR4 upwards and record what each range got. */ +value = SLJIT_SIMD_REG_128 | SLJIT_SIMD_ELEM_32 | SLJIT_SIMD_LANE_ZERO; + +for (k = 0; k < count; k++) + { + sljit_s32 low_reg = SLJIT_VR4 + (k * 2); + sljit_s32 span_reg = SLJIT_VR4 + (k * 2) + 1; + + single[k] = (ranges[k].low == ranges[k].high); + + sljit_emit_simd_lane_mov(compiler, value, low_reg, 0, SLJIT_IMM, + character_to_int32((PCRE2_UCHAR)ranges[k].low)); + sljit_emit_simd_lane_replicate(compiler, reg_type | SLJIT_SIMD_ELEM_32, low_reg, low_reg, 0); + low_ind[k] = sljit_get_register_index(reg_type, low_reg); + + span_ind[k] = low_ind[k]; + if (!single[k]) + { + sljit_emit_simd_lane_mov(compiler, value, span_reg, 0, SLJIT_IMM, + character_to_int32((PCRE2_UCHAR)(ranges[k].high - ranges[k].low))); + sljit_emit_simd_lane_replicate(compiler, reg_type | SLJIT_SIMD_ELEM_32, span_reg, span_reg, 0); + span_ind[k] = sljit_get_register_index(reg_type, span_reg); + } + } + +/* PXOR zero, zero */ +emit_sse2_op(compiler, 0xef, zero_ind, zero_ind); + +OP1(SLJIT_MOV, TMP2, 0, STR_PTR, 0); + +/* First part (unaligned start). */ +OP2(SLJIT_AND, STR_PTR, 0, STR_PTR, 0, SLJIT_IMM, ~0xf); +OP2(SLJIT_AND, TMP2, 0, TMP2, 0, SLJIT_IMM, 0xf); + +sljit_emit_simd_mov(compiler, reg_type | SLJIT_SIMD_MEM_ALIGNED_128, SLJIT_VR0, SLJIT_MEM1(STR_PTR), 0); + +emit_x86_class_range(compiler, acc_ind, data_ind, zero_ind, low_ind[0], span_ind[0], single[0]); +for (k = 1; k < count; k++) + { + emit_x86_class_range(compiler, tmp_ind, data_ind, zero_ind, low_ind[k], span_ind[k], single[k]); + /* POR acc, tmp */ + emit_sse2_op(compiler, 0xeb, acc_ind, tmp_ind); + } + +sljit_emit_simd_sign(compiler, SLJIT_SIMD_STORE | reg_type | SLJIT_SIMD_ELEM_8, SLJIT_VR2, TMP1, 0); +OP2(SLJIT_ADD, STR_PTR, 0, STR_PTR, 0, TMP2, 0); +OP2(SLJIT_LSHR, TMP1, 0, TMP1, 0, TMP2, 0); + +quit = CMP(SLJIT_NOT_ZERO, TMP1, 0, SLJIT_IMM, 0); + +OP2(SLJIT_SUB, STR_PTR, 0, STR_PTR, 0, TMP2, 0); + +/* Second part (aligned). */ +start = LABEL(); + +OP2(SLJIT_ADD, STR_PTR, 0, STR_PTR, 0, SLJIT_IMM, 16); + +add_jump(compiler, &common->failed_match, CMP(SLJIT_GREATER_EQUAL, STR_PTR, 0, STR_END, 0)); + +sljit_emit_simd_mov(compiler, reg_type | SLJIT_SIMD_MEM_ALIGNED_128, SLJIT_VR0, SLJIT_MEM1(STR_PTR), 0); + +emit_x86_class_range(compiler, acc_ind, data_ind, zero_ind, low_ind[0], span_ind[0], single[0]); +for (k = 1; k < count; k++) + { + emit_x86_class_range(compiler, tmp_ind, data_ind, zero_ind, low_ind[k], span_ind[k], single[k]); + emit_sse2_op(compiler, 0xeb, acc_ind, tmp_ind); + } + +sljit_emit_simd_sign(compiler, SLJIT_SIMD_STORE | reg_type | SLJIT_SIMD_ELEM_8, SLJIT_VR2, TMP1, 0); +CMPTO(SLJIT_ZERO, TMP1, 0, SLJIT_IMM, 0, start); + +JUMPHERE(quit); + +SLJIT_ASSERT(tmp1_reg_ind < 8); +/* BSF r32, r/m32 */ +instruction[0] = 0x0f; +instruction[1] = 0xbc; +instruction[2] = 0xc0 | (tmp1_reg_ind << 3) | tmp1_reg_ind; +sljit_emit_op_custom(compiler, instruction, 3); + +OP2(SLJIT_ADD, STR_PTR, 0, STR_PTR, 0, TMP1, 0); + +add_jump(compiler, &common->failed_match, CMP(SLJIT_GREATER_EQUAL, STR_PTR, 0, STR_END, 0)); +return TRUE; +} + +#endif /* SLJIT_CONFIG_X86_64 && PCRE2_CODE_UNIT_WIDTH == 8 */ + /* The AVX2 code path is currently disabled. #define JIT_HAS_FAST_REQUESTED_CHAR_SIMD (sljit_has_cpu_feature(SLJIT_HAS_SIMD)) */