diff --git a/src/pcre2_jit_compile.c b/src/pcre2_jit_compile.c index 9013b6864..c591d4def 100644 --- a/src/pcre2_jit_compile.c +++ b/src/pcre2_jit_compile.c @@ -6967,6 +6967,16 @@ if (common->match_end_ptr != 0) SELECT(SLJIT_GREATER, STR_END, TMP1, 0, STR_END); } +#ifdef JIT_HAS_FAST_FORWARD_START_BITS_SIMD +if (JIT_HAS_FAST_FORWARD_START_BITS_SIMD && common->mode == PCRE2_JIT_COMPLETE + && fast_forward_start_bits_simd(common, start_bits)) + { + if (common->match_end_ptr != 0) + OP1(SLJIT_MOV, STR_END, 0, RETURN_ADDR, 0); + return; + } +#endif + start = LABEL(); partial_quit = CMP(SLJIT_GREATER_EQUAL, STR_PTR, 0, STR_END, 0); diff --git a/src/pcre2_jit_simd_inc.h b/src/pcre2_jit_simd_inc.h index 0a24cdad8..b8fb98681 100644 --- a/src/pcre2_jit_simd_inc.h +++ b/src/pcre2_jit_simd_inc.h @@ -52,6 +52,52 @@ typedef enum { vector_compare_match2, } vector_compare_type; +#if defined(SLJIT_CONFIG_X86_64) && SLJIT_CONFIG_X86_64 && PCRE2_CODE_UNIT_WIDTH == 8 + +/* Largest number of ranges a start bitmap may be split into before the + vectorized scans give up on it. */ +#define MAX_START_BITS_RANGES 8 + +typedef struct { + sljit_u32 low; + sljit_u32 high; +} start_bits_range; + +/* Split a start bitmap into inclusive ranges of set bits, for the vectorized + start-bits scans below. Returns the number of ranges, or -1 if there are more + than max_ranges of them, in which case the caller must fall back to testing + the bitmap one code unit at a time. */ + +static int extract_start_bits_ranges(const sljit_u8 *bits, start_bits_range *ranges, int max_ranges) +{ +int count = 0; +int i = 0; + +SLJIT_ASSERT(max_ranges <= MAX_START_BITS_RANGES); + +while (i < 256) + { + if ((bits[i >> 3] & (1 << (i & 0x7))) == 0) + { + i++; + continue; + } + + if (count >= max_ranges) + return -1; + + ranges[count].low = (sljit_u32)i; + while (i < 256 && (bits[i >> 3] & (1 << (i & 0x7))) != 0) + i++; + ranges[count].high = (sljit_u32)(i - 1); + count++; + } + +return count; +} + +#endif /* SLJIT_CONFIG_X86_64 && PCRE2_CODE_UNIT_WIDTH == 8 */ + #if (defined SLJIT_CONFIG_X86 && SLJIT_CONFIG_X86) static SLJIT_INLINE sljit_s32 max_fast_forward_char_pair_offset(void) { @@ -373,6 +419,194 @@ if (common->utf && offset > 0) #endif } +#if defined(SLJIT_CONFIG_X86_64) && SLJIT_CONFIG_X86_64 && PCRE2_CODE_UNIT_WIDTH == 8 + +#define JIT_HAS_FAST_FORWARD_START_BITS_SIMD 1 + +/* Number of ranges the vectorized start-bits scan will handle. Each one costs + two vector registers for its bounds; beyond this the byte-at-a-time bitmap + scan is likely to be the better option anyway, since every extra range adds + three instructions to each loop iteration. */ +#define X86_START_BITS_MAX_RANGES 4 + +/* Largest number of accepted code units before the class counts as dense and + the byte-at-a-time scan is left to do the job. */ +#define X86_START_BITS_MAX_COVERED 48 + +/* Emit an SSE2 instruction operating on two vector registers. A REX prefix is + needed for xmm8 and above, which the start-bits scan reaches as soon as it + holds the bounds of more than two ranges. */ +static SLJIT_INLINE void emit_sse2_op(struct sljit_compiler *compiler, sljit_u8 opcode, + sljit_s32 dst_ind, sljit_s32 src_ind) +{ +sljit_u8 instruction[5]; +int size = 0; + +instruction[size++] = 0x66; + +if (dst_ind >= 8 || src_ind >= 8) + instruction[size++] = (sljit_u8)(0x40 | ((dst_ind >= 8) ? 0x4 : 0) | ((src_ind >= 8) ? 0x1 : 0)); + +instruction[size++] = 0x0f; +instruction[size++] = opcode; +instruction[size++] = (sljit_u8)(0xc0 | ((dst_ind & 0x7) << 3) | (src_ind & 0x7)); +sljit_emit_op_custom(compiler, instruction, size); +} + +/* Emit the byte mask for one range into dst, given the data in src. + For a single character this is one PCMPEQB. For a range it is the standard + unsigned-range idiom: subtract the low bound so the range starts at zero, + then a saturating subtract of the span leaves zero exactly for the bytes + inside it. Values below the low bound wrap to something larger than the span, + so they saturate to non-zero and are correctly excluded. */ +static void emit_x86_class_range(struct sljit_compiler *compiler, sljit_s32 dst_ind, + sljit_s32 src_ind, sljit_s32 zero_ind, sljit_s32 low_ind, sljit_s32 span_ind, BOOL single) +{ +/* MOVDQA dst, src */ +emit_sse2_op(compiler, 0x6f, dst_ind, src_ind); + +if (single) + { + /* PCMPEQB dst, low */ + emit_sse2_op(compiler, 0x74, dst_ind, low_ind); + return; + } + +/* PSUBB dst, low */ +emit_sse2_op(compiler, 0xf8, dst_ind, low_ind); +/* PSUBUSB dst, span */ +emit_sse2_op(compiler, 0xd8, dst_ind, span_ind); +/* PCMPEQB dst, zero */ +emit_sse2_op(compiler, 0x74, dst_ind, zero_ind); +} + +/* Scan for the first code unit whose start bit is set, 16 bytes at a time. + Returns FALSE without emitting anything if the bitmap does not reduce to few + enough ranges, in which case the caller emits its own scan. */ +static BOOL fast_forward_start_bits_simd(compiler_common *common, const sljit_u8 *start_bits) +{ +DEFINE_COMPILER; +sljit_u8 instruction[4]; +sljit_s32 reg_type = SLJIT_SIMD_REG_128; +struct sljit_label *start; +struct sljit_jump *quit; +start_bits_range ranges[X86_START_BITS_MAX_RANGES]; +sljit_s32 tmp1_reg_ind = sljit_get_register_index(SLJIT_GP_REGISTER, TMP1); +sljit_s32 data_ind = sljit_get_register_index(reg_type, SLJIT_VR0); +sljit_s32 zero_ind = sljit_get_register_index(reg_type, SLJIT_VR1); +sljit_s32 acc_ind = sljit_get_register_index(reg_type, SLJIT_VR2); +sljit_s32 tmp_ind = sljit_get_register_index(reg_type, SLJIT_VR3); +sljit_s32 low_ind[X86_START_BITS_MAX_RANGES]; +sljit_s32 span_ind[X86_START_BITS_MAX_RANGES]; +BOOL single[X86_START_BITS_MAX_RANGES]; +sljit_s32 value; +int count, covered, k; + +count = extract_start_bits_ranges(start_bits, ranges, X86_START_BITS_MAX_RANGES); +if (count <= 0) + return FALSE; + +/* A vector scan only pays when it gets to scan. The byte-at-a-time loop stops + at the first code unit in the class, so for a dense class it usually stops + immediately, while this loop has already tested a whole block. Count the + code units the class accepts and leave the dense ones alone: \w covers 63 of + them and is common enough in real subjects to matter. */ +covered = 0; +for (k = 0; k < count; k++) + covered += (int)(ranges[k].high - ranges[k].low) + 1; + +if (covered > X86_START_BITS_MAX_COVERED) + return FALSE; + +/* Load the bounds. A single character needs only one register, so pack the + pairs from VR4 upwards and record what each range got. */ +value = SLJIT_SIMD_REG_128 | SLJIT_SIMD_ELEM_32 | SLJIT_SIMD_LANE_ZERO; + +for (k = 0; k < count; k++) + { + sljit_s32 low_reg = SLJIT_VR4 + (k * 2); + sljit_s32 span_reg = SLJIT_VR4 + (k * 2) + 1; + + single[k] = (ranges[k].low == ranges[k].high); + + sljit_emit_simd_lane_mov(compiler, value, low_reg, 0, SLJIT_IMM, + character_to_int32((PCRE2_UCHAR)ranges[k].low)); + sljit_emit_simd_lane_replicate(compiler, reg_type | SLJIT_SIMD_ELEM_32, low_reg, low_reg, 0); + low_ind[k] = sljit_get_register_index(reg_type, low_reg); + + span_ind[k] = low_ind[k]; + if (!single[k]) + { + sljit_emit_simd_lane_mov(compiler, value, span_reg, 0, SLJIT_IMM, + character_to_int32((PCRE2_UCHAR)(ranges[k].high - ranges[k].low))); + sljit_emit_simd_lane_replicate(compiler, reg_type | SLJIT_SIMD_ELEM_32, span_reg, span_reg, 0); + span_ind[k] = sljit_get_register_index(reg_type, span_reg); + } + } + +/* PXOR zero, zero */ +emit_sse2_op(compiler, 0xef, zero_ind, zero_ind); + +OP1(SLJIT_MOV, TMP2, 0, STR_PTR, 0); + +/* First part (unaligned start). */ +OP2(SLJIT_AND, STR_PTR, 0, STR_PTR, 0, SLJIT_IMM, ~0xf); +OP2(SLJIT_AND, TMP2, 0, TMP2, 0, SLJIT_IMM, 0xf); + +sljit_emit_simd_mov(compiler, reg_type | SLJIT_SIMD_MEM_ALIGNED_128, SLJIT_VR0, SLJIT_MEM1(STR_PTR), 0); + +emit_x86_class_range(compiler, acc_ind, data_ind, zero_ind, low_ind[0], span_ind[0], single[0]); +for (k = 1; k < count; k++) + { + emit_x86_class_range(compiler, tmp_ind, data_ind, zero_ind, low_ind[k], span_ind[k], single[k]); + /* POR acc, tmp */ + emit_sse2_op(compiler, 0xeb, acc_ind, tmp_ind); + } + +sljit_emit_simd_sign(compiler, SLJIT_SIMD_STORE | reg_type | SLJIT_SIMD_ELEM_8, SLJIT_VR2, TMP1, 0); +OP2(SLJIT_ADD, STR_PTR, 0, STR_PTR, 0, TMP2, 0); +OP2(SLJIT_LSHR, TMP1, 0, TMP1, 0, TMP2, 0); + +quit = CMP(SLJIT_NOT_ZERO, TMP1, 0, SLJIT_IMM, 0); + +OP2(SLJIT_SUB, STR_PTR, 0, STR_PTR, 0, TMP2, 0); + +/* Second part (aligned). */ +start = LABEL(); + +OP2(SLJIT_ADD, STR_PTR, 0, STR_PTR, 0, SLJIT_IMM, 16); + +add_jump(compiler, &common->failed_match, CMP(SLJIT_GREATER_EQUAL, STR_PTR, 0, STR_END, 0)); + +sljit_emit_simd_mov(compiler, reg_type | SLJIT_SIMD_MEM_ALIGNED_128, SLJIT_VR0, SLJIT_MEM1(STR_PTR), 0); + +emit_x86_class_range(compiler, acc_ind, data_ind, zero_ind, low_ind[0], span_ind[0], single[0]); +for (k = 1; k < count; k++) + { + emit_x86_class_range(compiler, tmp_ind, data_ind, zero_ind, low_ind[k], span_ind[k], single[k]); + emit_sse2_op(compiler, 0xeb, acc_ind, tmp_ind); + } + +sljit_emit_simd_sign(compiler, SLJIT_SIMD_STORE | reg_type | SLJIT_SIMD_ELEM_8, SLJIT_VR2, TMP1, 0); +CMPTO(SLJIT_ZERO, TMP1, 0, SLJIT_IMM, 0, start); + +JUMPHERE(quit); + +SLJIT_ASSERT(tmp1_reg_ind < 8); +/* BSF r32, r/m32 */ +instruction[0] = 0x0f; +instruction[1] = 0xbc; +instruction[2] = 0xc0 | (tmp1_reg_ind << 3) | tmp1_reg_ind; +sljit_emit_op_custom(compiler, instruction, 3); + +OP2(SLJIT_ADD, STR_PTR, 0, STR_PTR, 0, TMP1, 0); + +add_jump(compiler, &common->failed_match, CMP(SLJIT_GREATER_EQUAL, STR_PTR, 0, STR_END, 0)); +return TRUE; +} + +#endif /* SLJIT_CONFIG_X86_64 && PCRE2_CODE_UNIT_WIDTH == 8 */ + /* The AVX2 code path is currently disabled. #define JIT_HAS_FAST_REQUESTED_CHAR_SIMD (sljit_has_cpu_feature(SLJIT_HAS_SIMD)) */