-
Notifications
You must be signed in to change notification settings - Fork 274
Scan character classes with SIMD in the JIT on x86-64 #941
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Open
mattst88
wants to merge
1
commit into
PCRE2Project:main
Choose a base branch
from
mattst88:jit-simd-class
base: main
Could not load branches
Branch not found: {{ refName }}
Loading
Could not load tags
Nothing to show
Loading
Are you sure you want to change the base?
Some commits from the old base branch may be removed from the timeline,
and old review comments may become outdated.
+244
−0
Open
Changes from all commits
Commits
File filter
Filter by extension
Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
There are no files selected for viewing
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -52,6 +52,52 @@ typedef enum { | |
| vector_compare_match2, | ||
| } vector_compare_type; | ||
|
|
||
| #if defined(SLJIT_CONFIG_X86_64) && SLJIT_CONFIG_X86_64 && PCRE2_CODE_UNIT_WIDTH == 8 | ||
|
|
||
| /* Largest number of ranges a start bitmap may be split into before the | ||
| vectorized scans give up on it. */ | ||
| #define MAX_START_BITS_RANGES 8 | ||
|
|
||
| typedef struct { | ||
| sljit_u32 low; | ||
| sljit_u32 high; | ||
| } start_bits_range; | ||
|
|
||
| /* Split a start bitmap into inclusive ranges of set bits, for the vectorized | ||
| start-bits scans below. Returns the number of ranges, or -1 if there are more | ||
| than max_ranges of them, in which case the caller must fall back to testing | ||
| the bitmap one code unit at a time. */ | ||
|
|
||
| static int extract_start_bits_ranges(const sljit_u8 *bits, start_bits_range *ranges, int max_ranges) | ||
| { | ||
| int count = 0; | ||
| int i = 0; | ||
|
|
||
| SLJIT_ASSERT(max_ranges <= MAX_START_BITS_RANGES); | ||
|
|
||
| while (i < 256) | ||
| { | ||
| if ((bits[i >> 3] & (1 << (i & 0x7))) == 0) | ||
| { | ||
| i++; | ||
| continue; | ||
| } | ||
|
|
||
| if (count >= max_ranges) | ||
| return -1; | ||
|
|
||
| ranges[count].low = (sljit_u32)i; | ||
| while (i < 256 && (bits[i >> 3] & (1 << (i & 0x7))) != 0) | ||
| i++; | ||
| ranges[count].high = (sljit_u32)(i - 1); | ||
| count++; | ||
| } | ||
|
|
||
| return count; | ||
| } | ||
|
|
||
| #endif /* SLJIT_CONFIG_X86_64 && PCRE2_CODE_UNIT_WIDTH == 8 */ | ||
|
|
||
| #if (defined SLJIT_CONFIG_X86 && SLJIT_CONFIG_X86) | ||
| static SLJIT_INLINE sljit_s32 max_fast_forward_char_pair_offset(void) | ||
| { | ||
|
|
@@ -373,6 +419,194 @@ if (common->utf && offset > 0) | |
| #endif | ||
| } | ||
|
|
||
| #if defined(SLJIT_CONFIG_X86_64) && SLJIT_CONFIG_X86_64 && PCRE2_CODE_UNIT_WIDTH == 8 | ||
|
|
||
| #define JIT_HAS_FAST_FORWARD_START_BITS_SIMD 1 | ||
|
|
||
| /* Number of ranges the vectorized start-bits scan will handle. Each one costs | ||
| two vector registers for its bounds; beyond this the byte-at-a-time bitmap | ||
| scan is likely to be the better option anyway, since every extra range adds | ||
| three instructions to each loop iteration. */ | ||
| #define X86_START_BITS_MAX_RANGES 4 | ||
|
|
||
| /* Largest number of accepted code units before the class counts as dense and | ||
| the byte-at-a-time scan is left to do the job. */ | ||
| #define X86_START_BITS_MAX_COVERED 48 | ||
|
|
||
| /* Emit an SSE2 instruction operating on two vector registers. A REX prefix is | ||
| needed for xmm8 and above, which the start-bits scan reaches as soon as it | ||
| holds the bounds of more than two ranges. */ | ||
| static SLJIT_INLINE void emit_sse2_op(struct sljit_compiler *compiler, sljit_u8 opcode, | ||
|
Collaborator
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Maybe such helpers could be used to improve code readability. |
||
| sljit_s32 dst_ind, sljit_s32 src_ind) | ||
| { | ||
| sljit_u8 instruction[5]; | ||
| int size = 0; | ||
|
|
||
| instruction[size++] = 0x66; | ||
|
|
||
| if (dst_ind >= 8 || src_ind >= 8) | ||
| instruction[size++] = (sljit_u8)(0x40 | ((dst_ind >= 8) ? 0x4 : 0) | ((src_ind >= 8) ? 0x1 : 0)); | ||
|
|
||
| instruction[size++] = 0x0f; | ||
| instruction[size++] = opcode; | ||
| instruction[size++] = (sljit_u8)(0xc0 | ((dst_ind & 0x7) << 3) | (src_ind & 0x7)); | ||
| sljit_emit_op_custom(compiler, instruction, size); | ||
| } | ||
|
|
||
| /* Emit the byte mask for one range into dst, given the data in src. | ||
| For a single character this is one PCMPEQB. For a range it is the standard | ||
| unsigned-range idiom: subtract the low bound so the range starts at zero, | ||
| then a saturating subtract of the span leaves zero exactly for the bytes | ||
| inside it. Values below the low bound wrap to something larger than the span, | ||
| so they saturate to non-zero and are correctly excluded. */ | ||
| static void emit_x86_class_range(struct sljit_compiler *compiler, sljit_s32 dst_ind, | ||
| sljit_s32 src_ind, sljit_s32 zero_ind, sljit_s32 low_ind, sljit_s32 span_ind, BOOL single) | ||
| { | ||
| /* MOVDQA dst, src */ | ||
| emit_sse2_op(compiler, 0x6f, dst_ind, src_ind); | ||
|
|
||
| if (single) | ||
| { | ||
| /* PCMPEQB dst, low */ | ||
| emit_sse2_op(compiler, 0x74, dst_ind, low_ind); | ||
| return; | ||
| } | ||
|
|
||
| /* PSUBB dst, low */ | ||
| emit_sse2_op(compiler, 0xf8, dst_ind, low_ind); | ||
| /* PSUBUSB dst, span */ | ||
| emit_sse2_op(compiler, 0xd8, dst_ind, span_ind); | ||
| /* PCMPEQB dst, zero */ | ||
| emit_sse2_op(compiler, 0x74, dst_ind, zero_ind); | ||
|
Collaborator
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. This looks like a good idea |
||
| } | ||
|
|
||
| /* Scan for the first code unit whose start bit is set, 16 bytes at a time. | ||
| Returns FALSE without emitting anything if the bitmap does not reduce to few | ||
| enough ranges, in which case the caller emits its own scan. */ | ||
| static BOOL fast_forward_start_bits_simd(compiler_common *common, const sljit_u8 *start_bits) | ||
| { | ||
| DEFINE_COMPILER; | ||
| sljit_u8 instruction[4]; | ||
| sljit_s32 reg_type = SLJIT_SIMD_REG_128; | ||
| struct sljit_label *start; | ||
| struct sljit_jump *quit; | ||
| start_bits_range ranges[X86_START_BITS_MAX_RANGES]; | ||
| sljit_s32 tmp1_reg_ind = sljit_get_register_index(SLJIT_GP_REGISTER, TMP1); | ||
| sljit_s32 data_ind = sljit_get_register_index(reg_type, SLJIT_VR0); | ||
| sljit_s32 zero_ind = sljit_get_register_index(reg_type, SLJIT_VR1); | ||
| sljit_s32 acc_ind = sljit_get_register_index(reg_type, SLJIT_VR2); | ||
| sljit_s32 tmp_ind = sljit_get_register_index(reg_type, SLJIT_VR3); | ||
| sljit_s32 low_ind[X86_START_BITS_MAX_RANGES]; | ||
| sljit_s32 span_ind[X86_START_BITS_MAX_RANGES]; | ||
| BOOL single[X86_START_BITS_MAX_RANGES]; | ||
| sljit_s32 value; | ||
| int count, covered, k; | ||
|
|
||
| count = extract_start_bits_ranges(start_bits, ranges, X86_START_BITS_MAX_RANGES); | ||
| if (count <= 0) | ||
| return FALSE; | ||
|
|
||
| /* A vector scan only pays when it gets to scan. The byte-at-a-time loop stops | ||
| at the first code unit in the class, so for a dense class it usually stops | ||
| immediately, while this loop has already tested a whole block. Count the | ||
| code units the class accepts and leave the dense ones alone: \w covers 63 of | ||
| them and is common enough in real subjects to matter. */ | ||
| covered = 0; | ||
| for (k = 0; k < count; k++) | ||
| covered += (int)(ranges[k].high - ranges[k].low) + 1; | ||
|
|
||
| if (covered > X86_START_BITS_MAX_COVERED) | ||
| return FALSE; | ||
|
|
||
| /* Load the bounds. A single character needs only one register, so pack the | ||
| pairs from VR4 upwards and record what each range got. */ | ||
| value = SLJIT_SIMD_REG_128 | SLJIT_SIMD_ELEM_32 | SLJIT_SIMD_LANE_ZERO; | ||
|
|
||
| for (k = 0; k < count; k++) | ||
| { | ||
| sljit_s32 low_reg = SLJIT_VR4 + (k * 2); | ||
| sljit_s32 span_reg = SLJIT_VR4 + (k * 2) + 1; | ||
|
|
||
| single[k] = (ranges[k].low == ranges[k].high); | ||
|
|
||
| sljit_emit_simd_lane_mov(compiler, value, low_reg, 0, SLJIT_IMM, | ||
| character_to_int32((PCRE2_UCHAR)ranges[k].low)); | ||
| sljit_emit_simd_lane_replicate(compiler, reg_type | SLJIT_SIMD_ELEM_32, low_reg, low_reg, 0); | ||
| low_ind[k] = sljit_get_register_index(reg_type, low_reg); | ||
|
|
||
| span_ind[k] = low_ind[k]; | ||
| if (!single[k]) | ||
| { | ||
| sljit_emit_simd_lane_mov(compiler, value, span_reg, 0, SLJIT_IMM, | ||
| character_to_int32((PCRE2_UCHAR)(ranges[k].high - ranges[k].low))); | ||
| sljit_emit_simd_lane_replicate(compiler, reg_type | SLJIT_SIMD_ELEM_32, span_reg, span_reg, 0); | ||
| span_ind[k] = sljit_get_register_index(reg_type, span_reg); | ||
| } | ||
| } | ||
|
|
||
| /* PXOR zero, zero */ | ||
| emit_sse2_op(compiler, 0xef, zero_ind, zero_ind); | ||
|
|
||
| OP1(SLJIT_MOV, TMP2, 0, STR_PTR, 0); | ||
|
|
||
| /* First part (unaligned start). */ | ||
| OP2(SLJIT_AND, STR_PTR, 0, STR_PTR, 0, SLJIT_IMM, ~0xf); | ||
| OP2(SLJIT_AND, TMP2, 0, TMP2, 0, SLJIT_IMM, 0xf); | ||
|
|
||
| sljit_emit_simd_mov(compiler, reg_type | SLJIT_SIMD_MEM_ALIGNED_128, SLJIT_VR0, SLJIT_MEM1(STR_PTR), 0); | ||
|
|
||
| emit_x86_class_range(compiler, acc_ind, data_ind, zero_ind, low_ind[0], span_ind[0], single[0]); | ||
| for (k = 1; k < count; k++) | ||
| { | ||
| emit_x86_class_range(compiler, tmp_ind, data_ind, zero_ind, low_ind[k], span_ind[k], single[k]); | ||
| /* POR acc, tmp */ | ||
| emit_sse2_op(compiler, 0xeb, acc_ind, tmp_ind); | ||
| } | ||
|
|
||
| sljit_emit_simd_sign(compiler, SLJIT_SIMD_STORE | reg_type | SLJIT_SIMD_ELEM_8, SLJIT_VR2, TMP1, 0); | ||
| OP2(SLJIT_ADD, STR_PTR, 0, STR_PTR, 0, TMP2, 0); | ||
| OP2(SLJIT_LSHR, TMP1, 0, TMP1, 0, TMP2, 0); | ||
|
|
||
| quit = CMP(SLJIT_NOT_ZERO, TMP1, 0, SLJIT_IMM, 0); | ||
|
|
||
| OP2(SLJIT_SUB, STR_PTR, 0, STR_PTR, 0, TMP2, 0); | ||
|
|
||
| /* Second part (aligned). */ | ||
| start = LABEL(); | ||
|
|
||
| OP2(SLJIT_ADD, STR_PTR, 0, STR_PTR, 0, SLJIT_IMM, 16); | ||
|
|
||
| add_jump(compiler, &common->failed_match, CMP(SLJIT_GREATER_EQUAL, STR_PTR, 0, STR_END, 0)); | ||
|
|
||
| sljit_emit_simd_mov(compiler, reg_type | SLJIT_SIMD_MEM_ALIGNED_128, SLJIT_VR0, SLJIT_MEM1(STR_PTR), 0); | ||
|
|
||
| emit_x86_class_range(compiler, acc_ind, data_ind, zero_ind, low_ind[0], span_ind[0], single[0]); | ||
| for (k = 1; k < count; k++) | ||
| { | ||
| emit_x86_class_range(compiler, tmp_ind, data_ind, zero_ind, low_ind[k], span_ind[k], single[k]); | ||
| emit_sse2_op(compiler, 0xeb, acc_ind, tmp_ind); | ||
| } | ||
|
|
||
| sljit_emit_simd_sign(compiler, SLJIT_SIMD_STORE | reg_type | SLJIT_SIMD_ELEM_8, SLJIT_VR2, TMP1, 0); | ||
| CMPTO(SLJIT_ZERO, TMP1, 0, SLJIT_IMM, 0, start); | ||
|
|
||
| JUMPHERE(quit); | ||
|
|
||
| SLJIT_ASSERT(tmp1_reg_ind < 8); | ||
| /* BSF r32, r/m32 */ | ||
| instruction[0] = 0x0f; | ||
| instruction[1] = 0xbc; | ||
| instruction[2] = 0xc0 | (tmp1_reg_ind << 3) | tmp1_reg_ind; | ||
| sljit_emit_op_custom(compiler, instruction, 3); | ||
|
|
||
| OP2(SLJIT_ADD, STR_PTR, 0, STR_PTR, 0, TMP1, 0); | ||
|
|
||
| add_jump(compiler, &common->failed_match, CMP(SLJIT_GREATER_EQUAL, STR_PTR, 0, STR_END, 0)); | ||
| return TRUE; | ||
| } | ||
|
|
||
| #endif /* SLJIT_CONFIG_X86_64 && PCRE2_CODE_UNIT_WIDTH == 8 */ | ||
|
|
||
| /* The AVX2 code path is currently disabled. | ||
| #define JIT_HAS_FAST_REQUESTED_CHAR_SIMD (sljit_has_cpu_feature(SLJIT_HAS_SIMD)) | ||
| */ | ||
|
|
||
Oops, something went wrong.
Add this suggestion to a batch that can be applied as a single commit.
This suggestion is invalid because no changes were made to the code.
Suggestions cannot be applied while the pull request is closed.
Suggestions cannot be applied while viewing a subset of changes.
Only one suggestion per line can be applied in a batch.
Add this suggestion to a batch that can be applied as a single commit.
Applying suggestions on deleted lines is not supported.
You must change the existing code in this line in order to create a valid suggestion.
Outdated suggestions cannot be applied.
This suggestion has been applied or marked resolved.
Suggestions cannot be applied from pending reviews.
Suggestions cannot be applied on multi-line comments.
Suggestions cannot be applied while the pull request is queued to merge.
Suggestion cannot be applied right now. Please check back later.
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
I think there is a similar function in the code somewhere. Not sure it can be used here.