/[pcre]/code/tags/pcre-8.37/pcre_jit_compile.c
ViewVC logotype

Diff of /code/tags/pcre-8.37/pcre_jit_compile.c

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1434 by zherczeg, Mon Jan 6 20:04:50 2014 UTC revision 1474 by zherczeg, Thu Apr 24 06:43:50 2014 UTC
# Line 362  typedef struct compiler_common { Line 362  typedef struct compiler_common {
362    sljit_sw lcc;    sljit_sw lcc;
363    /* Mode can be PCRE_STUDY_JIT_COMPILE and others. */    /* Mode can be PCRE_STUDY_JIT_COMPILE and others. */
364    int mode;    int mode;
365      /* TRUE, when minlength is greater than 0. */
366      BOOL might_be_empty;
367    /* \K is found in the pattern. */    /* \K is found in the pattern. */
368    BOOL has_set_som;    BOOL has_set_som;
369    /* (*SKIP:arg) is found in the pattern. */    /* (*SKIP:arg) is found in the pattern. */
# Line 396  typedef struct compiler_common { Line 398  typedef struct compiler_common {
398    struct sljit_label *quit_label;    struct sljit_label *quit_label;
399    struct sljit_label *forced_quit_label;    struct sljit_label *forced_quit_label;
400    struct sljit_label *accept_label;    struct sljit_label *accept_label;
401      struct sljit_label *ff_newline_shortcut;
402    stub_list *stubs;    stub_list *stubs;
403    label_addr_list *label_addrs;    label_addr_list *label_addrs;
404    recurse_entry *entries;    recurse_entry *entries;
# Line 789  while (cc < ccend) Line 792  while (cc < ccend)
792      {      {
793      case OP_SET_SOM:      case OP_SET_SOM:
794      common->has_set_som = TRUE;      common->has_set_som = TRUE;
795        common->might_be_empty = TRUE;
796      cc += 1;      cc += 1;
797      break;      break;
798    
# Line 2564  if (common->utf) Line 2568  if (common->utf)
2568    
2569  #if defined SUPPORT_UTF && defined COMPILE_PCRE8  #if defined SUPPORT_UTF && defined COMPILE_PCRE8
2570    
2571  static BOOL is_char7_bitset(const pcre_uint8* bitset, BOOL nclass)  static BOOL is_char7_bitset(const pcre_uint8 *bitset, BOOL nclass)
2572  {  {
2573  /* Tells whether the character codes below 128 are enough  /* Tells whether the character codes below 128 are enough
2574  to determine a match. */  to determine a match. */
# Line 3146  if (newlinecheck) Line 3150  if (newlinecheck)
3150  return mainloop;  return mainloop;
3151  }  }
3152    
3153  static int scan_prefix(compiler_common *common, pcre_uchar *cc, pcre_uint32 *chars, int max_chars)  #define MAX_N_CHARS 16
3154    #define MAX_N_BYTES 8
3155    
3156    static SLJIT_INLINE void add_prefix_byte(pcre_uint8 byte, pcre_uint8 *bytes)
3157    {
3158    pcre_uint8 len = bytes[0];
3159    int i;
3160    
3161    if (len == 255)
3162      return;
3163    
3164    if (len == 0)
3165      {
3166      bytes[0] = 1;
3167      bytes[1] = byte;
3168      return;
3169      }
3170    
3171    for (i = len; i > 0; i--)
3172      if (bytes[i] == byte)
3173        return;
3174    
3175    if (len >= MAX_N_BYTES - 1)
3176      {
3177      bytes[0] = 255;
3178      return;
3179      }
3180    
3181    len++;
3182    bytes[len] = byte;
3183    bytes[0] = len;
3184    }
3185    
3186    static int scan_prefix(compiler_common *common, pcre_uchar *cc, pcre_uint32 *chars, pcre_uint8 *bytes, int max_chars)
3187  {  {
3188  /* Recursive function, which scans prefix literals. */  /* Recursive function, which scans prefix literals. */
3189    BOOL last, any, caseless;
3190  int len, repeat, len_save, consumed = 0;  int len, repeat, len_save, consumed = 0;
3191  pcre_uint32 caseless, chr, mask;  pcre_uint32 chr, mask;
3192  pcre_uchar *alternative, *cc_save;  pcre_uchar *alternative, *cc_save, *oc;
3193  BOOL last, any;  #if defined SUPPORT_UTF && defined COMPILE_PCRE8
3194    pcre_uchar othercase[8];
3195    #elif defined SUPPORT_UTF && defined COMPILE_PCRE16
3196    pcre_uchar othercase[2];
3197    #else
3198    pcre_uchar othercase[1];
3199    #endif
3200    
3201  repeat = 1;  repeat = 1;
3202  while (TRUE)  while (TRUE)
3203    {    {
3204    last = TRUE;    last = TRUE;
3205    any = FALSE;    any = FALSE;
3206    caseless = 0;    caseless = FALSE;
3207    switch (*cc)    switch (*cc)
3208      {      {
3209      case OP_CHARI:      case OP_CHARI:
3210      caseless = 1;      caseless = TRUE;
3211      case OP_CHAR:      case OP_CHAR:
3212      last = FALSE;      last = FALSE;
3213      cc++;      cc++;
# Line 3184  while (TRUE) Line 3228  while (TRUE)
3228      cc++;      cc++;
3229      continue;      continue;
3230    
3231        case OP_ASSERT:
3232        case OP_ASSERT_NOT:
3233        case OP_ASSERTBACK:
3234        case OP_ASSERTBACK_NOT:
3235        cc = bracketend(cc);
3236        continue;
3237    
3238        case OP_PLUSI:
3239        case OP_MINPLUSI:
3240        case OP_POSPLUSI:
3241        caseless = TRUE;
3242      case OP_PLUS:      case OP_PLUS:
3243      case OP_MINPLUS:      case OP_MINPLUS:
3244      case OP_POSPLUS:      case OP_POSPLUS:
# Line 3191  while (TRUE) Line 3246  while (TRUE)
3246      break;      break;
3247    
3248      case OP_EXACTI:      case OP_EXACTI:
3249      caseless = 1;      caseless = TRUE;
3250      case OP_EXACT:      case OP_EXACT:
3251      repeat = GET2(cc, 1);      repeat = GET2(cc, 1);
3252      last = FALSE;      last = FALSE;
3253      cc += 1 + IMM2_SIZE;      cc += 1 + IMM2_SIZE;
3254      break;      break;
3255    
3256      case OP_PLUSI:      case OP_QUERYI:
3257      case OP_MINPLUSI:      case OP_MINQUERYI:
3258      case OP_POSPLUSI:      case OP_POSQUERYI:
3259      caseless = 1;      caseless = TRUE;
3260        case OP_QUERY:
3261        case OP_MINQUERY:
3262        case OP_POSQUERY:
3263        len = 1;
3264      cc++;      cc++;
3265    #ifdef SUPPORT_UTF
3266        if (common->utf && HAS_EXTRALEN(*cc)) len += GET_EXTRALEN(*cc);
3267    #endif
3268        max_chars = scan_prefix(common, cc + len, chars, bytes, max_chars);
3269        if (max_chars == 0)
3270          return consumed;
3271        last = FALSE;
3272      break;      break;
3273    
3274      case OP_KET:      case OP_KET:
# Line 3222  while (TRUE) Line 3288  while (TRUE)
3288      alternative = cc + GET(cc, 1);      alternative = cc + GET(cc, 1);
3289      while (*alternative == OP_ALT)      while (*alternative == OP_ALT)
3290        {        {
3291        max_chars = scan_prefix(common, alternative + 1 + LINK_SIZE, chars, max_chars);        max_chars = scan_prefix(common, alternative + 1 + LINK_SIZE, chars, bytes, max_chars);
3292        if (max_chars == 0)        if (max_chars == 0)
3293          return consumed;          return consumed;
3294        alternative += GET(alternative, 1);        alternative += GET(alternative, 1);
# Line 3234  while (TRUE) Line 3300  while (TRUE)
3300      continue;      continue;
3301    
3302      case OP_CLASS:      case OP_CLASS:
3303    #if defined SUPPORT_UTF && defined COMPILE_PCRE8
3304        if (common->utf && !is_char7_bitset((const pcre_uint8 *)(cc + 1), FALSE)) return consumed;
3305    #endif
3306        any = TRUE;
3307        cc += 1 + 32 / sizeof(pcre_uchar);
3308        break;
3309    
3310      case OP_NCLASS:      case OP_NCLASS:
3311    #if defined SUPPORT_UTF && !defined COMPILE_PCRE32
3312        if (common->utf) return consumed;
3313    #endif
3314      any = TRUE;      any = TRUE;
3315      cc += 1 + 32 / sizeof(pcre_uchar);      cc += 1 + 32 / sizeof(pcre_uchar);
3316      break;      break;
3317    
3318  #if defined SUPPORT_UTF || !defined COMPILE_PCRE8  #if defined SUPPORT_UTF || !defined COMPILE_PCRE8
3319      case OP_XCLASS:      case OP_XCLASS:
3320    #if defined SUPPORT_UTF && !defined COMPILE_PCRE32
3321        if (common->utf) return consumed;
3322    #endif
3323      any = TRUE;      any = TRUE;
3324      cc += GET(cc, 1);      cc += GET(cc, 1);
3325      break;      break;
3326  #endif  #endif
3327    
     case OP_NOT_DIGIT:  
3328      case OP_DIGIT:      case OP_DIGIT:
3329      case OP_NOT_WHITESPACE:  #if defined SUPPORT_UTF && defined COMPILE_PCRE8
3330        if (common->utf && !is_char7_bitset((const pcre_uint8 *)common->ctypes - cbit_length + cbit_digit, FALSE))
3331          return consumed;
3332    #endif
3333        any = TRUE;
3334        cc++;
3335        break;
3336    
3337      case OP_WHITESPACE:      case OP_WHITESPACE:
3338      case OP_NOT_WORDCHAR:  #if defined SUPPORT_UTF && defined COMPILE_PCRE8
3339        if (common->utf && !is_char7_bitset((const pcre_uint8 *)common->ctypes - cbit_length + cbit_space, FALSE))
3340          return consumed;
3341    #endif
3342        any = TRUE;
3343        cc++;
3344        break;
3345    
3346      case OP_WORDCHAR:      case OP_WORDCHAR:
3347    #if defined SUPPORT_UTF && defined COMPILE_PCRE8
3348        if (common->utf && !is_char7_bitset((const pcre_uint8 *)common->ctypes - cbit_length + cbit_word, FALSE))
3349          return consumed;
3350    #endif
3351        any = TRUE;
3352        cc++;
3353        break;
3354    
3355        case OP_NOT:
3356        case OP_NOTI:
3357        cc++;
3358        /* Fall through. */
3359        case OP_NOT_DIGIT:
3360        case OP_NOT_WHITESPACE:
3361        case OP_NOT_WORDCHAR:
3362      case OP_ANY:      case OP_ANY:
3363      case OP_ALLANY:      case OP_ALLANY:
3364    #if defined SUPPORT_UTF && !defined COMPILE_PCRE32
3365        if (common->utf) return consumed;
3366    #endif
3367      any = TRUE;      any = TRUE;
3368      cc++;      cc++;
3369      break;      break;
# Line 3261  while (TRUE) Line 3371  while (TRUE)
3371  #ifdef SUPPORT_UCP  #ifdef SUPPORT_UCP
3372      case OP_NOTPROP:      case OP_NOTPROP:
3373      case OP_PROP:      case OP_PROP:
3374    #if defined SUPPORT_UTF && !defined COMPILE_PCRE32
3375        if (common->utf) return consumed;
3376    #endif
3377      any = TRUE;      any = TRUE;
3378      cc += 1 + 2;      cc += 1 + 2;
3379      break;      break;
# Line 3271  while (TRUE) Line 3384  while (TRUE)
3384      cc += 1 + IMM2_SIZE;      cc += 1 + IMM2_SIZE;
3385      continue;      continue;
3386    
3387        case OP_NOTEXACT:
3388        case OP_NOTEXACTI:
3389    #if defined SUPPORT_UTF && !defined COMPILE_PCRE32
3390        if (common->utf) return consumed;
3391    #endif
3392        any = TRUE;
3393        repeat = GET2(cc, 1);
3394        cc += 1 + IMM2_SIZE + 1;
3395        break;
3396    
3397      default:      default:
3398      return consumed;      return consumed;
3399      }      }
3400    
3401    if (any)    if (any)
3402      {      {
 #ifdef SUPPORT_UTF  
     if (common->utf) return consumed;  
 #endif  
3403  #if defined COMPILE_PCRE8  #if defined COMPILE_PCRE8
3404      mask = 0xff;      mask = 0xff;
3405  #elif defined COMPILE_PCRE16  #elif defined COMPILE_PCRE16
# Line 3294  while (TRUE) Line 3414  while (TRUE)
3414        {        {
3415        chars[0] = mask;        chars[0] = mask;
3416        chars[1] = mask;        chars[1] = mask;
3417          bytes[0] = 255;
3418    
3419          consumed++;
3420        if (--max_chars == 0)        if (--max_chars == 0)
3421          return consumed;          return consumed;
       consumed++;  
3422        chars += 2;        chars += 2;
3423          bytes += MAX_N_BYTES;
3424        }        }
3425      while (--repeat > 0);      while (--repeat > 0);
3426    
# Line 3311  while (TRUE) Line 3433  while (TRUE)
3433    if (common->utf && HAS_EXTRALEN(*cc)) len += GET_EXTRALEN(*cc);    if (common->utf && HAS_EXTRALEN(*cc)) len += GET_EXTRALEN(*cc);
3434  #endif  #endif
3435    
3436    if (caseless != 0 && char_has_othercase(common, cc))    if (caseless && char_has_othercase(common, cc))
3437      {      {
3438      caseless = char_get_othercase_bit(common, cc);  #ifdef SUPPORT_UTF
3439      if (caseless == 0)      if (common->utf)
3440        return consumed;        {
3441  #ifdef COMPILE_PCRE8        GETCHAR(chr, cc);
3442      caseless = ((caseless & 0xff) << 8) | (len - (caseless >> 8));        if ((int)PRIV(ord2utf)(char_othercase(common, chr), othercase) != len)
3443  #else          return consumed;
3444      if ((caseless & 0x100) != 0)        }
       caseless = ((caseless & 0xff) << 16) | (len - (caseless >> 9));  
3445      else      else
       caseless = ((caseless & 0xff) << 8) | (len - (caseless >> 9));  
3446  #endif  #endif
3447          {
3448          chr = *cc;
3449          othercase[0] = TABLE_GET(chr, common->fcc, chr);
3450          }
3451      }      }
3452    else    else
3453      caseless = 0;      caseless = FALSE;
3454    
3455    len_save = len;    len_save = len;
3456    cc_save = cc;    cc_save = cc;
3457    while (TRUE)    while (TRUE)
3458      {      {
3459        oc = othercase;
3460      do      do
3461        {        {
3462        chr = *cc;        chr = *cc;
# Line 3339  while (TRUE) Line 3464  while (TRUE)
3464        if (SLJIT_UNLIKELY(chr == NOTACHAR))        if (SLJIT_UNLIKELY(chr == NOTACHAR))
3465          return consumed;          return consumed;
3466  #endif  #endif
3467          add_prefix_byte((pcre_uint8)chr, bytes);
3468    
3469        mask = 0;        mask = 0;
3470        if ((pcre_uint32)len == (caseless & 0xff))        if (caseless)
3471          {          {
3472          mask = caseless >> 8;          add_prefix_byte((pcre_uint8)*oc, bytes);
3473            mask = *cc ^ *oc;
3474          chr |= mask;          chr |= mask;
3475          }          }
3476    
3477    #ifdef COMPILE_PCRE32
3478          if (chars[0] == NOTACHAR && chars[1] == 0)
3479    #else
3480        if (chars[0] == NOTACHAR)        if (chars[0] == NOTACHAR)
3481    #endif
3482          {          {
3483          chars[0] = chr;          chars[0] = chr;
3484          chars[1] = mask;          chars[1] = mask;
# Line 3360  while (TRUE) Line 3492  while (TRUE)
3492          }          }
3493    
3494        len--;        len--;
3495          consumed++;
3496        if (--max_chars == 0)        if (--max_chars == 0)
3497          return consumed;          return consumed;
       consumed++;  
3498        chars += 2;        chars += 2;
3499          bytes += MAX_N_BYTES;
3500        cc++;        cc++;
3501          oc++;
3502        }        }
3503      while (len > 0);      while (len > 0);
3504    
# Line 3381  while (TRUE) Line 3515  while (TRUE)
3515    }    }
3516  }  }
3517    
 #define MAX_N_CHARS 16  
   
3518  static SLJIT_INLINE BOOL fast_forward_first_n_chars(compiler_common *common, BOOL firstline)  static SLJIT_INLINE BOOL fast_forward_first_n_chars(compiler_common *common, BOOL firstline)
3519  {  {
3520  DEFINE_COMPILER;  DEFINE_COMPILER;
3521  struct sljit_label *start;  struct sljit_label *start;
3522  struct sljit_jump *quit;  struct sljit_jump *quit;
3523  pcre_uint32 chars[MAX_N_CHARS * 2];  pcre_uint32 chars[MAX_N_CHARS * 2];
3524    pcre_uint8 bytes[MAX_N_CHARS * MAX_N_BYTES];
3525  pcre_uint8 ones[MAX_N_CHARS];  pcre_uint8 ones[MAX_N_CHARS];
 pcre_uint32 mask;  
 int i, max;  
3526  int offsets[3];  int offsets[3];
3527    pcre_uint32 mask;
3528    pcre_uint8 *byte_set, *byte_set_end;
3529    int i, max, from;
3530    int range_right = -1, range_len = 3 - 1;
3531    sljit_ub *update_table = NULL;
3532    BOOL in_range;
3533    
3534    /* This is even TRUE, if both are NULL. */
3535    SLJIT_ASSERT(common->read_only_data_ptr == common->read_only_data);
3536    
3537  for (i = 0; i < MAX_N_CHARS; i++)  for (i = 0; i < MAX_N_CHARS; i++)
3538    {    {
3539    chars[i << 1] = NOTACHAR;    chars[i << 1] = NOTACHAR;
3540    chars[(i << 1) + 1] = 0;    chars[(i << 1) + 1] = 0;
3541      bytes[i * MAX_N_BYTES] = 0;
3542    }    }
3543    
3544  max = scan_prefix(common, common->start, chars, MAX_N_CHARS);  max = scan_prefix(common, common->start, chars, bytes, MAX_N_CHARS);
3545    
3546  if (max <= 1)  if (max <= 1)
3547    return FALSE;    return FALSE;
# Line 3417  for (i = 0; i < max; i++) Line 3558  for (i = 0; i < max; i++)
3558      }      }
3559    }    }
3560    
3561    in_range = FALSE;
3562    from = 0;   /* Prevent compiler "uninitialized" warning */
3563    for (i = 0; i <= max; i++)
3564      {
3565      if (in_range && (i - from) > range_len && (bytes[(i - 1) * MAX_N_BYTES] <= 4))
3566        {
3567        range_len = i - from;
3568        range_right = i - 1;
3569        }
3570    
3571      if (i < max && bytes[i * MAX_N_BYTES] < 255)
3572        {
3573        if (!in_range)
3574          {
3575          in_range = TRUE;
3576          from = i;
3577          }
3578        }
3579      else if (in_range)
3580        in_range = FALSE;
3581      }
3582    
3583    if (range_right >= 0)
3584      {
3585      /* Since no data is consumed (see the assert in the beginning
3586      of this function), this space can be reallocated. */
3587      if (common->read_only_data)
3588        SLJIT_FREE(common->read_only_data);
3589    
3590      common->read_only_data_size += 256;
3591      common->read_only_data = (sljit_uw *)SLJIT_MALLOC(common->read_only_data_size);
3592      if (common->read_only_data == NULL)
3593        return TRUE;
3594    
3595      update_table = (sljit_ub *)common->read_only_data;
3596      common->read_only_data_ptr = (sljit_uw *)(update_table + 256);
3597      memset(update_table, IN_UCHARS(range_len), 256);
3598    
3599      for (i = 0; i < range_len; i++)
3600        {
3601        byte_set = bytes + ((range_right - i) * MAX_N_BYTES);
3602        SLJIT_ASSERT(byte_set[0] > 0 && byte_set[0] < 255);
3603        byte_set_end = byte_set + byte_set[0];
3604        byte_set++;
3605        while (byte_set <= byte_set_end)
3606          {
3607          if (update_table[*byte_set] > IN_UCHARS(i))
3608            update_table[*byte_set] = IN_UCHARS(i);
3609          byte_set++;
3610          }
3611        }
3612      }
3613    
3614  offsets[0] = -1;  offsets[0] = -1;
3615  /* Scan forward. */  /* Scan forward. */
3616  for (i = 0; i < max; i++)  for (i = 0; i < max; i++)
# Line 3425  for (i = 0; i < max; i++) Line 3619  for (i = 0; i < max; i++)
3619      break;      break;
3620    }    }
3621    
3622  if (offsets[0] == -1)  if (offsets[0] < 0 && range_right < 0)
3623    return FALSE;    return FALSE;
3624    
3625  /* Scan backward. */  if (offsets[0] >= 0)
 offsets[1] = -1;  
 for (i = max - 1; i > offsets[0]; i--)  
   if (ones[i] <= 2) {  
     offsets[1] = i;  
     break;  
   }  
   
 offsets[2] = -1;  
 if (offsets[1] >= 0)  
3626    {    {
3627    /* Scan from middle. */    /* Scan backward. */
3628    for (i = (offsets[0] + offsets[1]) / 2 + 1; i < offsets[1]; i++)    offsets[1] = -1;
3629      if (ones[i] <= 2)    for (i = max - 1; i > offsets[0]; i--)
3630        if (ones[i] <= 2 && i != range_right)
3631        {        {
3632        offsets[2] = i;        offsets[1] = i;
3633        break;        break;
3634        }        }
3635    
3636    if (offsets[2] == -1)    /* This case is handled better by fast_forward_first_char. */
3637      if (offsets[1] == -1 && offsets[0] == 0 && range_right < 0)
3638        return FALSE;
3639    
3640      offsets[2] = -1;
3641      /* We only search for a middle character if there is no range check. */
3642      if (offsets[1] >= 0 && range_right == -1)
3643      {      {
3644      for (i = (offsets[0] + offsets[1]) / 2; i > offsets[0]; i--)      /* Scan from middle. */
3645        for (i = (offsets[0] + offsets[1]) / 2 + 1; i < offsets[1]; i++)
3646        if (ones[i] <= 2)        if (ones[i] <= 2)
3647          {          {
3648          offsets[2] = i;          offsets[2] = i;
3649          break;          break;
3650          }          }
3651    
3652        if (offsets[2] == -1)
3653          {
3654          for (i = (offsets[0] + offsets[1]) / 2; i > offsets[0]; i--)
3655            if (ones[i] <= 2)
3656              {
3657              offsets[2] = i;
3658              break;
3659              }
3660          }
3661      }      }
   }  
3662    
3663  SLJIT_ASSERT(offsets[1] == -1 || (offsets[0] < offsets[1]));    SLJIT_ASSERT(offsets[1] == -1 || (offsets[0] < offsets[1]));
3664  SLJIT_ASSERT(offsets[2] == -1 || (offsets[0] < offsets[2] && offsets[1] > offsets[2]));    SLJIT_ASSERT(offsets[2] == -1 || (offsets[0] < offsets[2] && offsets[1] > offsets[2]));
3665    
3666  chars[0] = chars[offsets[0] << 1];    chars[0] = chars[offsets[0] << 1];
3667  chars[1] = chars[(offsets[0] << 1) + 1];    chars[1] = chars[(offsets[0] << 1) + 1];
3668  if (offsets[2] >= 0)    if (offsets[2] >= 0)
3669    {      {
3670    chars[2] = chars[offsets[2] << 1];      chars[2] = chars[offsets[2] << 1];
3671    chars[3] = chars[(offsets[2] << 1) + 1];      chars[3] = chars[(offsets[2] << 1) + 1];
3672    }      }
3673  if (offsets[1] >= 0)    if (offsets[1] >= 0)
3674    {      {
3675    chars[4] = chars[offsets[1] << 1];      chars[4] = chars[offsets[1] << 1];
3676    chars[5] = chars[(offsets[1] << 1) + 1];      chars[5] = chars[(offsets[1] << 1) + 1];
3677        }
3678    }    }
3679    
3680  max -= 1;  max -= 1;
3681  if (firstline)  if (firstline)
3682    {    {
3683    SLJIT_ASSERT(common->first_line_end != 0);    SLJIT_ASSERT(common->first_line_end != 0);
3684      OP1(SLJIT_MOV, TMP1, 0, SLJIT_MEM1(SLJIT_LOCALS_REG), common->first_line_end);
3685    OP1(SLJIT_MOV, TMP3, 0, STR_END, 0);    OP1(SLJIT_MOV, TMP3, 0, STR_END, 0);
3686    OP2(SLJIT_SUB, STR_END, 0, SLJIT_MEM1(SLJIT_LOCALS_REG), common->first_line_end, SLJIT_IMM, IN_UCHARS(max));    OP2(SLJIT_SUB, STR_END, 0, STR_END, 0, SLJIT_IMM, IN_UCHARS(max));
3687      quit = CMP(SLJIT_C_LESS_EQUAL, STR_END, 0, TMP1, 0);
3688      OP1(SLJIT_MOV, STR_END, 0, TMP1, 0);
3689      JUMPHERE(quit);
3690    }    }
3691  else  else
3692    OP2(SLJIT_SUB, STR_END, 0, STR_END, 0, SLJIT_IMM, IN_UCHARS(max));    OP2(SLJIT_SUB, STR_END, 0, STR_END, 0, SLJIT_IMM, IN_UCHARS(max));
3693    
3694    #if !(defined SLJIT_CONFIG_X86_32 && SLJIT_CONFIG_X86_32)
3695    if (range_right >= 0)
3696      OP1(SLJIT_MOV, RETURN_ADDR, 0, SLJIT_IMM, (sljit_sw)update_table);
3697    #endif
3698    
3699  start = LABEL();  start = LABEL();
3700  quit = CMP(SLJIT_C_GREATER_EQUAL, STR_PTR, 0, STR_END, 0);  quit = CMP(SLJIT_C_GREATER_EQUAL, STR_PTR, 0, STR_END, 0);
3701    
3702  OP1(MOV_UCHAR, TMP1, 0, SLJIT_MEM1(STR_PTR), IN_UCHARS(offsets[0]));  SLJIT_ASSERT(range_right >= 0 || offsets[0] >= 0);
 if (offsets[1] >= 0)  
   OP1(MOV_UCHAR, TMP2, 0, SLJIT_MEM1(STR_PTR), IN_UCHARS(offsets[1]));  
 OP2(SLJIT_ADD, STR_PTR, 0, STR_PTR, 0, SLJIT_IMM, IN_UCHARS(1));  
   
 if (chars[1] != 0)  
   OP2(SLJIT_OR, TMP1, 0, TMP1, 0, SLJIT_IMM, chars[1]);  
 CMPTO(SLJIT_C_NOT_EQUAL, TMP1, 0, SLJIT_IMM, chars[0], start);  
 if (offsets[2] >= 0)  
   OP1(MOV_UCHAR, TMP1, 0, SLJIT_MEM1(STR_PTR), IN_UCHARS(offsets[2] - 1));  
3703    
3704  if (offsets[1] >= 0)  if (range_right >= 0)
3705    {    {
3706    if (chars[5] != 0)  #if defined COMPILE_PCRE8 || (defined SLJIT_LITTLE_ENDIAN && SLJIT_LITTLE_ENDIAN)
3707      OP2(SLJIT_OR, TMP2, 0, TMP2, 0, SLJIT_IMM, chars[5]);    OP1(SLJIT_MOV_UB, TMP1, 0, SLJIT_MEM1(STR_PTR), IN_UCHARS(range_right));
3708    CMPTO(SLJIT_C_NOT_EQUAL, TMP2, 0, SLJIT_IMM, chars[4], start);  #else
3709      OP1(SLJIT_MOV_UB, TMP1, 0, SLJIT_MEM1(STR_PTR), IN_UCHARS(range_right + 1) - 1);
3710    #endif
3711    
3712    #if !(defined SLJIT_CONFIG_X86_32 && SLJIT_CONFIG_X86_32)
3713      OP1(SLJIT_MOV_UB, TMP1, 0, SLJIT_MEM2(RETURN_ADDR, TMP1), 0);
3714    #else
3715      OP1(SLJIT_MOV_UB, TMP1, 0, SLJIT_MEM1(TMP1), (sljit_sw)update_table);
3716    #endif
3717      OP2(SLJIT_ADD, STR_PTR, 0, STR_PTR, 0, TMP1, 0);
3718      CMPTO(SLJIT_C_NOT_EQUAL, TMP1, 0, SLJIT_IMM, 0, start);
3719    }    }
3720    
3721  if (offsets[2] >= 0)  if (offsets[0] >= 0)
3722    {    {
3723    if (chars[3] != 0)    OP1(MOV_UCHAR, TMP1, 0, SLJIT_MEM1(STR_PTR), IN_UCHARS(offsets[0]));
3724      OP2(SLJIT_OR, TMP1, 0, TMP1, 0, SLJIT_IMM, chars[3]);    if (offsets[1] >= 0)
3725    CMPTO(SLJIT_C_NOT_EQUAL, TMP1, 0, SLJIT_IMM, chars[2], start);      OP1(MOV_UCHAR, TMP2, 0, SLJIT_MEM1(STR_PTR), IN_UCHARS(offsets[1]));
3726      OP2(SLJIT_ADD, STR_PTR, 0, STR_PTR, 0, SLJIT_IMM, IN_UCHARS(1));
3727    
3728      if (chars[1] != 0)
3729        OP2(SLJIT_OR, TMP1, 0, TMP1, 0, SLJIT_IMM, chars[1]);
3730      CMPTO(SLJIT_C_NOT_EQUAL, TMP1, 0, SLJIT_IMM, chars[0], start);
3731      if (offsets[2] >= 0)
3732        OP1(MOV_UCHAR, TMP1, 0, SLJIT_MEM1(STR_PTR), IN_UCHARS(offsets[2] - 1));
3733    
3734      if (offsets[1] >= 0)
3735        {
3736        if (chars[5] != 0)
3737          OP2(SLJIT_OR, TMP2, 0, TMP2, 0, SLJIT_IMM, chars[5]);
3738        CMPTO(SLJIT_C_NOT_EQUAL, TMP2, 0, SLJIT_IMM, chars[4], start);
3739        }
3740    
3741      if (offsets[2] >= 0)
3742        {
3743        if (chars[3] != 0)
3744          OP2(SLJIT_OR, TMP1, 0, TMP1, 0, SLJIT_IMM, chars[3]);
3745        CMPTO(SLJIT_C_NOT_EQUAL, TMP1, 0, SLJIT_IMM, chars[2], start);
3746        }
3747      OP2(SLJIT_SUB, STR_PTR, 0, STR_PTR, 0, SLJIT_IMM, IN_UCHARS(1));
3748    }    }
 OP2(SLJIT_SUB, STR_PTR, 0, STR_PTR, 0, SLJIT_IMM, IN_UCHARS(1));  
3749    
3750  JUMPHERE(quit);  JUMPHERE(quit);
3751    
3752  if (firstline)  if (firstline)
3753      {
3754      if (range_right >= 0)
3755        OP1(SLJIT_MOV, TMP1, 0, SLJIT_MEM1(SLJIT_LOCALS_REG), common->first_line_end);
3756    OP1(SLJIT_MOV, STR_END, 0, TMP3, 0);    OP1(SLJIT_MOV, STR_END, 0, TMP3, 0);
3757      if (range_right >= 0)
3758        {
3759        quit = CMP(SLJIT_C_LESS_EQUAL, STR_PTR, 0, TMP1, 0);
3760        OP1(SLJIT_MOV, STR_PTR, 0, TMP1, 0);
3761        JUMPHERE(quit);
3762        }
3763      }
3764  else  else
3765    OP2(SLJIT_ADD, STR_END, 0, STR_END, 0, SLJIT_IMM, IN_UCHARS(max));    OP2(SLJIT_ADD, STR_END, 0, STR_END, 0, SLJIT_IMM, IN_UCHARS(max));
3766  return TRUE;  return TRUE;
3767  }  }
3768    
3769  #undef MAX_N_CHARS  #undef MAX_N_CHARS
3770    #undef MAX_N_BYTES
3771    
3772  static SLJIT_INLINE void fast_forward_first_char(compiler_common *common, pcre_uchar first_char, BOOL caseless, BOOL firstline)  static SLJIT_INLINE void fast_forward_first_char(compiler_common *common, pcre_uchar first_char, BOOL caseless, BOOL firstline)
3773  {  {
# Line 3628  if (common->nltype == NLTYPE_FIXED && co Line 3873  if (common->nltype == NLTYPE_FIXED && co
3873    JUMPHERE(lastchar);    JUMPHERE(lastchar);
3874    
3875    if (firstline)    if (firstline)
3876      OP1(SLJIT_MOV, STR_END, 0, SLJIT_MEM1(SLJIT_LOCALS_REG), POSSESSIVE0);      OP1(SLJIT_MOV, STR_END, 0, TMP3, 0);
3877    return;    return;
3878    }    }
3879    
# Line 3638  firstchar = CMP(SLJIT_C_LESS_EQUAL, STR_ Line 3883  firstchar = CMP(SLJIT_C_LESS_EQUAL, STR_
3883  skip_char_back(common);  skip_char_back(common);
3884    
3885  loop = LABEL();  loop = LABEL();
3886    common->ff_newline_shortcut = loop;
3887    
3888  read_char_range(common, common->nlmin, common->nlmax, TRUE);  read_char_range(common, common->nlmin, common->nlmax, TRUE);
3889  lastchar = CMP(SLJIT_C_GREATER_EQUAL, STR_PTR, 0, STR_END, 0);  lastchar = CMP(SLJIT_C_GREATER_EQUAL, STR_PTR, 0, STR_END, 0);
3890  if (common->nltype == NLTYPE_ANY || common->nltype == NLTYPE_ANYCRLF)  if (common->nltype == NLTYPE_ANY || common->nltype == NLTYPE_ANYCRLF)
# Line 7169  if (ket == OP_KETRMAX) Line 7416  if (ket == OP_KETRMAX)
7416    
7417  if (repeat_type == OP_EXACT)  if (repeat_type == OP_EXACT)
7418    {    {
7419      count_match(common);
7420    OP2(SLJIT_SUB | SLJIT_SET_E, SLJIT_MEM1(SLJIT_LOCALS_REG), repeat_ptr, SLJIT_MEM1(SLJIT_LOCALS_REG), repeat_ptr, SLJIT_IMM, 1);    OP2(SLJIT_SUB | SLJIT_SET_E, SLJIT_MEM1(SLJIT_LOCALS_REG), repeat_ptr, SLJIT_MEM1(SLJIT_LOCALS_REG), repeat_ptr, SLJIT_IMM, 1);
7421    JUMPTO(SLJIT_C_NOT_ZERO, rmax_label);    JUMPTO(SLJIT_C_NOT_ZERO, rmax_label);
7422    }    }
# Line 7853  if (*cc == OP_FAIL) Line 8101  if (*cc == OP_FAIL)
8101    return cc + 1;    return cc + 1;
8102    }    }
8103    
8104  if (*cc == OP_ASSERT_ACCEPT || common->currententry != NULL)  if (*cc == OP_ASSERT_ACCEPT || common->currententry != NULL || !common->might_be_empty)
8105    {    {
8106    /* No need to check notempty conditions. */    /* No need to check notempty conditions. */
8107    if (common->accept_label == NULL)    if (common->accept_label == NULL)
# Line 9496  sljit_uw total_length; Line 9744  sljit_uw total_length;
9744  label_addr_list *label_addr;  label_addr_list *label_addr;
9745  struct sljit_label *mainloop_label = NULL;  struct sljit_label *mainloop_label = NULL;
9746  struct sljit_label *continue_match_label;  struct sljit_label *continue_match_label;
9747  struct sljit_label *empty_match_found_label;  struct sljit_label *empty_match_found_label = NULL;
9748  struct sljit_label *empty_match_backtrack_label;  struct sljit_label *empty_match_backtrack_label = NULL;
9749  struct sljit_label *reset_match_label;  struct sljit_label *reset_match_label;
9750  struct sljit_label *quit_label;  struct sljit_label *quit_label;
9751  struct sljit_jump *jump;  struct sljit_jump *jump;
9752  struct sljit_jump *minlength_check_failed = NULL;  struct sljit_jump *minlength_check_failed = NULL;
9753  struct sljit_jump *reqbyte_notfound = NULL;  struct sljit_jump *reqbyte_notfound = NULL;
9754  struct sljit_jump *empty_match;  struct sljit_jump *empty_match = NULL;
9755    
9756  SLJIT_ASSERT((extra->flags & PCRE_EXTRA_STUDY_DATA) != 0);  SLJIT_ASSERT((extra->flags & PCRE_EXTRA_STUDY_DATA) != 0);
9757  study = extra->study_data;  study = extra->study_data;
# Line 9522  common->read_only_data_ptr = NULL; Line 9770  common->read_only_data_ptr = NULL;
9770  common->fcc = tables + fcc_offset;  common->fcc = tables + fcc_offset;
9771  common->lcc = (sljit_sw)(tables + lcc_offset);  common->lcc = (sljit_sw)(tables + lcc_offset);
9772  common->mode = mode;  common->mode = mode;
9773    common->might_be_empty = study->minlength == 0;
9774  common->nltype = NLTYPE_FIXED;  common->nltype = NLTYPE_FIXED;
9775  switch(re->options & PCRE_NEWLINE_BITS)  switch(re->options & PCRE_NEWLINE_BITS)
9776    {    {
# Line 9757  if ((re->options & PCRE_ANCHORED) == 0) Line 10006  if ((re->options & PCRE_ANCHORED) == 0)
10006    if ((re->options & PCRE_NO_START_OPTIMIZE) == 0)    if ((re->options & PCRE_NO_START_OPTIMIZE) == 0)
10007      {      {
10008      if (mode == JIT_COMPILE && fast_forward_first_n_chars(common, (re->options & PCRE_FIRSTLINE) != 0))      if (mode == JIT_COMPILE && fast_forward_first_n_chars(common, (re->options & PCRE_FIRSTLINE) != 0))
10009        { /* Do nothing */ }        {
10010          /* If read_only_data is reallocated, we might have an allocation failure. */
10011          if (common->read_only_data_size > 0 && common->read_only_data == NULL)
10012            {
10013            sljit_free_compiler(compiler);
10014            SLJIT_FREE(common->optimized_cbracket);
10015            SLJIT_FREE(common->private_data_ptrs);
10016            return;
10017            }
10018          }
10019      else if ((re->flags & PCRE_FIRSTSET) != 0)      else if ((re->flags & PCRE_FIRSTSET) != 0)
10020        fast_forward_first_char(common, (pcre_uchar)re->first_char, (re->flags & PCRE_FCH_CASELESS) != 0, (re->options & PCRE_FIRSTLINE) != 0);        fast_forward_first_char(common, (pcre_uchar)re->first_char, (re->flags & PCRE_FCH_CASELESS) != 0, (re->options & PCRE_FIRSTLINE) != 0);
10021      else if ((re->flags & PCRE_STARTLINE) != 0)      else if ((re->flags & PCRE_STARTLINE) != 0)
# Line 9815  if (SLJIT_UNLIKELY(sljit_get_compiler_er Line 10073  if (SLJIT_UNLIKELY(sljit_get_compiler_er
10073    return;    return;
10074    }    }
10075    
10076  empty_match = CMP(SLJIT_C_EQUAL, STR_PTR, 0, SLJIT_MEM1(SLJIT_LOCALS_REG), OVECTOR(0));  if (common->might_be_empty)
10077  empty_match_found_label = LABEL();    {
10078      empty_match = CMP(SLJIT_C_EQUAL, STR_PTR, 0, SLJIT_MEM1(SLJIT_LOCALS_REG), OVECTOR(0));
10079      empty_match_found_label = LABEL();
10080      }
10081    
10082  common->accept_label = LABEL();  common->accept_label = LABEL();
10083  if (common->accept != NULL)  if (common->accept != NULL)
# Line 9840  if (mode != JIT_COMPILE) Line 10101  if (mode != JIT_COMPILE)
10101    return_with_partial_match(common, common->quit_label);    return_with_partial_match(common, common->quit_label);
10102    }    }
10103    
10104  empty_match_backtrack_label = LABEL();  if (common->might_be_empty)
10105      empty_match_backtrack_label = LABEL();
10106  compile_backtrackingpath(common, rootbacktrack.top);  compile_backtrackingpath(common, rootbacktrack.top);
10107  if (SLJIT_UNLIKELY(sljit_get_compiler_error(compiler)))  if (SLJIT_UNLIKELY(sljit_get_compiler_error(compiler)))
10108    {    {
# Line 9876  OP1(SLJIT_MOV, STR_PTR, 0, SLJIT_MEM1(SL Line 10138  OP1(SLJIT_MOV, STR_PTR, 0, SLJIT_MEM1(SL
10138    
10139  if ((re->options & PCRE_ANCHORED) == 0)  if ((re->options & PCRE_ANCHORED) == 0)
10140    {    {
10141    if ((re->options & PCRE_FIRSTLINE) == 0)    if (common->ff_newline_shortcut != NULL)
10142      CMPTO(SLJIT_C_LESS, STR_PTR, 0, STR_END, 0, mainloop_label);      {
10143        if ((re->options & PCRE_FIRSTLINE) == 0)
10144          CMPTO(SLJIT_C_LESS, STR_PTR, 0, STR_END, 0, common->ff_newline_shortcut);
10145        /* There cannot be more newlines here. */
10146        }
10147    else    else
10148      CMPTO(SLJIT_C_LESS, STR_PTR, 0, TMP1, 0, mainloop_label);      {
10149        if ((re->options & PCRE_FIRSTLINE) == 0)
10150          CMPTO(SLJIT_C_LESS, STR_PTR, 0, STR_END, 0, mainloop_label);
10151        else
10152          CMPTO(SLJIT_C_LESS, STR_PTR, 0, TMP1, 0, mainloop_label);
10153        }
10154    }    }
10155    
10156  /* No more remaining characters. */  /* No more remaining characters. */
# Line 9894  JUMPTO(SLJIT_JUMP, common->quit_label); Line 10165  JUMPTO(SLJIT_JUMP, common->quit_label);
10165    
10166  flush_stubs(common);  flush_stubs(common);
10167    
10168  JUMPHERE(empty_match);  if (common->might_be_empty)
10169  OP1(SLJIT_MOV, TMP1, 0, ARGUMENTS, 0);    {
10170  OP1(SLJIT_MOV_UB, TMP2, 0, SLJIT_MEM1(TMP1), SLJIT_OFFSETOF(jit_arguments, notempty));    JUMPHERE(empty_match);
10171  CMPTO(SLJIT_C_NOT_EQUAL, TMP2, 0, SLJIT_IMM, 0, empty_match_backtrack_label);    OP1(SLJIT_MOV, TMP1, 0, ARGUMENTS, 0);
10172  OP1(SLJIT_MOV_UB, TMP2, 0, SLJIT_MEM1(TMP1), SLJIT_OFFSETOF(jit_arguments, notempty_atstart));    OP1(SLJIT_MOV_UB, TMP2, 0, SLJIT_MEM1(TMP1), SLJIT_OFFSETOF(jit_arguments, notempty));
10173  CMPTO(SLJIT_C_EQUAL, TMP2, 0, SLJIT_IMM, 0, empty_match_found_label);    CMPTO(SLJIT_C_NOT_EQUAL, TMP2, 0, SLJIT_IMM, 0, empty_match_backtrack_label);
10174  OP1(SLJIT_MOV, TMP2, 0, SLJIT_MEM1(TMP1), SLJIT_OFFSETOF(jit_arguments, str));    OP1(SLJIT_MOV_UB, TMP2, 0, SLJIT_MEM1(TMP1), SLJIT_OFFSETOF(jit_arguments, notempty_atstart));
10175  CMPTO(SLJIT_C_NOT_EQUAL, TMP2, 0, STR_PTR, 0, empty_match_found_label);    CMPTO(SLJIT_C_EQUAL, TMP2, 0, SLJIT_IMM, 0, empty_match_found_label);
10176  JUMPTO(SLJIT_JUMP, empty_match_backtrack_label);    OP1(SLJIT_MOV, TMP2, 0, SLJIT_MEM1(TMP1), SLJIT_OFFSETOF(jit_arguments, str));
10177      CMPTO(SLJIT_C_NOT_EQUAL, TMP2, 0, STR_PTR, 0, empty_match_found_label);
10178      JUMPTO(SLJIT_JUMP, empty_match_backtrack_label);
10179      }
10180    
10181  common->currententry = common->entries;  common->currententry = common->entries;
10182  common->local_exit = TRUE;  common->local_exit = TRUE;

Legend:
Removed from v.1434  
changed lines
  Added in v.1474

  ViewVC Help
Powered by ViewVC 1.1.5