ext/misc/regexp.c

   1 /*
   2 ** 2012-11-13
   3 **
   4 ** The author disclaims copyright to this source code.  In place of
   5 ** a legal notice, here is a blessing:
   6 **
   7 **    May you do good and not evil.
   8 **    May you find forgiveness for yourself and forgive others.
   9 **    May you share freely, never taking more than you give.
  10 **
  11 ******************************************************************************
  12 **
  13 ** The code in this file implements a compact but reasonably
  14 ** efficient regular-expression matcher for posix extended regular
  15 ** expressions against UTF8 text.
  16 **
  17 ** This file is an SQLite extension.  It registers a single function
  18 ** named "regexp(A,B)" where A is the regular expression and B is the
  19 ** string to be matched.  By registering this function, SQLite will also
  20 ** then implement the "B regexp A" operator.  Note that with the function
  21 ** the regular expression comes first, but with the operator it comes
  22 ** second.
  23 **
  24 **  The following regular expression syntax is supported:
  25 **
  26 **     X*      zero or more occurrences of X
  27 **     X+      one or more occurrences of X
  28 **     X?      zero or one occurrences of X
  29 **     X{p,q}  between p and q occurrences of X
  30 **     (X)     match X
  31 **     X|Y     X or Y
  32 **     ^X      X occurring at the beginning of the string
  33 **     X$      X occurring at the end of the string
  34 **     .       Match any single character
  35 **     \c      Character c where c is one of \{}()[]|*+?.
  36 **     \c      C-language escapes for c in afnrtv.  ex: \t or \n
  37 **     \uXXXX  Where XXXX is exactly 4 hex digits, unicode value XXXX
  38 **     \xXX    Where XX is exactly 2 hex digits, unicode value XX
  39 **     [abc]   Any single character from the set abc
  40 **     [^abc]  Any single character not in the set abc
  41 **     [a-z]   Any single character in the range a-z
  42 **     [^a-z]  Any single character not in the range a-z
  43 **     \b      Word boundary
  44 **     \w      Word character.  [A-Za-z0-9_]
  45 **     \W      Non-word character
  46 **     \d      Digit
  47 **     \D      Non-digit
  48 **     \s      Whitespace character
  49 **     \S      Non-whitespace character
  50 **
  51 ** A nondeterministic finite automaton (NFA) is used for matching, so the
  52 ** performance is bounded by O(N*M) where N is the size of the regular
  53 ** expression and M is the size of the input string.  The matcher never
  54 ** exhibits exponential behavior.  Note that the X{p,q} operator expands
  55 ** to p copies of X following by q-p copies of X? and that the size of the
  56 ** regular expression in the O(N*M) performance bound is computed after
  57 ** this expansion.
  58 */
  59 #include <string.h>
  60 #include <stdlib.h>
  61 #include "sqlite3ext.h"
  62 SQLITE_EXTENSION_INIT1
  63
  64 /*
  65 ** The following #defines change the names of some functions implemented in
  66 ** this file to prevent name collisions with C-library functions of the
  67 ** same name.
  68 */
  69 #define re_match   sqlite3re_match
  70 #define re_compile sqlite3re_compile
  71 #define re_free    sqlite3re_free
  72
  73 /* The end-of-input character */
  74 #define RE_EOF            0    /* End of input */
  75
  76 /* The NFA is implemented as sequence of opcodes taken from the following
  77 ** set.  Each opcode has a single integer argument.
  78 */
  79 #define RE_OP_MATCH       1    /* Match the one character in the argument */
  80 #define RE_OP_ANY         2    /* Match any one character.  (Implements ".") */
  81 #define RE_OP_ANYSTAR     3    /* Special optimized version of .* */
  82 #define RE_OP_FORK        4    /* Continue to both next and opcode at iArg */
  83 #define RE_OP_GOTO        5    /* Jump to opcode at iArg */
  84 #define RE_OP_ACCEPT      6    /* Halt and indicate a successful match */
  85 #define RE_OP_CC_INC      7    /* Beginning of a [...] character class */
  86 #define RE_OP_CC_EXC      8    /* Beginning of a [^...] character class */
  87 #define RE_OP_CC_VALUE    9    /* Single value in a character class */
  88 #define RE_OP_CC_RANGE   10    /* Range of values in a character class */
  89 #define RE_OP_WORD       11    /* Perl word character [A-Za-z0-9_] */
  90 #define RE_OP_NOTWORD    12    /* Not a perl word character */
  91 #define RE_OP_DIGIT      13    /* digit:  [0-9] */
  92 #define RE_OP_NOTDIGIT   14    /* Not a digit */
  93 #define RE_OP_SPACE      15    /* space:  [ \t\n\r\v\f] */
  94 #define RE_OP_NOTSPACE   16    /* Not a digit */
  95 #define RE_OP_BOUNDARY   17    /* Boundary between word and non-word */
  96
  97 /* Each opcode is a "state" in the NFA */
  98 typedef unsigned short ReStateNumber;
  99
 100 /* Because this is an NFA and not a DFA, multiple states can be active at
 101 ** once.  An instance of the following object records all active states in
 102 ** the NFA.  The implementation is optimized for the common case where the
 103 ** number of actives states is small.
 104 */
 105 typedef struct ReStateSet {
 106   unsigned nState;            /* Number of current states */
 107   ReStateNumber *aState;      /* Current states */
 108 } ReStateSet;
 109
 110 /* An input string read one character at a time.
 111 */
 112 typedef struct ReInput ReInput;
 113 struct ReInput {
 114   const unsigned char *z;  /* All text */
 115   int i;                   /* Next byte to read */
 116   int mx;                  /* EOF when i>=mx */
 117 };
 118
 119 /* A compiled NFA (or an NFA that is in the process of being compiled) is
 120 ** an instance of the following object.
 121 */
 122 typedef struct ReCompiled ReCompiled;
 123 struct ReCompiled {
 124   ReInput sIn;                /* Regular expression text */
 125   const char *zErr;           /* Error message to return */
 126   char *aOp;                  /* Operators for the virtual machine */
 127   int *aArg;                  /* Arguments to each operator */
 128   unsigned (*xNextChar)(ReInput*);  /* Next character function */
 129   unsigned char zInit[12];    /* Initial text to match */
 130   int nInit;                  /* Number of characters in zInit */
 131   unsigned nState;            /* Number of entries in aOp[] and aArg[] */
 132   unsigned nAlloc;            /* Slots allocated for aOp[] and aArg[] */
 133 };
 134
 135 /* Add a state to the given state set if it is not already there */
 136 static void re_add_state(ReStateSet *pSet, int newState){
 137   unsigned i;
 138   for(i=0; i<pSet->nState; i++) if( pSet->aState[i]==newState ) return;
 139   pSet->aState[pSet->nState++] = (ReStateNumber)newState;
 140 }
 141
 142 /* Extract the next unicode character from *pzIn and return it.  Advance
 143 ** *pzIn to the first byte past the end of the character returned.  To
 144 ** be clear:  this routine converts utf8 to unicode.  This routine is
 145 ** optimized for the common case where the next character is a single byte.
 146 */
 147 static unsigned re_next_char(ReInput *p){
 148   unsigned c;
 149   if( p->i>=p->mx ) return 0;
 150   c = p->z[p->i++];
 151   if( c>=0x80 ){
 152     if( (c&0xe0)==0xc0 && p->i<p->mx && (p->z[p->i]&0xc0)==0x80 ){
 153       c = (c&0x1f)<<6 | (p->z[p->i++]&0x3f);
 154       if( c<0x80 ) c = 0xfffd;
 155     }else if( (c&0xf0)==0xe0 && p->i+1<p->mx && (p->z[p->i]&0xc0)==0x80
 156            && (p->z[p->i+1]&0xc0)==0x80 ){
 157       c = (c&0x0f)<<12 | ((p->z[p->i]&0x3f)<<6) | (p->z[p->i+1]&0x3f);
 158       p->i += 2;
 159       if( c<=0x7ff || (c>=0xd800 && c<=0xdfff) ) c = 0xfffd;
 160     }else if( (c&0xf8)==0xf0 && p->i+3<p->mx && (p->z[p->i]&0xc0)==0x80
 161            && (p->z[p->i+1]&0xc0)==0x80 && (p->z[p->i+2]&0xc0)==0x80 ){
 162       c = (c&0x07)<<18 | ((p->z[p->i]&0x3f)<<12) | ((p->z[p->i+1]&0x3f)<<6)
 163                        | (p->z[p->i+2]&0x3f);
 164       p->i += 3;
 165       if( c<=0xffff || c>0x10ffff ) c = 0xfffd;
 166     }else{
 167       c = 0xfffd;
 168     }
 169   }
 170   return c;
 171 }
 172 static unsigned re_next_char_nocase(ReInput *p){
 173   unsigned c = re_next_char(p);
 174   if( c>='A' && c<='Z' ) c += 'a' - 'A';
 175   return c;
 176 }
 177
 178 /* Return true if c is a perl "word" character:  [A-Za-z0-9_] */
 179 static int re_word_char(int c){
 180   return (c>='0' && c<='9') || (c>='a' && c<='z')
 181       || (c>='A' && c<='Z') || c=='_';
 182 }
 183
 184 /* Return true if c is a "digit" character:  [0-9] */
 185 static int re_digit_char(int c){
 186   return (c>='0' && c<='9');
 187 }
 188
 189 /* Return true if c is a perl "space" character:  [ \t\r\n\v\f] */
 190 static int re_space_char(int c){
 191   return c==' ' || c=='\t' || c=='\n' || c=='\r' || c=='\v' || c=='\f';
 192 }
 193
 194 /* Run a compiled regular expression on the zero-terminated input
 195 ** string zIn[].  Return true on a match and false if there is no match.
 196 */
 197 static int re_match(ReCompiled *pRe, const unsigned char *zIn, int nIn){
 198   ReStateSet aStateSet[2], *pThis, *pNext;
 199   ReStateNumber aSpace[100];
 200   ReStateNumber *pToFree;
 201   unsigned int i = 0;
 202   unsigned int iSwap = 0;
 203   int c = RE_EOF+1;
 204   int cPrev = 0;
 205   int rc = 0;
 206   ReInput in;
 207
 208   in.z = zIn;
 209   in.i = 0;
 210   in.mx = nIn>=0 ? nIn : (int)strlen((char const*)zIn);
 211
 212   /* Look for the initial prefix match, if there is one. */
 213   if( pRe->nInit ){
 214     unsigned char x = pRe->zInit[0];
 215     while( in.i+pRe->nInit<=in.mx
 216      && (zIn[in.i]!=x ||
 217          strncmp((const char*)zIn+in.i, (const char*)pRe->zInit, pRe->nInit)!=0)
 218     ){
 219       in.i++;
 220     }
 221     if( in.i+pRe->nInit>in.mx ) return 0;
 222   }
 223
 224   if( pRe->nState<=(sizeof(aSpace)/(sizeof(aSpace[0])*2)) ){
 225     pToFree = 0;
 226     aStateSet[0].aState = aSpace;
 227   }else{
 228     pToFree = sqlite3_malloc64( sizeof(ReStateNumber)*2*pRe->nState );
 229     if( pToFree==0 ) return -1;
 230     aStateSet[0].aState = pToFree;
 231   }
 232   aStateSet[1].aState = &aStateSet[0].aState[pRe->nState];
 233   pNext = &aStateSet[1];
 234   pNext->nState = 0;
 235   re_add_state(pNext, 0);
 236   while( c!=RE_EOF && pNext->nState>0 ){
 237     cPrev = c;
 238     c = pRe->xNextChar(&in);
 239     pThis = pNext;
 240     pNext = &aStateSet[iSwap];
 241     iSwap = 1 - iSwap;
 242     pNext->nState = 0;
 243     for(i=0; i<pThis->nState; i++){
 244       int x = pThis->aState[i];
 245       switch( pRe->aOp[x] ){
 246         case RE_OP_MATCH: {
 247           if( pRe->aArg[x]==c ) re_add_state(pNext, x+1);
 248           break;
 249         }
 250         case RE_OP_ANY: {
 251           if( c!=0 ) re_add_state(pNext, x+1);
 252           break;
 253         }
 254         case RE_OP_WORD: {
 255           if( re_word_char(c) ) re_add_state(pNext, x+1);
 256           break;
 257         }
 258         case RE_OP_NOTWORD: {
 259           if( !re_word_char(c) && c!=0 ) re_add_state(pNext, x+1);
 260           break;
 261         }
 262         case RE_OP_DIGIT: {
 263           if( re_digit_char(c) ) re_add_state(pNext, x+1);
 264           break;
 265         }
 266         case RE_OP_NOTDIGIT: {
 267           if( !re_digit_char(c) && c!=0 ) re_add_state(pNext, x+1);
 268           break;
 269         }
 270         case RE_OP_SPACE: {
 271           if( re_space_char(c) ) re_add_state(pNext, x+1);
 272           break;
 273         }
 274         case RE_OP_NOTSPACE: {
 275           if( !re_space_char(c) && c!=0 ) re_add_state(pNext, x+1);
 276           break;
 277         }
 278         case RE_OP_BOUNDARY: {
 279           if( re_word_char(c)!=re_word_char(cPrev) ) re_add_state(pThis, x+1);
 280           break;
 281         }
 282         case RE_OP_ANYSTAR: {
 283           re_add_state(pNext, x);
 284           re_add_state(pThis, x+1);
 285           break;
 286         }
 287         case RE_OP_FORK: {
 288           re_add_state(pThis, x+pRe->aArg[x]);
 289           re_add_state(pThis, x+1);
 290           break;
 291         }
 292         case RE_OP_GOTO: {
 293           re_add_state(pThis, x+pRe->aArg[x]);
 294           break;
 295         }
 296         case RE_OP_ACCEPT: {
 297           rc = 1;
 298           goto re_match_end;
 299         }
 300         case RE_OP_CC_EXC: {
 301           if( c==0 ) break;
 302           /* fall-through */ goto re_op_cc_inc;
 303         }
 304         case RE_OP_CC_INC: re_op_cc_inc: {
 305           int j = 1;
 306           int n = pRe->aArg[x];
 307           int hit = 0;
 308           for(j=1; j>0 && j<n; j++){
 309             if( pRe->aOp[x+j]==RE_OP_CC_VALUE ){
 310               if( pRe->aArg[x+j]==c ){
 311                 hit = 1;
 312                 j = -1;
 313               }
 314             }else{
 315               if( pRe->aArg[x+j]<=c && pRe->aArg[x+j+1]>=c ){
 316                 hit = 1;
 317                 j = -1;
 318               }else{
 319                 j++;
 320               }
 321             }
 322           }
 323           if( pRe->aOp[x]==RE_OP_CC_EXC ) hit = !hit;
 324           if( hit ) re_add_state(pNext, x+n);
 325           break;
 326         }
 327       }
 328     }
 329   }
 330   for(i=0; i<pNext->nState; i++){
 331     if( pRe->aOp[pNext->aState[i]]==RE_OP_ACCEPT ){ rc = 1; break; }
 332   }
 333 re_match_end:
 334   sqlite3_free(pToFree);
 335   return rc;
 336 }
 337
 338 /* Resize the opcode and argument arrays for an RE under construction.
 339 */
 340 static int re_resize(ReCompiled *p, int N){
 341   char *aOp;
 342   int *aArg;
 343   aOp = sqlite3_realloc64(p->aOp, N*sizeof(p->aOp[0]));
 344   if( aOp==0 ) return 1;
 345   p->aOp = aOp;
 346   aArg = sqlite3_realloc64(p->aArg, N*sizeof(p->aArg[0]));
 347   if( aArg==0 ) return 1;
 348   p->aArg = aArg;
 349   p->nAlloc = N;
 350   return 0;
 351 }
 352
 353 /* Insert a new opcode and argument into an RE under construction.  The
 354 ** insertion point is just prior to existing opcode iBefore.
 355 */
 356 static int re_insert(ReCompiled *p, int iBefore, int op, int arg){
 357   int i;
 358   if( p->nAlloc<=p->nState && re_resize(p, p->nAlloc*2) ) return 0;
 359   for(i=p->nState; i>iBefore; i--){
 360     p->aOp[i] = p->aOp[i-1];
 361     p->aArg[i] = p->aArg[i-1];
 362   }
 363   p->nState++;
 364   p->aOp[iBefore] = (char)op;
 365   p->aArg[iBefore] = arg;
 366   return iBefore;
 367 }
 368
 369 /* Append a new opcode and argument to the end of the RE under construction.
 370 */
 371 static int re_append(ReCompiled *p, int op, int arg){
 372   return re_insert(p, p->nState, op, arg);
 373 }
 374
 375 /* Make a copy of N opcodes starting at iStart onto the end of the RE
 376 ** under construction.
 377 */
 378 static void re_copy(ReCompiled *p, int iStart, int N){
 379   if( p->nState+N>=p->nAlloc && re_resize(p, p->nAlloc*2+N) ) return;
 380   memcpy(&p->aOp[p->nState], &p->aOp[iStart], N*sizeof(p->aOp[0]));
 381   memcpy(&p->aArg[p->nState], &p->aArg[iStart], N*sizeof(p->aArg[0]));
 382   p->nState += N;
 383 }
 384
 385 /* Return true if c is a hexadecimal digit character:  [0-9a-fA-F]
 386 ** If c is a hex digit, also set *pV = (*pV)*16 + valueof(c).  If
 387 ** c is not a hex digit *pV is unchanged.
 388 */
 389 static int re_hex(int c, int *pV){
 390   if( c>='0' && c<='9' ){
 391     c -= '0';
 392   }else if( c>='a' && c<='f' ){
 393     c -= 'a' - 10;
 394   }else if( c>='A' && c<='F' ){
 395     c -= 'A' - 10;
 396   }else{
 397     return 0;
 398   }
 399   *pV = (*pV)*16 + (c & 0xff);
 400   return 1;
 401 }
 402
 403 /* A backslash character has been seen, read the next character and
 404 ** return its interpretation.
 405 */
 406 static unsigned re_esc_char(ReCompiled *p){
 407   static const char zEsc[] = "afnrtv\\()*.+?[$^{|}]";
 408   static const char zTrans[] = "\a\f\n\r\t\v";
 409   int i, v = 0;
 410   char c;
 411   if( p->sIn.i>=p->sIn.mx ) return 0;
 412   c = p->sIn.z[p->sIn.i];
 413   if( c=='u' && p->sIn.i+4<p->sIn.mx ){
 414     const unsigned char *zIn = p->sIn.z + p->sIn.i;
 415     if( re_hex(zIn[1],&v)
 416      && re_hex(zIn[2],&v)
 417      && re_hex(zIn[3],&v)
 418      && re_hex(zIn[4],&v)
 419     ){
 420       p->sIn.i += 5;
 421       return v;
 422     }
 423   }
 424   if( c=='x' && p->sIn.i+2<p->sIn.mx ){
 425     const unsigned char *zIn = p->sIn.z + p->sIn.i;
 426     if( re_hex(zIn[1],&v)
 427      && re_hex(zIn[2],&v)
 428     ){
 429       p->sIn.i += 3;
 430       return v;
 431     }
 432   }
 433   for(i=0; zEsc[i] && zEsc[i]!=c; i++){}
 434   if( zEsc[i] ){
 435     if( i<6 ) c = zTrans[i];
 436     p->sIn.i++;
 437   }else{
 438     p->zErr = "unknown \\ escape";
 439   }
 440   return c;
 441 }
 442
 443 /* Forward declaration */
 444 static const char *re_subcompile_string(ReCompiled*);
 445
 446 /* Peek at the next byte of input */
 447 static unsigned char rePeek(ReCompiled *p){
 448   return p->sIn.i<p->sIn.mx ? p->sIn.z[p->sIn.i] : 0;
 449 }
 450
 451 /* Compile RE text into a sequence of opcodes.  Continue up to the
 452 ** first unmatched ")" character, then return.  If an error is found,
 453 ** return a pointer to the error message string.
 454 */
 455 static const char *re_subcompile_re(ReCompiled *p){
 456   const char *zErr;
 457   int iStart, iEnd, iGoto;
 458   iStart = p->nState;
 459   zErr = re_subcompile_string(p);
 460   if( zErr ) return zErr;
 461   while( rePeek(p)=='|' ){
 462     iEnd = p->nState;
 463     re_insert(p, iStart, RE_OP_FORK, iEnd + 2 - iStart);
 464     iGoto = re_append(p, RE_OP_GOTO, 0);
 465     p->sIn.i++;
 466     zErr = re_subcompile_string(p);
 467     if( zErr ) return zErr;
 468     p->aArg[iGoto] = p->nState - iGoto;
 469   }
 470   return 0;
 471 }
 472
 473 /* Compile an element of regular expression text (anything that can be
 474 ** an operand to the "|" operator).  Return NULL on success or a pointer
 475 ** to the error message if there is a problem.
 476 */
 477 static const char *re_subcompile_string(ReCompiled *p){
 478   int iPrev = -1;
 479   int iStart;
 480   unsigned c;
 481   const char *zErr;
 482   while( (c = p->xNextChar(&p->sIn))!=0 ){
 483     iStart = p->nState;
 484     switch( c ){
 485       case '|':
 486       case '$':
 487       case ')': {
 488         p->sIn.i--;
 489         return 0;
 490       }
 491       case '(': {
 492         zErr = re_subcompile_re(p);
 493         if( zErr ) return zErr;
 494         if( rePeek(p)!=')' ) return "unmatched '('";
 495         p->sIn.i++;
 496         break;
 497       }
 498       case '.': {
 499         if( rePeek(p)=='*' ){
 500           re_append(p, RE_OP_ANYSTAR, 0);
 501           p->sIn.i++;
 502         }else{
 503           re_append(p, RE_OP_ANY, 0);
 504         }
 505         break;
 506       }
 507       case '*': {
 508         if( iPrev<0 ) return "'*' without operand";
 509         re_insert(p, iPrev, RE_OP_GOTO, p->nState - iPrev + 1);
 510         re_append(p, RE_OP_FORK, iPrev - p->nState + 1);
 511         break;
 512       }
 513       case '+': {
 514         if( iPrev<0 ) return "'+' without operand";
 515         re_append(p, RE_OP_FORK, iPrev - p->nState);
 516         break;
 517       }
 518       case '?': {
 519         if( iPrev<0 ) return "'?' without operand";
 520         re_insert(p, iPrev, RE_OP_FORK, p->nState - iPrev+1);
 521         break;
 522       }
 523       case '{': {
 524         int m = 0, n = 0;
 525         int sz, j;
 526         if( iPrev<0 ) return "'{m,n}' without operand";
 527         while( (c=rePeek(p))>='0' && c<='9' ){ m = m*10 + c - '0'; p->sIn.i++; }
 528         n = m;
 529         if( c==',' ){
 530           p->sIn.i++;
 531           n = 0;
 532           while( (c=rePeek(p))>='0' && c<='9' ){ n = n*10 + c-'0'; p->sIn.i++; }
 533         }
 534         if( c!='}' ) return "unmatched '{'";
 535         if( n>0 && n<m ) return "n less than m in '{m,n}'";
 536         p->sIn.i++;
 537         sz = p->nState - iPrev;
 538         if( m==0 ){
 539           if( n==0 ) return "both m and n are zero in '{m,n}'";
 540           re_insert(p, iPrev, RE_OP_FORK, sz+1);
 541           n--;
 542         }else{
 543           for(j=1; j<m; j++) re_copy(p, iPrev, sz);
 544         }
 545         for(j=m; j<n; j++){
 546           re_append(p, RE_OP_FORK, sz+1);
 547           re_copy(p, iPrev, sz);
 548         }
 549         if( n==0 && m>0 ){
 550           re_append(p, RE_OP_FORK, -sz);
 551         }
 552         break;
 553       }
 554       case '[': {
 555         int iFirst = p->nState;
 556         if( rePeek(p)=='^' ){
 557           re_append(p, RE_OP_CC_EXC, 0);
 558           p->sIn.i++;
 559         }else{
 560           re_append(p, RE_OP_CC_INC, 0);
 561         }
 562         while( (c = p->xNextChar(&p->sIn))!=0 ){
 563           if( c=='[' && rePeek(p)==':' ){
 564             return "POSIX character classes not supported";
 565           }
 566           if( c=='\\' ) c = re_esc_char(p);
 567           if( rePeek(p)=='-' ){
 568             re_append(p, RE_OP_CC_RANGE, c);
 569             p->sIn.i++;
 570             c = p->xNextChar(&p->sIn);
 571             if( c=='\\' ) c = re_esc_char(p);
 572             re_append(p, RE_OP_CC_RANGE, c);
 573           }else{
 574             re_append(p, RE_OP_CC_VALUE, c);
 575           }
 576           if( rePeek(p)==']' ){ p->sIn.i++; break; }
 577         }
 578         if( c==0 ) return "unclosed '['";
 579         p->aArg[iFirst] = p->nState - iFirst;
 580         break;
 581       }
 582       case '\\': {
 583         int specialOp = 0;
 584         switch( rePeek(p) ){
 585           case 'b': specialOp = RE_OP_BOUNDARY;   break;
 586           case 'd': specialOp = RE_OP_DIGIT;      break;
 587           case 'D': specialOp = RE_OP_NOTDIGIT;   break;
 588           case 's': specialOp = RE_OP_SPACE;      break;
 589           case 'S': specialOp = RE_OP_NOTSPACE;   break;
 590           case 'w': specialOp = RE_OP_WORD;       break;
 591           case 'W': specialOp = RE_OP_NOTWORD;    break;
 592         }
 593         if( specialOp ){
 594           p->sIn.i++;
 595           re_append(p, specialOp, 0);
 596         }else{
 597           c = re_esc_char(p);
 598           re_append(p, RE_OP_MATCH, c);
 599         }
 600         break;
 601       }
 602       default: {
 603         re_append(p, RE_OP_MATCH, c);
 604         break;
 605       }
 606     }
 607     iPrev = iStart;
 608   }
 609   return 0;
 610 }
 611
 612 /* Free and reclaim all the memory used by a previously compiled
 613 ** regular expression.  Applications should invoke this routine once
 614 ** for every call to re_compile() to avoid memory leaks.
 615 */
 616 static void re_free(ReCompiled *pRe){
 617   if( pRe ){
 618     sqlite3_free(pRe->aOp);
 619     sqlite3_free(pRe->aArg);
 620     sqlite3_free(pRe);
 621   }
 622 }
 623
 624 /*
 625 ** Compile a textual regular expression in zIn[] into a compiled regular
 626 ** expression suitable for us by re_match() and return a pointer to the
 627 ** compiled regular expression in *ppRe.  Return NULL on success or an
 628 ** error message if something goes wrong.
 629 */
 630 static const char *re_compile(ReCompiled **ppRe, const char *zIn, int noCase){
 631   ReCompiled *pRe;
 632   const char *zErr;
 633   int i, j;
 634
 635   *ppRe = 0;
 636   pRe = sqlite3_malloc( sizeof(*pRe) );
 637   if( pRe==0 ){
 638     return "out of memory";
 639   }
 640   memset(pRe, 0, sizeof(*pRe));
 641   pRe->xNextChar = noCase ? re_next_char_nocase : re_next_char;
 642   if( re_resize(pRe, 30) ){
 643     re_free(pRe);
 644     return "out of memory";
 645   }
 646   if( zIn[0]=='^' ){
 647     zIn++;
 648   }else{
 649     re_append(pRe, RE_OP_ANYSTAR, 0);
 650   }
 651   pRe->sIn.z = (unsigned char*)zIn;
 652   pRe->sIn.i = 0;
 653   pRe->sIn.mx = (int)strlen(zIn);
 654   zErr = re_subcompile_re(pRe);
 655   if( zErr ){
 656     re_free(pRe);
 657     return zErr;
 658   }
 659   if( rePeek(pRe)=='$' && pRe->sIn.i+1>=pRe->sIn.mx ){
 660     re_append(pRe, RE_OP_MATCH, RE_EOF);
 661     re_append(pRe, RE_OP_ACCEPT, 0);
 662     *ppRe = pRe;
 663   }else if( pRe->sIn.i>=pRe->sIn.mx ){
 664     re_append(pRe, RE_OP_ACCEPT, 0);
 665     *ppRe = pRe;
 666   }else{
 667     re_free(pRe);
 668     return "unrecognized character";
 669   }
 670
 671   /* The following is a performance optimization.  If the regex begins with
 672   ** ".*" (if the input regex lacks an initial "^") and afterwards there are
 673   ** one or more matching characters, enter those matching characters into
 674   ** zInit[].  The re_match() routine can then search ahead in the input
 675   ** string looking for the initial match without having to run the whole
 676   ** regex engine over the string.  Do not worry able trying to match
 677   ** unicode characters beyond plane 0 - those are very rare and this is
 678   ** just an optimization. */
 679   if( pRe->aOp[0]==RE_OP_ANYSTAR && !noCase ){
 680     for(j=0, i=1; j<(int)sizeof(pRe->zInit)-2 && pRe->aOp[i]==RE_OP_MATCH; i++){
 681       unsigned x = pRe->aArg[i];
 682       if( x<=127 ){
 683         pRe->zInit[j++] = (unsigned char)x;
 684       }else if( x<=0xfff ){
 685         pRe->zInit[j++] = (unsigned char)(0xc0 | (x>>6));
 686         pRe->zInit[j++] = 0x80 | (x&0x3f);
 687       }else if( x<=0xffff ){
 688         pRe->zInit[j++] = (unsigned char)(0xd0 | (x>>12));
 689         pRe->zInit[j++] = 0x80 | ((x>>6)&0x3f);
 690         pRe->zInit[j++] = 0x80 | (x&0x3f);
 691       }else{
 692         break;
 693       }
 694     }
 695     if( j>0 && pRe->zInit[j-1]==0 ) j--;
 696     pRe->nInit = j;
 697   }
 698   return pRe->zErr;
 699 }
 700
 701 /*
 702 ** Implementation of the regexp() SQL function.  This function implements
 703 ** the build-in REGEXP operator.  The first argument to the function is the
 704 ** pattern and the second argument is the string.  So, the SQL statements:
 705 **
 706 **       A REGEXP B
 707 **
 708 ** is implemented as regexp(B,A).
 709 */
 710 static void re_sql_func(
 711   sqlite3_context *context,
 712   int argc,
 713   sqlite3_value **argv
 714 ){
 715   ReCompiled *pRe;          /* Compiled regular expression */
 716   const char *zPattern;     /* The regular expression */
 717   const unsigned char *zStr;/* String being searched */
 718   const char *zErr;         /* Compile error message */
 719   int setAux = 0;           /* True to invoke sqlite3_set_auxdata() */
 720
 721   (void)argc;  /* Unused */
 722   pRe = sqlite3_get_auxdata(context, 0);
 723   if( pRe==0 ){
 724     zPattern = (const char*)sqlite3_value_text(argv[0]);
 725     if( zPattern==0 ) return;
 726     zErr = re_compile(&pRe, zPattern, sqlite3_user_data(context)!=0);
 727     if( zErr ){
 728       re_free(pRe);
 729       sqlite3_result_error(context, zErr, -1);
 730       return;
 731     }
 732     if( pRe==0 ){
 733       sqlite3_result_error_nomem(context);
 734       return;
 735     }
 736     setAux = 1;
 737   }
 738   zStr = (const unsigned char*)sqlite3_value_text(argv[1]);
 739   if( zStr!=0 ){
 740     sqlite3_result_int(context, re_match(pRe, zStr, -1));
 741   }
 742   if( setAux ){
 743     sqlite3_set_auxdata(context, 0, pRe, (void(*)(void*))re_free);
 744   }
 745 }
 746
 747 /*
 748 ** Invoke this routine to register the regexp() function with the
 749 ** SQLite database connection.
 750 */
 751 #ifdef _WIN32
 752 __declspec(dllexport)
 753 #endif
 754 int sqlite3_regexp_init(
 755   sqlite3 *db,
 756   char **pzErrMsg,
 757   const sqlite3_api_routines *pApi
 758 ){
 759   int rc = SQLITE_OK;
 760   SQLITE_EXTENSION_INIT2(pApi);
 761   (void)pzErrMsg;  /* Unused */
 762   rc = sqlite3_create_function(db, "regexp", 2, SQLITE_UTF8|SQLITE_INNOCUOUS,
 763                                0, re_sql_func, 0, 0);
 764   if( rc==SQLITE_OK ){
 765     /* The regexpi(PATTERN,STRING) function is a case-insensitive version
 766     ** of regexp(PATTERN,STRING). */
 767     rc = sqlite3_create_function(db, "regexpi", 2, SQLITE_UTF8|SQLITE_INNOCUOUS,
 768                                  (void*)db, re_sql_func, 0, 0);
 769   }
 770   return rc;
 771 }