Annotation of coherent/d/usr/bin/elvis/regexp.c, revision 1.1.1.1

1.1       root        1: /* regexp.c */
                      2: 
                      3: /* This file contains the code that compiles regular expressions and executes
                      4:  * them.  It supports the same syntax and features as vi's regular expression
                      5:  * code.  Specifically, the meta characters are:
                      6:  *     ^       matches the beginning of a line
                      7:  *     $       matches the end of a line
                      8:  *     \<      matches the beginning of a word
                      9:  *     \>      matches the end of a word
                     10:  *     .       matches any single character
                     11:  *     []      matches any character in a character class
                     12:  *     \(      delimits the start of a subexpression
                     13:  *     \)      delimits the end of a subexpression
                     14:  *     *       repeats the preceding 0 or more times
                     15:  * NOTE: You cannot follow a \) with a *.
                     16:  *
                     17:  * The physical structure of a compiled RE is as follows:
                     18:  *     - First, there is a one-byte value that says how many character classes
                     19:  *       are used in this regular expression
                     20:  *     - Next, each character class is stored as a bitmap that is 256 bits
                     21:  *       (32 bytes) long.
                     22:  *     - A mixture of literal characters and compiled meta characters follows.
                     23:  *       This begins with M_BEGIN(0) and ends with M_END(0).  All meta chars
                     24:  *       are stored as a \n followed by a one-byte code, so they take up two
                     25:  *       bytes apiece.  Literal characters take up one byte apiece.  \n can't
                     26:  *       be used as a literal character.
                     27:  *
                     28:  * If NO_MAGIC is defined, then a different set of functions is used instead.
                     29:  * That right, this file contains TWO versions of the code.
                     30:  */
                     31: 
                     32: #include <setjmp.h>
                     33: #include <ctype.h>
                     34: #include "config.h"
                     35: #include "vi.h"
                     36: #include "regexp.h"
                     37: 
                     38: 
                     39: 
                     40: static char    *previous;      /* the previous regexp, used when null regexp is given */
                     41: 
                     42: 
                     43: #ifndef NO_MAGIC
                     44: /* THE REAL REGEXP PACKAGE IS USED UNLESS "NO_MAGIC" IS DEFINED */
                     45: 
                     46: /* These are used to classify or recognize meta-characters */
                     47: #define META           '\0'
                     48: #define BASE_META(m)   ((m) - 256)
                     49: #define INT_META(c)    ((c) + 256)
                     50: #define IS_META(m)     ((m) >= 256)
                     51: #define IS_CLASS(m)    ((m) >= M_CLASS(0) && (m) <= M_CLASS(9))
                     52: #define IS_START(m)    ((m) >= M_START(0) && (m) <= M_START(9))
                     53: #define IS_END(m)      ((m) >= M_END(0) && (m) <= M_END(9))
                     54: #define IS_CLOSURE(m)  ((m) >= M_SPLAT && (m) <= M_QMARK)
                     55: #define ADD_META(s,m)  (*(s)++ = META, *(s)++ = BASE_META(m))
                     56: #define GET_META(s)    (*(s) == META ? INT_META(*++(s)) : *s)
                     57: 
                     58: /* These are the internal codes used for each type of meta-character */
                     59: #define M_BEGLINE      256             /* internal code for ^ */
                     60: #define M_ENDLINE      257             /* internal code for $ */
                     61: #define M_BEGWORD      258             /* internal code for \< */
                     62: #define M_ENDWORD      259             /* internal code for \> */
                     63: #define M_ANY          260             /* internal code for . */
                     64: #define M_SPLAT                261             /* internal code for * */
                     65: #define M_PLUS         262             /* internal code for \+ */
                     66: #define M_QMARK                263             /* internal code for \? */
                     67: #define M_CLASS(n)     (264+(n))       /* internal code for [] */
                     68: #define M_START(n)     (274+(n))       /* internal code for \( */
                     69: #define M_END(n)       (284+(n))       /* internal code for \) */
                     70: 
                     71: /* These are used during compilation */
                     72: static int     class_cnt;      /* used to assign class IDs */
                     73: static int     start_cnt;      /* used to assign start IDs */
                     74: static int     end_stk[NSUBEXP];/* used to assign end IDs */
                     75: static int     end_sp;
                     76: static char    *retext;        /* points to the text being compiled */
                     77: 
                     78: /* error-handling stuff */
                     79: jmp_buf        errorhandler;
                     80: #define FAIL(why)      regerror(why); longjmp(errorhandler, 1)
                     81: 
                     82: 
                     83: 
                     84: 
                     85: 
                     86: /* This function builds a bitmap for a particular class */
                     87: static char *makeclass(text, bmap)
                     88:        REG char        *text;  /* start of the class */
                     89:        REG char        *bmap;  /* the bitmap */
                     90: {
                     91:        REG int         i;
                     92:        int             complement = 0;
                     93: 
                     94: # if TRACE
                     95:        printf("makeclass(\"%s\", 0x%lx)\n", text, (long)bmap);
                     96: # endif
                     97: 
                     98:        /* zero the bitmap */
                     99:        for (i = 0; bmap && i < 32; i++)
                    100:        {
                    101:                bmap[i] = 0;
                    102:        }
                    103: 
                    104:        /* see if we're going to complement this class */
                    105:        if (*text == '^')
                    106:        {
                    107:                text++;
                    108:                complement = 1;
                    109:        }
                    110: 
                    111:        /* add in the characters */
                    112:        while (*text && *text != ']')
                    113:        {
                    114:                /* is this a span of characters? */
                    115:                if (text[1] == '-' && text[2])
                    116:                {
                    117:                        /* spans can't be backwards */
                    118:                        if (text[0] > text[2])
                    119:                        {
                    120:                                FAIL("Backwards span in []");
                    121:                        }
                    122: 
                    123:                        /* add each character in the span to the bitmap */
                    124:                        for (i = text[0]; bmap && i <= text[2]; i++)
                    125:                        {
                    126:                                bmap[i >> 3] |= (1 << (i & 7));
                    127:                        }
                    128: 
                    129:                        /* move past this span */
                    130:                        text += 3;
                    131:                }
                    132:                else
                    133:                {
                    134:                        /* add this single character to the span */
                    135:                        i = *text++;
                    136:                        if (bmap)
                    137:                        {
                    138:                                bmap[i >> 3] |= (1 << (i & 7));
                    139:                        }
                    140:                }
                    141:        }
                    142: 
                    143:        /* make sure the closing ] is missing */
                    144:        if (*text++ != ']')
                    145:        {
                    146:                FAIL("] missing");
                    147:        }
                    148: 
                    149:        /* if we're supposed to complement this class, then do so */
                    150:        if (complement && bmap)
                    151:        {
                    152:                for (i = 0; i < 32; i++)
                    153:                {
                    154:                        bmap[i] = ~bmap[i];
                    155:                }
                    156:        }
                    157: 
                    158:        return text;
                    159: }
                    160: 
                    161: 
                    162: 
                    163: 
                    164: /* This function gets the next character or meta character from a string.
                    165:  * The pointer is incremented by 1, or by 2 for \-quoted characters.  For [],
                    166:  * a bitmap is generated via makeclass() (if re is given), and the
                    167:  * character-class text is skipped.
                    168:  */
                    169: static int gettoken(sptr, re)
                    170:        char    **sptr;
                    171:        regexp  *re;
                    172: {
                    173:        int     c;
                    174: 
                    175:        c = **sptr;
                    176:        ++*sptr;
                    177:        if (c == '\\')
                    178:        {
                    179:                c = **sptr;
                    180:                ++*sptr;
                    181:                switch (c)
                    182:                {
                    183:                  case '<':
                    184:                        return M_BEGWORD;
                    185: 
                    186:                  case '>':
                    187:                        return M_ENDWORD;
                    188: 
                    189:                  case '(':
                    190:                        if (start_cnt >= NSUBEXP)
                    191:                        {
                    192:                                FAIL("Too many \\(s");
                    193:                        }
                    194:                        end_stk[end_sp++] = start_cnt;
                    195:                        return M_START(start_cnt++);
                    196: 
                    197:                  case ')':
                    198:                        if (end_sp <= 0)
                    199:                        {
                    200:                                FAIL("Mismatched \\)");
                    201:                        }
                    202:                        return M_END(end_stk[--end_sp]);
                    203: 
                    204:                  case '*':
                    205:                        return (*o_magic ? c : M_SPLAT);
                    206: 
                    207:                  case '.':
                    208:                        return (*o_magic ? c : M_ANY);
                    209: 
                    210:                  case '+':
                    211:                        return M_PLUS;
                    212: 
                    213:                  case '?':
                    214:                        return M_QMARK;
                    215: 
                    216:                  default:
                    217:                        return c;
                    218:                }
                    219:        }
                    220:        else if (*o_magic)
                    221:        {
                    222:                switch (c)
                    223:                {
                    224:                  case '^':
                    225:                        if (*sptr == retext + 1)
                    226:                        {
                    227:                                return M_BEGLINE;
                    228:                        }
                    229:                        return c;
                    230: 
                    231:                  case '$':
                    232:                        if (!**sptr)
                    233:                        {
                    234:                                return M_ENDLINE;
                    235:                        }
                    236:                        return c;
                    237: 
                    238:                  case '.':
                    239:                        return M_ANY;
                    240: 
                    241:                  case '*':
                    242:                        return M_SPLAT;
                    243: 
                    244:                  case '[':
                    245:                        /* make sure we don't have too many classes */
                    246:                        if (class_cnt >= 10)
                    247:                        {
                    248:                                FAIL("Too many []s");
                    249:                        }
                    250: 
                    251:                        /* process the character list for this class */
                    252:                        if (re)
                    253:                        {
                    254:                                /* generate the bitmap for this class */
                    255:                                *sptr = makeclass(*sptr, re->program + 1 + 32 * class_cnt);
                    256:                        }
                    257:                        else
                    258:                        {
                    259:                                /* skip to end of the class */
                    260:                                *sptr = makeclass(*sptr, (char *)0);
                    261:                        }
                    262:                        return M_CLASS(class_cnt++);
                    263: 
                    264:                  default:
                    265:                        return c;
                    266:                }
                    267:        }
                    268:        else    /* unquoted nomagic */
                    269:        {
                    270:                switch (c)
                    271:                {
                    272:                  case '^':
                    273:                        if (*sptr == retext + 1)
                    274:                        {
                    275:                                return M_BEGLINE;
                    276:                        }
                    277:                        return c;
                    278: 
                    279:                  case '$':
                    280:                        if (!**sptr)
                    281:                        {
                    282:                                return M_ENDLINE;
                    283:                        }
                    284:                        return c;
                    285: 
                    286:                  default:
                    287:                        return c;
                    288:                }
                    289:        }
                    290:        /*NOTREACHED*/
                    291: }
                    292: 
                    293: 
                    294: 
                    295: 
                    296: /* This function calculates the number of bytes that will be needed for a
                    297:  * compiled RE.  Its argument is the uncompiled version.  It is not clever
                    298:  * about catching syntax errors; that is done in a later pass.
                    299:  */
                    300: static unsigned calcsize(text)
                    301:        char            *text;
                    302: {
                    303:        unsigned        size;
                    304:        int             token;
                    305: 
                    306:        retext = text;
                    307:        class_cnt = 0;
                    308:        start_cnt = 1;
                    309:        end_sp = 0;
                    310:        size = 5;
                    311:        while ((token = gettoken(&text, (regexp *)0)) != 0)
                    312:        {
                    313:                if (IS_CLASS(token))
                    314:                {
                    315:                        size += 34;
                    316:                }
                    317:                else if (IS_META(token))
                    318:                {
                    319:                        size += 2;
                    320:                }
                    321:                else
                    322:                {
                    323:                        size++;
                    324:                }
                    325:        }
                    326: 
                    327:        return size;
                    328: }
                    329: 
                    330: 
                    331: 
                    332: /* This function compiles a regexp. */
                    333: regexp *regcomp(text)
                    334:        char            *text;
                    335: {
                    336:        int             needfirst;
                    337:        unsigned        size;
                    338:        int             token;
                    339:        int             peek;
                    340:        char            *build;
                    341:        regexp          *re;
                    342: 
                    343: 
                    344:        /* prepare for error handling */
                    345:        re = (regexp *)0;
                    346:        if (setjmp(errorhandler))
                    347:        {
                    348:                if (re)
                    349:                {
                    350:                        free(re);
                    351:                }
                    352:                return (regexp *)0;
                    353:        }
                    354: 
                    355:        /* if an empty regexp string was given, use the previous one */
                    356:        if (*text == 0)
                    357:        {
                    358:                if (!previous)
                    359:                {
                    360:                        FAIL("No previous RE");
                    361:                }
                    362:                text = previous;
                    363:        }
                    364:        else /* non-empty regexp given, so remember it */
                    365:        {
                    366:                if (previous)
                    367:                        free(previous);
                    368:                previous = (char *)malloc((unsigned)(strlen(text) + 1));
                    369:                if (previous)
                    370:                        strcpy(previous, text);
                    371:        }
                    372: 
                    373:        /* allocate memory */
                    374:        class_cnt = 0;
                    375:        start_cnt = 1;
                    376:        end_sp = 0;
                    377:        retext = text;
                    378:        size = calcsize(text) + sizeof(regexp);
                    379: #ifdef lint
                    380:        re = ((regexp *)0) + size;
                    381: #else
                    382:        re = (regexp *)malloc((unsigned)size);
                    383: #endif
                    384:        if (!re)
                    385:        {
                    386:                FAIL("Not enough memory for this RE");
                    387:        }
                    388: 
                    389:        /* compile it */
                    390:        build = &re->program[1 + 32 * class_cnt];
                    391:        re->program[0] = class_cnt;
                    392:        for (token = 0; token < NSUBEXP; token++)
                    393:        {
                    394:                re->startp[token] = re->endp[token] = (char *)0;
                    395:        }
                    396:        re->first = 0;
                    397:        re->bol = 0;
                    398:        re->minlen = 0;
                    399:        needfirst = 1;
                    400:        class_cnt = 0;
                    401:        start_cnt = 1;
                    402:        end_sp = 0;
                    403:        retext = text;
                    404:        for (token = M_START(0), peek = gettoken(&text, re);
                    405:             token;
                    406:             token = peek, peek = gettoken(&text, re))
                    407:        {
                    408:                /* special processing for the closure operator */
                    409:                if (IS_CLOSURE(peek))
                    410:                {
                    411:                        /* detect misuse of closure operator */
                    412:                        if (IS_START(token))
                    413:                        {
                    414:                                FAIL("* or \\+ or \\? follows nothing");
                    415:                        }
                    416:                        else if (IS_META(token) && token != M_ANY && !IS_CLASS(token))
                    417:                        {
                    418:                                FAIL("* or \\+ or \\? can only follow a normal character or . or []");
                    419:                        }
                    420: 
                    421:                        /* it is okay -- make it prefix instead of postfix */
                    422:                        ADD_META(build, peek);
                    423: 
                    424:                        /* take care of "needfirst" - is this the first char? */
                    425:                        if (needfirst && peek == M_PLUS && !IS_META(token))
                    426:                        {
                    427:                                re->first = token;
                    428:                        }
                    429:                        needfirst = 0;
                    430: 
                    431:                        /* we used "peek" -- need to refill it */
                    432:                        peek = gettoken(&text, re);
                    433:                        if (IS_CLOSURE(peek))
                    434:                        {
                    435:                                FAIL("* or \\+ or \\? doubled up");
                    436:                        }
                    437:                }
                    438:                else if (!IS_META(token))
                    439:                {
                    440:                        /* normal char is NOT argument of closure */
                    441:                        if (needfirst)
                    442:                        {
                    443:                                re->first = token;
                    444:                                needfirst = 0;
                    445:                        }
                    446:                        re->minlen++;
                    447:                }
                    448:                else if (token == M_ANY || IS_CLASS(token))
                    449:                {
                    450:                        /* . or [] is NOT argument of closure */
                    451:                        needfirst = 0;
                    452:                        re->minlen++;
                    453:                }
                    454: 
                    455:                /* the "token" character is not closure -- process it normally */
                    456:                if (token == M_BEGLINE)
                    457:                {
                    458:                        /* set the BOL flag instead of storing M_BEGLINE */
                    459:                        re->bol = 1;
                    460:                }
                    461:                else if (IS_META(token))
                    462:                {
                    463:                        ADD_META(build, token);
                    464:                }
                    465:                else
                    466:                {
                    467:                        *build++ = token;
                    468:                }
                    469:        }
                    470: 
                    471:        /* end it with a \) which MUST MATCH the opening \( */
                    472:        ADD_META(build, M_END(0));
                    473:        if (end_sp > 0)
                    474:        {
                    475:                FAIL("Not enough \\)s");
                    476:        }
                    477: 
                    478:        return re;
                    479: }
                    480: 
                    481: 
                    482: 
                    483: /*---------------------------------------------------------------------------*/
                    484: 
                    485: 
                    486: /* This function checks for a match between a character and a token which is
                    487:  * known to represent a single character.  It returns 0 if they match, or
                    488:  * 1 if they don't.
                    489:  */
                    490: int match1(re, ch, token)
                    491:        regexp          *re;
                    492:        REG char        ch;
                    493:        REG int         token;
                    494: {
                    495:        if (!ch)
                    496:        {
                    497:                /* the end of a line can't match any RE of width 1 */
                    498:                return 1;
                    499:        }
                    500:        if (token == M_ANY)
                    501:        {
                    502:                return 0;
                    503:        }
                    504:        else if (IS_CLASS(token))
                    505:        {
                    506:                if (re->program[1 + 32 * (token - M_CLASS(0)) + (ch >> 3)] & (1 << (ch & 7)))
                    507:                        return 0;
                    508:        }
                    509:        else if (ch == token
                    510:                || (*o_ignorecase && isupper(ch) && tolower(ch) == token))
                    511:        {
                    512:                return 0;
                    513:        }
                    514:        return 1;
                    515: }
                    516: 
                    517: 
                    518: 
                    519: /* This function checks characters up to and including the next closure, at
                    520:  * which point it does a recursive call to check the rest of it.  This function
                    521:  * returns 0 if everything matches, or 1 if something doesn't match.
                    522:  */
                    523: int match(re, str, prog, here)
                    524:        regexp          *re;    /* the regular expression */
                    525:        char            *str;   /* the string */
                    526:        REG char        *prog;  /* a portion of re->program, an compiled RE */
                    527:        REG char        *here;  /* a portion of str, the string to compare it to */
                    528: {
                    529:        REG int         token;
                    530:        REG int         nmatched;
                    531:        REG int         closure;
                    532: 
                    533:        for (token = GET_META(prog); !IS_CLOSURE(token); prog++, token = GET_META(prog))
                    534:        {
                    535:                switch (token)
                    536:                {
                    537:                /*case M_BEGLINE: can't happen; re->bol is used instead */
                    538:                  case M_ENDLINE:
                    539:                        if (*here)
                    540:                                return 1;
                    541:                        break;
                    542: 
                    543:                  case M_BEGWORD:
                    544:                        if (here != str &&
                    545:                           (here[-1] == '_' ||
                    546:                             isascii(here[-1]) && isalnum(here[-1])))
                    547:                                return 1;
                    548:                        break;
                    549: 
                    550:                  case M_ENDWORD:
                    551:                        if (here[0] == '_' || isascii(here[0]) && isalnum(here[0]))
                    552:                                return 1;
                    553:                        break;
                    554: 
                    555:                  case M_START(0):
                    556:                  case M_START(1):
                    557:                  case M_START(2):
                    558:                  case M_START(3):
                    559:                  case M_START(4):
                    560:                  case M_START(5):
                    561:                  case M_START(6):
                    562:                  case M_START(7):
                    563:                  case M_START(8):
                    564:                  case M_START(9):
                    565:                        re->startp[token - M_START(0)] = (char *)here;
                    566:                        break;
                    567: 
                    568:                  case M_END(0):
                    569:                  case M_END(1):
                    570:                  case M_END(2):
                    571:                  case M_END(3):
                    572:                  case M_END(4):
                    573:                  case M_END(5):
                    574:                  case M_END(6):
                    575:                  case M_END(7):
                    576:                  case M_END(8):
                    577:                  case M_END(9):
                    578:                        re->endp[token - M_END(0)] = (char *)here;
                    579:                        if (token == M_END(0))
                    580:                        {
                    581:                                return 0;
                    582:                        }
                    583:                        break;
                    584: 
                    585:                  default: /* literal, M_CLASS(n), or M_ANY */
                    586:                        if (match1(re, *here, token) != 0)
                    587:                                return 1;
                    588:                        here++;
                    589:                }
                    590:        }
                    591: 
                    592:        /* C L O S U R E */
                    593: 
                    594:        /* step 1: see what we have to match against, and move "prog" to point
                    595:         * the the remainder of the compiled RE.
                    596:         */
                    597:        closure = token;
                    598:        prog++, token = GET_META(prog);
                    599:        prog++;
                    600: 
                    601:        /* step 2: see how many times we can match that token against the string */
                    602:        for (nmatched = 0;
                    603:             (closure != M_QMARK || nmatched < 1) && *here && match1(re, *here, token) == 0;
                    604:             nmatched++, here++)
                    605:        {
                    606:        }
                    607: 
                    608:        /* step 3: try to match the remainder, and back off if it doesn't */
                    609:        while (nmatched >= 0 && match(re, str, prog, here) != 0)
                    610:        {
                    611:                nmatched--;
                    612:                here--;
                    613:        }
                    614: 
                    615:        /* so how did it work out? */
                    616:        if (nmatched >= ((closure == M_PLUS) ? 1 : 0))
                    617:                return 0;
                    618:        return 1;
                    619: }
                    620: 
                    621: 
                    622: 
                    623: /* This function searches through a string for text that matches an RE. */
                    624: int regexec(re, str, bol)
                    625:        regexp  *re;    /* the compiled regexp to search for */
                    626:        char    *str;   /* the string to search through */
                    627:        int     bol;    /* boolean: does str start at the beginning of a line? */
                    628: {
                    629:        char    *prog;  /* the entry point of re->program */
                    630:        int     len;    /* length of the string */
                    631:        REG char        *here;
                    632: 
                    633:        /* if must start at the beginning of a line, and this isn't, then fail */
                    634:        if (re->bol && !bol)
                    635:        {
                    636:                return 0;
                    637:        }
                    638: 
                    639:        len = strlen(str);
                    640:        prog = re->program + 1 + 32 * re->program[0];
                    641: 
                    642:        /* search for the RE in the string */
                    643:        if (re->bol)
                    644:        {
                    645:                /* must occur at BOL */
                    646:                if ((re->first
                    647:                        && match1(re, *(char *)str, re->first))/* wrong first letter? */
                    648:                 || len < re->minlen                    /* not long enough? */
                    649:                 || match(re, (char *)str, prog, str))  /* doesn't match? */
                    650:                        return 0;                       /* THEN FAIL! */
                    651:        }
                    652: #ifndef CRUNCH
                    653:        else if (!*o_ignorecase)
                    654:        {
                    655:                /* can occur anywhere in the line, noignorecase */
                    656:                for (here = (char *)str;
                    657:                     (re->first && re->first != *here)
                    658:                        || match(re, (char *)str, prog, here);
                    659:                     here++, len--)
                    660:                {
                    661:                        if (len < re->minlen)
                    662:                                return 0;
                    663:                }
                    664:        }
                    665: #endif
                    666:        else
                    667:        {
                    668:                /* can occur anywhere in the line, ignorecase */
                    669:                for (here = (char *)str;
                    670:                     (re->first && match1(re, *here, (int)re->first))
                    671:                        || match(re, (char *)str, prog, here);
                    672:                     here++, len--)
                    673:                {
                    674:                        if (len < re->minlen)
                    675:                                return 0;
                    676:                }
                    677:        }
                    678: 
                    679:        /* if we didn't fail, then we must have succeeded */
                    680:        return 1;
                    681: }
                    682: 
                    683: #else /* NO_MAGIC */
                    684: 
                    685: regexp *regcomp(exp)
                    686:        char    *exp;
                    687: {
                    688:        char    *src;
                    689:        char    *dest;
                    690:        regexp  *re;
                    691:        int     i;
                    692: 
                    693:        /* allocate a big enough regexp structure */
                    694: #ifdef lint
                    695:        re = (regexp *)0;
                    696: #else
                    697:        re = (regexp *)malloc((unsigned)(strlen(exp) + 1 + sizeof(struct regexp)));
                    698: #endif
                    699:        if (!re)
                    700:        {
                    701:                regerror("Could not malloc a regexp structure");
                    702:                return (regexp *)0;
                    703:        }
                    704: 
                    705:        /* initialize all fields of the structure */
                    706:        for (i = 0; i < NSUBEXP; i++)
                    707:        {
                    708:                re->startp[i] = re->endp[i] = (char *)0;
                    709:        }
                    710:        re->minlen = 0;
                    711:        re->first = 0;
                    712:        re->bol = 0;
                    713: 
                    714:        /* copy the string into it, translating ^ and $ as needed */
                    715:        for (src = exp, dest = re->program + 1; *src; src++)
                    716:        {
                    717:                switch (*src)
                    718:                {
                    719:                  case '^':
                    720:                        if (src == exp)
                    721:                        {
                    722:                                re->bol += 1;
                    723:                        }
                    724:                        else
                    725:                        {
                    726:                                *dest++ = '^';
                    727:                                re->minlen++;
                    728:                        }
                    729:                        break;
                    730: 
                    731:                  case '$':
                    732:                        if (!src[1])
                    733:                        {
                    734:                                re->bol += 2;
                    735:                        }
                    736:                        else
                    737:                        {
                    738:                                *dest++ = '$';
                    739:                                re->minlen++;
                    740:                        }
                    741:                        break;
                    742: 
                    743:                  case '\\':
                    744:                        if (src[1])
                    745:                        {
                    746:                                *dest++ = *++src;
                    747:                                re->minlen++;
                    748:                        }
                    749:                        else
                    750:                        {
                    751:                                regerror("extra \\ at end of regular expression");
                    752:                        }
                    753:                        break;
                    754: 
                    755:                  default:
                    756:                        *dest++ = *src;
                    757:                        re->minlen++;
                    758:                }
                    759:        }
                    760:        *dest = '\0';
                    761: 
                    762:        return re;
                    763: }
                    764: 
                    765: 
                    766: /* This "helper" function checks for a match at a given location.  It returns
                    767:  * 1 if it matches, 0 if it doesn't match here but might match later on in the
                    768:  * string, or -1 if it could not possibly match
                    769:  */
                    770: static int reghelp(prog, string, bolflag)
                    771:        struct regexp   *prog;
                    772:        char            *string;
                    773:        int             bolflag;
                    774: {
                    775:        char            *scan;
                    776:        char            *str;
                    777: 
                    778:        /* if ^, then require bolflag */
                    779:        if ((prog->bol & 1) && !bolflag)
                    780:        {
                    781:                return -1;
                    782:        }
                    783: 
                    784:        /* if it matches, then it will start here */
                    785:        prog->startp[0] = string;
                    786: 
                    787:        /* compare, possibly ignoring case */
                    788:        if (*o_ignorecase)
                    789:        {
                    790:                for (scan = &prog->program[1]; *scan; scan++, string++)
                    791:                        if (tolower(*scan) != tolower(*string))
                    792:                                return *string ? 0 : -1;
                    793:        }
                    794:        else
                    795:        {
                    796:                for (scan = &prog->program[1]; *scan; scan++, string++)
                    797:                        if (*scan != *string)
                    798:                                return *string ? 0 : -1;
                    799:        }
                    800: 
                    801:        /* if $, then require string to end here, too */
                    802:        if ((prog->bol & 2) && *string)
                    803:        {
                    804:                return 0;
                    805:        }
                    806: 
                    807:        /* if we get to here, it matches */
                    808:        prog->endp[0] = string;
                    809:        return 1;
                    810: }
                    811: 
                    812: 
                    813: 
                    814: int regexec(prog, string, bolflag)
                    815:        struct regexp   *prog;
                    816:        char            *string;
                    817:        int             bolflag;
                    818: {
                    819:        int             rc;
                    820: 
                    821:        /* keep trying to match it */
                    822:        for (rc = reghelp(prog, string, bolflag); rc == 0; rc = reghelp(prog, string, 0))
                    823:        {
                    824:                string++;
                    825:        }
                    826: 
                    827:        /* did we match? */
                    828:        return rc == 1;
                    829: }
                    830: #endif

unix.superglobalmegacorp.com

This archive runs on limited infrastructure. Preserving old code on modern bandwidth. Automated agents are requested to crawl responsibly.