Annotation of researchv8dc/cmd/spell/spellprog.c, revision 1.1.1.1

1.1       root        1: /*     @(#)spellprog.c 1.1     */
                      2: #include <stdio.h>
                      3: #include <ctype.h>
                      4: #define Tolower(c) (isupper(c)?tolower(c):c)
                      5: #define pair(a,b) (((a)<<8)|(b))
                      6: #define DLEV 2
                      7: 
                      8: char   *strcat();
                      9: int    strip();
                     10: int cstrip();
                     11: int    subst();
                     12: char   *skipv();
                     13: int    an();
                     14: int    s();
                     15: int    es();
                     16: int    ily();
                     17: int    CCe();
                     18: int    VCe();
                     19: int    bility();
                     20: int    tion();
                     21: int    ize();
                     22: int    y_to_e();
                     23: int    i_to_y();
                     24: int    nop();
                     25: 
                     26: struct suftab {
                     27:        char *suf;
                     28:        int (*p1)();
                     29:        int n1;
                     30:        char *d1;
                     31:        char *a1;
                     32:        int (*p2)();
                     33:        int n2;
                     34:        char *d2;
                     35:        char *a2;
                     36: };
                     37: struct suftab stabc[] = {
                     38:        {"citsi",strip,2,"","+ic"},
                     39:        {"citi",ize,1,"-e+ic",""},
                     40:        {"cihparg",i_to_y,1,"-y+ic",""},
                     41:        {"cipocs",ize,1,"-e+ic",""},
                     42:        {"cirtem",i_to_y,1,"-y+ic",""},
                     43:        {"cigol",i_to_y,1,"-y+ic",""},
                     44:        0
                     45: };
                     46: struct suftab stabd[] = {
                     47:        {"de",strip,1,"","+d",          i_to_y,2,"-y+ied","+ed"},
                     48:        {"dooh",ily,4,"-y+ihood","+hood"},
                     49:        0
                     50: };
                     51: struct suftab stabe[] = {
                     52:        {"ecn",subst,1,"-t+ce",""},
                     53:        {"elbaif",i_to_y,4,"-y+iable",""},
                     54:        {"elba",CCe,4,"-e+able","+able"},
                     55:        {"evi",subst,0,"-ion+ive",""},
                     56:        {"ezi",CCe,3,"-e+ize","+ize"},
                     57:        {"ekil",strip,4,"","+like"},
                     58:        0
                     59: };
                     60: struct suftab stabg[] = {
                     61:        {"gni",CCe,3,"-e+ing","+ing"},
                     62:        0
                     63: };
                     64: struct suftab stabl[] = {
                     65:        {"laci",strip,2,"","+al"},
                     66:        {"latnem",strip,2,"","+al"},
                     67:        {"lanoi",strip,2,"","+al"},
                     68:        {"luf",ily,3,"-y+iful","+ful"},
                     69:        0
                     70: };
                     71: struct suftab stabm[] = {
                     72:        {"msi",CCe,3,"-e+ism","ism"},
                     73:        0
                     74: };
                     75: struct suftab stabn[] = {
                     76:        {"noitacifi",i_to_y,6,"-y+ication",""},
                     77:        {"noitazi",ize,4,"-e+ation",""},
                     78:        {"noit",tion,3,"-e+ion","+ion"},
                     79:        {"naino",an,3,"","+ian"},
                     80:        {"na",an,1,"","+n"},
                     81:        0
                     82: };
                     83: struct suftab stabp[] = {
                     84:        {"pihs",strip,4,"","+ship"},
                     85:        0
                     86: };
                     87: struct suftab stabr[] = {
                     88:        {"reta",nop,0,"",""},
                     89:        {"retc",nop,0,"",""},
                     90:        {"re",strip,1,"","+r",          i_to_y,2,"-y+ier","+er"},
                     91:        {"rota",tion,2,"-e+or",""},
                     92:        {"rotc",tion,2,"","+or"},
                     93:        0
                     94: };
                     95: struct suftab stabs[] = {
                     96:        {"ssen",ily,4,"-y+iness","+ness" },
                     97:        {"ssel",ily,4,"-y+i+less","+less" },
                     98:        {"se",s,1,"","+s",              es,2,"-y+ies","+es" },
                     99:        {"s'",s,2,"","+'s"},
                    100:        {"s",s,1,"","+s"},
                    101:        0
                    102: };
                    103: struct suftab stabt[] = {
                    104:        {"tnem",strip,4,"","+ment"},
                    105:        {"tse",strip,2,"","+st",        i_to_y,3,"-y+iest","+est"},
                    106:        {"tsigol",i_to_y,2,"-y+ist",""},
                    107:        {"tsi",CCe,3,"-e+ist","+ist"},
                    108:        0
                    109: };
                    110: struct suftab staby[] = {
                    111:        {"ycn",subst,1,"-t+cy",""},
                    112:        {"ytilb",nop,0,"",""},
                    113:        {"ytilib",bility,5,"-le+ility",""},
                    114:        {"yti",CCe,3,"-e+ity","+ity"},
                    115:        {"ylb",y_to_e,1,"-e+y",""},
                    116:        {"ylc",nop,0,"",""},
                    117:        {"yl",ily,2,"-y+ily","+ly"},
                    118:        {"yrtem",subst,0,"-er+ry",""},
                    119:        0
                    120: };
                    121: struct suftab stabz[] = {
                    122:        0
                    123: };
                    124: struct suftab *suftab[] = {
                    125:        stabz,
                    126:        stabz,
                    127:        stabc,
                    128:        stabd,
                    129:        stabe,
                    130:        stabz,
                    131:        stabg,
                    132:        stabz,
                    133:        stabz,
                    134:        stabz,
                    135:        stabz,
                    136:        stabl,
                    137:        stabm,
                    138:        stabn,
                    139:        stabz,
                    140:        stabp,
                    141:        stabz,
                    142:        stabr,
                    143:        stabs,
                    144:        stabt,
                    145:        stabz,
                    146:        stabz,
                    147:        stabz,
                    148:        stabz,
                    149:        staby,
                    150:        stabz,
                    151: };
                    152: 
                    153: 
                    154: char *ptaba[] = {
                    155:        "anti",
                    156:        "auto",
                    157:        0
                    158: };
                    159: char *ptabb[] = {
                    160:        "bio",
                    161:        0
                    162: };
                    163: char *ptabc[] = {
                    164:        "counter",
                    165:        0
                    166: };
                    167: char *ptabd[] = {
                    168:        "dis",
                    169:        0
                    170: };
                    171: char *ptabe[] = {
                    172:        "electro",
                    173:        "en",
                    174:        0
                    175: };
                    176: char *ptabf[] = {
                    177:        "femto",
                    178:        "fore",
                    179:        0
                    180: };
                    181: char *ptabg[] = {
                    182:        "geo",
                    183:        "giga",
                    184:        0
                    185: };
                    186: char *ptabh[] = {
                    187:        "hyper",
                    188:        0
                    189: };
                    190: char *ptabi[] = {
                    191:        "intra",
                    192:        "inter",
                    193:        "iso",
                    194:        0
                    195: };
                    196: char *ptabj[] = {
                    197:        0
                    198: };
                    199: char *ptabk[] = {
                    200:        "kilo",
                    201:        0
                    202: };
                    203: char *ptabl[] = {
                    204:        0
                    205: };
                    206: char *ptabm[] = {
                    207:        "magneto",
                    208:        "mega",
                    209:        "meta",
                    210:        "micro",
                    211:        "mid",
                    212:        "milli",
                    213:        "mis",
                    214:        "mono",
                    215:        "multi",
                    216:        0
                    217: };
                    218: char *ptabn[] = {
                    219:        "nano",
                    220:        "non",
                    221:        0
                    222: };
                    223: char *ptabo[] = {
                    224:        "out",
                    225:        "over",
                    226:        0
                    227: };
                    228: char *ptabp[] = {
                    229:        "photo",
                    230:        "pico",
                    231:        "poly",
                    232:        "pre",
                    233:        "pseudo",
                    234:        "psycho",
                    235:        0
                    236: };
                    237: char *ptabq[] = {
                    238:        "quasi",
                    239:        0
                    240: };
                    241: char *ptabr[] = {
                    242:        "re",
                    243:        0
                    244: };
                    245: char *ptabs[] = {
                    246:        "semi",
                    247:        "stereo",
                    248:        "sub",
                    249:        "super",
                    250:        0
                    251: };
                    252: char *ptabt[] = {
                    253:        "tele",
                    254:        "thermo",
                    255:        0
                    256: };
                    257: char *ptabu[] = {
                    258:        "ultra",
                    259:        "under",        /*must precede un*/
                    260:        "un",
                    261:        0
                    262: };
                    263: char *ptabv[] = {
                    264:        0
                    265: };
                    266: char *ptabw[] = {
                    267:        0
                    268: };
                    269: char *ptabx[] = {
                    270:        0
                    271: };
                    272: char *ptaby[] = {
                    273:        0
                    274: };
                    275: char *ptabz[] = {
                    276:        0
                    277: };
                    278: 
                    279: char **preftab[] = {
                    280:        ptaba,
                    281:        ptabb,
                    282:        ptabc,
                    283:        ptabd,
                    284:        ptabe,
                    285:        ptabf,
                    286:        ptabg,
                    287:        ptabh,
                    288:        ptabi,
                    289:        ptabj,
                    290:        ptabk,
                    291:        ptabl,
                    292:        ptabm,
                    293:        ptabn,
                    294:        ptabo,
                    295:        ptabp,
                    296:        ptabq,
                    297:        ptabr,
                    298:        ptabs,
                    299:        ptabt,
                    300:        ptabu,
                    301:        ptabv,
                    302:        ptabw,
                    303:        ptabx,
                    304:        ptaby,
                    305:        ptabz,
                    306: };
                    307: 
                    308: int bflag;
                    309: int vflag;
                    310: int xflag;
                    311: char word[100];
                    312: char original[100];
                    313: char *deriv[40];
                    314: char affix[40];
                    315: char errmsg[] = "spell: cannot initialize hash table\n";
                    316: FILE *found;
                    317: /*     deriv is stack of pointers to notes like +micro +ed
                    318:  *     affix is concatenated string of notes
                    319:  *     the buffer size 141 stems from the sizes of original and affix.
                    320: */
                    321: 
                    322: /*
                    323:  *     in an attempt to defray future maintenance misunderstandings, here is an attempt to
                    324:  *     describe the input/output expectations of the spell program.
                    325:  *
                    326:  *     spellprog is intended to be called from the shell file spell.
                    327:  *     because of this, there is little error checking (this is historical, not
                    328:  *     necessarily advisable).
                    329:  *
                    330:  *     spellprog hashed-list pass options
                    331:  *
                    332:  *     the hashed-list is a list of the form made by spellin.
                    333:  *     there are 2 types of hashed lists:
                    334:  *             1. a stop list: this specifies words that by the rules embodied in
                    335:  *                spellprog would be recognized as correct, BUT are really errors.
                    336:  *             2. a dictionary of correctly spelled words.
                    337:  *     the pass number determines how the words found in the specified hashed-list
                    338:  *     are treated. If the pass number is 1, the hashed-list is treated as the stop-list,
                    339:  *     otherwise, it is treated as the regular dictionary list. in this case, the
                    340:  *     value of "pass" is a filename. Found words are written to this file.
                    341:  *     In the normal case, the filename = /dev/null. However, if the v option is
                    342:  *     specified, the derivations are written to this file.
                    343:  *     the spellprog looks up words in the hashed-list; if a word is found, it is
                    344:  *     printed to the stdout. if the hashed-list was the stop-list, the words found
                    345:  *     are presumed to be misspellings. in this case,
                    346:  *     a control character is printed ( a "-" is appended to the word.
                    347:  *     a hyphen will never occur naturally in the input list because deroff
                    348:  *     is used in the shell file before calling spellprog.)
                    349:  *     if the regualar spelling list was used (hlista or hlistb), the words are correct,
                    350:  *     and may be ditched. (unless the -v option was used - see the manual page).
                    351: 
                    352:  *     spellprog should be called twice : first with the stop-list, to flag all
                    353:  *     a priori incorrectly spelled words; second with the dictionary.
                    354:  *
                    355:  *     spellprog hstop 1 |\
                    356:  *     spellprog hlista /dev/null 
                    357:  *
                    358:  *     for a complete scenario, see the shell file: spell.
                    359:  *
                    360:  */
                    361: main(argc,argv)
                    362: char **argv;
                    363: {
                    364:        register char *ep, *cp;
                    365:        register char *dp;
                    366:        int fold;
                    367:        int j;
                    368:        int pass;
                    369:        if(!prime(argc,argv)) {
                    370:                fwrite(errmsg, sizeof(*errmsg), sizeof(errmsg), stderr);
                    371:                exit(1);
                    372:        }
                    373: /*
                    374:  *     if pass is not 1, it is assumed to be a filename.
                    375:  *     found words are written to this file.
                    376:  */
                    377:        pass = argv[2][0] ;
                    378:        if ( pass != '1' )
                    379:                found = fopen(argv[2],"w");
                    380:        for(argc-=3,argv+=3; argc>0 && argv[0][0]=='-'; argc--,argv++)
                    381:                switch(argv[0][1]) {
                    382:                case 'b':
                    383:                        bflag++;
                    384:                        ise();
                    385:                        break;
                    386:                case 'v':
                    387:                        vflag++;
                    388:                        break;
                    389:                case 'x':
                    390:                        xflag++;
                    391:                        break;
                    392:                }
                    393:        for(;;) {
                    394:                affix[0] = 0;
                    395:                for(ep=word;(*ep=j=getchar())!='\n';ep++)
                    396:                        if(j == EOF)
                    397:                                exit(0);
                    398:                for(cp=word,dp=original; cp<ep; )
                    399:                        *dp++ = *cp++;
                    400:                *dp = 0;
                    401: /*
                    402:  *     here is the hyphen processing. these words were found in the stop
                    403:  *     list. however, if they exist as is, (no derivations tried) in the
                    404:  *     dictionary, let them through as correct.
                    405:  *
                    406:  */
                    407:                if(ep[-1]=='-') {
                    408:                        *--dp = *--ep = 0;
                    409:                        if(tryword(word,ep,0))
                    410:                                continue;
                    411:                        if(!isupper(word[0])) 
                    412:                                goto notfound;
                    413:                        if(isupper(word[1])) {
                    414:                                for(cp=word+1;cp<ep;cp++) {
                    415:                                        if(islower(*cp))
                    416:                                                goto notfound;
                    417:                                        *cp = Tolower(*cp);
                    418:                                }
                    419:                        }
                    420:                        if(tryword(word,ep,0))
                    421:                                continue;
                    422:                        word[0] = Tolower(word[0]);
                    423:                        if(tryword(word,ep,0))
                    424:                                continue;
                    425:                        goto notfound;
                    426:                }
                    427:                fold = 0;
                    428:                for(cp=word;cp<ep;cp++)
                    429:                        if(islower(*cp))
                    430:                                goto lcase;
                    431:                if(trypref(ep,".",0))
                    432:                        goto foundit;
                    433:                ++fold;
                    434:                for(cp=original+1,dp=word+1;dp<ep;dp++,cp++)
                    435:                        *dp = Tolower(*cp);
                    436: lcase:
                    437:                if(trypref(ep,".",0)||trysuff(ep,0))
                    438:                        goto foundit;
                    439:                if(isupper(word[0])) {
                    440:                        for(cp=original,dp=word; *dp = *cp++; dp++)
                    441:                                if (fold) *dp = Tolower(*dp);
                    442:                        word[0] = Tolower(word[0]);
                    443:                        goto lcase;
                    444:                }
                    445: notfound:
                    446:                printf("%s\n", original);
                    447:                continue;
                    448: 
                    449: foundit:
                    450:                if(pass=='1')
                    451:                        printf("%s-\n", original);
                    452:                else if(affix[0]!=0 && affix[0]!='.') {
                    453:                        fprintf(found, "%s\t%s\n", affix, original);
                    454:                }
                    455:        }
                    456: }
                    457: 
                    458: /*     strip exactly one suffix and do
                    459:  *     indicated routine(s), which may recursively
                    460:  *     strip suffixes
                    461: */
                    462: trysuff(ep,lev)
                    463: char *ep;
                    464: {
                    465:        register struct suftab *t;
                    466:        register char *cp, *sp;
                    467:        int initchar = ep[-1];
                    468:        lev += DLEV;
                    469:        deriv[lev] = deriv[lev-1] = 0;
                    470:        if(!islower(initchar))
                    471:                return(0);
                    472:        for(t=suftab[initchar-'a'];sp=t->suf;t++) {
                    473:                cp = ep;
                    474:                while(*sp)
                    475:                        if(*--cp!=*sp++)
                    476:                                goto next;
                    477:                for(sp=cp; --sp>=word&&!vowel(*sp); ) ;
                    478:                if(sp<word)
                    479:                        return(0);
                    480:                if((*t->p1)(ep-t->n1,t->d1,t->a1,lev+1))
                    481:                        return(1);
                    482:                if(t->p2!=0) {
                    483:                        deriv[lev] = deriv[lev+1] = 0;
                    484:                        return((*t->p2)(ep-t->n2,t->d2,t->a2,lev));
                    485:                }
                    486:                return(0);
                    487: next:          ;
                    488:        }
                    489:        return(0);
                    490: }
                    491: 
                    492: nop()
                    493: {
                    494:        return(0);
                    495: }
                    496: 
                    497: cstrip(ep,d,a,lev)
                    498: char *ep,*d,*a;
                    499: {
                    500:        register temp = ep[0];
                    501:        if(vowel(temp)&&vowel(ep[-1])) {
                    502:                switch(pair(ep[-1],ep[0])) {
                    503:                case pair('a', 'a'):
                    504:                case pair('a', 'e'):
                    505:                case pair('a', 'i'):
                    506:                case pair('e', 'a'):
                    507:                case pair('e', 'e'):
                    508:                case pair('e', 'i'):
                    509:                case pair('i', 'i'):
                    510:                case pair('o', 'a'):
                    511:                        return(0);
                    512:                }
                    513:        } else if(temp==ep[-1]&&temp==ep[-2])
                    514:                return(0);
                    515:        return(strip(ep,d,a,lev));
                    516: }
                    517: 
                    518: strip(ep,d,a,lev)
                    519: char *ep,*d,*a;
                    520: {
                    521:        return(trypref(ep,a,lev)||trysuff(ep,lev));
                    522: }
                    523: 
                    524: s(ep,d,a,lev)
                    525: char *ep,*d,*a;
                    526: {
                    527:        if(lev>DLEV+1)
                    528:                return(0);
                    529:        if(*ep=='s') {
                    530:                switch(ep[-1]) {
                    531:                case 'y':
                    532:                        if(vowel(ep[-2]))
                    533:                                break;
                    534:                case 'x':
                    535:                case 'z':
                    536:                case 's':
                    537:                        return(0);
                    538:                case 'h':
                    539:                        switch(ep[-2]) {
                    540:                        case 'c':
                    541:                        case 's':
                    542:                                return(0);
                    543:                        }
                    544:                }
                    545:        }
                    546:        return(strip(ep,d,a,lev));
                    547: }
                    548: 
                    549: an(ep,d,a,lev)
                    550: char *ep,*d,*a;
                    551: {
                    552:        if(!isupper(*word))     /*must be proper name*/
                    553:                return(0);
                    554:        return(trypref(ep,a,lev));
                    555: }
                    556: 
                    557: ize(ep,d,a,lev)
                    558: char *ep,*d,*a;
                    559: {
                    560:        register temp = ep[-1];
                    561:        register val;
                    562:        ep[-1] = 'e';
                    563:        val = strip(ep,"",d,lev);
                    564:        ep[-1] = temp;
                    565:        return(val);
                    566: }
                    567: 
                    568: y_to_e(ep,d,a,lev)
                    569: char *ep,*d,*a;
                    570: {
                    571:        register val, temp;
                    572:        switch(ep[-1]) {
                    573:        case 'a':
                    574:        case 'e':
                    575:        case 'i':
                    576:                return 0;
                    577:        }
                    578:        temp = *ep;
                    579:        *ep++ = 'e';
                    580:        val = strip(ep,"",d,lev);
                    581:        *--ep = temp;
                    582:        return(val);
                    583: }
                    584: 
                    585: ily(ep,d,a,lev)
                    586: char *ep,*d,*a;
                    587: {
                    588:        register temp = ep[0];
                    589:        if(temp==ep[-1]&&temp==ep[-2])
                    590:                return(0);
                    591:        if(ep[-1]=='i')
                    592:                return(i_to_y(ep,d,a,lev));
                    593:        else
                    594:                return(cstrip(ep,d,a,lev));
                    595: }
                    596: 
                    597: bility(ep,d,a,lev)
                    598: char *ep,*d,*a;
                    599: {
                    600:        *ep++ = 'l';
                    601:        return(y_to_e(ep,d,a,lev));
                    602: }
                    603: 
                    604: i_to_y(ep,d,a,lev)
                    605: char *ep,*d,*a;
                    606: {
                    607:        register val, temp;
                    608:        if((temp=ep[-1])=='i'&&!vowel(ep[-2])) {
                    609:                ep[-1] = 'y';
                    610:                a = d;
                    611:        }
                    612:        val = cstrip(ep,"",a,lev);
                    613:        ep[-1] = temp;
                    614:        return(val);
                    615: }
                    616: 
                    617: es(ep,d,a,lev)
                    618: char *ep,*d,*a;
                    619: {
                    620:        if(lev>DLEV)
                    621:                return(0);
                    622:        switch(ep[-1]) {
                    623:        default:
                    624:                return(0);
                    625:        case 'i':
                    626:                return(i_to_y(ep,d,a,lev));
                    627:        case 's':
                    628:        case 'h':
                    629:        case 'z':
                    630:        case 'x':
                    631:                return(strip(ep,d,a,lev));
                    632:        }
                    633: }
                    634: 
                    635: subst(ep,d,a,lev)
                    636: char *ep, *d, *a;
                    637: {
                    638:        char *u,*t;
                    639:        int val;
                    640:        if(skipv(skipv(ep-1))<word)
                    641:                return(0);
                    642:        for(t=d;*t!='+';t++)
                    643:                continue;
                    644:        for(u=ep;*--t!='-'; )
                    645:                *--u = *t;
                    646:        val = strip(ep,"",d,lev);
                    647:        while(*++t!='+')
                    648:                continue;
                    649:        while(*++t)
                    650:                *u++ = *t;
                    651:        return(val);
                    652: }
                    653: 
                    654: 
                    655: tion(ep,d,a,lev)
                    656: char *ep,*d,*a;
                    657: {
                    658:        switch(ep[-2]) {
                    659:        case 'c':
                    660:        case 'r':
                    661:                return(trypref(ep,a,lev));
                    662:        case 'a':
                    663:                return(y_to_e(ep,d,a,lev));
                    664:        }
                    665:        return(0);
                    666: }
                    667: 
                    668: /*     possible consonant-consonant-e ending*/
                    669: CCe(ep,d,a,lev)
                    670: char *ep,*d,*a;
                    671: {
                    672:        switch(ep[-1]) {
                    673:        case 'l':
                    674:                if(vowel(ep[-2]))
                    675:                        break;
                    676:                switch(ep[-2]) {
                    677:                case 'l':
                    678:                case 'r':
                    679:                case 'w':
                    680:                        break;
                    681:                default:
                    682:                        return(y_to_e(ep,d,a,lev));
                    683:                }
                    684:                break;
                    685:        case 's':
                    686:                if(ep[-2]=='s')
                    687:                        break;
                    688:        case 'c':
                    689:        case 'g':
                    690:                if(*ep=='a')
                    691:                        return(0);
                    692:        case 'v':
                    693:        case 'z':
                    694:                if(vowel(ep[-2]))
                    695:                        break;
                    696:        case 'u':
                    697:                if(y_to_e(ep,d,a,lev))
                    698:                        return(1);
                    699:                if(!(ep[-2]=='n'&&ep[-1]=='g'))
                    700:                        return(0);
                    701:        }
                    702:        return(VCe(ep,d,a,lev));
                    703: }
                    704: 
                    705: /*     possible consonant-vowel-consonant-e ending*/
                    706: VCe(ep,d,a,lev)
                    707: char *ep,*d,*a;
                    708: {
                    709:        char c;
                    710:        c = ep[-1];
                    711:        if(c=='e')
                    712:                return(0);
                    713:        if(!vowel(c) && vowel(ep[-2])) {
                    714:                if(bflag&&c=='l'&&ep[-2]=='e')  /* UK modelled vs US modeled */
                    715:                        return(0);
                    716:                c = *ep;
                    717:                *ep++ = 'e';
                    718:                if(trypref(ep,d,lev)||trysuff(ep,lev))
                    719:                        return(1);
                    720:                ep--;
                    721:                *ep = c;
                    722:        }
                    723:        return(cstrip(ep,d,a,lev));
                    724: }
                    725: 
                    726: char *lookuppref(wp,ep)
                    727: char **wp;
                    728: char *ep;
                    729: {
                    730:        register char **sp;
                    731:        register char *bp,*cp;
                    732:        int initchar = Tolower(**wp);
                    733:        if(!isalpha(initchar))
                    734:                return(0);
                    735:        for(sp=preftab[initchar-'a'];*sp;sp++) {
                    736:                bp = *wp;
                    737:                for(cp= *sp;*cp;cp++,bp++)
                    738:                        if(Tolower(*bp)!=*cp)
                    739:                                goto next;
                    740:                for(cp=bp;cp<ep;cp++) 
                    741:                        if(vowel(*cp)) {
                    742:                                *wp = bp;
                    743:                                return(*sp);
                    744:                        }
                    745: next:  ;
                    746:        }
                    747:        return(0);
                    748: }
                    749: 
                    750: /*     while word is not in dictionary try stripping
                    751:  *     prefixes. Fail if no more prefixes.
                    752: */
                    753: trypref(ep,a,lev)
                    754: char *ep,*a;
                    755: {
                    756:        register char *cp;
                    757:        char *bp;
                    758:        register char *pp;
                    759:        int val = 0;
                    760:        char space[20];
                    761:        deriv[lev] = a;
                    762:        if(tryword(word,ep,lev))
                    763:                return(1);
                    764:        bp = word;
                    765:        pp = space;
                    766:        deriv[lev+1] = pp;
                    767:        while(cp=lookuppref(&bp,ep)) {
                    768:                *pp++ = '+';
                    769:                while(*pp = *cp++)
                    770:                        pp++;
                    771:                if(tryword(bp,ep,lev+1)) {
                    772:                        val = 1;
                    773:                        break;
                    774:                }
                    775:        }
                    776:        deriv[lev+1] = deriv[lev+2] = 0;
                    777:        return(val);
                    778: }
                    779: 
                    780: tryword(bp,ep,lev)
                    781: char *bp,*ep;
                    782: {
                    783:        register i, j;
                    784:        char duple[3];
                    785:        if(ep-bp<=1)
                    786:                return(0);
                    787:        if(vowel(*ep)) {
                    788:                if(monosyl(bp,ep))
                    789:                        return(0);
                    790:        }
                    791:        i = dict(bp,ep);
                    792:        /* doubled consonant in monosyllables and UK 'modelled', etc */
                    793:        if(i==0&&vowel(*ep)&&ep[-1]==ep[-2]&&
                    794:                (monosyl(bp,ep-1)||bflag&&ep[-1]=='l'&&ep[-3]=='e')) {
                    795:                ep--;
                    796:                deriv[++lev] = duple;
                    797:                duple[0] = '+';
                    798:                duple[1] = *ep;
                    799:                duple[2] = 0;
                    800:                i = dict(bp,ep);
                    801:        }
                    802:        if(vflag==0||i==0)
                    803:                return(i);
                    804:        /*      when derivations are wanted, collect them
                    805:         *      for printing
                    806:        */
                    807:        j = lev;
                    808:        do {
                    809:                if(deriv[j])
                    810:                        strcat(affix,deriv[j]);
                    811:        } while(--j>0);
                    812:        return(i);
                    813: }
                    814: 
                    815: 
                    816: monosyl(bp,ep)
                    817: char *bp, *ep;
                    818: {
                    819:        if(ep<bp+2)
                    820:                return(0);
                    821:        if(vowel(*--ep)||!vowel(*--ep)
                    822:                ||ep[1]=='x'||ep[1]=='w')
                    823:                return(0);
                    824:        while(--ep>=bp)
                    825:                if(vowel(*ep)&&!(ep[-1]=='q'&&*ep=='u'))
                    826:                        return(0);
                    827:        return(1);
                    828: }
                    829: 
                    830: char *
                    831: skipv(s)
                    832: char *s;
                    833: {
                    834:        if(s>=word&&vowel(*s))
                    835:                s--;
                    836:        while(s>=word&&!vowel(*s))
                    837:                s--;
                    838:        return(s);
                    839: }
                    840: 
                    841: vowel(c)
                    842: {
                    843:        switch(Tolower(c)) {
                    844:        case 'a':
                    845:        case 'e':
                    846:        case 'i':
                    847:        case 'o':
                    848:        case 'u':
                    849:        case 'y':
                    850:                return(1);
                    851:        }
                    852:        return(0);
                    853: }
                    854: 
                    855: /* crummy way to Britishise */
                    856: ise()
                    857: {
                    858:        register struct suftab *p;
                    859:        register i;
                    860:        for(i=0; i<26; i++) {
                    861:                for(p = suftab[i];p->suf;p++) {
                    862:                        ztos(p->suf);
                    863:                        ztos(p->d1);
                    864:                        ztos(p->a1);
                    865:                }
                    866:        }
                    867: }
                    868: ztos(s)
                    869: char *s;
                    870: {
                    871:        for(;*s;s++)
                    872:                if(*s=='z')
                    873:                        *s = 's';
                    874: }
                    875: 
                    876: dict(bp,ep)
                    877: char *bp, *ep;
                    878: {
                    879:        register temp, result;
                    880:        if(xflag)
                    881:                fprintf(stderr, "=%.*s\n", ep-bp, bp);
                    882:        temp = *ep;
                    883:        *ep = 0;
                    884:        result = hashlook(bp);
                    885:        *ep = temp;
                    886:        return(result);
                    887: }

unix.superglobalmegacorp.com

This archive runs on limited infrastructure. Preserving old code on modern bandwidth. Automated agents are requested to crawl responsibly.