|
|
1.1 ! root 1: /* ! 2: * Prepare text for statistical processing ! 3: * by breaking it into words (and possibly ! 4: * also punctuation marks) and discarding ! 5: * certain words if this is desired. ! 6: */ ! 7: ! 8: #include <stdio.h> ! 9: #include <ctype.h> ! 10: ! 11: #define NHASH 64 /* Hash buckets for ignore and only */ ! 12: #define NWORD 400 /* Longest word */ ! 13: ! 14: typedef struct WORDS { ! 15: struct WORDS *w_next; ! 16: char w_name[]; ! 17: } WORDS; ! 18: ! 19: WORDS *words[NHASH]; ! 20: ! 21: int pflag; /* Print punctuation as well */ ! 22: int dflag; /* Print (input) word numbers */ ! 23: int fflag; /* Fold upper into lower case */ ! 24: int iflag; /* Ignore case for '-i' */ ! 25: int nignore; /* Number of ignored words */ ! 26: int nonly; /* Number of only words */ ! 27: long wordno; /* Input word number */ ! 28: ! 29: char wordbuf[NWORD]; ! 30: ! 31: char missing[] = "Missing `%s' file argument"; ! 32: char onlyone[] = "Only one of `-i' or `-o' may be given"; ! 33: ! 34: main(argc, argv) ! 35: int argc; ! 36: char *argv[]; ! 37: { ! 38: register char *ap; ! 39: register int i; ! 40: register int estat = 0; ! 41: register FILE *fp; ! 42: ! 43: while (argc>1 && *argv[1]=='-') { ! 44: for (ap = &argv[1][1]; *ap != '\0'; ap++) ! 45: switch (*ap) { ! 46: case 'd': ! 47: dflag = 1; ! 48: break; ! 49: ! 50: case 'f': ! 51: fflag = 1; ! 52: break; ! 53: ! 54: case 'p': ! 55: pflag = 1; ! 56: break; ! 57: ! 58: case 'i': ! 59: if (nonly) ! 60: preperr(onlyone); ! 61: if (argc < 3) ! 62: preperr(missing, "ignore"); ! 63: argv++; ! 64: argc--; ! 65: nignore += enter(argv[1]); ! 66: iflag = 1; ! 67: break; ! 68: ! 69: case 'o': ! 70: if (nignore) ! 71: preperr(onlyone); ! 72: if (argc < 3) ! 73: preperr(missing, "only"); ! 74: argv++; ! 75: argc--; ! 76: nonly += enter(argv[1]); ! 77: break; ! 78: ! 79: default: ! 80: usage(); ! 81: } ! 82: argv++; ! 83: argc--; ! 84: } ! 85: if (argc > 1) ! 86: for (i=1; i<argc; i++) { ! 87: if ((fp = fopen(argv[i], "r")) == NULL) ! 88: preperr("Cannot open `%s'", argv[i]); ! 89: estat |= prep(fp); ! 90: fclose(fp); ! 91: } ! 92: else ! 93: estat = prep(stdin); ! 94: exit(estat); ! 95: } ! 96: ! 97: /* ! 98: * Run prep on each input file. ! 99: */ ! 100: prep(fp) ! 101: FILE *fp; ! 102: { ! 103: register char *cp; ! 104: register int c; ! 105: register int inword = 0; ! 106: ! 107: while ((c = getc(fp)) != EOF) { ! 108: if (!isascii(c)) ! 109: c = '\0'; ! 110: if (fflag && isupper(c)) ! 111: c = tolower(c); ! 112: if (inword) { ! 113: if (isalpha(c) || c=='\'') { ! 114: *cp++ = c; ! 115: continue; ! 116: } ! 117: if (c == '-') { ! 118: if ((c = getc(fp)) == '\n') ! 119: continue; ! 120: ungetc(c, fp); ! 121: c = '-'; ! 122: } ! 123: *cp = '\0'; ! 124: inword = 0; ! 125: wordno++; ! 126: print(wordbuf); ! 127: } ! 128: if (isalpha(c) || c=='\'') { ! 129: inword++; ! 130: cp = wordbuf; ! 131: *cp++ = c; ! 132: } else if (pflag && ispunct(c)) { ! 133: putchar(c); ! 134: putchar('\n'); ! 135: } ! 136: } ! 137: } ! 138: ! 139: /* ! 140: * Print out a word. ! 141: */ ! 142: print(word) ! 143: char *word; ! 144: { ! 145: if ((nignore && lookup(word)) || (nonly && !lookup(word))) ! 146: return; ! 147: if (dflag) ! 148: printf("%ld\t", wordno); ! 149: printf("%s\n", wordbuf); ! 150: } ! 151: ! 152: /* ! 153: * Enter words from the given file ! 154: * into the hash table. ! 155: */ ! 156: enter(fn) ! 157: char *fn; ! 158: { ! 159: register char *cp; ! 160: register WORDS *wp; ! 161: register int c; ! 162: register unsigned hash; ! 163: register int nword = 0; ! 164: register FILE *fp; ! 165: ! 166: if ((fp = fopen(fn, "r")) == NULL) ! 167: preperr("Cannot open `%s'", fn); ! 168: while (fgets(wordbuf, NWORD, fp) != NULL) { ! 169: hash = 0; ! 170: cp = wordbuf; ! 171: while ((c = *cp++) != '\0') { ! 172: if (c == '\n') { ! 173: cp[-1] = '\0'; ! 174: break; ! 175: } ! 176: if (isupper(c)) ! 177: cp[-1] = c = tolower(c); ! 178: hash += c; ! 179: } ! 180: if ((wp = (WORDS *)malloc(sizeof(WORDS) + cp-wordbuf)) == NULL) ! 181: preperr("Out of memory for words from `%s'", fn); ! 182: strcpy(wp->w_name, wordbuf); ! 183: wp->w_next = words[hash %= NHASH]; ! 184: words[hash] = wp; ! 185: nword++; ! 186: } ! 187: fclose(fp); ! 188: return (nword); ! 189: } ! 190: ! 191: /* ! 192: * Lookup a word in either the only ! 193: * or exception list. ! 194: */ ! 195: lookup(word) ! 196: char *word; ! 197: { ! 198: register WORDS *wp; ! 199: register char *cp; ! 200: register unsigned hash = 0; ! 201: char wordbuf[256]; /* Keep a temporary copy of a world */ ! 202: int i; ! 203: ! 204: cp = word; ! 205: while (*cp != '\0') ! 206: if (isupper(*cp)) ! 207: hash += tolower(*cp++); else ! 208: hash += *cp++; ! 209: for (wp = words[hash%NHASH]; wp != NULL; wp = wp->w_next) { ! 210: if (iflag && !fflag) { ! 211: if (strlen(word) > 256) ! 212: preperr("word too long %s", word); ! 213: for (cp = word, i = 0; *cp != '\0'; cp++) ! 214: if (isupper(*cp)) ! 215: wordbuf[i++] = tolower(*cp); ! 216: else ! 217: wordbuf[i++] = *cp; ! 218: wordbuf[i] = '\0'; ! 219: cp = wordbuf; ! 220: if (strcmp(wp->w_name, cp) == 0) ! 221: return (1); ! 222: } else { ! 223: if (strcmp(wp->w_name, word) == 0) ! 224: return (1); ! 225: } ! 226: } ! 227: return (0); ! 228: } ! 229: ! 230: /* VARARGS */ ! 231: preperr(x) ! 232: { ! 233: fprintf(stderr, "prep: %r\n", &x); ! 234: exit(1); ! 235: } ! 236: ! 237: usage() ! 238: { ! 239: fprintf(stderr, "Usage: prep [-dfp] [-i file] [-o file] [file ...]\n"); ! 240: exit(1); ! 241: }
This archive runs on limited infrastructure. Preserving old code on modern bandwidth. Automated agents are requested to crawl responsibly.