|
|
1.1 ! root 1: /* ! 2: * sh/lex.c ! 3: * Bourne shell. ! 4: * Lexical analysis. ! 5: */ ! 6: ! 7: #include "sh.h" ! 8: #include "y.tab.h" ! 9: ! 10: /* ! 11: * Local externals. ! 12: */ ! 13: int lastget = '\0'; /* Pushed back character */ ! 14: int eolflag = 0; /* End of line */ ! 15: ! 16: /* ! 17: * For processing here documents. ! 18: */ ! 19: char *hereeof = NULL; /* Here document EOF mark */ ! 20: int herefd; /* Here document fd */ ! 21: char *heretmp; /* Here document tempfile name */ ! 22: int hereqflag; /* Here document quoted */ ! 23: ! 24: /* ! 25: * Keyword table. ! 26: */ ! 27: typedef struct key { ! 28: int k_hash; /* Hash */ ! 29: int k_lexv; /* Lexical value */ ! 30: char *k_name; /* Keyword name */ ! 31: } KEY; ! 32: ! 33: /* ! 34: * Keyword table. ! 35: */ ! 36: KEY keytab[] ={ ! 37: 0, _CASE, "case", ! 38: 0, _DO, "do", ! 39: 0, _DONE, "done", ! 40: 0, _ELIF, "elif", ! 41: 0, _ELSE, "else", ! 42: 0, _ESAC, "esac", ! 43: 0, _FI, "fi", ! 44: 0, _FOR, "for", ! 45: 0, _IF, "if", ! 46: 0, _IN, "in", ! 47: 0, _RET, "return", ! 48: 0, _THEN, "then", ! 49: 0, _UNTIL, "until", ! 50: 0, _WHILE, "while", ! 51: 0, _OBRAC, "{", ! 52: 0, _CBRAC, "}" ! 53: }; ! 54: #define NKEYS (sizeof(keytab) / sizeof(keytab[0])) ! 55: ! 56: /* ! 57: * Get the next lexical token. ! 58: */ ! 59: yylex() ! 60: { ! 61: register int c; ! 62: register KEY *kp; ! 63: int hash; ! 64: ! 65: if (keytab[0].k_hash == 0) ! 66: for (kp = & keytab [0] ; kp < & keytab [NKEYS] ; kp ++) ! 67: kp->k_hash = ihash (kp->k_name); ! 68: again: ! 69: while ((c = getn ()) == ' ' || c == '\t') ! 70: /* DO NOTHING */ ; ! 71: strp = strt; ! 72: if (c == '#' && readflag == 0) { ! 73: /* ! 74: * Ignore a '#'-delimited comment line. ! 75: * Lines which begin with a ':' token are lexed as usual; ! 76: * the built-in function s_colon() executes (i.e. ignores) ! 77: * lines starting with ':', while other ':' tokens get passed. ! 78: * The built-in "read" does not ignore comment lines. ! 79: */ ! 80: do ! 81: c = getn (); ! 82: while (c > 0 && c != '\n'); ! 83: return c; ! 84: } else if (class (c, MDIGI)) { ! 85: * strp ++ = c; ! 86: c = getn (); ! 87: if (c == '>' || c == '<') { ! 88: * strp ++ = c; ! 89: return lexiors (c); ! 90: } ! 91: ungetn (c); ! 92: return lexname (); ! 93: } ! 94: if (! class (c, MNAME)) { ! 95: if (c >= 0) ! 96: ungetn (c); ! 97: if ((c = lexname ()) == 0) ! 98: goto again; ! 99: else if (c < 0) ! 100: return c; ! 101: ! 102: hash = ihash (strt); ! 103: ! 104: if (keyflag) { ! 105: for (kp = keytab; kp < & keytab [NKEYS] ; kp ++) ! 106: if (hash == kp->k_hash && ! 107: strcmp (strt, kp->k_name) == 0) { ! 108: return kp->k_lexv; ! 109: } ! 110: } ! 111: ! 112: return c; ! 113: } ! 114: * strp ++ = c; ! 115: * strp = '\0'; ! 116: ! 117: switch (c) { ! 118: case ';': ! 119: return isnext (c, _DSEMI); ! 120: ! 121: case '>': ! 122: return lexiors (c); ! 123: ! 124: case '<': ! 125: return lexiors (c); ! 126: ! 127: case '&': ! 128: return isnext (c, _ANDF); ! 129: ! 130: case '|': ! 131: #ifdef NAMEPIPE ! 132: if ( ! isnext (')', 0)) ! 133: return _NCLOSE; ! 134: #endif ! 135: return isnext (c, _ORF); ! 136: ! 137: #ifdef NAMEPIPE ! 138: case '(': ! 139: return isnext ('|', _NOPEN); ! 140: #else ! 141: case '(': ! 142: return isnext (')', _PARENS); ! 143: #endif ! 144: ! 145: default: ! 146: if (hereeof != NULL) { ! 147: /* Read here document. */ ! 148: for (;;) { ! 149: strp = strt; ! 150: if ((c = collect ('\n', NO_ERRORS)) < 0) ! 151: break; ! 152: * strp = '\0'; ! 153: if (strcmp (strt, hereeof) == 0) ! 154: break; ! 155: if (herefd < 0) ! 156: continue; ! 157: if (! hereqflag && strp > strt + 1 && ! 158: strp [-2] == '\\') ! 159: * (strp -= 2) = '\0'; ! 160: if (! hereqflag && * strt == '\\' && ! 161: strcmp (hereeof, strt + 1) == 0) { ! 162: write (herefd, strt + 1, ! 163: strp - strt - 1); ! 164: } else ! 165: write (herefd, strt, strp - strt); ! 166: } ! 167: close (herefd); ! 168: remember_temp (heretmp); ! 169: hereeof = NULL; ! 170: return '\n'; ! 171: } ! 172: ! 173: return c; ! 174: } ! 175: } ! 176: ! 177: isnext(c, t1) ! 178: register int c; ! 179: { ! 180: register int c2; ! 181: ! 182: if ((c2 = getn ()) == c) { ! 183: * strp ++ = c2; ! 184: * strp = '\0'; ! 185: return t1; ! 186: } ! 187: ungetn (c2); ! 188: return strp [-1]; ! 189: } ! 190: ! 191: /* ! 192: * Read stuff delimited by (possibly nested) '{' '}' pairs. ! 193: */ ! 194: ! 195: void getcurlies () { ! 196: int c; ! 197: int quote = 0; ! 198: ! 199: for (;;) { ! 200: if ((c = getn ()) < 0 || c == '\n') ! 201: emisschar (); ! 202: ! 203: if (strp >= strt + STRSIZE) ! 204: etoolong ("in getcurlies ()"); ! 205: ! 206: switch (* strp ++ = c) { ! 207: case '}': ! 208: if (! quote) ! 209: return; ! 210: continue; ! 211: ! 212: case '"': ! 213: quote ^= 1; ! 214: continue; ! 215: ! 216: case '\'': ! 217: if ((c = collect ('\'', NO_BACKSLASH)) != '\'') ! 218: break; ! 219: continue; ! 220: ! 221: case '\\': ! 222: if ((c = getn ()) < 0) { ! 223: syntax (); ! 224: break; ! 225: } ! 226: if (c == '\n') { ! 227: strp --; ! 228: continue; ! 229: } ! 230: * strp ++ = c; ! 231: continue; ! 232: ! 233: case '$': ! 234: c = getn (); ! 235: if (c == '{') { ! 236: * strp ++ = c; ! 237: getcurlies (); ! 238: } else ! 239: ungetn(c); ! 240: continue; ! 241: ! 242: case '`': ! 243: if ((c = collect ('`', BACKSLASH_END)) != '`') ! 244: break; ! 245: continue; ! 246: } ! 247: } ! 248: } ! 249: ! 250: ! 251: /* ! 252: * Scan a single argument. ! 253: * Return 0 if it's an escaped newline, EOF if EOF is found, ! 254: * or _NAME or _ASGN if any part of an argument is found. ! 255: */ ! 256: lexname() ! 257: { ! 258: int q, asgn; ! 259: register int c, m; ! 260: register char *cp; ! 261: ! 262: q = 0; ! 263: asgn = 0; ! 264: m = MNQUO; ! 265: cp = strp; ! 266: for (;;) { ! 267: c = getn (); ! 268: if (asgn == 0) ! 269: asgn = class (c, MBVAR) ? 1 : -1; ! 270: else if (asgn == 1) ! 271: asgn = class (c, MRVAR) ? 1 : (c == '=' ? 2 : -1); ! 272: if (cp >= strt + STRSIZE) ! 273: etoolong ("in lexname ()"); ! 274: else ! 275: * cp++ = c; ! 276: if (! class (c, m)) ! 277: continue; ! 278: switch (c) { ! 279: case '"': ! 280: m = (q ^= 1) ? MDQUO : MNQUO; ! 281: continue; ! 282: ! 283: case '\'': ! 284: strp = cp; ! 285: c = collect ('\'', NO_BACKSLASH); ! 286: cp = strp; ! 287: if (c != '\'') ! 288: break; ! 289: continue; ! 290: ! 291: case '\\': ! 292: if ((c = getn ()) < 0) { ! 293: syntax (); ! 294: break; ! 295: } ! 296: if (c == '\n') { ! 297: ungetn ((c = getn ()) < 0 ? '\n' : c); ! 298: if (-- cp == strp) ! 299: return 0; ! 300: continue; ! 301: } ! 302: * cp ++ = c; ! 303: continue; ! 304: ! 305: case '$': ! 306: c = getn (); ! 307: if (c == '{') { ! 308: * cp ++ = c; ! 309: strp = cp; ! 310: getcurlies (); ! 311: cp = strp; ! 312: } else ! 313: ungetn (c); ! 314: continue; ! 315: ! 316: case '`': ! 317: strp = cp; ! 318: c = collect ('`', BACKSLASH_END); ! 319: cp = strp; ! 320: if (c != '`') ! 321: break; ! 322: continue; ! 323: ! 324: case '\n': ! 325: if (q) ! 326: continue; ! 327: break; ! 328: } ! 329: break; ! 330: } ! 331: ! 332: #if 0 ! 333: if (c < 0) ! 334: return c; ! 335: #endif ! 336: ! 337: if (q) { ! 338: emisschar ('"'); ! 339: * cp = '\0'; ! 340: } else ! 341: * -- cp = '\0'; ! 342: ! 343: if (c >= 0) ! 344: ungetn (c); ! 345: strp = cp; ! 346: ! 347: #ifdef VERBOSE ! 348: if (vflag) ! 349: prints("\t<%d> <%s> %s\n", getpid(), (asgn==2 ? "ASGN" : "NAME"), strt); ! 350: #endif ! 351: if (errflag) ! 352: return _NULL; ! 353: else if (asgn == 2) ! 354: return _ASGN; ! 355: else ! 356: return _NAME; ! 357: } ! 358: ! 359: /* ! 360: * Lex an io redirection string, including the file name if any. ! 361: * Called with one '>' or '<' in buffer, optionally preceded by ! 362: * a digit. ! 363: */ ! 364: lexiors(c1) ! 365: { ! 366: register int c; ! 367: register char *name; ! 368: char *iors; ! 369: ! 370: * strp ++ = c = getn (); ! 371: if (c == '&') { ! 372: * strp ++ = c = getn (); ! 373: * strp = '\0'; ! 374: if (c < 0) ! 375: return c; ! 376: if (c != '-' && ! class (c, MDIGI)) ! 377: eredir (); ! 378: return _IORS; ! 379: } ! 380: if (c == c1) ! 381: c1 += 0200; ! 382: else { ! 383: * -- strp = '\0'; ! 384: ungetn (c); ! 385: } ! 386: ! 387: /* Collect file name */ ! 388: while ((c = getn ()) == ' ' || c == '\t') ! 389: * strp ++ = c; ! 390: ungetn (c); ! 391: name = strp; ! 392: if (c == '\n') { ! 393: eredir (); ! 394: return _IORS; ! 395: } ! 396: ! 397: while ((c = lexname ()) == 0) ! 398: /* DO NOTHING */ ; ! 399: ! 400: if (c < 0) ! 401: return c; ! 402: ! 403: if (c1 != '<' + 0200) ! 404: return _IORS; ! 405: ! 406: /* ! 407: * Set up here document processing. ! 408: * Modified by steve 1/25/91 so that ! 409: * the actual processing happens at the '\n' ending the line, ! 410: * otherwise the common "foo <<SHAR_EOF >baz\n" does not work. ! 411: * This code is anything but obvious, it could doubtless be simpler. ! 412: */ ! 413: strp = strt; ! 414: /* Simplify quoted here document iors from ?<<file to ?<file. */ ! 415: if ((hereqflag = strpbrk (name, "\"\\'") != NULL) != 0) ! 416: * ++ strp = * strt; ! 417: heretmp = name; ! 418: name = duplstr (name, 0); ! 419: strcpy (heretmp, shtmp ()); ! 420: iors = duplstr (strp, 0); ! 421: heretmp += iors - strp; ! 422: eval (name, EWORD); ! 423: hereeof = duplstr (strcat (strt, "\n"), 0); ! 424: if ((herefd = creat (heretmp, 0666)) < 0) ! 425: ecantmake (heretmp); ! 426: strcpy (strt, iors); ! 427: return _IORS; ! 428: } ! 429: ! 430: /* ! 431: * Collect characters until the end character is found. If 'f' is ! 432: * CONSUME_BACKSLASH, '\' escapes the next character and newline is ! 433: * not allowed. If 'f' is BACKSLASH_END, backslashes are retained but ! 434: * suppress recognition of the end-character. If 'f' is NO_BACKSLASH, ! 435: * all characters are passed through. If 'f' is NO_ERRORS, behave as ! 436: * NO_BACKSLASH but be quiet about errors. ! 437: */ ! 438: collect(ec, f) ! 439: register int ec; ! 440: { ! 441: register int c; ! 442: register char *cp; ! 443: ! 444: cp = strp; ! 445: while ((c = getn ()) != ec) { ! 446: backslashed: ! 447: /* ! 448: * NIGEL: Originally, this routine complained about missing ! 449: * characters at EOF, but System V just implicitly closes off ! 450: * any strings in this situation. In order to stroke the ! 451: * lexer, we mutate EOF's into EOL's. ! 452: */ ! 453: if (c < 0) { ! 454: if (f != NO_ERRORS) { ! 455: c = ec; ! 456: break; ! 457: } ! 458: return c; ! 459: } ! 460: if (c == '\n' && f == CONSUME_BACKSLASH) { ! 461: if (f != NO_ERRORS) ! 462: emisschar (ec); ! 463: return c; ! 464: } ! 465: if (c == '\\' && f == CONSUME_BACKSLASH) { ! 466: if ((c = getn ()) < 0) { ! 467: syntax (); ! 468: return c; ! 469: } ! 470: if (c == '\n') ! 471: continue; ! 472: } ! 473: if (cp >= strt + STRSIZE) ! 474: etoolong ("in collect ()"); ! 475: else ! 476: * cp ++ = c; ! 477: if (c == '\\' && f == BACKSLASH_END) { ! 478: if ((c = getn ()) != '\n') ! 479: goto backslashed; ! 480: cp --; ! 481: } ! 482: } ! 483: * cp ++ = ec; ! 484: strp = cp; ! 485: return ec; ! 486: } ! 487: ! 488: ! 489: /* ! 490: * Get a character. ! 491: */ ! 492: ! 493: getn() ! 494: { ! 495: register int c; ! 496: register int t; ! 497: ! 498: if (lastget != '\0') { ! 499: c = lastget; ! 500: lastget = '\0'; ! 501: return c; ! 502: } ! 503: switch (t = sesp->s_type) { ! 504: case SSTR: ! 505: case SFILE: ! 506: yyline += eolflag; ! 507: eolflag = 0; ! 508: if (prpflag && sesp->s_flag) { ! 509: if (prpflag -= 1) { ! 510: prompt("\n"); ! 511: prpflag -= 1; ! 512: } ! 513: prompt (comflag ? vps1 : vps2); ! 514: comflag = 0; ! 515: } ! 516: if ((c = getc (sesp->s_ifp)) == '\n') { ! 517: if (sesp->s_flag) { ! 518: prpflag = 1; ! 519: yyline = 1; ! 520: } else ! 521: eolflag = 1; ! 522: } ! 523: if (vflag) ! 524: putc(c, stderr); ! 525: return c; ! 526: case SARGS: ! 527: case SARGV: ! 528: if (sesp->s_flag) ! 529: return EOF; ! 530: if ((c = * sesp->s_strp ++) == '\0') { ! 531: if (t == SARGV && ! 532: (sesp->s_strp = * ++ sesp->s_argv) != NULL) ! 533: c = ' '; ! 534: else { ! 535: sesp->s_flag = 1; ! 536: c = EOF; ! 537: } ! 538: } ! 539: if (vflag) ! 540: putc (c, stderr); ! 541: return c; ! 542: } ! 543: } ! 544: ! 545: ! 546: /* ! 547: * Unget a character. ! 548: */ ! 549: ! 550: ungetn(c) ! 551: { ! 552: lastget = c; ! 553: }
This archive runs on limited infrastructure. Preserving old code on modern bandwidth. Automated agents are requested to crawl responsibly.