|
|
1.1 root 1: /*
2: * sh/lex.c
3: * Bourne shell.
4: * Lexical analysis.
5: */
6:
7: #include "sh.h"
8: #include <y.tab.h>
9:
10: /*
11: * Local externals.
12: */
13: int lastget = '\0'; /* Pushed back character */
14: int eolflag = 0; /* End of line */
15:
16: /*
17: * For processing here documents.
18: */
19: char *hereeof = NULL; /* Here document EOF mark */
20: int herefd; /* Here document fd */
21: char *heretmp; /* Here document tempfile name */
22: int hereqflag; /* Here document quoted */
23:
24: /*
25: * Keyword table.
26: */
27: typedef struct key {
28: int k_hash; /* Hash */
29: int k_lexv; /* Lexical value */
30: char *k_name; /* Keyword name */
31: } KEY;
32:
33: /*
34: * Keyword table.
35: */
36: KEY keytab[] ={
37: 0, _CASE, "case",
38: 0, _DO, "do",
39: 0, _DONE, "done",
40: 0, _ELIF, "elif",
41: 0, _ELSE, "else",
42: 0, _ESAC, "esac",
43: 0, _FI, "fi",
44: 0, _FOR, "for",
45: 0, _IF, "if",
46: 0, _IN, "in",
47: 0, _RET, "return",
48: 0, _THEN, "then",
49: 0, _UNTIL, "until",
50: 0, _WHILE, "while",
51: 0, _OBRAC, "{",
52: 0, _CBRAC, "}"
53: };
54: #define NKEYS (sizeof(keytab) / sizeof(keytab[0]))
55:
56: /*
57: * Get the next lexical token.
58: */
59: yylex()
60: {
61: register int c;
62: register KEY *kp;
63: int hash;
64:
65: if (keytab[0].k_hash == 0)
66: for (kp = &keytab[0]; kp < &keytab[NKEYS]; kp++)
67: kp->k_hash = ihash(kp->k_name);
68: again:
69: while ((c=getn())==' ' || c=='\t') ;
70: strp = strt;
71: if (c == '#' && readflag == 0) {
72: /*
73: * Ignore a '#'-delimited comment line.
74: * Lines which begin with a ':' token are lexed as usual;
75: * the built-in function s_colon() executes (i.e. ignores)
76: * lines starting with ':', while other ':' tokens get passed.
77: * The built-in "read" does not ignore comment lines.
78: */
79: do
80: c = getn();
81: while (c > 0 && c != '\n');
82: return c;
83: } else if (class(c, MDIGI)) {
84: *strp++ = c;
85: c = getn();
86: if (c=='>' || c=='<') {
87: *strp++ = c;
88: return lexiors(c);
89: }
90: ungetn(c);
91: return lexname();
92: }
93: if (!class(c, MNAME)) {
94: ungetn(c);
95: if ((c = lexname()) == 0)
96: goto again;
97: else if (c < 0)
98: return c;
99: hash = ihash(strt);
100: if (keyflag) {
101: for (kp = keytab; kp < &keytab[NKEYS]; kp++)
102: if (hash == kp->k_hash && strcmp(strt, kp->k_name) == 0)
103: return kp->k_lexv;
104: #if 0
105: /*
106: * As documented elsewhere, ':' is required to actually
107: * evaluate the rest of the line, and a builtin with
108: * name ':' then discards the result. This code belongs
109: * to the age before builtins.
110: */
111: if (strcmp(strt, ":") == 0) {
112: /* On-line comment, eat everything to newline. */
113: do
114: c = getn();
115: while (c > 0 && c != '\n');
116: if (c == '\n')
117: ungetn(c);
118: c = _NAME;
119: }
120: #endif
121: }
122: return c;
123: }
124: *strp++ = c;
125: *strp = '\0';
126: switch (c) {
127: case ';':
128: return isnext(c, _DSEMI);
129: case '>':
130: return lexiors(c);
131: case '<':
132: return lexiors(c);
133: case '&':
134: return isnext(c, _ANDF);
135: case '|':
136: #ifdef NAMEPIPE
137: if ( ! isnext(')', 0))
138: return _NCLOSE;
139: #endif
140: return isnext(c, _ORF);
141: #ifdef NAMEPIPE
142: case '(':
143: return isnext('|', _NOPEN);
144: #else
145: case '(':
146: return isnext(')', _PARENS);
147: #endif
148: default:
149: if (hereeof != NULL) {
150: /* Read here document. */
151: for (;;) {
152: strp = strt;
153: if ((c = collect('\n', NO_ERRORS)) < 0)
154: break;
155: *strp = '\0';
156: if (strcmp(strt, hereeof)==0)
157: break;
158: if (herefd < 0)
159: continue;
160: if (!hereqflag && strp > strt + 1 && strp[-2]=='\\')
161: *(strp-=2) = '\0';
162: if (!hereqflag && *strt=='\\' && strcmp(hereeof, strt+1)==0)
163: write(herefd, strt+1, strp-strt-1);
164: else
165: write(herefd, strt, strp-strt);
166: }
167: close(herefd);
168: remember_temp (heretmp);
169: hereeof = NULL;
170: return '\n';
171: }
172: return c;
173: }
174: }
175:
176: isnext(c, t1)
177: register int c;
178: {
179: register int c2;
180:
181: if ((c2=getn()) == c) {
182: *strp++ = c2;
183: *strp = '\0';
184: return t1;
185: }
186: ungetn(c2);
187: return strp[-1];
188: }
189:
190: /*
191: * Read stuff delimited by (possibly nested) '{' '}' pairs.
192: */
193:
194: void getcurlies () {
195: int c;
196: char * cp = strp;
197: int quote = 0;
198:
199: for (;;) {
200: if ((c = getn ()) < 0 || c == '\n')
201: emisschar ();
202:
203: if (cp >= strt + STRSIZE)
204: etoolong();
205:
206: switch (* cp ++ = c) {
207: case '}':
208: if (! quote)
209: return;
210: continue;
211:
212: case '"':
213: quote ^= 1;
214: continue;
215:
216: case '\'':
217: strp = cp;
218: if ((c = collect('\'', NO_BACKSLASH)) != '\'')
219: break;
220: cp = strp;
221: continue;
222:
223: case '\\':
224: if ((c = getn ()) < 0) {
225: syntax ();
226: break;
227: }
228: if (c == '\n') {
229: cp --;
230: continue;
231: }
232: * cp ++ = c;
233: continue;
234:
235: case '$':
236: c = getn ();
237: if (c == '{') {
238: * cp ++ = c;
239: strp = cp;
240: getcurlies ();
241: cp = strp;
242: } else
243: ungetn(c);
244: continue;
245:
246: case '`':
247: strp = cp;
248: if ((c = collect('`', BACKSLASH_END)) != '`')
249: break;
250: cp = strp;
251: continue;
252:
253: case '\n':
254: if (quote)
255: continue;
256: break;
257: }
258:
259: break;
260: }
261:
262: if (quote)
263: emisschar ('"');
264: strp = cp;
265: }
266:
267: /*
268: * Scan a single argument.
269: * Return 0 if it's an escaped newline, EOF if EOF is found,
270: * or _NAME or _ASGN if any part of an argument is found.
271: */
272: lexname()
273: {
274: int q, asgn;
275: register int c, m;
276: register char *cp;
277:
278: q = 0;
279: asgn = 0;
280: m = MNQUO;
281: cp = strp;
282: for (;;) {
283: c = getn();
284: if (asgn==0)
285: asgn = class(c, MBVAR) ? 1 : -1;
286: else if (asgn==1)
287: asgn = class(c, MRVAR) ? 1 : (c=='=' ? 2 : -1);
288: if (cp >= strt + STRSIZE)
289: etoolong();
290: else
291: *cp++ = c;
292: if (!class(c, m))
293: continue;
294: switch (c) {
295: case '"':
296: m = (q^=1) ? MDQUO : MNQUO;
297: continue;
298: case '\'':
299: strp = cp;
300: if ((c = collect('\'', NO_BACKSLASH)) != '\'')
301: break;
302: cp = strp;
303: continue;
304: case '\\':
305: if ((c=getn()) < 0) {
306: syntax();
307: break;
308: }
309: if (c == '\n') {
310: ungetn((c=getn())<0 ? '\n' : c);
311: if (--cp == strp)
312: return 0;
313: continue;
314: }
315: *cp++ = c;
316: continue;
317: case '$':
318: c = getn ();
319: if (c == '{') {
320: * cp ++ = c;
321: strp = cp;
322: getcurlies ();
323: cp = strp;
324: } else
325: ungetn(c);
326: continue;
327: case '`':
328: strp = cp;
329: if ((c = collect('`', BACKSLASH_END)) != '`')
330: break;
331: cp = strp;
332: continue;
333: case '\n':
334: if (q)
335: continue;
336: break;
337: }
338: break;
339: }
340: if (c < 0)
341: return c;
342: if (q) {
343: emisschar('"');
344: *cp = '\0';
345: } else {
346: *--cp = '\0';
347: }
348: ungetn(c);
349: strp = cp;
350: #ifdef VERBOSE
351: if (vflag)
352: prints("\t<%d> <%s> %s\n", getpid(), (asgn==2 ? "ASGN" : "NAME"), strt);
353: #endif
354: if (errflag)
355: return _NULL;
356: else if (asgn==2)
357: return _ASGN;
358: else
359: return _NAME;
360: }
361:
362: /*
363: * Lex an io redirection string, including the file name if any.
364: * Called with one '>' or '<' in buffer, optionally preceded by
365: * a digit.
366: */
367: lexiors(c1)
368: {
369: register int c;
370: register char *name;
371: char *iors;
372:
373: *strp++ = c = getn();
374: if (c=='&') {
375: *strp++ = c = getn();
376: *strp = '\0';
377: if (c < 0) return c;
378: if (c!='-' && !class(c, MDIGI))
379: eredir();
380: return _IORS;
381: }
382: if (c==c1)
383: c1 += 0200;
384: else {
385: *--strp = '\0';
386: ungetn(c);
387: }
388: /* Collect file name */
389: while ((c=getn())==' '||c=='\t')
390: *strp++ = c;
391: ungetn(c);
392: name = strp;
393: if (c=='\n') {
394: eredir();
395: return _IORS;
396: }
397: while ((c = lexname())==0);
398: if (c < 0) return c;
399: if (c1!='<'+0200)
400: return _IORS;
401: #if 1
402: /*
403: * Set up here document processing.
404: * Modified by steve 1/25/91 so that
405: * the actual processing happens at the '\n' ending the line,
406: * otherwise the common "foo <<SHAR_EOF >baz\n" does not work.
407: * This code is anything but obvious, it could doubtless be simpler.
408: */
409: strp = strt;
410: /* Simplify quoted here document iors from ?<<file to ?<file. */
411: if (hereqflag = (strpbrk(name, "\"\\'") != NULL))
412: *++strp = *strt;
413: heretmp = name;
414: name = duplstr(name, 0);
415: strcpy(heretmp, shtmp());
416: iors = duplstr(strp, 0);
417: heretmp += iors - strp;
418: eval(name, EWORD);
419: hereeof = duplstr(strcat(strt, "\n"), 0);
420: if ((herefd = creat(heretmp, 0666)) < 0)
421: ecantmake(heretmp);
422: strcpy(strt, iors);
423: #else
424: /* Collect here document */
425: if ((c=getn())!='\n') {
426: eredir();
427: ++strp;
428: c = collect('\n', NO_BACKSLASH);
429: }
430: if (c < 0) return c;
431: bpp = savebuf();
432: strp = strt;
433: /* Simplify quoted to ?<file from ?<<file */
434: if (quote = (strpbrk(name, "\"\\'") != NULL))
435: *++strp = *strt;
436: tmp = name;
437: name = duplstr(name, 0);
438: strcpy(tmp, shtmp());
439: iors = duplstr(strp, 0);
440: tmp += iors - strp;
441: eval(name, EWORD);
442: name = duplstr(strcat(strt, "\n"), 0);
443: if ((hfd = creat(tmp, 0666)) < 0)
444: ecantmake(tmp);
445: for (;;) {
446: strp = strt;
447: if ((c = collect('\n', NO_ERRORS)) < 0)
448: break;
449: *strp = '\0';
450: if (strcmp(strt, name)==0)
451: break;
452: if (hfd < 0)
453: continue;
454: if (! quote && strp > strt + 1 && strp[-2]=='\\')
455: *(strp-=2) = '\0';
456: if (! quote && *strt=='\\' && strcmp(name, strt+1)==0)
457: write(hfd, strt+1, strp-strt-1);
458: else
459: write(hfd, strt, strp-strt);
460: }
461: close(hfd);
462: cleanup(0, tmp);
463: ungetn('\n');
464: strcpy(strt, iors);
465: freebuf(bpp);
466: /* Check for interrupt, since EOF is legal for once */
467: if (c < 0 && ! recover(ILEX)) return c;
468: #endif
469: return _IORS;
470: }
471:
472: /*
473: * Collect characters until the end character is found. If 'f' is
474: * CONSUME_BACKSLASH, '\' escapes the next character and newline is
475: * not allowed. If 'f' is BACKSLASH_END, backslashes are retained but
476: * suppress recognition of the end-character. If 'f' is NO_BACKSLASH,
477: * all characters are passed through. If 'f' is NO_ERRORS, behave as
478: * NO_BACKSLASH but be quiet about errors.
479: */
480: collect(ec, f)
481: register int ec;
482: {
483: register int c;
484: register char *cp;
485:
486: cp = strp;
487: while ((c=getn()) != ec) {
488: backslashed:
489: if (c<0 || (c=='\n' && f == CONSUME_BACKSLASH)) {
490: if (f != NO_ERRORS)
491: emisschar(ec);
492: return c;
493: }
494: if (c=='\\' && f == CONSUME_BACKSLASH) {
495: if ((c=getn()) < 0) {
496: syntax();
497: return c;
498: }
499: if (c == '\n')
500: continue;
501: }
502: if (cp >= strt + STRSIZE)
503: etoolong();
504: else
505: *cp++ = c;
506: if (c == '\\' && f == BACKSLASH_END) {
507: if ((c = getn ()) != '\n')
508: goto backslashed;
509: cp --;
510: }
511: }
512: *cp++ = ec;
513: strp = cp;
514: return ec;
515: }
516:
517: /*
518: * Get a character.
519: */
520: getn()
521: {
522: register int c;
523: register int t;
524:
525: if (lastget != '\0') {
526: c = lastget;
527: lastget = '\0';
528: return c;
529: }
530: switch (t = sesp->s_type) {
531: case SSTR:
532: case SFILE:
533: yyline += eolflag;
534: eolflag = 0;
535: if (prpflag && sesp->s_flag) {
536: if (prpflag -= 1) {
537: prompt("\n");
538: prpflag -= 1;
539: }
540: prompt(comflag ? vps1 : vps2);
541: comflag = 0;
542: }
543: if ((c=getc(sesp->s_ifp))=='\n') {
544: if (sesp->s_flag) {
545: prpflag = 1;
546: yyline = 1;
547: } else
548: eolflag = 1;
549: }
550: if (vflag)
551: putc(c, stderr);
552: return c;
553: case SARGS:
554: case SARGV:
555: if (sesp->s_flag)
556: return EOF;
557: if ((c=*sesp->s_strp++) == '\0') {
558: if (t == SARGV
559: && (sesp->s_strp=*++sesp->s_argv) != NULL)
560: c = ' ';
561: else {
562: sesp->s_flag = 1;
563: c = '\n';
564: }
565: }
566: if (vflag)
567: putc(c, stderr);
568: return c;
569: }
570: }
571:
572: /*
573: * Unget a character.
574: */
575: ungetn(c)
576: {
577: lastget = c;
578: }
579:
580: /* end of sh/lex.c */
This archive runs on limited infrastructure. Preserving old code on modern bandwidth. Automated agents are requested to crawl responsibly.