Annotation of nono/util/iasm/scan.cpp, revision 1.1.1.1

1.1       root        1: //
                      2: // XM6i
                      3: // Copyright (C) 2013 [email protected]
                      4: //
                      5: // 簡易アセンブラ: 字句解析
                      6: //
                      7: 
                      8: #include <ctype.h>
                      9: #include <errno.h>
                     10: #include <stdint.h>
                     11: #include <stdio.h>
                     12: #include <stdlib.h>
                     13: #include <limits.h>
                     14: #include "iasm.h"
                     15: 
                     16: 
                     17: /* コンストラクタ */
                     18: Token::Token()
                     19: {
                     20:        clear();
                     21: }
                     22: 
                     23: /* クリア */
                     24: void
                     25: Token::clear()
                     26: {
                     27:        type = T_END;
                     28:        word.erase();
                     29:        number = 0;
                     30: }
                     31: 
                     32: /* トークン種別 t の表示用文字列を返す (static 関数) */
                     33: const char *
                     34: Token::type_name(token_type_t t)
                     35: {
                     36:        if (t < 0 || t >= T_MAX) {
                     37:                return "?";
                     38:        }
                     39:        return type_str[t];
                     40: }
                     41: 
                     42: /* 現在のトークンの種別文字列を返す */
                     43: const char *
                     44: Token::type_name() const
                     45: {
                     46:        return type_name(type);
                     47: }
                     48: 
                     49: /* トークン種別の表示用文字列 */
                     50: /* token_type_t の順序と合わせること orz */
                     51: const char *Token::type_str[] = {
                     52:        "END",
                     53:        "SPACE",
                     54:        "CHAR",
                     55:        "STRING",
                     56:        "NUMBER",
                     57:        "LITERAL",
                     58:        "SIZE",
                     59:        "SHARP",
                     60:        "COLON",
                     61:        "COMMA",
                     62:        "LPAREN",
                     63:        "RPAREN",
                     64:        "PLUS",
                     65:        "MINUS",
                     66:        "MULTI",
                     67:        "DIVIDE",
                     68:        "LT",
                     69:        "GT",
                     70: };
                     71: 
                     72: 
                     73: /* 1文字の記号テーブル */
                     74: Scanner::mark_t Scanner::mark[] = {
                     75:        { '#', T_SHARP, },
                     76:        { ':', T_COLON, },
                     77:        { ',', T_COMMA, },
                     78:        { '(', T_LPAREN, },
                     79:        { ')', T_RPAREN, },
                     80:        { '+', T_PLUS, },
                     81:        { '-', T_MINUS, },
                     82:        { '*', T_MULTI, },
                     83:        { '/', T_DIVIDE, },
                     84:        { '<', T_LT, },
                     85:        { '>', T_GT, },
                     86: };
                     87: 
                     88: /* コンストラクタ */
                     89: Scanner::Scanner(String& src)
                     90: {
                     91:        buf = src;
                     92:        ptr = 0;
                     93: }
                     94: 
                     95: /* デストラクタ */
                     96: Scanner::~Scanner()
                     97: {
                     98: }
                     99: 
                    100: /* リテラルの1文字目を判定 (ドットを含み数字は含まない) */
                    101: bool
                    102: Scanner::isLabel1(int c)
                    103: {
                    104:        return (isalpha(c) || c == '_' || c == '.');
                    105: }
                    106: 
                    107: /* リテラルの2文字目以降を判定 (数字は含みドットを含まない) */
                    108: bool
                    109: Scanner::isLabel2(int c)
                    110: {
                    111:        return (isalpha(c) || c == '_' || isdigit(c));
                    112: }
                    113: 
                    114: /*
                    115:  * サイズサフィックス文字を判定。
                    116:  * 正しいものならバイト数を返す。
                    117:  * bra.s (と fmove.s ?) のため .S なら 0 を返すのでそっちで判断しろ下さい。
                    118:  * 正しくなければ -1 を返す。
                    119:  */
                    120: int
                    121: Scanner::isSize(int c) const
                    122: {
                    123:        c = tolower(c);
                    124: 
                    125:        if (c == 's') {
                    126:                return 0;
                    127:        }
                    128:        if (c == 'b') {
                    129:                return 1;
                    130:        }
                    131:        if (c == 'w') {
                    132:                return 2;
                    133:        }
                    134:        if (c == 'l') {
                    135:                return 4;
                    136:        }
                    137: 
                    138:        return -1;
                    139: }
                    140: 
                    141: /* トークンを1つ切り出す */
                    142: bool
                    143: Scanner::get_token(Token& token)
                    144: {
                    145:        int c;
                    146: 
                    147:        /* 戻り値を初期化 */
                    148:        token.clear();
                    149:        errmsg.erase();
                    150: 
                    151:        c = getch();
                    152: 
                    153:        /* 終端 */
                    154:        if (c == 0) {
                    155:                token.type = T_END;
                    156:                return true;
                    157:        }
                    158: 
                    159:        /* 連続する空白は1つの空白 */
                    160:        if (isspace(c)) {
                    161:                for(; isspace((c = getch())); ptr++) {
                    162:                        token.word += c;
                    163:                }
                    164:                token.type = T_SPACE;
                    165:                return true;
                    166:        }
                    167: 
                    168:        /* シングルクォートなら1文字 */
                    169:        if (c == '\'') {
                    170:                token.word = c;
                    171:                ptr++;
                    172:                /* 文字コンテキスト */
                    173:                bool escape = false;
                    174:                bool terminate = false;
                    175:                int charnum = 0;
                    176:                for (; (c = getch()) != 0; ptr++) {
                    177:                        if (escape) {
                    178:                                token.word += c;
                    179:                                /* 制御文字のコードをここで解釈 */
                    180:                                if (c == 't') {
                    181:                                        token.number = '\t';
                    182:                                } else if (c == 'r') {
                    183:                                        token.number = '\r';
                    184:                                } else if (c == 'n') {
                    185:                                        token.number = '\n';
                    186:                                } else if (c == '0') {
                    187:                                        token.number = '\0';
                    188:                                } else {
                    189:                                        token.number = c;
                    190:                                }
                    191:                                charnum++;
                    192:                                escape = false;
                    193:                                continue;
                    194:                        }
                    195:                        if (c == '\\') {
                    196:                                token.word += c;
                    197:                                escape = true;
                    198:                                continue;
                    199:                        }
                    200:                        if (c == '\'') {
                    201:                                token.word += c;
                    202:                                terminate = true;
                    203:                                ptr++;
                    204:                                break;
                    205:                        }
                    206:                        token.word += c;
                    207:                        token.number = c;
                    208:                        charnum++;
                    209:                }
                    210:                if (!terminate) {
                    211:                        errmsg = "single char not terminated";
                    212:                        return false;
                    213:                }
                    214:                if (charnum != 1) {
                    215:                        errmsg = "invalid char";
                    216:                        return false;
                    217:                }
                    218:                token.type = T_CHAR;
                    219:                return true;
                    220:        }
                    221: 
                    222:        /* ダブルクォートなら文字列 */
                    223:        if (c == '\x22') {
                    224:                token.word = c;
                    225:                ptr++;
                    226:                /* 文字列コンテキスト */
                    227:                bool escape = false;
                    228:                bool terminate = false;
                    229:                for (; (c = getch()) != 0; ptr++) {
                    230:                        if (escape) {
                    231:                                token.word += c;
                    232:                                escape = false;
                    233:                                continue;
                    234:                        }
                    235:                        if (c == '\\') {
                    236:                                token.word += c;
                    237:                                escape = true;
                    238:                                continue;
                    239:                        }
                    240:                        if (c == '\x22') {
                    241:                                token.word += c;
                    242:                                terminate = true;
                    243:                                ptr++;
                    244:                                break;
                    245:                        }
                    246:                        token.word += c;
                    247:                }
                    248:                if (!terminate) {
                    249:                        errmsg = "string not terminated";
                    250:                        return false;
                    251:                }
                    252:                token.type = T_STRING;
                    253:                return true;
                    254:        }
                    255: 
                    256:        /* 16進数 */
                    257:        if (c == '$') {
                    258:                token.word += c;
                    259:                ptr++;
                    260:                for (; isxdigit((c = getch())); ptr++) {
                    261:                        token.word += c;
                    262:                }
                    263:                /* 8桁('$'含めて9文字)越えたらアウトっていう判定だけでいいはず */
                    264:                if (token.word.size() > 9) {
                    265:                        errmsg = "too large number";
                    266:                        return false;
                    267:                }
                    268:                /* 変換できるはず */
                    269:                token.number = (uint32_t)strtol(token.word.c_str() + 1, NULL, 16);
                    270:                token.type = T_NUMBER;
                    271:                return true;
                    272:        }
                    273: 
                    274:        /* 10進数 */
                    275:        if (isdigit(c)) {
                    276:                for (; isdigit((c = getch())); ptr++) {
                    277:                        token.word += c;
                    278:                }
                    279:                errno = 0;
                    280:                long lval;
                    281:                lval = strtol(token.word.c_str(), NULL, 10);
                    282:                if (errno == ERANGE && (lval == LONG_MAX || lval == LONG_MIN)) {
                    283:                        errmsg = "too large number";
                    284:                        return false;
                    285:                }
                    286:                token.number = (uint32_t)lval;
                    287:                token.type = T_NUMBER;
                    288:                return true;
                    289:        }
                    290: 
                    291:        /* プレインなところでのセミコロンはコメント開始記号 */
                    292:        if (c == ';') {
                    293:                /*
                    294:                 * コメント以降を字句解析すると文法エラーになる可能性があるので
                    295:                 * ここで解析をやめる。
                    296:                 */
                    297:                token.word = c;
                    298:                token.type = T_END;
                    299:                return true;
                    300:        }
                    301: 
                    302:        /* 1文字の記号 */
                    303:        for (int i = 0; i < countof(mark); i++) {
                    304:                if (mark[i].chr == c) {
                    305:                        token.word = c;
                    306:                        ptr++;
                    307:                        token.type = mark[i].type;
                    308:                        return true;
                    309:                }
                    310:        }
                    311: 
                    312:        /* サイズサフィックス */
                    313:        if (c == '.') {
                    314:                int c1 = getch(1);
                    315:                int c2 = getch(2);
                    316:                int size;
                    317:                if ((size = isSize(c1)) != -1 && !isalpha(c2)) {
                    318:                        token.word += c;
                    319:                        token.word += c1;
                    320:                        ptr += 2;
                    321:                        token.type = T_SIZE;
                    322:                        token.number = size;
                    323:                        return true;
                    324:                }
                    325: 
                    326:                /* そうでなければリテラル(擬似命令)かも */
                    327:                /* PASSTHROUGH */
                    328:        }
                    329: 
                    330:        /* ラベル (1文字目と2文字目以降は判定条件が違う) */
                    331:        if (isLabel1(c)) {
                    332:                token.word = c;
                    333:                ptr++;
                    334:                for (; isLabel2((c = getch())); ptr++) {
                    335:                        token.word += c;
                    336:                }
                    337:                token.type = T_LITERAL;
                    338:                return true;
                    339:        }
                    340: 
                    341:        errmsg = "syntax error";
                    342:        return false;
                    343: }
                    344: 
                    345: /* 現在位置の文字を返す。文字列の終端以降なら 0 を返す */
                    346: int
                    347: Scanner::getch(int offset) const
                    348: {
                    349:        int c;
                    350: 
                    351:        if (ptr + offset < buf.size()) {
                    352:                c = (int)buf[ptr + offset];
                    353:        } else {
                    354:                c = 0;
                    355:        }
                    356:        return c;
                    357: }

unix.superglobalmegacorp.com

This archive runs on limited infrastructure. Preserving old code on modern bandwidth. Automated agents are requested to crawl responsibly.