#include "lexer.h" #include "source.h" #include "error.h" #include "util.h" #include #include #include #include #include struct lexer { source_t* source; pos_t cur_pos; }; lexer_t* lexer_alloc(source_t *source) { lexer_t *it = malloc(sizeof (struct lexer)); if (it == NULL) { die("out of memory: lexer allocation failed"); } it->source = source; it->cur_pos.line = 0; it->cur_pos.col = 0; it->cur_pos.idx = 0; return it; } pos_t lexer_cur_pos(lexer_t* lexer) { return lexer->cur_pos; } source_t* lexer_source(lexer_t *lexer) { return lexer->source; } token_t lexer_lex(lexer_t* lexer) { const char* current; start: current = source_contents_from(lexer->source, lexer->cur_pos); if (*current == '\0') { token_t tok; tok.len = 0; tok.start = current; tok.type = TOKEN_EOF; return tok; } if (starts_with_any(current, " ","\t","\n","\r")) { lexer->cur_pos = source_next_pos(lexer->source, lexer->cur_pos); goto start; } if (starts_with_any(current,"//")) { while (*current != '\n') { current++; lexer->cur_pos = source_next_pos(lexer->source, lexer->cur_pos); } goto start; } token_t tok; tok.len = 1; tok.start = current; tok.pos = lexer->cur_pos; lexer->cur_pos = source_next_pos(lexer->source, lexer->cur_pos); switch (*current) { case '(': tok.type = TOKEN_LPAREN; return tok; case ')': tok.type = TOKEN_RPAREN; return tok; case '{': tok.type = TOKEN_LBRACE; return tok; case '}': tok.type = TOKEN_RBRACE; return tok; case ':': if (*(current+1) == '=') { tok.type = TOKEN_ASSIGN; tok.len = 2; lexer->cur_pos = source_next_pos(lexer->source, lexer->cur_pos); return tok; } else { tok.type = TOKEN_COLON; return tok; } case '>': if (*(current+1) == '=') { tok.type = TOKEN_GREATEREQ; tok.len = 2; lexer->cur_pos = source_next_pos(lexer->source, lexer->cur_pos); return tok; } else { tok.type = TOKEN_GREATER; return tok; } case '<': if (*(current+1) == '=') { tok.type = TOKEN_LESSEQ; tok.len = 2; lexer->cur_pos = source_next_pos(lexer->source, lexer->cur_pos); return tok; } else { tok.type = TOKEN_LESS; return tok; } case '/': if (*(current+1) == '=') { tok.type = TOKEN_NOTEQ; tok.len = 2; lexer->cur_pos = source_next_pos(lexer->source, lexer->cur_pos); return tok; } else { tok.type = TOKEN_DIV; return tok; } if (*(current+1) == '=') { tok.type = TOKEN_NOTEQ; tok.len = 2; lexer->cur_pos = source_next_pos(lexer->source, lexer->cur_pos); return tok; } else { tok.type = TOKEN_DIV; return tok; } case '|': tok.type = TOKEN_PIPE; return tok; case ',': tok.type = TOKEN_COMMA; return tok; case ';': tok.type = TOKEN_SEMI; return tok; case '-': if (isdigit(*(current+1))) { current++; tok.len++; lexer->cur_pos = source_next_pos(lexer->source, lexer->cur_pos); break; } else { tok.type = TOKEN_MINUS; return tok; } case '+': tok.type = TOKEN_PLUS; return tok; case '=': tok.type = TOKEN_EQ; return tok; case '^': tok.type = TOKEN_EXP; return tok; case '*': tok.type = TOKEN_MULT; return tok; case '.': if (isalnum(*(current+1)) || *(current+1) == '_') { current++; while (isalnum(*current) || *current == '_') { tok.len++; current++; lexer->cur_pos = source_next_pos(lexer->source, lexer->cur_pos); } tok.type = TOKEN_CONSTRUCTOR; return tok; } else break; case '\"': { current++; bool escaped = false; while (*current != '\"' || escaped) { if (*current == '\0') { tok.type = TOKEN_UNRECOGNISED; error_report( error(ERROR_UNTERMINATED_STRING_LITERAL,lexer->source,tok.pos,1) ); return tok; } escaped = false; if (*current == '\\') escaped = true; current++; tok.len++; lexer->cur_pos = source_next_pos(lexer->source, lexer->cur_pos); } current++; tok.len++; lexer->cur_pos = source_next_pos(lexer->source, lexer->cur_pos); tok.type = TOKEN_STRING_LIT; return tok; } } if (isdigit(*current)) { current++; bool dot_seen = false; bool e_seen = false; tok.type = TOKEN_INT_LIT; while (isdigit(*current) || !dot_seen && *current == '.' || !e_seen && *current == 'E' || !e_seen && *current == 'e') { if (*current == '.') { dot_seen = true; tok.type = TOKEN_FLOAT_LIT; } if (*current == 'e' || *current == 'E') { e_seen = true; tok.type = TOKEN_FLOAT_LIT; } current++; tok.len++; lexer->cur_pos = source_next_pos(lexer->source, lexer->cur_pos); } return tok; } else if (isalpha(*current) || *current == '_') { current++; while (isalnum(*current) || *current == '_') { current++; tok.len++; lexer->cur_pos = source_next_pos(lexer->source, lexer->cur_pos); } if (tok.len == 2 && starts_with_any(tok.start,"if")) tok.type = TOKEN_IF; else if (tok.len == 3 && starts_with_any(tok.start,"and")) tok.type = TOKEN_AND; else if (tok.len == 3 && starts_with_any(tok.start,"mod")) tok.type = TOKEN_MOD; else if (tok.len == 3 && starts_with_any(tok.start,"div")) tok.type = TOKEN_IDIV; else if (tok.len == 2 && starts_with_any(tok.start,"or")) tok.type = TOKEN_OR; else tok.type = TOKEN_IDENT; return tok; } tok.type = TOKEN_UNRECOGNISED; error_report( error(ERROR_UNRECOGNISED_TOKEN,lexer->source,tok.pos,1) ); return tok; } token_t lexer_peek(lexer_t* lexer) { pos_t pos = lexer->cur_pos; token_t tok = lexer_lex(lexer); printf("%s\n",lexer_token_type_to_string(tok)); lexer->cur_pos = pos; return tok; } void lexer_dealloc(lexer_t* lexer) { free(lexer); } #define enum_case_str(n) case n: return #n const char* lexer_token_type_to_string(token_t to_print) { switch (to_print.type) { enum_case_str(TOKEN_LPAREN); enum_case_str(TOKEN_RPAREN); enum_case_str(TOKEN_LBRACE); enum_case_str(TOKEN_RBRACE); enum_case_str(TOKEN_STRING_LIT); enum_case_str(TOKEN_INT_LIT); enum_case_str(TOKEN_FLOAT_LIT); enum_case_str(TOKEN_IDENT); enum_case_str(TOKEN_ASSIGN); enum_case_str(TOKEN_SEMI); enum_case_str(TOKEN_PIPE); enum_case_str(TOKEN_IF); enum_case_str(TOKEN_AND); enum_case_str(TOKEN_OR); enum_case_str(TOKEN_MULT); enum_case_str(TOKEN_PLUS); enum_case_str(TOKEN_EXP); enum_case_str(TOKEN_MINUS); enum_case_str(TOKEN_DIV); enum_case_str(TOKEN_IDIV); enum_case_str(TOKEN_EQ); enum_case_str(TOKEN_NOTEQ); enum_case_str(TOKEN_MOD); enum_case_str(TOKEN_LESSEQ); enum_case_str(TOKEN_GREATEREQ); enum_case_str(TOKEN_LESS); enum_case_str(TOKEN_GREATER); enum_case_str(TOKEN_COMMA); enum_case_str(TOKEN_COLON); enum_case_str(TOKEN_CONSTRUCTOR); enum_case_str(TOKEN_UNRECOGNISED); enum_case_str(TOKEN_EOF); } }