diff options
Diffstat (limited to 'src/mcpp-lexer.c')
| -rw-r--r-- | src/mcpp-lexer.c | 371 |
1 files changed, 371 insertions, 0 deletions
diff --git a/src/mcpp-lexer.c b/src/mcpp-lexer.c new file mode 100644 index 0000000..067e1a2 --- /dev/null +++ b/src/mcpp-lexer.c @@ -0,0 +1,371 @@ +#include <defs.h> + +void +mcpu_pp_lexer_init( mcpu_pp_lexer_state *state ) +{ + if( state ) + { + state->in_block_comment = 0; + state->options = NULL; + state->filename = NULL; + state->line_number = 0; + state->splice_offsets = NULL; + state->splice_count = 0; + } +} + +void +mcpu_pp_lexer_set_diagnostics( mcpu_pp_lexer_state *state, + const mcpp_options *options, + const char *filename, + unsigned line_number, + const size_t *splice_offsets, + size_t splice_count ) +{ + if( state == NULL ) + return; + + state->options = options; + state->filename = filename; + state->line_number = line_number; + state->splice_offsets = splice_offsets; + state->splice_count = splice_count; +} + +static unsigned +lexer_warning_line( const mcpu_pp_lexer_state *state, size_t offset ) +{ + size_t i; + unsigned line; + + line = state->line_number; + for( i = 0; i < state->splice_count; ++i ) + if( state->splice_offsets[i] <= offset ) + ++line; + + return( line ); +} + +static int +lexer_has_splice_after( const mcpu_pp_lexer_state *state, size_t offset ) +{ + size_t i; + + for( i = 0; i < state->splice_count; ++i ) + if( state->splice_offsets[i] >= offset ) + return( 1 ); + + return( 0 ); +} + +static int +lexer_comment_warning( const mcpu_pp_lexer_state *state, size_t offset, + const char *message ) +{ + if( state->options == NULL || !state->options->warn_comments ) + return( 0 ); + + return( mcpp_diagnostic_warning(state->options, + state->filename, + lexer_warning_line(state, offset), + "%s", message) ); +} + +int +mcpu_pp_is_identifier_start( __mpu_char16_t c ) +{ + return( c == '_' || mpu_ucs2_is_xid_start(c) ); +} + +int +mcpu_pp_is_identifier_char( __mpu_char16_t c ) +{ + return( c == '_' || c == '$' || mpu_ucs2_is_xid_continue(c) ); +} + +static int +is_quote16( __mpu_char16_t c, enum mcpu_language language ) +{ + if( c == '"' || c == '`' ) + return( 1 ); + + if( c == '\'' && language != MCPU_LANG_DIFF ) + return( 1 ); + + return( 0 ); +} + +static int +append_space_once( mcpu_text *text ) +{ + if( text->length != 0 ) + { + __mpu_char16_t last = text->data[text->length - 1]; + if( last == ' ' || last == '\t' || last == '\f' || last == '\v' || + last == '\r' || last == '\n' ) + return( 0 ); + } + + return( mcpu_text_append_char(text, ' ') ); +} + +static int +is_blank16( __mpu_char16_t c ) +{ + return( c == ' ' || c == '\t' || c == '\f' || + c == '\v' || c == '\r' ); +} + +static void +remove_comment_tail_space( mcpu_text *text ) +{ + size_t end; + int newline; + + if( text == NULL || text->length == 0 ) + return; + + newline = text->data[text->length - 1] == '\n'; + end = newline ? text->length - 1 : text->length; + + while( end != 0 && is_blank16(text->data[end - 1]) ) + --end; + + if( newline ) + text->data[end++] = '\n'; + + text->length = end; + text->data[text->length] = 0; +} + +int +mcpu_pp_prepare_line( mcpu_pp_lexer_state *state, + const __mpu_char16_t *line, + size_t length, + enum mcpu_language language, + mcpu_text *prepared ) +{ + size_t p = 0; + __mpu_char16_t quote = 0; + int escaped = 0; + int first_token = 1; + int saw_hash = 0; + int reading_directive = 0; + int directive_done = 0; + int include_directive = 0; + int include_argument_pending = 0; + int in_include_angle = 0; + __mpu_char16_t directive[32]; + size_t directive_length = 0; + int comment_tail; + + if( state == NULL || line == NULL || prepared == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + mcpu_text_free( prepared ); + mcpu_text_init( prepared ); + comment_tail = state->in_block_comment; + + while( p < length ) + { + __mpu_char16_t c = line[p]; + + if( state->in_block_comment ) + { + if( p + 1 < length && c == '/' && line[p + 1] == '*' ) + { + if( lexer_comment_warning(state, p, "\"/*\" within comment") != 0 ) + return( -1 ); + } + + if( p + 1 < length && c == '*' && line[p + 1] == '/' ) + { + state->in_block_comment = 0; + p += 2; + } + else + { + if( c == '\n' && mcpu_text_append_char(prepared, c) != 0 ) + return( -1 ); + ++p; + } + continue; + } + + if( in_include_angle ) + { + if( mcpu_text_append_char(prepared, c) != 0 ) + return( -1 ); + if( c == '>' || c == '\n' ) + in_include_angle = 0; + ++p; + continue; + } + + if( quote ) + { + if( mcpu_text_append_char(prepared, c) != 0 ) + return( -1 ); + + if( escaped ) + escaped = 0; + else if( c == '\\' ) + escaped = 1; + else if( c == quote || c == '\n' ) + quote = 0; + + ++p; + continue; + } + + if( p + 1 < length && c == '/' && line[p + 1] == '*' ) + { + if( append_space_once(prepared) != 0 ) + return( -1 ); + comment_tail = 1; + state->in_block_comment = 1; + p += 2; + continue; + } + + if( p + 1 < length && c == '/' && line[p + 1] == '/' ) + { + if( lexer_has_splice_after(state, p + 2) && + lexer_comment_warning(state, p, "multi-line comment") != 0 ) + return( -1 ); + + if( append_space_once(prepared) != 0 ) + return( -1 ); + comment_tail = 1; + while( p < length && line[p] != '\n' ) + ++p; + continue; + } + + if( comment_tail && !is_blank16(c) && c != '\n' ) + comment_tail = 0; + + if( first_token ) + { + if( c == ' ' || c == '\t' || c == '\f' || c == '\v' || c == '\r' ) + { + if( mcpu_text_append_char(prepared, c) != 0 ) + return( -1 ); + ++p; + continue; + } + + first_token = 0; + if( c == '#' ) + { + saw_hash = 1; + reading_directive = 1; + } + } + + if( reading_directive && !directive_done && saw_hash && c != '#' ) + { + if( directive_length == 0 && + (c == ' ' || c == '\t' || c == '\f' || c == '\v' || c == '\r') ) + { + if( mcpu_text_append_char(prepared, c) != 0 ) + return( -1 ); + ++p; + continue; + } + + if( mcpu_pp_is_identifier_char(c) ) + { + if( directive_length + 1 < sizeof(directive) / sizeof(directive[0]) ) + directive[directive_length++] = c; + if( mcpu_text_append_char(prepared, c) != 0 ) + return( -1 ); + ++p; + continue; + } + + directive_done = 1; + reading_directive = 0; + if( (directive_length == 7 && + directive[0] == 'i' && directive[1] == 'n' && + directive[2] == 'c' && directive[3] == 'l' && + directive[4] == 'u' && directive[5] == 'd' && + directive[6] == 'e') || + (directive_length == 12 && + directive[0] == 'i' && directive[1] == 'n' && + directive[2] == 'c' && directive[3] == 'l' && + directive[4] == 'u' && directive[5] == 'd' && + directive[6] == 'e' && directive[7] == '_' && + directive[8] == 'n' && directive[9] == 'e' && + directive[10] == 'x' && directive[11] == 't') ) + { + include_directive = 1; + include_argument_pending = 1; + } + } + + if( include_directive && include_argument_pending ) + { + if( c == ' ' || c == '\t' || c == '\f' || c == '\v' || c == '\r' ) + { + if( mcpu_text_append_char(prepared, c) != 0 ) + return( -1 ); + ++p; + continue; + } + + include_argument_pending = 0; + if( c == '<' ) + in_include_angle = 1; + } + + if( is_quote16(c, language) ) + quote = c; + + if( mcpu_text_append_char(prepared, c) != 0 ) + return( -1 ); + ++p; + } + + if( comment_tail ) + remove_comment_tail_space( prepared ); + + return( 0 ); +} + +int +mcpu_pp_find_directive( const __mpu_char16_t *line, + size_t length, + size_t *hash_offset ) +{ + size_t p = 0; + + if( line == NULL || hash_offset == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + while( p < length ) + { + if( line[p] == ' ' || line[p] == '\t' || line[p] == '\f' || + line[p] == '\v' || line[p] == '\r' ) + { + ++p; + continue; + } + + if( line[p] == '#' ) + { + *hash_offset = p; + return( 1 ); + } + + break; + } + + return( 0 ); +} |
