summaryrefslogtreecommitdiff
path: root/src/mcpp-lexer.c
diff options
context:
space:
mode:
Diffstat (limited to 'src/mcpp-lexer.c')
-rw-r--r--src/mcpp-lexer.c371
1 files changed, 371 insertions, 0 deletions
diff --git a/src/mcpp-lexer.c b/src/mcpp-lexer.c
new file mode 100644
index 0000000..067e1a2
--- /dev/null
+++ b/src/mcpp-lexer.c
@@ -0,0 +1,371 @@
+#include <defs.h>
+
+void
+mcpu_pp_lexer_init( mcpu_pp_lexer_state *state )
+{
+ if( state )
+ {
+ state->in_block_comment = 0;
+ state->options = NULL;
+ state->filename = NULL;
+ state->line_number = 0;
+ state->splice_offsets = NULL;
+ state->splice_count = 0;
+ }
+}
+
+void
+mcpu_pp_lexer_set_diagnostics( mcpu_pp_lexer_state *state,
+ const mcpp_options *options,
+ const char *filename,
+ unsigned line_number,
+ const size_t *splice_offsets,
+ size_t splice_count )
+{
+ if( state == NULL )
+ return;
+
+ state->options = options;
+ state->filename = filename;
+ state->line_number = line_number;
+ state->splice_offsets = splice_offsets;
+ state->splice_count = splice_count;
+}
+
+static unsigned
+lexer_warning_line( const mcpu_pp_lexer_state *state, size_t offset )
+{
+ size_t i;
+ unsigned line;
+
+ line = state->line_number;
+ for( i = 0; i < state->splice_count; ++i )
+ if( state->splice_offsets[i] <= offset )
+ ++line;
+
+ return( line );
+}
+
+static int
+lexer_has_splice_after( const mcpu_pp_lexer_state *state, size_t offset )
+{
+ size_t i;
+
+ for( i = 0; i < state->splice_count; ++i )
+ if( state->splice_offsets[i] >= offset )
+ return( 1 );
+
+ return( 0 );
+}
+
+static int
+lexer_comment_warning( const mcpu_pp_lexer_state *state, size_t offset,
+ const char *message )
+{
+ if( state->options == NULL || !state->options->warn_comments )
+ return( 0 );
+
+ return( mcpp_diagnostic_warning(state->options,
+ state->filename,
+ lexer_warning_line(state, offset),
+ "%s", message) );
+}
+
+int
+mcpu_pp_is_identifier_start( __mpu_char16_t c )
+{
+ return( c == '_' || mpu_ucs2_is_xid_start(c) );
+}
+
+int
+mcpu_pp_is_identifier_char( __mpu_char16_t c )
+{
+ return( c == '_' || c == '$' || mpu_ucs2_is_xid_continue(c) );
+}
+
+static int
+is_quote16( __mpu_char16_t c, enum mcpu_language language )
+{
+ if( c == '"' || c == '`' )
+ return( 1 );
+
+ if( c == '\'' && language != MCPU_LANG_DIFF )
+ return( 1 );
+
+ return( 0 );
+}
+
+static int
+append_space_once( mcpu_text *text )
+{
+ if( text->length != 0 )
+ {
+ __mpu_char16_t last = text->data[text->length - 1];
+ if( last == ' ' || last == '\t' || last == '\f' || last == '\v' ||
+ last == '\r' || last == '\n' )
+ return( 0 );
+ }
+
+ return( mcpu_text_append_char(text, ' ') );
+}
+
+static int
+is_blank16( __mpu_char16_t c )
+{
+ return( c == ' ' || c == '\t' || c == '\f' ||
+ c == '\v' || c == '\r' );
+}
+
+static void
+remove_comment_tail_space( mcpu_text *text )
+{
+ size_t end;
+ int newline;
+
+ if( text == NULL || text->length == 0 )
+ return;
+
+ newline = text->data[text->length - 1] == '\n';
+ end = newline ? text->length - 1 : text->length;
+
+ while( end != 0 && is_blank16(text->data[end - 1]) )
+ --end;
+
+ if( newline )
+ text->data[end++] = '\n';
+
+ text->length = end;
+ text->data[text->length] = 0;
+}
+
+int
+mcpu_pp_prepare_line( mcpu_pp_lexer_state *state,
+ const __mpu_char16_t *line,
+ size_t length,
+ enum mcpu_language language,
+ mcpu_text *prepared )
+{
+ size_t p = 0;
+ __mpu_char16_t quote = 0;
+ int escaped = 0;
+ int first_token = 1;
+ int saw_hash = 0;
+ int reading_directive = 0;
+ int directive_done = 0;
+ int include_directive = 0;
+ int include_argument_pending = 0;
+ int in_include_angle = 0;
+ __mpu_char16_t directive[32];
+ size_t directive_length = 0;
+ int comment_tail;
+
+ if( state == NULL || line == NULL || prepared == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ mcpu_text_free( prepared );
+ mcpu_text_init( prepared );
+ comment_tail = state->in_block_comment;
+
+ while( p < length )
+ {
+ __mpu_char16_t c = line[p];
+
+ if( state->in_block_comment )
+ {
+ if( p + 1 < length && c == '/' && line[p + 1] == '*' )
+ {
+ if( lexer_comment_warning(state, p, "\"/*\" within comment") != 0 )
+ return( -1 );
+ }
+
+ if( p + 1 < length && c == '*' && line[p + 1] == '/' )
+ {
+ state->in_block_comment = 0;
+ p += 2;
+ }
+ else
+ {
+ if( c == '\n' && mcpu_text_append_char(prepared, c) != 0 )
+ return( -1 );
+ ++p;
+ }
+ continue;
+ }
+
+ if( in_include_angle )
+ {
+ if( mcpu_text_append_char(prepared, c) != 0 )
+ return( -1 );
+ if( c == '>' || c == '\n' )
+ in_include_angle = 0;
+ ++p;
+ continue;
+ }
+
+ if( quote )
+ {
+ if( mcpu_text_append_char(prepared, c) != 0 )
+ return( -1 );
+
+ if( escaped )
+ escaped = 0;
+ else if( c == '\\' )
+ escaped = 1;
+ else if( c == quote || c == '\n' )
+ quote = 0;
+
+ ++p;
+ continue;
+ }
+
+ if( p + 1 < length && c == '/' && line[p + 1] == '*' )
+ {
+ if( append_space_once(prepared) != 0 )
+ return( -1 );
+ comment_tail = 1;
+ state->in_block_comment = 1;
+ p += 2;
+ continue;
+ }
+
+ if( p + 1 < length && c == '/' && line[p + 1] == '/' )
+ {
+ if( lexer_has_splice_after(state, p + 2) &&
+ lexer_comment_warning(state, p, "multi-line comment") != 0 )
+ return( -1 );
+
+ if( append_space_once(prepared) != 0 )
+ return( -1 );
+ comment_tail = 1;
+ while( p < length && line[p] != '\n' )
+ ++p;
+ continue;
+ }
+
+ if( comment_tail && !is_blank16(c) && c != '\n' )
+ comment_tail = 0;
+
+ if( first_token )
+ {
+ if( c == ' ' || c == '\t' || c == '\f' || c == '\v' || c == '\r' )
+ {
+ if( mcpu_text_append_char(prepared, c) != 0 )
+ return( -1 );
+ ++p;
+ continue;
+ }
+
+ first_token = 0;
+ if( c == '#' )
+ {
+ saw_hash = 1;
+ reading_directive = 1;
+ }
+ }
+
+ if( reading_directive && !directive_done && saw_hash && c != '#' )
+ {
+ if( directive_length == 0 &&
+ (c == ' ' || c == '\t' || c == '\f' || c == '\v' || c == '\r') )
+ {
+ if( mcpu_text_append_char(prepared, c) != 0 )
+ return( -1 );
+ ++p;
+ continue;
+ }
+
+ if( mcpu_pp_is_identifier_char(c) )
+ {
+ if( directive_length + 1 < sizeof(directive) / sizeof(directive[0]) )
+ directive[directive_length++] = c;
+ if( mcpu_text_append_char(prepared, c) != 0 )
+ return( -1 );
+ ++p;
+ continue;
+ }
+
+ directive_done = 1;
+ reading_directive = 0;
+ if( (directive_length == 7 &&
+ directive[0] == 'i' && directive[1] == 'n' &&
+ directive[2] == 'c' && directive[3] == 'l' &&
+ directive[4] == 'u' && directive[5] == 'd' &&
+ directive[6] == 'e') ||
+ (directive_length == 12 &&
+ directive[0] == 'i' && directive[1] == 'n' &&
+ directive[2] == 'c' && directive[3] == 'l' &&
+ directive[4] == 'u' && directive[5] == 'd' &&
+ directive[6] == 'e' && directive[7] == '_' &&
+ directive[8] == 'n' && directive[9] == 'e' &&
+ directive[10] == 'x' && directive[11] == 't') )
+ {
+ include_directive = 1;
+ include_argument_pending = 1;
+ }
+ }
+
+ if( include_directive && include_argument_pending )
+ {
+ if( c == ' ' || c == '\t' || c == '\f' || c == '\v' || c == '\r' )
+ {
+ if( mcpu_text_append_char(prepared, c) != 0 )
+ return( -1 );
+ ++p;
+ continue;
+ }
+
+ include_argument_pending = 0;
+ if( c == '<' )
+ in_include_angle = 1;
+ }
+
+ if( is_quote16(c, language) )
+ quote = c;
+
+ if( mcpu_text_append_char(prepared, c) != 0 )
+ return( -1 );
+ ++p;
+ }
+
+ if( comment_tail )
+ remove_comment_tail_space( prepared );
+
+ return( 0 );
+}
+
+int
+mcpu_pp_find_directive( const __mpu_char16_t *line,
+ size_t length,
+ size_t *hash_offset )
+{
+ size_t p = 0;
+
+ if( line == NULL || hash_offset == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ while( p < length )
+ {
+ if( line[p] == ' ' || line[p] == '\t' || line[p] == '\f' ||
+ line[p] == '\v' || line[p] == '\r' )
+ {
+ ++p;
+ continue;
+ }
+
+ if( line[p] == '#' )
+ {
+ *hash_offset = p;
+ return( 1 );
+ }
+
+ break;
+ }
+
+ return( 0 );
+}