diff options
Diffstat (limited to 'src/mcpp-source.c')
| -rw-r--r-- | src/mcpp-source.c | 332 |
1 files changed, 332 insertions, 0 deletions
diff --git a/src/mcpp-source.c b/src/mcpp-source.c new file mode 100644 index 0000000..c40e370 --- /dev/null +++ b/src/mcpp-source.c @@ -0,0 +1,332 @@ +#include <defs.h> + +static char * +read_stream_bytes( FILE *fp, size_t *file_length ) +{ + char *data = NULL; + size_t length = 0; + size_t capacity = 0; + + if( fp == NULL || file_length == NULL ) + { + errno = EINVAL; + return( NULL ); + } + + *file_length = 0; + + for( ;; ) + { + size_t n; + + if( capacity - length < 4096 ) + { + size_t new_capacity = capacity ? capacity * 2 : 8192; + char *q = (char *)realloc( data, new_capacity + 1 ); + if( q == NULL ) + { + free( data ); + return( NULL ); + } + data = q; + capacity = new_capacity; + } + + n = fread( data + length, 1, capacity - length, fp ); + length += n; + + if( n == 0 ) + { + if( ferror(fp) ) + { + free( data ); + return( NULL ); + } + break; + } + } + + if( data == NULL ) + { + data = (char *)calloc( 1, 1 ); + if( data == NULL ) + return( NULL ); + } + else + data[length] = 0; + + *file_length = length; + return( data ); +} + +static int +normalize_newlines( mcpu_text *text ) +{ + size_t r; + size_t w = 0; + + if( text == NULL || text->data == NULL ) + return( 0 ); + + for( r = 0; r < text->length; ++r ) + { + if( text->data[r] == '\r' ) + { + if( r + 1 < text->length && text->data[r + 1] == '\n' ) + ++r; + text->data[w++] = '\n'; + } + else + text->data[w++] = text->data[r]; + } + + text->length = w; + text->data[w] = 0; + + return( 0 ); +} + +void +mcpu_source_init( mcpu_source *source ) +{ + if( source == NULL ) + return; + + source->filename = NULL; + mcpu_text_init( &source->text ); + mcpu_text_init( &source->logical_line ); + source->splice_offsets = NULL; + source->splice_count = 0; + source->splice_capacity = 0; + source->offset = 0; + source->line = 1; +} + +void +mcpu_source_free( mcpu_source *source ) +{ + if( source == NULL ) + return; + + free( source->filename ); + source->filename = NULL; + mcpu_text_free( &source->text ); + mcpu_text_free( &source->logical_line ); + free( source->splice_offsets ); + source->splice_offsets = NULL; + source->splice_count = 0; + source->splice_capacity = 0; + source->offset = 0; + source->line = 1; +} + +static int +source_from_bytes( mcpu_source *source, char *bytes, size_t byte_length, + const char *name ) +{ + const __mpu_char8_t *input = (const __mpu_char8_t *)bytes; + const __mpu_char8_t *p; + + if( memchr( bytes, 0, byte_length ) != NULL ) + { + fprintf( stderr, "%s: NUL character is not allowed in source text\n", name ); + return( -1 ); + } + + if( byte_length >= 3 && + (unsigned char)bytes[0] == 0xef && + (unsigned char)bytes[1] == 0xbb && + (unsigned char)bytes[2] == 0xbf ) + input += 3; + + if( !mpu_utf8valid(input) ) + { + fprintf( stderr, "%s: invalid UTF-8\n", name ); + return( -1 ); + } + + p = input; + while( *p != 0 ) + { + __mpu_char32_t value; + const __mpu_char8_t *next; + __mpu_char16_t c; + + next = mpu_utf8get( p, &value ); + if( next == NULL ) + { + fprintf( stderr, "%s: invalid UTF-8\n", name ); + return( -1 ); + } + + if( value <= 0xffff ) + c = (__mpu_char16_t)value; + else + c = MCPU_CPP_NON_UCS2_SENTINEL; + + if( mcpu_text_append_char(&source->text, c) != 0 ) + return( -1 ); + + p = next; + } + + if( normalize_newlines(&source->text) != 0 ) + return( -1 ); + + source->filename = strdup( name ); + if( source->filename == NULL ) + return( -1 ); + + return( 0 ); +} + +int +mcpu_source_valid_ucs2( const __mpu_char16_t *text, size_t length ) +{ + size_t i; + + if( text == NULL && length != 0 ) + return( 0 ); + + for( i = 0; i < length; ++i ) + if( text[i] == MCPU_CPP_NON_UCS2_SENTINEL ) + return( 0 ); + + return( 1 ); +} + +int +mcpu_source_open_stream( mcpu_source *source, FILE *stream, const char *name ) +{ + char *bytes; + size_t byte_length; + int rc; + + if( source == NULL || stream == NULL || name == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + mcpu_source_free( source ); + mcpu_source_init( source ); + + bytes = read_stream_bytes( stream, &byte_length ); + if( bytes == NULL ) + return( -1 ); + + rc = source_from_bytes( source, bytes, byte_length, name ); + free( bytes ); + return( rc ); +} + +int +mcpu_source_open( mcpu_source *source, const char *filename ) +{ + FILE *fp; + int rc; + + if( source == NULL || filename == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + fp = fopen( filename, "rb" ); + if( fp == NULL ) + return( -1 ); + + rc = mcpu_source_open_stream( source, fp, filename ); + if( fclose(fp) != 0 && rc == 0 ) + rc = -1; + + return( rc ); +} + +static int +append_splice_offset( mcpu_source *source, size_t offset ) +{ + size_t *offsets; + size_t capacity; + + if( source->splice_count < source->splice_capacity ) + { + source->splice_offsets[source->splice_count++] = offset; + return( 0 ); + } + + capacity = source->splice_capacity ? source->splice_capacity * 2 : 8; + offsets = (size_t *)realloc( source->splice_offsets, + capacity * sizeof(*offsets) ); + if( offsets == NULL ) + return( -1 ); + + source->splice_offsets = offsets; + source->splice_capacity = capacity; + source->splice_offsets[source->splice_count++] = offset; + return( 0 ); +} + + +int +mcpu_source_next_line( mcpu_source *source, + const __mpu_char16_t **line, + size_t *length, + unsigned *line_number, + unsigned *next_line_number, + int *spliced ) +{ + unsigned first_line; + + if( source == NULL || line == NULL || length == NULL || + line_number == NULL || next_line_number == NULL || spliced == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + if( source->offset >= source->text.length ) + return( 0 ); + + mcpu_text_free( &source->logical_line ); + mcpu_text_init( &source->logical_line ); + source->splice_count = 0; + first_line = source->line; + *spliced = 0; + + while( source->offset < source->text.length ) + { + __mpu_char16_t c = source->text.data[source->offset++]; + + /* + * Backslash-newline deletion is preprocessing phase 2 and therefore + * happens before comments, directives and macro recognition. + */ + if( c == '\\' && source->offset < source->text.length && + source->text.data[source->offset] == '\n' ) + { + if( append_splice_offset(source, source->logical_line.length) != 0 ) + return( -1 ); + ++source->offset; + ++source->line; + *spliced = 1; + continue; + } + + if( mcpu_text_append_char( &source->logical_line, c ) != 0 ) + return( -1 ); + + if( c == '\n' ) + { + ++source->line; + break; + } + } + + *line = source->logical_line.data; + *length = source->logical_line.length; + *line_number = first_line; + *next_line_number = source->line; + + return( 1 ); +} |
