#include static char * read_stream_bytes( FILE *fp, size_t *file_length ) { char *data = NULL; size_t length = 0; size_t capacity = 0; if( fp == NULL || file_length == NULL ) { errno = EINVAL; return( NULL ); } *file_length = 0; for( ;; ) { size_t n; if( capacity - length < 4096 ) { size_t new_capacity = capacity ? capacity * 2 : 8192; char *q = (char *)realloc( data, new_capacity + 1 ); if( q == NULL ) { free( data ); return( NULL ); } data = q; capacity = new_capacity; } n = fread( data + length, 1, capacity - length, fp ); length += n; if( n == 0 ) { if( ferror(fp) ) { free( data ); return( NULL ); } break; } } if( data == NULL ) { data = (char *)calloc( 1, 1 ); if( data == NULL ) return( NULL ); } else data[length] = 0; *file_length = length; return( data ); } static int normalize_newlines( mcpu_text *text ) { size_t r; size_t w = 0; if( text == NULL || text->data == NULL ) return( 0 ); for( r = 0; r < text->length; ++r ) { if( text->data[r] == '\r' ) { if( r + 1 < text->length && text->data[r + 1] == '\n' ) ++r; text->data[w++] = '\n'; } else text->data[w++] = text->data[r]; } text->length = w; text->data[w] = 0; return( 0 ); } void mcpu_source_init( mcpu_source *source ) { if( source == NULL ) return; source->filename = NULL; mcpu_text_init( &source->text ); mcpu_text_init( &source->logical_line ); source->splice_offsets = NULL; source->splice_count = 0; source->splice_capacity = 0; source->offset = 0; source->line = 1; } void mcpu_source_free( mcpu_source *source ) { if( source == NULL ) return; free( source->filename ); source->filename = NULL; mcpu_text_free( &source->text ); mcpu_text_free( &source->logical_line ); free( source->splice_offsets ); source->splice_offsets = NULL; source->splice_count = 0; source->splice_capacity = 0; source->offset = 0; source->line = 1; } static int source_from_bytes( mcpu_source *source, char *bytes, size_t byte_length, const char *name ) { const __mpu_char8_t *input = (const __mpu_char8_t *)bytes; const __mpu_char8_t *p; if( memchr( bytes, 0, byte_length ) != NULL ) { fprintf( stderr, "%s: NUL character is not allowed in source text\n", name ); return( -1 ); } if( byte_length >= 3 && (unsigned char)bytes[0] == 0xef && (unsigned char)bytes[1] == 0xbb && (unsigned char)bytes[2] == 0xbf ) input += 3; if( !mpu_utf8valid(input) ) { fprintf( stderr, "%s: invalid UTF-8\n", name ); return( -1 ); } p = input; while( *p != 0 ) { __mpu_char32_t value; const __mpu_char8_t *next; __mpu_char16_t c; next = mpu_utf8get( p, &value ); if( next == NULL ) { fprintf( stderr, "%s: invalid UTF-8\n", name ); return( -1 ); } if( value <= 0xffff ) c = (__mpu_char16_t)value; else c = MCPU_CPP_NON_UCS2_SENTINEL; if( mcpu_text_append_char(&source->text, c) != 0 ) return( -1 ); p = next; } if( normalize_newlines(&source->text) != 0 ) return( -1 ); source->filename = strdup( name ); if( source->filename == NULL ) return( -1 ); return( 0 ); } int mcpu_source_valid_ucs2( const __mpu_char16_t *text, size_t length ) { size_t i; if( text == NULL && length != 0 ) return( 0 ); for( i = 0; i < length; ++i ) if( text[i] == MCPU_CPP_NON_UCS2_SENTINEL ) return( 0 ); return( 1 ); } int mcpu_source_open_stream( mcpu_source *source, FILE *stream, const char *name ) { char *bytes; size_t byte_length; int rc; if( source == NULL || stream == NULL || name == NULL ) { errno = EINVAL; return( -1 ); } mcpu_source_free( source ); mcpu_source_init( source ); bytes = read_stream_bytes( stream, &byte_length ); if( bytes == NULL ) return( -1 ); rc = source_from_bytes( source, bytes, byte_length, name ); free( bytes ); return( rc ); } int mcpu_source_open( mcpu_source *source, const char *filename ) { FILE *fp; int rc; if( source == NULL || filename == NULL ) { errno = EINVAL; return( -1 ); } fp = fopen( filename, "rb" ); if( fp == NULL ) return( -1 ); rc = mcpu_source_open_stream( source, fp, filename ); if( fclose(fp) != 0 && rc == 0 ) rc = -1; return( rc ); } static int append_splice_offset( mcpu_source *source, size_t offset ) { size_t *offsets; size_t capacity; if( source->splice_count < source->splice_capacity ) { source->splice_offsets[source->splice_count++] = offset; return( 0 ); } capacity = source->splice_capacity ? source->splice_capacity * 2 : 8; offsets = (size_t *)realloc( source->splice_offsets, capacity * sizeof(*offsets) ); if( offsets == NULL ) return( -1 ); source->splice_offsets = offsets; source->splice_capacity = capacity; source->splice_offsets[source->splice_count++] = offset; return( 0 ); } int mcpu_source_next_line( mcpu_source *source, const __mpu_char16_t **line, size_t *length, unsigned *line_number, unsigned *next_line_number, int *spliced ) { unsigned first_line; if( source == NULL || line == NULL || length == NULL || line_number == NULL || next_line_number == NULL || spliced == NULL ) { errno = EINVAL; return( -1 ); } if( source->offset >= source->text.length ) return( 0 ); mcpu_text_free( &source->logical_line ); mcpu_text_init( &source->logical_line ); source->splice_count = 0; first_line = source->line; *spliced = 0; while( source->offset < source->text.length ) { __mpu_char16_t c = source->text.data[source->offset++]; /* * Backslash-newline deletion is preprocessing phase 2 and therefore * happens before comments, directives and macro recognition. */ if( c == '\\' && source->offset < source->text.length && source->text.data[source->offset] == '\n' ) { if( append_splice_offset(source, source->logical_line.length) != 0 ) return( -1 ); ++source->offset; ++source->line; *spliced = 1; continue; } if( mcpu_text_append_char( &source->logical_line, c ) != 0 ) return( -1 ); if( c == '\n' ) { ++source->line; break; } } *line = source->logical_line.data; *length = source->logical_line.length; *line_number = first_line; *next_line_number = source->line; return( 1 ); }