diff options
Diffstat (limited to 'src/mcpp-expr.zubr')
| -rw-r--r-- | src/mcpp-expr.zubr | 723 |
1 files changed, 723 insertions, 0 deletions
diff --git a/src/mcpp-expr.zubr b/src/mcpp-expr.zubr new file mode 100644 index 0000000..1ec5823 --- /dev/null +++ b/src/mcpp-expr.zubr @@ -0,0 +1,723 @@ +/*************************************************************** + MCPP_EXPR.C + + This file containt the grammar & procedure for + parse expressions for MCPU-CPP . + + PART OF : MCPU-CPP - MCPU language preproccessor . + + NOTE : NONE . + + Copyright (C) 1998 - 2026 by Andrey V.Kosteltsev. + All Rights Reserved. + ***************************************************************/ + +%{ + +#include <defs.h> + +static int mcpp_zubr_lex( void ); +static void mcpp_zubr_error( char *s ); + +static mcpp_integer expression_value; +static mcpp_semantic_context semantic_context; + +/************************************************************ + Nonzero means do not evaluate this expression. + This is a count, since unevaluated expressions can nest. + ************************************************************/ +static int skip_evaluation; + +/************************************************************ + During parsing of an MCPU-CPP expression, LEXPTR points to + the next UCS-2 character and LEXEND points one character + past the input expression. + ************************************************************/ +static const __mpu_char16_t *lexptr; +static const __mpu_char16_t *lexend; + +%} + +%union +{ + mcpp_integer integer; + struct name + { + const __mpu_char16_t *address; + size_t length; + } name; +} + +%type <integer> exp exp1 start +%token <integer> INT CHAR +%token <name> NAME +%token <integer> ERROR + +%right '?' ':' +%left ',' +%left OR +%left AND +%left '|' +%left '^' +%left '&' +%left EQUAL NOTEQUAL +%left '<' '>' LEQ GEQ +%left LSH RSH +%left '+' '-' +%left '*' '/' '%' +%right UNARY + +%% + +start: exp1 + { + expression_value = $1; + } + ; + +/* Expressions, including the comma operator. */ +exp1: exp + | exp1 ',' exp + { + $$ = $3; + } + ; + +/* Expressions, not including the comma operator. */ +exp: '-' exp %prec UNARY + { + $$ = mcpp_semantic_neg( $2 ); + } + | '!' exp %prec UNARY + { + $$ = mcpp_semantic_not( $2 ); + } + | '+' exp %prec UNARY + { + $$ = $2; + } + | '~' exp %prec UNARY + { + $$ = mcpp_semantic_compl( $2 ); + } + | '(' exp1 ')' + { + $$ = $2; + } + ; + +/* Binary operators in order of decreasing precedence. */ +exp: exp '*' exp + { + $$ = mcpp_semantic_mul( $1, $3 ); + } + | exp '/' exp + { + $$ = mcpp_semantic_div( &semantic_context, $1, $3, + !skip_evaluation ); + } + | exp '%' exp + { + $$ = mcpp_semantic_mod( &semantic_context, $1, $3, + !skip_evaluation ); + } + | exp '+' exp + { + $$ = mcpp_semantic_add( $1, $3 ); + } + | exp '-' exp + { + $$ = mcpp_semantic_sub( $1, $3 ); + } + | exp LSH exp + { + $$ = mcpp_semantic_lshift( $1, $3 ); + } + | exp RSH exp + { + $$ = mcpp_semantic_rshift( $1, $3 ); + } + | exp EQUAL exp + { + $$ = mcpp_semantic_equal( $1, $3 ); + } + | exp NOTEQUAL exp + { + $$ = mcpp_semantic_notequal( $1, $3 ); + } + | exp LEQ exp + { + $$ = mcpp_semantic_leq( $1, $3 ); + } + | exp GEQ exp + { + $$ = mcpp_semantic_geq( $1, $3 ); + } + | exp '<' exp + { + $$ = mcpp_semantic_less( $1, $3 ); + } + | exp '>' exp + { + $$ = mcpp_semantic_greater( $1, $3 ); + } + | exp '&' exp + { + $$ = mcpp_semantic_bitand( $1, $3 ); + } + | exp '^' exp + { + $$ = mcpp_semantic_bitxor( $1, $3 ); + } + | exp '|' exp + { + $$ = mcpp_semantic_bitor( $1, $3 ); + } + | exp AND + { + skip_evaluation += !mcpp_semantic_true( $1 ); + } + exp + { + skip_evaluation -= !mcpp_semantic_true( $1 ); + $$ = mcpp_semantic_and( $1, $4 ); + } + | exp OR + { + skip_evaluation += !!mcpp_semantic_true( $1 ); + } + exp + { + skip_evaluation -= !!mcpp_semantic_true( $1 ); + $$ = mcpp_semantic_or( $1, $4 ); + } + | exp '?' + { + skip_evaluation += !mcpp_semantic_true( $1 ); + } + exp ':' + { + skip_evaluation += !!mcpp_semantic_true( $1 ) - + !mcpp_semantic_true( $1 ); + } + exp + { + skip_evaluation -= !!mcpp_semantic_true( $1 ); + $$ = mcpp_semantic_conditional( $1, $4, $7 ); + } + | INT + { + $$ = $1; + } + | CHAR + { + $$ = $1; + } + | NAME + { + $$ = mcpp_semantic_make( 0, 0 ); + } + ; + +/******* END OF GRAMMAR *******/ +%% + +struct token +{ + const char *operator; + int token; +}; + +static struct token tokentab2[] = +{ + { "&&", AND }, + { "||", OR }, + { "<<", LSH }, + { ">>", RSH }, + { "==", EQUAL }, + { "!=", NOTEQUAL }, + { "<=", LEQ }, + { ">=", GEQ }, + { "++", ERROR }, + { "--", ERROR }, + { NULL, ERROR } +}; + +static int +mcpp_expr_space( __mpu_char16_t c ) +{ + return( c == ' ' || c == '\t' || c == '\r' || c == '\n' || + c == '\f' || c == '\v' ); +} + +static int +mcpp_expr_hex_digit( __mpu_char16_t c ) +{ + if( c >= '0' && c <= '9' ) return( c - '0' ); + if( c >= 'a' && c <= 'f' ) return( c - 'a' + 10 ); + if( c >= 'A' && c <= 'F' ) return( c - 'A' + 10 ); + return( -1 ); +} + +static int +mcpp_expr_error_token( const char *message ) +{ + mcpp_zubr_error( (char *)message ); + return( ERROR ); +} + +static __mpu_uint16_t +mcpp_expr_escape( int *failed ) +{ + __mpu_char16_t c; + __mpu_uint32_t value; + int digit; + unsigned count; + + *failed = 0; + if( lexptr >= lexend ) + { + *failed = 1; + mcpp_zubr_error( "incomplete escape sequence in #if expression" ); + return( 0 ); + } + + c = *lexptr++; + switch( c ) + { + case 'a': return( (__mpu_uint16_t)'\a' ); + case 'b': return( (__mpu_uint16_t)'\b' ); + case 'e': + case 'E': return( (__mpu_uint16_t)033 ); + case 'f': return( (__mpu_uint16_t)'\f' ); + case 'n': return( (__mpu_uint16_t)'\n' ); + case 'r': return( (__mpu_uint16_t)'\r' ); + case 't': return( (__mpu_uint16_t)'\t' ); + case 'v': return( (__mpu_uint16_t)'\v' ); + case '\\': return( (__mpu_uint16_t)'\\' ); + case '\'': return( (__mpu_uint16_t)'\'' ); + case '"': return( (__mpu_uint16_t)'"' ); + case '?': return( (__mpu_uint16_t)'?' ); + + case 'x': + case 'X': + value = 0; + count = 0; + while( lexptr < lexend ) + { + digit = mcpp_expr_hex_digit( *lexptr ); + if( digit < 0 ) break; + if( value > 0xffffU >> 4 ) + { + *failed = 1; + mcpp_zubr_error( "hex escape sequence out of UCS-2 range" ); + return( 0 ); + } + value = (value << 4) + (unsigned)digit; + ++lexptr; + ++count; + } + if( count == 0 ) + { + *failed = 1; + mcpp_zubr_error( "\\x used with no following hex digits" ); + return( 0 ); + } + return( (__mpu_uint16_t)value ); + + default: + if( c >= '0' && c <= '7' ) + { + value = c - '0'; + for( count = 1; count < 6 && lexptr < lexend; ++count ) + { + c = *lexptr; + if( c < '0' || c > '7' ) break; + if( value > 0xffffU >> 3 ) + { + *failed = 1; + mcpp_zubr_error( "octal escape sequence out of UCS-2 range" ); + return( 0 ); + } + value = (value << 3) + (c - '0'); + ++lexptr; + } + if( value > 0xffffU ) + { + *failed = 1; + mcpp_zubr_error( "octal escape sequence out of UCS-2 range" ); + return( 0 ); + } + return( (__mpu_uint16_t)value ); + } + return( (__mpu_uint16_t)c ); + } +} + +static int +mcpp_expr_character_constant( void ) +{ + __mpu_uint64_t result = 0; + __mpu_uint16_t value; + __mpu_char16_t c; + unsigned count = 0; + int failed; + + ++lexptr; + + while( lexptr < lexend ) + { + c = *lexptr++; + if( c == '\'' ) + break; + + if( c == '\n' ) + return( mcpp_expr_error_token( + "unterminated character constant in #if expression") ); + + if( c == '\\' ) + { + value = mcpp_expr_escape( &failed ); + if( failed ) return( ERROR ); + } + else + value = (__mpu_uint16_t)c; + + ++count; + if( count > 4 ) + return( mcpp_expr_error_token( + "character constant too long in #if expression") ); + + result = (result << 16) | (__mpu_uint64_t)value; + } + + if( lexptr == lexend && (lexptr == 0 || lexptr[-1] != '\'') ) + return( mcpp_expr_error_token( + "unterminated character constant in #if expression") ); + + if( count == 0 ) + return( mcpp_expr_error_token( + "empty character constant in #if expression") ); + + mcpp_zubr_lval.integer = mcpp_semantic_make( result, 1 ); + return( CHAR ); +} + +static int +mcpp_expr_valid_integer_width( unsigned width ) +{ + return( width >= 8 && width <= MPU_REAL_IO_LIMIT && + (width & (width - 1)) == 0 ); +} + +static __mpu_uint64_t +mcpp_expr_apply_integer_width( __mpu_uint64_t value, unsigned width, + int unsignedp ) +{ + __mpu_uint64_t mask; + __mpu_uint64_t sign; + + if( width >= 64 ) + return( value ); + + mask = (((__mpu_uint64_t)1 << width) - 1); + value &= mask; + + if( unsignedp ) + return( value ); + + sign = ((__mpu_uint64_t)1 << (width - 1)); + if( value & sign ) + value |= ~mask; + + return( value ); +} + +static int +mcpp_expr_number( void ) +{ + const __mpu_char16_t *start = lexptr; + const __mpu_char16_t *end; + const __mpu_char16_t *number_end; + const __mpu_char16_t *p; + __mpu_char8_t *ascii; + __mpu_uint64_t value = 0; + size_t number_length; + unsigned base = 10; + unsigned width = 0; + int digit; + int have_digit = 0; + int unsignedp = 0; + int have_width = 0; + int width_over_64 = 0; + + while( lexptr < lexend && + (mcpu_pp_is_identifier_char(*lexptr) || *lexptr == '.') ) + ++lexptr; + end = lexptr; + + for( p = start; p < end; ++p ) + if( *p == '.' ) + return( mcpp_expr_error_token( + "floating point numbers not allowed in #if expressions") ); + + p = start; + if( end - p >= 2 && p[0] == '0' && + (p[1] == 'x' || p[1] == 'X') ) + { + base = 16; + p += 2; + } + else if( end - p >= 2 && p[0] == '0' && + (p[1] == 'b' || p[1] == 'B') ) + { + base = 2; + p += 2; + } + else if( end - p > 1 && p[0] == '0' ) + base = 8; + + for( ; p < end; ++p ) + { + if( *p > 0x7f ) + break; + digit = mcpp_expr_hex_digit( *p ); + if( digit < 0 || (unsigned)digit >= base ) + break; + have_digit = 1; + } + number_end = p; + + if( !have_digit ) + return( mcpp_expr_error_token( + "invalid integer constant in #if expression") ); + + if( p < end && (*p == 'z' || *p == 'Z') ) + { + have_width = 1; + ++p; + if( p == end || *p < '0' || *p > '9' ) + return( mcpp_expr_error_token( + "integer width suffix requires decimal digits after z/Z") ); + + while( p < end && *p >= '0' && *p <= '9' ) + { + unsigned d = (unsigned)(*p - '0'); + + if( !width_over_64 ) + { + if( width > (64U - d) / 10U ) + width_over_64 = 1; + else + { + width = width * 10U + d; + if( width > 64U ) + width_over_64 = 1; + } + } + ++p; + } + + if( p < end && (*p == 'u' || *p == 'U') ) + { + unsignedp = 1; + ++p; + } + + if( p != end ) + return( mcpp_expr_error_token( + "invalid characters after integer width suffix") ); + + if( width_over_64 ) + return( mcpp_expr_error_token( + "integer constants wider than 64 bits are not allowed in conditional directives") ); + + if( !mcpp_expr_valid_integer_width(width) ) + { + mcpp_semantic_warning( + &semantic_context, + "invalid zNNN integer-width suffix; suffix ignored" ); + have_width = 0; + } + } + else if( p < end && (*p == 'u' || *p == 'U') ) + { + unsignedp = 1; + ++p; + if( p != end ) + return( mcpp_expr_error_token( + "invalid characters after integer suffix") ); + } + else if( p != end ) + return( mcpp_expr_error_token( + "invalid integer suffix in #if expression") ); + + number_length = (size_t)(number_end - start); + ascii = (__mpu_char8_t *)malloc( number_length + 1 ); + if( ascii == NULL ) + { + mcpp_zubr_error( "out of memory while parsing #if expression" ); + return( ERROR ); + } + + for( p = start; p < number_end; ++p ) + { + if( *p > 0x7f ) + { + free( ascii ); + return( mcpp_expr_error_token( + "non-ASCII character in integer constant") ); + } + ascii[p - start] = (__mpu_char8_t)*p; + } + ascii[number_length] = 0; + + __mpu_clo(); + iatoui( (mpu_int *)&value, ascii, (int)sizeof(value) ); + free( ascii ); + + if( __mpu_gto() ) + return( mcpp_expr_error_token( + "integer constant does not fit in 64 bits") ); + + if( have_width ) + { + value = mcpp_expr_apply_integer_width( value, width, unsignedp ); + mcpp_zubr_lval.integer = mcpp_semantic_make( value, unsignedp ); + } + else + mcpp_zubr_lval.integer = + mcpp_semantic_make( value, + unsignedp || value > (__mpu_uint64_t)INT64_MAX ); + + return( INT ); +} + +static void +mcpp_zubr_error( char *s ) +{ + mcpp_semantic_error( &semantic_context, s ); + skip_evaluation = 0; +} + +/**************************************************** + Read one token, getting UCS-2 characters through + LEXPTR. + ****************************************************/ +static int +mcpp_zubr_lex( void ) +{ + const __mpu_char16_t *tokstart; + struct token *toktab; + __mpu_char16_t c; + +retry: + while( lexptr < lexend && mcpp_expr_space(*lexptr) ) + ++lexptr; + + if( lexptr >= lexend ) + return( 0 ); + + tokstart = lexptr; + c = *tokstart; + + for( toktab = tokentab2; toktab->operator != NULL; ++toktab ) + { + if( lexend - tokstart >= 2 && + c == (__mpu_char16_t)(unsigned char)toktab->operator[0] && + tokstart[1] == (__mpu_char16_t)(unsigned char)toktab->operator[1] ) + { + lexptr += 2; + if( toktab->token == ERROR ) + return( mcpp_expr_error_token( + "increment/decrement operator not allowed in #if expression") ); + return( toktab->token ); + } + } + + if( c >= '0' && c <= '9' ) + return( mcpp_expr_number() ); + + if( c == '\'' ) + return( mcpp_expr_character_constant() ); + + if( mcpu_pp_is_identifier_start(c) ) + { + ++lexptr; + while( lexptr < lexend && mcpu_pp_is_identifier_char(*lexptr) ) + ++lexptr; + mcpp_zubr_lval.name.address = tokstart; + mcpp_zubr_lval.name.length = (size_t)(lexptr - tokstart); + return( NAME ); + } + + ++lexptr; + switch( c ) + { + case '(': + case ')': + case '?': + case ':': + case ',': + case '*': + case '/': + case '%': + case '~': + case '^': + return( (int)c ); + + case '+': + case '-': + return( (int)c ); + + case '!': + case '<': + case '>': + case '&': + case '|': + return( (int)c ); + + case '"': + case '`': + return( mcpp_expr_error_token( + "string constants not allowed in #if expressions") ); + + default: + mcpp_zubr_error( "invalid token in #if expression" ); + goto retry; + } +} + +int +mcpp_expr_parse( const __mpu_char16_t *text, size_t length, + const char *filename, unsigned line_number, + const mcpp_options *options, int *result ) +{ + int rc; + + if( text == NULL || filename == NULL || result == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + mcpp_semantic_context_init( &semantic_context, filename, line_number, + options ); + expression_value = mcpp_semantic_make( 0, 0 ); + skip_evaluation = 0; + lexptr = text; + lexend = text + length; + + if( length == 0 ) + { + mcpp_zubr_error( "empty #if expression" ); + return( -1 ); + } + + rc = mcpp_zubr_parse(); + if( rc != 0 || mcpp_semantic_failed(&semantic_context) ) + return( -1 ); + + *result = mcpp_semantic_true( expression_value ); + return( 0 ); +} |
