/*************************************************************** MCPP_EXPR.C This file containt the grammar & procedure for parse expressions for MCPU-CPP . PART OF : MCPU-CPP - MCPU language preproccessor . NOTE : NONE . Copyright (C) 1998 - 2026 by Andrey V.Kosteltsev. All Rights Reserved. ***************************************************************/ %{ #include static int mcpp_zubr_lex( void ); static void mcpp_zubr_error( char *s ); static mcpp_integer expression_value; static mcpp_semantic_context semantic_context; /************************************************************ Nonzero means do not evaluate this expression. This is a count, since unevaluated expressions can nest. ************************************************************/ static int skip_evaluation; /************************************************************ During parsing of an MCPU-CPP expression, LEXPTR points to the next UCS-2 character and LEXEND points one character past the input expression. ************************************************************/ static const __mpu_char16_t *lexptr; static const __mpu_char16_t *lexend; %} %union { mcpp_integer integer; struct name { const __mpu_char16_t *address; size_t length; } name; } %type exp exp1 start %token INT CHAR %token NAME %token ERROR %right '?' ':' %left ',' %left OR %left AND %left '|' %left '^' %left '&' %left EQUAL NOTEQUAL %left '<' '>' LEQ GEQ %left LSH RSH %left '+' '-' %left '*' '/' '%' %right UNARY %% start: exp1 { expression_value = $1; } ; /* Expressions, including the comma operator. */ exp1: exp | exp1 ',' exp { $$ = $3; } ; /* Expressions, not including the comma operator. */ exp: '-' exp %prec UNARY { $$ = mcpp_semantic_neg( $2 ); } | '!' exp %prec UNARY { $$ = mcpp_semantic_not( $2 ); } | '+' exp %prec UNARY { $$ = $2; } | '~' exp %prec UNARY { $$ = mcpp_semantic_compl( $2 ); } | '(' exp1 ')' { $$ = $2; } ; /* Binary operators in order of decreasing precedence. */ exp: exp '*' exp { $$ = mcpp_semantic_mul( $1, $3 ); } | exp '/' exp { $$ = mcpp_semantic_div( &semantic_context, $1, $3, !skip_evaluation ); } | exp '%' exp { $$ = mcpp_semantic_mod( &semantic_context, $1, $3, !skip_evaluation ); } | exp '+' exp { $$ = mcpp_semantic_add( $1, $3 ); } | exp '-' exp { $$ = mcpp_semantic_sub( $1, $3 ); } | exp LSH exp { $$ = mcpp_semantic_lshift( $1, $3 ); } | exp RSH exp { $$ = mcpp_semantic_rshift( $1, $3 ); } | exp EQUAL exp { $$ = mcpp_semantic_equal( $1, $3 ); } | exp NOTEQUAL exp { $$ = mcpp_semantic_notequal( $1, $3 ); } | exp LEQ exp { $$ = mcpp_semantic_leq( $1, $3 ); } | exp GEQ exp { $$ = mcpp_semantic_geq( $1, $3 ); } | exp '<' exp { $$ = mcpp_semantic_less( $1, $3 ); } | exp '>' exp { $$ = mcpp_semantic_greater( $1, $3 ); } | exp '&' exp { $$ = mcpp_semantic_bitand( $1, $3 ); } | exp '^' exp { $$ = mcpp_semantic_bitxor( $1, $3 ); } | exp '|' exp { $$ = mcpp_semantic_bitor( $1, $3 ); } | exp AND { skip_evaluation += !mcpp_semantic_true( $1 ); } exp { skip_evaluation -= !mcpp_semantic_true( $1 ); $$ = mcpp_semantic_and( $1, $4 ); } | exp OR { skip_evaluation += !!mcpp_semantic_true( $1 ); } exp { skip_evaluation -= !!mcpp_semantic_true( $1 ); $$ = mcpp_semantic_or( $1, $4 ); } | exp '?' { skip_evaluation += !mcpp_semantic_true( $1 ); } exp ':' { skip_evaluation += !!mcpp_semantic_true( $1 ) - !mcpp_semantic_true( $1 ); } exp { skip_evaluation -= !!mcpp_semantic_true( $1 ); $$ = mcpp_semantic_conditional( $1, $4, $7 ); } | INT { $$ = $1; } | CHAR { $$ = $1; } | NAME { $$ = mcpp_semantic_make( 0, 0 ); } ; /******* END OF GRAMMAR *******/ %% struct token { const char *operator; int token; }; static struct token tokentab2[] = { { "&&", AND }, { "||", OR }, { "<<", LSH }, { ">>", RSH }, { "==", EQUAL }, { "!=", NOTEQUAL }, { "<=", LEQ }, { ">=", GEQ }, { "++", ERROR }, { "--", ERROR }, { NULL, ERROR } }; static int mcpp_expr_space( __mpu_char16_t c ) { return( c == ' ' || c == '\t' || c == '\r' || c == '\n' || c == '\f' || c == '\v' ); } static int mcpp_expr_hex_digit( __mpu_char16_t c ) { if( c >= '0' && c <= '9' ) return( c - '0' ); if( c >= 'a' && c <= 'f' ) return( c - 'a' + 10 ); if( c >= 'A' && c <= 'F' ) return( c - 'A' + 10 ); return( -1 ); } static int mcpp_expr_error_token( const char *message ) { mcpp_zubr_error( (char *)message ); return( ERROR ); } static __mpu_uint16_t mcpp_expr_escape( int *failed ) { __mpu_char16_t c; __mpu_uint32_t value; int digit; unsigned count; *failed = 0; if( lexptr >= lexend ) { *failed = 1; mcpp_zubr_error( "incomplete escape sequence in #if expression" ); return( 0 ); } c = *lexptr++; switch( c ) { case 'a': return( (__mpu_uint16_t)'\a' ); case 'b': return( (__mpu_uint16_t)'\b' ); case 'e': case 'E': return( (__mpu_uint16_t)033 ); case 'f': return( (__mpu_uint16_t)'\f' ); case 'n': return( (__mpu_uint16_t)'\n' ); case 'r': return( (__mpu_uint16_t)'\r' ); case 't': return( (__mpu_uint16_t)'\t' ); case 'v': return( (__mpu_uint16_t)'\v' ); case '\\': return( (__mpu_uint16_t)'\\' ); case '\'': return( (__mpu_uint16_t)'\'' ); case '"': return( (__mpu_uint16_t)'"' ); case '?': return( (__mpu_uint16_t)'?' ); case 'x': case 'X': value = 0; count = 0; while( lexptr < lexend ) { digit = mcpp_expr_hex_digit( *lexptr ); if( digit < 0 ) break; if( value > 0xffffU >> 4 ) { *failed = 1; mcpp_zubr_error( "hex escape sequence out of UCS-2 range" ); return( 0 ); } value = (value << 4) + (unsigned)digit; ++lexptr; ++count; } if( count == 0 ) { *failed = 1; mcpp_zubr_error( "\\x used with no following hex digits" ); return( 0 ); } return( (__mpu_uint16_t)value ); default: if( c >= '0' && c <= '7' ) { value = c - '0'; for( count = 1; count < 6 && lexptr < lexend; ++count ) { c = *lexptr; if( c < '0' || c > '7' ) break; if( value > 0xffffU >> 3 ) { *failed = 1; mcpp_zubr_error( "octal escape sequence out of UCS-2 range" ); return( 0 ); } value = (value << 3) + (c - '0'); ++lexptr; } if( value > 0xffffU ) { *failed = 1; mcpp_zubr_error( "octal escape sequence out of UCS-2 range" ); return( 0 ); } return( (__mpu_uint16_t)value ); } return( (__mpu_uint16_t)c ); } } static int mcpp_expr_character_constant( void ) { __mpu_uint64_t result = 0; __mpu_uint16_t value; __mpu_char16_t c; unsigned count = 0; int failed; ++lexptr; while( lexptr < lexend ) { c = *lexptr++; if( c == '\'' ) break; if( c == '\n' ) return( mcpp_expr_error_token( "unterminated character constant in #if expression") ); if( c == '\\' ) { value = mcpp_expr_escape( &failed ); if( failed ) return( ERROR ); } else value = (__mpu_uint16_t)c; ++count; if( count > 4 ) return( mcpp_expr_error_token( "character constant too long in #if expression") ); result = (result << 16) | (__mpu_uint64_t)value; } if( lexptr == lexend && (lexptr == 0 || lexptr[-1] != '\'') ) return( mcpp_expr_error_token( "unterminated character constant in #if expression") ); if( count == 0 ) return( mcpp_expr_error_token( "empty character constant in #if expression") ); mcpp_zubr_lval.integer = mcpp_semantic_make( result, 1 ); return( CHAR ); } static int mcpp_expr_valid_integer_width( unsigned width ) { return( width >= 8 && width <= MPU_REAL_IO_LIMIT && (width & (width - 1)) == 0 ); } static __mpu_uint64_t mcpp_expr_apply_integer_width( __mpu_uint64_t value, unsigned width, int unsignedp ) { __mpu_uint64_t mask; __mpu_uint64_t sign; if( width >= 64 ) return( value ); mask = (((__mpu_uint64_t)1 << width) - 1); value &= mask; if( unsignedp ) return( value ); sign = ((__mpu_uint64_t)1 << (width - 1)); if( value & sign ) value |= ~mask; return( value ); } static int mcpp_expr_number( void ) { const __mpu_char16_t *start = lexptr; const __mpu_char16_t *end; const __mpu_char16_t *number_end; const __mpu_char16_t *p; __mpu_char8_t *ascii; __mpu_uint64_t value = 0; size_t number_length; unsigned base = 10; unsigned width = 0; int digit; int have_digit = 0; int unsignedp = 0; int have_width = 0; int width_over_64 = 0; while( lexptr < lexend && (mcpu_pp_is_identifier_char(*lexptr) || *lexptr == '.') ) ++lexptr; end = lexptr; for( p = start; p < end; ++p ) if( *p == '.' ) return( mcpp_expr_error_token( "floating point numbers not allowed in #if expressions") ); p = start; if( end - p >= 2 && p[0] == '0' && (p[1] == 'x' || p[1] == 'X') ) { base = 16; p += 2; } else if( end - p >= 2 && p[0] == '0' && (p[1] == 'b' || p[1] == 'B') ) { base = 2; p += 2; } else if( end - p > 1 && p[0] == '0' ) base = 8; for( ; p < end; ++p ) { if( *p > 0x7f ) break; digit = mcpp_expr_hex_digit( *p ); if( digit < 0 || (unsigned)digit >= base ) break; have_digit = 1; } number_end = p; if( !have_digit ) return( mcpp_expr_error_token( "invalid integer constant in #if expression") ); if( p < end && (*p == 'z' || *p == 'Z') ) { have_width = 1; ++p; if( p == end || *p < '0' || *p > '9' ) return( mcpp_expr_error_token( "integer width suffix requires decimal digits after z/Z") ); while( p < end && *p >= '0' && *p <= '9' ) { unsigned d = (unsigned)(*p - '0'); if( !width_over_64 ) { if( width > (64U - d) / 10U ) width_over_64 = 1; else { width = width * 10U + d; if( width > 64U ) width_over_64 = 1; } } ++p; } if( p < end && (*p == 'u' || *p == 'U') ) { unsignedp = 1; ++p; } if( p != end ) return( mcpp_expr_error_token( "invalid characters after integer width suffix") ); if( width_over_64 ) return( mcpp_expr_error_token( "integer constants wider than 64 bits are not allowed in conditional directives") ); if( !mcpp_expr_valid_integer_width(width) ) { mcpp_semantic_warning( &semantic_context, "invalid zNNN integer-width suffix; suffix ignored" ); have_width = 0; } } else if( p < end && (*p == 'u' || *p == 'U') ) { unsignedp = 1; ++p; if( p != end ) return( mcpp_expr_error_token( "invalid characters after integer suffix") ); } else if( p != end ) return( mcpp_expr_error_token( "invalid integer suffix in #if expression") ); number_length = (size_t)(number_end - start); ascii = (__mpu_char8_t *)malloc( number_length + 1 ); if( ascii == NULL ) { mcpp_zubr_error( "out of memory while parsing #if expression" ); return( ERROR ); } for( p = start; p < number_end; ++p ) { if( *p > 0x7f ) { free( ascii ); return( mcpp_expr_error_token( "non-ASCII character in integer constant") ); } ascii[p - start] = (__mpu_char8_t)*p; } ascii[number_length] = 0; __mpu_clo(); iatoui( (mpu_int *)&value, ascii, (int)sizeof(value) ); free( ascii ); if( __mpu_gto() ) return( mcpp_expr_error_token( "integer constant does not fit in 64 bits") ); if( have_width ) { value = mcpp_expr_apply_integer_width( value, width, unsignedp ); mcpp_zubr_lval.integer = mcpp_semantic_make( value, unsignedp ); } else mcpp_zubr_lval.integer = mcpp_semantic_make( value, unsignedp || value > (__mpu_uint64_t)INT64_MAX ); return( INT ); } static void mcpp_zubr_error( char *s ) { mcpp_semantic_error( &semantic_context, s ); skip_evaluation = 0; } /**************************************************** Read one token, getting UCS-2 characters through LEXPTR. ****************************************************/ static int mcpp_zubr_lex( void ) { const __mpu_char16_t *tokstart; struct token *toktab; __mpu_char16_t c; retry: while( lexptr < lexend && mcpp_expr_space(*lexptr) ) ++lexptr; if( lexptr >= lexend ) return( 0 ); tokstart = lexptr; c = *tokstart; for( toktab = tokentab2; toktab->operator != NULL; ++toktab ) { if( lexend - tokstart >= 2 && c == (__mpu_char16_t)(unsigned char)toktab->operator[0] && tokstart[1] == (__mpu_char16_t)(unsigned char)toktab->operator[1] ) { lexptr += 2; if( toktab->token == ERROR ) return( mcpp_expr_error_token( "increment/decrement operator not allowed in #if expression") ); return( toktab->token ); } } if( c >= '0' && c <= '9' ) return( mcpp_expr_number() ); if( c == '\'' ) return( mcpp_expr_character_constant() ); if( mcpu_pp_is_identifier_start(c) ) { ++lexptr; while( lexptr < lexend && mcpu_pp_is_identifier_char(*lexptr) ) ++lexptr; mcpp_zubr_lval.name.address = tokstart; mcpp_zubr_lval.name.length = (size_t)(lexptr - tokstart); return( NAME ); } ++lexptr; switch( c ) { case '(': case ')': case '?': case ':': case ',': case '*': case '/': case '%': case '~': case '^': return( (int)c ); case '+': case '-': return( (int)c ); case '!': case '<': case '>': case '&': case '|': return( (int)c ); case '"': case '`': return( mcpp_expr_error_token( "string constants not allowed in #if expressions") ); default: mcpp_zubr_error( "invalid token in #if expression" ); goto retry; } } int mcpp_expr_parse( const __mpu_char16_t *text, size_t length, const char *filename, unsigned line_number, const mcpp_options *options, int *result ) { int rc; if( text == NULL || filename == NULL || result == NULL ) { errno = EINVAL; return( -1 ); } mcpp_semantic_context_init( &semantic_context, filename, line_number, options ); expression_value = mcpp_semantic_make( 0, 0 ); skip_evaluation = 0; lexptr = text; lexend = text + length; if( length == 0 ) { mcpp_zubr_error( "empty #if expression" ); return( -1 ); } rc = mcpp_zubr_parse(); if( rc != 0 || mcpp_semantic_failed(&semantic_context) ) return( -1 ); *result = mcpp_semantic_true( expression_value ); return( 0 ); }