summaryrefslogtreecommitdiff
path: root/src/mcpp-expr.zubr
diff options
context:
space:
mode:
authorkx <kx@radix-linux.su>2026-10-01 12:02:28 +0300
committerkx <kx@radix-linux.su>2026-10-01 12:02:28 +0300
commitba1b04d64bdfaf377915b22f77520216ebe47674 (patch)
tree044343035c28b2b04ca9725f26e9b7c39342d19c /src/mcpp-expr.zubr
parent124b140798456778e2a96851cc7268aa9a726698 (diff)
downloadmcpu-cpp-1.0.2.tar.xz
Version 1.0.21.0.2
Diffstat (limited to 'src/mcpp-expr.zubr')
-rw-r--r--src/mcpp-expr.zubr723
1 files changed, 723 insertions, 0 deletions
diff --git a/src/mcpp-expr.zubr b/src/mcpp-expr.zubr
new file mode 100644
index 0000000..1ec5823
--- /dev/null
+++ b/src/mcpp-expr.zubr
@@ -0,0 +1,723 @@
+/***************************************************************
+ MCPP_EXPR.C
+
+ This file containt the grammar & procedure for
+ parse expressions for MCPU-CPP .
+
+ PART OF : MCPU-CPP - MCPU language preproccessor .
+
+ NOTE : NONE .
+
+ Copyright (C) 1998 - 2026 by Andrey V.Kosteltsev.
+ All Rights Reserved.
+ ***************************************************************/
+
+%{
+
+#include <defs.h>
+
+static int mcpp_zubr_lex( void );
+static void mcpp_zubr_error( char *s );
+
+static mcpp_integer expression_value;
+static mcpp_semantic_context semantic_context;
+
+/************************************************************
+ Nonzero means do not evaluate this expression.
+ This is a count, since unevaluated expressions can nest.
+ ************************************************************/
+static int skip_evaluation;
+
+/************************************************************
+ During parsing of an MCPU-CPP expression, LEXPTR points to
+ the next UCS-2 character and LEXEND points one character
+ past the input expression.
+ ************************************************************/
+static const __mpu_char16_t *lexptr;
+static const __mpu_char16_t *lexend;
+
+%}
+
+%union
+{
+ mcpp_integer integer;
+ struct name
+ {
+ const __mpu_char16_t *address;
+ size_t length;
+ } name;
+}
+
+%type <integer> exp exp1 start
+%token <integer> INT CHAR
+%token <name> NAME
+%token <integer> ERROR
+
+%right '?' ':'
+%left ','
+%left OR
+%left AND
+%left '|'
+%left '^'
+%left '&'
+%left EQUAL NOTEQUAL
+%left '<' '>' LEQ GEQ
+%left LSH RSH
+%left '+' '-'
+%left '*' '/' '%'
+%right UNARY
+
+%%
+
+start: exp1
+ {
+ expression_value = $1;
+ }
+ ;
+
+/* Expressions, including the comma operator. */
+exp1: exp
+ | exp1 ',' exp
+ {
+ $$ = $3;
+ }
+ ;
+
+/* Expressions, not including the comma operator. */
+exp: '-' exp %prec UNARY
+ {
+ $$ = mcpp_semantic_neg( $2 );
+ }
+ | '!' exp %prec UNARY
+ {
+ $$ = mcpp_semantic_not( $2 );
+ }
+ | '+' exp %prec UNARY
+ {
+ $$ = $2;
+ }
+ | '~' exp %prec UNARY
+ {
+ $$ = mcpp_semantic_compl( $2 );
+ }
+ | '(' exp1 ')'
+ {
+ $$ = $2;
+ }
+ ;
+
+/* Binary operators in order of decreasing precedence. */
+exp: exp '*' exp
+ {
+ $$ = mcpp_semantic_mul( $1, $3 );
+ }
+ | exp '/' exp
+ {
+ $$ = mcpp_semantic_div( &semantic_context, $1, $3,
+ !skip_evaluation );
+ }
+ | exp '%' exp
+ {
+ $$ = mcpp_semantic_mod( &semantic_context, $1, $3,
+ !skip_evaluation );
+ }
+ | exp '+' exp
+ {
+ $$ = mcpp_semantic_add( $1, $3 );
+ }
+ | exp '-' exp
+ {
+ $$ = mcpp_semantic_sub( $1, $3 );
+ }
+ | exp LSH exp
+ {
+ $$ = mcpp_semantic_lshift( $1, $3 );
+ }
+ | exp RSH exp
+ {
+ $$ = mcpp_semantic_rshift( $1, $3 );
+ }
+ | exp EQUAL exp
+ {
+ $$ = mcpp_semantic_equal( $1, $3 );
+ }
+ | exp NOTEQUAL exp
+ {
+ $$ = mcpp_semantic_notequal( $1, $3 );
+ }
+ | exp LEQ exp
+ {
+ $$ = mcpp_semantic_leq( $1, $3 );
+ }
+ | exp GEQ exp
+ {
+ $$ = mcpp_semantic_geq( $1, $3 );
+ }
+ | exp '<' exp
+ {
+ $$ = mcpp_semantic_less( $1, $3 );
+ }
+ | exp '>' exp
+ {
+ $$ = mcpp_semantic_greater( $1, $3 );
+ }
+ | exp '&' exp
+ {
+ $$ = mcpp_semantic_bitand( $1, $3 );
+ }
+ | exp '^' exp
+ {
+ $$ = mcpp_semantic_bitxor( $1, $3 );
+ }
+ | exp '|' exp
+ {
+ $$ = mcpp_semantic_bitor( $1, $3 );
+ }
+ | exp AND
+ {
+ skip_evaluation += !mcpp_semantic_true( $1 );
+ }
+ exp
+ {
+ skip_evaluation -= !mcpp_semantic_true( $1 );
+ $$ = mcpp_semantic_and( $1, $4 );
+ }
+ | exp OR
+ {
+ skip_evaluation += !!mcpp_semantic_true( $1 );
+ }
+ exp
+ {
+ skip_evaluation -= !!mcpp_semantic_true( $1 );
+ $$ = mcpp_semantic_or( $1, $4 );
+ }
+ | exp '?'
+ {
+ skip_evaluation += !mcpp_semantic_true( $1 );
+ }
+ exp ':'
+ {
+ skip_evaluation += !!mcpp_semantic_true( $1 ) -
+ !mcpp_semantic_true( $1 );
+ }
+ exp
+ {
+ skip_evaluation -= !!mcpp_semantic_true( $1 );
+ $$ = mcpp_semantic_conditional( $1, $4, $7 );
+ }
+ | INT
+ {
+ $$ = $1;
+ }
+ | CHAR
+ {
+ $$ = $1;
+ }
+ | NAME
+ {
+ $$ = mcpp_semantic_make( 0, 0 );
+ }
+ ;
+
+/******* END OF GRAMMAR *******/
+%%
+
+struct token
+{
+ const char *operator;
+ int token;
+};
+
+static struct token tokentab2[] =
+{
+ { "&&", AND },
+ { "||", OR },
+ { "<<", LSH },
+ { ">>", RSH },
+ { "==", EQUAL },
+ { "!=", NOTEQUAL },
+ { "<=", LEQ },
+ { ">=", GEQ },
+ { "++", ERROR },
+ { "--", ERROR },
+ { NULL, ERROR }
+};
+
+static int
+mcpp_expr_space( __mpu_char16_t c )
+{
+ return( c == ' ' || c == '\t' || c == '\r' || c == '\n' ||
+ c == '\f' || c == '\v' );
+}
+
+static int
+mcpp_expr_hex_digit( __mpu_char16_t c )
+{
+ if( c >= '0' && c <= '9' ) return( c - '0' );
+ if( c >= 'a' && c <= 'f' ) return( c - 'a' + 10 );
+ if( c >= 'A' && c <= 'F' ) return( c - 'A' + 10 );
+ return( -1 );
+}
+
+static int
+mcpp_expr_error_token( const char *message )
+{
+ mcpp_zubr_error( (char *)message );
+ return( ERROR );
+}
+
+static __mpu_uint16_t
+mcpp_expr_escape( int *failed )
+{
+ __mpu_char16_t c;
+ __mpu_uint32_t value;
+ int digit;
+ unsigned count;
+
+ *failed = 0;
+ if( lexptr >= lexend )
+ {
+ *failed = 1;
+ mcpp_zubr_error( "incomplete escape sequence in #if expression" );
+ return( 0 );
+ }
+
+ c = *lexptr++;
+ switch( c )
+ {
+ case 'a': return( (__mpu_uint16_t)'\a' );
+ case 'b': return( (__mpu_uint16_t)'\b' );
+ case 'e':
+ case 'E': return( (__mpu_uint16_t)033 );
+ case 'f': return( (__mpu_uint16_t)'\f' );
+ case 'n': return( (__mpu_uint16_t)'\n' );
+ case 'r': return( (__mpu_uint16_t)'\r' );
+ case 't': return( (__mpu_uint16_t)'\t' );
+ case 'v': return( (__mpu_uint16_t)'\v' );
+ case '\\': return( (__mpu_uint16_t)'\\' );
+ case '\'': return( (__mpu_uint16_t)'\'' );
+ case '"': return( (__mpu_uint16_t)'"' );
+ case '?': return( (__mpu_uint16_t)'?' );
+
+ case 'x':
+ case 'X':
+ value = 0;
+ count = 0;
+ while( lexptr < lexend )
+ {
+ digit = mcpp_expr_hex_digit( *lexptr );
+ if( digit < 0 ) break;
+ if( value > 0xffffU >> 4 )
+ {
+ *failed = 1;
+ mcpp_zubr_error( "hex escape sequence out of UCS-2 range" );
+ return( 0 );
+ }
+ value = (value << 4) + (unsigned)digit;
+ ++lexptr;
+ ++count;
+ }
+ if( count == 0 )
+ {
+ *failed = 1;
+ mcpp_zubr_error( "\\x used with no following hex digits" );
+ return( 0 );
+ }
+ return( (__mpu_uint16_t)value );
+
+ default:
+ if( c >= '0' && c <= '7' )
+ {
+ value = c - '0';
+ for( count = 1; count < 6 && lexptr < lexend; ++count )
+ {
+ c = *lexptr;
+ if( c < '0' || c > '7' ) break;
+ if( value > 0xffffU >> 3 )
+ {
+ *failed = 1;
+ mcpp_zubr_error( "octal escape sequence out of UCS-2 range" );
+ return( 0 );
+ }
+ value = (value << 3) + (c - '0');
+ ++lexptr;
+ }
+ if( value > 0xffffU )
+ {
+ *failed = 1;
+ mcpp_zubr_error( "octal escape sequence out of UCS-2 range" );
+ return( 0 );
+ }
+ return( (__mpu_uint16_t)value );
+ }
+ return( (__mpu_uint16_t)c );
+ }
+}
+
+static int
+mcpp_expr_character_constant( void )
+{
+ __mpu_uint64_t result = 0;
+ __mpu_uint16_t value;
+ __mpu_char16_t c;
+ unsigned count = 0;
+ int failed;
+
+ ++lexptr;
+
+ while( lexptr < lexend )
+ {
+ c = *lexptr++;
+ if( c == '\'' )
+ break;
+
+ if( c == '\n' )
+ return( mcpp_expr_error_token(
+ "unterminated character constant in #if expression") );
+
+ if( c == '\\' )
+ {
+ value = mcpp_expr_escape( &failed );
+ if( failed ) return( ERROR );
+ }
+ else
+ value = (__mpu_uint16_t)c;
+
+ ++count;
+ if( count > 4 )
+ return( mcpp_expr_error_token(
+ "character constant too long in #if expression") );
+
+ result = (result << 16) | (__mpu_uint64_t)value;
+ }
+
+ if( lexptr == lexend && (lexptr == 0 || lexptr[-1] != '\'') )
+ return( mcpp_expr_error_token(
+ "unterminated character constant in #if expression") );
+
+ if( count == 0 )
+ return( mcpp_expr_error_token(
+ "empty character constant in #if expression") );
+
+ mcpp_zubr_lval.integer = mcpp_semantic_make( result, 1 );
+ return( CHAR );
+}
+
+static int
+mcpp_expr_valid_integer_width( unsigned width )
+{
+ return( width >= 8 && width <= MPU_REAL_IO_LIMIT &&
+ (width & (width - 1)) == 0 );
+}
+
+static __mpu_uint64_t
+mcpp_expr_apply_integer_width( __mpu_uint64_t value, unsigned width,
+ int unsignedp )
+{
+ __mpu_uint64_t mask;
+ __mpu_uint64_t sign;
+
+ if( width >= 64 )
+ return( value );
+
+ mask = (((__mpu_uint64_t)1 << width) - 1);
+ value &= mask;
+
+ if( unsignedp )
+ return( value );
+
+ sign = ((__mpu_uint64_t)1 << (width - 1));
+ if( value & sign )
+ value |= ~mask;
+
+ return( value );
+}
+
+static int
+mcpp_expr_number( void )
+{
+ const __mpu_char16_t *start = lexptr;
+ const __mpu_char16_t *end;
+ const __mpu_char16_t *number_end;
+ const __mpu_char16_t *p;
+ __mpu_char8_t *ascii;
+ __mpu_uint64_t value = 0;
+ size_t number_length;
+ unsigned base = 10;
+ unsigned width = 0;
+ int digit;
+ int have_digit = 0;
+ int unsignedp = 0;
+ int have_width = 0;
+ int width_over_64 = 0;
+
+ while( lexptr < lexend &&
+ (mcpu_pp_is_identifier_char(*lexptr) || *lexptr == '.') )
+ ++lexptr;
+ end = lexptr;
+
+ for( p = start; p < end; ++p )
+ if( *p == '.' )
+ return( mcpp_expr_error_token(
+ "floating point numbers not allowed in #if expressions") );
+
+ p = start;
+ if( end - p >= 2 && p[0] == '0' &&
+ (p[1] == 'x' || p[1] == 'X') )
+ {
+ base = 16;
+ p += 2;
+ }
+ else if( end - p >= 2 && p[0] == '0' &&
+ (p[1] == 'b' || p[1] == 'B') )
+ {
+ base = 2;
+ p += 2;
+ }
+ else if( end - p > 1 && p[0] == '0' )
+ base = 8;
+
+ for( ; p < end; ++p )
+ {
+ if( *p > 0x7f )
+ break;
+ digit = mcpp_expr_hex_digit( *p );
+ if( digit < 0 || (unsigned)digit >= base )
+ break;
+ have_digit = 1;
+ }
+ number_end = p;
+
+ if( !have_digit )
+ return( mcpp_expr_error_token(
+ "invalid integer constant in #if expression") );
+
+ if( p < end && (*p == 'z' || *p == 'Z') )
+ {
+ have_width = 1;
+ ++p;
+ if( p == end || *p < '0' || *p > '9' )
+ return( mcpp_expr_error_token(
+ "integer width suffix requires decimal digits after z/Z") );
+
+ while( p < end && *p >= '0' && *p <= '9' )
+ {
+ unsigned d = (unsigned)(*p - '0');
+
+ if( !width_over_64 )
+ {
+ if( width > (64U - d) / 10U )
+ width_over_64 = 1;
+ else
+ {
+ width = width * 10U + d;
+ if( width > 64U )
+ width_over_64 = 1;
+ }
+ }
+ ++p;
+ }
+
+ if( p < end && (*p == 'u' || *p == 'U') )
+ {
+ unsignedp = 1;
+ ++p;
+ }
+
+ if( p != end )
+ return( mcpp_expr_error_token(
+ "invalid characters after integer width suffix") );
+
+ if( width_over_64 )
+ return( mcpp_expr_error_token(
+ "integer constants wider than 64 bits are not allowed in conditional directives") );
+
+ if( !mcpp_expr_valid_integer_width(width) )
+ {
+ mcpp_semantic_warning(
+ &semantic_context,
+ "invalid zNNN integer-width suffix; suffix ignored" );
+ have_width = 0;
+ }
+ }
+ else if( p < end && (*p == 'u' || *p == 'U') )
+ {
+ unsignedp = 1;
+ ++p;
+ if( p != end )
+ return( mcpp_expr_error_token(
+ "invalid characters after integer suffix") );
+ }
+ else if( p != end )
+ return( mcpp_expr_error_token(
+ "invalid integer suffix in #if expression") );
+
+ number_length = (size_t)(number_end - start);
+ ascii = (__mpu_char8_t *)malloc( number_length + 1 );
+ if( ascii == NULL )
+ {
+ mcpp_zubr_error( "out of memory while parsing #if expression" );
+ return( ERROR );
+ }
+
+ for( p = start; p < number_end; ++p )
+ {
+ if( *p > 0x7f )
+ {
+ free( ascii );
+ return( mcpp_expr_error_token(
+ "non-ASCII character in integer constant") );
+ }
+ ascii[p - start] = (__mpu_char8_t)*p;
+ }
+ ascii[number_length] = 0;
+
+ __mpu_clo();
+ iatoui( (mpu_int *)&value, ascii, (int)sizeof(value) );
+ free( ascii );
+
+ if( __mpu_gto() )
+ return( mcpp_expr_error_token(
+ "integer constant does not fit in 64 bits") );
+
+ if( have_width )
+ {
+ value = mcpp_expr_apply_integer_width( value, width, unsignedp );
+ mcpp_zubr_lval.integer = mcpp_semantic_make( value, unsignedp );
+ }
+ else
+ mcpp_zubr_lval.integer =
+ mcpp_semantic_make( value,
+ unsignedp || value > (__mpu_uint64_t)INT64_MAX );
+
+ return( INT );
+}
+
+static void
+mcpp_zubr_error( char *s )
+{
+ mcpp_semantic_error( &semantic_context, s );
+ skip_evaluation = 0;
+}
+
+/****************************************************
+ Read one token, getting UCS-2 characters through
+ LEXPTR.
+ ****************************************************/
+static int
+mcpp_zubr_lex( void )
+{
+ const __mpu_char16_t *tokstart;
+ struct token *toktab;
+ __mpu_char16_t c;
+
+retry:
+ while( lexptr < lexend && mcpp_expr_space(*lexptr) )
+ ++lexptr;
+
+ if( lexptr >= lexend )
+ return( 0 );
+
+ tokstart = lexptr;
+ c = *tokstart;
+
+ for( toktab = tokentab2; toktab->operator != NULL; ++toktab )
+ {
+ if( lexend - tokstart >= 2 &&
+ c == (__mpu_char16_t)(unsigned char)toktab->operator[0] &&
+ tokstart[1] == (__mpu_char16_t)(unsigned char)toktab->operator[1] )
+ {
+ lexptr += 2;
+ if( toktab->token == ERROR )
+ return( mcpp_expr_error_token(
+ "increment/decrement operator not allowed in #if expression") );
+ return( toktab->token );
+ }
+ }
+
+ if( c >= '0' && c <= '9' )
+ return( mcpp_expr_number() );
+
+ if( c == '\'' )
+ return( mcpp_expr_character_constant() );
+
+ if( mcpu_pp_is_identifier_start(c) )
+ {
+ ++lexptr;
+ while( lexptr < lexend && mcpu_pp_is_identifier_char(*lexptr) )
+ ++lexptr;
+ mcpp_zubr_lval.name.address = tokstart;
+ mcpp_zubr_lval.name.length = (size_t)(lexptr - tokstart);
+ return( NAME );
+ }
+
+ ++lexptr;
+ switch( c )
+ {
+ case '(':
+ case ')':
+ case '?':
+ case ':':
+ case ',':
+ case '*':
+ case '/':
+ case '%':
+ case '~':
+ case '^':
+ return( (int)c );
+
+ case '+':
+ case '-':
+ return( (int)c );
+
+ case '!':
+ case '<':
+ case '>':
+ case '&':
+ case '|':
+ return( (int)c );
+
+ case '"':
+ case '`':
+ return( mcpp_expr_error_token(
+ "string constants not allowed in #if expressions") );
+
+ default:
+ mcpp_zubr_error( "invalid token in #if expression" );
+ goto retry;
+ }
+}
+
+int
+mcpp_expr_parse( const __mpu_char16_t *text, size_t length,
+ const char *filename, unsigned line_number,
+ const mcpp_options *options, int *result )
+{
+ int rc;
+
+ if( text == NULL || filename == NULL || result == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ mcpp_semantic_context_init( &semantic_context, filename, line_number,
+ options );
+ expression_value = mcpp_semantic_make( 0, 0 );
+ skip_evaluation = 0;
+ lexptr = text;
+ lexend = text + length;
+
+ if( length == 0 )
+ {
+ mcpp_zubr_error( "empty #if expression" );
+ return( -1 );
+ }
+
+ rc = mcpp_zubr_parse();
+ if( rc != 0 || mcpp_semantic_failed(&semantic_context) )
+ return( -1 );
+
+ *result = mcpp_semantic_true( expression_value );
+ return( 0 );
+}