diff options
137 files changed, 24608 insertions, 0 deletions
diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..d2ad001 --- /dev/null +++ b/.gitignore @@ -0,0 +1,67 @@ + +# Files generated by ./bootstrap + +aclocal.m4 +m4/*.m4 +config.h.in +configure +Makefile.in +src/Makefile.in +tests/Makefile.in +doc/Makefile.in +man/Makefile.in +man/ru/Makefile.in +etc/Makefile.in +compile +config.guess +config.sub +depcomp +install-sh +missing +test-driver +src/mcpp-expr.c + + +# Autotools/cache/configure state: + +autom4te.cache +config.h +config.log +config.status +stamp-h1 +Makefile +src/Makefile +tests/Makefile +doc/Makefile +man/Makefile +man/ru/Makefile +etc/Makefile +etc/mcpu-cpp.conf + + +# Build products: + +.deps +src/.deps +tests/.deps +src/mcpu-cpp +tests/mcpp-options-test +tests/*.log +tests/*.trs +tests/test-suite.log +z.output + + +# Distribution products: + +mcpu-cpp-*.tar.gz +mcpu-cpp-*.tar.xz + + +# Editor/backup files: + +*~ +*.swp +*.swo + +*.o @@ -0,0 +1,4 @@ + +Authors of mcpu-cpp (in chronological order of initial contribution) + +Andrey V.Kosteltsev Main author diff --git a/ChangeLog b/ChangeLog new file mode 100644 index 0000000..c902a5c --- /dev/null +++ b/ChangeLog @@ -0,0 +1,381 @@ +2026-10-01 Andrey V. Kosteltsev + + * mcpu-cpp 1.0.2: implement LibMPU, LibMPUIO congiguring check + using libmpu.m4, libmpuio.m4. + +2026-09-30 Andrey V. Kosteltsev + + * mcpu-cpp 1.0.1: integrate English and Russian mcpu-cpp(1) manual + pages into the Automake build and install them through the standard + system man directory. + +2026-09-30 Andrey V. Kosteltsev + + * mcpu-cpp 1.0.0: first public release of the preprocessor. + +2026-09-30 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.51: remove Russian preprocessing-directive aliases while + retaining full Unicode identifier/text support. Update the synchronized + manuals with a GNU-compatible-features section and remove predecessor + references. Adjust only alias-related and version-sensitive tests. + +2026-09-29 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.50: replace the single normative manual with synchronized + Russian and English editions, update the manual to the complete 0.0.49 + feature set, add a LibMPU/LibMPUIO-style developer bootstrap that + regenerates the ZUBR parser and Autotools files, add a .gitignore for + reproducible/generated Git-tree files, and keep release archives and the + existing ordinary build model self-contained and unchanged. + +2026-09-29 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.49: compact invisible source-line gaps using GNU CPP's + 0..7-newline versus 8+-linemarker policy while preserving logical source + coordinates, include enter/return markers, __LINE__, diagnostics and #line. + +2026-09-29 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.48: normalize whitespace belonging to macro replacement lists + without reformatting ordinary source text or actual macro arguments; keep + literal contents and preprocessing-token boundaries intact. + +2026-09-29 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.47: implement standard __VA_OPT__(pp-tokens) with expanded + variadic-emptiness semantics, balanced parentheses, #/##/placemarker/rescan + integration, and explicitly omit historical GNU comma swallowing. + +2026-09-29 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.46: add C99-style variadic function-like macros with ... + and __VA_ARGS__; integrate the variable argument with ordinary prescan, + stringification, token concatenation, placemarkers, rescanning and macro + dumps; intentionally leave GNU named args... and __VA_OPT__ unsupported, + document the contract and add dedicated regression coverage. + +2026-09-29 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.45: implement GNU-like -MG for -M/-MM; integrate missing + generated headers and forced files with the ordered dependency registry + through a separate unresolved textual identity domain, preserve physical + st_dev/st_ino deduplication for real files and GNU-like user/system + filtering, document the model and add dedicated regression coverage. + +2026-09-29 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.44: implement GNU-like -MT/-MQ dependency targets with + repeated and attached forms, exact -MT output, Make-quoted -MQ output and + multiple targets per rule; document the 0.0.43 -include/-imacros forced- + file contract and the new dependency-target behavior in doc/mcpu-cpp.md, + and add dedicated regression coverage. + +2026-09-29 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.42: implement -w as the global warning-output gate; suppress + every warning class before -Werror promotion regardless of command-line + order, leave real errors unaffected, expose -w in --help, document the + warning policy, and add regression coverage for all existing warning + sources. + +2026-09-29 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.41: clean the public command-line interface, remove + obsolete compatibility options, document --object-suffix, add selected + Russian directive aliases (#определить, #строка, #включить, + #включить_следующий, #отменить, #управление, #язык), rename the product + description to "MCPU languages preprocessor", and remove predecessor- + oriented wording from comments and documentation. + +2026-09-29 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.40: implement warning control with -Wcomment/-Wcomments, + -Wno-comment/-Wno-comments, -Wall, -Werror and -Wno-error; preserve + specific-option priority over -Wall, promote every emitted warning via + one diagnostic policy, add comment-lexing diagnostics and regression + coverage, and document the warning model as a separate normative section. + +2026-09-29 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.39: split final macro dumping by macro origin; make -dM + report only source/include and command-line macros, implement -dMP as + predefined macros first followed by ordinary macros, preserve #undef + final-state semantics and deterministic per-group ordering, leave -dD + unchanged, and add documentation/regression coverage. + +2026-09-29 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.38: add -MD/-MMD side-effect dependency generation and + -MF dependency-output selection; derive default .d names from the input + basename or ordinary -o output, preserve the established -M/-MM graph + and system classification, and add documentation/regression coverage. + +2026-09-29 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.37: replace the versioned, compile-time-bound installation + root with a relocatable `$libdir/mcpu` ecosystem tree; derive the runtime + root from the physical executable path, derive the packaged config and + default system include tree from that root, retain configuration overrides, + add physical relocation regression coverage, and document relocatability as + a general MCPU ecosystem principle for cpp/as/ld/run and future libraries. + +2026-09-29 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.36: fix --no-config so it suppresses only configuration-file + reads while retaining the compiled MCPU_CPP_SYSTEM_INCLUDE_PATH default; + keep -nostdinc as the mechanism that removes the standard-system include + tree; add regression coverage for -dsearch-dirs, -dconfig, -v and explicit + include classes, and document the corrected configuration contract. + +2026-09-29 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.35: implement basic -M/-MM dependency generation directly + from the existing include pipeline; classify system dependencies by real + include provenance and propagate system context transitively; deduplicate + physical files by device/inode identity; keep logical #line names out of + make rules; add documentation and regression coverage. + +2026-09-28 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.33: align configure output with the LibMPU/LibMPUIO/ZUBR + package family by adding the common headline and AC_MSG_CFG_PART section + headings while preserving the detailed final summary; normalize every + textual occurrence of the author given name to Andrey. + +2026-09-28 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.32: make -v report one effective include-configuration + snapshot instead of logging every assignment from every configuration + file; add -dsearch-dirs and regression coverage for semantic search-order + dumping, compiled system defaults and -nostdinc; document both interfaces. + +2026-09-28 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.31: introduce versioned MCPU installation under + $libdir/mcpu-0.0.31, install a public $bindir/mcpu-cpp symlink, move + the packaged configuration into the versioned tree, treat /etc/mcpu as + an optional administrator override, keep the unversioned per-user config + at $HOME/.mcpu/mcpu-cpp.conf with highest priority, and restore + MCPU_CPP_SYSTEM_INCLUDE_PATH as a completely replaceable system-header + root. Document the configuration hierarchy, sandbox use case and + #include_next interaction as normative contracts. + +2026-09-28 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.30: make command-line include classes override configured + defaults and define the complete normative search chain. Rename the + configured system root to MCPU_CPP_SYSTEM_INCLUDE_DIR, derive fixed + language subdirectories internally, allow an empty root to disable the + configured system tree, keep AFTER language-independent, extend + #include_next across every class, add full-order regression coverage and + document the wrapper-header/search contract in a dedicated section. + +2026-09-28 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.29: implement GNU-style #include_next. Preserve the + physical include search entry that supplied each nested header and resume + searching strictly after it across user, explicit-system, configured-system + and idirafter classes. #include_next does not re-search the current source + directory and treats quoted/angle operands identically; macro-expanded + operands remain supported. Add dedicated wrapper-header regression tests. + +2026-09-28 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.28: implement #error/#warning diagnostics with logical + source locations, no macro expansion of diagnostic text, whitespace + normalization outside quoted tokens, conditional suppression and the + historical Russian aliases #ошибка/#предупреждение. Add regression + coverage and document the diagnostic-directive contract. + +2026-09-28 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.27: implement zNNN/ZNNN integer-width suffixes with an + optional U/u in #if expressions. Parse NNN as decimal, normalize valid + 8/16/32/64-bit literals by sign or zero extension to the fixed 64-bit + evaluator, reject widths above 64, warn and ignore invalid smaller widths, + reject C L/LL suffixes and malformed suffix continuations. Remove the + obsolete assertion -A action and -dMA modifier. Expand the manual with + the normative conversion model and add dedicated regressions. + +2026-09-28 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.26: replace the temporary hand-written #if parser with + mcpp-expr.zubr generated by ZUBR 4.1.0; retain the current defined()/ + macro-expansion pipeline, add an UCS-2 lexer using LibMPU iatoui() for + numeric conversion, move fixed 64-bit expression semantics into + mcpp-semantic.c/h, preserve short-circuit and historical shift behavior, + and add dedicated regression coverage. + +2026-09-28 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.25: allow `$` as a preprocessing-identifier continuation + character but never as identifier start; parse -D macro declarators + separately from replacement values so invalid suffixes are silently + discarded instead of entering the replacement list; apply matching -U + prefix handling; document shell quoting and add regressions. + +2026-09-28 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.24: fix -dD provenance for command-line macros. Track + builtin/source/command-line macro origin and emit # 0 "<command-line>" + for definitions installed through -D while preserving # 0 "<built-in>" + for predefined/builtin definitions. Add regression coverage. + +2026-09-28 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.23: implement -D/-U command-line macros. Decode only + their payloads from UTF-8 to strict UCS-2, reuse the ordinary #define/ + #undef parser and macro engine, preserve action order after builtin/ + predefined installation, support Unicode XID identifiers and function- + like macros, reject invalid UTF-8/non-UCS-2 payloads, and add dedicated + regression coverage. + +2026-09-28 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.22: replace the ASCII-only preprocessing-identifier + predicates with LibMPUIO 1.0.4 Unicode 18.0.0 XID_Start/XID_Continue + classification. Keep underscore as an explicit language identifier + character, preserve case-sensitive macro names, require the UCS-2 ctype + API at configure time, and add regression coverage for Cyrillic, Latin, + Greek, combining marks, non-ASCII digits, macro parameters, #if/defined, + ## rescanning. + +2026-09-27 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.21: implement GNU-compatible ## token concatenation for + object-like and function-like macros; use raw arguments next to ##, keep + empty arguments as placemarkers, validate the pasted preprocessing token, + rescan the resulting replacement list, preserve # stringification + interaction, diagnose invalid/end-position uses, and add regressions. + +2026-09-27 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.20: keep the quoted #lang implementation contract unchanged, + strengthen its case-insensitive regression coverage, and synchronize every + package-version regression expectation with the release version. + +2026-09-27 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.19: require a quoted #lang language name; validate it + case-insensitively against the internal language table only, reject empty + or whitespace-containing names and multi-physical-line forms, preserve the + original spelling inside quotes, normalize external whitespace in output, + and add dedicated regression coverage. + +2026-09-27 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.18: implement conditional-compilation control flow; + consume #if/#ifdef/#ifndef/#elif/#else/#endif instead of emitting them, + skip inactive branches, evaluate #if expressions with defined and the + historical operator precedence. Restore GNU-compatible line control: + generated line information uses # N "file" linemarkers, include entry and + return use flags 1/2, and input #line updates __LINE__/__FILE__ instead of + passing through to output. Add dedicated regression coverage. + +2026-09-27 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.17: finish comment whitespace cleanup. Remove blanks left + before end-of-line comments, including multi-line block comments, while + preserving the separator required by comments between adjacent tokens. + +2026-09-27 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.16: implement -dD output; keep source #define directives + and emit predefined definitions with built-in markers; remove whitespace + left on comment-only lines; defer the UCS-2 range check until after + comment removal and add dedicated regressions. + +2026-09-25 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.13: implement function-macro stringification (#); + preserve raw arguments for stringification, lazily expand ordinary + argument occurrences, handle empty actual arguments correctly, keep ## + rejected, and add dedicated stringification/dump/error regressions. + +2026-09-25 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.12: correct INT/UINT DECIMAL_DIG semantics by computing + decimal digits from the known language type width, excluding signs and + terminating NUL characters; add exact full-range regression coverage. + +2026-09-25 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.11: publish integer decimal-digit metadata for every + LibMPU integer width and Real decimal/mantissa precision for every Real + width through MPU_REAL_IO_LIMIT; replace FLOAT_WORD_ORDER with explicit + MCPU byte/word-order macros, add maximum-width capability macros, and + rename SIZEOF_PTRDIFF_T to SIZEOF_PTRDIFF. + +2026-09-25 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.10: move configured size/ssize metadata fully into the + MCPU namespace; use char8/char16-native SIZEOF names; add Real exponent + storage size and maximum textual character-count metadata for every + configured Real width through MPU_REAL_IO_LIMIT; extend -dM regressions. + +2026-09-25 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.9: correct ptrdiff to signed int64; add the MCPU ssize + profile; extend TYPE/WIDTH/SIZEOF metadata through NB_I_MAX * 8 for + integers and MPU_REAL_IO_LIMIT for Real/Complex; define Complex WIDTH + as the language parameter while keeping double storage size; add + exhaustive predefined-range regression coverage. + +2026-09-25 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.8: refine the predefined MCPU ABI contract: rename the + preprocessor/version and LibMPU-profile macros, use char8/char16, + cap textual numeric predefines at 256 bits, add LibMPU-derived decimal + precision macros, fix pointer/ptrdiff width at 64 bits, and derive + active-language subdirectories from configured system include roots. + +2026-09-25 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.7: establish the first serious predefined ABI/environment + layer using the LibMPU/LibMPUIO configure model and GNU GCC probes; + add MCPU/fixed-width integer/Real/Complex/type-size/endian predefines, + lazy LibMPU-generated Real limits, and the -dM/-dconfig inspection + options. + +2026-09-25 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.6: implement the neutral predefined-macro mechanism: __FILE__, + __LINE__, __BASE_FILE__, __INCLUDE_LEVEL__, __DATE__, __TIME__ and __VERSION__; + preserve include nesting, fixed startup timestamp and non-rescanned special + expansion semantics. + +2026-09-25 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.5: move project-owned M4 logic to acsite.m4, reserve m4/ + for external macros, document the future explicit `make parser` Bison + workflow, and implement function-like macro definitions/calls and argument + expansion. + +2026-09-25 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.4: implement the first macro-engine layer, add + backslash-newline/comment preprocessing phases, object-like #define/#undef, + recursive object-macro expansion and computed #include; + add doc/mcpu-cpp.md as the implementation-order manual. + +2026-09-25 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.3: make MCPU_CPP_INCLUDE_PATH a common include root + visible in every language state, keep switched-language include paths + as additional higher-priority configured user paths, and move the + configuration hierarchy to /etc/mcpu/mcpu-cpp.conf and + $HOME/.mcpu/mcpu-cpp.conf. + +2026-09-25 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.2: restore LANG_0 as the initial base language state, replace + historical vasm with as, remove the temporary C-specific include path, + define MCPU_CPP_INCLUDE_PATH as the include root of language 0, tighten + the canonical #lang names, and add + the public `make tests` target. + +2026-09-24 Andrey V. Kosteltsev + + * mcpu-cpp 0.0.1: initial implementation based on the architecture and + semantics required by the MCPU project with a UTF-8/UCS-2 text model. diff --git a/Makefile.am b/Makefile.am new file mode 100644 index 0000000..6b4ac3b --- /dev/null +++ b/Makefile.am @@ -0,0 +1,20 @@ + +ACLOCAL_AMFLAGS = -I m4 + +SUBDIRS = src tests doc man etc + +EXTRA_DIST = AUTHORS \ + LICENSE \ + README \ + NEWS \ + ChangeLog \ + acsite.m4 \ + bootstrap \ + m4/README + +.PHONY: tests +tests: all + $(MAKE) $(AM_MAKEFLAGS) -C tests check-TESTS + +distclean-local: + -rm -rf $(top_srcdir)/autom4te.cache @@ -0,0 +1,748 @@ +mcpu-cpp 1.0.2 +=============== + +Implemented LibMPU, LibMPUIO congiguring check using libmpu.m4, libmpuio.m4. + + +mcpu-cpp 1.0.1 +=============== + +Integrate the mcpu-cpp(1) manual pages. + +* Add English man/mcpu-cpp.1 and Russian man/ru/mcpu-cpp.1. +* Install the pages through Automake's standard man directory hierarchy. + +mcpu-cpp 1.0.0 +=============== + +Remove Russian preprocessing-directive aliases and prepare the public interface +for another verification cycle before 1.0.0. + +* Accept only canonical English preprocessing-directive names; Unicode support + for identifiers and ordinary source text is unchanged. +* Remove Russian-alias regression cases without changing the underlying macro, + conditional, include, diagnostic, language-stack, or output engines. +* Retitle manual section 10 as "GNU-compatible features" and summarize the + documented behaviors intentionally aligned with GNU CPP. +* Remove references to the predecessor preprocessor from both normative manuals. + +mcpu-cpp 0.0.50 +=============== + +Refresh and bilingualize the normative manual and add a developer bootstrap. + +* Replace the single doc/mcpu-cpp.md with synchronized Russian and English + manuals: doc/mcpu-cpp-ru.md and doc/mcpu-cpp-en.md. +* Bring the manual up to the 0.0.49 feature set, fix stale future-work text and + conditional-expression subsection numbering, and document the current + intentionally unsupported legacy features. +* Add a root bootstrap script, modeled after the LibMPU/LibMPUIO bootstrap + workflow, that regenerates the ZUBR parser and Autotools-generated files. +* Add a root .gitignore describing files that a developer Git tree may omit; + keep release archives self-contained and the ordinary release build unchanged. + +mcpu-cpp 0.0.49 +=============== + +Compact invisible output lines while preserving exact source coordinates. + +* Track the next visible logical source position instead of materializing every + consumed directive/comment/blank source line in .E output. +* Match GNU CPP's 0..7 newline versus 8+ corrective-linemarker boundary. +* Preserve include entry/return markers and avoid synthetic progress markers + for completely silent included files. +* Keep __LINE__, diagnostics, #line and logical coordinates independent of the + compacted physical output representation. + +mcpu-cpp 0.0.48 +=============== + +Normalize whitespace that belongs to macro replacement lists. + +* Canonicalize replacement-list whitespace to one ASCII space after line + splicing while preserving ordinary source formatting and actual-argument + whitespace. +* Preserve literal contents and preprocessing-token boundaries around #, ##, + __VA_ARGS__, __VA_OPT__ and placemarkers. + +mcpu-cpp 0.0.47 +=============== + +Add the standard __VA_OPT__(pp-tokens) mechanism. + +* Decide variadic emptiness after normal macro substitution. +* Support balanced parentheses, stringification, token concatenation, + placemarkers and rescan integration. +* Intentionally omit GNU named variadics and the historical comma-swallow + extension , ## __VA_ARGS__. + +mcpu-cpp 0.0.46 +=============== + +Add C99-style variadic function-like macros. + +* Accept ... as the final macro parameter and substitute __VA_ARGS__. +* Preserve ordinary prescan, raw stringification and token-paste semantics for + the complete variadic tail, including empty-tail placemarkers. +* Keep the old GNU named form args... and __VA_OPT__ outside this release. +* Preserve variadic definitions in macro dumps and document the exact contract. +* Add dedicated regression coverage for expansion, #, ##, dumps and errors. + +mcpu-cpp 0.0.45 +=============== + +Add GNU-like missing-generated dependency handling. + +* Implement -MG for dependency-only -M/-MM modes. +* Record unresolved includes in the existing ordered dependency registry while + keeping their textual identity separate from physical st_dev/st_ino identity. +* Preserve GNU-like user/system filtering for missing quoted/angle includes and + system-header context; tolerate missing -include/-imacros operands under -MG. +* Keep -MG invalid with -MD/-MMD or without -M/-MM. +* Document the unresolved-dependency model and add dedicated regression coverage. + +mcpu-cpp 0.0.44 +=============== + +Add explicit dependency targets and complete the forced-file documentation. + +* Implement GNU-like -MT TARGET and -MQ TARGET, including attached forms. +* Allow repeated -MT/-MQ options and multiple targets in one dependency rule. +* Keep -MT literal while -MQ applies Make quoting; suppress the default target + whenever an explicit target is present. +* Document -include/-imacros and -MT/-MQ in doc/mcpu-cpp.md. +* Add dedicated dependency-target regression coverage. + +mcpu-cpp 0.0.41 +=============== + +Clean the public interface and formalize selected Russian directive aliases. + +* Remove obsolete compatibility options from the command-line parser so they + are diagnosed as unknown options instead of remaining dormant handlers. +* Keep -E as the intentional compiler-driver compatibility no-op. +* Document --object-suffix in --help and require its argument. +* Add Russian aliases for define, line, include, include_next, undef, pragma + and lang; normalize non-once #управление output to #pragma. +* Rename the product description to "MCPU languages preprocessor" and remove + predecessor-oriented wording from source comments and documentation. +* Add interface-cleanup and Russian-alias regression coverage. + +mcpu-cpp 0.0.40 +=============== + +Add explicit warning-control semantics. + +* Implement -Wcomment/-Wcomments and negative forms for nested block-comment + openers and backslash-newline continuation of // comments. +* Make -Wall enable the optional comment-warning class while preserving the + priority of an explicit -Wno-comment regardless of option order. +* Implement -Werror/-Wno-error for every warning actually emitted by mcpu-cpp, + without making -Werror enable optional warning classes. +* Route #warning, zNNN width warnings, macro redefinition and invalid ## paste + through the common warning policy. +* Add dedicated regression coverage and a normative documentation section. + +mcpu-cpp 0.0.39 +=============== + +Split final macro dumping into ordinary and predefined views. + +* Make -dM dump only source/include and command-line macros. +* Implement -dMP as predefined macros first, followed by ordinary macros. +* Preserve deterministic name sorting inside each group and final #undef state. +* Keep context-dependent special macros out of the static dump. +* Leave -dD semantics unchanged. +* Add dedicated regression coverage and document the macro-origin contract. + +mcpu-cpp 0.0.38 +=============== + +Extend dependency generation with side-effect dependency files. + +* Add GNU-like -MD and -MMD without suppressing normal preprocessing output. +* Add -MF FILE, including attached -MFfile spelling and -MF - for stdout. +* Derive the default .d filename from the input basename, or from ordinary -o + output when present. +* Reuse the existing -M/-MM dependency graph and system-header classification. +* Add dedicated regression coverage and document the dependency-output rules. + +mcpu-cpp 0.0.37 +=============== + +Make the MCPU installation tree relocatable and shared by the future toolchain. + +* Install under `$libdir/mcpu` rather than `$libdir/mcpu-VERSION`. +* Resolve the real executable at run time and derive `<root>/include` and + `<root>/etc/mcpu-cpp.conf` from the physical MCPU root. +* Keep `/etc/mcpu/mcpu-cpp.conf` and `$HOME/.mcpu/mcpu-cpp.conf` as higher + priority overrides, including full replacement of MCPU_CPP_SYSTEM_INCLUDE_PATH. +* Remove the absolute installation default from the packaged config and runtime + binary contract. +* Add a regression that physically moves a complete MCPU tree and verifies the + new root, include search, packaged config and symlink invocation. +* Document one relocatable `bin/etc/include/lib` root as a general MCPU + ecosystem principle for mcpu-cpp, mcpu-as, mcpu-ld and mcpu-run. + +mcpu-cpp 0.0.36 +=============== + +Correct --no-config so it suppresses configuration-file reads without +discarding the compiled installation defaults. + +* Keep the compiled MCPU_CPP_SYSTEM_INCLUDE_PATH active under --no-config. +* Make -dsearch-dirs, -dconfig and -v expose that built-in default even when + every configuration file is ignored. +* Preserve explicit -I/-isystem/-idirafter ordering around the compiled system + tree and keep -nostdinc as the switch that suppresses standard-system paths. +* Add dedicated regression coverage and document the distinction between + compiled defaults, configuration files and -nostdinc. + +mcpu-cpp 0.0.35 +=============== + +Implement the first dependency-generation stage. + +* Add -M make-rule output including system headers. +* Add -MM user-dependency output excluding system headers and descendants + reachable only from system-header contexts. +* Collect dependencies directly from the existing include/#include_next + pipeline and deduplicate physical files by device/inode identity. +* Keep #line logical filenames out of dependency output and preserve a file in + -MM when it is also reached through a user include context. +* Add dedicated dependency-generation regression coverage and documentation. + +mcpu-cpp 0.0.33 +=============== + +Align the configure-time console presentation with the LibMPU/LibMPUIO/ZUBR +package family and normalize the author's given name spelling. + +* Add the shared-style mcpu-cpp configure headline and AC_MSG_CFG_PART section + headings to make configure output follow the same visual structure as the + other MCPU/LibMPU projects. +* Keep the existing detailed configuration summary as its own final section. +* Normalize every textual occurrence of the author given name to `Andrey` + throughout the maintained project tree. + +mcpu-cpp 0.0.32 +=============== + +Polish include-path diagnostics after the 0.0.31 installation/search contract. + +* Make -v print each effective include configuration variable once, after all + configuration layers have been resolved. +* Add -dsearch-dirs to print effective include search directories in semantic + priority order and stop without preprocessing. +* Document the distinction between effective configuration values and the + expanded global search-directory chain. + +mcpu-cpp 0.0.31 +=============== + +Introduce the versioned MCPU installation/configuration contract. + +* Install mcpu-cpp under `$libdir/mcpu-0.0.31/bin` and its packaged + configuration under `$libdir/mcpu-0.0.31/etc`. +* Install `$bindir/mcpu-cpp` as a symbolic link to the versioned executable. +* Do not install or create `/etc/mcpu`; treat `/etc/mcpu/mcpu-cpp.conf` only as + an optional distributor/administrator override. +* Load configuration in increasing priority: compiled versioned defaults, + versioned packaged config, optional `/etc` override, then the unversioned + `$HOME/.mcpu/mcpu-cpp.conf` user override. +* Restore `MCPU_CPP_SYSTEM_INCLUDE_PATH` as the replaceable effective system + include root, defaulting to `$libdir/mcpu-0.0.31/include`. Derive + `<root>/<lang>` internally; no per-language system variables exist. +* Let a higher-priority config replace the system tree completely or disable it + with an empty value. +* Keep the normative include and #include_next search order established in + 0.0.30, now using the effective `MCPU_CPP_SYSTEM_INCLUDE_PATH`. +* Document the versioned tree, configuration precedence, sandbox workflow and + wrapper-header rationale as normative interfaces. + +mcpu-cpp 0.0.30 +=============== + +Define and enforce the normative include-search contract discovered by manual +wrapper-header testing. + +* Give explicit command-line -I, -isystem and -idirafter directories priority + over persistent configuration classes. +* Search configured user language paths before MCPU_CPP_INCLUDE_PATH. +* Replace MCPU_CPP_SYSTEM_INCLUDE_PATH with the singular + MCPU_CPP_SYSTEM_INCLUDE_DIR system-tree root. Derive the fixed <root>/<lang> + directory internally and then search <root>; no per-language system config + variables exist. +* Allow an empty MCPU_CPP_SYSTEM_INCLUDE_DIR to disable the configured system + tree. -nostdinc suppresses the same configured system classes for one run + without suppressing explicit -isystem. +* Keep explicit -idirafter before MCPU_CPP_AFTER_INCLUDE_PATH and never derive + language-specific AFTER directories. +* Make #include_next traverse this exact effective chain by physical provenance. +* Add a dedicated manual-style regression that places the same header in all + eight search classes and verifies the complete #include_next order. +* Add a separate normative documentation section explaining the search order, + configurable user paths, fixed system subdirectory names and wrapper-header + purpose. + +mcpu-cpp 0.0.29 +=============== + +Implement GNU-style #include_next wrapper-header search. + +* Track the exact configured include-directory entry that supplied each nested + header and continue #include_next strictly after that entry. +* Preserve the semantic search order across -I, explicit -isystem, configured + system directories and -idirafter directories. +* Do not re-search the current source-file directory for #include_next; the + directive intentionally does not distinguish between "file" and <file>. +* Support macro-expanded #include_next operands through the ordinary include + operand parser and macro-expansion path. +* Keep physical include provenance independent of logical #line filenames. +* Add regression coverage for the wrapper -> configured-system -> idirafter + chain, quoted-source origin, macro operands, inactive branches and failure + when no later header exists. + +mcpu-cpp 0.0.28 +=============== + +Implement #error and #warning diagnostics. + +* Active #error emits a source-location error diagnostic and stops preprocessing. +* Active #warning emits a source-location warning and preprocessing continues. +* Diagnostic directive arguments are not macro-expanded, following the directive contract; whitespace between tokens is folded while quoted text is kept. +* Diagnostic directives are consumed and never copied to normal or -dD output. +* Inactive conditional branches suppress both directives completely. +* Historical Russian aliases #ошибка and #предупреждение are recognized. + +mcpu-cpp 0.0.27 +=============== + +Define the MCPU integer-width suffix contract for conditional expressions and +remove the obsolete assertion command-line heritage. + +* Accept `zNNN`/`ZNNN` and optional trailing `U`/`u` on integer constants. + NNN is always decimal, including forms with leading zeroes. +* `z8`, `z16`, `z32` and `z64` normalize the low N bits by sign extension; + an optional U/u selects zero extension. All later expression evaluation is + still exactly 64-bit and does not retain the source width. +* Widths above 64 are errors in conditional directives. Invalid widths up to + 64 are ignored with a warning; a trailing U/u remains effective. +* Remove C-style L/LL integer suffix acceptance from the #if lexer. +* Diagnose a suffix that runs into an identifier instead of splitting it into + unrelated tokens. +* Remove the obsolete `-A` assertion option, the assertion action kind and the + `-dMA` assertion-dump modifier. +* Document the complete normalization and 64-bit evaluation model. +* Add regression coverage for width suffixes, sign/zero extension, truncation, + 64-bit post-normalization operations, errors and removed assertion options. + +mcpu-cpp 0.0.26 +=============== + +Use the ZUBR grammar architecture for #if expression parsing. + +* Add mcpp-expr.zubr with the grammar and UCS-2 lexical analyzer; generated + mcpp-expr.c uses the mcpp_zubr_* prefix from ZUBR 4.1.0. +* Keep the established defined() preprocessing and macro-expansion front end. +* Move fixed-width #if arithmetic to mcpp-semantic.c/h; expression integers + use LibMPU 64-bit types rather than host long/intmax_t. +* Parse binary/octal/decimal/hexadecimal integer lexemes through LibMPU + iatoui() after validating/copying the ASCII subset from UCS-2. +* Character units are UCS-2 __mpu_uint16_t values; multi-character + packing is retained up to the 64-bit expression width. +* Preserve the defined precedence, negative-shift direction and short-circuit + behavior for &&, || and ?:. +* Ship the generated C source; ZUBR is needed only when the .zubr grammar is + regenerated. The obsolete code-page -T option is not used. +* Add a dedicated ZUBR expression regression. + +mcpu-cpp 0.0.25 +=============== + +Refine preprocessing identifiers and command-line macro declarator parsing. + +* Permit `$` after the first preprocessing-identifier character, while keeping + `$` forbidden as an identifier start. +* Apply the same `$` continuation rule throughout macro names, parameters, + conditionals, expansion, # stringification and ## rescanning. +* Parse the -D declarator separately from its replacement value so an invalid + tail before '=' is discarded instead of becoming replacement text. +* Apply the same valid-identifier-prefix rule to -U command-line actions. +* Document shell quoting explicitly: single quotes already protect `$`; an + unquoted or double-quoted `$` must be escaped when literal text is intended. +* Extend Unicode/command-line macro regressions for `$`, invalid tails, -U and + forbidden `$` at identifier start. + +mcpu-cpp 0.0.24 +=============== + +Correct -dD origin markers for command-line macro definitions. + +* Macros installed by -D are tagged as command-line definitions internally. +* -dD emits `# 0 "<command-line>"` before command-line definitions instead of + incorrectly reporting them as `# 0 "<built-in>"`. +* Predefined/builtin macros retain the `# 0 "<built-in>"` marker. +* Added a regression covering Unicode -D names and both marker classes. + +mcpu-cpp 0.0.23 +=============== + +Implement command-line macro definition and undefinition. + +* -DNAME defines NAME as 1; -DNAME=VALUE defines VALUE, including an empty + replacement after an explicit '='. +* Function-like command-line macros use the same parser and macro engine as + source #define directives, including # stringification and ## concatenation. +* -D and -U payloads alone are decoded from UTF-8 to strict UCS-2; filenames, + include paths and all other command-line arguments remain byte strings. +* Unicode command-line macro names use the same XID_Start/XID_Continue rules + as source identifiers. Invalid UTF-8 and non-UCS-2 characters are rejected. +* -D/-U actions are applied in command-line order after predefined/builtin + macros are installed, so -U may remove a predefined macro and later -D may + define it again. +* Added dedicated regression coverage for ASCII and Unicode command-line + macros, order, function macros, #/##, conditionals and invalid encodings. + +mcpu-cpp 0.0.22 +=============== + +Use LibMPUIO 1.0.4 strict UCS-2 Unicode 18.0.0 XID classification for +preprocessing identifiers. + +* Identifier start is `_` or Unicode XID_Start. +* Identifier continuation is `_` or Unicode XID_Continue. +* Macro names and parameters may therefore use UCS-2 letters from supported + scripts, combining marks in continuation positions, and non-ASCII decimal + digits in continuation positions. +* The same identifier rules are used by #define/#undef, #ifdef/#ifndef, + defined(), ordinary macro expansion, stringification, token concatenation + rescanning. +* Classification is locale-independent and inherits LibMPUIO strict UCS-2 + semantics; surrogate code units are never identifiers. +* Configure now verifies that the required LibMPUIO UCS-2 ctype API is present. + +mcpu-cpp 0.0.21 +=============== + +Implement GNU-compatible token concatenation (##). + +* ## now pastes preprocessing tokens in both object-like and function-like + macro replacement lists. +* A formal parameter adjacent to ## uses its raw, unexpanded actual argument; + the completed replacement list is rescanned afterwards. This preserves the + standard two-level CAT/XCAT expansion technique. +* Empty pasted arguments use placemarker semantics, and multi-token arguments + paste only the token adjacent to ##. +* Pasting can form identifiers, preprocessing numbers, multi-character + punctuators and prefixed string/character tokens. +* An invalid paste is diagnosed and the original two tokens are emitted; ## at + either end of a replacement list is rejected when the macro is defined. +* Stringification (#) and concatenation (##) may be used in the same macro. +* Added dedicated token-concatenation regression coverage and updated macro + error tests. + +mcpu-cpp 0.0.20 +=============== + +Corrective release of the quoted #lang contract introduced in 0.0.19. + +* The #lang implementation contract is unchanged: exactly one quoted language + name, no whitespace inside the quotes, no escape processing, same-physical-line + termination, ASCII case-insensitive validation against the internal language + table only, original spelling preserved, and normalized external whitespace. +* Regression coverage explicitly checks the accepted spellings "diff", "Diff", + "DIFF" and "dIfF", as well as all other internal languages. +* All package-version regression expectations are synchronized with 0.0.20; + this fixes the stale-version test failures seen in the first 0.0.19 test run. + +mcpu-cpp 0.0.19 +=============== + +Tighten and normalize the #lang directive contract. + +* #lang now requires a quoted language name, for example `#lang "diff"`; + the unquoted form is rejected. +* The quoted text must be one non-empty word without whitespace, must close on + the same physical source line, and may be followed only by whitespace. +* Escape sequences are not interpreted inside the #lang string. +* Language lookup is ASCII case-insensitive but is restricted to the internal + language table: diff, dift, alg, as, avm and ACS. +* Output preserves the spelling inside the quotes while normalizing external + whitespace to exactly `#lang "name"`. +* Updated existing language tests and added a dedicated #lang string-contract + regression. + +mcpu-cpp 0.0.18 +=============== + +Implement conditional-compilation control flow. + +* #if, #ifdef, #ifndef, #elif, #else and #endif are consumed by the + preprocessor and never copied to normal or -dD output. +* Inactive branches are skipped without executing definitions, includes or + ordinary macro expansion; nested conditionals remain balanced. +* #if expressions follow the expr.zubr precedence model, including + defined, unknown identifiers as zero, integer operators, ?: and short-circuit + evaluation. +* Include files keep independent conditional-stack boundaries and diagnose + unbalanced or unterminated groups. +* Generated source-location information now uses GNU linemarkers (`# N "file"`) + instead of emitting `#line`; include entry/return markers carry flags 1 and 2. +* Input #line directives are consumed, macro-expanded and update __LINE__ and + __FILE__; the resulting location is emitted as a GNU linemarker. +* Added regression coverage for include guards, nested groups, all conditional + directives, defined, macro-expanded expressions, malformed groups and line + control. + +mcpu-cpp 0.0.17 +=============== + +Finish comment whitespace cleanup. + +* Removing a // or /* ... */ comment at the end of a non-empty source line no + longer leaves a trailing blank before the newline. +* The same rule applies when a block comment begins after program text and + continues onto following physical lines. +* Inline comments between tokens still preserve a separating blank where it is + required to prevent accidental token concatenation. + +mcpu-cpp 0.0.16 +=============== + +Implement -dD and correct comment/UCS-2 phase behavior. + +* -dD preserves normal preprocessed output, emits predefined macro definitions + with <built-in> markers, and keeps encountered #define directives. +* Comment-only lines are left truly empty after // or /* ... */ removal while + inline comments still preserve token separation. +* Valid UTF-8 scalars outside UCS-2 are accepted inside comments and rejected + only if they remain in program text after comment removal. +* Added regression coverage for -dD, comment-only whitespace and non-UCS-2 + characters inside versus outside comments. + +mcpu-cpp 0.0.13 +=============== + +Implement macro stringification (#). + +* Function-like macro replacement lists now support #parameter stringification. +* Stringification uses the raw, unexpanded actual argument, trims leading and + trailing whitespace, folds internal whitespace outside quoted tokens, and + escapes quotes/backslashes in the generated string literal. +* Ordinary argument expansion is now lazy, matching the established raw-vs-expanded + argument semantics and preserving nested calls of the same macro. +* Empty actual arguments are represented correctly; zero-argument macros keep + their established behavior. +* Invalid # uses are diagnosed at definition time. ## remains deliberately + rejected for the next separate token-concatenation port. +* Added regression coverage for raw-vs-expanded stringification, whitespace, + quoted text, empty arguments, diff apostrophe semantics, diagnostics and -dM. + +mcpu-cpp 0.0.12 +=============== + +Correct integer DECIMAL_DIG semantics. + +* Replaced direct use of LibMPU _int_digs() for predefined integer precision + metadata with mcpu-cpp integer-only helpers based on the known type width. +* __INT<bits>_DECIMAL_DIG__ now counts decimal digits of INT<bits>_MAX only; + it excludes both the sign and any terminating NUL. +* __UINT<bits>_DECIMAL_DIG__ now counts decimal digits of UINT<bits>_MAX only; + it excludes any terminating NUL. +* Added exact regression values for every integer family width from 8 through + 65536 bits. + +mcpu-cpp 0.0.11 +=============== + +Complete predefined precision metadata and MCPU byte/word-order naming. + +* Added __INT<bits>_DECIMAL_DIG__ and __UINT<bits>_DECIMAL_DIG__ from LibMPU + _int_digs() for every configured integer width through NB_I_MAX * 8; MAX + values remain capped at 256 bits. +* Added __REAL<bits>_DECIMAL_DIG__ and __REAL<bits>_MANT_DIG__ for every Real + width through MPU_REAL_IO_LIMIT; large numeric MAX/MIN/EPSILON/exponent + values remain capped at 256 bits. +* Replaced __FLOAT_WORD_ORDER__ with __MCPU_BYTE_ORDER__ and + __MCPU_WORD_ORDER__; __BYTE_ORDER__ remains as an alias to + __MCPU_BYTE_ORDER__. +* Added __MCPU_INT_MAX_WIDTH__, __MCPU_REAL_MAX_WIDTH__ and + __MCPU_COMPLEX_MAX_WIDTH__ capability macros. +* Renamed __SIZEOF_PTRDIFF_T__ to language-native __SIZEOF_PTRDIFF__. +* Extended predefined ABI/range regression coverage for the complete integer + and Real families. + +mcpu-cpp 0.0.10 +=============== + +Predefined ABI naming cleanup and complete Real text-buffer metadata. + +* Replaced the C-style public size family with __MCPU_SIZE_TYPE__, WIDTH, + SIZEOF_SIZE and MAX; renamed __MCPU_SIZEOF_SSIZE_T__ to + __MCPU_SIZEOF_SSIZE__. +* Renamed __SIZEOF_CHAR8_T__/__SIZEOF_CHAR16_T__ to the language-native + __SIZEOF_CHAR8__/__SIZEOF_CHAR16__. +* Added __SIZEOF_REAL<bits>_EXP__ from LibMPU _sizeof_exp() for every Real + width through MPU_REAL_IO_LIMIT. +* Added __REAL<bits>_MAX_STRLEN__ from _real_max_string() for every Real width; + the value is a character count, not a byte count. +* Extended -dM regression coverage to verify the new names and the largest + configured Real family. + +mcpu-cpp 0.0.9 +============== + +Complete structural predefined metadata across LibMPU type ranges. + +* Corrected MCPU ptrdiff to signed int64 with max 0x7fffffffffffffff. +* Added the MCPU-specific signed-size family: __MCPU_SSIZE_TYPE__, WIDTH, + SIZEOF and MAX from the configured LibMPU __mpu_ssize_t profile. +* Integer TYPE/WIDTH/SIZEOF families now extend through NB_I_MAX * 8; + numeric MAX/DECIMAL_DIG values remain capped at 256 bits. +* Real/Complex TYPE/WIDTH/SIZEOF families now extend through + MPU_REAL_IO_LIMIT; numeric Real characteristics remain capped at 256 bits. +* Corrected __COMPLEX<bits>_WIDTH__ to the language type parameter <bits>; + __SIZEOF_COMPLEX<bits>__ remains twice the corresponding Real storage. +* Added exhaustive -dM regression coverage across all advertised integer, + Real and Complex structural families. + +mcpu-cpp 0.0.8 +============== + +Predefined ABI refinement and derived system language include paths. + +* Renamed __VERSION__ to __MCPU_CPP_VERSION__ and report the mcpu-cpp version. +* Renamed LibMPU-profile predefines to the __MCPU_* namespace. +* Character type spellings are now char8 and char16. +* MCPU pointer and ptrdiff contracts are fixed at 64 bits independently of host; + ptrdiff is currently uint64 by ABI decision. +* Integer/Real textual value constants are emitted only through 256 bits. + Wider families retain only type and width macros. +* Removed computed integer MIN predefines. +* Added integer DECIMAL_DIG values from _int_digs(). +* Replaced REAL<bits>_DIG with REAL<bits>_DECIMAL_DIG from _real_digs(); + MANT_DIG comes from _real_mant_digs(). +* Configured system include roots automatically add active-language + subdirectories such as /usr/include/mcpu/diff without extra config variables. +* Added regression coverage for the refined predefined environment and derived + system language include search. + +mcpu-cpp 0.0.7 +============== + +Predefined ABI/environment layer and inspection tools. + +* GNU GCC is now an explicit mandatory build compiler. +* Ported the LibMPU/LibMPUIO GCC type-size, endian and machine-register probes + into mcpu-cpp acsite.m4 under project-specific macro names. +* Configure reads and verifies the installed LibMPU real-I/O/math limits, + byte/word order, addressable-unit width, size_t and ptrdiff_t profile. +* Added _ARCH_MCPU and GCC-compatible byte-order, pointer, size_t/ptrdiff_t and + assembler-prefix predefined macros. +* Added fixed-width int/uint predefined families through 65536 bits. +* Added Real/Complex type families through 65536 bits; Real numeric limits are + generated with LibMPU and real_to_ascii up to MPU_REAL_IO_LIMIT. +* Large Real numeric predefined values are generated lazily so ordinary + preprocessing remains fast. +* __VERSION__ is now the future high-level compiler version placeholder 0.0.0. +* Added `-dM` for deterministic macro-table dumps and `-dconfig` for effective + configuration dumps; neither requires an input file. +* Added ABI, dump-macro and dump-config regression tests. + +mcpu-cpp 0.0.6 +============== + +Implement predefined macros. + +* Added the dynamic predefined macros __FILE__, __LINE__, __BASE_FILE__, + __INCLUDE_LEVEL__, __DATE__, __TIME__ and __VERSION__. +* __FILE__/__LINE__/__INCLUDE_LEVEL__ follow nested #include processing; + __BASE_FILE__ remains the original translation-unit input file. +* __DATE__ and __TIME__ are fixed once at preprocessing startup, matching the + translation-unit timestamp model. +* __VERSION__ expands to the mcpu-cpp package version, as preprocessor used its own + package version for the corresponding macro. +* Predefined expansions are emitted without rescanning, following the established + special-symbol behavior. +* Built-ins share the normal macro table and can therefore be undefined or + explicitly redefined. +* K-specific legacy built-ins (__STDK__, __K__, __kxLab__, and related names) + are not carried into the MCPU language family. + +mcpu-cpp 0.0.5 +============== + +Extend the macro engine. + +* Moved project-owned Autoconf macros from m4/ to the root acsite.m4; m4/ is + reserved for external/vendor macros. +* Documented the future developer-only Bison workflow: make parser, make tests, + then make dist. Normal release builds do not require Bison. +* Added function-like #define macros with zero or more formal parameters. +* Added nested argument parsing with defined parenthesis/comma rules. +* Actual arguments are macro-expanded before ordinary substitution, including + nested calls such as min(min(a,b),c). +* Added diagnostics for duplicate parameters, malformed definitions, incomplete + calls and wrong argument counts. +* # and ## are explicitly rejected until their semantics are implemented, avoiding silent partial preprocessing. + +mcpu-cpp 0.0.4 +============== + +Implement the first macro-processing layer. + +* Added doc/mcpu-cpp.md as a living implementation-order manual as the normative MCPU preprocessing manual. +* Added the preprocessing phase needed before directive and macro processing: + backslash-newline splicing and comment removal with preserved source lines. +* Added object-like #define and #undef. +* Added recursive/cascaded object-macro expansion with recursion blocking. +* Macro names are not expanded inside quoted strings; the historical diff + apostrophe rule remains in force. +* Added computed #include through macro expansion. +* Preserved literal #include <...> lexical behavior for comment-like text. +* Began reusing macro regression cases; the self-referential macro + case is now part of the mcpu-cpp test suite. + +mcpu-cpp 0.0.3 +============== + +Include-root and configuration-layout correction. + +* MCPU_CPP_INCLUDE_PATH is now a common user include root visible in every + language state. +* Active language-specific include directories are additional search paths + and precede the common configured root. +* Explicit names such as <diff/a.h> can therefore be used through the common + root regardless of the current #lang state. +* The system configuration moved to /etc/mcpu/mcpu-cpp.conf. +* The per-user configuration moved to $HOME/.mcpu/mcpu-cpp.conf. + +mcpu-cpp 0.0.2 +============== + +Language-contract cleanup before macro-engine development. + +* Establish initial language state 0. +* Language 0 is active at startup and cannot be named by #lang. +* Canonical #lang set is: diff, dift, alg, as, avm, ACS. +* Historical vasm was replaced by as for the real MCPU assembler. +* MCPU_CPP_VASM_INCLUDE_PATH became MCPU_CPP_AS_INCLUDE_PATH. +* Removed the temporary MCPU_CPP_C_INCLUDE_PATH. The root + MCPU_CPP_INCLUDE_PATH is the main include directory of the unnamed base + base MCPU language (state 0) and is selected only in that state. +* Added the public `make tests` target. + +mcpu-cpp 0.0.1 +============== + +Initial architectural release. + +* New UTF-8 external / UCS-2 internal source model with strict NUL rejection. +* No legacy code-page machinery. +* Hand-written preprocessing scanner; flex is not used. +* Scanner honors quoted strings and the diff apostrophe rule. +* Source/include stack with #include support. +* #lang/#endlang language stack defined for the MCPU language stack. +* Configuration-file and command-line include search paths. +* Initial language set: c, diff, dift, alg, vasm, avm, ACS. @@ -0,0 +1,480 @@ +mcpu-cpp +======== + +mcpu-cpp is the preprocessor/front-end source manager for MCPU programming +languages. It is an independent component of the LibMPU/LibMPUIO/LibMCPU +software platform and provides preprocessing, language selection, include-path +management, dependency generation and preprocessing diagnostics for the MCPU +toolchain. + +Text model +---------- + +External text files are UTF-8. Internally mcpu-cpp uses strict UCS-2 through +LibMPUIO. UTF-8 scalar values which cannot be represented by UCS-2 are +rejected. Legacy external code pages are not supported; the external representation is UTF-8. + +Lexical analysis +---------------- + +mcpu-cpp does not use flex. Its preprocessing scanners are hand-written and +operate on the internal UCS-2 text model. The `#if` expression grammar and its +expression lexer are generated by ZUBR 4.1.0. + +Parser architecture +------------------- + +The generated `src/mcpp-expr.c` is shipped in release archives, so ordinary +builds do not require ZUBR. In the developer Git tree the generated parser +may be omitted; the root `./bootstrap` script regenerates it from +`src/mcpp-expr.zubr` with ZUBR 4.1.0 before regenerating the Autotools files. + +Documentation +------------- + +The normative preprocessing manual is maintained in two synchronized forms: + + doc/mcpu-cpp-en.md English + doc/mcpu-cpp-ru.md Russian + +Both documents describe the same public contract and are distributed together. + +Configuration +------------- + +mcpu-cpp reads UTF-8 configuration files using a simple `NAME = value;` +syntax. Shell-style `$NAME` and `${NAME}` references are expanded from values +already defined in the effective configuration and then from the process +environment. + +MCPU uses one relocatable installation tree rather than a versioned directory +per tool. With `--prefix=/usr --libdir=/usr/lib64`, `make install` creates: + +```text +/usr/lib64/mcpu/ + bin/mcpu-cpp + etc/mcpu-cpp.conf + include/ + lib/ +``` + +`/usr/bin/mcpu-cpp` is a public symbolic link to +`../lib64/mcpu/bin/mcpu-cpp`. The absolute `/usr/lib64/mcpu` path is an +install-time choice only; it is not embedded as the runtime MCPU root. + +On Linux, mcpu-cpp resolves its real executable through `/proc/self/exe`, takes +the parent of the executable directory as the MCPU runtime root, and derives +`<root>/etc/mcpu-cpp.conf` and `<root>/include` from it. A fallback based on +`argv[0]`, `PATH`, and `realpath(3)` is used only when `/proc/self/exe` cannot +be read. Consequently a complete MCPU tree may be copied or moved without +rebuilding mcpu-cpp. + +This is the intended ecosystem-wide layout for future `mcpu-as`, `mcpu-ld`, +`mcpu-run`, libraries and CRT components as well: tool versions do not define +separate roots; a coherent MCPU environment is identified by one physical +runtime tree. + +Configuration layers are applied in this order: + +```text +runtime-derived defaults +<runtime-root>/etc/mcpu-cpp.conf +/etc/mcpu/mcpu-cpp.conf optional +$HOME/.mcpu/mcpu-cpp.conf optional, highest priority +``` + +The packaged config deliberately does not store an absolute default system +include path. Before reading any config file mcpu-cpp sets +`MCPU_CPP_SYSTEM_INCLUDE_PATH=<runtime-root>/include`; any higher-priority +configuration may replace that value or set it empty. + +`--config-file FILE` reads only FILE on top of the runtime-derived defaults. +`--no-config` reads no configuration files at all but keeps those runtime +defaults. Use `-nostdinc` when the effective standard-system include tree +itself must be suppressed for one invocation. + +Installation does not create `/etc/mcpu`; that directory is reserved for an +optional distributor or system-administrator override. The per-user +configuration is deliberately not versioned. + +Recognized path variables are: + + MCPU_CPP_INCLUDE_PATH + MCPU_CPP_DIFF_INCLUDE_PATH + MCPU_CPP_DIFT_INCLUDE_PATH + MCPU_CPP_ALG_INCLUDE_PATH + MCPU_CPP_AS_INCLUDE_PATH + MCPU_CPP_AVM_INCLUDE_PATH + MCPU_CPP_ACS_INCLUDE_PATH + MCPU_CPP_SYSTEM_INCLUDE_PATH + MCPU_CPP_AFTER_INCLUDE_PATH + +User and AFTER path lists use the host PATH separator (`:` on UNIX systems). +`MCPU_CPP_SYSTEM_INCLUDE_PATH` is different: it is one replaceable root of the +MCPU system-header tree. Its runtime-derived default is +`<runtime-root>/include`. When a switched language is active mcpu-cpp +searches `<root>/<lang>` and then `<root>`. A higher-priority configuration can +replace the root completely for a developer/tester sandbox, or set it to an +empty value to disable the configured system tree. + +Command line +------------ + + mcpu-cpp [options] [input [output]] + +Important options: + + -o FILE write output to FILE instead of stdout + -D NAME[=VALUE] define a command-line macro + -U NAME undefine a command-line macro + -imacros FILE preprocess FILE for macro state; discard its output + -include FILE preprocess FILE before the primary input + -I DIR, -IDIR add a user include directory + -isystem DIR add an explicit system include directory + -idirafter DIR add a directory searched after system directories + -nostdinc suppress the effective standard-system include tree + -dM dump non-predefined macros to stdout + -dMP dump predefined macros first, then other macros + -dD preserve #define directives in normal output + -dconfig dump effective configuration variables to stdout + -dsearch-dirs dump effective include search directories and exit + -M output make dependencies including system headers + -MM output make dependencies excluding system headers + -MD write dependencies and keep preprocessing output + -MMD like -MD but exclude system headers + -MF FILE write dependencies to FILE ('-' means stdout) + -MT TARGET set unquoted make dependency target + -MQ TARGET set make-quoted dependency target + --object-suffix SFX set object suffix used for dependency targets + -w suppress all warnings + -Wcomment[s] warn about nested /* and multi-line // comments + -Wno-comment[s] disable comment warnings even under -Wall + -Wall enable all optional warning classes + -Werror promote every emitted warning to an error + -Wno-error keep emitted warnings as warnings + --config-file FILE use only FILE as the configuration file + --no-config do not read config files; keep runtime-derived defaults + -v, --verbose print configuration and include activity + --help print help + --version print version + +Include search +-------------- + +For `#include "file"`, the physical directory containing the current source +file is searched first. For `#include <file>`, that first step is omitted. +The remaining include search order is normative: + +```text +explicit -I +explicit -isystem +MCPU_CPP_<LANG>_INCLUDE_PATH +MCPU_CPP_INCLUDE_PATH +MCPU_CPP_SYSTEM_INCLUDE_PATH/<lang> +MCPU_CPP_SYSTEM_INCLUDE_PATH +explicit -idirafter +MCPU_CPP_AFTER_INCLUDE_PATH +``` + +Command-line include directories therefore override persistent configuration. +The language-specific user paths are freely configurable. The system path is +a single effective root; mcpu-cpp derives the fixed `<root>/<lang>` directory +itself, so there are intentionally no +`MCPU_CPP_SYSTEM_<LANG>_INCLUDE_PATH` variables. Replacing +`MCPU_CPP_SYSTEM_INCLUDE_PATH` replaces the complete installed system-header +tree rather than adding another directory. An empty effective value disables +the configured system tree. Neither `-idirafter` nor +`MCPU_CPP_AFTER_INCLUDE_PATH` acquires automatic language subdirectories. + +`#include_next` is intended for wrapper headers. It remembers the exact +physical entry of this effective search chain that supplied the current header +and resumes at the following entry. This allows an explicit wrapper to alter +policy and then continue into a sandbox or installed system tree without +copying the original header or hard-coding its absolute pathname. The +`"file"` and `<file>` forms are equivalent for `#include_next`, and logical +names established by `#line` do not affect physical search provenance. + +`#pragma once` marks the current physical file as processed for the remainder +of the preprocessing run. Identity is the filesystem device/inode pair, not +the pathname spelling, so the same file reached through another relative name, +a symbolic link or a hard link is skipped. The exact active directive is +consumed by mcpu-cpp; an inactive `#pragma once` has no effect. Other pragmas +remain in output for later compiler stages. + +Command-line forced files +------------------------- + +`-imacros FILE` and `-include FILE` use one deterministic preprocessing +pipeline. Predefined macros are installed first, all command-line `-D`/`-U` +actions are then applied in their own command-line order, every `-imacros` +file is processed in command-line order, every `-include` file is processed in +command-line order, and only then does preprocessing enter the primary input. +The relative placement of `-imacros` and `-include` options in argv therefore +does not interleave the two classes: every `-imacros` always precedes every +`-include`. + +An `-imacros` file goes through the ordinary preprocessing machinery, including +`#include`, macro definition/undefinition, conditional directives, `#lang`, +`#pragma once`, diagnostics and dependency tracking. Its normal preprocessing +output, including output from headers reached from that file, is discarded. +The resulting macro table and other preprocessing state remain available to +subsequent forced files and to the primary input. `-include` uses the same +preprocessing machinery but retains its normal output, as if the header had +been included immediately before the primary source. + +The operand of either forced-file option has GNU-style command-line search +semantics. An absolute path is used directly. A relative path is searched +first in the current working directory and then through the ordinary quoted +include chain shown above. The physical directory of the primary input does +not receive the special first priority that a literal `#include "file"` inside +that source would receive. Once a forced file has been found, quoted includes +inside it are resolved normally relative to that file's physical directory, +and `#include_next` retains the search-chain provenance of the entry that found +it. + +Forced files are ordinary physical dependencies. Files reached through CWD or +user include classes remain user dependencies; files found through `-isystem`, +the configured system tree, `-idirafter` or configured AFTER paths retain +system dependency class for `-MM`/`-MMD`. + + +Dependency generation +--------------------- + +`-M` preprocesses the translation unit and writes one make rule instead of +normal preprocessor output. The rule contains the main source file and every +physical header actually reached through the ordinary `#include` / +`#include_next` pipeline, including system headers. Repeated spellings, +symbolic links and hard links to the same physical file are recorded once. +Logical names established by `#line` are never dependency names. + +`-MM` uses the same traversal but omits headers found through `-isystem`, the +configured MCPU system tree, `-idirafter` and configured AFTER paths, and also +omits headers reachable only as descendants of such a system header. Quote +versus angle include spelling does not decide whether a dependency is system. +If the same physical file is also reached directly through a user include +context, it remains a user dependency. + +The default target is the input basename with its suffix replaced by the +configured object suffix (normally `.o`). In dependency-only `-M`/`-MM` mode, +`-MF FILE` selects the make-rule destination; without `-MF`, the existing +mcpu-cpp `-o FILE` destination remains available. + +`-MD` and `-MMD` generate dependencies as a side effect and do **not** suppress +normal preprocessing output. `-MD` includes system headers like `-M`; `-MMD` +applies the same user-only filter as `-MM`. Without `-MF`, the dependency file +is `<input-basename>.d` in the current directory, or is derived from ordinary +`-o` output by replacing its suffix with `.d`. `-MF FILE` overrides that +automatic name, and `-MF -` sends the dependency rule to stdout. + +Options `-MT` and `-MQ` are intentionally left for the next dependency stage. + +#lang contract +-------------- + +The initial language state is `0`. It is the base MCPU language state and is +installed before preprocessing begins. `0` is deliberately not a parameter of +`#lang`. + +`#lang` requires exactly one quoted language name: + + #lang "diff" + +The text inside the quotes is not escape-decoded. It must be a non-empty +single word without whitespace, and the closing quote must occur on the same +physical source line. After the closing quote only whitespace is allowed. +The accepted names come only from the internal language table: `diff`, `dift`, +`alg`, `as`, `avm`, and `ACS`. Matching is ASCII case-insensitive, so for +example `"diff"`, `"Diff"`, `"DIFF"` and `"dIfF"` all select the same +language. Unsupported names such as `"0"`, `"c"` and `"vasm"` are errors. + +External whitespace is normalized when `#lang` is copied to output. For +example: + + # lang "DiFf" + +is emitted as: + + #lang "DiFf" + +The spelling inside the quotes is preserved as written. + +`#lang` and `#endlang` form one translation-unit-wide stack. The stack is not +reset at an `#include` boundary. Therefore a `#lang` in one file and its +matching `#endlang` in another included file are intentionally legal, exactly +as required by the macro-expansion contract. + +Both directives are passed through to output so the later language dispatcher +and parser layer can observe the same language boundaries. `#lang` is emitted +in the normalized form described above. + +Macro operators +--------------- + +Function-macro stringification (`#`) is supported. +The operator uses the raw, unexpanded actual argument, removes leading and +trailing whitespace, folds internal whitespace outside quoted tokens to one +space, and escapes double quotes and backslashes as required by the resulting +quoted string. `##` token concatenation is also supported and rescans the concatenated token +through the ordinary macro-expansion path. + +The completed predefined ABI/environment layer from 0.0.12 is preserved. +INT/UINT DECIMAL_DIG counts only decimal digits of the corresponding numeric +maximum, without a sign or terminating NUL. Integer decimal precision and +Real DECIMAL_DIG/MANT_DIG metadata cover every configured type width; +large MAX/MIN/EPSILON textual values remain capped at 256 bits. + +`__MCPU_CPP_VERSION__` is the mcpu-cpp package version. `_ARCH_MCPU` identifies +the target. The LibMPU profile macros use the MCPU namespace: +`__MCPU_MACHINE_REGISTER_WIDTH__`, `__MCPU_REAL_IO_LIMIT__` and +`__MCPU_MATH_FN_LIMIT__`. + +MCPU pointers are always 64 bits independently of the host. `intptr` and +`ptrdiff` are signed `int64`; `uintptr` is `uint64`: + + __INTPTR_TYPE__ int64 + __INTPTR_WIDTH__ 64 + __INTPTR_MAX__ 0x7fffffffffffffff + __UINTPTR_TYPE__ uint64 + __UINTPTR_WIDTH__ 64 + __UINTPTR_MAX__ 0xffffffffffffffff + __PTRDIFF_TYPE__ int64 + __PTRDIFF_WIDTH__ 64 + __PTRDIFF_MAX__ 0x7fffffffffffffff + +The configured LibMPU size and signed-size types are exposed only in the MCPU +namespace. For a 64-bit profile the public contract is: + + __MCPU_SIZE_TYPE__ uint64 + __MCPU_SIZE_WIDTH__ 64 + __MCPU_SIZEOF_SIZE__ 8 + __MCPU_SIZE_MAX__ 0xffffffffffffffff + __MCPU_SSIZE_TYPE__ int64 + __MCPU_SSIZE_WIDTH__ 64 + __MCPU_SIZEOF_SSIZE__ 8 + __MCPU_SSIZE_MAX__ 0x7fffffffffffffff + +Integer TYPE/WIDTH/SIZEOF metadata is generated for every power-of-two LibMPU +integer family through `NB_I_MAX * 8`. Real and Complex TYPE/WIDTH/SIZEOF +metadata is generated through the configured `MPU_REAL_IO_LIMIT`. Complex +WIDTH is the language type parameter: `complex128` has WIDTH 128 but SIZEOF 32 +bytes, because it stores two real128 components. + +Every available Real family also exports two compact conversion/layout +properties through `MPU_REAL_IO_LIMIT`: `__SIZEOF_REAL<bits>_EXP__` comes from +LibMPU `_sizeof_exp()`, while `__REAL<bits>_MAX_STRLEN__` comes from +`_real_max_string()`. MAX_STRLEN is a number of characters, not bytes; a +zero-terminated `char8` or `char16` buffer therefore needs at least +`MAX_STRLEN + 1` elements. + +Large numeric textual values remain deliberately capped at 256 bits. Integer +MAX and Real MAX/MIN/EPSILON/exponent-value macros are not emitted above that +width. Integer DECIMAL_DIG and Real DECIMAL_DIG/MANT_DIG remain available for +the complete configured families. Integer MIN expressions are never +predefined. This keeps `-dMP` compact while preserving useful precision, +structural and text-buffer metadata for large LibMPU types. + +The future language uses fixed-width names. Character types are `char8` and +`char16`; their sizes are `__SIZEOF_CHAR8__` and `__SIZEOF_CHAR16__`. Ordinary +C `char`, `short`, `int`, `long`, `wchar_t`, and the C-style `char*_t` names are +not part of the target language type model. + +The effective `MCPU_CPP_SYSTEM_INCLUDE_PATH` automatically contributes a +language-specific subdirectory for each active switched language. For example, +with the runtime-derived root and `#lang "diff"`, +`<runtime-root>/include/diff` is searched before +`<runtime-root>/include`. These derived directories need not exist and +require no extra configuration variables. + +`-dM` dumps only the current non-predefined macro table in deterministic +`#define` form. `-dMP` first dumps active predefined macros and then the +non-predefined macros; each group is sorted by name. +`-dconfig` dumps the effective configuration variables in sorted `NAME = value;` +form. `-v` prints the include-related effective configuration values once, +after all configuration layers have been resolved, and then preserves the usual +include/language runtime trace. `-dsearch-dirs` prints `search: DIR` lines for +the effective global include directories in semantic priority order and exits +without preprocessing. The dynamic source-directory step used by quoted +includes is not part of that global dump. None of these dump actions requires +an input file. + +The dynamic source macros `__FILE__`, `__LINE__`, `__BASE_FILE__`, +`__INCLUDE_LEVEL__`, `__DATE__` and `__TIME__` use one translation-unit +source stack and timestamp. Stringification, concatenation and conditional +compilation are part of the current preprocessing contract. + +Parser generation +----------------- + +The #if expression parser is generated by ZUBR 4.1.0 from +`src/mcpp-expr.zubr`. The grammar also contains the UCS-2 lexical analyzer. +The generated source `src/mcpp-expr.c` is included in release archives, so a +normal build from a release archive does not require ZUBR. + +The conditional-expression evaluator is deliberately 64-bit only. Integer +literals may use `U`/`u`, or the MCPU width suffix `zNNN[Uu]` / `ZNNN[Uu]`. +For valid widths up to 64, the low N bits are taken and then sign-extended +(`zNNN`) or zero-extended (`zNNNu`) to 64 bits; all later operations remain +64-bit and the original width is forgotten. Widths above 64 are rejected in +conditional directives. Invalid widths not exceeding 64 produce a warning +and the width suffix is ignored. C `L`/`LL` integer suffixes are not accepted. + +`#error` and `#warning` are implemented as diagnostic directives. Their +arguments are not macro-expanded. Outside quoted tokens, whitespace sequences +are folded to one space for the diagnostic text. `#error` stops preprocessing; +`#warning` continues. Both directives disappear from normal and `-dD` output, +and both are ignored in inactive conditional branches. + +Warning control follows a compact GNU-like model. `-Wcomment` and `-Wcomments` +enable warnings for `/*` inside an existing block comment and for a +backslash-newline continuing a `//` comment; `-Wall` currently enables this +optional warning class. The specific `-Wno-comment`/`-Wno-comments` setting +overrides `-Wall` regardless of command-line order. `-w` globally suppresses +all warnings, including `#warning`, invalid `zNNN` widths, macro redefinition, +invalid `##` paste and enabled comment diagnostics. It does not suppress +errors. `-Werror` promotes every warning that is actually emitted to an error +and unsuccessful preprocessing; because `-w` prevents warning emission first, +`-w -Werror` and `-Werror -w` are equivalent and successful when no independent +error occurs. Likewise, warning classes may be enabled by `-Wcomment` or +`-Wall`, but remain silent under `-w` regardless of option order. `-Wno-error` +restores ordinary warning severity. `-Werror` does not by itself enable +optional warning classes. + +When `mcpp-expr.zubr` is changed, the ordinary Automake `.zubr.c` rule +regenerates the C source with: + + zubr -vl -s -Bmcpp_ -o mcpp-expr.c mcpp-expr.zubr + + +Developer bootstrap +------------------- + +A Git checkout may omit files that are regenerated mechanically, including +`configure`, `Makefile.in`, Automake helper scripts, `config.h.in`, `aclocal.m4` +and `src/mcpp-expr.c`. Regenerate them with: + +```text +./bootstrap +``` + +`./bootstrap --target-dest-dir=DIR` follows the LibMPU/LibMPUIO convention for +using Autoconf macro/header directories from a target ROOTFS. Release archives +remain self-contained and do not require bootstrap before `configure`. + +Build +----- + +mcpu-cpp uses Autoconf/Automake and obtains LibMPUIO compilation and linker +flags from `mpuio-config`, following the conventions of LibMPU and LibMPUIO. +No pkg-config or flex dependency is introduced. The 1.0.2 release is +developed and tested against LibMPU 1.0.25 and LibMPUIO 1.0.4. The LibMPUIO +UCS-2 ctype API is required. ZUBR 4.1.0 is the parser-regeneration tool; it is +not required for a normal build from a release archive containing the generated +`src/mcpp-expr.c`. + +A normal build is: + + ./configure --prefix=/usr --libdir=/usr/lib64 + make + make tests + make install diff --git a/acsite.m4 b/acsite.m4 new file mode 100644 index 0000000..8b4c011 --- /dev/null +++ b/acsite.m4 @@ -0,0 +1,797 @@ +dnl ============================================================ +dnl Force configure to run under /bin/bash +dnl ============================================================ +m4_define([AC_MCPU_CPP_REQUIRE_BASH], [dnl +m4_divert_push([M4SH-SANITIZE])dnl + +if test ! -x /bin/bash; then + echo "configure: error: /bin/bash is required" >&2 + exit 1 +fi + +CONFIG_SHELL=/bin/bash +export CONFIG_SHELL + +m4_divert_pop([M4SH-SANITIZE])dnl +])dnl + + +dnl ============================================================ +dnl Support for Configuration Headers +dnl +dnl configure.ac: +dnl AC_MCPU_CPP_HEADLINE(<short-name>, <long-name>, +dnl <vers-var>, <copyright>) +dnl +dnl NOTE: +dnl ==== +dnl See the paragraph 8.3.3 of Autoconf Documentation at: +dnl +dnl https://www.gnu.org/software/autoconf/manual/autoconf.html#Diversion-support +dnl ============================================================ +m4_define([AC_MCPU_CPP_HEADLINE], [dnl +m4_divert_push([M4SH-INIT])dnl +{ + if test ".`echo dummy [$]@ | grep help`" = .; then + + ####### нахождение escape последовательностей для + ####### обозначения начала и конца выделяемого текста + ####### Use a `Quadrigaph'. '@<:@' gives you [ and '@:>@' gives you ] : + TB=`echo -n -e '\033@<:@1m'` + TN=`echo -n -e '\033@<:@0m'` + + ####### получение короткого номера версии продукта + ####### из AC_INIT() + $3="AC_PACKAGE_VERSION" + AC_SUBST($3) + + ####### печать заголовка + echo "Configuring:" + echo "" + echo "${TB}$1${TN} ($2), Version ${TB}${$3}${TN}" + echo "$4" + echo "" + fi +} +m4_divert_pop([M4SH-INIT])dnl +])dnl + + +dnl ============================================================ +dnl Display Configuration Headers +dnl +dnl configure.ac: +dnl AC_MSG_CFG_PART(<text>) +dnl ============================================================ +AC_DEFUN([AC_MSG_CFG_PART],[dnl + AC_MSG_RESULT() + AC_MSG_RESULT([${TB}$1:${TN}]) +])dnl + + +dnl ============================================================ +dnl GCC is a required part of the mcpu-cpp build environment. +dnl ============================================================ +AC_DEFUN([MCPU_CPP_REQUIRE_GCC], [dnl + AC_MSG_CHECKING([whether $CC is GNU GCC]) + AC_COMPILE_IFELSE([ + AC_LANG_SOURCE([[ +#if !defined(__GNUC__) || defined(__clang__) || defined(__INTEL_COMPILER) || defined(__INTEL_LLVM_COMPILER) +#error mcpu-cpp requires GNU GCC +#endif +int main( void ) { return 0; } + ]]) + ], [ + AC_MSG_RESULT([yes]) + ], [ + AC_MSG_RESULT([no]) + AC_MSG_ERROR([mcpu-cpp requires GNU GCC]) + ]) +]) +dnl ============================================================ +dnl Test for WIDTH of MACHINE REGISTER using GCC predefines: +dnl ================== +dnl +dnl configure.ac: +dnl MCPU_CPP_GCC_REGISTER_WIDTH +dnl +dnl ============================================================ +AC_DEFUN([MCPU_CPP_GCC_REGISTER_WIDTH], +[dnl +AC_MSG_CHECKING(for CPP predefined macro __INT_FAST32_WIDTH__ ) +AC_CACHE_VAL(ac_cv_int_fast32_width, +[ac_cv_int_fast32_width=`$CC -dM -E - < /dev/null | grep __INT_FAST32_WIDTH__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_int_fast32_width" != ""; then + AC_MSG_RESULT($ac_cv_int_fast32_width) +else + AC_MSG_RESULT(not defined) +fi + +AC_MSG_CHECKING(for CPP predefined macro __INT_FAST64_WIDTH__ ) +AC_CACHE_VAL(ac_cv_int_fast64_width, +[ac_cv_int_fast64_width=`$CC -dM -E - < /dev/null | grep __INT_FAST64_WIDTH__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_int_fast64_width" != ""; then + AC_MSG_RESULT($ac_cv_int_fast64_width) +else + AC_MSG_RESULT(not defined) +fi + +AC_MSG_CHECKING(for MACHINE REGISTER WIDTH ) +AC_CACHE_VAL(ac_cv_machine_register_width, +[dnl +if test "$ac_cv_int_fast64_width" != "" -a "$ac_cv_int_fast32_width" != "" ; then + if test "$ac_cv_int_fast64_width" -gt "$ac_cv_int_fast32_width" ; then + ac_cv_machine_register_width=$ac_cv_int_fast32_width + else + ac_cv_machine_register_width=$ac_cv_int_fast64_width + fi +elif test "$ac_cv_int_fast64_width" != "" ; then + ac_cv_machine_register_width=$ac_cv_int_fast64_width +elif test "$ac_cv_int_fast32_width" != "" ; then + ac_cv_machine_register_width=$ac_cv_int_fast32_width +else + ac_cv_machine_register_width=32 +fi +])dnl +if test "$ac_cv_machine_register_width" != ""; then + MACHINE_REGISTER_WIDTH=$ac_cv_machine_register_width + AC_MSG_RESULT($ac_cv_machine_register_width) +else + MACHINE_REGISTER_WIDTH= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(MACHINE_REGISTER_WIDTH)dnl +AC_DEFINE_UNQUOTED([MACHINE_REGISTER_WIDTH], [$ac_cv_machine_register_width], [The size of Machine Register in bits.])dnl +]) + + +dnl ============================================================ +dnl Test for GCC Types: +dnl ================== +dnl +dnl configure.ac: +dnl MCPU_CPP_GCC_TYPES +dnl +dnl ============================================================ +AC_DEFUN([MCPU_CPP_GCC_TYPES], +[dnl +AC_MSG_CHECKING(for CPP predefined macro __CHAR16_TYPE__ ) +AC_CACHE_VAL(ac_cv_char16_type, +[ac_cv_char16_type=`$CC -dM -E - < /dev/null | grep __CHAR16_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_char16_type" != ""; then + GCC_CHAR16_TYPE=$ac_cv_char16_type + AC_MSG_RESULT($ac_cv_char16_type) +else + GCC_CHAR16_TYPE= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_CHAR16_TYPE)dnl + +AC_MSG_CHECKING(for CPP predefined macro __CHAR32_TYPE__ ) +AC_CACHE_VAL(ac_cv_char32_type, +[ac_cv_char32_type=`$CC -dM -E - < /dev/null | grep __CHAR32_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_char32_type" != ""; then + GCC_CHAR32_TYPE=$ac_cv_char32_type + AC_MSG_RESULT($ac_cv_char32_type) +else + GCC_CHAR32_TYPE= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_CHAR32_TYPE)dnl + +AC_MSG_CHECKING(for CPP predefined macro __WCHAR_TYPE__ ) +AC_CACHE_VAL(ac_cv_wchar_type, +[ac_cv_wchar_type=`$CC -dM -E - < /dev/null | grep __WCHAR_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_wchar_type" != ""; then + GCC_WCHAR_TYPE=$ac_cv_wchar_type + AC_MSG_RESULT($ac_cv_wchar_type) +else + GCC_WCHAR_TYPE= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_WCHAR_TYPE)dnl + +AC_MSG_CHECKING(for CPP predefined macro __INT8_TYPE__ ) +AC_CACHE_VAL(ac_cv_int8_type, +[ac_cv_int8_type=`$CC -dM -E - < /dev/null | grep __INT8_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_int8_type" != ""; then + GCC_INT8_TYPE=$ac_cv_int8_type + AC_MSG_RESULT($ac_cv_int8_type) +else + GCC_INT8_TYPE= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_INT8_TYPE)dnl + +AC_MSG_CHECKING(for CPP predefined macro __UINT8_TYPE__ ) +AC_CACHE_VAL(ac_cv_uint8_type, +[ac_cv_uint8_type=`$CC -dM -E - < /dev/null | grep __UINT8_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_uint8_type" != ""; then + GCC_UINT8_TYPE=$ac_cv_uint8_type + AC_MSG_RESULT($ac_cv_uint8_type) +else + GCC_UINT8_TYPE= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_UINT8_TYPE)dnl + +AC_MSG_CHECKING(for CPP predefined macro __INT16_TYPE__ ) +AC_CACHE_VAL(ac_cv_int16_type, +[ac_cv_int16_type=`$CC -dM -E - < /dev/null | grep __INT16_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_int16_type" != ""; then + GCC_INT16_TYPE=$ac_cv_int16_type + AC_MSG_RESULT($ac_cv_int16_type) +else + GCC_INT16_TYPE= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_INT16_TYPE)dnl + +AC_MSG_CHECKING(for CPP predefined macro __UINT16_TYPE__ ) +AC_CACHE_VAL(ac_cv_uint16_type, +[ac_cv_uint16_type=`$CC -dM -E - < /dev/null | grep __UINT16_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_uint16_type" != ""; then + GCC_UINT16_TYPE=$ac_cv_uint16_type + AC_MSG_RESULT($ac_cv_uint16_type) +else + GCC_UINT16_TYPE= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_UINT16_TYPE)dnl + +AC_MSG_CHECKING(for CPP predefined macro __INT32_TYPE__ ) +AC_CACHE_VAL(ac_cv_int32_type, +[ac_cv_int32_type=`$CC -dM -E - < /dev/null | grep __INT32_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_int32_type" != ""; then + GCC_INT32_TYPE=$ac_cv_int32_type + AC_MSG_RESULT($ac_cv_int32_type) +else + GCC_INT32_TYPE= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_INT32_TYPE)dnl + +AC_MSG_CHECKING(for CPP predefined macro __UINT32_TYPE__ ) +AC_CACHE_VAL(ac_cv_uint32_type, +[ac_cv_uint32_type=`$CC -dM -E - < /dev/null | grep __UINT32_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +AC_CACHE_VAL(ac_cv_uint32_const_suffix, +[ac_cv_uint32_const_suffix=`$CC -dM -E - < /dev/null | grep __UINT32_C | cut -f5- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_uint32_type" != ""; then + GCC_UINT32_TYPE=$ac_cv_uint32_type + AC_MSG_RESULT($ac_cv_uint32_type) +else + GCC_UINT32_TYPE= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_UINT32_TYPE)dnl + +AC_MSG_CHECKING(for CPP predefined macro __INT64_TYPE__ ) +AC_CACHE_VAL(ac_cv_int64_type, +[ac_cv_int64_type=`$CC -dM -E - < /dev/null | grep __INT64_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_int64_type" != ""; then + GCC_INT64_TYPE=$ac_cv_int64_type + AC_MSG_RESULT($ac_cv_int64_type) +else + GCC_INT64_TYPE= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_INT64_TYPE)dnl + +AC_MSG_CHECKING(for CPP predefined macro __UINT64_TYPE__ ) +AC_CACHE_VAL(ac_cv_uint64_type, +[ac_cv_uint64_type=`$CC -dM -E - < /dev/null | grep __UINT64_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +AC_CACHE_VAL(ac_cv_uint64_const_suffix, +[ac_cv_uint64_const_suffix=`$CC -dM -E - < /dev/null | grep __UINT64_C | cut -f5- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_uint64_type" != ""; then + GCC_UINT64_TYPE=$ac_cv_uint64_type + AC_MSG_RESULT($ac_cv_uint64_type) +else + GCC_UINT64_TYPE= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_UINT64_TYPE)dnl + +AC_MSG_CHECKING(for CPP predefined macro __WINT_TYPE__ ) +AC_CACHE_VAL(ac_cv_wint_type, +[ac_cv_wint_type=`$CC -dM -E - < /dev/null | grep __WINT_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_wint_type" != ""; then + GCC_WINT_TYPE=$ac_cv_wint_type + AC_MSG_RESULT($ac_cv_wint_type) +else + GCC_WINT_TYPE= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_WINT_TYPE)dnl + +AC_MSG_CHECKING(for CPP predefined macro __INTMAX_TYPE__ ) +AC_CACHE_VAL(ac_cv_intmax_type, +[ac_cv_intmax_type=`$CC -dM -E - < /dev/null | grep __INTMAX_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_intmax_type" != ""; then + GCC_INTMAX_TYPE=$ac_cv_intmax_type + AC_MSG_RESULT($ac_cv_intmax_type) +else + GCC_INTMAX_TYPE= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_INTMAX_TYPE)dnl + +AC_MSG_CHECKING(for CPP predefined macro __UINTMAX_TYPE__ ) +AC_CACHE_VAL(ac_cv_uintmax_type, +[ac_cv_uintmax_type=`$CC -dM -E - < /dev/null | grep __UINTMAX_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_uintmax_type" != ""; then + GCC_UINTMAX_TYPE=$ac_cv_uintmax_type + AC_MSG_RESULT($ac_cv_uintmax_type) +else + GCC_UINTMAX_TYPE= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_UINTMAX_TYPE)dnl + +AC_MSG_CHECKING(for CPP predefined macro __SIZE_TYPE__ ) +AC_CACHE_VAL(ac_cv_size_type, +[ac_cv_size_type=`$CC -dM -E - < /dev/null | grep __SIZE_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_size_type" != ""; then + GCC_SIZE_TYPE=$ac_cv_size_type + AC_MSG_RESULT($ac_cv_size_type) +else + GCC_SIZE_TYPE= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_SIZE_TYPE)dnl + +AC_MSG_CHECKING(for CPP predefined macro __INTPTR_TYPE__ ) +AC_CACHE_VAL(ac_cv_intptr_type, +[ac_cv_intptr_type=`$CC -dM -E - < /dev/null | grep __INTPTR_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_intptr_type" != ""; then + GCC_INTPTR_TYPE=$ac_cv_intptr_type + AC_MSG_RESULT($ac_cv_intptr_type) +else + GCC_INTPTR_TYPE= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_INTPTR_TYPE)dnl + +AC_MSG_CHECKING(for CPP predefined macro __UINTPTR_TYPE__ ) +AC_CACHE_VAL(ac_cv_uintptr_type, +[ac_cv_uintptr_type=`$CC -dM -E - < /dev/null | grep __UINTPTR_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_uintptr_type" != ""; then + GCC_UINTPTR_TYPE=$ac_cv_uintptr_type + AC_MSG_RESULT($ac_cv_uintptr_type) +else + GCC_UINTPTR_TYPE= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_UINTPTR_TYPE)dnl + +AC_MSG_CHECKING(for CPP predefined macro __PTRDIFF_TYPE__ ) +AC_CACHE_VAL(ac_cv_ptrdiff_type, +[ac_cv_ptrdiff_type=`$CC -dM -E - < /dev/null | grep __PTRDIFF_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_ptrdiff_type" != ""; then + GCC_PTRDIFF_TYPE=$ac_cv_ptrdiff_type + AC_MSG_RESULT($ac_cv_ptrdiff_type) +else + GCC_PTRDIFF_TYPE= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_PTRDIFF_TYPE)dnl + +AC_MSG_CHECKING(for CPP predefined macro __SIG_ATOMIC_TYPE__ ) +AC_CACHE_VAL(ac_cv_sig_atomic_type, +[ac_cv_sig_atomic_type=`$CC -dM -E - < /dev/null | grep __SIG_ATOMIC_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_sig_atomic_type" != ""; then + GCC_SIG_ATOMIC_TYPE=$ac_cv_sig_atomic_type + AC_MSG_RESULT($ac_cv_sig_atomic_type) +else + GCC_SIG_ATOMIC_TYPE= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_SIG_ATOMIC_TYPE)dnl +]) + + +dnl ============================================================ +dnl Test for GCC Sizeof Types: +dnl ========================= +dnl +dnl configure.ac: +dnl MCPU_CPP_GCC_SIZEOF_TYPES +dnl +dnl ============================================================ +AC_DEFUN([MCPU_CPP_GCC_SIZEOF_TYPES], +[dnl +AC_MSG_CHECKING(for CPP predefined macro __SIZEOF_WCHAR_T__ ) +AC_CACHE_VAL(ac_cv_sizeof_wchar_t, +[ac_cv_sizeof_wchar_t=`$CC -dM -E - < /dev/null | grep __SIZEOF_WCHAR_T__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_sizeof_wchar_t" != ""; then + GCC_SIZEOF_WCHAR_T=$ac_cv_sizeof_wchar_t + AC_MSG_RESULT($ac_cv_sizeof_wchar_t) +else + GCC_SIZEOF_WCHAR_T= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_SIZEOF_WCHAR_T)dnl + +AC_MSG_CHECKING(for CPP predefined macro __SIZEOF_SHORT__ ) +AC_CACHE_VAL(ac_cv_sizeof_short, +[ac_cv_sizeof_short=`$CC -dM -E - < /dev/null | grep __SIZEOF_SHORT__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_sizeof_short" != ""; then + GCC_SIZEOF_SHORT=$ac_cv_sizeof_short + AC_MSG_RESULT($ac_cv_sizeof_short) +else + GCC_SIZEOF_SHORT= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_SIZEOF_SHORT)dnl + +AC_MSG_CHECKING(for CPP predefined macro __SIZEOF_INT__ ) +AC_CACHE_VAL(ac_cv_sizeof_int, +[ac_cv_sizeof_int=`$CC -dM -E - < /dev/null | grep __SIZEOF_INT__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_sizeof_int" != ""; then + GCC_SIZEOF_INT=$ac_cv_sizeof_int + AC_MSG_RESULT($ac_cv_sizeof_int) +else + GCC_SIZEOF_INT= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_SIZEOF_INT)dnl + +AC_MSG_CHECKING(for CPP predefined macro __SIZEOF_WINT_T__ ) +AC_CACHE_VAL(ac_cv_sizeof_wint_t, +[ac_cv_sizeof_wint_t=`$CC -dM -E - < /dev/null | grep __SIZEOF_WINT_T__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_sizeof_wint_t" != ""; then + GCC_SIZEOF_WINT_T=$ac_cv_sizeof_wint_t + AC_MSG_RESULT($ac_cv_sizeof_wint_t) +else + GCC_SIZEOF_WINT_T= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_SIZEOF_WINT_T)dnl + +AC_MSG_CHECKING(for CPP predefined macro __SIZEOF_LONG__ ) +AC_CACHE_VAL(ac_cv_sizeof_long, +[ac_cv_sizeof_long=`$CC -dM -E - < /dev/null | grep __SIZEOF_LONG__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_sizeof_long" != ""; then + GCC_SIZEOF_LONG=$ac_cv_sizeof_long + AC_MSG_RESULT($ac_cv_sizeof_long) +else + GCC_SIZEOF_LONG= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_SIZEOF_LONG)dnl + +AC_MSG_CHECKING(for CPP predefined macro __SIZEOF_LONG_LONG__ ) +AC_CACHE_VAL(ac_cv_sizeof_long_long, +[ac_cv_sizeof_long_long=`$CC -dM -E - < /dev/null | grep __SIZEOF_LONG_LONG__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_sizeof_long_long" != ""; then + GCC_SIZEOF_LONG_LONG=$ac_cv_sizeof_long_long + AC_MSG_RESULT($ac_cv_sizeof_long_long) +else + GCC_SIZEOF_LONG_LONG= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_SIZEOF_LONG_LONG)dnl + +AC_MSG_CHECKING(for CPP predefined macro __SIZEOF_FLOAT__ ) +AC_CACHE_VAL(ac_cv_sizeof_float, +[ac_cv_sizeof_float=`$CC -dM -E - < /dev/null | grep __SIZEOF_FLOAT__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_sizeof_float" != ""; then + GCC_SIZEOF_FLOAT=$ac_cv_sizeof_float + AC_MSG_RESULT($ac_cv_sizeof_float) +else + GCC_SIZEOF_FLOAT= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_SIZEOF_FLOAT)dnl + +AC_MSG_CHECKING(for CPP predefined macro __SIZEOF_DOUBLE__ ) +AC_CACHE_VAL(ac_cv_sizeof_double, +[ac_cv_sizeof_double=`$CC -dM -E - < /dev/null | grep __SIZEOF_DOUBLE__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_sizeof_double" != ""; then + GCC_SIZEOF_DOUBLE=$ac_cv_sizeof_double + AC_MSG_RESULT($ac_cv_sizeof_double) +else + GCC_SIZEOF_DOUBLE= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_SIZEOF_DOUBLE)dnl + +AC_MSG_CHECKING(for CPP predefined macro __SIZEOF_LONG_DOUBLE__ ) +AC_CACHE_VAL(ac_cv_sizeof_long_double, +[ac_cv_sizeof_long_double=`$CC -dM -E - < /dev/null | grep __SIZEOF_LONG_DOUBLE__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_sizeof_long_double" != ""; then + GCC_SIZEOF_LONG_DOUBLE=$ac_cv_sizeof_long_double + AC_MSG_RESULT($ac_cv_sizeof_long_double) +else + GCC_SIZEOF_LONG_DOUBLE= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_SIZEOF_LONG_DOUBLE)dnl + +AC_MSG_CHECKING(for CPP predefined macro __SIZEOF_SIZE_T__ ) +AC_CACHE_VAL(ac_cv_sizeof_size_t, +[ac_cv_sizeof_size_t=`$CC -dM -E - < /dev/null | grep __SIZEOF_SIZE_T__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_sizeof_size_t" != ""; then + GCC_SIZEOF_SIZE_T=$ac_cv_sizeof_size_t + AC_MSG_RESULT($ac_cv_sizeof_size_t) +else + GCC_SIZEOF_SIZE_T= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_SIZEOF_SIZE_T)dnl + +AC_MSG_CHECKING(for CPP predefined macro __SIZEOF_POINTER__ ) +AC_CACHE_VAL(ac_cv_sizeof_pointer, +[ac_cv_sizeof_pointer=`$CC -dM -E - < /dev/null | grep __SIZEOF_POINTER__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_sizeof_pointer" != ""; then + GCC_SIZEOF_POINTER=$ac_cv_sizeof_pointer + AC_MSG_RESULT($ac_cv_sizeof_pointer) +else + GCC_SIZEOF_POINTER= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_SIZEOF_POINTER)dnl + +AC_MSG_CHECKING(for CPP predefined macro __SIZEOF_PTRDIFF_T__ ) +AC_CACHE_VAL(ac_cv_sizeof_ptrdiff_t, +[ac_cv_sizeof_ptrdiff_t=`$CC -dM -E - < /dev/null | grep __SIZEOF_PTRDIFF_T__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_sizeof_ptrdiff_t" != ""; then + GCC_SIZEOF_PTRDIFF_T=$ac_cv_sizeof_ptrdiff_t + AC_MSG_RESULT($ac_cv_sizeof_ptrdiff_t) +else + GCC_SIZEOF_PTRDIFF_T= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_SIZEOF_PTRDIFF_T)dnl +]) + + +dnl ============================================================ +dnl Test for GCC Width of Types: +dnl =========================== +dnl +dnl configure.ac: +dnl MCPU_CPP_GCC_WIDTH_OF_TYPES +dnl +dnl ============================================================ +AC_DEFUN([MCPU_CPP_GCC_WIDTH_OF_TYPES], +[dnl +AC_MSG_CHECKING(for CPP predefined macro __CHAR_BIT__ ) +AC_CACHE_VAL(ac_cv_char_width, +[ac_cv_char_width=`$CC -dM -E - < /dev/null | grep __CHAR_BIT__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_char_width" != ""; then + GCC_CHAR_WIDTH=$ac_cv_char_width + AC_MSG_RESULT($ac_cv_char_width) +else + GCC_CHAR_WIDTH= + AC_MSG_RESULT(not defined) +fi +AC_SUBST(GCC_CHAR_WIDTH)dnl +]) + + +dnl ============================================================ +dnl Test for GCC Byte Order: +dnl ======================= +dnl +dnl configure.ac: +dnl MCPU_CPP_GCC_BYTE_ORDER +dnl +dnl ============================================================ +AC_DEFUN([MCPU_CPP_GCC_BYTE_ORDER], +[dnl +AC_MSG_CHECKING(for CPP predefined macro __BYTE_ORDER__ ) +AC_CACHE_VAL(ac_cv_order_little_endian, +[ac_cv_order_little_endian=`$CC -dM -E - < /dev/null | grep __ORDER_LITTLE_ENDIAN__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +AC_CACHE_VAL(ac_cv_order_big_endian, +[ac_cv_order_big_endian=`$CC -dM -E - < /dev/null | grep __ORDER_BIG_ENDIAN__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +AC_CACHE_VAL(ac_cv_byte_order, +[ac_cv_byte_order=`$CC -dM -E - < /dev/null | grep __BYTE_ORDER__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_byte_order" = "__ORDER_LITTLE_ENDIAN__"; then + GCC_BYTE_ORDER=1234 + GCC_BYTE_ORDER_LITTLE_ENDIAN=1 + GCC_BYTE_ORDER_BIG_ENDIAN=0 + AC_MSG_RESULT(Little-endian) +else + GCC_BYTE_ORDER=4321 + GCC_BYTE_ORDER_LITTLE_ENDIAN=0 + GCC_BYTE_ORDER_BIG_ENDIAN=1 + AC_MSG_RESULT(Big-endian) +fi +AC_SUBST(GCC_BYTE_ORDER)dnl +AC_SUBST(GCC_BYTE_ORDER_LITTLE_ENDIAN)dnl +AC_SUBST(GCC_BYTE_ORDER_BIG_ENDIAN)dnl +]) + + +dnl ============================================================ +dnl Test for GCC Float Word Order: +dnl ============================= +dnl +dnl configure.ac: +dnl MCPU_CPP_GCC_FLOAT_WORD_ORDER +dnl +dnl ============================================================ +AC_DEFUN([MCPU_CPP_GCC_FLOAT_WORD_ORDER], +[dnl +AC_MSG_CHECKING(for CPP predefined macro __FLOAT_WORD_ORDER__ ) +AC_CACHE_VAL(ac_cv_order_little_endian, +[ac_cv_order_little_endian=`$CC -dM -E - < /dev/null | grep __ORDER_LITTLE_ENDIAN__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +AC_CACHE_VAL(ac_cv_order_big_endian, +[ac_cv_order_big_endian=`$CC -dM -E - < /dev/null | grep __ORDER_BIG_ENDIAN__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +AC_CACHE_VAL(ac_cv_float_word_order, +[ac_cv_float_word_order=`$CC -dM -E - < /dev/null | grep __FLOAT_WORD_ORDER__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl +if test "$ac_cv_float_word_order" = "__ORDER_LITTLE_ENDIAN__"; then + GCC_FLOAT_WORD_ORDER=1234 + GCC_FLOAT_WORD_ORDER_LITTLE_ENDIAN=1 + GCC_FLOAT_WORD_ORDER_BIG_ENDIAN=0 + AC_MSG_RESULT(Little-endian) +else + GCC_FLOAT_WORD_ORDER=4321 + GCC_FLOAT_WORD_ORDER_LITTLE_ENDIAN=0 + GCC_FLOAT_WORD_ORDER_BIG_ENDIAN=1 + AC_MSG_RESULT(Big-endian) +fi +AC_SUBST(GCC_FLOAT_WORD_ORDER)dnl +AC_SUBST(GCC_FLOAT_WORD_ORDER_LITTLE_ENDIAN)dnl +AC_SUBST(GCC_FLOAT_WORD_ORDER_BIG_ENDIAN)dnl +]) + + +dnl ======================================================================= +dnl AC_GCC_PROGRAMMING_MODEL +dnl +dnl Determine programming model corresponding to following list: +dnl +dnl PROGRAMMING MODELS: +dnl ------------------ +dnl +dnl PM: | 1 | 2 | 3 | 4 | 5 | 6 +dnl -----------+-------+-------+-----------+-------+-----------+-------- +dnl | LP32 | ILP32 | LP64 | ILP64 | LLP64 | SILP64 +dnl Data type | | | (I32LP64) | | (IL32P64) | +dnl ===========|=======|=======|===========|=======|===========|======== +dnl char | 8 | 8 | 8 | 8 | 8 | 8 +dnl -----------+-------+-------+-----------+-------+-----------+-------- +dnl short | 16 | 16 | 16 | 16 | 16 | 64 +dnl -----------+-------+-------+-----------+-------+-----------+-------- +dnl int | 16 | 32 | 32 | 64 | 32 | 64 +dnl -----------+-------+-------+-----------+-------+-----------+-------- +dnl long | 32 | 32 | 64 | 64 | 32 | 64 +dnl -----------+-------+-------+-----------+-------+-----------+-------- +dnl long long | 64 +dnl -----------+-------+-------+-----------+-------+-----------+-------- +dnl pointer | 32 | 32 | 64 | 64 | 64 | 64 +dnl -----------+-------+-------+-----------+-------+-----------+-------- +dnl ptrdiff_t | 32 | 32 | 64 | 64 | 64 | 64 +dnl ===========+=======+=======+===========+=======+===========+======== +dnl CPU: | | i686 | x86_64 | | | +dnl +dnl Many 64-bit platforms today use an LP64 model (including Solaris, +dnl AIX, HP-UX, Linux, macOS, BSD, and IBM z/OS). Microsoft Windows +dnl uses an LLP64 model. The disadvantage of the LP64 model is that +dnl storing a long into an int may overflow. On the other hand, +dnl converting a pointer to a long will “work” in LP64. In the LLP64 +dnl model, the reverse is true. These are not problems which affect +dnl fully standard-compliant code, but code is often written with +dnl implicit assumptions about the widths of data types. C code +dnl should prefer (u)intptr_t instead of long when casting pointers +dnl into integer objects. +dnl + + +dnl ============================================================ +dnl Read the ABI constants exported by the configured LibMPU. +dnl mpuio-config supplies the include path and LibMPUIO includes +dnl <libmpu.h>, so the values below are the exact constants used +dnl by the installed LibMPU rather than independent guesses. +dnl ============================================================ +AC_DEFUN([MCPU_CPP_CHECK_LIBMPU_ABI], [dnl + AC_REQUIRE([MCPU_CPP_GCC_REGISTER_WIDTH])dnl + AC_REQUIRE([MCPU_CPP_GCC_TYPES])dnl + AC_REQUIRE([MCPU_CPP_GCC_SIZEOF_TYPES])dnl + AC_REQUIRE([MCPU_CPP_GCC_WIDTH_OF_TYPES])dnl + AC_REQUIRE([MCPU_CPP_GCC_BYTE_ORDER])dnl + + mcpu_cpp_save_CPPFLAGS="$CPPFLAGS" + mcpu_cpp_save_CFLAGS="$CFLAGS" + CFLAGS="$CFLAGS $MPUIO_CFLAGS" + + AC_MSG_CHECKING([LibMPU real I/O limit]) + AC_COMPUTE_INT([MCPU_CPP_MPU_REAL_IO_LIMIT], [MPU_REAL_IO_LIMIT], + [[#include <libmpuio.h>]], + [AC_MSG_ERROR([cannot determine MPU_REAL_IO_LIMIT])]) + AC_MSG_RESULT([$MCPU_CPP_MPU_REAL_IO_LIMIT]) + + AC_MSG_CHECKING([LibMPU math-function limit]) + AC_COMPUTE_INT([MCPU_CPP_MPU_MATH_FN_LIMIT], [MPU_MATH_FN_LIMIT], + [[#include <libmpuio.h>]], + [AC_MSG_ERROR([cannot determine MPU_MATH_FN_LIMIT])]) + AC_MSG_RESULT([$MCPU_CPP_MPU_MATH_FN_LIMIT]) + + AC_MSG_CHECKING([LibMPU integer maximum width]) + AC_COMPUTE_INT([MCPU_CPP_MPU_INT_MAX_WIDTH], [NB_I_MAX * BITS_PER_UNIT_T], + [[#include <libmpuio.h>]], + [AC_MSG_ERROR([cannot determine NB_I_MAX * BITS_PER_UNIT_T])]) + AC_MSG_RESULT([$MCPU_CPP_MPU_INT_MAX_WIDTH]) + + AC_MSG_CHECKING([LibMPU byte order]) + AC_COMPUTE_INT([MCPU_CPP_MPU_BYTE_ORDER], [MPU_BYTE_ORDER], + [[#include <libmpuio.h>]], + [AC_MSG_ERROR([cannot determine MPU_BYTE_ORDER])]) + AC_MSG_RESULT([$MCPU_CPP_MPU_BYTE_ORDER]) + + AC_MSG_CHECKING([LibMPU word order]) + AC_COMPUTE_INT([MCPU_CPP_MPU_WORD_ORDER], [MPU_WORD_ORDER], + [[#include <libmpuio.h>]], + [AC_MSG_ERROR([cannot determine MPU_WORD_ORDER])]) + AC_MSG_RESULT([$MCPU_CPP_MPU_WORD_ORDER]) + + AC_MSG_CHECKING([LibMPU machine-register width]) + AC_COMPUTE_INT([MCPU_CPP_MPU_REGISTER_WIDTH], [BITS_PER_MACHINE_REGISTER], + [[#include <libmpuio.h>]], + [AC_MSG_ERROR([cannot determine BITS_PER_MACHINE_REGISTER])]) + AC_MSG_RESULT([$MCPU_CPP_MPU_REGISTER_WIDTH]) + + AC_MSG_CHECKING([LibMPU addressable-unit width]) + AC_COMPUTE_INT([MCPU_CPP_MPU_UNIT_WIDTH], [BITS_PER_UNIT_T], + [[#include <libmpuio.h>]], + [AC_MSG_ERROR([cannot determine BITS_PER_UNIT_T])]) + AC_MSG_RESULT([$MCPU_CPP_MPU_UNIT_WIDTH]) + + AC_MSG_CHECKING([sizeof(__mpu_size_t)]) + AC_COMPUTE_INT([MCPU_CPP_SIZEOF_SIZE_T], [sizeof(__mpu_size_t)], + [[#include <libmpuio.h>]], + [AC_MSG_ERROR([cannot determine sizeof(__mpu_size_t)])]) + AC_MSG_RESULT([$MCPU_CPP_SIZEOF_SIZE_T]) + + AC_MSG_CHECKING([sizeof(__mpu_ssize_t)]) + AC_COMPUTE_INT([MCPU_CPP_SIZEOF_SSIZE_T], [sizeof(__mpu_ssize_t)], + [[#include <libmpuio.h>]], + [AC_MSG_ERROR([cannot determine sizeof(__mpu_ssize_t)])]) + AC_MSG_RESULT([$MCPU_CPP_SIZEOF_SSIZE_T]) + + CPPFLAGS="$mcpu_cpp_save_CPPFLAGS" + CFLAGS="$mcpu_cpp_save_CFLAGS" + + MCPU_CPP_SIZE_WIDTH=`expr "$MCPU_CPP_SIZEOF_SIZE_T" \* "$MCPU_CPP_MPU_UNIT_WIDTH"` + MCPU_CPP_SSIZE_WIDTH=`expr "$MCPU_CPP_SIZEOF_SSIZE_T" \* "$MCPU_CPP_MPU_UNIT_WIDTH"` + + case "$MCPU_CPP_SIZE_WIDTH" in + 8|16|32|64|128|256|512|1024|2048|4096|8192|16384|32768|65536) ;; + *) AC_MSG_ERROR([unsupported __mpu_size_t width: $MCPU_CPP_SIZE_WIDTH]) ;; + esac + case "$MCPU_CPP_SSIZE_WIDTH" in + 8|16|32|64|128|256|512|1024|2048|4096|8192|16384|32768|65536) ;; + *) AC_MSG_ERROR([unsupported __mpu_ssize_t width: $MCPU_CPP_SSIZE_WIDTH]) ;; + esac + + if test "x$MCPU_CPP_MPU_BYTE_ORDER" != "x$GCC_BYTE_ORDER"; then + AC_MSG_ERROR([LibMPU byte order disagrees with GCC target byte order]) + fi + if test "x$MCPU_CPP_MPU_REGISTER_WIDTH" != "x$MACHINE_REGISTER_WIDTH"; then + AC_MSG_ERROR([LibMPU machine-register width disagrees with GCC target]) + fi + + AC_DEFINE_UNQUOTED([MCPU_CPP_MPU_REAL_IO_LIMIT], [$MCPU_CPP_MPU_REAL_IO_LIMIT], + [LibMPU real I/O limit used by mcpu-cpp.]) + AC_DEFINE_UNQUOTED([MCPU_CPP_MPU_MATH_FN_LIMIT], [$MCPU_CPP_MPU_MATH_FN_LIMIT], + [LibMPU math-function limit used by mcpu-cpp.]) + AC_DEFINE_UNQUOTED([MCPU_CPP_MPU_INT_MAX_WIDTH], [$MCPU_CPP_MPU_INT_MAX_WIDTH], + [Maximum LibMPU integer width in bits used by mcpu-cpp.]) + AC_DEFINE_UNQUOTED([MCPU_CPP_MPU_BYTE_ORDER], [$MCPU_CPP_MPU_BYTE_ORDER], + [LibMPU byte order used by mcpu-cpp.]) + AC_DEFINE_UNQUOTED([MCPU_CPP_MPU_WORD_ORDER], [$MCPU_CPP_MPU_WORD_ORDER], + [LibMPU word order used by mcpu-cpp.]) + AC_DEFINE_UNQUOTED([MCPU_CPP_MPU_REGISTER_WIDTH], [$MCPU_CPP_MPU_REGISTER_WIDTH], + [LibMPU machine-register width used by mcpu-cpp.]) + AC_DEFINE_UNQUOTED([MCPU_CPP_MPU_UNIT_WIDTH], [$MCPU_CPP_MPU_UNIT_WIDTH], + [LibMPU addressable-unit width used by mcpu-cpp.]) + AC_DEFINE_UNQUOTED([MCPU_CPP_SIZEOF_SIZE_T], [$MCPU_CPP_SIZEOF_SIZE_T], + [Size of __mpu_size_t in bytes.]) + AC_DEFINE_UNQUOTED([MCPU_CPP_SIZE_WIDTH], [$MCPU_CPP_SIZE_WIDTH], + [Width of __mpu_size_t in bits.]) + AC_DEFINE_UNQUOTED([MCPU_CPP_SIZEOF_SSIZE_T], [$MCPU_CPP_SIZEOF_SSIZE_T], + [Size of __mpu_ssize_t in bytes.]) + AC_DEFINE_UNQUOTED([MCPU_CPP_SSIZE_WIDTH], [$MCPU_CPP_SSIZE_WIDTH], + [Width of __mpu_ssize_t in bits.]) +]) diff --git a/auto-clean b/auto-clean new file mode 100755 index 0000000..346660a --- /dev/null +++ b/auto-clean @@ -0,0 +1,42 @@ +#!/bin/bash + +CWD=`pwd` + +program=`basename $0` + +usage() { + cat << EOF + +Usage: $program [options] + +Options: + -h,--help Display this message. + +EOF +} + +if [ -f "${CWD}/Makefile" ] ; then + make distclean +fi + +gitignore='.gitignore' + +if [ -f "$gitignore" ] ; then + while read ln; do + line=`echo "${ln}" | sed 's,^[ \t],,' | sed 's,[ \t]$,,'` + if [ "x$line" != "x" -a "${line:0:1}" != "#" ] ; then + if `echo "${line}" | grep -q '\*~$'` ; then + find "`dirname "${line}"`" -type f -iname '*~' -print0 | while IFS= read -r -d '' file ; do + rm -f "$file" + done + elif `echo "${line}" | grep -q '\*'` ; then + find "`dirname "${line}"`" -type f -iname "`basename "${line}"`" -print0 | while IFS= read -r -d '' file ; do + rm -f "$file" + done + else + if [ -d "${line}" ] ; then rm -rf "${line}" ; fi + if [ -f "${line}" ] ; then rm -f "${line}" ; fi + fi + fi + done < ${CWD}/${gitignore} +fi diff --git a/bootstrap b/bootstrap new file mode 100755 index 0000000..0557f2d --- /dev/null +++ b/bootstrap @@ -0,0 +1,106 @@ +#!/bin/bash + +CWD=`pwd` +program=`basename $0` + +usage() { + cat << EOF_USAGE + +Usage: $program [options] + +Options: + -h,--help Display this message. + -d,--target-dest-dir=DIR The target ROOTFS directory + [default: DIR=/]. + +The script regenerates files that do not need to be stored in Git: + src/mcpp-expr.c from src/mcpp-expr.zubr using ZUBR 4.1.0; + aclocal.m4 and m4 macros using aclocal; + config.h.in using autoheader; + Makefile.in and helper files using automake; + configure using autoconf. + +EOF_USAGE +} + +TARGET_DEST_DIR=/ +ACDIR=usr/share/aclocal +INCDIR=usr/include +SYSTEM_ACDIR= +SYSTEM_INCDIR= + +while [ 0 ] ; do + if [ "$1" = "-h" -o "$1" = "--help" ] ; then + usage + exit 0 + elif [ "$1" = "-d" -o "$1" = "--target-dest-dir" ] ; then + if [ "$2" = "" ] ; then + echo -e "\n${program}: ERROR: --target-dest-dir is not specified.\n" + usage + exit 1 + fi + TARGET_DEST_DIR="$2" + shift 2 + elif [[ $1 == --target-dest-dir=* ]] ; then + TARGET_DEST_DIR="`echo $1 | cut -f2 -d'='`" + shift 1 + else + if [ "$1" != "" ] ; then + echo -e "\n${program}: ERROR: Unknown argument: $1.\n" + usage + exit 1 + fi + break + fi +done + +if [ ! -d "${TARGET_DEST_DIR}" ] ; then + echo -e "\n${program}: ERROR: --target-dest-dir is not a directory.\n" + usage + exit 1 +fi + +# Absolute path: +if [ "${TARGET_DEST_DIR:0:1}" != "/" ] ; then + TARGET_DEST_DIR=${CWD}/${TARGET_DEST_DIR} +fi + +# Remove last '/' char except for the root itself: +if [ "${TARGET_DEST_DIR}" != "/" -a "${TARGET_DEST_DIR: -1}" = "/" ] ; then + len=${#TARGET_DEST_DIR} + let "len = len - 1" + tmp="${TARGET_DEST_DIR:0:$len}" + TARGET_DEST_DIR=${tmp} +fi + +if [ "${TARGET_DEST_DIR}" = "/" ] ; then + SYSTEM_ACDIR=/${ACDIR} + SYSTEM_INCDIR=/${INCDIR} +else + SYSTEM_ACDIR="${TARGET_DEST_DIR}/${ACDIR}" + SYSTEM_INCDIR="${TARGET_DEST_DIR}/${INCDIR}" +fi + +if ! command -v zubr >/dev/null 2>&1 ; then + echo -e "\n${program}: ERROR: ZUBR 4.1.0 is required to regenerate src/mcpp-expr.c.\n" + exit 1 +fi + +( + cd "$CWD/src" || exit 1 + echo -e "\n${program}: producing './src/mcpu-expr.c'" + zubr -vl -s -Bmcpp_ -o mcpp-expr.c mcpp-expr.zubr +) || exit 1 + +aclocal --install -I m4 --force --system-acdir=${SYSTEM_ACDIR} || exit 1 +autoheader --include=${SYSTEM_INCDIR} || exit 1 +automake --gnu --add-missing --copy --force-missing || exit 1 +autoconf --force || exit 1 + +################################################################ +# Remove cache and backup files: +# +rm -rf autom4te.cache src/z.output *~ src/*~ tests/*~ doc/*~ etc/*~ +# +# End of Cleanup. +################################################################ diff --git a/configure.ac b/configure.ac new file mode 100644 index 0000000..87f5503 --- /dev/null +++ b/configure.ac @@ -0,0 +1,190 @@ +AC_PREREQ([2.69]) + +m4_include([acsite.m4]) + +AC_INIT([mcpu-cpp], [1.0.2], [], [mcpu-cpp]) + +AC_MCPU_CPP_REQUIRE_BASH + +AC_MCPU_CPP_HEADLINE([mcpu-cpp], [MCPU Languages Preprocessor], [MCPU_CPP_VERSION],dnl +[Copyright (c) 1998-2026 Andrey V.Kosteltsev])dnl + +AC_MSG_CFG_PART(Getting the canonical system type) +AC_CANONICAL_HOST + +AC_MSG_CFG_PART(Init Automake environment) +AC_CONFIG_SRCDIR([src/main.c]) +AC_CONFIG_HEADERS([config.h]) +AC_CONFIG_MACRO_DIR([m4]) +AM_INIT_AUTOMAKE([foreign dist-xz no-dist-gzip subdir-objects]) + +dnl With the project convention --prefix=/usr, system configuration belongs +dnl in /etc rather than /usr/etc unless the caller explicitly selects another +dnl sysconfdir. +if test "x$sysconfdir" = 'x${prefix}/etc'; then + if test "x$prefix" = xNONE || test "x$prefix" = x/usr; then + sysconfdir=/etc + fi +fi + +AC_MSG_CFG_PART(Test for GNU Compiler Collection) +AC_USE_SYSTEM_EXTENSIONS +AC_PROG_CC +AC_PROG_LN_S +MCPU_CPP_REQUIRE_GCC + +AC_MSG_CFG_PART(Test for LibMPUIO) +#MCPU_CPP_CHECK_LIBMPUIO +dnl ============================================================ +dnl Check for LibMPU +dnl ============================================================ +AC_CHECK_LIBMPU( + [1.0.25], + [yes], + [ + AC_DEFINE( + [HAVE_LIBMPU], + [1], + [Define to 1 if LibMPU is available.] + ) + LIBMPU_VERSION=`$LIBMPU_CONFIG --version 2>/dev/null` + AC_SUBST([LIBMPU_VERSION]) + ], + [ + AC_MSG_ERROR([LibMPU version 1.0.25 or later is required.]) + ] +) + +dnl ============================================================ +dnl Check for LibMPUIO +dnl ============================================================ +AC_CHECK_LIBMPUIO( + [1.0.4], + [yes], + [ + AC_DEFINE( + [HAVE_LIBMPUIO], + [1], + [Define to 1 if LibMPUIO is available.] + ) + LIBMPUIO_VERSION=`$LIBMPUIO_CONFIG --version 2>/dev/null` + AC_SUBST([LIBMPUIO_VERSION]) + ], + [ + AC_MSG_ERROR([LibMPUIO version 1.0.4 or later is required.]) + ] +) + +AC_MSG_CFG_PART(Test for LibMPU ABI) +MCPU_CPP_CHECK_LIBMPU_ABI + +AC_MSG_CFG_PART(Define MCPU installation layout) + +mcpu_cpp_prefix="$prefix" +if test "x$mcpu_cpp_prefix" = xNONE; then + mcpu_cpp_prefix="$ac_default_prefix" +fi + +mcpu_cpp_exec_prefix="$exec_prefix" +if test "x$mcpu_cpp_exec_prefix" = xNONE; then + mcpu_cpp_exec_prefix="$mcpu_cpp_prefix" +fi + +mcpu_cpp_save_prefix="$prefix" +mcpu_cpp_save_exec_prefix="$exec_prefix" +prefix="$mcpu_cpp_prefix" +exec_prefix="$mcpu_cpp_exec_prefix" +eval mcpu_cpp_bindir="\"$bindir\"" +eval mcpu_cpp_libdir="\"$libdir\"" +eval mcpu_cpp_sysconfdir="\"$sysconfdir\"" +prefix="$mcpu_cpp_save_prefix" +exec_prefix="$mcpu_cpp_save_exec_prefix" + +mcpu_cpp_root="$mcpu_cpp_libdir/mcpu" +mcpu_cpp_private_bindir="$mcpu_cpp_root/bin" +mcpu_cpp_private_etcdir="$mcpu_cpp_root/etc" +mcpu_cpp_private_includedir="$mcpu_cpp_root/include" + +MCPU_CPP_ROOT="$mcpu_cpp_root" +MCPU_CPP_BINDIR="$mcpu_cpp_private_bindir" +MCPU_CPP_ETCDIR="$mcpu_cpp_private_etcdir" +MCPU_CPP_INCLUDEDIR="$mcpu_cpp_private_includedir" +AC_SUBST([MCPU_CPP_ROOT]) +AC_SUBST([MCPU_CPP_BINDIR]) +AC_SUBST([MCPU_CPP_ETCDIR]) +AC_SUBST([MCPU_CPP_INCLUDEDIR]) + +mcpu_cpp_public_link_target="$mcpu_cpp_private_bindir/mcpu-cpp" +case "$mcpu_cpp_bindir:$mcpu_cpp_private_bindir" in + "$mcpu_cpp_exec_prefix"/*:"$mcpu_cpp_exec_prefix"/*) + mcpu_cpp_bindir_rel=${mcpu_cpp_bindir#"$mcpu_cpp_exec_prefix/"} + mcpu_cpp_target_rel=${mcpu_cpp_private_bindir#"$mcpu_cpp_exec_prefix/"} + mcpu_cpp_up= + mcpu_cpp_rest=$mcpu_cpp_bindir_rel + while :; do + mcpu_cpp_up="../$mcpu_cpp_up" + case "$mcpu_cpp_rest" in + */*) mcpu_cpp_rest=${mcpu_cpp_rest#*/} ;; + *) break ;; + esac + done + mcpu_cpp_public_link_target="$mcpu_cpp_up$mcpu_cpp_target_rel/mcpu-cpp" + ;; +esac +MCPU_CPP_PUBLIC_LINK_TARGET="$mcpu_cpp_public_link_target" +AC_SUBST([MCPU_CPP_PUBLIC_LINK_TARGET]) + +AC_DEFINE_UNQUOTED([MCPU_CPP_SYSTEM_CONFIG_FILE], + ["$mcpu_cpp_sysconfdir/mcpu/mcpu-cpp.conf"], + [Optional system-wide mcpu-cpp configuration override.]) + + +AC_MSG_CFG_PART(OUTPUT Substitutions) +AC_CONFIG_FILES([ + Makefile + src/Makefile + tests/Makefile + doc/Makefile + man/Makefile + man/ru/Makefile + etc/Makefile + etc/mcpu-cpp.conf +]) +AC_OUTPUT + +AC_MSG_CFG_PART(Configuration summary) +AC_MSG_NOTICE([mcpu-cpp configuration:]) +AC_MSG_NOTICE([ version : $PACKAGE_VERSION]) +AC_MSG_NOTICE([ host : $host]) +AC_MSG_NOTICE([ prefix : $prefix]) +AC_MSG_NOTICE([ install MCPU root : $mcpu_cpp_root]) +AC_MSG_NOTICE([ runtime config : <runtime-root>/etc/mcpu-cpp.conf]) +AC_MSG_NOTICE([ system override : $mcpu_cpp_sysconfdir/mcpu/mcpu-cpp.conf]) +AC_MSG_NOTICE([ runtime MCPU include : <runtime-root>/include]) +AC_MSG_NOTICE([ public command : $mcpu_cpp_bindir/mcpu-cpp]) +AC_MSG_NOTICE([ public symlink : $mcpu_cpp_public_link_target]) +AC_MSG_NOTICE([ mpuio-config : $LIBMPUIO_CONFIG]) +AC_MSG_NOTICE([ LibMPU version : $LIBMPU_VERSION]) +AC_MSG_NOTICE([ LibMPUIO version : $LIBMPUIO_VERSION]) +AC_MSG_NOTICE([ LibMPUIO CFLAGS : $LIBMPUIO_CFLAGS]) +AC_MSG_NOTICE([ LibMPUIO LDFLAGS : $LIBMPUIO_LDFLAGS]) +AC_MSG_NOTICE([ LibMPUIO LIBS : $LIBMPUIO_LIBS]) +AC_MSG_NOTICE([ LibMPU real I/O limit : $MCPU_CPP_MPU_REAL_IO_LIMIT]) +AC_MSG_NOTICE([ LibMPU math fn limit : $MCPU_CPP_MPU_MATH_FN_LIMIT]) +AC_MSG_NOTICE([ LibMPU int max width : $MCPU_CPP_MPU_INT_MAX_WIDTH]) +AC_MSG_NOTICE([ LibMPU byte order : $MCPU_CPP_MPU_BYTE_ORDER]) +AC_MSG_NOTICE([ LibMPU word order : $MCPU_CPP_MPU_WORD_ORDER]) +AC_MSG_NOTICE([ machine reg width : $MCPU_CPP_MPU_REGISTER_WIDTH]) +AC_MSG_NOTICE([ size_t width : $MCPU_CPP_SIZE_WIDTH]) +AC_MSG_NOTICE([ ssize_t width : $MCPU_CPP_SSIZE_WIDTH]) + +if test -f "config.h"; then + echo "" + echo "Now you can run:" + echo " \`${TB}make${TN}' to compile," + echo " \`${TB}make install${TN}' to make and install ${TB}mcpu-cpp${TN}," + echo " \`${TB}make dist${TN}' to create distributable tarballs, or" + echo " \`${TB}make distclean${TN}' to clean before configure for another target." + echo "Enjoy." + echo "" +fi diff --git a/doc/Makefile.am b/doc/Makefile.am new file mode 100644 index 0000000..7f354f3 --- /dev/null +++ b/doc/Makefile.am @@ -0,0 +1,3 @@ +EXTRA_DIST = \ + mcpu-cpp-ru.md \ + mcpu-cpp-en.md diff --git a/doc/mcpu-cpp-en.md b/doc/mcpu-cpp-en.md new file mode 100644 index 0000000..5429456 --- /dev/null +++ b/doc/mcpu-cpp-en.md @@ -0,0 +1,2169 @@ +# mcpu-cpp + +`mcpu-cpp` is the preprocessor for MCPU programming languages. It is an +independent component of the LibMPU/LibMPUIO/LibMCPU ecosystem and is not tied +to the name of any single language: the active language is selected with the +`#lang` directive. + +This document defines the normative behavior of `mcpu-cpp`: its text model, +directives, macro engine, include pipeline, configuration, diagnostics, and +dependency generation for MCPU tools. + +## 1. Text model + +External source files and configuration files are encoded in UTF-8. The UTF-8 +must be valid. For source programs, the check that characters belong to the +UCS-2 range is performed after comments have been removed. Therefore a valid +Unicode scalar value above `U+FFFF` is permitted inside a comment, but remains +an error in program text. After this stage, source text is processed as a +sequence of `__mpu_char16_t` values. An input UTF-8 BOM is accepted and +removed. An embedded NUL in a source file is forbidden. + +`CRLF` and `CR` line endings are normalized to `LF`. + +## 2. Actions performed independently of directives + +`mcpu-cpp` performs several transformations before directives are parsed. + +### 2.1. Backslash-newline + +A `\\` immediately followed by a newline is removed before comments, +directives, and macros are recognized. For example, + +```text +#defi\ +ne FOO 10\ +20 +``` + +is equivalent to the logical line + +```text +#define FOO 1020 +``` + +Physical line numbers continue to contribute to the current source position. +Unless the user changes that position with `#line`, those physical positions +are the ones reflected in generated line markers. + +### 2.2. Comments + +`/* ... */` and `// ...` comments are removed before subsequent processing. +Where needed to keep adjacent tokens separate, a whitespace separator is +preserved. If a comment terminates a nonempty line, neither a synthetic +separator nor whitespace that immediately preceded the comment is retained +after the comment is removed: the line ends at its last significant +character. The same rule applies to a multi-line comment that starts after +program text. If comment removal leaves a line containing only whitespace, the +line becomes genuinely empty. A comment between two tokens still leaves the +separator required to keep the tokens from being joined. Newlines are +preserved so source coordinates are not destroyed. + +Comments are not recognized inside string or character constants. In the +`diff` language, an apostrophe is not treated as the beginning of a character +constant because it is used in derivative notation. + +Within a literal `#include <...>` operand, `/*` and `//` sequences are treated +as part of the file name. + +## 3. Directives and the output stream + +A directive begins with `#` when only whitespace or comments precede it on the +logical line. Whitespace is permitted between `#` and the directive name. + +Source-position service information in the output stream uses GNU **line +markers**: + +```text +# line-number "file-name" [flags] +``` + +This is not the input directive `#line`. Entering an included file adds flag +`1` to the line marker, and returning to the file that contained the +`#include` adds flag `2`. These values have the same meaning as in GNU CPP: +`1` means entering a new file and `2` means returning to the previous file. +Flag `2` is not a nesting count or include level. + +For example: + +```text +# 1 "main.c" +# 1 "defs.h" 1 +... +# 2 "main.c" 2 +``` + +The input directive + +```text +#line 62 "main.y" +``` + +is not copied to the output stream. It changes the logical values of +`__LINE__` and `__FILE__` for subsequent text and is represented in output by +a line marker: + +```text +# 62 "main.y" +``` + +The arguments of `#line` undergo macro expansion according to the line-control +model. If an `#include` follows such a `#line`, the return marker receives flag +`2`, for example `# 65 "main.y" 2`. A name installed by `#line` becomes the +logical name used by `__FILE__` and line markers; it does not change the +directory used to resolve a quoted `#include`. + +Preprocessor directives use canonical English names only. Unicode remains fully supported in identifiers, strings, comments, and other user text. + +## 4. Header files + +The following forms are supported: + +```text +#include "file" +#include <file> +#include_next "file" +#include_next <file> +#pragma once +``` + +For ordinary `#include "file"`, the directory of the **physical** current +source file is always checked first. A logical name established by `#line` +does not affect this step. For `#include <file>`, the directory of the current +file is not checked. + +### 4.1. Relocatable MCPU root as an ecosystem-wide principle + +Starting with release 0.0.37, the MCPU installation directory **does not +contain a version number of a particular tool** and is not an absolute runtime +constant compiled into the binary. A version belongs to `mcpu-cpp`, +`mcpu-as`, `mcpu-ld`, `mcpu-run`, or a library; it does not define the root of +the shared MCPU environment. + +For a typical configuration: + +```text +./configure --prefix=/usr --libdir=/usr/lib64 +``` + +`make install` creates: + +```text +/usr/lib64/mcpu/ +├── bin/ +│ └── mcpu-cpp +├── etc/ +│ └── mcpu-cpp.conf +├── include/ +│ ├── diff/ +│ ├── dift/ +│ ├── alg/ +│ ├── as/ +│ ├── avm/ +│ └── acs/ +└── lib/ # common directory for future MCPU libraries +``` + +The public program name lives in `$bindir`: + +```text +/usr/bin/mcpu-cpp -> ../lib64/mcpu/bin/mcpu-cpp +``` + +The absolute `/usr/lib64/mcpu` path is **not part of the MCPU-CPP runtime +ABI**. It is only the configure-time installation location selected by +`make install`. + +On every normal invocation, MCPU-CPP determines the actual path of its own +executable through Linux `/proc/self/exe`. The public-command symlink does not +interfere with this: `/proc/self/exe` names the binary that is actually being +executed. If `/proc/self/exe` is unavailable, a fallback resolves `argv[0]` +through `PATH` and `realpath(3)`; there is no fallback to a compiled-in +configure-time installation root. + +For an executable + +```text +<root>/bin/mcpu-cpp +``` + +the runtime root is derived as: + +```text +executable = <root>/bin/mcpu-cpp +executable dir = <root>/bin +MCPU runtime root = <root> +``` + +and the following paths are derived from it automatically: + +```text +<root>/etc/mcpu-cpp.conf +<root>/include +``` + +Therefore the whole tree can be physically moved, for example from + +```text +/usr/lib64/mcpu/ +``` + +to + +```text +/opt/mcpu-test/ +``` + +or + +```text +$HOME/devel/mcpu-next/ +``` + +and `<new-root>/bin/mcpu-cpp` immediately starts using +`<new-root>/etc/mcpu-cpp.conf` and `<new-root>/include` without being +reconfigured. The old absolute path is retained neither in runtime defaults +nor in the installed `mcpu-cpp.conf`. + +This is not a preprocessor-specific trick; it is a **general MCPU ecosystem +principle**. Future `mcpu-as`, `mcpu-ld`, `mcpu-run`, libraries, CRT, and other +components are expected to share one relocatable root: + +```text +<root>/bin +<root>/etc +<root>/include +<root>/lib +``` + +Their own versions may differ. Consistency of a particular MCPU environment is +defined by all components residing in one runtime tree, not by matching +version suffixes in directory names. + +### 4.2. Runtime defaults, configuration layers, and the system include root + +Before reading any configuration file, MCPU-CPP creates the runtime-derived +value: + +```text +MCPU_CPP_SYSTEM_INCLUDE_PATH = <runtime-root>/include +``` + +Configuration layers are then applied in order of increasing priority: + +```text +runtime-derived defaults + ↓ +<runtime-root>/etc/mcpu-cpp.conf + ↓ +/etc/mcpu/mcpu-cpp.conf + ↓ +$HOME/.mcpu/mcpu-cpp.conf +``` + +`<runtime-root>/etc/mcpu-cpp.conf` is installed with MCPU-CPP, but deliberately +does not contain an absolute default `MCPU_CPP_SYSTEM_INCLUDE_PATH`: otherwise +moving the tree would restore the old path. `/etc/mcpu/mcpu-cpp.conf` is an +optional machine-wide override; `make install` does not create `/etc/mcpu`. +`$HOME/.mcpu/mcpu-cpp.conf` is also optional, is not versioned, and has the +highest configuration priority. + +If one variable is defined more than once, the last definition wins, including +an empty definition. Therefore `MCPU_CPP_SYSTEM_INCLUDE_PATH` remains a fully +replaceable system root. For example: + +```text +MCPU_CPP_SYSTEM_INCLUDE_PATH = $HOME/mcpu-next/include; +``` + +completely replaces the runtime-derived `<runtime-root>/include`. For active +`#lang "as"`, the following locations are then checked: + +```text +$HOME/mcpu-next/include/as +$HOME/mcpu-next/include +``` + +Standard language subdirectories are always derived by the preprocessor from +one root; there are no variables named +`MCPU_CPP_SYSTEM_<LANG>_INCLUDE_PATH`. + +An empty effective value: + +```text +MCPU_CPP_SYSTEM_INCLUDE_PATH = ; +``` + +removes the configured system stage entirely. A higher-priority configuration +file may later enable it again with a nonempty value. + +`--config-file FILE` applies an explicitly selected file on top of the +runtime-derived default. `--no-config` disables **only configuration-file +reading**: `<runtime-root>/etc/mcpu-cpp.conf`, `/etc/mcpu/mcpu-cpp.conf`, and +`$HOME/.mcpu/mcpu-cpp.conf` are not read, but `<runtime-root>/include` remains +the standard system root. Only `-nostdinc` removes the effective standard +system tree from include search for one invocation; an explicit `-isystem` +still remains a command-line directory. + +### 4.3. Normative include-file search order + +Search order is part of the MCPU-CPP contract. Explicit command-line +parameters have priority over persistent configuration. After the optional +directory of the current physical file, the effective chain is strictly: + +```text +explicit -I + ↓ +explicit -isystem + ↓ +MCPU_CPP_<LANG>_INCLUDE_PATH + ↓ +MCPU_CPP_INCLUDE_PATH + ↓ +MCPU_CPP_SYSTEM_INCLUDE_PATH/<lang> + ↓ +MCPU_CPP_SYSTEM_INCLUDE_PATH + ↓ +explicit -idirafter + ↓ +MCPU_CPP_AFTER_INCLUDE_PATH +``` + +Entries that are absent or do not contain the requested file are skipped. + +`MCPU_CPP_<LANG>_INCLUDE_PATH` denotes user-configurable language-specific path +lists: + +```text +MCPU_CPP_DIFF_INCLUDE_PATH +MCPU_CPP_DIFT_INCLUDE_PATH +MCPU_CPP_ALG_INCLUDE_PATH +MCPU_CPP_AS_INCLUDE_PATH +MCPU_CPP_AVM_INCLUDE_PATH +MCPU_CPP_ACS_INCLUDE_PATH +``` + +The user fully controls the names and locations of these directories. +`MCPU_CPP_INCLUDE_PATH` is a common user path list visible in every language +state. + +`-idirafter` and `MCPU_CPP_AFTER_INCLUDE_PATH` form a common fallback area. +MCPU-CPP does not automatically derive `<lang>` subdirectories for them. The +user controls their internal layout and may, for example, write: + +```text +#include <vendor/device.h> +``` + +Priority is determined by the semantic class, not by the relative appearance +of different classes in argv or configuration. Within one class, insertion +order is preserved. + +### 4.4. `#include_next` and wrapper headers + +`#include_next` is intended primarily for wrapper headers. It allows a local +header to precede a system header, adjust local policy, and then continue the +search for a same-named header along the normative chain without copying the +system file or using an absolute name. + +For example: + +```text +mcpu-cpp -isystem $HOME/mcpu-wrapper ... +``` + +with `$HOME/mcpu-wrapper/math.h`: + +```text +#ifndef SOME_SYSTEM_MACRO +#define SOME_SYSTEM_MACRO temporary_value +#define REMOVE_SOME_SYSTEM_MACRO 1 +#endif + +#include_next <math.h> + +#ifdef REMOVE_SOME_SYSTEM_MACRO +#undef SOME_SYSTEM_MACRO +#undef REMOVE_SOME_SYSTEM_MACRO +#endif +``` + +If the home configuration also specifies: + +```text +MCPU_CPP_SYSTEM_INCLUDE_PATH = $HOME/mcpu-next/include; +``` + +a wrapper found through `-isystem` continues `#include_next` through the +configured user paths, then through `$HOME/mcpu-next/include/<lang>` and +`$HOME/mcpu-next/include`. The old system tree at the original installation +location does not participate. This is the intended way for a system developer +or tester to work in a private sandbox. + +MCPU-CPP stores the exact **physical element of the effective search chain** +from which the current header was found. `#include_next` starts at the next +element. The `"file"` and `<file>` forms of `#include_next` are equivalent; the +directory of the current file is not checked again. If the current file was +found by ordinary quoted search relative to its containing file and therefore +has no search-chain provenance, `#include_next` starts at the first element of +the configured chain. + +The operand may be produced by macro expansion. A logical name installed by +`#line` does not affect physical provenance. If no suitable file exists after +the current entry, preprocessing fails. + +### 4.5. `#pragma once` + +An active + +```text +#pragma once +``` + +directive marks the **physical file** as already processed during the current +MCPU-CPP invocation. A later attempt to include the same physical file skips +its contents. The directive itself is consumed by the preprocessor and is not +copied to output, including in `-dD` mode. + +Identity is determined by the file-system `st_dev`/`st_ino` pair, not by the +path string. Therefore the same file cannot bypass `#pragma once` by being +reached as `./file.h`, through a symbolic link, or through another hard-link +name. A logical name installed by `#line` also has no effect on this physical +identity. + +The mark takes effect immediately when the active directive is processed. +Therefore a header may include itself after `#pragma once`: the repeated +include is skipped and recursion does not occur. A directive in an inactive +conditional branch has no effect. + +MCPU-CPP recognizes only the exact `#pragma once` form, with optional +whitespace. Other `#pragma` directives are not interpreted by the preprocessor +and are preserved for later compiler stages; for example, `#pragma pack(...)` +continues to be passed through to output. + +`#pragma once` supplements, but does not modify, the normative +`#include`/`#include_next` search chain. The ordinary search mechanism first +finds a physical file, then the `once` registry decides whether its contents +must be processed. + +### 4.6. Forced files: `-imacros FILE` and `-include FILE` + +The command-line options + +```text +-imacros FILE +-include FILE +``` + +process a file before the primary input. They use the ordinary preprocessing +engine, not a separate simplified parser. + +The normative start-of-translation-unit order is: + +```text +predefined macros + -> -D/-U in command-line order + -> all -imacros in command-line order + -> all -include in command-line order + -> primary input +``` + +Thus the relative interleaving of `-imacros` and `-include` in `argv` does not +interleave the two groups: **all** `-imacros` files are always processed before +**all** `-include` files. + +`-imacros FILE` fully preprocesses the file. Its `#define`/`#undef`, +conditional directives, `#lang`/`#endlang`, `#include`, `#include_next`, +`#pragma once`, and diagnostics have normal semantics. However, all normal +preprocessing output from this forced file, including line markers and text +from nested headers, is discarded. The resulting macro-table state and other +preprocessing state are retained for later forced files and for the primary +input. + +`-include FILE` uses the same machinery, but preserves normal output, as if the +located header had been included immediately before the primary source. A +forced include is a real include boundary: inside it `__INCLUDE_LEVEL__ == 1`, +inside a header that it includes the level is `2`, and the primary input +remains at level `0`. `__BASE_FILE__` inside forced files remains the name of +the primary input. + +An absolute forced-file operand is used directly. A relative operand is first +searched for in the **current working directory**, then along the ordinary +include chain: + +```text +explicit -I +explicit -isystem +MCPU_CPP_<LANG>_INCLUDE_PATH +MCPU_CPP_INCLUDE_PATH +MCPU_CPP_SYSTEM_INCLUDE_PATH/<lang> +MCPU_CPP_SYSTEM_INCLUDE_PATH +explicit -idirafter +MCPU_CPP_AFTER_INCLUDE_PATH +``` + +The directory of the primary input receives no special priority when resolving +an `-imacros`/`-include` operand. Once a forced file is found, ordinary quoted +`#include "file"` inside it is again resolved relative to the physical +directory of that forced file. If the forced file was found through an element +of the include chain, its provenance is retained and `#include_next` continues +at the next chain element. + +Forced files and the headers actually reached from them participate in the +ordinary physical dependency registry. Their user/system classification is +derived from the same search provenance, so `-MM`/`-MMD` filter system forced +headers exactly as they filter ordinary system headers. A missing forced file +is an error. + +### 4.7. Dependency generation: `-M`, `-MM`, `-MG`, `-MD`, `-MMD`, `-MF`, `-MT`, `-MQ` + +`-M` and `-MM` use **the same include-pipeline pass** as ordinary preprocessing. +There is no second, independent header search. The dependency graph therefore +inherits the normative search order, `#include_next`, macro-expanded include +operands, conditional compilation, and `#pragma once` automatically. + +`-M` suppresses normal preprocessing output and emits one Make rule: + +```make +file.o: file.c header1.h header2.h +``` + +The list contains the primary source and all physical headers actually reached, +including system headers. A physical file appears only once. Identity is +`st_dev + st_ino`, so an alternate relative spelling, symbolic link, or hard +link does not create a duplicate dependency. A logical name established by +`#line` is only a source name and never enters the dependency list. + +`-MM` builds the same graph but removes system dependencies. System context +includes headers found through explicit `-isystem`, the configured system tree +`MCPU_CPP_SYSTEM_INCLUDE_PATH/<lang>` / `MCPU_CPP_SYSTEM_INCLUDE_PATH`, explicit +`-idirafter`, and `MCPU_CPP_AFTER_INCLUDE_PATH`, as well as the entire branch of +headers included directly or indirectly from such a system header. The syntax +`#include "file"` versus `#include <file>` does not by itself determine whether +a dependency is a system dependency. If one physical file is reached from a +system branch and is later included directly from user context, it remains a +user dependency and is present in `-MM` output. + +The default target is derived from the basename of the primary source: its +suffix is replaced with the object suffix (`.o` by default). Paths and the +default target are Make-quoted. For stdin, the GNU-like form is `-: -`. + +`-MD` and `-MMD` use the same dependency graph but, unlike `-M` and `-MM`, +**do not suppress normal preprocessing output**. `-MD` includes system headers +like `-M`; `-MMD` applies the user-only filter of `-MM`. One pass can therefore +produce both preprocessed text and a side-effect dependency file. + +If `-MF` is not specified, side-effect mode chooses the `.d` file name +automatically: + +* without `-o`, the input basename loses its suffix and receives `.d`; input + pathname directories are not copied into the dependency-file name; +* with an ordinary `-o FILE`, the output-file suffix is replaced by `.d`; +* stdin uses `-.d`. + +`-MF FILE` overrides the automatic dependency-file name. `-MF -` means stdout. +`-MF` also works with dependency-only `-M`/`-MM`; in that case it takes +priority over the ordinary destination of the Make rule. `-MF` by itself, +without one of `-M`, `-MM`, `-MD`, or `-MMD`, is an error. + +The semantics intentionally follow GNU CPP: `-MD`/`-MMD` do not accept their +own argument; `-MF` is a separate dependency-output option. + +`-MT TARGET` replaces the automatic target with `TARGET` **exactly as supplied**. +No Make quoting is performed. Thus one `-MT` argument may contain multiple +targets separated by spaces: + +```text +-MT 'obj/a.o obj/a.pic.o' +``` + +and repeated `-MT` options also append targets to the same rule: + +```text +-MT obj/a.o -MT obj/a.pic.o +``` + +`-MQ TARGET` has the same target-selection semantics but quotes characters that +are special to Make. For example: + +```text +-MQ '$(OBJDIR)/foo.o' +``` + +produces the left-hand side: + +```make +$$(OBJDIR)/foo.o: +``` + +Both separate arguments (`-MT TARGET`, `-MQ TARGET`) and attached forms +(`-MTTARGET`, `-MQTARGET`) are supported. + +If at least one `-MT` or `-MQ` is present, the automatic default target is not +emitted. In particular, `--object-suffix` affects only the automatic target and +does not rewrite explicit targets. When no explicit target is present, the +default target is Make-quoted as with `-MQ`. + +Repeated and mixed `-MT`/`-MQ` options are allowed. As in GNU CPP, all `-MT` +targets are emitted first in their command-line order, followed by all `-MQ` +targets in their command-line order. All of them form the left-hand side of +**one** dependency rule. + +`-MT` and `-MQ` are meaningful only with one of `-M`, `-MM`, `-MD`, or `-MMD`. +Using either without dependency generation is a command-line error. + +`-MG` changes only the handling of **missing** include files in dependency-only +`-M` and `-MM` modes. Without `-MG`, an unresolved `#include` remains an error. +With `-M -MG` or `-MM -MG`, a missing header is treated as a future generated +file: preprocessing does not fail and the directive operand is added to the +dependency rule **exactly as obtained after macro expansion**, without +prepending a guessed include directory. For example: + +```text +#include "generated.h" +``` + +adds `generated.h` under `-M -MG`, even when that file does not yet exist. A +macro-expanded include behaves the same way: the dependency receives the +expanded name. `-MG` is valid only with `-M` or `-MM`; combinations with +`-MD`/`-MMD`, or use without dependency-only mode, are command-line errors. + +Unresolved dependencies are integrated into **the same ordered dependency +registry** as physical files, but occupy a separate identity domain. For a +found file, the registry still uses `st_dev/st_ino` and physical provenance. +For a missing file there is no such information, so an `-MG` entry performs no +`stat()` and is deduplicated by the exact include-operand text. This is +essential: an identically named file in the current working directory must not +turn an unresolved `<name>` into a false physical match when angle search did +not find that file. Different unresolved spellings, such as `generated.h` and +`./generated.h`, are distinct dependencies. + +For `-MM`, an unresolved dependency gets its user/system class from search +context: a missing `<file>` is system-class and a missing `"file"` is +user-class when the including unit is not itself a system header; any missing +include reached from a system header remains system-class. If the same +unresolved operand appears more than once, the classification of its first +occurrence is retained, matching GNU CPP. Physical dependencies keep the +existing rule that if the same inode is later reached from user context, it is +no longer system-only. + +`-MG` also applies to missing command-line forced files `-include FILE` and +`-imacros FILE`: their operand enters the unresolved registry as a user +dependency without a synthetic search prefix. When the file exists, +`-include`/`-imacros` continue to use the normal physical dependency registry +and search provenance. + + +## 5. Language switching + +The preprocessor starts in language state `0`. This is the unnamed primary +C-like language and it is not a valid argument of `#lang`. + +Supported languages are: + +| Name | Purpose | +|---|---| +| `diff` | differential equations | +| `dift` | difference equations | +| `alg` | algebraic equations | +| `as` | MCPU assembler (`mcpu-as`) | +| `avm` | analog-computer schemes | +| `ACS` | block diagrams of automatic-control systems | + +`#lang` must be followed by a string constant containing one nonempty word: + +```text +#lang "diff" +``` + +The name is checked only against the internal language list above and is +compared case-insensitively in ASCII. Thus `"diff"`, `"Diff"`, `"DIFF"`, and +`"dIfF"` select the same language. The spelling inside the quotes is preserved +in output. + +Whitespace inside the string constant is forbidden: `" diff"`, `"diff "`, and +`"di ff"` are errors. Escape sequences are not interpreted inside this +constant. The closing quote must occur on the same physical source line. Only +whitespace is permitted between the closing quote and the end of the line. + +Whitespace outside the string is normalized. For example: + +```text + # lang "DiFf" +``` + +becomes: + +```text +#lang "DiFf" +``` + +`#lang` pushes a new language onto the language stack; `#endlang` restores the +previous language. The stack is not reset by `#include`, so a language block +may begin and end in different files. `#lang` and `#endlang` remain in the +output stream for the later frontend dispatcher; `#lang` is emitted in +normalized form. + +## 6. Object-like macro definitions + +Starting with 0.0.4, object-like macros are supported: + +```text +#define BUFFER_SIZE 1024 +#define NAME value +#define EMPTY +``` + +The `#define` directive itself is not copied to normal output. In ordinary +text, a macro identifier is replaced with its replacement list. That +replacement is rescanned for macro names, so cascaded expansion works: + +```text +#define A B +#define B 10 +A +``` + +produces `10`. + +While a particular macro is being expanded, that macro is temporarily +disabled. Therefore self-referential and mutually recursive definitions do not +cause infinite recursion. + +Macro names are not expanded inside string or character constants. For the +`diff` language, an apostrophe keeps its language-specific meaning and does +not protect following text as a C character constant. + +A multi-line definition using backslash-newline is supported because splicing +occurs before `#define` is parsed. + +### 6.1. `#undef` + +```text +#undef NAME +``` + +removes an object-like macro definition. Undefining a nonexistent macro is not +an error. + +### 6.2. Computed `#include` + +An `#include` argument that does not begin directly with `"` or `<` is first +macro-expanded. Therefore both of these are valid: + +```text +#define HEADER <diff/model.h> +#include HEADER +``` + +and + +```text +#define HEADER "local.h" +#include HEADER +``` + +The expansion result must have the form `"file"` or `<file>`. + +## 7. Function-like macros + +The macro engine supports function-like macros: + +```text +#define identifier( argument-list ) replacement +``` + +The opening parenthesis in the definition must follow the macro name +**immediately**. Thus + +```text +#define F(X) X +``` + +defines a function-like macro, while + +```text +#define F (X) +``` + +defines an object-like macro with replacement `(X)`. + +At a use site, whitespace is permitted between a function-like macro name and +the opening parenthesis. If `(` does not follow, the identifier is not a call +of that macro and remains in output. + +For an ordinary function-like macro, the number of actual arguments must equal +the number of formal parameters. For a variadic macro, every fixed argument +must be present, while the variadic tail may contain any number of arguments, +including an empty tail. Nested parentheses are tracked while parsing actual +arguments; a comma inside them does not separate arguments. Square brackets do +not have this property; this is part of the adopted macro-expansion semantics. + +For example: + +```text +#define min(X, Y) ((X) < (Y) ? (X) : (Y)) +min(1, 2) +``` + +produces: + +```text +((1) < (2) ? (1) : (2)) +``` + +Before substitution, an ordinary actual argument itself undergoes macro +expansion. Cascaded and nested calls therefore work naturally: + +```text +#define A 7 +#define min(X, Y) ((X) < (Y) ? (X) : (Y)) +min(min(A, 3), 10) +``` + +A formal parameter may occur any number of times in the replacement list. An +expression with side effects in an actual argument may therefore be evaluated +multiple times by the later compiler; the preprocessor does not attempt to +repair such source code. + +Macros with no formal parameters are supported: + +```text +#define READY() 1 +``` + +They expand only when called as `READY()` (whitespace between the name and `(` +is permitted at a use site); the standalone identifier `READY` does not +expand. + +Formal parameter names must be distinct. An unterminated parameter list, +invalid punctuation, and too few or too many actual arguments are errors. + +### 7.1. Stringification `#` + +The stringification operator (`#`) is supported for parameters of +function-like macros: + +```text +#define STR(X) #X +STR(alpha + beta) +``` + +produces: + +```text +"alpha + beta" +``` + +Stringification uses the **raw actual argument before macro expansion**. Thus: + +```text +#define A 7 +#define STR(X) #X +#define XSTR(X) STR(X) + +STR(A) -> "A" +XSTR(A) -> "7" +``` + +Leading and trailing whitespace in the argument is removed. Internal +whitespace sequences are collapsed to one space except inside string/character +tokens of the active language. Double quotes and backslashes inside quoted +tokens are escaped so that the result remains one valid string constant. + +In a function-like replacement list, `#` must refer, directly or after +whitespace, to a formal parameter name. Inside a quoted token, `#` is not an +operator. An empty actual argument is allowed and stringifies as `""`. + +### 7.2. Token concatenation `##` + +Starting with 0.0.21, token concatenation (`##`) is supported with semantics +aligned with GNU CPP and the macro engine's `collect_expansion()` / +`macroexpand()` model. The operator combines two adjacent preprocessing tokens +into one token, after which the resulting replacement list is rescanned for +macro expansion. + +For example: + +```text +#define CAT(A, B) A ## B +CAT(foo, bar) +``` + +produces `foobar`. Concatenation may form an identifier, preprocessing number, +or multi-character punctuator. For example: + +```text +CAT(1.5, e3) -> 1.5e3 +CAT(+, =) -> += +``` + +When a formal parameter is directly adjacent to `##`, its actual argument is +substituted **without preliminary macro expansion**. This is the same raw +argument principle used by stringification. To expand first and concatenate +second, use the normal two-level GNU CPP pattern: + +```text +#define AFTERX(X) X_ ## X +#define XAFTERX(X) AFTERX(X) +#define TABLESIZE 1024 +#define BUFSIZE TABLESIZE + +AFTERX(BUFSIZE) -> X_BUFSIZE +XAFTERX(BUFSIZE) -> X_1024 +``` + +An empty actual argument adjacent to `##` behaves as a placemarker: it adds no +token, and concatenation on that side leaves the remaining operand unchanged. +If an actual argument contains multiple preprocessing tokens, only the edge +token directly adjacent to `##` is concatenated; the others are preserved and +participate in the subsequent rescan. + +`#` and `##` may be used in the same function-like macro, for example: + +```text +#define COMMAND(NAME) #NAME | NAME ## _command +``` + +Here `#NAME` uses the raw spelling of the argument for stringification, while +`NAME ## _command` uses the same raw argument for concatenation. + +Inside a quoted token, `##` is not an operator. Comments have already become +whitespace by the time macro expansion occurs, so comments cannot be created +by concatenating `/` and `*`. Whitespace may originally appear between `##` +and its operands; it does not participate in the concatenation. + +If the two operands do not form one valid preprocessing token, a diagnostic is +issued and the original tokens are retained; whether whitespace appears +between them after that diagnostic is not part of the contract. `##` at the +beginning or end of a replacement list is a macro-definition error. + +### 7.3. Variadic macros: `...` and `__VA_ARGS__` + +Starting with 0.0.46, variadic function-like macros are supported in the modern +C99-compatible form: + +```text +#define LOG(...) output(__VA_ARGS__) +#define LOGF(format, ...) output(format, __VA_ARGS__) +``` + +The `...` marker may be the only parameter or the final element after one or +more fixed parameters. The old GNU extension with a named variadic parameter, + +```text +#define LOG(args...) ... +``` + +is intentionally not supported in 0.0.46. `__VA_OPT__` was also not part of +that particular release. + +At invocation, every token after the last fixed parameter, including commas +that separate those tokens, forms one logical variable argument and is +substituted for `__VA_ARGS__`. In an ordinary position, that variable argument +undergoes macro expansion before substitution, just like an ordinary actual +argument: + +```text +#define A 7 +#define V(...) <__VA_ARGS__> +#define F(first, ...) first | __VA_ARGS__ + +V(A, 2, 3) -> <7, 2, 3> +F(1, A, 3) -> 1 | 7, 3 +``` + +The variadic tail may be empty. Both + +```text +F(1) +F(1,) +``` + +are valid and substitute an empty `__VA_ARGS__`. This does **not** imply that a +comma written explicitly in the replacement list is removed automatically. +For example, with + +```text +#define E(format, ...) output(format, __VA_ARGS__) +``` + +`E("ok")` leaves the comma before the empty tail. The historical GNU +`, ## __VA_ARGS__` comma-swallowing behavior is deliberately outside the 0.0.46 +contract and remains unsupported; modern code should use `__VA_OPT__(,)`. + +`__VA_ARGS__` participates in the existing `#` and `##` semantics as a real +macro parameter. Stringification uses the raw spelling of the whole variadic +tail: + +```text +#define STRV(...) #__VA_ARGS__ +STRV(A, b + c) -> "A, b + c" +``` + +When adjacent to `##`, the variadic argument is likewise substituted without +prescan; the ordinary placemarker, token-concatenation, and rescan rules then +apply. For example: + +```text +#define L(...) pre ## __VA_ARGS__ +#define R(...) __VA_ARGS__ ## post + +L(fix) -> prefix +R(fix) -> fixpost +``` + +If the variadic argument contains multiple preprocessing tokens, only the edge +token immediately adjacent to `##` is concatenated and the remaining tokens +are preserved, exactly as for an ordinary parameter. An empty variadic tail +next to `##` behaves as a placemarker. + +The name `__VA_ARGS__` is reserved for the variable argument and is not +accepted as an ordinary formal parameter name. `#__VA_ARGS__` is valid only in +a variadic macro. Dump modes preserve the variadic form of the definition, for +example: + +```text +#define F(first,...) first | __VA_ARGS__ +``` + +### 7.4. `__VA_OPT__` + +Starting with 0.0.47, variadic macros support the standard conditional fragment +`__VA_OPT__(pp-tokens)`. If the variable argument contains no preprocessing +tokens after normal macro substitution, the entire `__VA_OPT__(...)` expands +to an empty sequence. If the variable argument is nonempty, the parenthesized +contents participate in the replacement list: + +```text +#define DEBUG(format, ...) \ + fprintf(stderr, format __VA_OPT__(,) __VA_ARGS__) + +DEBUG("ready") -> fprintf(stderr, "ready") +DEBUG("x=%d", x) -> fprintf(stderr, "x=%d", x) +``` + +Emptiness is decided **after expansion of the variable argument**, not from its +raw spelling. Therefore a macro that itself expands to an empty sequence does +not activate `__VA_OPT__`: + +```text +#define EMPTY +#define HAS(...) [__VA_OPT__(yes)] + +HAS() -> [] +HAS(EMPTY) -> [] +HAS(token) -> [yes] +``` + +The contents of `__VA_OPT__` may contain balanced nested parentheses. The +closing `)` of the `__VA_OPT__` construct is found with nesting taken into +account. A nested `__VA_OPT__` inside another `__VA_OPT__` is deliberately +forbidden. + +`__VA_OPT__` is integrated with the existing rules for parameter substitution, +stringification, token concatenation, placemarkers, and rescan. For example: + +```text +#define X 123 +#define S(...) #__VA_OPT__(__VA_ARGS__) +#define L(...) pre ## __VA_OPT__(__VA_ARGS__) + +S() -> "" +S(X) -> "123" +L() -> pre +L(X) -> pre123 +``` + +With `#__VA_OPT__(...)`, parameter substitution inside the fragment happens +first, including prescan of ordinary parameters, but arbitrary macro names in +the fragment are not additionally rescanned before stringification. Thus: + +```text +#define X 123 +#define S(a, ...) #__VA_OPT__(a X) + +S(X, y) -> "123 X" +``` + +If a parameter inside `__VA_OPT__` participates directly in an internal `##`, +prescan is suppressed for that parameter in the usual way; the paste is +performed before later rescan. An outer `##` adjacent to `__VA_OPT__` receives +the edge token of the already prepared fragment. An empty `__VA_OPT__` result +next to `##` behaves as a placemarker. + +`__VA_OPT__` is valid only in the replacement list of a variadic function-like +macro and must immediately introduce a parenthesized fragment. `##` cannot be +the first or last preprocessing token inside that fragment. + +The historical GNU extension + +```text +, ## __VA_ARGS__ +``` + +is intentionally **not implemented** by `mcpu-cpp`. Use the modern +`__VA_OPT__(,)` form for a conditional comma. The old GNU named variadic +parameter form `args...` also remains unsupported. + +### 7.5. Whitespace normalization in replacement lists + +Starting with 0.0.48, `mcpu-cpp` does not carry alignment whitespace from a +multi-line macro definition into the expansion result. After `\\` + newline +has been removed, a whitespace sequence belonging to the replacement list +itself is canonicalized to one ASCII space. This is particularly important for +definitions whose backslashes are visually aligned in one column: + +```text +#define TRACE(x) \ + do \ + { \ + output(x); \ + done(); \ + } \ + while( 0 ) +``` + +Such a definition expands to the compact replacement: + +```text +do { output(x); done(); } while( 0 ) +``` + +rather than preserving dozens of spaces before each former physical-line +boundary. + +Normalization applies **only to whitespace belonging to the replacement +list**. `mcpu-cpp` is not a source formatter: whitespace in ordinary input text +is preserved. Whitespace inside an actual macro argument is likewise not +reformatted merely because the argument is substituted into a macro: + +```text +#define ID(x) x + +ID(a + b) -> a + b +``` + +String and character literal contents are preserved verbatim, so: + +```text +#define S "left right" +``` + +still contains five spaces inside the string. + +The presence of whitespace between preprocessing tokens is preserved as one +space. This prevents accidental retokenization such as turning `+ +` into +`++`, `- >` into `->`, or `< <` into `<<`. The `#` and `##` operators, +placemarkers, `__VA_ARGS__`, `__VA_OPT__`, and later rescan keep their existing +rules; the policy changes only the amount of ordinary replacement-list +whitespace. + +Dump modes (`-dM`, `-dD`) show the same canonical replacement-list form stored +in the internal macro table. + +### 7.6. Invisible-line compaction and line markers + +Starting with 0.0.49, `mcpu-cpp` uses the same model as GNU CPP for vertical +whitespace: **remove it, but do not forget it**. Source lines that produce no +output preprocessing token after preprocessing need not remain as physical +blank lines in the `.E` output, but their source position still contributes to +line markers and to `__LINE__`. + +Why a line is invisible does not matter. It may be a consumed directive, an +inactive `#if` branch, a single-line or multi-line comment, an ordinary blank +line, or any mixture of these. The emitter compares its current output source +position with the position of the next line that will actually be emitted. + +If the next position is fewer than eight lines away, the gap is represented by +ordinary newlines. If the distance is eight lines or greater, the long run of +blank lines is replaced by a corrective line marker: + +```text +# N "file" +``` + +and the next content line immediately belongs to source line `N`. The behavior +therefore matches the GNU CPP boundary: gaps 0 through 7 use newlines; a gap of +8 or more uses a line marker. + +Structural enter/return markers for included files keep their ordinary +meaning: + +```text +# 1 "header.h" 1 +# 4 "source.c" 2 +``` + +If an included file produces no output, `mcpu-cpp` does not invent a marker +reporting how far the preprocessor progressed internally through that header. +An enter marker may be followed immediately by its return marker. The real +position is corrected again only when some following content must be emitted. + +This optimization changes only the representation of the output stream. +Source coordinates, `__LINE__`, diagnostics, `#line`, include enter/return +semantics, and macro processing remain tied to the logical source stream, not +to the number of physical lines in the compacted `.E` file. + +## 8. Predefined macros + +Starting with 0.0.6, the historical predefined-macro mechanism was restored in +the preprocessor. It is treated as a separate ABI/environment layer for the +future unnamed C-like language. These definitions are not decorative: their +names and values must match either GNU CPP semantics or an explicitly +documented MCPU/LibMPU contract. + +### 8.1. Dynamic source macros + +The following predefined macros are evaluated at the point of use: + +| Macro | Expansion | +|---|---| +| `__FILE__` | string constant containing the name of the current input file | +| `__LINE__` | decimal number of the current source line | +| `__BASE_FILE__` | string constant containing the primary input file name of the translation unit | +| `__INCLUDE_LEVEL__` | `#include` nesting level; `0` in the primary file | +| `__DATE__` | preprocessor start date in the form `"Mmm dd yyyy"` | +| `__TIME__` | preprocessor start time in the form `"hh:mm:ss"` | + +`__DATE__` and `__TIME__` share one timestamp for the whole translation unit. +Their special expansion is emitted without another macro rescan. + +These names reside in the ordinary macro table, so `#undef` followed by +`#define` may deliberately replace a builtin. + +### 8.2. Preprocessor version + +Starting with 0.0.8, the standalone preprocessor does not define GCC's +`__VERSION__`. That name belongs to a compiler environment, which does not yet +exist for the future high-level language. The version of `mcpu-cpp` has its +own unambiguous name: + +```text +#define __MCPU_CPP_VERSION__ "1.0.2" +``` + +The value is obtained automatically from `PACKAGE_VERSION`. When a compiler +frontend/driver appears, its version contract will be defined separately and +will not be mixed with the version of the standalone preprocessor. + +### 8.3. ABI sources of truth + +`mcpu-cpp` is built only with GNU GCC. During `configure`, the project follows +the established LibMPU/LibMPUIO `acsite.m4` approach: GCC predefined macros +describe native type sizes, byte/word order, and machine-register width, while +the installed `<libmpu.h>` is the final source of truth for LibMPU +configuration. + +In particular, the following values are captured and checked: + +```text +MPU_REAL_IO_LIMIT +MPU_MATH_FN_LIMIT +MPU_BYTE_ORDER +MPU_WORD_ORDER +BITS_PER_MACHINE_REGISTER +BITS_PER_UNIT_T +sizeof(__mpu_size_t) +sizeof(__mpu_ptrdiff_t) +``` + +`configure` additionally verifies that the byte order and +`BITS_PER_MACHINE_REGISTER` recorded by LibMPU agree with the GCC target used +to build `mcpu-cpp`. `MPU_WORD_ORDER` is taken directly from the configured +LibMPU profile and describes word order in the MCPU data environment. + +`MPU_REAL_IO_LIMIT` and `MPU_MATH_FN_LIMIT` serve different purposes. For +example, a library may support Real I/O up to 65536 bits while providing +mathematical functions only up to 16384 bits. Therefore `MPU_MATH_FN_LIMIT` +is not used as the limit on existence of Real types. + +### 8.4. MCPU architecture and assembler prefixes + +The target architecture is identified by: + +```text +#define _ARCH_MCPU 1 +``` + +MCPU PTR64 is 64 bits wide, so `__SIZEOF_POINTER__`, +`__MCPU_POINTER_WIDTH__`, `__INTPTR_TYPE__`, `__UINTPTR_TYPE__`, and the +corresponding width/max macros are defined accordingly. + +Assembler-prefix macros follow GNU CPP meaning rather than the first letter of +a register-view name. `mcpu-as` syntax uses no extra sigil before a register, +label, or immediate value. The letters `r` and `c` belong to MCPU register +syntax; they are not a `REGISTER_PREFIX`. Therefore: + +```text +#define __REGISTER_PREFIX__ +#define __LOCAL_LABEL_PREFIX__ +#define __USER_LABEL_PREFIX__ +#define __IMMEDIATE_PREFIX__ +``` + +all four expand to an empty sequence. `.L...` remains a compiler naming +convention and is not an assembler-ABI local-label prefix: LOCAL/GLOBAL binding +is determined by symbol directives. + +### 8.5. Byte order and word order + +The basic numeric byte-order values are compatible with GNU CPP: + +```text +__ORDER_LITTLE_ENDIAN__ +__ORDER_BIG_ENDIAN__ +__ORDER_PDP_ENDIAN__ +``` + +The target environment publishes its own MCPU names: + +```text +#define __MCPU_BYTE_ORDER__ __ORDER_LITTLE_ENDIAN__ +#define __MCPU_WORD_ORDER__ __ORDER_LITTLE_ENDIAN__ +#define __BYTE_ORDER__ __MCPU_BYTE_ORDER__ +``` + +The actual values of `__MCPU_BYTE_ORDER__` and `__MCPU_WORD_ORDER__` come from +the configured LibMPU profile (`MPU_BYTE_ORDER` and `MPU_WORD_ORDER`). They +therefore follow the host data representation for which LibMPU was built. This +does not alter the separate architectural contract for MCPU instruction +bytecode encoding. + +The GNU/C-specific name `__FLOAT_WORD_ORDER__` is not defined because the +future MCPU language has no `float` type. + +LibMPU/MCPU environment parameters are published in the MCPU namespace: + +```text +__MCPU_MACHINE_REGISTER_WIDTH__ +__MCPU_REAL_IO_LIMIT__ +__MCPU_MATH_FN_LIMIT__ +__MCPU_INT_MAX_WIDTH__ +__MCPU_REAL_MAX_WIDTH__ +__MCPU_COMPLEX_MAX_WIDTH__ +``` + +`__MCPU_INT_MAX_WIDTH__` is `NB_I_MAX * 8`; the Real/Complex maximum width is +the configured `MPU_REAL_IO_LIMIT`. `__MCPU_MACHINE_REGISTER_WIDTH__` is the +`BITS_PER_MACHINE_REGISTER` value of the installed LibMPU. Real I/O and math +limits are deliberately kept separate: `MPU_REAL_IO_LIMIT` controls existence +of Real/Complex type families and text conversion, while `MPU_MATH_FN_LIMIT` +controls availability of mathematical functions at a given width. + +### 8.6. MCPU size/ssize, `ptrdiff`, and pointers + +The future language does not inherit variable-width C names such as `short`, +`int`, and `long`, and it does not use the C-style name `size_t` as part of its +own ABI. The unsigned LibMPU size type and signed byte-count/error type are +published symmetrically in the MCPU namespace. For a 64-bit configured profile, +for example: + +```text +#define __MCPU_SIZE_TYPE__ uint64 +#define __MCPU_SIZE_WIDTH__ 64 +#define __MCPU_SIZEOF_SIZE__ 8 +#define __MCPU_SIZE_MAX__ 0xffffffffffffffff + +#define __MCPU_SSIZE_TYPE__ int64 +#define __MCPU_SSIZE_WIDTH__ 64 +#define __MCPU_SIZEOF_SSIZE__ 8 +#define __MCPU_SSIZE_MAX__ 0x7fffffffffffffff +``` + +This is an MCPU-specific family, not an attempt to invent a nonexistent GNU CPP +`__SSIZE_*` contract. + +The MCPU pointer ABI is independent of the host: PTR64 is always 64 bits wide: + +```text +#define __INTPTR_TYPE__ int64 +#define __UINTPTR_TYPE__ uint64 +#define __INTPTR_WIDTH__ 64 +#define __UINTPTR_WIDTH__ 64 +#define __INTPTR_MAX__ 0x7fffffffffffffff +#define __UINTPTR_MAX__ 0xffffffffffffffff +#define __SIZEOF_POINTER__ 8 +#define __MCPU_POINTER_WIDTH__ 64 +``` + +The difference between two MCPU pointers is signed and also fixed independently +of the host: + +```text +#define __PTRDIFF_TYPE__ int64 +#define __PTRDIFF_WIDTH__ 64 +#define __SIZEOF_PTRDIFF__ 8 +#define __PTRDIFF_MAX__ 0x7fffffffffffffff +``` + +Computed MIN expressions such as `(-__PTRDIFF_MAX__ - 1)` are not added to the +predefined table. + +### 8.7. Character types + +The future language has no ordinary C `char`. Therefore `__CHAR_TYPE__` and +`__WCHAR_TYPE__` are not defined. Language types are named without C/C++ `_t` +suffixes: + +```text +#define __CHAR8_TYPE__ char8 +#define __CHAR16_TYPE__ char16 +#define __CHAR8_WIDTH__ 8 +#define __CHAR16_WIDTH__ 16 +#define __SIZEOF_CHAR8__ 1 +#define __SIZEOF_CHAR16__ 2 +``` + +These are types of the future language. The implementation of `mcpu-cpp` +itself continues to use LibMPUIO `__mpu_char16_t` and the strict UCS-2 text +model internally. + +### 8.8. LibMPU integer families + +Complete structural metadata for integer families is generated up to the +actual `NB_I_MAX * 8` of the installed LibMPU rather than stopping at a +hard-coded final type. For every power-of-two width starting at 8 bits, TYPE, +WIDTH, and SIZEOF are defined: + +```text +#define __INT1024_TYPE__ int1024 +#define __UINT1024_TYPE__ uint1024 +#define __INT1024_WIDTH__ 1024 +#define __UINT1024_WIDTH__ 1024 +#define __SIZEOF_INT1024__ 128 +#define __SIZEOF_UINT1024__ 128 +``` + +With the current LibMPU 1.0.25, `NB_I_MAX == 8192`, so the family extends to +`int65536`/`uint65536`, with `__SIZEOF_INT65536__ == 8192`. + +Decimal-digit metadata is defined for **every** permitted integer width: + +```text +__INT<bits>_DECIMAL_DIG__ +__UINT<bits>_DECIMAL_DIG__ +``` + +The value is computed by `mcpu-cpp` integer-only helpers from the known bit +width. It is the exact number of decimal digits in the maximum value of the +type; neither a sign nor a terminating NUL is included in `DECIMAL_DIG`. For +unsigned values the maximum is `2^bits - 1`; for signed values it is +`2^(bits-1) - 1`. This differs from LibMPU `_int_digs()`, which estimates a +string-buffer size and includes room for a terminating NUL. + +For example: + +```text +#define __INT64_DECIMAL_DIG__ 19 +#define __UINT64_DECIMAL_DIG__ 20 +#define __INT256_DECIMAL_DIG__ 77 +#define __UINT256_DECIMAL_DIG__ 78 +``` + +Only the textual maxima themselves are deliberately limited to widths +`bits <= 256`: + +```text +__INT128_MAX__ +__UINT128_MAX__ +``` + +Maxima are produced through LibMPU `iuitoa()`. Macros named +`__INT<bits>_MIN__` are not generated: the predefined table must not contain +computed expressions such as `(-__INT<bits>_MAX__ - 1)`. For widths above 256 +bits, only MAX is absent; TYPE/WIDTH/SIZEOF/DECIMAL_DIG continue through the +full `NB_I_MAX * 8` range. + +### 8.9. LibMPU Real and Complex families + +Real/Complex structural metadata is generated for every power-of-two width +from 32 bits through the actual configured `MPU_REAL_IO_LIMIT`. TYPE, WIDTH, +and SIZEOF are published for all of these types. + +For Complex, WIDTH denotes the type parameter, not total storage width: + +```text +#define __COMPLEX128_TYPE__ complex128 +#define __COMPLEX128_WIDTH__ 128 +#define __SIZEOF_COMPLEX128__ 32 +``` + +`complex128` consists of two `real128` components, so its storage size is 32 +bytes. With `MPU_REAL_IO_LIMIT == 65536`, the top of the family is: + +```text +#define __COMPLEX65536_TYPE__ complex65536 +#define __COMPLEX65536_WIDTH__ 65536 +#define __SIZEOF_COMPLEX65536__ 16384 +``` + +For Real: + +```text +#define __REAL65536_TYPE__ real65536 +#define __REAL65536_WIDTH__ 65536 +#define __SIZEOF_REAL65536__ 8192 +``` + +Precision metadata is defined for **all** allowed Real widths up to +`MPU_REAL_IO_LIMIT`. Macro names correspond directly to LibMPU helpers: + +```text +__REAL<bits>_DECIMAL_DIG__ -> _real_digs(bits/8) +__REAL<bits>_MANT_DIG__ -> _real_mant_digs(bits/8) +``` + +`__REAL<bits>_DIG__` is intentionally absent. The `bits <= 256` restriction +applies only to large textual numeric constants. For widths up to 256 bits, +the following are also defined: + +```text +__REAL<bits>_MAX__ +__REAL<bits>_MIN__ +__REAL<bits>_EPSILON__ +__REAL<bits>_MAX_EXP__ +__REAL<bits>_MIN_EXP__ +__REAL<bits>_MAX_10_EXP__ +__REAL<bits>_MIN_10_EXP__ +``` + +For example, with LibMPU 1.0.25, the current `real128` profile gives values of +the form: + +```text +#define __REAL128_EPSILON__ 2.524354896707237777317531409e-29 +#define __REAL128_MAX__ 4.197157432934775384808581951e+323228496 +#define __REAL128_MIN__ 9.530259619551804292864984035e-323228497 +#define __REAL128_MAX_10_EXP__ 323228496 +#define __REAL128_MAX_EXP__ 1073741823 +#define __REAL128_MIN_10_EXP__ -323228524 +#define __REAL128_MIN_EXP__ -1073741822 +``` + +MAX/MIN/EPSILON are created by LibMPU itself and converted through +`real_to_ascii()`. Exponent constants are obtained from LibMPU exponent helpers +and integer conversion. For widths above 256 bits, these numeric predefines are +absent, but TYPE/WIDTH/SIZEOF/DECIMAL_DIG/MANT_DIG continue through +`MPU_REAL_IO_LIMIT`. + +For every supported Real type through `MPU_REAL_IO_LIMIT`, two compact +characteristics are also published: + +```text +#define __SIZEOF_REAL128_EXP__ 4 +#define __REAL128_MAX_STRLEN__ 60 +``` + +`__SIZEOF_REALxxx_EXP__` is obtained directly from `_sizeof_exp(NB_Rxxx)`. +`__REALxxx_MAX_STRLEN__` comes from `_real_max_string(NB_Rxxx)` and is the +maximum **number of characters** in the textual representation, not a byte +count. A zero-terminated string therefore needs at least +`__REALxxx_MAX_STRLEN__ + 1` elements: for `char8` that is the same number of +bytes, while for `char16` the physical byte count is twice as large. These two +metadata macros are also defined for Real widths above 256 bits because their +own values remain small. + +### 8.10. Macro dumps: `-dM`, `-dMP` + +The command: + +```text +mcpu-cpp -dM input.c +``` + +prints only final **non-predefined** macros in `#define ...` form. This group +includes definitions from the primary file and included headers, as well as +command-line `-D` definitions. MCPU-CPP's own predefined macros are not printed +by `-dM`. This mode is therefore intended primarily for a compact inspection of +macro state created by the user program. + +The command: + +```text +mcpu-cpp -dMP input.c +``` + +adds active MCPU-CPP predefined macros to the same final state. Output contains +two consecutive groups: predefined macros first, then non-predefined macros. +Definitions inside each group are sorted deterministically by name. This is +useful for system development because it exposes the preprocessing ABI and +architectural properties of the current MCPU environment without mixing them +with user definitions. + +Group membership is determined by macro origin, not by spelling. A macro +created by `-D` or `#define` is ordinary even if its name looks system-like. If +a predefined macro is removed with `#undef`, it is not printed. If the user +then defines the same name again, the new definition belongs to the ordinary +group and appears in the corresponding part of `-dMP`, and also in `-dM`. +Thus both modes display the **final macro state**. + +Context-dependent `__FILE__`, `__LINE__`, `__DATE__`, `__TIME__`, +`__BASE_FILE__`, and `__INCLUDE_LEVEL__` are not printed by the static dump. +Static ABI/architecture predefined macros and computed static Real metadata are +printed by `-dMP`. + +When an input file is supplied, it is fully preprocessed first and the final +macro state is printed afterward; ordinary preprocessed text is not emitted in +`-dM`/`-dMP` modes. Without an input file, stdin is used, so empty stdin with +`-dM` gives an empty dump while `-dMP` provides the active static predefined +macros of the current MCPU environment. + +`-dD` has different semantics and is unaffected by this distinction. + +### 8.11. Definition dump: `-dD` + +The command: + +```text +mcpu-cpp -dD input.c +``` + +preserves ordinary preprocessing output and additionally emits encountered +`#define` directives. Before primary input starts, static predefined macro +definitions are printed. Each such definition is preceded by a marker: + +```text +# 0 "<built-in>" +#define NAME value +``` + +and the predefined block itself is preceded by an input-file marker of the form +`# 0 "input.c"`. Context-dependent `__FILE__`, `__LINE__`, `__DATE__`, +`__TIME__`, `__BASE_FILE__`, and `__INCLUDE_LEVEL__` are not included in the +initial built-in block. + +### 8.12. Configuration dump: `-dconfig` + +The command: + +```text +mcpu-cpp -dconfig +``` + +requires no input file and prints the effective configuration-variable layer +after runtime configuration, optional system override, home user override, or +a selected `--config-file` have been read, including `$NAME`/`${NAME}` +expansion. Lines are sorted by name and printed as: + +```text +NAME = value; +``` + +This makes it possible to inspect actual include paths without manually +searching `<runtime-root>/etc/mcpu-cpp.conf`, `/etc/mcpu/mcpu-cpp.conf`, and +`$HOME/.mcpu/mcpu-cpp.conf`. + +### 8.13. Verbose configuration snapshot: `-v` + +With `-v`, MCPU-CPP retains its runtime trace for `#lang`, `#include`, and +`#include_next`, but configuration variables are printed only once, after all +configuration layers have been read and priority rules applied. Verbose output +therefore shows only **effective values**; intermediate values from runtime +root, system, and user configuration are not duplicated. + +The configuration block follows include-policy order: language-specific user +paths, the common user path, the system root, and the AFTER path. A variable +that is absent from every configuration layer is not printed. The +runtime-derived default `MCPU_CPP_SYSTEM_INCLUDE_PATH` is a full lowest-priority +value and is therefore visible under `-v` even when no `mcpu-cpp.conf` exists +**or all configuration files are disabled with `--no-config`**. + +The line form is: + +```text +config: NAME=value +``` + +### 8.14. Effective search directories: `-dsearch-dirs` + +The command: + +```text +mcpu-cpp -dsearch-dirs +``` + +requires no input file, prints the effective global search directories, and +exits without preprocessing. The format is intentionally simple: + +```text +search: /path/to/directory +``` + +Directories are printed in semantic search-class order: + +```text +explicit -I +explicit -isystem +configured language-specific user directories +MCPU_CPP_INCLUDE_PATH +MCPU_CPP_SYSTEM_INCLUDE_PATH/<lang> +MCPU_CPP_SYSTEM_INCLUDE_PATH +explicit -idirafter +MCPU_CPP_AFTER_INCLUDE_PATH +``` + +Language-specific entries are printed for every supported language in their +canonical order. During a real `#include`, only the directory corresponding to +the active `#lang` participates. The directory of the current physical file is +not printed by `-dsearch-dirs`: it exists only dynamically for a particular +`#include "..."` and changes with the include stack. `--no-config` does not +remove the runtime-derived system root, so even without configuration files the +dump still contains `<runtime-root>/include/<lang>` and +`<runtime-root>/include`. `-nostdinc` removes the effective system `<lang>` +entries and system root from the dump, but not explicit `-isystem`. A directory +that does not exist in the file system is still displayed because it remains +part of the effective search configuration and will simply be skipped during a +real file search. + +`-dsearch-dirs` accounts for `-I`, `-isystem`, `-idirafter`, every +configuration layer, and the replacement semantics of +`MCPU_CPP_SYSTEM_INCLUDE_PATH`. Using `-o` with this action is an error. + +### 8.15. Conditional compilation + +The directives `#if`, `#ifdef`, `#ifndef`, `#elif`, `#else`, and `#endif` are +processed as preprocessor control directives and are never copied to the +output stream, including under `-dD`. Inactive branches are skipped without +executing `#define`, `#undef`, or `#include` directives within them; nested +conditional groups are still tracked correctly. + +An `#if` expression first processes the `defined` operator, then undergoes macro +expansion, and any remaining identifiers evaluate to `0`. Arithmetic, +bitwise, comparison, and logical operators are supported, as are `?:` and +short-circuit semantics for `&&`, `||`, and `?:`. + +Starting with 0.0.26, expression syntax is parsed by a parser generated by ZUBR +4.1.0 from `src/mcpp-expr.zubr`; the same file contains the UCS-2 lexical +analyzer. `defined` preprocessing and macro expansion take place before the +parser is entered. Arithmetic semantics live in `mcpp-semantic.c/h` and do not +depend on the integer sizes of the host system. Generated `mcpp-expr.c` is +included in releases, so ZUBR is required only when the grammar changes. + +#### 8.15.1. The only evaluation width is 64 bits + +MCPU-CPP is a preprocessor, not a general-purpose language compiler. All +integer computation in conditional directives uses only 64-bit arithmetic. +The preprocessor does not perform arbitrary-width LibMPU arithmetic, floating +point, or complex-number computation. + +When a programmer does not need explicit control over the binary representation +of a literal, ordinary integer constants with optional `U`/`u` are sufficient. +For example: + +```c +#if 2 > 1 +#if 0xffffffffffffffffU > 1 +``` + +A numeric lexeme remains in UCS-2 until classification, after which its ASCII +portion is passed to LibMPU `iatoui()`. Binary `0b...`, octal `0...`, decimal, +and hexadecimal `0x...` forms are supported. A value that does not fit in 64 +bits is an error. Old C suffixes `L`, `l`, `LL`, and `ll` are not supported. + +#### 8.15.2. Width suffix `zNNN[Uu]` + +MCPU-CPP understands the width suffix shared by MCPU languages: + +```text +zNNN +ZNNN +zNNNu +zNNNU +ZNNNu +ZNNNU +``` + +`NNN` is a nonempty sequence of decimal digits and is **always** interpreted in +decimal, even with leading zeroes. Thus `z8`, `z08`, and `z008` all denote the +same width of 8 bits. + +In the general MCPU syntax, a valid width must be a power of two from 8 through +`MPU_REAL_IO_LIMIT`. MCPU-CPP, however, deliberately limits evaluation to 64 +bits: + +* `z8`, `z16`, `z32`, `z64`, in either letter case, are valid; +* `NNN > 64` is immediately an error: conditional preprocessing does not accept + numeric constants wider than 64 bits; +* if `NNN <= 64` but is not a valid power-of-two width, such as `z24`, a warning + is issued and the `zNNN` part itself is ignored; +* a following optional `U`/`u` selects unsigned interpretation and retains that + meaning even when an invalid `zNNN` has been ignored. + +The numeric preprocessing token must end after the complete suffix. An +operator or punctuation character begins the next token, so `1z32u+2`, +`(1z32u)`, and `1z32u==1` are valid. Forms such as `1z32undefined`, +`1z32ufoo`, and `1z32$foo` are errors and are not artificially split into a +number followed by a name. + +#### 8.15.3. Literal normalization + +The width suffix acts **exactly once, while the value of the literal itself is +formed**. The width is not retained in the semantic value and has no role in +later operations. + +For `VALUEzNNN`, the value is treated as a signed N-bit two's-complement number: + +1. retain the low `NNN` bits; +2. sign-extend the result to 64 bits. + +For `VALUEzNNNu`/`VALUEzNNNU`, the low `NNN` bits are retained and then +zero-extended to 64 bits. + +For example: + +```text +0x7fz8 -> 0x000000000000007f -> 127 +0x80z8 -> 0xffffffffffffff80 -> -128 +0xffz8 -> 0xffffffffffffffff -> -1 +0x80z8u -> 0x0000000000000080 -> 128 +0xffz8u -> 0x00000000000000ff -> 255 +0x1ffz8 -> 0xffffffffffffffff -> -1 +0x1ffz8u -> 0x00000000000000ff -> 255 +``` + +The last two examples are deliberate: `zNNN` specifies the width of the +**binary representation**, not a mathematical range check. Bits above N are +discarded before extension. + +After this normalization there is no remaining `z8`, `z16`, or `z32` concept +in the evaluation model. The internal value contains only a 64-bit bit pattern +and signed/unsigned state. + +#### 8.15.4. All subsequent operations are 64-bit + +After literal normalization, every arithmetic, bitwise, comparison, and logical +operation uses 64-bit operands. An operation result is not truncated back to +the width of the original suffix. Therefore: + +```text +0x7fz8 + 1 -> 128 +0xffz8u + 1 -> 256 +``` + +not `-128` and `0`. Likewise `~0xffz8u` inverts all 64 bits and gives +`0xffffffffffffff00`. + +For binary operations where signedness matters, the presence of an unsigned +operand selects 64-bit unsigned interpretation. Comparisons return `0` or `1`. +Logical `!`, `&&`, and `||` also return signed 64-bit `0` or `1`; short-circuit +evaluation does not evaluate an unselected operand. + +Shifts happen after 64-bit normalization. Right shift of a negative signed +value is arithmetic; right shift of an unsigned value is logical. For example: + +```text +0x80z8 >> 1 -> -64 +0x80z8u >> 1 -> 64 +``` + +The historical MCPU-CPP rule for a negative shift count is preserved: +`A << -N` is equivalent to `A >> N`, and `A >> -N` is equivalent to `A << N`. + +Thus `zNNN` does not turn the preprocessor into a compiler with integer +promotions over multiple widths. It only allows the binary representation of +the source literal to be stated explicitly; the expression then evaluates in +one simple 64-bit model. + +#### 8.15.5. Character constants + +A character unit has type `__mpu_uint16_t`, matching the internal UCS-2 +representation, and is zero-extended to 64 bits before evaluation. Subsequent +arithmetic is again ordinary 64-bit arithmetic. + +Conditional-compilation state is stored on a separate stack; a conditional +group may not cross an include-file boundary. + +### 8.16. Diagnostic directives `#error` and `#warning` + +MCPU-CPP supports the standard diagnostic directives: + +```text +#error message +#warning message +``` + +`#error` emits an error diagnostic using the current logical file name and line +number and immediately terminates preprocessing unsuccessfully. `#warning` +emits a warning with the same source-location information and preprocessing +continues. A preceding `#line` therefore affects both diagnostics. + +The remainder of the line after the directive name **does not undergo macro +expansion**. For example: + +```c +#define MESSAGE expanded +#warning MESSAGE +``` + +prints `MESSAGE`, not `expanded`. This distinguishes diagnostic directives from +`#if` and `#line`, where macro expansion is part of the relevant contract. + +Comments are removed by the ordinary preprocessing phase before the directive +is processed. Leading and trailing whitespace in the message is removed and +whitespace sequences between preprocessing tokens are collapsed to one space. +Whitespace inside quotes is preserved. For example: + +```c +#warning one /* comment */ two +#warning "a b" +``` + +produce `one two` and `"a b"`, respectively. Unicode text passes through the +internal UCS-2 representation and is written to the external diagnostic as +UTF-8. + +Both directives are control directives and are never copied to normal output +or to `-dD`. In an inactive `#if` branch they are ignored completely, so the +usual protective pattern behaves as expected: + +```c +#if 0 +#error this error is inactive +#endif +``` + +### 8.17. Warning control: `-Wcomment`, `-Wall`, `-Werror` + +MCPU-CPP distinguishes mandatory warnings that are part of established +preprocessing semantics from optional warning classes enabled by the user. +Warning control does not alter `-dD`, macro expansion, conditional compilation, +or include search semantics. + +`-Wcomment` and `-Wcomments` are exact aliases and enable two lexical warnings: + +* a `/*` sequence seen while already inside an open `/* ... */` comment; +* backslash-newline inside a `//` comment, causing that single-line comment to + continue physically onto the next source line. + +This optional class is disabled by default. `-Wall` enables all optional +MCPU-CPP warning classes; in version 0.0.40 this class is `-Wcomment`. +`-Wno-comment` and `-Wno-comments` disable it. As in the GNU warning model, a +more specific setting has priority over a group setting regardless of argument +order. Therefore both: + +```text +mcpu-cpp -Wall -Wno-comment file.c +mcpu-cpp -Wno-comment -Wall file.c +``` + +leave comment warnings disabled. Between settings of equal specificity, the +last option wins; for example `-Wno-comment -Wcomment` enables the class. + +`-Werror` does not enable any new warning class. It promotes to an error every +warning that would actually be emitted during that invocation, causing an +unsuccessful result. This applies both to optional comment warnings and to +existing mandatory MCPU-CPP warnings, including: + +* an active `#warning` directive; +* an invalid `zNNN` width not exceeding 64 bits; +* redefinition of a macro with a different replacement list; +* a `##` result that does not form a single preprocessing token. + +For example: + +```text +mcpu-cpp -Wcomment -Werror file.c +``` + +turns a detected comment warning into an error. By contrast, `-Werror` alone, +without `-Wcomment`/`-Wall`, does not cause MCPU-CPP to search for optional +comment warnings. + +`-Wno-error` restores ordinary warning severity. Between `-Werror` and +`-Wno-error`, which have the same specificity, the last command-line option +wins. Thus `-Werror -Wno-error` leaves warnings as warnings, while +`-Wno-error -Werror` promotes them again. + +Version 0.0.40 deliberately did not introduce `-Werror=<class>`, +`-Wno-error=<class>`, `-Wundef`, `-Wunused-macros`, `-Wtraditional`, or other +compiler-oriented classes. The MCPU-CPP warning interface remains compact and +is extended only when a class is actually needed by the preprocessing +language itself. + +### 8.18. UCS-2 identifiers + +Starting with 0.0.22, preprocessing identifiers are no longer restricted to +ASCII. Inside `mcpu-cpp`, text is already strict UCS-2, and characters are +classified by locale-independent LibMPUIO 1.0.4 functions based on Unicode +18.0.0. The first identifier character must be `_` or have the `XID_Start` +property; following characters must be `_`, `$`, or have `XID_Continue`. +`$` is an `mcpu-cpp` extension: it is allowed only after the first character +and may not start an identifier. One rule is used consistently for macro names +and parameters, `#undef`, `#ifdef`/`#ifndef`, `defined`, ordinary macro +expansion, and `#`/`##`. Names remain case-sensitive. Surrogate code units +`U+D800..U+DFFF` are not valid identifier characters. + +For example, all of these are valid: + +```c +#define АНДРЕЙ 1 +#define résumé 2 +#define ΩМЕГА 3 +#define VALUE$OLD 4 +``` + +`VALUE$OLD` is valid, while `$VALUE` is invalid because `$` is not an +identifier-start character. + +Combining marks and non-ASCII decimal digits may appear in `XID_Continue` +positions but do not automatically become valid initial characters. Numeric +constant syntax is unaffected: it follows the rules of the active language, +not Unicode `isdigit`. + +### 8.19. Command-line macros `-D` and `-U` + +Starting with 0.0.23, `-D` and `-U` are full preprocessor actions. Supported +forms are: + +```text +-DNAME +-DNAME=VALUE +-D'FUNC(a,b)=a+b' +-UNAME +``` + +`-DNAME` is equivalent to `#define NAME 1`; an `=` with an empty right-hand +side defines an empty replacement list. Function-like command-line definitions +use the same macro engine as ordinary `#define`, including parameters, `#`, +`##`, and subsequent rescanning. `-U` uses the same identifier contract as +`#undef`. `-D`/`-U` actions are executed in command-line order after predefined +macros have been installed. + +Only the payload of `-D` and `-U` is interpreted as UTF-8 and converted to +strict UCS-2. File names, `-I`, other pathname arguments, and all other +command-line arguments remain the original byte strings and undergo no Unicode +conversion. + +Starting with 0.0.25, `$` is allowed inside a macro name, but not in its first +position. When `$` is passed through a shell, the user must account for shell +rules: the shell processes `$` **before `mcpu-cpp` starts**. Single quotes +fully protect `$`, for example: + +```sh +mcpu-cpp '-DАНДРЕЙ$_Y=62' input.c +``` + +Without quotes, `$` must be escaped: + +```sh +mcpu-cpp -DАНДРЕЙ\$_Y=62 input.c +``` + +or double quotes may be used with escaping: + +```sh +mcpu-cpp -D"АНДРЕЙ\$_Y=62" input.c +``` + +The unprotected form: + +```sh +mcpu-cpp -DАНДРЕЙ$_Y=62 input.c +``` + +does not pass the spelling literally: `$...` is expanded by the shell first, +and `mcpu-cpp` receives the already modified `argv`. Inside single quotes, a +backslash before `$` is unnecessary and would become an ordinary argument +character. + +For command-line `-D`, the left-hand side up to the first `=` is parsed as a +separate macro declarator. If an invalid tail occurs after a valid name (or a +completed formal-parameter list of a function-like macro) but before `=`, that +tail is silently discarded and **never becomes part of the replacement list**. +For example: + +```text +-D'АНДРЕЙ@XYZ=62' +``` + +is equivalent to: + +```c +#define АНДРЕЙ 62 +``` + +not the invalid `#define АНДРЕЙ @XYZ 62`. The same valid-identifier-prefix rule +applies to `-U`. If the very first character is not a valid identifier-start +character, such as `$` or a digit, the definition remains an error. + +Under `-dD`, definitions originating from `-D` are marked separately from +predefined macros: + +```text +# 0 "<command-line>" +#define NAME value +``` + +while predefined macros continue to use `<built-in>`. + +### 8.20. Public command-line interface + +`mcpu-cpp` supports only the current options documented by `--help`. Obsolete +compatibility flags do not form a hidden interface and are diagnosed as +`unknown option`. `-E` is the exception: it is silently accepted and ignored +because a compiler driver may pass it while invoking a standalone +preprocessor. + +`--object-suffix SUFFIX` selects the object-target suffix used when generating +Make dependencies; the argument is mandatory. + +## 9. Build system and generators + +Project-owned Autoconf macros live in the root `acsite.m4`. The `m4/` directory +is reserved for external/vendor M4 files. This follows the convention used by +MCPU libraries and keeps project configure code separate from imported macros. + +The `#if` expression parser is generated by ZUBR 4.1.0 from +`src/mcpp-expr.zubr`. Release archives contain both the grammar and the already +generated `src/mcpp-expr.c`, so an ordinary release build does not require +ZUBR. After the grammar changes, a developer build uses the normal Automake +rule: + +```text +zubr -vl -s -Bmcpp_ -o mcpp-expr.c mcpp-expr.zubr +``` + +Before a release, generated C must correspond to the grammar, and the full test +suite plus `make distcheck` must complete without errors. + +### 9.1. Developer bootstrap and the Git source tree + +Starting with 0.0.50, the root `./bootstrap` script makes it unnecessary to +store files in Git when they are completely reproducible from source. The +script first generates `src/mcpp-expr.c` from `src/mcpp-expr.zubr` using ZUBR +4.1.0, then runs `aclocal`, `autoheader`, `automake`, and `autoconf` in the +style used by LibMPU and LibMPUIO. `--target-dest-dir=DIR` selects a target +ROOTFS used for the system Autoconf macro/include directories. + +This convention applies specifically to the developer Git tree. **Release +archives remain self-contained**, exactly as before: they contain `configure`, +`Makefile.in`, Automake helper scripts, and the generated `src/mcpp-expr.c`. +Therefore an ordinary release build requires neither a preliminary `bootstrap` +run nor ZUBR. + +The root `.gitignore` lists reproducible bootstrap files and ordinary +configure/build state. It does not change the existing release/build model; it +only allows a cleaner Git repository. + +## 10. GNU-compatible features + +`mcpu-cpp` is an independent MCPU preprocessor, but it intentionally follows +GNU CPP behavior for a number of well-known operations. Compatibility applies +to the documented features; it does not imply complete CLI or language +interchangeability with GCC. + +GNU-compatible behavior is used for, in particular: + +* object-like and function-like macros, macro rescan, `#`, and `##`; +* variadic macros `...` / `__VA_ARGS__` and standard `__VA_OPT__`; +* `#if`, `#ifdef`, `#ifndef`, `#elif`, `#else`, `#endif`, and `defined`; +* `#include`, `#include_next`, `#pragma once`, `#line`, and GNU linemarkers; +* compact output mapping: up to seven invisible lines are represented by + newlines, while a gap of eight or more uses a corrective linemarker; +* forced files `-include` / `-imacros` and dependency options `-M`, `-MM`, + `-MD`, `-MMD`, `-MF`, `-MT`, `-MQ`, and `-MG`; +* warning controls `-w`, `-Wall`, `-Werror`, and the supported `-Wcomment` forms. + +MCPU-specific facilities, including `#lang` / `#endlang`, the `zNNN` numeric +suffix, and ABI predefined macros, remain native `mcpu-cpp` extensions. diff --git a/doc/mcpu-cpp-ru.md b/doc/mcpu-cpp-ru.md new file mode 100644 index 0000000..f8f762a --- /dev/null +++ b/doc/mcpu-cpp-ru.md @@ -0,0 +1,2167 @@ +# mcpu-cpp + +`mcpu-cpp` — препроцессор языков программирования MCPU. Он является +самостоятельным компонентом экосистемы LibMPU/LibMPUIO/LibMCPU и не привязан +к названию одного конкретного языка: активный язык выбирается директивой +`#lang`. + +Этот документ задаёт нормативное поведение `mcpu-cpp`: текстовую модель, +директивы, macro engine, include pipeline, конфигурацию, диагностику и +генерацию зависимостей для инструментов MCPU. + +## 1. Текстовая модель + +Внешние исходные файлы и конфигурационные файлы имеют кодировку UTF-8. +UTF-8 должен быть корректным. Для исходных программ проверка принадлежности +символов диапазону UCS-2 выполняется после удаления комментариев: поэтому +корректный Unicode scalar value выше `U+FFFF` допустим внутри комментария, но +остаётся ошибкой в программном тексте. После этой стадии исходный текст +обрабатывается как последовательность `__mpu_char16_t`. Входной UTF-8 BOM +допускается и удаляется. Встроенный NUL в исходном файле запрещён. + +Переводы строк `CRLF` и `CR` нормализуются в `LF`. + +## 2. Действия, выполняемые независимо от директив + +`mcpu-cpp` выполняет несколько преобразований до разбора +директив. + +### 2.1. Backslash-newline + +Последовательность `\\` непосредственно перед переводом строки удаляется до +распознавания комментариев, директив и макросов. Поэтому, например, + +```text +#defi\ +ne FOO 10\ +20 +``` + +эквивалентно логической строке + +```text +#define FOO 1020 +``` + +При этом физические номера строк продолжают учитываться при формировании +текущей позиции; если пользователь не менял её директивой `#line`, они и будут +видны в генерируемых line marker-ах. + +### 2.2. Комментарии + +Комментарии `/* ... */` и `// ...` удаляются до последующей обработки. Там, +где это необходимо для разделения соседних токенов, сохраняется пробельный +разделитель. Если комментарий завершает непустую строку, после его удаления не +сохраняются ни синтетический разделитель, ни пробелы, предшествовавшие +комментарию: строка заканчивается последним значащим символом. Это относится и +к многострочному комментарию, начавшемуся после программного текста. Если после +удаления комментария строка вообще не содержит ничего кроме пробелов, она +становится действительно пустой строкой. При этом комментарий между двумя +токенами по-прежнему оставляет необходимый разделитель и не склеивает их. +Переводы строк сохраняются, чтобы не разрушать координаты исходного текста. + +Комментарий не распознаётся внутри строковой или символьной константы. Для +языка `diff` апостроф не считается началом символьной константы, поскольку +используется в обозначениях производных. + +В буквальном аргументе `#include <...>` последовательности `/*` и `//` +рассматриваются как часть имени файла. + +## 3. Директивы и выходной поток + +Директива начинается символом `#`, если до него в логической строке находятся +только пробельные символы или комментарии. Между `#` и именем директивы +допускаются пробелы. + +Служебная информация о позиции в выходном потоке представлена в форме GNU +**line marker**: + +```text +# номер "имя-файла" [флаги] +``` + +Это не входная директива `#line`. При входе во включаемый файл к line marker-у +добавляется флаг `1`, а при возврате в файл, содержащий `#include`, — флаг `2`. +Эти значения имеют тот же смысл, что и в GNU CPP: `1` означает вход в новый +файл, `2` — возврат в предыдущий файл. Флаг `2` не является числом или уровнем +вложенности. + +Например: + +```text +# 1 "main.c" +# 1 "defs.h" 1 +... +# 2 "main.c" 2 +``` + +Входная директива + +```text +#line 62 "main.y" +``` + +сама в выходной поток не копируется. Она изменяет логические значения +`__LINE__` и `__FILE__` для последующего текста, а в выходе представляется +line marker-ом: + +```text +# 62 "main.y" +``` + +Аргументы `#line` предварительно подвергаются macro expansion, как в +принятой модели line control. Если после такого `#line` происходит `#include`, то после +возврата marker получает флаг `2`, например `# 65 "main.y" 2`. Имя, заданное +через `#line`, становится логическим именем для `__FILE__` и line marker-ов; оно +не меняет каталог, относительно которого ищется quoted `#include`. + +Директивы препроцессора имеют только канонические английские имена. Unicode остаётся полностью допустимым в идентификаторах, строках, комментариях и другом пользовательском тексте. + +## 4. Заголовочные файлы + +Поддерживаются: + +```text +#include "file" +#include <file> +#include_next "file" +#include_next <file> +#pragma once +``` + +Для обычного `#include "file"` первым всегда проверяется каталог **физического** +текущего исходного файла. Логическое имя, установленное через `#line`, на этот +шаг не влияет. Для `#include <file>` каталог текущего файла не проверяется. + +### 4.1. Перемещаемый корень MCPU как общий принцип экосистемы + +Начиная с выпуска 0.0.37 каталог установки MCPU **не содержит версию +конкретного инструмента** и не является абсолютной runtime-константой, +зашитой в бинарный файл. Версия относится к самому `mcpu-cpp`, `mcpu-as`, +`mcpu-ld`, `mcpu-run` или библиотеке, но не определяет корень единой среды +MCPU. + +При типичной конфигурации: + +```text +./configure --prefix=/usr --libdir=/usr/lib64 +``` + +`make install` создаёт дерево: + +```text +/usr/lib64/mcpu/ +├── bin/ +│ └── mcpu-cpp +├── etc/ +│ └── mcpu-cpp.conf +├── include/ +│ ├── diff/ +│ ├── dift/ +│ ├── alg/ +│ ├── as/ +│ ├── avm/ +│ └── acs/ +└── lib/ # общий каталог будущих библиотек MCPU +``` + +Публичное имя программы находится в `$bindir`: + +```text +/usr/bin/mcpu-cpp -> ../lib64/mcpu/bin/mcpu-cpp +``` + +Абсолютный `/usr/lib64/mcpu` при этом **не является частью runtime ABI +MCPU-CPP**. Он используется только `make install` как выбранное configure-time +место размещения файлов. + +При каждом обычном запуске MCPU-CPP определяет фактический путь собственного +исполняемого файла через Linux `/proc/self/exe`. Символическая ссылка публичной +команды не мешает этому: `/proc/self/exe` указывает на реально выполняемый +бинарный файл. Если `/proc/self/exe` недоступен, используется резервное +разрешение `argv[0]` через `PATH` и `realpath(3)`; возврата к зашитому +configure-time installation root нет. + +Для бинарного файла: + +```text +<root>/bin/mcpu-cpp +``` + +runtime-корень определяется как: + +```text +executable = <root>/bin/mcpu-cpp +executable dir = <root>/bin +MCPU runtime root = <root> +``` + +Из него автоматически выводятся: + +```text +<root>/etc/mcpu-cpp.conf +<root>/include +``` + +Следовательно всё дерево можно физически перенести, например из: + +```text +/usr/lib64/mcpu/ +``` + +в: + +```text +/opt/mcpu-test/ +``` + +или: + +```text +$HOME/devel/mcpu-next/ +``` + +и `<new-root>/bin/mcpu-cpp` без переконфигурирования начнёт использовать +`<new-root>/etc/mcpu-cpp.conf` и `<new-root>/include`. Старый абсолютный путь +не сохраняется ни в runtime default, ни в штатном `mcpu-cpp.conf`. + +Это не частная особенность препроцессора, а **общий принцип экосистемы MCPU**. +Будущие `mcpu-as`, `mcpu-ld`, `mcpu-run`, библиотеки, CRT и другие компоненты +должны разделять один перемещаемый корень: + +```text +<root>/bin +<root>/etc +<root>/include +<root>/lib +``` + +Их собственные версии могут отличаться, но согласованность конкретной среды +MCPU определяется тем, что все компоненты находятся в одном runtime tree, а +не совпадением version suffix в именах каталогов. + +### 4.2. Runtime defaults, уровни конфигурации и системный include root + +До чтения любого конфигурационного файла MCPU-CPP создаёт runtime-derived +значение: + +```text +MCPU_CPP_SYSTEM_INCLUDE_PATH = <runtime-root>/include +``` + +После этого конфигурационные слои применяются в порядке возрастающего +приоритета: + +```text +runtime-derived defaults + ↓ +<runtime-root>/etc/mcpu-cpp.conf + ↓ +/etc/mcpu/mcpu-cpp.conf + ↓ +$HOME/.mcpu/mcpu-cpp.conf +``` + +`<runtime-root>/etc/mcpu-cpp.conf` устанавливается вместе с MCPU-CPP, но сам +файл намеренно не содержит абсолютного штатного `MCPU_CPP_SYSTEM_INCLUDE_PATH`: +иначе перенос всего дерева восстановил бы старый путь. `/etc/mcpu/mcpu-cpp.conf` +является необязательным machine-wide override: `make install` каталог +`/etc/mcpu` не создаёт. Домашний `$HOME/.mcpu/mcpu-cpp.conf` также необязателен, +не версионируется и имеет максимальный config-приоритет. + +Если одна переменная определена несколько раз, побеждает последнее +определение, включая пустое. Поэтому `MCPU_CPP_SYSTEM_INCLUDE_PATH` остаётся +полностью заменяемым system root. Например: + +```text +MCPU_CPP_SYSTEM_INCLUDE_PATH = $HOME/mcpu-next/include; +``` + +полностью заменяет runtime-derived `<runtime-root>/include`. Для активного +`#lang "as"` тогда проверяются: + +```text +$HOME/mcpu-next/include/as +$HOME/mcpu-next/include +``` + +Штатные language-подкаталоги всегда выводятся самим препроцессором из одного +root; переменных вида `MCPU_CPP_SYSTEM_<LANG>_INCLUDE_PATH` нет. + +Пустое effective значение: + +```text +MCPU_CPP_SYSTEM_INCLUDE_PATH = ; +``` + +удаляет configured system stage полностью. Более приоритетный config может +после этого снова включить его непустым значением. + +`--config-file FILE` применяет явно выбранный файл поверх runtime-derived +default. `--no-config` отключает **только чтение файлов конфигурации**: +`<runtime-root>/etc/mcpu-cpp.conf`, `/etc/mcpu/mcpu-cpp.conf` и +`$HOME/.mcpu/mcpu-cpp.conf` не читаются, но `<runtime-root>/include` остаётся +штатным system root. Только `-nostdinc` удаляет effective standard-system tree +из include search для конкретного запуска; явно переданный `-isystem` при этом +остаётся command-line каталогом. + +### 4.3. Нормативный порядок поиска include-файлов + +Порядок поиска является частью контракта MCPU-CPP. Явно заданные параметры +командной строки имеют приоритет над persistent configuration. После +необязательного каталога текущего физического файла эффективная цепочка имеет +строго следующий вид: + +```text +explicit -I + ↓ +explicit -isystem + ↓ +MCPU_CPP_<LANG>_INCLUDE_PATH + ↓ +MCPU_CPP_INCLUDE_PATH + ↓ +MCPU_CPP_SYSTEM_INCLUDE_PATH/<lang> + ↓ +MCPU_CPP_SYSTEM_INCLUDE_PATH + ↓ +explicit -idirafter + ↓ +MCPU_CPP_AFTER_INCLUDE_PATH +``` + +Элементы, которых нет или которые не содержат требуемого файла, пропускаются. + +`MCPU_CPP_<LANG>_INCLUDE_PATH` — свободно настраиваемые пользователем +language-specific path-list'ы: + +```text +MCPU_CPP_DIFF_INCLUDE_PATH +MCPU_CPP_DIFT_INCLUDE_PATH +MCPU_CPP_ALG_INCLUDE_PATH +MCPU_CPP_AS_INCLUDE_PATH +MCPU_CPP_AVM_INCLUDE_PATH +MCPU_CPP_ACS_INCLUDE_PATH +``` + +Пользователь полностью распоряжается именами и расположением этих каталогов. +`MCPU_CPP_INCLUDE_PATH` — общий пользовательский path-list, видимый во всех +языковых состояниях. + +`-idirafter` и `MCPU_CPP_AFTER_INCLUDE_PATH` являются общим fallback-карманом. +MCPU-CPP не строит для них автоматических `<lang>`-подкаталогов. Пользователь +сам организует их внутреннюю структуру и при необходимости пишет, например: + +```text +#include <vendor/device.h> +``` + +Именно semantic class, а не порядок появления разных классов в argv/config, +определяет приоритет. Внутри одного класса сохраняется порядок добавления. + +### 4.4. `#include_next` и wrapper headers + +`#include_next` предназначен прежде всего для заголовков-обёрток (wrapper +headers). Он позволяет поставить локальный header раньше системного, изменить +локальную политику и затем продолжить поиск одноимённого header по нормативной +цепочке без копирования системного файла и без абсолютного имени. + +Например: + +```text +mcpu-cpp -isystem $HOME/mcpu-wrapper ... +``` + +и `$HOME/mcpu-wrapper/math.h`: + +```text +#ifndef SOME_SYSTEM_MACRO +#define SOME_SYSTEM_MACRO temporary_value +#define REMOVE_SOME_SYSTEM_MACRO 1 +#endif + +#include_next <math.h> + +#ifdef REMOVE_SOME_SYSTEM_MACRO +#undef SOME_SYSTEM_MACRO +#undef REMOVE_SOME_SYSTEM_MACRO +#endif +``` + +Если домашний config одновременно задаёт: + +```text +MCPU_CPP_SYSTEM_INCLUDE_PATH = $HOME/mcpu-next/include; +``` + +wrapper найденный через `-isystem` продолжит `#include_next` уже через +configured user paths, затем через +`$HOME/mcpu-next/include/<lang>` и `$HOME/mcpu-next/include`; старое system tree исходного места установки при этом не участвует. Именно такой сценарий позволяет +системному разработчику или тестеру жить в собственной sandbox. + +MCPU-CPP хранит конкретный **физический элемент effective search chain**, из +которого найден текущий header. `#include_next` начинает со следующего элемента. +Формы `"file"` и `<file>` для `#include_next` эквивалентны; каталог текущего +файла повторно не проверяется. Если текущий файл найден обычным quoted-поиском +относительно содержащего файла и не имеет search-chain provenance, +`#include_next` начинает с первого элемента configured chain. + +Операнд может быть получен macro expansion. Логическое имя после `#line` не +влияет на физический provenance. Если после текущего entry подходящего файла +нет, preprocessing завершается ошибкой. + +### 4.5. `#pragma once` + +Активная директива + +```text +#pragma once +``` + +помечает **физический файл** как уже обработанный в текущем запуске +MCPU-CPP. При последующей попытке включить тот же физический файл его +содержимое повторно не обрабатывается. Сама директива потребляется +препроцессором и в выходной поток не копируется, в том числе при `-dD`. + +Идентичность определяется по паре `st_dev`/`st_ino`, полученной файловой +системой, а не по строковому имени пути. Поэтому один и тот же файл не может +обойти `#pragma once`, если он достигнут как `./file.h`, через символическую +ссылку или через другое жёсткое имя (hard link). Логическое имя после `#line` +также не влияет на эту физическую идентичность. + +Пометка действует сразу в момент обработки активной директивы. Поэтому +заголовок может после `#pragma once` включить самого себя: повторное включение +будет пропущено и рекурсия не возникнет. Директива внутри неактивной ветви +условной компиляции никакого действия не имеет. + +MCPU-CPP распознаёт только точную форму `#pragma once` с необязательными +пробелами. Остальные `#pragma` не интерпретируются препроцессором и сохраняются +для последующих стадий компиляции; например, `#pragma pack(...)` продолжает +передаваться в выходной поток. + +`#pragma once` дополняет, но не изменяет нормативную search-chain +`#include`/`#include_next`: сначала обычный механизм поиска находит физический +файл, затем registry `once` решает, надо ли обрабатывать его содержимое. + +### 4.6. Принудительные файлы: `-imacros FILE` и `-include FILE` + +Опции командной строки + +```text +-imacros FILE +-include FILE +``` + +обрабатывают файл до главного input. Они используют обычный preprocessing +engine, а не отдельный облегчённый parser. + +Нормативный порядок начала translation unit: + +```text +predefined macros + -> -D/-U в порядке командной строки + -> все -imacros в порядке командной строки + -> все -include в порядке командной строки + -> главный input +``` + +Таким образом, взаимное расположение `-imacros` и `-include` в `argv` не +перемешивает эти две группы: **все** `-imacros` всегда выполняются раньше +**всех** `-include`. + +`-imacros FILE` полностью обрабатывает файл: его `#define`/`#undef`, +условные директивы, `#lang`/`#endlang`, `#include`, `#include_next`, +`#pragma once` и диагностика имеют обычную семантику. Однако весь normal +preprocessing output этого forced-файла, включая line markers и текст +вложенных headers, отбрасывается. Полученное состояние macro table и других +preprocessing-механизмов сохраняется для последующих forced-файлов и главного +input. + +`-include FILE` использует тот же механизм, но normal output сохраняется, как +если бы найденный header был включён непосредственно перед главным source. +Forced include является настоящей include-границей: внутри него +`__INCLUDE_LEVEL__ == 1`, внутри включённого им header уровень равен `2`, а +главный input остаётся на уровне `0`. `__BASE_FILE__` внутри forced-файлов +остаётся именем главного input. + +Абсолютный operand forced-файла используется непосредственно. Относительный +operand ищется сначала в **current working directory**, затем по обычной +include-chain: + +```text +explicit -I +explicit -isystem +MCPU_CPP_<LANG>_INCLUDE_PATH +MCPU_CPP_INCLUDE_PATH +MCPU_CPP_SYSTEM_INCLUDE_PATH/<lang> +MCPU_CPP_SYSTEM_INCLUDE_PATH +explicit -idirafter +MCPU_CPP_AFTER_INCLUDE_PATH +``` + +Каталог главного input не получает специального приоритета при поиске operand +`-imacros`/`-include`. После нахождения forced-файла обычный quoted +`#include "file"` внутри него снова разрешается относительно физического +каталога этого файла. Если forced-файл найден через элемент include-chain, +его provenance сохраняется и `#include_next` продолжает поиск со следующего +элемента цепочки. + +Forced-файлы и реально достигнутые из них headers входят в обычный physical +dependency registry. Их user/system classification определяется тем же +search provenance, поэтому `-MM`/`-MMD` фильтруют system forced headers так же, +как обычные system headers. Отсутствующий forced-файл является ошибкой. + + +### 4.7. Генерация зависимостей: `-M`, `-MM`, `-MG`, `-MD`, `-MMD`, `-MF`, `-MT`, `-MQ` + +Опции `-M` и `-MM` используют **тот же самый проход include pipeline**, что и +обычная preprocessing. Отдельного повторного поиска заголовков не выполняется. +Поэтому dependency graph автоматически наследует нормативный порядок путей, +`#include_next`, macro-expanded include operands, conditional compilation и +`#pragma once`. + +`-M` подавляет обычный preprocessing output и выводит одно правило Make: + +```make +file.o: file.c header1.h header2.h +``` + +В список входят главный source-файл и все реально достигнутые физические +headers, включая system headers. Один физический файл записывается один раз; +идентичность определяется как `st_dev + st_ino`, поэтому другое относительное +имя, symbolic link или hard link не создают дополнительную dependency. Имя, +назначенное директивой `#line`, является только logical source name и в +dependency list не попадает. + +`-MM` строит тот же граф, но исключает system dependencies. System-контекстом +считаются headers, найденные через explicit `-isystem`, configured system tree +`MCPU_CPP_SYSTEM_INCLUDE_PATH/<lang>` / `MCPU_CPP_SYSTEM_INCLUDE_PATH`, explicit +`-idirafter` и `MCPU_CPP_AFTER_INCLUDE_PATH`, а также вся ветвь headers, +включённая непосредственно или косвенно из такого system header. Форма +`#include "file"` или `#include <file>` сама по себе не определяет system-ness. +Если один и тот же физический файл был достигнут из system-ветви, но затем +также включён непосредственно из user-контекста, он остаётся пользовательской +dependency и присутствует в `-MM`. + +Default target образуется из basename главного source-файла: его suffix +заменяется object suffix (`.o` по умолчанию). Пути и target экранируются для +Make. Для stdin используется GNU-подобная форма `-: -`. + +`-MD` и `-MMD` используют тот же dependency graph, но, в отличие от `-M` и +`-MM`, **не подавляют обычный preprocessing output**. `-MD` включает system +headers, как `-M`; `-MMD` применяет user-only фильтр `-MM`. Это позволяет одним +проходом получить и препроцессированный текст, и side-effect dependency file. + +Если `-MF` не задан, side-effect режим выбирает имя `.d` автоматически: + +* без `-o` из basename входного файла удаляется suffix и добавляется `.d`; + каталоги входного pathname в имя dependency-файла не переносятся; +* при обычном `-o FILE` suffix output-файла заменяется на `.d`; +* для stdin используется имя `-.d`. + +`-MF FILE` переопределяет автоматическое имя dependency-файла. Значение +`-MF -` означает stdout. `-MF` работает также с dependency-only `-M`/`-MM`; +в этом случае оно имеет приоритет над обычным destination для make-rule. Само +по себе `-MF` без одного из `-M`, `-MM`, `-MD`, `-MMD` является ошибкой. + +Семантика намеренно следует GNU CPP: `-MD`/`-MMD` не принимают собственный +аргумент, а `-MF` является отдельной опцией назначения dependency output. + +`-MT TARGET` заменяет автоматический target правила строкой `TARGET` **точно +как она передана**. Make quoting при этом не выполняется. Поэтому один argument +`-MT` может сам содержать несколько targets, разделённых пробелами: + +```text +-MT 'obj/a.o obj/a.pic.o' +``` + +и повторные `-MT` также добавляют targets одного и того же правила: + +```text +-MT obj/a.o -MT obj/a.pic.o +``` + +`-MQ TARGET` имеет ту же семантику выбора target, но экранирует специальные +для Make символы. Например, + +```text +-MQ '$(OBJDIR)/foo.o' +``` + +даёт левую часть правила: + +```make +$$(OBJDIR)/foo.o: +``` + +Поддерживаются как отдельные arguments (`-MT TARGET`, `-MQ TARGET`), так и +attached forms (`-MTTARGET`, `-MQTARGET`). + +Если задан хотя бы один `-MT` или `-MQ`, автоматический default target не +выводится. В частности, `--object-suffix` влияет только на автоматический +target и не переписывает явно заданные targets. Если явных targets нет, +default target экранируется для Make так же, как при `-MQ`. + +Разрешены повторные и смешанные `-MT`/`-MQ`. В соответствии с GNU CPP сначала +выводятся все `-MT` targets в их command-line order, затем все `-MQ` targets в +их command-line order. Все они образуют левую часть **одного** dependency +rule. + +`-MT` и `-MQ` имеют смысл только вместе с одним из dependency-generation +режимов `-M`, `-MM`, `-MD` или `-MMD`. Без такого режима это ошибка командной +строки. + +`-MG` изменяет только обработку **отсутствующих** include-файлов при +dependency-only режимах `-M` и `-MM`. Без `-MG` неразрешённый `#include` остаётся +ошибкой. С `-M -MG` или `-MM -MG` отсутствующий header считается будущим +generated file: preprocessing не завершается ошибкой, а operand директивы +добавляется в dependency rule **ровно в том виде, который получен после macro +expansion**, без приписывания предполагаемого include-directory. Например: + +```text +#include "generated.h" +``` + +при `-M -MG` добавляет dependency `generated.h`, даже если такого файла ещё нет. +Macro-expanded include ведёт себя аналогично: dependency получает уже +развёрнутое имя. `-MG` разрешён только вместе с `-M` или `-MM`; комбинации с +`-MD`/`-MMD` и использование без dependency-only режима являются ошибкой +командной строки. + +Unresolved dependencies интегрированы в **тот же упорядоченный dependency +registry**, что и физически найденные файлы, но образуют отдельный identity +domain. Для найденного файла registry по-прежнему использует `st_dev/st_ino` и +physical provenance. Для отсутствующего файла этих данных нет, поэтому `-MG` +entry не выполняет `stat()` и дедуплицируется по точному тексту include operand. +Это принципиально: наличие одноимённого файла в CWD не должно превращать +неразрешённый `<name>` в ложное physical совпадение, если angle-search этот файл +не находил. Разные unresolved spellings (`generated.h` и `./generated.h`) +считаются разными dependencies. + +Для `-MM` unresolved dependency получает user/system class из контекста поиска: +отсутствующий `<file>` является system-class, отсутствующий `"file"` — user-class, +если сама включающая единица не является system header; любой missing include, +достигнутый из system header, остаётся system-class. При повторении одного и того +же unresolved operand сохраняется классификация его первого появления, что +соответствует GNU CPP. Физически найденные зависимости сохраняют прежнее правило: +если один и тот же inode позднее достигается из user-контекста, он перестаёт быть +system-only. + +`-MG` распространяется также на отсутствующие command-line forced files +`-include FILE` и `-imacros FILE`: их operand заносится в unresolved registry как +user dependency без синтетического search prefix. При наличии реального файла +`-include`/`-imacros` продолжают использовать обычный physical dependency +registry и search provenance. + + +## 5. Переключение языков + +Препроцессор запускается в состоянии `0`. Это безымянный основной C-подобный +язык и он не является допустимым аргументом `#lang`. + +Допустимые языки: + +| Имя | Назначение | +|---|---| +| `diff` | дифференциальные уравнения | +| `dift` | разностные уравнения | +| `alg` | алгебраические уравнения | +| `as` | MCPU assembler (`mcpu-as`) | +| `avm` | схемы аналоговых вычислительных машин | +| `ACS` | структурные схемы систем автоматического управления | + +После `#lang` обязательна строковая константа с одним непустым словом: + +```text +#lang "diff" +``` + +Имя проверяется только по внутреннему списку языков выше и сравнивается без +учёта ASCII-регистра. Поэтому `"diff"`, `"Diff"`, `"DIFF"` и `"dIfF"` +эквивалентны при выборе языка. Исходное написание внутри кавычек при этом +сохраняется в выходном потоке. + +Пробелы внутри строковой константы запрещены: `" diff"`, `"diff "` и +`"di ff"` являются ошибками. Escape-последовательности внутри неё не +разбираются. Закрывающая кавычка обязана находиться на той же физической строке +исходного файла. После неё до конца строки допустимы только пробельные символы. + +Внешние пробелы директивы нормализуются. Например: + +```text + # lang "DiFf" +``` + +превращается в: + +```text +#lang "DiFf" +``` + +`#lang` помещает новый язык в стек, `#endlang` восстанавливает предыдущий. +Стек не сбрасывается при `#include`, поэтому начало и конец языкового блока +могут находиться в разных файлах. Директивы `#lang` и `#endlang` сохраняются в +выходном потоке для последующего frontend dispatcher; `#lang` сохраняется в +нормализованной форме. + +## 6. Простые макроопределения + +Начиная с 0.0.4 поддерживаются object-like macros: + +```text +#define BUFFER_SIZE 1024 +#define NAME value +#define EMPTY +``` + +Директива `#define` сама в выходной поток не попадает. В обычном тексте +идентификатор-макро заменяется его replacement list. Replacement затем снова +просматривается на макроимена, поэтому допускается каскадное раскрытие: + +```text +#define A B +#define B 10 +A +``` + +даёт `10`. + +Во время раскрытия конкретное макро временно блокируется. Поэтому +самоссылочные и взаимно-рекурсивные определения не вызывают бесконечной +рекурсии. + +Макроимена не раскрываются внутри строковых и символьных констант. Для `diff` +апостроф сохраняет специальную языковую семантику и не защищает последующий +текст как C character constant. + +Многострочное определение через backslash-newline поддерживается, поскольку +splice выполняется раньше `#define`. + +### 6.1. `#undef` + +```text +#undef NAME +``` + +удаляет object-like macro. Отмена несуществующего определения не является +ошибкой. + +### 6.2. Вычисляемый `#include` + +Аргумент `#include`, который не начинается непосредственно с `"` или `<`, +сначала проходит macro expansion. Поэтому допустимо: + +```text +#define HEADER <diff/model.h> +#include HEADER +``` + +или + +```text +#define HEADER "local.h" +#include HEADER +``` + +Результат раскрытия обязан иметь форму `"file"` или `<file>`. + +## 7. Макро с аргументами + +Macro engine поддерживает +function-like macros: + +```text +#define идентификатор( список аргументов ) текст +``` + +Открывающая скобка в определении должна идти **непосредственно** +после имени макро. Поэтому + +```text +#define F(X) X +``` + +задаёт макро с аргументом, а + +```text +#define F (X) +``` + +задаёт простое object-like macro со строкой замены `(X)`. + +В месте использования между именем function-like macro и открывающей скобкой +пробельные символы допустимы. Если `(` не следует, идентификатор не считается +вызовом данного макро и остаётся в выходном тексте. + +Для обычного function-like macro число фактических аргументов должно совпадать +с числом формальных. Для variadic macro должны присутствовать все фиксированные +аргументы, а variadic tail может содержать произвольное число аргументов, включая +пустой tail. При разборе списка фактических аргументов вложенные круглые скобки +учитываются; запятая внутри них не разделяет аргументы. Квадратные скобки такого +свойства не имеют — это является частью принятой семантики macro expansion. + +Например, + +```text +#define min(X, Y) ((X) < (Y) ? (X) : (Y)) +min(1, 2) +``` + +даёт + +```text +((1) < (2) ? (1) : (2)) +``` + +Перед подстановкой обычный фактический аргумент сам проходит macro expansion. +Поэтому каскадные и вложенные вызовы работают естественно: + +```text +#define A 7 +#define min(X, Y) ((X) < (Y) ? (X) : (Y)) +min(min(A, 3), 10) +``` + +Формальный параметр может встречаться в replacement list произвольное число +раз. Это означает, что выражение с побочным эффектом в +фактическом аргументе также может быть вычислено несколько раз уже последующим +компилятором; препроцессор не пытается исправлять такую программу. + +Поддерживаются макро без формальных параметров: + +```text +#define READY() 1 +``` + +Они раскрываются только как вызов `READY()` (пробел между именем и `(` при +использовании допустим), но самостоятельный идентификатор `READY` не +раскрывается. + +Имена формальных параметров должны быть различны. Незавершённый список, +неверная пунктуация, недостаточное или избыточное число фактических аргументов +диагностируются как ошибки. + +### 7.1. Stringification `#` + +Поддерживается оператор +stringification (`#`) для параметров function-like macro: + +```text +#define STR(X) #X +STR(alpha + beta) +``` + +даёт + +```text +"alpha + beta" +``` + +Stringification использует **сырой фактический аргумент до macro expansion**. +Поэтому: + +```text +#define A 7 +#define STR(X) #X +#define XSTR(X) STR(X) + +STR(A) -> "A" +XSTR(A) -> "7" +``` + +Ведущие и завершающие пробелы аргумента удаляются. Последовательности +пробельных символов внутри аргумента сворачиваются в один пробел, кроме +пробелов внутри строковых/символьных токенов соответствующего активного +языка. Двойные кавычки и обратные косые черты внутри quoted tokens экранируются +так, чтобы результат оставался одной корректной строковой константой. + +Оператор `#` в replacement list function-like macro обязан непосредственно или +через пробельные символы ссылаться на имя формального параметра. Внутри quoted +token символ `#` оператором не является. Пустой фактический аргумент допустим и +stringify-ится как `""`. + +### 7.2. Token concatenation `##` + +Начиная с 0.0.21 поддерживается оператор token concatenation (`##`) в модели, +согласованной с GNU CPP и механизмом `collect_expansion()` / `macroexpand()` +macro engine. Оператор объединяет два соседних preprocessing token в +один token, после чего получившийся replacement list снова проходит macro +expansion. + +Например: + +```text +#define CAT(A, B) A ## B +CAT(foo, bar) +``` + +даёт `foobar`. Склеивание может образовывать identifier, preprocessing number +или многосимвольный punctuator. Поэтому, например, допустимы: + +```text +CAT(1.5, e3) -> 1.5e3 +CAT(+, =) -> += +``` + +Если формальный параметр непосредственно примыкает к `##`, его фактический +аргумент подставляется **без предварительного macro expansion**. Это тот же +raw-argument принцип, который используется для stringification. Для получения +сначала expansion, а затем concatenation применяется обычный двухуровневый +приём GNU CPP: + +```text +#define AFTERX(X) X_ ## X +#define XAFTERX(X) AFTERX(X) +#define TABLESIZE 1024 +#define BUFSIZE TABLESIZE + +AFTERX(BUFSIZE) -> X_BUFSIZE +XAFTERX(BUFSIZE) -> X_1024 +``` + +Пустой фактический аргумент рядом с `##` ведёт себя как placemarker: сам по +себе он не добавляет token, а `##` с такой стороны не изменяет оставшийся +операнд. Если фактический аргумент содержит несколько preprocessing tokens, +склеивается только крайний token, непосредственно соседний с `##`; остальные +tokens сохраняются и затем участвуют в общем rescan. + +`#` и `##` могут использоваться в одном function-like macro, например: + +```text +#define COMMAND(NAME) #NAME | NAME ## _command +``` + +При этом `#NAME` использует raw spelling аргумента для stringification, а +`NAME ## _command` — тот же raw argument для concatenation. + +`##` внутри quoted token оператором не является. Комментарии к моменту macro +expansion уже заменены whitespace, поэтому они не могут быть созданы +склеиванием `/` и `*`. Между `##` и его операндами исходно может находиться +whitespace; при склеивании он не участвует. + +Если два операнда не образуют один допустимый preprocessing token, выдаётся +диагностика, а сами исходные tokens сохраняются; наличие whitespace между ними +после такой диагностики не является частью контракта. `##` в начале или в +конце replacement list является ошибкой определения macro. + +### 7.3. Variadic macros: `...` и `__VA_ARGS__` + +Начиная с 0.0.46 поддерживаются variadic function-like macros в современной +C99-совместимой форме: + +```text +#define LOG(...) output(__VA_ARGS__) +#define LOGF(format, ...) output(format, __VA_ARGS__) +``` + +Маркер `...` может быть единственным параметром либо последним элементом после +одного или нескольких фиксированных параметров. Старое GNU-расширение с +именованным variadic parameter + +```text +#define LOG(args...) ... +``` + +в 0.0.46 намеренно не поддерживается. `__VA_OPT__` также не является частью +этого релиза. + +При вызове все tokens после последнего фиксированного параметра, включая +разделяющие их запятые, образуют один logical variable argument и подставляются +вместо `__VA_ARGS__`. В обычной позиции этот variable argument предварительно +проходит macro expansion так же, как обычный фактический аргумент: + +```text +#define A 7 +#define V(...) <__VA_ARGS__> +#define F(first, ...) first | __VA_ARGS__ + +V(A, 2, 3) -> <7, 2, 3> +F(1, A, 3) -> 1 | 7, 3 +``` + +Variadic tail может быть пустым. Поэтому оба вызова + +```text +F(1) +F(1,) +``` + +допустимы и подставляют пустой `__VA_ARGS__`. Это **не** означает автоматическое +удаление запятой, явно записанной в replacement list. Например для + +```text +#define E(format, ...) output(format, __VA_ARGS__) +``` + +вызов `E("ok")` оставляет запятую перед пустым tail. Специальная историческая +GNU-семантика `, ## __VA_ARGS__`, удаляющая такую запятую, в контракт 0.0.46 не +входит; если она понадобится, её следует вводить отдельным явно документированным +расширением. + +`__VA_ARGS__` участвует в уже существующей семантике `#` и `##` как настоящий +macro parameter. Stringification использует raw spelling всего variadic tail: + +```text +#define STRV(...) #__VA_ARGS__ +STRV(A, b + c) -> "A, b + c" +``` + +При соседстве с `##` variadic argument также подставляется без prescan; затем +работают обычные правила placemarker, token concatenation и общего rescan. +Например: + +```text +#define L(...) pre ## __VA_ARGS__ +#define R(...) __VA_ARGS__ ## post + +L(fix) -> prefix +R(fix) -> fixpost +``` + +Если variadic argument содержит несколько preprocessing tokens, склеивается +только крайний token, непосредственно соседний с `##`, а остальные tokens +сохраняются, как и для обычного параметра. Пустой tail рядом с `##` ведёт себя +как placemarker. + +Имя `__VA_ARGS__` зарезервировано для variable argument и не принимается как +обычное имя формального параметра. Оператор `#__VA_ARGS__` допустим только в +variadic macro. Dump-режимы сохраняют variadic форму определения, например: + +```text +#define F(first,...) first | __VA_ARGS__ +``` + + +### 7.4. `__VA_OPT__` + +Начиная с 0.0.47 variadic macros поддерживают стандартный условный fragment +`__VA_OPT__(pp-tokens)`. Если variable argument после обычной macro substitution +не содержит preprocessing tokens, весь `__VA_OPT__(...)` раскрывается в пустую +последовательность. Если variable argument непуст, содержимое круглых скобок +участвует в replacement list: + +```text +#define DEBUG(format, ...) \ + fprintf(stderr, format __VA_OPT__(,) __VA_ARGS__) + +DEBUG("ready") -> fprintf(stderr, "ready") +DEBUG("x=%d", x) -> fprintf(stderr, "x=%d", x) +``` + +Решение о непустоте принимается **после expansion variable argument**, а не по +его исходному spelling. Поэтому macro, который сам раскрывается в пустую +последовательность, не активирует `__VA_OPT__`: + +```text +#define EMPTY +#define HAS(...) [__VA_OPT__(yes)] + +HAS() -> [] +HAS(EMPTY) -> [] +HAS(token) -> [yes] +``` + +Содержимое `__VA_OPT__` может включать сбалансированные вложенные круглые скобки. +Закрывающая `)` самого `__VA_OPT__` определяется с учётом их вложенности. +Вложенный `__VA_OPT__` внутри другого `__VA_OPT__` намеренно запрещён. + +`__VA_OPT__` интегрирован с существующими правилами parameter substitution, +stringification, token concatenation, placemarker и rescan. Например: + +```text +#define X 123 +#define S(...) #__VA_OPT__(__VA_ARGS__) +#define L(...) pre ## __VA_OPT__(__VA_ARGS__) + +S() -> "" +S(X) -> "123" +L() -> pre +L(X) -> pre123 +``` + +При `#__VA_OPT__(...)` сначала выполняется parameter substitution внутри +fragment, включая prescan обычных параметров, но произвольные macro names самого +fragment до stringification дополнительно не rescanning-ятся. Поэтому: + +```text +#define X 123 +#define S(a, ...) #__VA_OPT__(a X) + +S(X, y) -> "123 X" +``` + +Если parameter внутри `__VA_OPT__` непосредственно участвует во внутреннем +`##`, для него, как обычно, prescan подавляется; paste выполняется до дальнейшего +rescan. Внешний `##`, соседний с `__VA_OPT__`, получает крайний token уже +подготовленного fragment. Пустой результат `__VA_OPT__` рядом с `##` ведёт себя +как placemarker. + +`__VA_OPT__` допустим только в replacement list variadic function-like macro и +должен непосредственно задавать parenthesized fragment. `##` не может быть +первым или последним preprocessing token внутри самого `__VA_OPT__`. + +Историческое GNU-расширение + +```text +, ## __VA_ARGS__ +``` + +в `mcpu-cpp` намеренно **не реализуется**. Для условной запятой следует +использовать современную форму `__VA_OPT__(,)`. Старое GNU-расширение с +именованным variadic parameter `args...` также остаётся неподдерживаемым. + + + +### 7.5. Нормализация пробелов в replacement list + +Начиная с 0.0.48 `mcpu-cpp` не переносит в результат разворачивания +служебное выравнивание многострочного macro. После удаления `\` + newline +последовательность пробельных символов, принадлежащая самому replacement list, +канонизируется в один ASCII-пробел. Это особенно важно для определений, где +обратные косые черты визуально выровнены в одну колонку: + +```text +#define TRACE(x) \ + do \ + { \ + output(x); \ + done(); \ + } \ + while( 0 ) +``` + +При разворачивании такое определение выдаёт компактный replacement: + +```text +do { output(x); done(); } while( 0 ) +``` + +а не сохраняет десятки пробелов перед каждой бывшей границей физической +строки. + +Нормализация относится **только к whitespace самого replacement list**. +`mcpu-cpp` не является formatter-ом исходной программы: пробелы в обычном +тексте input сохраняются. Пробелы внутри фактического macro argument также не +переформатируются только потому, что argument был подставлен в macro: + +```text +#define ID(x) x + +ID(a + b) -> a + b +``` + +Содержимое string/character literals сохраняется буквально, поэтому: + +```text +#define S "left right" +``` + +по-прежнему содержит пять пробелов внутри строки. + +Наличие whitespace между preprocessing tokens сохраняется как один пробел. +Это не позволяет случайно изменить tokenization, например превратить `+ +` в +`++`, `- >` в `->` или `< <` в `<<`. Операторы `#` и `##`, placemarkers, +`__VA_ARGS__`, `__VA_OPT__` и последующий rescan продолжают использовать свои +существующие правила; новая политика меняет только количество обычного +replacement-list whitespace. + +Dump-режимы (`-dM`, `-dD`) показывают ту же каноническую форму replacement +list, которая хранится во внутренней таблице macro. + + +### 7.6. Компактификация невидимых строк и linemarkers + +Начиная с 0.0.49 `mcpu-cpp` использует для вертикального whitespace ту же +модель, что GNU CPP: **удаляем, но не забываем**. Строки, которые после +preprocessing не породили ни одного выводимого preprocessing token, не обязаны +оставаться физическими пустыми строками в `.E`, однако их исходная позиция +продолжает учитываться при построении linemarkers и значении `__LINE__`. + +Причина невидимости не имеет значения. Это могут быть удалённые directives, +неактивные ветви `#if`, однострочные и многострочные comments, обычные пустые +строки или их смесь. Emitter сравнивает текущую output source position с +позицией следующей реально выдаваемой строки. + +Если следующая позиция находится менее чем через восемь строк, разрыв +представляется обычными newline. Если расстояние равно восьми строкам или +больше, вместо длинной последовательности пустых строк выдаётся корректирующий +linemarker: + +```text +# N "file" +``` + +и следующая содержательная строка сразу относится к source line `N`. Таким +образом граница поведения совместима с GNU CPP: gaps 0..7 сохраняются через +newline, gap 8 и больше заменяется linemarker. + +Structural markers входа и возврата из include-файла сохраняют обычный смысл: + +```text +# 1 "header.h" 1 +# 4 "source.c" 2 +``` + +Если included file не породил никакого output, `mcpu-cpp` не создаёт +искусственный marker, сообщающий, до какой внутренней строки header дошёл +препроцессор. После enter-marker сразу может следовать return-marker. Реальная +позиция снова уточняется только тогда, когда требуется выдать следующий +содержательный текст. + +Эта оптимизация меняет только представление output stream. Source coordinates, +`__LINE__`, diagnostics, `#line`, include enter/return semantics и обработка +macro остаются привязаны к исходному логическому потоку, а не к количеству +физических строк в сжатом `.E`. + + +## 8. Предопределённые макро + +Начиная с 0.0.6 был перенесён исторический механизм predefined macros из +препроцессора. Этот механизм оформлен как отдельный ABI/environment layer +будущего безымянного C-подобного языка. Эти определения не являются +декоративными: их имена и значения должны соответствовать либо семантике GNU +CPP, либо явно документированному MCPU/LibMPU contract. + +### 8.1. Динамические source macros + +Следующие predefined macros вычисляются в точке использования: + +| Макро | Раскрытие | +|---|---| +| `__FILE__` | строковая константа с именем текущего входного файла | +| `__LINE__` | десятичный номер текущей строки | +| `__BASE_FILE__` | строковая константа с именем главного входного файла translation unit | +| `__INCLUDE_LEVEL__` | уровень вложенности `#include`; для главного файла равен `0` | +| `__DATE__` | дата запуска препроцессора в форме `"Mmm dd yyyy"` | +| `__TIME__` | время запуска препроцессора в форме `"hh:mm:ss"` | + +`__DATE__` и `__TIME__` получают один timestamp на весь +translation unit. Специальное раскрытие помещается в output без повторного macro +rescan. + +Эти имена находятся в общей macro table, поэтому `#undef` и последующий +`#define` могут осознанно заменить builtin. + +### 8.2. Версия препроцессора + +Начиная с 0.0.8 standalone preprocessor не определяет GCC-имя `__VERSION__`. +Оно относится к compiler environment, которого для будущего high-level языка +пока нет. Собственная версия `mcpu-cpp` имеет отдельное однозначное имя: + +```text +#define __MCPU_CPP_VERSION__ "1.0.2" +``` + +Значение автоматически берётся из `PACKAGE_VERSION`. Когда появится compiler +frontend/driver, его version contract будет определён отдельно и не будет +смешиваться с версией standalone preprocessor. + +### 8.3. Источники истины ABI + +`mcpu-cpp` собирается только GNU GCC. Во время `configure` проект использует +проверенные приёмы из `LibMPU`/`LibMPUIO` `acsite.m4`: GCC predefined macros +определяют native type sizes, byte/word order и machine-register width, а +установленный `<libmpu.h>` является окончательным источником настроек LibMPU. + +В частности, фиксируются и проверяются: + +```text +MPU_REAL_IO_LIMIT +MPU_MATH_FN_LIMIT +MPU_BYTE_ORDER +MPU_WORD_ORDER +BITS_PER_MACHINE_REGISTER +BITS_PER_UNIT_T +sizeof(__mpu_size_t) +sizeof(__mpu_ptrdiff_t) +``` + +`configure` дополнительно проверяет, что byte order и +`BITS_PER_MACHINE_REGISTER`, записанные в LibMPU, согласованы с GCC target, +которым собирается `mcpu-cpp`. `MPU_WORD_ORDER` берётся непосредственно из +configured LibMPU profile и описывает порядок слов MCPU data environment. + +Пределы `MPU_REAL_IO_LIMIT` и `MPU_MATH_FN_LIMIT` имеют разные назначения. +Например, библиотека может иметь Real I/O до 65536 бит и математические функции +только до 16384 бит. Поэтому `MPU_MATH_FN_LIMIT` не используется как предел +существования типов Real. + +### 8.4. MCPU architecture и assembler prefixes + +Целевая архитектура определяется макро: + +```text +#define _ARCH_MCPU 1 +``` + +MCPU PTR64 имеет ширину 64 бита, поэтому определены `__SIZEOF_POINTER__`, +`__MCPU_POINTER_WIDTH__`, `__INTPTR_TYPE__`, `__UINTPTR_TYPE__`, соответствующие +width/max macros. + +Смысл assembler-prefix macros согласован с GNU CPP, а не с первой буквой имени +register view. В синтаксисе `mcpu-as` дополнительного sigil перед register, +label или immediate нет. `r` и `c` являются частью MCPU register syntax, а не +`REGISTER_PREFIX`. Поэтому: + +```text +#define __REGISTER_PREFIX__ +#define __LOCAL_LABEL_PREFIX__ +#define __USER_LABEL_PREFIX__ +#define __IMMEDIATE_PREFIX__ +``` + +все четыре раскрываются в пустую последовательность. `.L...` остаётся +compiler naming convention и не является assembler ABI local-label prefix: +LOCAL/GLOBAL binding определяется symbol directives. + +### 8.5. Byte order и word order + +Базовые числовые значения порядка байт совместимы с GNU CPP: + +```text +__ORDER_LITTLE_ENDIAN__ +__ORDER_BIG_ENDIAN__ +__ORDER_PDP_ENDIAN__ +``` + +Но целевая среда публикует собственные MCPU names: + +```text +#define __MCPU_BYTE_ORDER__ __ORDER_LITTLE_ENDIAN__ +#define __MCPU_WORD_ORDER__ __ORDER_LITTLE_ENDIAN__ +#define __BYTE_ORDER__ __MCPU_BYTE_ORDER__ +``` + +Фактические значения `__MCPU_BYTE_ORDER__` и `__MCPU_WORD_ORDER__` получают из +configured LibMPU profile (`MPU_BYTE_ORDER` и `MPU_WORD_ORDER`). Поэтому они +следуют host data representation, с которой собрана LibMPU. Это не меняет +отдельный архитектурный контракт кодировки MCPU instruction bytecode. + +GNU/C-specific имя `__FLOAT_WORD_ORDER__` не определяется: типа `float` в +будущем языке MCPU нет. + +Параметры LibMPU/MCPU environment публикуются в MCPU namespace: + +```text +__MCPU_MACHINE_REGISTER_WIDTH__ +__MCPU_REAL_IO_LIMIT__ +__MCPU_MATH_FN_LIMIT__ +__MCPU_INT_MAX_WIDTH__ +__MCPU_REAL_MAX_WIDTH__ +__MCPU_COMPLEX_MAX_WIDTH__ +``` + +`__MCPU_INT_MAX_WIDTH__` равен `NB_I_MAX * 8`, а Real/Complex maximum width +равен configured `MPU_REAL_IO_LIMIT`. `__MCPU_MACHINE_REGISTER_WIDTH__` является +значением `BITS_PER_MACHINE_REGISTER` установленной LibMPU. Пределы Real I/O и +math functions не смешиваются: `MPU_REAL_IO_LIMIT` определяет существование +Real/Complex type family и text conversion, а `MPU_MATH_FN_LIMIT` — наличие +математических функций соответствующей ширины. + +### 8.6. MCPU size/ssize, `ptrdiff` и pointers + +Будущий язык не наследует variable-width C names `short`, `int`, `long` и +не использует C-style имя `size_t` как часть собственного ABI. Беззнаковый +LibMPU size type и знаковый byte-count/error type публикуются симметрично в +MCPU namespace. Например для 64-bit configured profile: + +```text +#define __MCPU_SIZE_TYPE__ uint64 +#define __MCPU_SIZE_WIDTH__ 64 +#define __MCPU_SIZEOF_SIZE__ 8 +#define __MCPU_SIZE_MAX__ 0xffffffffffffffff + +#define __MCPU_SSIZE_TYPE__ int64 +#define __MCPU_SSIZE_WIDTH__ 64 +#define __MCPU_SIZEOF_SSIZE__ 8 +#define __MCPU_SSIZE_MAX__ 0x7fffffffffffffff +``` + +Это MCPU-specific family, а не попытка приписать GNU CPP несуществующий +стандартный `__SSIZE_*` contract. + +MCPU pointer ABI от host не зависит: PTR64 всегда имеет ширину 64 бита: + +```text +#define __INTPTR_TYPE__ int64 +#define __UINTPTR_TYPE__ uint64 +#define __INTPTR_WIDTH__ 64 +#define __UINTPTR_WIDTH__ 64 +#define __INTPTR_MAX__ 0x7fffffffffffffff +#define __UINTPTR_MAX__ 0xffffffffffffffff +#define __SIZEOF_POINTER__ 8 +#define __MCPU_POINTER_WIDTH__ 64 +``` + +Разность MCPU pointers является знаковой и также фиксирована независимо от +host: + +```text +#define __PTRDIFF_TYPE__ int64 +#define __PTRDIFF_WIDTH__ 64 +#define __SIZEOF_PTRDIFF__ 8 +#define __PTRDIFF_MAX__ 0x7fffffffffffffff +``` + +Computed MIN expressions вроде `(-__PTRDIFF_MAX__ - 1)` в predefined table не +создаются. + +### 8.7. Character types + +Обычного C `char` в будущем языке нет. Поэтому `__CHAR_TYPE__` и +`__WCHAR_TYPE__` не определяются. Типы языка называются без C/C++ suffix `_t`: + +```text +#define __CHAR8_TYPE__ char8 +#define __CHAR16_TYPE__ char16 +#define __CHAR8_WIDTH__ 8 +#define __CHAR16_WIDTH__ 16 +#define __SIZEOF_CHAR8__ 1 +#define __SIZEOF_CHAR16__ 2 +``` + +Это типы будущего языка. Внутренняя реализация самого `mcpu-cpp` по-прежнему +использует LibMPUIO `__mpu_char16_t` и strict UCS-2 text model. + +### 8.8. Integer families LibMPU + +Полная structural metadata integer families строится не по жёстко записанному +последнему типу, а до `NB_I_MAX * 8` фактически установленной LibMPU. Для +каждой power-of-two ширины от 8 бит определяются TYPE, WIDTH и SIZEOF: + +```text +#define __INT1024_TYPE__ int1024 +#define __UINT1024_TYPE__ uint1024 +#define __INT1024_WIDTH__ 1024 +#define __UINT1024_WIDTH__ 1024 +#define __SIZEOF_INT1024__ 128 +#define __SIZEOF_UINT1024__ 128 +``` + +На текущей LibMPU 1.0.25 `NB_I_MAX == 8192`, поэтому family доходит до +`int65536`/`uint65536`, а `__SIZEOF_INT65536__ == 8192`. + +Decimal-digit metadata определяется для **каждой** разрешённой integer width: + +```text +__INT<bits>_DECIMAL_DIG__ +__UINT<bits>_DECIMAL_DIG__ +``` + +Значение вычисляется собственными integer-only helpers `mcpu-cpp` из известной +ширины типа. Оно означает точное число десятичных цифр максимального значения +соответствующего типа: знак и завершающий NUL в `DECIMAL_DIG` не входят. Для +unsigned используется максимум `2^bits - 1`, для signed — `2^(bits-1) - 1`. +Это отличается от LibMPU `_int_digs()`, которая предназначена для оценки +строкового буфера и включает место для завершающего NUL. + +Например: + +```text +#define __INT64_DECIMAL_DIG__ 19 +#define __UINT64_DECIMAL_DIG__ 20 +#define __INT256_DECIMAL_DIG__ 77 +#define __UINT256_DECIMAL_DIG__ 78 +``` + +Только сами textual maxima намеренно ограничены шириной `bits <= 256`: + +```text +__INT128_MAX__ +__UINT128_MAX__ +``` + +Максимумы строятся через LibMPU `iuitoa()`. Макро `__INT<bits>_MIN__` не +создаются: predefined table не должна содержать вычисляемые выражения вида +`(-__INT<bits>_MAX__ - 1)`. Для widths больше 256 бит отсутствуют только MAX; +TYPE/WIDTH/SIZEOF/DECIMAL_DIG сохраняются до полного `NB_I_MAX * 8`. + +### 8.9. Real и Complex families LibMPU + +Real/Complex structural metadata генерируется для каждой power-of-two ширины от +32 бит до фактического configured `MPU_REAL_IO_LIMIT`. Для всех этих типов +публикуются TYPE, WIDTH и SIZEOF. + +Для Complex WIDTH означает параметр типа, а не суммарную storage width: + +```text +#define __COMPLEX128_TYPE__ complex128 +#define __COMPLEX128_WIDTH__ 128 +#define __SIZEOF_COMPLEX128__ 32 +``` + +`complex128` состоит из двух компонентов `real128`, поэтому его storage size +равен 32 байтам. При `MPU_REAL_IO_LIMIT == 65536` верх family имеет вид: + +```text +#define __COMPLEX65536_TYPE__ complex65536 +#define __COMPLEX65536_WIDTH__ 65536 +#define __SIZEOF_COMPLEX65536__ 16384 +``` + +Для Real соответственно: + +```text +#define __REAL65536_TYPE__ real65536 +#define __REAL65536_WIDTH__ 65536 +#define __SIZEOF_REAL65536__ 8192 +``` + +Precision metadata определяется для **всех** разрешённых Real widths вплоть +до `MPU_REAL_IO_LIMIT`. Имена macros согласованы с LibMPU helpers: + +```text +__REAL<bits>_DECIMAL_DIG__ -> _real_digs(bits/8) +__REAL<bits>_MANT_DIG__ -> _real_mant_digs(bits/8) +``` + +`__REAL<bits>_DIG__` намеренно отсутствует. Ограничение `bits <= 256` относится +только к большим textual numeric constants. Для размеров до 256 бит также +определяются: + +```text +__REAL<bits>_MAX__ +__REAL<bits>_MIN__ +__REAL<bits>_EPSILON__ +__REAL<bits>_MAX_EXP__ +__REAL<bits>_MIN_EXP__ +__REAL<bits>_MAX_10_EXP__ +__REAL<bits>_MIN_10_EXP__ +``` + +Например, на LibMPU 1.0.25 для `real128` текущий profile даёт значения вида: + +```text +#define __REAL128_EPSILON__ 2.524354896707237777317531409e-29 +#define __REAL128_MAX__ 4.197157432934775384808581951e+323228496 +#define __REAL128_MIN__ 9.530259619551804292864984035e-323228497 +#define __REAL128_MAX_10_EXP__ 323228496 +#define __REAL128_MAX_EXP__ 1073741823 +#define __REAL128_MIN_10_EXP__ -323228524 +#define __REAL128_MIN_EXP__ -1073741822 +``` + +MAX/MIN/EPSILON создаются самой LibMPU и преобразуются через +`real_to_ascii()`. Exponent constants получают значения через LibMPU exponent +helpers и integer conversion. Для widths больше 256 бит эти numeric predefines отсутствуют, но +TYPE/WIDTH/SIZEOF/DECIMAL_DIG/MANT_DIG продолжаются до `MPU_REAL_IO_LIMIT`. + +Для каждого разрешённого Real type вплоть до `MPU_REAL_IO_LIMIT` также +публикуются две компактные характеристики: + +```text +#define __SIZEOF_REAL128_EXP__ 4 +#define __REAL128_MAX_STRLEN__ 60 +``` + +`__SIZEOF_REALxxx_EXP__` непосредственно получает `_sizeof_exp(NB_Rxxx)`. +`__REALxxx_MAX_STRLEN__` получает `_real_max_string(NB_Rxxx)` и означает +максимальное **количество символов** текстового представления, а не количество +байт. Поэтому для zero-terminated строки нужно резервировать не менее +`__REALxxx_MAX_STRLEN__ + 1` элементов: для `char8` это столько же bytes, а для +`char16` физический объём в bytes вдвое больше. Эти два metadata-macro +определяются и для Real widths больше 256, поскольку сами их значения малы. + +### 8.10. Dump macros: `-dM`, `-dMP` + +Опция: + +```text +mcpu-cpp -dM input.c +``` + +печатает только итоговые **непредопределённые** macros в форме `#define ...`. +К этой группе относятся определения из основного файла и включённых headers, а +также определения командной строки `-D`. Предопределённые macros самого +MCPU-CPP в `-dM` не выводятся. Поэтому `-dM` предназначен прежде всего для +короткой инспекции macro-state, созданного пользовательской программой. + +Опция: + +```text +mcpu-cpp -dMP input.c +``` + +добавляет к тому же итоговому состоянию активные predefined macros MCPU-CPP. +Вывод имеет две последовательные группы: сначала все predefined macros, затем +все непредопределённые macros. Внутри каждой группы определения +детерминированно сортируются по имени. Такое разделение удобно системному +разработчику для инспекции preprocessing ABI и архитектурных свойств текущей +MCPU environment, не смешивая их с пользовательскими определениями. + +Принадлежность к группе определяется происхождением macro, а не его именем. +Macro, заданный через `-D` или `#define`, является обычным даже если его имя +похоже на системное. Если predefined macro был удалён через `#undef`, он не +печатается. Если после этого то же имя снова определено пользователем, новое +определение относится к обычной группе и выводится в её части `-dMP`, а также +в `-dM`. Тем самым оба режима показывают именно **итоговый macro-state**. + +Context-dependent `__FILE__`, `__LINE__`, `__DATE__`, `__TIME__`, +`__BASE_FILE__` и `__INCLUDE_LEVEL__` в статическом dump не печатаются. +Статические ABI/architecture predefined macros и вычисляемые static Real +metadata выводятся в `-dMP`. + +Если input file указан, он сначала полностью препроцессируется, после чего +выводится итоговый macro-state; обычный preprocessed text в режимах `-dM` и +`-dMP` не выдаётся. Без input file используется stdin, поэтому пустой stdin с +`-dM` даёт пустой dump, а `-dMP` позволяет получить набор активных static +predefined macros текущей MCPU environment. + +`-dD` имеет другую семантику и этим разделением не затрагивается. + +### 8.11. Dump definitions: `-dD` + +Опция: + +```text +mcpu-cpp -dD input.c +``` + +сохраняет обычный результат препроцессирования и одновременно выводит +встреченные директивы `#define`. Перед началом основного входного текста +печатаются статические предопределённые macro definitions. Каждой такой +дефиниции предшествует marker: + +```text +# 0 "<built-in>" +#define NAME value +``` + +а перед блоком предопределённых macro выводится marker исходного файла вида +`# 0 "input.c"`. Context-dependent `__FILE__`, `__LINE__`, `__DATE__`, +`__TIME__`, `__BASE_FILE__` и `__INCLUDE_LEVEL__` в начальный built-in block +не включаются. + +### 8.12. Dump configuration: `-dconfig` + +Опция: + +```text +mcpu-cpp -dconfig +``` + +не требует input file и выводит effective variables configuration layer после чтения runtime config, необязательного system override, домашнего +user override или выбранного `--config-file`, включая expansion +`$NAME`/`${NAME}`. Строки сортируются по имени и печатаются в форме: + +```text +NAME = value; +``` + +Это позволяет проверить реальные include paths без ручного поиска +`<runtime-root>/etc/mcpu-cpp.conf`, `/etc/mcpu/mcpu-cpp.conf` и +`$HOME/.mcpu/mcpu-cpp.conf`. + +### 8.13. Verbose configuration snapshot: `-v` + +При `-v` MCPU-CPP сохраняет прежний runtime trace для `#lang`, `#include` и +`#include_next`, но конфигурационные переменные печатаются только один раз — +после чтения всех уровней configuration и применения правил приоритета. Поэтому +в verbose output видны только **effective values**, а промежуточные значения из +runtime-root, system и user config не дублируются. + +Config-блок выводится в порядке include policy: language-specific user paths, +общий user path, system root и AFTER path. Переменная, отсутствующая во всех +уровнях configuration, не печатается. Runtime-derived default +`MCPU_CPP_SYSTEM_INCLUDE_PATH` является полноценным самым нижним значением и +поэтому виден при `-v`, даже если ни один `mcpu-cpp.conf` не найден **или все +config-файлы отключены опцией `--no-config`**. + +Форма строки: + +```text +config: NAME=value +``` + +### 8.14. Effective search directories: `-dsearch-dirs` + +Опция: + +```text +mcpu-cpp -dsearch-dirs +``` + +не требует input file, печатает effective глобальные каталоги поиска и +завершает работу без preprocessing. Формат намеренно прост: + +```text +search: /path/to/directory +``` + +Каталоги выводятся в семантическом порядке классов поиска: + +```text +explicit -I +explicit -isystem +configured language-specific user directories +MCPU_CPP_INCLUDE_PATH +MCPU_CPP_SYSTEM_INCLUDE_PATH/<lang> +MCPU_CPP_SYSTEM_INCLUDE_PATH +explicit -idirafter +MCPU_CPP_AFTER_INCLUDE_PATH +``` + +Language-specific entries печатаются для всех поддерживаемых языков в их +каноническом порядке. Во время реального `#include` из этой группы участвует +только каталог активного `#lang`. Каталог текущего физического файла в +`-dsearch-dirs` не выводится: он существует только динамически для конкретного +`#include "..."` и меняется вместе с include stack. `--no-config` не удаляет +runtime-derived system root, поэтому без конфигурационных файлов dump всё равно +содержит `<runtime-root>/include/<lang>` и `<runtime-root>/include`. `-nostdinc` удаляет +из dump effective system `<lang>` entries и system root, но не explicit +`-isystem`. Не существующий на filesystem каталог всё равно показывается, +поскольку он является элементом effective search configuration и просто будет +пропущен при реальном поиске файла. + +`-dsearch-dirs` учитывает `-I`, `-isystem`, `-idirafter`, все уровни config и +replacement-семантику `MCPU_CPP_SYSTEM_INCLUDE_PATH`. Опция `-o` вместе с ним +является ошибкой. + +### 8.15. Условная компиляция + +Директивы `#if`, `#ifdef`, `#ifndef`, `#elif`, `#else` и `#endif` обрабатываются +как управляющие директивы препроцессора и в выходной поток не копируются, в +том числе при `-dD`. Неактивные ветви пропускаются без выполнения находящихся +в них `#define`, `#undef` и `#include`; вложенные условные группы при этом +учитываются корректно. + +Выражение `#if` сначала обрабатывает оператор +`defined`, затем выполняется macro expansion, а оставшиеся идентификаторы +имеют значение `0`. Поддерживаются арифметические, битовые, сравнительные и +логические операции, `?:` и short-circuit semantics для `&&`, `||` и `?:`. + +Начиная с 0.0.26 синтаксис выражения разбирается parser-ом, генерируемым +ZUBR 4.1.0 из `src/mcpp-expr.zubr`; в том же файле находится UCS-2 lexical +analyzer. Предварительная обработка `defined` и macro expansion выполняются до +входа в parser. Арифметическая семантика вынесена в `mcpp-semantic.c/h` и не +зависит от размеров целых типов host-системы. Generated `mcpp-expr.c` +включается в release, поэтому ZUBR требуется только при изменении grammar. + +#### 8.15.1. Единственная вычислительная разрядность — 64 бита + +MCPU-CPP является препроцессором, а не компилятором языка общего назначения. +Все целочисленные вычисления в директивах условной компиляции выполняются +только в 64-разрядной арифметике. Препроцессор не выполняет арифметику LibMPU +произвольной разрядности, вещественные или комплексные вычисления. + +Если программисту не требуется управлять двоичным представлением литерала, +достаточно обычных целых констант и необязательного `U`/`u`. Например: + +```c +#if 2 > 1 +#if 0xffffffffffffffffU > 1 +``` + +Числовой lexeme хранится в UCS-2 до классификации, после чего его ASCII-часть +передаётся LibMPU `iatoui()`. Поддерживаются `0b...`, `0...`, decimal и +`0x...`. Значение, не помещающееся в 64 бита, является ошибкой. Старые C +suffixes `L`, `l`, `LL`, `ll` не поддерживаются. + +#### 8.15.2. Суффикс разрядности `zNNN[Uu]` + +MCPU-CPP понимает общий для MCPU-языков суффикс разрядности: + +```text +zNNN +ZNNN +zNNNu +zNNNU +ZNNNu +ZNNNU +``` + +`NNN` — непустая последовательность десятичных цифр и **всегда** читается как +десятичное число, даже если начинается с нулей. Поэтому `z8`, `z08` и `z008` +задают одну и ту же разрядность 8 бит. + +В общем синтаксисе MCPU корректная разрядность должна быть степенью двойки от +8 до `MPU_REAL_IO_LIMIT`. MCPU-CPP, однако, сознательно ограничен 64-битными +вычислениями: + +* `z8`, `z16`, `z32`, `z64` и варианты регистра допустимы; +* значение `NNN > 64` немедленно является ошибкой: препроцессор не допускает + числовые константы разрядности выше 64 бит в директивах условной компиляции; +* если `NNN <= 64`, но не задаёт допустимую степень двойки, например `z24`, + выводится warning и сам `zNNN` игнорируется; +* необязательный следующий `U`/`u` задаёт unsigned и сохраняет своё значение + даже если некорректный `zNNN` был проигнорирован. + +После полного суффикса должна заканчиваться числовая preprocessing token. +Оператор или punctuation начинает следующий token, поэтому допустимы +`1z32u+2`, `(1z32u)` и `1z32u==1`. Записи вроде `1z32undefined`, `1z32ufoo` и +`1z32$foo` являются ошибками и не разбиваются искусственно на число и имя. + +#### 8.15.3. Нормализация литерала + +Суффикс разрядности действует **только один раз — при формировании значения +самой константы**. Разрядность не сохраняется в semantic value и не участвует +в последующих операциях. + +Для `VALUEzNNN` значение считается знаковым N-битным числом в дополнительном +коде: + +1. сохраняются младшие `NNN` бит; +2. результат расширяется со знаком до 64 бит. + +Для `VALUEzNNNu`/`VALUEzNNNU` сохраняются младшие `NNN` бит, после чего +выполняется нулевое расширение до 64 бит. + +Например: + +```text +0x7fz8 -> 0x000000000000007f -> 127 +0x80z8 -> 0xffffffffffffff80 -> -128 +0xffz8 -> 0xffffffffffffffff -> -1 +0x80z8u -> 0x0000000000000080 -> 128 +0xffz8u -> 0x00000000000000ff -> 255 +0x1ffz8 -> 0xffffffffffffffff -> -1 +0x1ffz8u -> 0x00000000000000ff -> 255 +``` + +Последние два примера намеренны: `zNNN` задаёт разрядность **двоичного +представления**, а не проверку математического диапазона. Биты старше N +отбрасываются до расширения. + +После этой нормализации никакой `z8`, `z16` или `z32` в вычислительной модели +уже не существует. Внутреннее значение содержит только 64-битный битовый +образ и признак signed/unsigned. + +#### 8.15.4. Все последующие операции — 64-битные + +После нормализации все арифметические, побитовые, сравнительные и логические +операции выполняются над 64-битными операндами. Результат операции не +усекается обратно до разрядности исходного suffix. Поэтому: + +```text +0x7fz8 + 1 -> 128 +0xffz8u + 1 -> 256 +``` + +а не `-128` и `0` соответственно. Аналогично `~0xffz8u` инвертирует все 64 +бита и даёт `0xffffffffffffff00`. + +Для бинарных операций, где signedness имеет значение, наличие unsigned +операнда переводит операцию в 64-битную unsigned-интерпретацию. Сравнения +возвращают `0` или `1`. Логические `!`, `&&`, `||` также возвращают signed +64-битные `0` или `1`; short-circuit не вычисляет невыбранную часть. + +Сдвиги выполняются после 64-битной нормализации. Правый сдвиг signed +отрицательного значения является арифметическим, unsigned — логическим. +Например: + +```text +0x80z8 >> 1 -> -64 +0x80z8u >> 1 -> 64 +``` + +Историческое правило MCPU-CPP для отрицательного счётчика сдвига сохраняется: +`A << -N` эквивалентно `A >> N`, а `A >> -N` — `A << N`. + +Таким образом, `zNNN` не превращает препроцессор в компилятор с системой +integer promotions разных размеров. Он лишь позволяет явно описать битовый +образ исходного литерала; затем выражение вычисляется в единственной простой +64-битной модели. + +#### 8.15.5. Символьные константы + +Символьная единица имеет тип `__mpu_uint16_t`, соответствующий внутреннему +UCS-2 представлению, и перед вычислением расширяется нулями до 64 бит. +Последующая арифметика снова является обычной 64-битной арифметикой. + +Состояние условной компиляции хранится в отдельном стеке; условная группа не +может пересекать границу include-файла. + +### 8.16. Диагностические директивы `#error` и `#warning` + +MCPU-CPP поддерживает стандартные диагностические директивы: + +```text +#error сообщение +#warning сообщение +``` + +`#error` выдаёт diagnostic уровня error с текущими логическими именем файла и +номером строки и немедленно завершает preprocessing с ошибкой. `#warning` +выдаёт warning с той же source-location information, после чего preprocessing +продолжается. Поэтому предшествующий `#line` влияет на координаты обеих +диагностик. + +Остаток строки после имени директивы +**не подвергается macro expansion**. Например: + +```c +#define MESSAGE expanded +#warning MESSAGE +``` + +печатает `MESSAGE`, а не `expanded`. Это отличает диагностические директивы от +`#if` и `#line`, где macro expansion является частью соответствующего +контракта. + +Комментарии удаляются на обычной preprocessing phase до обработки директивы. +Начальные и конечные пробелы сообщения удаляются, последовательности пробельных +символов между preprocessing tokens сворачиваются в один пробел. Пробелы внутри +кавычек сохраняются. Например: + +```c +#warning one /* comment */ two +#warning "a b" +``` + +дают сообщения соответственно `one two` и `"a b"`. Unicode-текст проходит +через внутреннее UCS-2 представление и выводится во внешнюю диагностику в UTF-8. + +Обе директивы являются управляющими и никогда не копируются в обычный выходной +поток или в `-dD`. В неактивной ветви `#if` они полностью игнорируются, поэтому +обычная защитная конструкция работает ожидаемо: + +```c +#if 0 +#error this error is inactive +#endif +``` + +### 8.17. Управление предупреждениями: `-Wcomment`, `-Wall`, `-Werror` + +MCPU-CPP разделяет обязательные предупреждения, являющиеся частью уже +зафиксированной preprocessing-семантики, и дополнительные классы предупреждений, +которые включаются пользователем. Управление предупреждениями не изменяет +семантику `-dD`, macro expansion, conditional compilation или include search. + +Опции `-Wcomment` и `-Wcomments` являются полными синонимами и включают два +лексических предупреждения: + +* последовательность `/*`, встретившуюся внутри уже открытого `/* ... */` + комментария; +* backslash-newline внутри `//` комментария, из-за которого однострочный + комментарий физически продолжается на следующую строку. + +По умолчанию этот дополнительный класс выключен. `-Wall` включает все +дополнительные warning classes MCPU-CPP; в версии 0.0.40 таким классом является +`-Wcomment`. Формы `-Wno-comment` и `-Wno-comments` выключают его. Как в GNU +warning model, более специфическая настройка имеет приоритет над групповой +независимо от порядка аргументов. Поэтому обе команды: + +```text +mcpu-cpp -Wall -Wno-comment file.c +mcpu-cpp -Wno-comment -Wall file.c +``` + +оставляют comment warnings выключенными. Между настройками одинаковой +специфичности действует последнее указание, например `-Wno-comment -Wcomment` +включает этот класс. + +`-Werror` не включает никаких новых warning classes. Он повышает до error любое +предупреждение, которое в данном запуске действительно было бы выдано, и такой +запуск завершается неуспешно. Это относится как к дополнительным comment +warnings, так и к уже существующим обязательным предупреждениям MCPU-CPP: + +* активной директиве `#warning`; +* недопустимой, но не превышающей 64 бита ширине `zNNN`; +* переопределению macro другим replacement list; +* результату `##`, не образующему один preprocessing token. + +Например: + +```text +mcpu-cpp -Wcomment -Werror file.c +``` + +превращает найденный comment warning в error. В то же время один `-Werror` без +`-Wcomment`/`-Wall` не заставляет MCPU-CPP искать optional comment warnings. + +`-Wno-error` возвращает обычную severity warning. Для `-Werror` и `-Wno-error`, +имеющих одинаковую специфичность, действует последняя опция командной строки. +Так, `-Werror -Wno-error` оставляет warnings предупреждениями, а +`-Wno-error -Werror` снова повышает их до errors. + +В 0.0.40 намеренно не вводятся `-Werror=<class>`, `-Wno-error=<class>`, +`-Wundef`, `-Wunused-macros`, `-Wtraditional` и другие компиляторные классы. +Warning interface MCPU-CPP остаётся компактным и расширяется только тогда, когда +новый класс действительно нужен самому preprocessing language. + +### 8.18. Идентификаторы UCS-2 + +Начиная с 0.0.22 имена preprocessing identifiers больше не ограничены ASCII. +Внутри `mcpu-cpp` текст уже представлен строгим UCS-2, а классификация символов +выполняется locale-independent функциями LibMPUIO 1.0.4, построенными по Unicode +18.0.0. Первый символ идентификатора должен быть `_` или иметь свойство +`XID_Start`; последующие символы должны быть `_`, `$` или иметь свойство +`XID_Continue`. Символ `$` является расширением `mcpu-cpp`: он разрешён только +после первого символа и не может начинать identifier. Это правило едино для +имён и параметров macro, `#undef`, `#ifdef`/`#ifndef`, `defined`, обычного macro +expansion, `#`/`##`. Имена остаются case-sensitive. Surrogate code units +`U+D800..U+DFFF` не являются допустимыми символами identifiers. + +Например, допустимы: + +```c +#define АНДРЕЙ 1 +#define résumé 2 +#define ΩМЕГА 3 +#define VALUE$OLD 4 +``` + +Например, `VALUE$OLD` допустим, а `$VALUE` недопустим, поскольку `$` не является +identifier-start character. + +Combining marks и не-ASCII decimal digits могут входить в identifier в позициях +`XID_Continue`, но не становятся автоматически допустимыми первыми символами. +Синтаксис числовых констант от этого не меняется: его правила остаются правилами +соответствующего языка, а не Unicode `isdigit`. + +### 8.19. Макросы командной строки `-D` и `-U` + +Начиная с 0.0.23 опции `-D` и `-U` являются полноценными действиями +препроцессора. Поддерживаются формы: + +```text +-DNAME +-DNAME=VALUE +-D'FUNC(a,b)=a+b' +-UNAME +``` + +`-DNAME` эквивалентна `#define NAME 1`; наличие `=` с пустой правой частью +задаёт пустой replacement list. Function-like определения используют тот же +macro engine, что и обычный `#define`, включая параметры, `#`, `##` и +последующий rescanning. `-U` использует тот же identifier contract, что и +`#undef`. Действия `-D`/`-U` выполняются в порядке командной строки после +установки predefined macros. + +Только payload опций `-D` и `-U` интерпретируется как UTF-8 и преобразуется в +строгий UCS-2. Имена файлов, `-I`, другие pathname arguments и остальные +аргументы командной строки остаются исходными byte strings и не подвергаются +Unicode-конвертации. + +Начиная с 0.0.25 символ `$` разрешён внутри имени macro, но не в первой +позиции. При передаче `$` из shell пользователь обязан учитывать правила самого +shell: shell обрабатывает `$` **до запуска `mcpu-cpp`**. Одинарные кавычки уже +полностью защищают `$`, например: + +```sh +mcpu-cpp '-DАНДРЕЙ$_Y=62' input.c +``` + +Без кавычек `$` следует экранировать: + +```sh +mcpu-cpp -DАНДРЕЙ\$_Y=62 input.c +``` + +или использовать двойные кавычки с экранированием: + +```sh +mcpu-cpp -D"АНДРЕЙ\$_Y=62" input.c +``` + +Вариант без защиты: + +```sh +mcpu-cpp -DАНДРЕЙ$_Y=62 input.c +``` + +не передаёт написанное имя буквально: `$...` сначала раскрывается shell и +`mcpu-cpp` получает уже изменённый `argv`. Внутри одинарных кавычек обратная +косая черта перед `$` не нужна и стала бы обычным символом аргумента. + +Для command-line `-D` левая часть до первого `=` разбирается как отдельный +macro declarator. Если после допустимого имени (или завершённого списка +параметров function-like macro) до `=` встречается недопустимый хвост, этот +хвост молча отбрасывается и **никогда не превращается в replacement list**. +Например: + +```text +-D'АНДРЕЙ@XYZ=62' +``` + +эквивалентно: + +```c +#define АНДРЕЙ 62 +``` + +а не ошибочной форме `#define АНДРЕЙ @XYZ 62`. Аналогичное правило допустимого +identifier-prefix применяется к `-U`. Если же первый символ вообще не является +допустимым identifier-start character (например `$` или цифра), определение +остаётся ошибочным. + +При `-dD` определения, пришедшие через `-D`, маркируются отдельно от +предопределённых macro: + +```text +# 0 "<command-line>" +#define NAME value +``` + +в то время как predefined macros продолжают использовать `<built-in>`. + +### 8.20. Публичный интерфейс командной строки + +`mcpu-cpp` поддерживает только актуальные опции, описанные `--help`. Устаревшие +compatibility-флаги не образуют скрытый интерфейс и диагностируются как +`unknown option`. Опция `-E` является исключением: она молча принимается и +игнорируется, поскольку может передаваться compiler driver при запуске +отдельного препроцессора. + +Опция `--object-suffix SUFFIX` задаёт суффикс object target, используемый при +генерации make-зависимостей; аргумент обязателен. + +## 9. Build-system и генераторы + +Собственные Autoconf-макросы проекта находятся в корневом `acsite.m4`. +Каталог `m4/` зарезервирован для внешних/vendor M4-файлов. Такой порядок +повторяет принятую в библиотеках MCPU схему и не смешивает собственный +configure-код с импортированными макросами. + +Парсер выражений `#if` генерируется ZUBR 4.1.0 из `src/mcpp-expr.zubr`. +Release archive содержит и грамматику, и уже сгенерированный `src/mcpp-expr.c`, +поэтому обычная сборка не требует установленного ZUBR. После изменения +грамматики developer build использует штатное правило Automake: + +```text +zubr -vl -s -Bmcpp_ -o mcpp-expr.c mcpp-expr.zubr +``` + +Перед выпуском release generated C должен соответствовать грамматике, полный +test suite и `make distcheck` должны проходить без ошибок. + +### 9.1. Developer bootstrap и Git source tree + +Начиная с 0.0.50 корневой скрипт `./bootstrap` позволяет не хранить в Git файлы, +которые полностью воспроизводятся из исходников. Скрипт сначала генерирует +`src/mcpp-expr.c` из `src/mcpp-expr.zubr` с помощью ZUBR 4.1.0, затем выполняет +`aclocal`, `autoheader`, `automake` и `autoconf` в стиле библиотек LibMPU и +LibMPUIO. Опция `--target-dest-dir=DIR` задаёт target ROOTFS для системных +Autoconf macro/include directories. + +Это правило относится именно к developer Git tree. **Release archive остаётся +самодостаточным**, как и раньше: он содержит `configure`, `Makefile.in`, helper +scripts Automake и уже сгенерированный `src/mcpp-expr.c`, поэтому обычная сборка +релиза не требует предварительного запуска `bootstrap` и не требует ZUBR. + +Корневой `.gitignore` перечисляет воспроизводимые bootstrap-файлы и обычный +configure/build state. Он не меняет существующую release/build model, а только +позволяет поддерживать более чистый Git repository. + +## 10. GNU-compatible features + +`mcpu-cpp` является самостоятельным препроцессором MCPU, но для ряда хорошо +известных операций намеренно повторяет поведение GNU CPP. Совместимость +относится к документированным возможностям, а не означает полную CLI- или +языковую взаимозаменяемость с GCC. + +В частности, GNU-compatible поведение используется для: + +* object-like и function-like macro, повторного macro rescan, `#` и `##`; +* variadic macro `...` / `__VA_ARGS__` и стандартного `__VA_OPT__`; +* `#if`, `#ifdef`, `#ifndef`, `#elif`, `#else`, `#endif` и `defined`; +* `#include`, `#include_next`, `#pragma once`, `#line` и GNU linemarkers; +* compact output mapping: до семи невидимых строк представляются newline, а + разрыв в восемь и более строк — корректирующим linemarker; +* forced files `-include` / `-imacros` и dependency options `-M`, `-MM`, `-MD`, + `-MMD`, `-MF`, `-MT`, `-MQ`, `-MG`; +* warning controls `-w`, `-Wall`, `-Werror` и поддерживаемых `-Wcomment` forms. + +MCPU-specific возможности, включая `#lang` / `#endlang`, числовой суффикс +`zNNN` и ABI predefined macros, остаются собственными расширениями `mcpu-cpp`. diff --git a/etc/Makefile.am b/etc/Makefile.am new file mode 100644 index 0000000..6c8989b --- /dev/null +++ b/etc/Makefile.am @@ -0,0 +1,4 @@ +mcpuprivateetcdir = @MCPU_CPP_ETCDIR@ +mcpuprivateetc_DATA = mcpu-cpp.conf + +EXTRA_DIST = mcpu-cpp.conf.in diff --git a/etc/mcpu-cpp.conf.in b/etc/mcpu-cpp.conf.in new file mode 100644 index 0000000..1791a59 --- /dev/null +++ b/etc/mcpu-cpp.conf.in @@ -0,0 +1,41 @@ +/* + * mcpu-cpp configuration. + * + * External text is UTF-8. Path values may use $NAME or ${NAME} + * references. A missing path is harmless; it is simply skipped when + * include files are searched. + */ + +/* + * Main user include directory. It is available in every language state. + * Switched languages add their own language-specific include directories. + * Thus, for example, <diff/a.h> may always be named explicitly through + * this common include root. + */ +MCPU_CPP_INCLUDE_PATH = $HOME/.mcpu/include; +MCPU_CPP_DIFF_INCLUDE_PATH = $HOME/.mcpu/include/diff; +MCPU_CPP_DIFT_INCLUDE_PATH = $HOME/.mcpu/include/dift; +MCPU_CPP_ALG_INCLUDE_PATH = $HOME/.mcpu/include/alg; +MCPU_CPP_AS_INCLUDE_PATH = $HOME/.mcpu/include/as; +MCPU_CPP_AVM_INCLUDE_PATH = $HOME/.mcpu/include/avm; +MCPU_CPP_ACS_INCLUDE_PATH = $HOME/.mcpu/include/acs; + +/* + * Effective root of the MCPU system include tree. + * + * The default is not stored as an absolute installation path in this file. + * At run time mcpu-cpp locates its real executable, takes the parent of its + * executable directory as the MCPU installation root, and uses + * <runtime-root>/include. A higher-priority system or per-user configuration + * may replace that root completely by defining MCPU_CPP_SYSTEM_INCLUDE_PATH. + * Set an empty value in an override configuration to disable the configured + * system tree. + * + * Example override: + * MCPU_CPP_SYSTEM_INCLUDE_PATH = $HOME/mcpu-next/include; + */ +/* + * General fallback path-list searched after explicit -idirafter directories. + * No automatic language subdirectories are added here. + */ +MCPU_CPP_AFTER_INCLUDE_PATH = ; diff --git a/m4/README b/m4/README new file mode 100644 index 0000000..4c2cd60 --- /dev/null +++ b/m4/README @@ -0,0 +1,4 @@ +This directory is reserved for external/vendor Autoconf M4 macros. + +Project-owned mcpu-cpp macros belong in ../acsite.m4 so locally maintained +configure logic remains clearly separated from imported macro packages. diff --git a/man/Makefile.am b/man/Makefile.am new file mode 100644 index 0000000..19d0e63 --- /dev/null +++ b/man/Makefile.am @@ -0,0 +1,6 @@ + +SUBDIRS = ru + +MAN1 = mcpu-cpp.1 + +dist_man_MANS = $(MAN1) diff --git a/man/mcpu-cpp.1 b/man/mcpu-cpp.1 new file mode 100644 index 0000000..489de41 --- /dev/null +++ b/man/mcpu-cpp.1 @@ -0,0 +1,432 @@ +.TH MCPU-CPP 1 "October 2026" "MCPU-CPP 1.0.2" "User Commands" +.SH NAME +mcpu-cpp \- preprocessor for MCPU languages +.SH SYNOPSIS +.B mcpu-cpp +.RI [ options ] +.RI [ input +.RI [ output ]] +.SH DESCRIPTION +.B mcpu-cpp +is the preprocessor for the MCPU toolchain. It processes source text before +language-specific frontends, expands macros, evaluates conditional compilation, +resolves include files, maintains source-location information, and can generate +Make dependencies. +.PP +External text is UTF-8. Text is represented internally as strict UCS-2 through +LibMPUIO. Preprocessing identifiers are Unicode-aware: the first character +must be underscore or have the Unicode XID_Start property; following characters +may also be dollar sign or have XID_Continue. The dollar sign cannot start an +identifier. +.PP +Preprocessor directives use canonical English names only. Unicode remains +available in identifiers, strings, comments, diagnostics, and ordinary source +text. +.PP +If +.I input +is omitted or is +.BR - , +standard input is read. If +.I output +is omitted or is +.BR - , +normal preprocessing output is written to standard output. +.SH TEXT PROCESSING +Backslash-newline splicing is performed before directive parsing. C block +comments and C++-style line comments are removed by the preprocessing phase. +Long invisible source regions are represented compactly while preserving source +coordinates: short forward gaps are emitted as newlines, while larger gaps use +corrective GNU-style line markers. +.SH DIRECTIVES +The principal supported directives are: +.TP +.B #define +Define an object-like or function-like macro. +.TP +.B #undef +Remove a macro definition. +.TP +.BR #if , " #ifdef" , " #ifndef" , " #elif" , " #else" , " #endif" +Control conditional compilation. The +.B defined +operator is supported in +.B #if +expressions. +.TP +.B #include +Include a quoted or angle-bracket header. The operand may be produced by macro +expansion. +.TP +.B #include_next +Continue header lookup after the search-chain element that found the current +header. It is intended primarily for wrapper headers. +.TP +.B #pragma once +Process a physical header only once. Physical file identity is used rather +than source spelling. +.TP +.B #line +Set the logical source line and, optionally, logical source file name. Its +arguments undergo macro expansion; output uses GNU-style line markers. +.TP +.B #error +Emit an error diagnostic and terminate preprocessing unsuccessfully. +.TP +.B #warning +Emit a warning diagnostic and continue unless warning policy promotes it to an +error. +.TP +.B #lang +Push an MCPU language state. The argument is one quoted language name. +.TP +.B #endlang +Restore the previous MCPU language state. +.PP +.B #lang +and +.B #endlang +remain in the output stream for the later frontend dispatcher. The supported +language names are +.BR diff , +.BR dift , +.BR alg , +.BR as , +.BR avm , +and +.BR ACS . +Language matching is case-insensitive in ASCII, while the quoted spelling is +preserved in normalized +.B #lang +output. +.SH MACROS +Object-like and function-like macros are supported, including recursive rescan, +stringification with +.BR # , +token concatenation with +.BR ## , +variadic macros using +.B ... +and +.BR __VA_ARGS__ , +and standard +.BR __VA_OPT__ . +.PP +Ordinary macro arguments are expanded before substitution except where raw +arguments are required by stringification or token concatenation. Replacement +list horizontal whitespace is normalized without changing whitespace inside +quoted tokens or actual arguments. +.SH CONDITIONAL EXPRESSIONS +.B #if +expressions support integer and character constants, the +.B defined +operator, unary, arithmetic, shift, relational, equality, bitwise, logical, +conditional +.BR ?: , +and comma operators, with short-circuit evaluation for +.BR && , +.BR || , +and +.BR ?: . +.PP +Expression evaluation uses one 64-bit model. MCPU also supports the +.B zNNN +and +.B zNNNU +width suffixes for source integer literals with widths not exceeding 64 bits. +A negative shift count reverses shift direction according to the MCPU +preprocessing contract. +.SH INCLUDE SEARCH +For +.BR "#include \"file\"" , +the physical directory containing the current source file is searched first. +This step is omitted for +.BR "#include <file>" . +The remaining effective search order is: +.PP +.nf +explicit -I +explicit -isystem +MCPU_CPP_<LANG>_INCLUDE_PATH +MCPU_CPP_INCLUDE_PATH +MCPU_CPP_SYSTEM_INCLUDE_PATH/<lang> +MCPU_CPP_SYSTEM_INCLUDE_PATH +explicit -idirafter +MCPU_CPP_AFTER_INCLUDE_PATH +.fi +.PP +Within one search class, insertion order is preserved. The logical file name +set by +.B #line +does not alter quoted-header lookup. +.SH OPTIONS +.TP +.BI -o " FILE" +Write normal preprocessing output to +.IR FILE . +.TP +.BI -D " NAME[=VALUE]" +Define a command-line macro. If no value is given, the replacement is +.BR 1 . +.TP +.BI -U " NAME" +Undefine a command-line macro. +.TP +.BI -imacros " FILE" +Preprocess +.I FILE +for its macro and preprocessing state, but discard its ordinary output. All +.B -imacros +files are processed before all +.B -include +files. +.TP +.BI -include " FILE" +Preprocess +.I FILE +before the primary input. +.TP +.BR "-I DIR" ", " "-IDIR" +Add a user include directory. +.TP +.BI -isystem " DIR" +Add an explicit system include directory. +.TP +.BI -idirafter " DIR" +Add an include directory searched after the configured system tree. +.TP +.B -nostdinc +Suppress the effective standard-system include tree. Explicit +.B -isystem +directories remain active. +.TP +.B -dM +After preprocessing, dump non-predefined macro definitions. +.TP +.B -dMP +Dump predefined macros first, followed by the other macro definitions. +.TP +.B -dD +Preserve +.B #define +directives in normal preprocessing output. +.TP +.B -dconfig +Print the effective MCPU-CPP configuration and exit. +.TP +.B -dsearch-dirs +Print the effective include search directories and exit. +.TP +.B -M +Write one Make dependency rule including system headers and suppress normal +preprocessing output. +.TP +.B -MM +Like +.BR -M , +but omit system dependencies. +.TP +.B -MG +With dependency-only +.B -M +or +.BR -MM , +treat missing headers and missing forced files as generated dependencies rather +than errors. It is not valid with +.B -MD +or +.BR -MMD . +.TP +.B -MD +Generate dependencies including system headers while retaining normal +preprocessing output. +.TP +.B -MMD +Like +.BR -MD , +but omit system dependencies. +.TP +.BI -MF " FILE" +Write dependencies to +.IR FILE . +A file name of +.B - +means standard output. This option requires dependency generation. +.TP +.BI -MT " TARGET" +Set an explicit Make dependency target without Make quoting. The option may be +repeated. +.TP +.BI -MQ " TARGET" +Set an explicit Make dependency target with Make quoting. The option may be +repeated. +.TP +.BI --object-suffix " SFX" +Set the suffix used for the automatically derived dependency target. The +default is +.BR .o . +.TP +.B -w +Suppress all warnings. +.TP +.BR -Wcomment , " -Wcomments" +Enable warnings for nested +.B /* +inside a block comment and for backslash-newline inside a +.B // +comment. +.TP +.BR -Wno-comment , " -Wno-comments" +Disable comment warnings, including when +.B -Wall +is present. +.TP +.B -Wall +Enable all optional MCPU-CPP warning classes. +.TP +.B -Werror +Promote every warning that would be emitted to an error. This does not enable +new warning classes. +.TP +.B -Wno-error +Keep emitted warnings as warnings. +.TP +.BI --config-file " FILE" +Read only +.I FILE +as the explicit configuration layer on top of runtime-derived defaults. +.TP +.B --no-config +Do not read configuration files. Runtime-derived defaults remain active. +.TP +.BR -v , " --verbose" +Print configuration and include activity. +.TP +.B --help +Print command-line help and exit successfully. +.TP +.B --version +Print the program version and exit successfully. +.TP +.B -E +Accepted and ignored for compiler-driver compatibility. It is intentionally +not listed by +.BR --help . +.SH DEPENDENCIES +Dependency generation uses the same include traversal as ordinary +preprocessing, including conditional compilation, computed includes, +.BR #include_next , +and +.BR "#pragma once" . +Physical dependencies are deduplicated by physical file identity. +.PP +With +.B -M +or +.BR -MM , +normal preprocessing output is suppressed. With +.B -MD +or +.BR -MMD , +normal preprocessing output is retained and a side-effect dependency file is +written. If +.B -MF +is not specified, the dependency file name is derived from the input or +ordinary +.B -o +output name and receives a +.B .d +suffix. +.SH CONFIGURATION +Before reading configuration files, MCPU-CPP derives +.B MCPU_CPP_SYSTEM_INCLUDE_PATH +as +.IR <runtime-root>/include . +The runtime root is determined from the actual executable location, normally +through Linux +.IR /proc/self/exe . +.PP +Configuration layers are applied in increasing priority: +.PP +.nf +<runtime-root>/etc/mcpu-cpp.conf +/etc/mcpu/mcpu-cpp.conf +$HOME/.mcpu/mcpu-cpp.conf +.fi +.PP +The user configuration path variables are +.BR MCPU_CPP_INCLUDE_PATH , +.BR MCPU_CPP_AFTER_INCLUDE_PATH , +.BR MCPU_CPP_DIFF_INCLUDE_PATH , +.BR MCPU_CPP_DIFT_INCLUDE_PATH , +.BR MCPU_CPP_ALG_INCLUDE_PATH , +.BR MCPU_CPP_AS_INCLUDE_PATH , +.BR MCPU_CPP_AVM_INCLUDE_PATH , +and +.BR MCPU_CPP_ACS_INCLUDE_PATH . +The effective system root is controlled by +.BR MCPU_CPP_SYSTEM_INCLUDE_PATH . +A later definition replaces an earlier one, including an empty value. +.SH SOURCE LOCATIONS +MCPU-CPP emits GNU-style line markers of the form: +.PP +.nf +# line "file" [flags] +.fi +.PP +Flag 1 denotes entry into an included file and flag 2 denotes return to the +including file. Input +.B #line +directives change the logical values observed by +.B __LINE__ +and +.BR __FILE__ . +The predefined source macros also include +.B __BASE_FILE__ +and +.BR __INCLUDE_LEVEL__ . +.SH GNU-COMPATIBLE FEATURES +MCPU-CPP is an independent MCPU preprocessor, but intentionally follows GNU CPP +behavior for documented operations including object-like and function-like +macros, rescan, +.BR # , +.BR ## , +variadic macros, +.BR __VA_OPT__ , +conditional directives, +.BR #include , +.BR #include_next , +.BR "#pragma once" , +.BR #line , +GNU line markers, forced files, Make dependency options, and supported warning +controls. +.PP +MCPU-specific facilities such as +.BR "#lang / #endlang" , +the +.B zNNN +integer-literal suffix, and MCPU ABI predefined macros are native extensions. +.SH FILES +.TP +.I <runtime-root>/etc/mcpu-cpp.conf +Configuration installed with the relocatable MCPU runtime tree. +.TP +.I /etc/mcpu/mcpu-cpp.conf +Optional machine-wide configuration override. +.TP +.I $HOME/.mcpu/mcpu-cpp.conf +Optional per-user configuration override with the highest normal configuration +priority. +.SH EXIT STATUS +.B mcpu-cpp +returns zero after successful preprocessing or after successful informational +operations such as +.B --help +and +.BR --version . +It returns nonzero when command-line processing, configuration, preprocessing, +diagnostics promoted to errors, or output generation fails. +.SH SEE ALSO +.BR libmpu (7), +.BR libmpuio (7), +.BR zubr (1) diff --git a/man/ru/Makefile.am b/man/ru/Makefile.am new file mode 100644 index 0000000..cf13dfc --- /dev/null +++ b/man/ru/Makefile.am @@ -0,0 +1,8 @@ + +LANG = ru + +mandir = @mandir@/$(LANG) + +MAN1 = mcpu-cpp.1 + +dist_man_MANS = $(MAN1) diff --git a/man/ru/mcpu-cpp.1 b/man/ru/mcpu-cpp.1 new file mode 100644 index 0000000..c1bf7e3 --- /dev/null +++ b/man/ru/mcpu-cpp.1 @@ -0,0 +1,428 @@ +.TH MCPU-CPP 1 "Октябрь 2026" "MCPU-CPP 1.0.2" "Команды пользователя" +.SH ИМЯ +mcpu-cpp \- препроцессор языков MCPU +.SH СИНТАКСИС +.B mcpu-cpp +.RI [ параметры ] +.RI [ входной-файл +.RI [ выходной-файл ]] +.SH ОПИСАНИЕ +.B mcpu-cpp +\- препроцессор инструментария MCPU. Он обрабатывает исходный текст до передачи +языковым frontend, раскрывает макро, вычисляет условия условной компиляции, +выполняет поиск include-файлов, поддерживает информацию о позиции в исходном +тексте и может формировать зависимости Make. +.PP +Внешний текст имеет кодировку UTF-8. Внутри текст представлен строгим UCS-2 +средствами LibMPUIO. Идентификаторы препроцессора поддерживают Unicode: первый +символ должен быть подчёркиванием или иметь свойство Unicode XID_Start; +последующие символы дополнительно могут быть знаком доллара или иметь +XID_Continue. Знак доллара не может начинать идентификатор. +.PP +Директивы препроцессора имеют только канонические английские имена. Unicode +полностью сохраняется в идентификаторах, строках, комментариях, диагностике и +обычном исходном тексте. +.PP +Если +.I входной-файл +не задан или равен +.BR - , +читается стандартный ввод. Если +.I выходной-файл +не задан или равен +.BR - , +обычный результат препроцессирования записывается в стандартный вывод. +.SH ОБРАБОТКА ТЕКСТА +Склейка backslash-newline выполняется до разбора директив. Блочные комментарии +C и однострочные комментарии C++ удаляются фазой препроцессирования. Длинные +невидимые участки исходного файла представляются компактно с сохранением +исходных координат: короткие переходы задаются переводами строк, большие +переходы \- корректирующими GNU-style linemarkers. +.SH ДИРЕКТИВЫ +Основные поддерживаемые директивы: +.TP +.B #define +Определяет object-like или function-like macro. +.TP +.B #undef +Удаляет определение макро. +.TP +.BR #if , " #ifdef" , " #ifndef" , " #elif" , " #else" , " #endif" +Управляют условной компиляцией. В выражениях +.B #if +поддерживается оператор +.BR defined . +.TP +.B #include +Включает заголовочный файл в кавычках или угловых скобках. Operand может быть +получен macro expansion. +.TP +.B #include_next +Продолжает поиск header после элемента search chain, которым был найден текущий +header. Директива предназначена прежде всего для wrapper headers. +.TP +.B #pragma once +Обрабатывает физический header только один раз. Используется физическая +идентичность файла, а не написание его имени в исходном тексте. +.TP +.B #line +Задаёт логический номер строки и, необязательно, логическое имя исходного файла. +Аргументы проходят macro expansion; в выходе используются GNU-style +linemarkers. +.TP +.B #error +Выдаёт сообщение об ошибке и немедленно завершает препроцессирование неуспешно. +.TP +.B #warning +Выдаёт предупреждение и продолжает работу, если политика предупреждений не +повышает его до ошибки. +.TP +.B #lang +Помещает состояние языка MCPU в стек. Аргументом является одно имя языка в +кавычках. +.TP +.B #endlang +Восстанавливает предыдущее состояние языка MCPU. +.PP +.B #lang +и +.B #endlang +сохраняются в выходном потоке для последующего frontend dispatcher. +Поддерживаются имена языков +.BR diff , +.BR dift , +.BR alg , +.BR as , +.BR avm +и +.BR ACS . +Сравнение имени языка выполняется без учёта ASCII-регистра, а написание внутри +кавычек сохраняется в нормализованном выводе +.BR #lang . +.SH МАКРО +Поддерживаются object-like и function-like macros, повторный macro rescan, +stringification оператором +.BR # , +token concatenation оператором +.BR ## , +variadic macros с +.B ... +и +.BR __VA_ARGS__ , +а также стандартный +.BR __VA_OPT__ . +.PP +Обычные фактические аргументы раскрываются до подстановки, кроме случаев, когда +stringification или token concatenation требуют сырого аргумента. Горизонтальные +пробелы replacement list нормализуются без изменения пробелов внутри quoted +tokens и фактических аргументов. +.SH УСЛОВНЫЕ ВЫРАЖЕНИЯ +Выражения +.B #if +поддерживают целые и символьные константы, оператор +.BR defined , +унарные, арифметические, shift, relational, equality, bitwise, logical, +условный оператор +.B ?: +и оператор comma. Для +.BR && , +.BR || +и +.B ?: +используется short-circuit evaluation. +.PP +Вычисление выражений выполняется в единой 64-битной модели. MCPU дополнительно +поддерживает суффиксы разрядности +.B zNNN +и +.B zNNNU +для исходных целых литералов с разрядностью не более 64 бит. Отрицательное +значение shift count меняет направление сдвига согласно контракту +препроцессора MCPU. +.SH ПОИСК INCLUDE-ФАЙЛОВ +Для +.B "#include \"file\"" +сначала проверяется физический каталог текущего исходного файла. Для +.B "#include <file>" +этот шаг отсутствует. Остальная effective search chain имеет строгий порядок: +.PP +.nf +explicit -I +explicit -isystem +MCPU_CPP_<LANG>_INCLUDE_PATH +MCPU_CPP_INCLUDE_PATH +MCPU_CPP_SYSTEM_INCLUDE_PATH/<lang> +MCPU_CPP_SYSTEM_INCLUDE_PATH +explicit -idirafter +MCPU_CPP_AFTER_INCLUDE_PATH +.fi +.PP +Внутри одного search class сохраняется порядок добавления. Логическое имя, +заданное +.BR #line , +не изменяет поиск quoted header. +.SH ПАРАМЕТРЫ +.TP +.BI -o " ФАЙЛ" +Записывает обычный результат препроцессирования в +.IR ФАЙЛ . +.TP +.BI -D " ИМЯ[=ЗНАЧЕНИЕ]" +Определяет макро командной строки. +.TP +.BI -U " ИМЯ" +Отменяет определение макро командной строки. +.TP +.BI -imacros " ФАЙЛ" +Полностью препроцессирует +.I ФАЙЛ +для изменения macro/preprocessing state, но отбрасывает его обычный вывод. Все +.B -imacros +обрабатываются раньше всех +.BR -include . +.TP +.BI -include " ФАЙЛ" +Препроцессирует +.I ФАЙЛ +до основного входного файла. +.TP +.BR "-I КАТАЛОГ" ", " "-IКАТАЛОГ" +Добавляет пользовательский include-каталог. +.TP +.BI -isystem " КАТАЛОГ" +Добавляет явный системный include-каталог. +.TP +.BI -idirafter " КАТАЛОГ" +Добавляет include-каталог, который ищется после configured system tree. +.TP +.B -nostdinc +Исключает effective standard-system include tree. Явные каталоги +.B -isystem +остаются активными. +.TP +.B -dM +После препроцессирования выводит определения непредопределённых макро. +.TP +.B -dMP +Сначала выводит предопределённые макро, затем остальные определения. +.TP +.B -dD +Сохраняет директивы +.B #define +в обычном выходном потоке. +.TP +.B -dconfig +Выводит effective configuration MCPU-CPP и завершает работу. +.TP +.B -dsearch-dirs +Выводит effective include search directories и завершает работу. +.TP +.B -M +Выводит одно правило зависимостей Make с системными headers и подавляет обычный +результат препроцессирования. +.TP +.B -MM +Аналог +.BR -M , +но без системных зависимостей. +.TP +.B -MG +В dependency-only режимах +.B -M +или +.B -MM +считает отсутствующие headers и forced files генерируемыми зависимостями, а не +ошибками. С +.B -MD +и +.B -MMD +не используется. +.TP +.B -MD +Формирует зависимости с системными headers, сохраняя обычный результат +препроцессирования. +.TP +.B -MMD +Аналог +.BR -MD , +но без системных зависимостей. +.TP +.BI -MF " ФАЙЛ" +Записывает зависимости в +.IR ФАЙЛ . +Имя +.B - +означает стандартный вывод. Параметр имеет смысл только при генерации +зависимостей. +.TP +.BI -MT " ЦЕЛЬ" +Задаёт явную цель правила Make без Make quoting. Параметр можно повторять. +.TP +.BI -MQ " ЦЕЛЬ" +Задаёт явную цель правила Make с Make quoting. Параметр можно повторять. +.TP +.BI --object-suffix " СУФФИКС" +Задаёт суффикс автоматически формируемой цели зависимости. По умолчанию +используется +.BR .o . +.TP +.B -w +Подавляет все предупреждения. +.TP +.BR -Wcomment , " -Wcomments" +Включает предупреждения о вложенном +.B /* +внутри блочного комментария и о backslash-newline внутри комментария +.BR // . +.TP +.BR -Wno-comment , " -Wno-comments" +Выключает comment warnings, в том числе при наличии +.BR -Wall . +.TP +.B -Wall +Включает все optional warning classes MCPU-CPP. +.TP +.B -Werror +Повышает каждое реально выдаваемое предупреждение до ошибки, но само не +включает новые классы предупреждений. +.TP +.B -Wno-error +Оставляет выдаваемые предупреждения предупреждениями. +.TP +.BI --config-file " ФАЙЛ" +Использует только явно выбранный +.I ФАЙЛ +как configuration layer поверх runtime-derived defaults. +.TP +.B --no-config +Не читает configuration files. Runtime-derived defaults остаются активными. +.TP +.BR -v , " --verbose" +Выводит информацию об effective configuration и include activity. +.TP +.B --help +Выводит краткую справку по командной строке и успешно завершает работу. +.TP +.B --version +Выводит версию программы и успешно завершает работу. +.TP +.B -E +Принимается и игнорируется для совместимости с compiler drivers. Параметр +намеренно не показывается в +.BR --help . +.SH ЗАВИСИМОСТИ +Генерация зависимостей использует тот же include traversal, что и обычное +препроцессирование, включая условную компиляцию, computed includes, +.BR #include_next +и +.BR "#pragma once" . +Физические зависимости дедуплицируются по физической идентичности файла. +.PP +В режимах +.B -M +и +.B -MM +обычный результат препроцессирования подавляется. В режимах +.B -MD +и +.B -MMD +он сохраняется, а dependency file формируется как side effect. Если +.B -MF +не задан, имя dependency file выводится из имени входного файла или обычного +выхода +.B -o +и получает суффикс +.BR .d . +.SH КОНФИГУРАЦИЯ +До чтения configuration files MCPU-CPP формирует +.B MCPU_CPP_SYSTEM_INCLUDE_PATH +как +.IR <runtime-root>/include . +Runtime root определяется из фактического расположения исполняемого файла, +обычно через Linux +.IR /proc/self/exe . +.PP +Configuration layers применяются в порядке возрастания приоритета: +.PP +.nf +<runtime-root>/etc/mcpu-cpp.conf +/etc/mcpu/mcpu-cpp.conf +$HOME/.mcpu/mcpu-cpp.conf +.fi +.PP +Пользовательские переменные путей: +.BR MCPU_CPP_INCLUDE_PATH , +.BR MCPU_CPP_AFTER_INCLUDE_PATH , +.BR MCPU_CPP_DIFF_INCLUDE_PATH , +.BR MCPU_CPP_DIFT_INCLUDE_PATH , +.BR MCPU_CPP_ALG_INCLUDE_PATH , +.BR MCPU_CPP_AS_INCLUDE_PATH , +.BR MCPU_CPP_AVM_INCLUDE_PATH +и +.BR MCPU_CPP_ACS_INCLUDE_PATH . +Effective system root задаётся +.BR MCPU_CPP_SYSTEM_INCLUDE_PATH . +Более позднее определение полностью заменяет раннее, включая пустое значение. +.SH ПОЗИЦИИ В ИСХОДНОМ ТЕКСТЕ +MCPU-CPP выводит GNU-style linemarkers вида: +.PP +.nf +# line "file" [flags] +.fi +.PP +Флаг 1 означает вход во включённый файл, флаг 2 \- возврат в включающий файл. +Входная директива +.B #line +изменяет логические значения +.B __LINE__ +и +.BR __FILE__ . +К предопределённым source macros также относятся +.B __BASE_FILE__ +и +.BR __INCLUDE_LEVEL__ . +.SH GNU-COMPATIBLE FEATURES +MCPU-CPP является самостоятельным препроцессором MCPU, но для документированных +операций намеренно повторяет поведение GNU CPP. К ним относятся object-like и +function-like macros, rescan, +.BR # , +.BR ## , +variadic macros, +.BR __VA_OPT__ , +условные директивы, +.BR #include , +.BR #include_next , +.BR "#pragma once" , +.BR #line , +GNU linemarkers, forced files, параметры генерации зависимостей Make и +поддерживаемые warning controls. +.PP +MCPU-специфичные средства +.BR "#lang / #endlang" , +суффикс целых литералов +.B zNNN +и предопределённые макро ABI MCPU являются собственными расширениями. +.SH ФАЙЛЫ +.TP +.I <runtime-root>/etc/mcpu-cpp.conf +Конфигурация, установленная вместе с перемещаемым runtime tree MCPU. +.TP +.I /etc/mcpu/mcpu-cpp.conf +Необязательная общесистемная configuration override. +.TP +.I $HOME/.mcpu/mcpu-cpp.conf +Необязательная пользовательская configuration override с наивысшим обычным +приоритетом. +.SH КОД ЗАВЕРШЕНИЯ +.B mcpu-cpp +возвращает ноль после успешного препроцессирования и после успешных +информационных операций, например +.B --help +и +.BR --version . +При ошибке командной строки, конфигурации, препроцессирования, диагностике, +повышенной до ошибки, или ошибке вывода возвращается ненулевое значение. +.SH СМ. ТАКЖЕ +.BR libmpu (7), +.BR libmpuio (7), +.BR zubr (1) diff --git a/src/Makefile.am b/src/Makefile.am new file mode 100644 index 0000000..b5e2ea2 --- /dev/null +++ b/src/Makefile.am @@ -0,0 +1,64 @@ + +mcpuprivatebindir = @MCPU_CPP_BINDIR@ +mcpuprivatebin_PROGRAMS = mcpu-cpp + +mcpu_cpp_SOURCES = \ + main.c \ + defs.h \ + mcpp-options.c \ + mcpp-options.h \ + mcpp-diagnostic.c \ + mcpp-diagnostic.h \ + mcpp-text.c \ + mcpp-text.h \ + mcpp-config-file.c \ + mcpp-config-file.h \ + mcpp-runtime.c \ + mcpp-runtime.h \ + mcpp-language.c \ + mcpp-language.h \ + mcpp-include-path.c \ + mcpp-include-path.h \ + mcpp-source.c \ + mcpp-source.h \ + mcpp-lexer.c \ + mcpp-lexer.h \ + mcpp-macro.c \ + mcpp-macro.h \ + mcpp-semantic.c \ + mcpp-semantic.h \ + mcpp-expr.c \ + mcpp-expr.h \ + mcpp-expression.c \ + mcpp-expression.h \ + mcpp-predefined.c \ + mcpp-predefined.h \ + mcpp-lib.c \ + mcpp-lib.h + +mcpu_cpp_CFLAGS = $(LIBMPUIO_CFLAGS) +mcpu_cpp_LDFLAGS = $(LIBMPUIO_LDFLAGS) +mcpu_cpp_LDADD = $(LIBMPUIO_LIBS) + +EXTRA_DIST = mcpp-expr.zubr + +ZUBR = zubr +SUFFIXES = .zubr .c + +.zubr.c: + $(ZUBR) -vl -s -Bmcpp_ -o $@ $< + +CLEANFILES = z.output + + +install-exec-hook: + $(MKDIR_P) "$(DESTDIR)$(bindir)" + rm -f "$(DESTDIR)$(bindir)/mcpu-cpp" + $(LN_S) "@MCPU_CPP_PUBLIC_LINK_TARGET@" "$(DESTDIR)$(bindir)/mcpu-cpp" + +uninstall-hook: + rm -f "$(DESTDIR)$(bindir)/mcpu-cpp" + +distclean-local: + -rm -rf $(DEPDIR) + diff --git a/src/defs.h b/src/defs.h new file mode 100644 index 0000000..2b22f15 --- /dev/null +++ b/src/defs.h @@ -0,0 +1,42 @@ +#ifndef __MCPU_CPP_DEFS_H__ +#define __MCPU_CPP_DEFS_H__ 1 + +#ifdef HAVE_CONFIG_H +#include <config.h> +#endif + +#include <errno.h> +#include <limits.h> +#include <signal.h> +#include <stdint.h> +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <time.h> + +#include <libmpuio.h> + +#define MCPU_CPP_LANG_STACK_SIZE 400 +#define MCPU_CPP_INCLUDE_STACK_SIZE 400 +#define MCPU_CPP_CONDITIONAL_STACK_SIZE 400 + +#ifndef PATH_MAX +#define PATH_MAX 4096 +#endif + +#include <mcpp-text.h> +#include <mcpp-config-file.h> +#include <mcpp-runtime.h> +#include <mcpp-language.h> +#include <mcpp-include-path.h> +#include <mcpp-lexer.h> +#include <mcpp-macro.h> +#include <mcpp-semantic.h> +#include <mcpp-expression.h> +#include <mcpp-predefined.h> +#include <mcpp-source.h> +#include <mcpp-options.h> +#include <mcpp-diagnostic.h> +#include <mcpp-lib.h> + +#endif /* __MCPU_CPP_DEFS_H__ */ diff --git a/src/main.c b/src/main.c new file mode 100644 index 0000000..0882904 --- /dev/null +++ b/src/main.c @@ -0,0 +1,412 @@ +#include <defs.h> + +static void +welcome( mcpp_options *opts ) +{ + if( opts == NULL ) + return; + + printf( "mcpu-cpp %s\n", PACKAGE_VERSION ); + printf( "MCPU languages preprocessor\n" ); +} + +static void +usage( mcpp_options *opts ) +{ + printf( "Usage: %s [options] [input [output]]\n\n", opts->progname ); + printf( "Options:\n" ); + printf( " -o FILE write output to FILE\n" ); + printf( " -D NAME[=VALUE] define a command-line macro\n" ); + printf( " -U NAME undefine a command-line macro\n" ); + printf( " -imacros FILE preprocess FILE for macro state; discard its output\n" ); + printf( " -include FILE preprocess FILE before the primary input\n" ); + printf( " -I DIR, -IDIR add user include directory\n" ); + printf( " -isystem DIR add explicit system include directory\n" ); + printf( " -idirafter DIR add include directory searched last\n" ); + printf( " -nostdinc suppress effective standard-system include tree\n" ); + printf( " -dM dump non-predefined macros after preprocessing\n" ); + printf( " -dMP dump predefined macros first, then other macros\n" ); + printf( " -dD preserve #define directives in normal output\n" ); + printf( " -dconfig dump effective mcpu-cpp configuration\n" ); + printf( " -dsearch-dirs dump effective include search directories\n" ); + printf( " -M output make dependencies including system headers\n" ); + printf( " -MM output make dependencies excluding system headers\n" ); + printf( " -MG treat missing headers as generated dependencies\n" ); + printf( " -MD write dependencies and keep preprocessing output\n" ); + printf( " -MMD like -MD but exclude system headers\n" ); + printf( " -MF FILE write dependencies to FILE ('-' means stdout)\n" ); + printf( " -MT TARGET set unquoted make dependency target\n" ); + printf( " -MQ TARGET set make-quoted dependency target\n" ); + printf( " --object-suffix SFX set object suffix used for dependency targets\n" ); + printf( " -w suppress all warnings\n" ); + printf( " -Wcomment[s] warn about nested /* and multi-line // comments\n" ); + printf( " -Wno-comment[s] disable comment warnings even under -Wall\n" ); + printf( " -Wall enable all optional warning classes\n" ); + printf( " -Werror promote every emitted warning to an error\n" ); + printf( " -Wno-error keep emitted warnings as warnings\n" ); + printf( " --config-file FILE use only FILE as configuration\n" ); + printf( " --no-config do not read configuration files; keep runtime defaults\n" ); + printf( " -v, --verbose print configuration/include activity\n" ); + printf( " --help display this help and exit\n" ); + printf( " --version display version and exit\n\n" ); + printf( "If input is omitted or '-', read standard input.\n" ); + printf( "If output is omitted or '-', write standard output.\n" ); +} + +static void +pipe_closed( int sig ) +{ + (void)sig; + _Exit( 0 ); +} + +static int +apply_include_options( mcpp *cpp, const mcpp_options *opts ) +{ + size_t i; + + for( i = 0; i < opts->action_count; ++i ) + { + const mcpp_option_action *action = &opts->actions[i]; + enum mcpu_include_class class_type; + + switch( action->kind ) + { + case MCPP_ACTION_INCLUDE_USER: + if( action->argument[0] == 0 ) + continue; + class_type = MCPU_INCLUDE_USER_EXPLICIT; + break; + case MCPP_ACTION_INCLUDE_SYSTEM: + class_type = MCPU_INCLUDE_SYSTEM_EXPLICIT; + break; + case MCPP_ACTION_INCLUDE_AFTER: + class_type = MCPU_INCLUDE_AFTER_EXPLICIT; + break; + default: + continue; + } + + if( mcpu_include_paths_add(&cpp->include_paths, + action->argument, class_type) != 0 ) + return( -1 ); + } + + return( 0 ); +} + +static int +dump_verbose_config( const mcpp *cpp, FILE *stream ) +{ + enum mcpu_language language; + const char *name; + const char *value; + + if( cpp == NULL || stream == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + for( language = MCPU_LANG_DIFF; + language < MCPU_LANG__COUNT; + language = (enum mcpu_language)(language + 1) ) + { + name = mcpu_language_config_variable( language ); + value = mcpu_config_get( &cpp->config, name ); + if( value != NULL && fprintf(stream, "config: %s=%s\n", name, value) < 0 ) + return( -1 ); + } + + name = "MCPU_CPP_INCLUDE_PATH"; + value = mcpu_config_get( &cpp->config, name ); + if( value != NULL && fprintf(stream, "config: %s=%s\n", name, value) < 0 ) + return( -1 ); + + name = "MCPU_CPP_SYSTEM_INCLUDE_PATH"; + value = mcpu_config_get( &cpp->config, name ); + if( value != NULL && fprintf(stream, "config: %s=%s\n", name, value) < 0 ) + return( -1 ); + + name = "MCPU_CPP_AFTER_INCLUDE_PATH"; + value = mcpu_config_get( &cpp->config, name ); + if( value != NULL && fprintf(stream, "config: %s=%s\n", name, value) < 0 ) + return( -1 ); + + return( ferror(stream) ? -1 : 0 ); +} + + +static FILE * +open_output( const mcpp_options *opts, int *close_stream ) +{ + FILE *stream; + + *close_stream = 0; + if( opts->out_fname == NULL || opts->out_fname[0] == 0 ) + return( stdout ); + + stream = fopen( opts->out_fname, "wb" ); + if( stream != NULL ) + *close_stream = 1; + + return( stream ); +} + + +static char * +dependency_default_filename( const mcpp_options *opts ) +{ + const char *source; + const char *base; + const char *dot; + size_t stem_length; + char *filename; + + if( opts == NULL ) + { + errno = EINVAL; + return( NULL ); + } + + if( opts->out_fname != NULL && opts->out_fname[0] != 0 ) + { + source = opts->out_fname; + base = source; + } + else if( opts->in_fname != NULL && opts->in_fname[0] != 0 ) + { + base = strrchr( opts->in_fname, '/' ); + source = base == NULL ? opts->in_fname : base + 1; + base = source; + } + else + { + source = "-"; + base = source; + } + + dot = strrchr( base, '.' ); + stem_length = dot != NULL && dot != base ? (size_t)(dot - source) : strlen(source); + + filename = (char *)malloc( stem_length + 3 ); + if( filename == NULL ) + return( NULL ); + + memcpy( filename, source, stem_length ); + memcpy( filename + stem_length, ".d", 3 ); + return( filename ); +} + + +static FILE * +open_dependency_output( const mcpp_options *opts, int side_effect, + char **allocated_name, int *close_stream ) +{ + const char *name = NULL; + FILE *stream; + + if( opts == NULL || allocated_name == NULL || close_stream == NULL ) + { + errno = EINVAL; + return( NULL ); + } + + *allocated_name = NULL; + *close_stream = 0; + + if( opts->deps_file != NULL ) + name = opts->deps_file; + else if( side_effect ) + { + *allocated_name = dependency_default_filename( opts ); + if( *allocated_name == NULL ) + return( NULL ); + name = *allocated_name; + } + else if( opts->out_fname != NULL && opts->out_fname[0] != 0 ) + name = opts->out_fname; + + if( name == NULL || name[0] == 0 || strcmp(name, "-") == 0 ) + return( stdout ); + + stream = fopen( name, "wb" ); + if( stream != NULL ) + *close_stream = 1; + else + fprintf( stderr, "%s: cannot open dependency output '%s': %s\n", + opts->progname, name, strerror(errno) ); + return( stream ); +} + +static int +run_preprocessor( mcpp *cpp, mcpp_options *opts ) +{ + FILE *stream; + int close_stream; + int rc; + + cpp->options_data = opts; + cpp->verbose = opts->verbose; + cpp->include_paths.no_standard_includes = opts->no_standard_includes; + + if( mcpp_load_config(cpp, opts->config_fname, opts->no_config) != 0 ) + return( -1 ); + + if( cpp->verbose && dump_verbose_config(cpp, stderr) != 0 ) + return( -1 ); + + if( mcpu_include_paths_from_config(&cpp->include_paths, &cpp->config) != 0 || + apply_include_options(cpp, opts) != 0 ) + return( -1 ); + + if( opts->dump_config ) + { + if( opts->out_fname[0] != 0 ) + { + fprintf( stderr, "%s: -o/output file has no meaning with -dconfig\n", + opts->progname ); + return( -1 ); + } + return( mcpu_config_dump(&cpp->config, stdout) ); + } + + if( opts->dump_search_dirs ) + { + if( opts->out_fname[0] != 0 ) + { + fprintf( stderr, "%s: -o/output file has no meaning with -dsearch-dirs\n", + opts->progname ); + return( -1 ); + } + return( mcpu_include_paths_dump(&cpp->include_paths, stdout) ); + } + + if( mcpp_options_has_unimplemented(opts) ) + { + fprintf( stderr, + "%s: requested option is parsed but its handler is not implemented yet\n", + opts->progname ); + return( -1 ); + } + + if( opts->in_fname[0] != 0 ) + rc = mcpp_process( cpp, opts->in_fname ); + else + rc = mcpp_process_stream( cpp, stdin, "<stdin>" ); + if( rc != 0 ) + return( -1 ); + + if( opts->print_deps && opts->inhibit_output ) + { + char *allocated_name; + + stream = open_dependency_output( opts, 0, &allocated_name, &close_stream ); + if( stream == NULL ) + { + free( allocated_name ); + return( -1 ); + } + + rc = mcpp_write_dependencies( cpp, stream, opts->print_deps == 2 ); + if( close_stream && fclose(stream) != 0 ) + rc = -1; + free( allocated_name ); + return( rc ); + } + + if( opts->dump_macros == MCPP_DUMP_ONLY ) + { + stream = open_output( opts, &close_stream ); + if( stream == NULL ) + return( -1 ); + + rc = mcpp_dump_macros( cpp, stream ); + if( close_stream && fclose(stream) != 0 ) + rc = -1; + return( rc ); + } + + rc = mcpp_write_output(cpp, + opts->out_fname[0] != 0 ? opts->out_fname : NULL); + if( rc != 0 || !opts->print_deps ) + return( rc ); + + { + char *allocated_name; + + stream = open_dependency_output( opts, 1, &allocated_name, &close_stream ); + if( stream == NULL ) + { + free( allocated_name ); + return( -1 ); + } + + rc = mcpp_write_dependencies( cpp, stream, opts->print_deps == 2 ); + if( close_stream && fclose(stream) != 0 ) + rc = -1; + free( allocated_name ); + } + + return( rc ); +} + +int +main( int argc, char **argv ) +{ + mcpp cpp; + mcpp_options options; + const char *p; + int mpu_initialized = 0; + int rc = 1; + + mcpp_options_init( &options, argc ); + if( options.actions == NULL ) + return( 1 ); + + p = argv[0] + strlen( argv[0] ); + while( p != argv[0] && p[-1] != '/' ) --p; + options.progname = p; + +#ifdef SIGPIPE + signal( SIGPIPE, pipe_closed ); +#endif + + if( mcpp_handle_options(&options, argc - 1, argv + 1) != 0 ) + goto done_options; + + if( options.version ) + { + welcome( &options ); + rc = 0; + goto done_options; + } + + if( options.help ) + { + welcome( &options ); + usage( &options ); + rc = 0; + goto done_options; + } + + __mpu_init(); + mpu_initialized = 1; + mcpp_init( &cpp ); + + if( mcpp_runtime_paths_discover(&cpp.runtime_paths, argv[0]) != 0 ) + { + fprintf( stderr, "%s: cannot determine MCPU installation root: %s\n", + options.progname, strerror(errno) ); + } + else if( run_preprocessor(&cpp, &options) == 0 ) + rc = 0; + + mcpp_free( &cpp ); + if( mpu_initialized ) + __mpu_free_context(); + +done_options: + mcpp_options_free( &options ); + return( rc ); +} diff --git a/src/mcpp-config-file.c b/src/mcpp-config-file.c new file mode 100644 index 0000000..8022af7 --- /dev/null +++ b/src/mcpp-config-file.c @@ -0,0 +1,512 @@ +#include <defs.h> + +static char * +read_bytes( const char *filename, size_t *file_length ) +{ + FILE *fp; + char *data; + long size; + size_t n; + + if( file_length == NULL ) + { + errno = EINVAL; + return( NULL ); + } + + *file_length = 0; + + fp = fopen( filename, "rb" ); + if( fp == NULL ) + return( NULL ); + + if( fseek( fp, 0, SEEK_END ) != 0 ) + { + fclose( fp ); + return( NULL ); + } + + size = ftell( fp ); + if( size < 0 || fseek( fp, 0, SEEK_SET ) != 0 ) + { + fclose( fp ); + return( NULL ); + } + + data = (char *)calloc( (size_t)size + 1, 1 ); + if( data == NULL ) + { + fclose( fp ); + return( NULL ); + } + + n = fread( data, 1, (size_t)size, fp ); + if( n != (size_t)size && ferror(fp) ) + { + free( data ); + fclose( fp ); + return( NULL ); + } + + data[n] = 0; + fclose( fp ); + *file_length = n; + + return( data ); +} + +static int +is_space16( __mpu_char16_t c ) +{ + return( c == ' ' || c == '\t' || c == '\r' || c == '\n' || + c == '\f' || c == '\v' ); +} + +static int +is_name16( __mpu_char16_t c ) +{ + return( (c >= 'a' && c <= 'z') || + (c >= 'A' && c <= 'Z') || + (c >= '0' && c <= '9') || c == '_' ); +} + +static void +skip_space_and_comments( const __mpu_char16_t *s, size_t n, size_t *pos ) +{ + size_t p = *pos; + + for( ;; ) + { + while( p < n && is_space16(s[p]) ) + ++p; + + if( p + 1 < n && s[p] == '/' && s[p + 1] == '*' ) + { + p += 2; + while( p + 1 < n && !(s[p] == '*' && s[p + 1] == '/') ) + ++p; + if( p + 1 < n ) + p += 2; + continue; + } + + if( p + 1 < n && s[p] == '/' && s[p + 1] == '/' ) + { + p += 2; + while( p < n && s[p] != '\n' ) + ++p; + continue; + } + + if( p < n && s[p] == '#' ) + { + while( p < n && s[p] != '\n' ) + ++p; + continue; + } + + break; + } + + *pos = p; +} + +static char * +expand_value( const mcpu_config *config, const char *value ) +{ + size_t length = strlen( value ); + size_t capacity = length + 64; + char *out = (char *)calloc( capacity, 1 ); + size_t i = 0; + size_t o = 0; + + if( out == NULL ) + return( NULL ); + + while( i < length ) + { + if( value[i] == '$' ) + { + char name[256]; + size_t ni = 0; + size_t j = i + 1; + int braced = 0; + const char *replacement; + + if( j < length && value[j] == '{' ) + { + braced = 1; + ++j; + } + + while( j < length && ni + 1 < sizeof(name) && + ((value[j] >= 'a' && value[j] <= 'z') || + (value[j] >= 'A' && value[j] <= 'Z') || + (value[j] >= '0' && value[j] <= '9') || value[j] == '_') ) + name[ni++] = value[j++]; + + if( braced ) + { + if( j >= length || value[j] != '}' ) + { + free( out ); + errno = EINVAL; + return( NULL ); + } + ++j; + } + + if( ni == 0 ) + { + out[o++] = value[i++]; + continue; + } + + name[ni] = 0; + replacement = mcpu_config_get( config, name ); + if( replacement == NULL ) + replacement = getenv( name ); + if( replacement == NULL ) + replacement = ""; + + { + size_t rl = strlen( replacement ); + if( o + rl + 1 > capacity ) + { + char *p; + while( o + rl + 1 > capacity ) + capacity *= 2; + p = (char *)realloc( out, capacity ); + if( p == NULL ) + { + free( out ); + return( NULL ); + } + out = p; + } + memcpy( out + o, replacement, rl ); + o += rl; + } + + i = j; + continue; + } + + if( o + 2 > capacity ) + { + char *p; + capacity *= 2; + p = (char *)realloc( out, capacity ); + if( p == NULL ) + { + free( out ); + return( NULL ); + } + out = p; + } + + out[o++] = value[i++]; + } + + out[o] = 0; + return( out ); +} + +void +mcpu_config_init( mcpu_config *config ) +{ + if( config == NULL ) + return; + + config->first = NULL; + config->verbose = 0; +} + +void +mcpu_config_free( mcpu_config *config ) +{ + mcpu_config_entry *p; + + if( config == NULL ) + return; + + p = config->first; + while( p ) + { + mcpu_config_entry *next = p->next; + free( p->name ); + free( p->value ); + free( p ); + p = next; + } + + config->first = NULL; +} + +const char * +mcpu_config_get( const mcpu_config *config, const char *name ) +{ + mcpu_config_entry *p; + + if( config == NULL || name == NULL ) + return( NULL ); + + for( p = config->first; p; p = p->next ) + if( strcmp( p->name, name ) == 0 ) + return( p->value ); + + return( NULL ); +} + +int +mcpu_config_set( mcpu_config *config, const char *name, + const char *value ) +{ + mcpu_config_entry *p; + char *new_value; + + if( config == NULL || name == NULL || value == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + new_value = expand_value( config, value ); + if( new_value == NULL ) + return( -1 ); + + for( p = config->first; p; p = p->next ) + { + if( strcmp( p->name, name ) == 0 ) + { + free( p->value ); + p->value = new_value; + return( 0 ); + } + } + + p = (mcpu_config_entry *)calloc( 1, sizeof(*p) ); + if( p == NULL ) + { + free( new_value ); + return( -1 ); + } + + p->name = strdup( name ); + if( p->name == NULL ) + { + free( new_value ); + free( p ); + return( -1 ); + } + + p->value = new_value; + p->next = config->first; + config->first = p; + + return( 0 ); +} + + +static int +config_entry_compare( const void *a, const void *b ) +{ + const mcpu_config_entry *ea = *(const mcpu_config_entry * const *)a; + const mcpu_config_entry *eb = *(const mcpu_config_entry * const *)b; + return( strcmp(ea->name, eb->name) ); +} + + +int +mcpu_config_dump( const mcpu_config *config, FILE *stream ) +{ + const mcpu_config_entry *p; + mcpu_config_entry **list; + size_t count = 0; + size_t i = 0; + + if( config == NULL || stream == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + for( p = config->first; p; p = p->next ) + ++count; + + list = count ? (mcpu_config_entry **)calloc( count, sizeof(*list) ) : NULL; + if( count != 0 && list == NULL ) + return( -1 ); + + for( p = config->first; p; p = p->next ) + list[i++] = (mcpu_config_entry *)p; + + if( count > 1 ) + qsort( list, count, sizeof(*list), config_entry_compare ); + + for( i = 0; i < count; ++i ) + { + if( fprintf(stream, "%s = %s;\n", list[i]->name, list[i]->value) < 0 ) + { + free( list ); + return( -1 ); + } + } + + free( list ); + return( ferror(stream) ? -1 : 0 ); +} + + +int +mcpu_config_read( mcpu_config *config, const char *filename, + int missing_is_error ) +{ + char *bytes; + size_t byte_length; + mcpu_text text; + size_t p = 0; + int rc = -1; + + if( config == NULL || filename == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + bytes = read_bytes( filename, &byte_length ); + if( bytes == NULL ) + { + if( !missing_is_error && errno == ENOENT ) + return( 0 ); + return( -1 ); + } + + mcpu_text_init( &text ); + { + const char *input = bytes; + + if( memchr( bytes, 0, byte_length ) != NULL ) + { + fprintf( stderr, "%s: NUL character is not allowed in configuration text\n", + filename ); + free( bytes ); + return( -1 ); + } + + if( byte_length >= 3 && + (unsigned char)bytes[0] == 0xef && + (unsigned char)bytes[1] == 0xbb && + (unsigned char)bytes[2] == 0xbf ) + input += 3; + + if( mcpu_text_from_utf8( &text, input ) != 0 ) + { + fprintf( stderr, "%s: invalid UTF-8 or non-UCS-2 character\n", + filename ); + free( bytes ); + return( -1 ); + } + } + free( bytes ); + + while( p < text.length ) + { + size_t name_start; + size_t name_end; + size_t value_start; + size_t value_end; + char *name = NULL; + char *value = NULL; + + skip_space_and_comments( text.data, text.length, &p ); + if( p >= text.length ) + break; + + name_start = p; + while( p < text.length && is_name16(text.data[p]) ) + ++p; + name_end = p; + + if( name_end == name_start ) + { + fprintf( stderr, "%s: malformed configuration variable\n", filename ); + goto done; + } + + skip_space_and_comments( text.data, text.length, &p ); + if( p >= text.length || text.data[p] != '=' ) + { + fprintf( stderr, "%s: '=' expected after configuration variable\n", filename ); + goto done; + } + ++p; + + while( p < text.length && is_space16(text.data[p]) ) + ++p; + + value_start = p; + if( p < text.length && (text.data[p] == '"' || text.data[p] == '\'') ) + { + __mpu_char16_t quote = text.data[p++]; + value_start = p; + while( p < text.length && text.data[p] != quote ) + { + if( text.data[p] == '\\' && p + 1 < text.length ) + p += 2; + else + ++p; + } + if( p >= text.length ) + { + fprintf( stderr, "%s: unterminated quoted configuration value\n", filename ); + goto done; + } + value_end = p++; + while( p < text.length && is_space16(text.data[p]) ) + ++p; + } + else + { + while( p < text.length && text.data[p] != ';' && text.data[p] != '\n' ) + ++p; + value_end = p; + while( value_end > value_start && is_space16(text.data[value_end - 1]) ) + --value_end; + } + + if( p >= text.length || text.data[p] != ';' ) + { + fprintf( stderr, "%s: ';' expected after configuration value\n", filename ); + goto done; + } + ++p; + + name = mcpu_text_to_utf8( text.data + name_start, name_end - name_start ); + value = mcpu_text_to_utf8( text.data + value_start, value_end - value_start ); + if( name == NULL || value == NULL ) + { + free( name ); + free( value ); + goto done; + } + + if( mcpu_config_set( config, name, value ) != 0 ) + { + free( name ); + free( value ); + goto done; + } + + free( name ); + free( value ); + name = NULL; + value = NULL; + } + + rc = 0; + +done: + mcpu_text_free( &text ); + return( rc ); +} diff --git a/src/mcpp-config-file.h b/src/mcpp-config-file.h new file mode 100644 index 0000000..b2923ab --- /dev/null +++ b/src/mcpp-config-file.h @@ -0,0 +1,30 @@ +#ifndef __MCPU_CPP_CONFIG_FILE_H__ +#define __MCPU_CPP_CONFIG_FILE_H__ 1 + +#include <defs.h> + +typedef struct mcpu_config_entry mcpu_config_entry; +struct mcpu_config_entry +{ + char *name; + char *value; + mcpu_config_entry *next; +}; + +typedef struct mcpu_config mcpu_config; +struct mcpu_config +{ + mcpu_config_entry *first; + int verbose; +}; + +void mcpu_config_init( mcpu_config *config ); +void mcpu_config_free( mcpu_config *config ); +int mcpu_config_read( mcpu_config *config, const char *filename, + int missing_is_error ); +const char *mcpu_config_get( const mcpu_config *config, const char *name ); +int mcpu_config_set( mcpu_config *config, const char *name, + const char *value ); +int mcpu_config_dump( const mcpu_config *config, FILE *stream ); + +#endif /* __MCPU_CPP_CONFIG_FILE_H__ */ diff --git a/src/mcpp-diagnostic.c b/src/mcpp-diagnostic.c new file mode 100644 index 0000000..76e0c25 --- /dev/null +++ b/src/mcpp-diagnostic.c @@ -0,0 +1,28 @@ +#include <defs.h> +#include <stdarg.h> + +int +mcpp_diagnostic_warning( const mcpp_options *options, + const char *filename, + unsigned line_number, + const char *format, ... ) +{ + va_list ap; + int as_error; + + if( options != NULL && options->inhibit_warnings ) + return( 0 ); + + as_error = options != NULL && options->warnings_are_errors; + + fprintf( stderr, "%s:%u: %s: ", + filename != NULL ? filename : "<input>", line_number, + as_error ? "error" : "warning" ); + + va_start( ap, format ); + vfprintf( stderr, format, ap ); + va_end( ap ); + fputc( '\n', stderr ); + + return( as_error ? -1 : 0 ); +} diff --git a/src/mcpp-diagnostic.h b/src/mcpp-diagnostic.h new file mode 100644 index 0000000..93cccb1 --- /dev/null +++ b/src/mcpp-diagnostic.h @@ -0,0 +1,11 @@ +#ifndef __MCPU_CPP_DIAGNOSTIC_H__ +#define __MCPU_CPP_DIAGNOSTIC_H__ 1 + +typedef struct mcpp_options mcpp_options; + +int mcpp_diagnostic_warning( const mcpp_options *options, + const char *filename, + unsigned line_number, + const char *format, ... ); + +#endif /* __MCPU_CPP_DIAGNOSTIC_H__ */ diff --git a/src/mcpp-expr.h b/src/mcpp-expr.h new file mode 100644 index 0000000..6c5a962 --- /dev/null +++ b/src/mcpp-expr.h @@ -0,0 +1,10 @@ +#ifndef __MCPU_CPP_EXPR_H__ +#define __MCPU_CPP_EXPR_H__ 1 + +#include <defs.h> + +int mcpp_expr_parse( const __mpu_char16_t *text, size_t length, + const char *filename, unsigned line_number, + const mcpp_options *options, int *result ); + +#endif /* __MCPU_CPP_EXPR_H__ */ diff --git a/src/mcpp-expr.zubr b/src/mcpp-expr.zubr new file mode 100644 index 0000000..1ec5823 --- /dev/null +++ b/src/mcpp-expr.zubr @@ -0,0 +1,723 @@ +/*************************************************************** + MCPP_EXPR.C + + This file containt the grammar & procedure for + parse expressions for MCPU-CPP . + + PART OF : MCPU-CPP - MCPU language preproccessor . + + NOTE : NONE . + + Copyright (C) 1998 - 2026 by Andrey V.Kosteltsev. + All Rights Reserved. + ***************************************************************/ + +%{ + +#include <defs.h> + +static int mcpp_zubr_lex( void ); +static void mcpp_zubr_error( char *s ); + +static mcpp_integer expression_value; +static mcpp_semantic_context semantic_context; + +/************************************************************ + Nonzero means do not evaluate this expression. + This is a count, since unevaluated expressions can nest. + ************************************************************/ +static int skip_evaluation; + +/************************************************************ + During parsing of an MCPU-CPP expression, LEXPTR points to + the next UCS-2 character and LEXEND points one character + past the input expression. + ************************************************************/ +static const __mpu_char16_t *lexptr; +static const __mpu_char16_t *lexend; + +%} + +%union +{ + mcpp_integer integer; + struct name + { + const __mpu_char16_t *address; + size_t length; + } name; +} + +%type <integer> exp exp1 start +%token <integer> INT CHAR +%token <name> NAME +%token <integer> ERROR + +%right '?' ':' +%left ',' +%left OR +%left AND +%left '|' +%left '^' +%left '&' +%left EQUAL NOTEQUAL +%left '<' '>' LEQ GEQ +%left LSH RSH +%left '+' '-' +%left '*' '/' '%' +%right UNARY + +%% + +start: exp1 + { + expression_value = $1; + } + ; + +/* Expressions, including the comma operator. */ +exp1: exp + | exp1 ',' exp + { + $$ = $3; + } + ; + +/* Expressions, not including the comma operator. */ +exp: '-' exp %prec UNARY + { + $$ = mcpp_semantic_neg( $2 ); + } + | '!' exp %prec UNARY + { + $$ = mcpp_semantic_not( $2 ); + } + | '+' exp %prec UNARY + { + $$ = $2; + } + | '~' exp %prec UNARY + { + $$ = mcpp_semantic_compl( $2 ); + } + | '(' exp1 ')' + { + $$ = $2; + } + ; + +/* Binary operators in order of decreasing precedence. */ +exp: exp '*' exp + { + $$ = mcpp_semantic_mul( $1, $3 ); + } + | exp '/' exp + { + $$ = mcpp_semantic_div( &semantic_context, $1, $3, + !skip_evaluation ); + } + | exp '%' exp + { + $$ = mcpp_semantic_mod( &semantic_context, $1, $3, + !skip_evaluation ); + } + | exp '+' exp + { + $$ = mcpp_semantic_add( $1, $3 ); + } + | exp '-' exp + { + $$ = mcpp_semantic_sub( $1, $3 ); + } + | exp LSH exp + { + $$ = mcpp_semantic_lshift( $1, $3 ); + } + | exp RSH exp + { + $$ = mcpp_semantic_rshift( $1, $3 ); + } + | exp EQUAL exp + { + $$ = mcpp_semantic_equal( $1, $3 ); + } + | exp NOTEQUAL exp + { + $$ = mcpp_semantic_notequal( $1, $3 ); + } + | exp LEQ exp + { + $$ = mcpp_semantic_leq( $1, $3 ); + } + | exp GEQ exp + { + $$ = mcpp_semantic_geq( $1, $3 ); + } + | exp '<' exp + { + $$ = mcpp_semantic_less( $1, $3 ); + } + | exp '>' exp + { + $$ = mcpp_semantic_greater( $1, $3 ); + } + | exp '&' exp + { + $$ = mcpp_semantic_bitand( $1, $3 ); + } + | exp '^' exp + { + $$ = mcpp_semantic_bitxor( $1, $3 ); + } + | exp '|' exp + { + $$ = mcpp_semantic_bitor( $1, $3 ); + } + | exp AND + { + skip_evaluation += !mcpp_semantic_true( $1 ); + } + exp + { + skip_evaluation -= !mcpp_semantic_true( $1 ); + $$ = mcpp_semantic_and( $1, $4 ); + } + | exp OR + { + skip_evaluation += !!mcpp_semantic_true( $1 ); + } + exp + { + skip_evaluation -= !!mcpp_semantic_true( $1 ); + $$ = mcpp_semantic_or( $1, $4 ); + } + | exp '?' + { + skip_evaluation += !mcpp_semantic_true( $1 ); + } + exp ':' + { + skip_evaluation += !!mcpp_semantic_true( $1 ) - + !mcpp_semantic_true( $1 ); + } + exp + { + skip_evaluation -= !!mcpp_semantic_true( $1 ); + $$ = mcpp_semantic_conditional( $1, $4, $7 ); + } + | INT + { + $$ = $1; + } + | CHAR + { + $$ = $1; + } + | NAME + { + $$ = mcpp_semantic_make( 0, 0 ); + } + ; + +/******* END OF GRAMMAR *******/ +%% + +struct token +{ + const char *operator; + int token; +}; + +static struct token tokentab2[] = +{ + { "&&", AND }, + { "||", OR }, + { "<<", LSH }, + { ">>", RSH }, + { "==", EQUAL }, + { "!=", NOTEQUAL }, + { "<=", LEQ }, + { ">=", GEQ }, + { "++", ERROR }, + { "--", ERROR }, + { NULL, ERROR } +}; + +static int +mcpp_expr_space( __mpu_char16_t c ) +{ + return( c == ' ' || c == '\t' || c == '\r' || c == '\n' || + c == '\f' || c == '\v' ); +} + +static int +mcpp_expr_hex_digit( __mpu_char16_t c ) +{ + if( c >= '0' && c <= '9' ) return( c - '0' ); + if( c >= 'a' && c <= 'f' ) return( c - 'a' + 10 ); + if( c >= 'A' && c <= 'F' ) return( c - 'A' + 10 ); + return( -1 ); +} + +static int +mcpp_expr_error_token( const char *message ) +{ + mcpp_zubr_error( (char *)message ); + return( ERROR ); +} + +static __mpu_uint16_t +mcpp_expr_escape( int *failed ) +{ + __mpu_char16_t c; + __mpu_uint32_t value; + int digit; + unsigned count; + + *failed = 0; + if( lexptr >= lexend ) + { + *failed = 1; + mcpp_zubr_error( "incomplete escape sequence in #if expression" ); + return( 0 ); + } + + c = *lexptr++; + switch( c ) + { + case 'a': return( (__mpu_uint16_t)'\a' ); + case 'b': return( (__mpu_uint16_t)'\b' ); + case 'e': + case 'E': return( (__mpu_uint16_t)033 ); + case 'f': return( (__mpu_uint16_t)'\f' ); + case 'n': return( (__mpu_uint16_t)'\n' ); + case 'r': return( (__mpu_uint16_t)'\r' ); + case 't': return( (__mpu_uint16_t)'\t' ); + case 'v': return( (__mpu_uint16_t)'\v' ); + case '\\': return( (__mpu_uint16_t)'\\' ); + case '\'': return( (__mpu_uint16_t)'\'' ); + case '"': return( (__mpu_uint16_t)'"' ); + case '?': return( (__mpu_uint16_t)'?' ); + + case 'x': + case 'X': + value = 0; + count = 0; + while( lexptr < lexend ) + { + digit = mcpp_expr_hex_digit( *lexptr ); + if( digit < 0 ) break; + if( value > 0xffffU >> 4 ) + { + *failed = 1; + mcpp_zubr_error( "hex escape sequence out of UCS-2 range" ); + return( 0 ); + } + value = (value << 4) + (unsigned)digit; + ++lexptr; + ++count; + } + if( count == 0 ) + { + *failed = 1; + mcpp_zubr_error( "\\x used with no following hex digits" ); + return( 0 ); + } + return( (__mpu_uint16_t)value ); + + default: + if( c >= '0' && c <= '7' ) + { + value = c - '0'; + for( count = 1; count < 6 && lexptr < lexend; ++count ) + { + c = *lexptr; + if( c < '0' || c > '7' ) break; + if( value > 0xffffU >> 3 ) + { + *failed = 1; + mcpp_zubr_error( "octal escape sequence out of UCS-2 range" ); + return( 0 ); + } + value = (value << 3) + (c - '0'); + ++lexptr; + } + if( value > 0xffffU ) + { + *failed = 1; + mcpp_zubr_error( "octal escape sequence out of UCS-2 range" ); + return( 0 ); + } + return( (__mpu_uint16_t)value ); + } + return( (__mpu_uint16_t)c ); + } +} + +static int +mcpp_expr_character_constant( void ) +{ + __mpu_uint64_t result = 0; + __mpu_uint16_t value; + __mpu_char16_t c; + unsigned count = 0; + int failed; + + ++lexptr; + + while( lexptr < lexend ) + { + c = *lexptr++; + if( c == '\'' ) + break; + + if( c == '\n' ) + return( mcpp_expr_error_token( + "unterminated character constant in #if expression") ); + + if( c == '\\' ) + { + value = mcpp_expr_escape( &failed ); + if( failed ) return( ERROR ); + } + else + value = (__mpu_uint16_t)c; + + ++count; + if( count > 4 ) + return( mcpp_expr_error_token( + "character constant too long in #if expression") ); + + result = (result << 16) | (__mpu_uint64_t)value; + } + + if( lexptr == lexend && (lexptr == 0 || lexptr[-1] != '\'') ) + return( mcpp_expr_error_token( + "unterminated character constant in #if expression") ); + + if( count == 0 ) + return( mcpp_expr_error_token( + "empty character constant in #if expression") ); + + mcpp_zubr_lval.integer = mcpp_semantic_make( result, 1 ); + return( CHAR ); +} + +static int +mcpp_expr_valid_integer_width( unsigned width ) +{ + return( width >= 8 && width <= MPU_REAL_IO_LIMIT && + (width & (width - 1)) == 0 ); +} + +static __mpu_uint64_t +mcpp_expr_apply_integer_width( __mpu_uint64_t value, unsigned width, + int unsignedp ) +{ + __mpu_uint64_t mask; + __mpu_uint64_t sign; + + if( width >= 64 ) + return( value ); + + mask = (((__mpu_uint64_t)1 << width) - 1); + value &= mask; + + if( unsignedp ) + return( value ); + + sign = ((__mpu_uint64_t)1 << (width - 1)); + if( value & sign ) + value |= ~mask; + + return( value ); +} + +static int +mcpp_expr_number( void ) +{ + const __mpu_char16_t *start = lexptr; + const __mpu_char16_t *end; + const __mpu_char16_t *number_end; + const __mpu_char16_t *p; + __mpu_char8_t *ascii; + __mpu_uint64_t value = 0; + size_t number_length; + unsigned base = 10; + unsigned width = 0; + int digit; + int have_digit = 0; + int unsignedp = 0; + int have_width = 0; + int width_over_64 = 0; + + while( lexptr < lexend && + (mcpu_pp_is_identifier_char(*lexptr) || *lexptr == '.') ) + ++lexptr; + end = lexptr; + + for( p = start; p < end; ++p ) + if( *p == '.' ) + return( mcpp_expr_error_token( + "floating point numbers not allowed in #if expressions") ); + + p = start; + if( end - p >= 2 && p[0] == '0' && + (p[1] == 'x' || p[1] == 'X') ) + { + base = 16; + p += 2; + } + else if( end - p >= 2 && p[0] == '0' && + (p[1] == 'b' || p[1] == 'B') ) + { + base = 2; + p += 2; + } + else if( end - p > 1 && p[0] == '0' ) + base = 8; + + for( ; p < end; ++p ) + { + if( *p > 0x7f ) + break; + digit = mcpp_expr_hex_digit( *p ); + if( digit < 0 || (unsigned)digit >= base ) + break; + have_digit = 1; + } + number_end = p; + + if( !have_digit ) + return( mcpp_expr_error_token( + "invalid integer constant in #if expression") ); + + if( p < end && (*p == 'z' || *p == 'Z') ) + { + have_width = 1; + ++p; + if( p == end || *p < '0' || *p > '9' ) + return( mcpp_expr_error_token( + "integer width suffix requires decimal digits after z/Z") ); + + while( p < end && *p >= '0' && *p <= '9' ) + { + unsigned d = (unsigned)(*p - '0'); + + if( !width_over_64 ) + { + if( width > (64U - d) / 10U ) + width_over_64 = 1; + else + { + width = width * 10U + d; + if( width > 64U ) + width_over_64 = 1; + } + } + ++p; + } + + if( p < end && (*p == 'u' || *p == 'U') ) + { + unsignedp = 1; + ++p; + } + + if( p != end ) + return( mcpp_expr_error_token( + "invalid characters after integer width suffix") ); + + if( width_over_64 ) + return( mcpp_expr_error_token( + "integer constants wider than 64 bits are not allowed in conditional directives") ); + + if( !mcpp_expr_valid_integer_width(width) ) + { + mcpp_semantic_warning( + &semantic_context, + "invalid zNNN integer-width suffix; suffix ignored" ); + have_width = 0; + } + } + else if( p < end && (*p == 'u' || *p == 'U') ) + { + unsignedp = 1; + ++p; + if( p != end ) + return( mcpp_expr_error_token( + "invalid characters after integer suffix") ); + } + else if( p != end ) + return( mcpp_expr_error_token( + "invalid integer suffix in #if expression") ); + + number_length = (size_t)(number_end - start); + ascii = (__mpu_char8_t *)malloc( number_length + 1 ); + if( ascii == NULL ) + { + mcpp_zubr_error( "out of memory while parsing #if expression" ); + return( ERROR ); + } + + for( p = start; p < number_end; ++p ) + { + if( *p > 0x7f ) + { + free( ascii ); + return( mcpp_expr_error_token( + "non-ASCII character in integer constant") ); + } + ascii[p - start] = (__mpu_char8_t)*p; + } + ascii[number_length] = 0; + + __mpu_clo(); + iatoui( (mpu_int *)&value, ascii, (int)sizeof(value) ); + free( ascii ); + + if( __mpu_gto() ) + return( mcpp_expr_error_token( + "integer constant does not fit in 64 bits") ); + + if( have_width ) + { + value = mcpp_expr_apply_integer_width( value, width, unsignedp ); + mcpp_zubr_lval.integer = mcpp_semantic_make( value, unsignedp ); + } + else + mcpp_zubr_lval.integer = + mcpp_semantic_make( value, + unsignedp || value > (__mpu_uint64_t)INT64_MAX ); + + return( INT ); +} + +static void +mcpp_zubr_error( char *s ) +{ + mcpp_semantic_error( &semantic_context, s ); + skip_evaluation = 0; +} + +/**************************************************** + Read one token, getting UCS-2 characters through + LEXPTR. + ****************************************************/ +static int +mcpp_zubr_lex( void ) +{ + const __mpu_char16_t *tokstart; + struct token *toktab; + __mpu_char16_t c; + +retry: + while( lexptr < lexend && mcpp_expr_space(*lexptr) ) + ++lexptr; + + if( lexptr >= lexend ) + return( 0 ); + + tokstart = lexptr; + c = *tokstart; + + for( toktab = tokentab2; toktab->operator != NULL; ++toktab ) + { + if( lexend - tokstart >= 2 && + c == (__mpu_char16_t)(unsigned char)toktab->operator[0] && + tokstart[1] == (__mpu_char16_t)(unsigned char)toktab->operator[1] ) + { + lexptr += 2; + if( toktab->token == ERROR ) + return( mcpp_expr_error_token( + "increment/decrement operator not allowed in #if expression") ); + return( toktab->token ); + } + } + + if( c >= '0' && c <= '9' ) + return( mcpp_expr_number() ); + + if( c == '\'' ) + return( mcpp_expr_character_constant() ); + + if( mcpu_pp_is_identifier_start(c) ) + { + ++lexptr; + while( lexptr < lexend && mcpu_pp_is_identifier_char(*lexptr) ) + ++lexptr; + mcpp_zubr_lval.name.address = tokstart; + mcpp_zubr_lval.name.length = (size_t)(lexptr - tokstart); + return( NAME ); + } + + ++lexptr; + switch( c ) + { + case '(': + case ')': + case '?': + case ':': + case ',': + case '*': + case '/': + case '%': + case '~': + case '^': + return( (int)c ); + + case '+': + case '-': + return( (int)c ); + + case '!': + case '<': + case '>': + case '&': + case '|': + return( (int)c ); + + case '"': + case '`': + return( mcpp_expr_error_token( + "string constants not allowed in #if expressions") ); + + default: + mcpp_zubr_error( "invalid token in #if expression" ); + goto retry; + } +} + +int +mcpp_expr_parse( const __mpu_char16_t *text, size_t length, + const char *filename, unsigned line_number, + const mcpp_options *options, int *result ) +{ + int rc; + + if( text == NULL || filename == NULL || result == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + mcpp_semantic_context_init( &semantic_context, filename, line_number, + options ); + expression_value = mcpp_semantic_make( 0, 0 ); + skip_evaluation = 0; + lexptr = text; + lexend = text + length; + + if( length == 0 ) + { + mcpp_zubr_error( "empty #if expression" ); + return( -1 ); + } + + rc = mcpp_zubr_parse(); + if( rc != 0 || mcpp_semantic_failed(&semantic_context) ) + return( -1 ); + + *result = mcpp_semantic_true( expression_value ); + return( 0 ); +} diff --git a/src/mcpp-expression.c b/src/mcpp-expression.c new file mode 100644 index 0000000..1f9532d --- /dev/null +++ b/src/mcpp-expression.c @@ -0,0 +1,178 @@ +#include <defs.h> +#include <mcpp-expr.h> + +/* + * The front end keeps the already established mcpu-cpp preprocessing order: + * handle the special defined operator first, expand macros second, then pass + * the resulting UCS-2 expression to the ZUBR parser generated from + * mcpp-expr.zubr. + */ + +static int +expr_identifier_start( __mpu_char16_t c ) +{ + return( mcpu_pp_is_identifier_start(c) ); +} + +static int +expr_identifier_char( __mpu_char16_t c ) +{ + return( mcpu_pp_is_identifier_char(c) ); +} + +static int +expr_space( __mpu_char16_t c ) +{ + return( c == ' ' || c == '\t' || c == '\r' || c == '\n' || + c == '\f' || c == '\v' ); +} + +static int +append_defined_value( mcpu_macro_table *macros, + const __mpu_char16_t *text, size_t length, + size_t *pos, mcpu_text *output, + const char *filename, unsigned line_number ) +{ + size_t p = *pos; + size_t start; + size_t end; + int parenthesized = 0; + + while( p < length && expr_space(text[p]) ) ++p; + if( p < length && text[p] == '(' ) + { + parenthesized = 1; + ++p; + while( p < length && expr_space(text[p]) ) ++p; + } + + if( p >= length || !expr_identifier_start(text[p]) ) + { + fprintf( stderr, "%s:%u: error: identifier expected after defined\n", + filename, line_number ); + return( -1 ); + } + + start = p++; + while( p < length && expr_identifier_char(text[p]) ) ++p; + end = p; + + while( p < length && expr_space(text[p]) ) ++p; + if( parenthesized ) + { + if( p >= length || text[p] != ')' ) + { + fprintf( stderr, "%s:%u: error: missing ')' after defined\n", + filename, line_number ); + return( -1 ); + } + ++p; + } + + if( mcpu_text_append_ascii( + output, mcpu_macro_find(macros, text + start, end - start) ? "1" : "0") != 0 ) + return( -1 ); + + *pos = p; + return( 0 ); +} + +static int +replace_defined_operators( mcpu_macro_table *macros, + const __mpu_char16_t *text, size_t length, + mcpu_text *output, + const char *filename, unsigned line_number ) +{ + size_t p = 0; + __mpu_char16_t quote = 0; + int escaped = 0; + + mcpu_text_free( output ); + mcpu_text_init( output ); + + while( p < length ) + { + __mpu_char16_t c = text[p]; + + if( quote ) + { + if( mcpu_text_append_char(output, c) != 0 ) return( -1 ); + if( escaped ) + escaped = 0; + else if( c == '\\' ) + escaped = 1; + else if( c == quote ) + quote = 0; + ++p; + continue; + } + + if( c == '\'' || c == '"' ) + { + quote = c; + if( mcpu_text_append_char(output, c) != 0 ) return( -1 ); + ++p; + continue; + } + + if( expr_identifier_start(c) ) + { + size_t start = p++; + while( p < length && expr_identifier_char(text[p]) ) ++p; + + if( p - start == 7 && + mcpu_text_equal_ascii(text + start, p - start, "defined") ) + { + if( append_defined_value(macros, text, length, &p, output, + filename, line_number) != 0 ) + return( -1 ); + } + else if( mcpu_text_append(output, text + start, p - start) != 0 ) + return( -1 ); + continue; + } + + if( mcpu_text_append_char(output, c) != 0 ) + return( -1 ); + ++p; + } + + return( 0 ); +} + +int +mcpp_eval_if_expression( mcpu_macro_table *macros, + const __mpu_char16_t *text, size_t length, + enum mcpu_language language, + const mcpu_macro_expansion_context *context, + const char *filename, unsigned line_number, + int *result ) +{ + mcpu_text defined; + mcpu_text expanded; + int rc = -1; + + if( macros == NULL || text == NULL || context == NULL || + filename == NULL || result == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + mcpu_text_init( &defined ); + mcpu_text_init( &expanded ); + + if( replace_defined_operators(macros, text, length, &defined, + filename, line_number) != 0 || + mcpu_macro_expand(macros, defined.data, defined.length, + language, context, &expanded) != 0 ) + goto done; + + rc = mcpp_expr_parse( expanded.data, expanded.length, + filename, line_number, context->options, result ); + +done: + mcpu_text_free( &defined ); + mcpu_text_free( &expanded ); + return( rc ); +} diff --git a/src/mcpp-expression.h b/src/mcpp-expression.h new file mode 100644 index 0000000..3fc1c96 --- /dev/null +++ b/src/mcpp-expression.h @@ -0,0 +1,15 @@ +#ifndef __MCPU_CPP_EXPRESSION_H__ +#define __MCPU_CPP_EXPRESSION_H__ 1 + +#include <defs.h> + +int mcpp_eval_if_expression( mcpu_macro_table *macros, + const __mpu_char16_t *text, + size_t length, + enum mcpu_language language, + const mcpu_macro_expansion_context *context, + const char *filename, + unsigned line_number, + int *result ); + +#endif /* __MCPU_CPP_EXPRESSION_H__ */ diff --git a/src/mcpp-include-path.c b/src/mcpp-include-path.c new file mode 100644 index 0000000..1b26137 --- /dev/null +++ b/src/mcpp-include-path.c @@ -0,0 +1,467 @@ +#include <defs.h> + +static char * +join_path( const char *dir, const char *name ) +{ + size_t dl; + size_t nl; + char *p; + int slash; + + if( dir == NULL || name == NULL ) + { + errno = EINVAL; + return( NULL ); + } + + dl = strlen( dir ); + nl = strlen( name ); + slash = dl != 0 && dir[dl - 1] != '/'; + + p = (char *)malloc( dl + (size_t)slash + nl + 1 ); + if( p == NULL ) + return( NULL ); + + memcpy( p, dir, dl ); + if( slash ) + p[dl++] = '/'; + memcpy( p + dl, name, nl + 1 ); + + return( p ); +} + +static int +file_exists( const char *path ) +{ + FILE *fp = fopen( path, "rb" ); + + if( fp == NULL ) + return( 0 ); + + fclose( fp ); + return( 1 ); +} + +static char * +source_directory( const char *filename ) +{ + const char *slash; + size_t n; + char *dir; + + if( filename == NULL ) + return( NULL ); + + slash = strrchr( filename, '/' ); + if( slash == NULL ) + return( strdup(".") ); + + if( slash == filename ) + return( strdup("/") ); + + n = (size_t)(slash - filename); + dir = (char *)malloc( n + 1 ); + if( dir == NULL ) + return( NULL ); + + memcpy( dir, filename, n ); + dir[n] = 0; + + return( dir ); +} + +void +mcpu_include_paths_init( mcpu_include_paths *paths ) +{ + if( paths == NULL ) + return; + + paths->first = NULL; + paths->last = NULL; + paths->no_standard_includes = 0; +} + +void +mcpu_include_paths_free( mcpu_include_paths *paths ) +{ + mcpu_include_dir *p; + + if( paths == NULL ) + return; + + p = paths->first; + while( p ) + { + mcpu_include_dir *next = p->next; + free( p->path ); + free( p ); + p = next; + } + + paths->first = NULL; + paths->last = NULL; +} + +static int +add_dir( mcpu_include_paths *paths, const char *path, + enum mcpu_include_class class_type, + int language_specific, enum mcpu_language language ) +{ + mcpu_include_dir *dir; + + if( paths == NULL || path == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + if( *path == 0 ) + return( 0 ); + + dir = (mcpu_include_dir *)calloc( 1, sizeof(*dir) ); + if( dir == NULL ) + return( -1 ); + + dir->path = strdup( path ); + if( dir->path == NULL ) + { + free( dir ); + return( -1 ); + } + + dir->class_type = class_type; + dir->language = language; + dir->language_specific = language_specific; + + if( paths->last ) + paths->last->next = dir; + else + paths->first = dir; + paths->last = dir; + + return( 0 ); +} + +int +mcpu_include_paths_add( mcpu_include_paths *paths, const char *path, + enum mcpu_include_class class_type ) +{ + return( add_dir( paths, path, class_type, 0, MCPU_LANG_0 ) ); +} + +int +mcpu_include_paths_add_language( mcpu_include_paths *paths, + const char *path, + enum mcpu_include_class class_type, + enum mcpu_language language ) +{ + return( add_dir( paths, path, class_type, 1, language ) ); +} + +int +mcpu_include_paths_add_list( mcpu_include_paths *paths, + const char *list, + enum mcpu_include_class class_type ) +{ + const char *p; + const char *start; + + if( paths == NULL || list == NULL ) + return( 0 ); + + p = start = list; + for( ;; ) + { + if( *p == ':' || *p == 0 ) + { + size_t n = (size_t)(p - start); + if( n != 0 ) + { + char *dir = (char *)malloc( n + 1 ); + int rc; + + if( dir == NULL ) + return( -1 ); + memcpy( dir, start, n ); + dir[n] = 0; + rc = mcpu_include_paths_add( paths, dir, class_type ); + free( dir ); + if( rc != 0 ) + return( -1 ); + } + + if( *p == 0 ) + break; + start = p + 1; + } + ++p; + } + + return( 0 ); +} + +static int +add_language_list( mcpu_include_paths *paths, const char *list, + enum mcpu_include_class class_type, + enum mcpu_language language ) +{ + const char *p; + const char *start; + + if( list == NULL ) + return( 0 ); + + p = start = list; + for( ;; ) + { + if( *p == ':' || *p == 0 ) + { + size_t n = (size_t)(p - start); + if( n != 0 ) + { + char *dir = (char *)malloc( n + 1 ); + int rc; + + if( dir == NULL ) + return( -1 ); + memcpy( dir, start, n ); + dir[n] = 0; + rc = mcpu_include_paths_add_language( paths, dir, + class_type, language ); + free( dir ); + if( rc != 0 ) + return( -1 ); + } + if( *p == 0 ) + break; + start = p + 1; + } + ++p; + } + + return( 0 ); +} + +static int +add_system_root( mcpu_include_paths *paths, const char *root ) +{ + enum mcpu_language language; + + if( root == NULL || *root == 0 ) + return( 0 ); + + for( language = MCPU_LANG_DIFF; + language < MCPU_LANG__COUNT; + language = (enum mcpu_language)(language + 1) ) + { + const char *dirname = mcpu_language_directory_name( language ); + char *dir; + + if( dirname == NULL ) + continue; + + dir = join_path( root, dirname ); + if( dir == NULL ) + return( -1 ); + + if( mcpu_include_paths_add_language(paths, dir, + MCPU_INCLUDE_SYSTEM_CONFIG_LANGUAGE, + language) != 0 ) + { + free( dir ); + return( -1 ); + } + free( dir ); + } + + return( mcpu_include_paths_add(paths, root, + MCPU_INCLUDE_SYSTEM_CONFIG) ); +} + + + +int +mcpu_include_paths_dump( const mcpu_include_paths *paths, FILE *stream ) +{ + enum mcpu_include_class class_type; + const mcpu_include_dir *dir; + + if( paths == NULL || stream == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + for( class_type = MCPU_INCLUDE_USER_EXPLICIT; + class_type <= MCPU_INCLUDE_AFTER_CONFIG; + class_type = (enum mcpu_include_class)(class_type + 1) ) + { + if( paths->no_standard_includes && + (class_type == MCPU_INCLUDE_SYSTEM_CONFIG_LANGUAGE || + class_type == MCPU_INCLUDE_SYSTEM_CONFIG) ) + continue; + + for( dir = paths->first; dir; dir = dir->next ) + { + if( dir->class_type != class_type ) + continue; + + if( fprintf(stream, "search: %s\n", dir->path) < 0 ) + return( -1 ); + } + } + + return( ferror(stream) ? -1 : 0 ); +} + + +int +mcpu_include_paths_from_config( mcpu_include_paths *paths, + const mcpu_config *config ) +{ + enum mcpu_language language; + const char *value; + + if( paths == NULL || config == NULL ) + return( 0 ); + + /* + * Language-specific directories precede the common include root. The + * common MCPU_CPP_INCLUDE_PATH itself is language-independent and remains + * visible in every language state. + */ + for( language = MCPU_LANG_DIFF; + language < MCPU_LANG__COUNT; + language = (enum mcpu_language)(language + 1) ) + { + const char *name = mcpu_language_config_variable( language ); + value = mcpu_config_get( config, name ); + if( add_language_list(paths, value, + MCPU_INCLUDE_USER_CONFIG_LANGUAGE, + language) != 0 ) + return( -1 ); + } + + value = mcpu_config_get( config, "MCPU_CPP_INCLUDE_PATH" ); + if( mcpu_include_paths_add_list( paths, value, + MCPU_INCLUDE_USER_CONFIG ) != 0 ) + return( -1 ); + + value = mcpu_config_get( config, "MCPU_CPP_SYSTEM_INCLUDE_PATH" ); + if( add_system_root(paths, value) != 0 ) + return( -1 ); + + value = mcpu_config_get( config, "MCPU_CPP_AFTER_INCLUDE_PATH" ); + if( mcpu_include_paths_add_list( paths, value, + MCPU_INCLUDE_AFTER_CONFIG ) != 0 ) + return( -1 ); + + return( 0 ); +} + +char * +mcpu_include_find( const mcpu_include_paths *paths, + const char *source_filename, + const char *include_filename, + int quoted, + enum mcpu_language language, + int include_next, + const mcpu_include_dir *after_dir, + const mcpu_include_dir **found_dir ) +{ + mcpu_include_dir *dir; + int past_after = after_dir == NULL; + + if( found_dir != NULL ) + *found_dir = NULL; + + if( include_filename == NULL ) + return( NULL ); + + if( include_filename[0] == '/' ) + { + if( file_exists(include_filename) ) + return( strdup(include_filename) ); + return( NULL ); + } + + /* + * #include_next never searches the directory of the current source file. + * It continues in the configured include chain after the directory entry + * that supplied the containing header. If the containing file came from + * its source directory (or has no search-chain origin), the configured + * chain is searched from its beginning. + */ + if( !include_next && quoted && source_filename ) + { + char *base = source_directory( source_filename ); + char *path; + + if( base == NULL ) + return( NULL ); + path = join_path( base, include_filename ); + free( base ); + if( path && file_exists(path) ) + return( path ); + free( path ); + } + + if( paths == NULL ) + return( NULL ); + + { + static const enum mcpu_include_class order[] = + { + MCPU_INCLUDE_USER_EXPLICIT, + MCPU_INCLUDE_SYSTEM_EXPLICIT, + MCPU_INCLUDE_USER_CONFIG_LANGUAGE, + MCPU_INCLUDE_USER_CONFIG, + MCPU_INCLUDE_SYSTEM_CONFIG_LANGUAGE, + MCPU_INCLUDE_SYSTEM_CONFIG, + MCPU_INCLUDE_AFTER_EXPLICIT, + MCPU_INCLUDE_AFTER_CONFIG + }; + size_t pass; + + for( pass = 0; pass < sizeof(order) / sizeof(order[0]); ++pass ) + { + for( dir = paths->first; dir; dir = dir->next ) + { + char *path; + + if( dir->class_type != order[pass] ) + continue; + + if( !past_after ) + { + if( dir == after_dir ) + past_after = 1; + continue; + } + + if( dir->language_specific && dir->language != language ) + continue; + + if( paths->no_standard_includes && + (dir->class_type == MCPU_INCLUDE_SYSTEM_CONFIG_LANGUAGE || + dir->class_type == MCPU_INCLUDE_SYSTEM_CONFIG) ) + continue; + + path = join_path( dir->path, include_filename ); + if( path == NULL ) + return( NULL ); + + if( file_exists(path) ) + { + if( found_dir != NULL ) + *found_dir = dir; + return( path ); + } + + free( path ); + } + } + } + + return( NULL ); +} diff --git a/src/mcpp-include-path.h b/src/mcpp-include-path.h new file mode 100644 index 0000000..98001f8 --- /dev/null +++ b/src/mcpp-include-path.h @@ -0,0 +1,61 @@ +#ifndef __MCPU_CPP_INCLUDE_PATH_H__ +#define __MCPU_CPP_INCLUDE_PATH_H__ 1 + +#include <defs.h> +#include <mcpp-config-file.h> +#include <mcpp-language.h> + +enum mcpu_include_class +{ + MCPU_INCLUDE_USER_EXPLICIT = 0, + MCPU_INCLUDE_SYSTEM_EXPLICIT, + MCPU_INCLUDE_USER_CONFIG_LANGUAGE, + MCPU_INCLUDE_USER_CONFIG, + MCPU_INCLUDE_SYSTEM_CONFIG_LANGUAGE, + MCPU_INCLUDE_SYSTEM_CONFIG, + MCPU_INCLUDE_AFTER_EXPLICIT, + MCPU_INCLUDE_AFTER_CONFIG +}; + +typedef struct mcpu_include_dir mcpu_include_dir; +struct mcpu_include_dir +{ + char *path; + enum mcpu_include_class class_type; + enum mcpu_language language; + int language_specific; + mcpu_include_dir *next; +}; + +typedef struct mcpu_include_paths mcpu_include_paths; +struct mcpu_include_paths +{ + mcpu_include_dir *first; + mcpu_include_dir *last; + int no_standard_includes; +}; + +void mcpu_include_paths_init( mcpu_include_paths *paths ); +void mcpu_include_paths_free( mcpu_include_paths *paths ); +int mcpu_include_paths_add( mcpu_include_paths *paths, const char *path, + enum mcpu_include_class class_type ); +int mcpu_include_paths_add_language( mcpu_include_paths *paths, + const char *path, + enum mcpu_include_class class_type, + enum mcpu_language language ); +int mcpu_include_paths_add_list( mcpu_include_paths *paths, + const char *list, + enum mcpu_include_class class_type ); +int mcpu_include_paths_from_config( mcpu_include_paths *paths, + const mcpu_config *config ); +int mcpu_include_paths_dump( const mcpu_include_paths *paths, FILE *stream ); +char *mcpu_include_find( const mcpu_include_paths *paths, + const char *source_filename, + const char *include_filename, + int quoted, + enum mcpu_language language, + int include_next, + const mcpu_include_dir *after_dir, + const mcpu_include_dir **found_dir ); + +#endif /* __MCPU_CPP_INCLUDE_PATH_H__ */ diff --git a/src/mcpp-language.c b/src/mcpp-language.c new file mode 100644 index 0000000..d224bc8 --- /dev/null +++ b/src/mcpp-language.c @@ -0,0 +1,108 @@ +#include <defs.h> + +static int +text_equal_ascii_nocase( const __mpu_char16_t *text, size_t length, + const char *ascii ) +{ + size_t i; + size_t ascii_length; + + if( text == NULL || ascii == NULL ) + return( 0 ); + + ascii_length = strlen( ascii ); + if( length != ascii_length ) + return( 0 ); + + for( i = 0; i < length; ++i ) + { + __mpu_char16_t c = text[i]; + unsigned char a = (unsigned char)ascii[i]; + + if( c >= 'A' && c <= 'Z' ) + c = (__mpu_char16_t)(c - 'A' + 'a'); + if( a >= 'A' && a <= 'Z' ) + a = (unsigned char)(a - 'A' + 'a'); + + if( c != (__mpu_char16_t)a ) + return( 0 ); + } + + return( 1 ); +} + +const char * +mcpu_language_name( enum mcpu_language language ) +{ + switch( language ) + { + case MCPU_LANG_0: return( "0" ); + case MCPU_LANG_DIFF: return( "diff" ); + case MCPU_LANG_DIFT: return( "dift" ); + case MCPU_LANG_ALG: return( "alg" ); + case MCPU_LANG_AS: return( "as" ); + case MCPU_LANG_AVM: return( "avm" ); + case MCPU_LANG_ACS: return( "ACS" ); + default: return( "unknown" ); + } +} + +int +mcpu_language_from_text( const __mpu_char16_t *text, size_t length, + enum mcpu_language *language ) +{ + enum mcpu_language candidate; + + if( text == NULL || language == NULL ) + return( -1 ); + + /* + * MCPU_LANG_0 is the initial base language state. It is deliberately + * not a valid #lang argument. Every accepted name comes directly from + * the internal switched-language table and is matched case-insensitively. + */ + for( candidate = MCPU_LANG_DIFF; + candidate < MCPU_LANG__COUNT; + candidate = (enum mcpu_language)(candidate + 1) ) + { + if( text_equal_ascii_nocase(text, length, + mcpu_language_name(candidate)) ) + { + *language = candidate; + return( 0 ); + } + } + + return( -1 ); +} + +const char * +mcpu_language_config_variable( enum mcpu_language language ) +{ + switch( language ) + { + case MCPU_LANG_DIFF: return( "MCPU_CPP_DIFF_INCLUDE_PATH" ); + case MCPU_LANG_DIFT: return( "MCPU_CPP_DIFT_INCLUDE_PATH" ); + case MCPU_LANG_ALG: return( "MCPU_CPP_ALG_INCLUDE_PATH" ); + case MCPU_LANG_AS: return( "MCPU_CPP_AS_INCLUDE_PATH" ); + case MCPU_LANG_AVM: return( "MCPU_CPP_AVM_INCLUDE_PATH" ); + case MCPU_LANG_ACS: return( "MCPU_CPP_ACS_INCLUDE_PATH" ); + default: return( NULL ); + } +} + + +const char * +mcpu_language_directory_name( enum mcpu_language language ) +{ + switch( language ) + { + case MCPU_LANG_DIFF: return( "diff" ); + case MCPU_LANG_DIFT: return( "dift" ); + case MCPU_LANG_ALG: return( "alg" ); + case MCPU_LANG_AS: return( "as" ); + case MCPU_LANG_AVM: return( "avm" ); + case MCPU_LANG_ACS: return( "acs" ); + default: return( NULL ); + } +} diff --git a/src/mcpp-language.h b/src/mcpp-language.h new file mode 100644 index 0000000..a758e37 --- /dev/null +++ b/src/mcpp-language.h @@ -0,0 +1,24 @@ +#ifndef __MCPU_CPP_LANGUAGE_H__ +#define __MCPU_CPP_LANGUAGE_H__ 1 + +#include <defs.h> + +enum mcpu_language +{ + MCPU_LANG_0 = 0, + MCPU_LANG_DIFF, + MCPU_LANG_DIFT, + MCPU_LANG_ALG, + MCPU_LANG_AS, + MCPU_LANG_AVM, + MCPU_LANG_ACS, + MCPU_LANG__COUNT +}; + +const char *mcpu_language_name( enum mcpu_language language ); +int mcpu_language_from_text( const __mpu_char16_t *text, size_t length, + enum mcpu_language *language ); +const char *mcpu_language_config_variable( enum mcpu_language language ); +const char *mcpu_language_directory_name( enum mcpu_language language ); + +#endif /* __MCPU_CPP_LANGUAGE_H__ */ diff --git a/src/mcpp-lexer.c b/src/mcpp-lexer.c new file mode 100644 index 0000000..067e1a2 --- /dev/null +++ b/src/mcpp-lexer.c @@ -0,0 +1,371 @@ +#include <defs.h> + +void +mcpu_pp_lexer_init( mcpu_pp_lexer_state *state ) +{ + if( state ) + { + state->in_block_comment = 0; + state->options = NULL; + state->filename = NULL; + state->line_number = 0; + state->splice_offsets = NULL; + state->splice_count = 0; + } +} + +void +mcpu_pp_lexer_set_diagnostics( mcpu_pp_lexer_state *state, + const mcpp_options *options, + const char *filename, + unsigned line_number, + const size_t *splice_offsets, + size_t splice_count ) +{ + if( state == NULL ) + return; + + state->options = options; + state->filename = filename; + state->line_number = line_number; + state->splice_offsets = splice_offsets; + state->splice_count = splice_count; +} + +static unsigned +lexer_warning_line( const mcpu_pp_lexer_state *state, size_t offset ) +{ + size_t i; + unsigned line; + + line = state->line_number; + for( i = 0; i < state->splice_count; ++i ) + if( state->splice_offsets[i] <= offset ) + ++line; + + return( line ); +} + +static int +lexer_has_splice_after( const mcpu_pp_lexer_state *state, size_t offset ) +{ + size_t i; + + for( i = 0; i < state->splice_count; ++i ) + if( state->splice_offsets[i] >= offset ) + return( 1 ); + + return( 0 ); +} + +static int +lexer_comment_warning( const mcpu_pp_lexer_state *state, size_t offset, + const char *message ) +{ + if( state->options == NULL || !state->options->warn_comments ) + return( 0 ); + + return( mcpp_diagnostic_warning(state->options, + state->filename, + lexer_warning_line(state, offset), + "%s", message) ); +} + +int +mcpu_pp_is_identifier_start( __mpu_char16_t c ) +{ + return( c == '_' || mpu_ucs2_is_xid_start(c) ); +} + +int +mcpu_pp_is_identifier_char( __mpu_char16_t c ) +{ + return( c == '_' || c == '$' || mpu_ucs2_is_xid_continue(c) ); +} + +static int +is_quote16( __mpu_char16_t c, enum mcpu_language language ) +{ + if( c == '"' || c == '`' ) + return( 1 ); + + if( c == '\'' && language != MCPU_LANG_DIFF ) + return( 1 ); + + return( 0 ); +} + +static int +append_space_once( mcpu_text *text ) +{ + if( text->length != 0 ) + { + __mpu_char16_t last = text->data[text->length - 1]; + if( last == ' ' || last == '\t' || last == '\f' || last == '\v' || + last == '\r' || last == '\n' ) + return( 0 ); + } + + return( mcpu_text_append_char(text, ' ') ); +} + +static int +is_blank16( __mpu_char16_t c ) +{ + return( c == ' ' || c == '\t' || c == '\f' || + c == '\v' || c == '\r' ); +} + +static void +remove_comment_tail_space( mcpu_text *text ) +{ + size_t end; + int newline; + + if( text == NULL || text->length == 0 ) + return; + + newline = text->data[text->length - 1] == '\n'; + end = newline ? text->length - 1 : text->length; + + while( end != 0 && is_blank16(text->data[end - 1]) ) + --end; + + if( newline ) + text->data[end++] = '\n'; + + text->length = end; + text->data[text->length] = 0; +} + +int +mcpu_pp_prepare_line( mcpu_pp_lexer_state *state, + const __mpu_char16_t *line, + size_t length, + enum mcpu_language language, + mcpu_text *prepared ) +{ + size_t p = 0; + __mpu_char16_t quote = 0; + int escaped = 0; + int first_token = 1; + int saw_hash = 0; + int reading_directive = 0; + int directive_done = 0; + int include_directive = 0; + int include_argument_pending = 0; + int in_include_angle = 0; + __mpu_char16_t directive[32]; + size_t directive_length = 0; + int comment_tail; + + if( state == NULL || line == NULL || prepared == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + mcpu_text_free( prepared ); + mcpu_text_init( prepared ); + comment_tail = state->in_block_comment; + + while( p < length ) + { + __mpu_char16_t c = line[p]; + + if( state->in_block_comment ) + { + if( p + 1 < length && c == '/' && line[p + 1] == '*' ) + { + if( lexer_comment_warning(state, p, "\"/*\" within comment") != 0 ) + return( -1 ); + } + + if( p + 1 < length && c == '*' && line[p + 1] == '/' ) + { + state->in_block_comment = 0; + p += 2; + } + else + { + if( c == '\n' && mcpu_text_append_char(prepared, c) != 0 ) + return( -1 ); + ++p; + } + continue; + } + + if( in_include_angle ) + { + if( mcpu_text_append_char(prepared, c) != 0 ) + return( -1 ); + if( c == '>' || c == '\n' ) + in_include_angle = 0; + ++p; + continue; + } + + if( quote ) + { + if( mcpu_text_append_char(prepared, c) != 0 ) + return( -1 ); + + if( escaped ) + escaped = 0; + else if( c == '\\' ) + escaped = 1; + else if( c == quote || c == '\n' ) + quote = 0; + + ++p; + continue; + } + + if( p + 1 < length && c == '/' && line[p + 1] == '*' ) + { + if( append_space_once(prepared) != 0 ) + return( -1 ); + comment_tail = 1; + state->in_block_comment = 1; + p += 2; + continue; + } + + if( p + 1 < length && c == '/' && line[p + 1] == '/' ) + { + if( lexer_has_splice_after(state, p + 2) && + lexer_comment_warning(state, p, "multi-line comment") != 0 ) + return( -1 ); + + if( append_space_once(prepared) != 0 ) + return( -1 ); + comment_tail = 1; + while( p < length && line[p] != '\n' ) + ++p; + continue; + } + + if( comment_tail && !is_blank16(c) && c != '\n' ) + comment_tail = 0; + + if( first_token ) + { + if( c == ' ' || c == '\t' || c == '\f' || c == '\v' || c == '\r' ) + { + if( mcpu_text_append_char(prepared, c) != 0 ) + return( -1 ); + ++p; + continue; + } + + first_token = 0; + if( c == '#' ) + { + saw_hash = 1; + reading_directive = 1; + } + } + + if( reading_directive && !directive_done && saw_hash && c != '#' ) + { + if( directive_length == 0 && + (c == ' ' || c == '\t' || c == '\f' || c == '\v' || c == '\r') ) + { + if( mcpu_text_append_char(prepared, c) != 0 ) + return( -1 ); + ++p; + continue; + } + + if( mcpu_pp_is_identifier_char(c) ) + { + if( directive_length + 1 < sizeof(directive) / sizeof(directive[0]) ) + directive[directive_length++] = c; + if( mcpu_text_append_char(prepared, c) != 0 ) + return( -1 ); + ++p; + continue; + } + + directive_done = 1; + reading_directive = 0; + if( (directive_length == 7 && + directive[0] == 'i' && directive[1] == 'n' && + directive[2] == 'c' && directive[3] == 'l' && + directive[4] == 'u' && directive[5] == 'd' && + directive[6] == 'e') || + (directive_length == 12 && + directive[0] == 'i' && directive[1] == 'n' && + directive[2] == 'c' && directive[3] == 'l' && + directive[4] == 'u' && directive[5] == 'd' && + directive[6] == 'e' && directive[7] == '_' && + directive[8] == 'n' && directive[9] == 'e' && + directive[10] == 'x' && directive[11] == 't') ) + { + include_directive = 1; + include_argument_pending = 1; + } + } + + if( include_directive && include_argument_pending ) + { + if( c == ' ' || c == '\t' || c == '\f' || c == '\v' || c == '\r' ) + { + if( mcpu_text_append_char(prepared, c) != 0 ) + return( -1 ); + ++p; + continue; + } + + include_argument_pending = 0; + if( c == '<' ) + in_include_angle = 1; + } + + if( is_quote16(c, language) ) + quote = c; + + if( mcpu_text_append_char(prepared, c) != 0 ) + return( -1 ); + ++p; + } + + if( comment_tail ) + remove_comment_tail_space( prepared ); + + return( 0 ); +} + +int +mcpu_pp_find_directive( const __mpu_char16_t *line, + size_t length, + size_t *hash_offset ) +{ + size_t p = 0; + + if( line == NULL || hash_offset == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + while( p < length ) + { + if( line[p] == ' ' || line[p] == '\t' || line[p] == '\f' || + line[p] == '\v' || line[p] == '\r' ) + { + ++p; + continue; + } + + if( line[p] == '#' ) + { + *hash_offset = p; + return( 1 ); + } + + break; + } + + return( 0 ); +} diff --git a/src/mcpp-lexer.h b/src/mcpp-lexer.h new file mode 100644 index 0000000..cc18799 --- /dev/null +++ b/src/mcpp-lexer.h @@ -0,0 +1,38 @@ +#ifndef __MCPU_CPP_LEXER_H__ +#define __MCPU_CPP_LEXER_H__ 1 + +#include <defs.h> +#include <mcpp-language.h> +#include <mcpp-text.h> + +typedef struct mcpp_options mcpp_options; +typedef struct mcpu_pp_lexer_state mcpu_pp_lexer_state; +struct mcpu_pp_lexer_state +{ + int in_block_comment; + const mcpp_options *options; + const char *filename; + unsigned line_number; + const size_t *splice_offsets; + size_t splice_count; +}; + +void mcpu_pp_lexer_init( mcpu_pp_lexer_state *state ); +void mcpu_pp_lexer_set_diagnostics( mcpu_pp_lexer_state *state, + const mcpp_options *options, + const char *filename, + unsigned line_number, + const size_t *splice_offsets, + size_t splice_count ); +int mcpu_pp_prepare_line( mcpu_pp_lexer_state *state, + const __mpu_char16_t *line, + size_t length, + enum mcpu_language language, + mcpu_text *prepared ); +int mcpu_pp_find_directive( const __mpu_char16_t *line, + size_t length, + size_t *hash_offset ); +int mcpu_pp_is_identifier_start( __mpu_char16_t c ); +int mcpu_pp_is_identifier_char( __mpu_char16_t c ); + +#endif /* __MCPU_CPP_LEXER_H__ */ diff --git a/src/mcpp-lib.c b/src/mcpp-lib.c new file mode 100644 index 0000000..a0879f2 --- /dev/null +++ b/src/mcpp-lib.c @@ -0,0 +1,3443 @@ +#include <defs.h> + +#include <sys/stat.h> + + +struct mcpp_dependency +{ + char *path; + dev_t device; + ino_t inode; + int has_identity; + int unresolved; + int system_only; + mcpp_dependency *next; +}; + + +static int +include_dir_is_system( const mcpu_include_dir *dir ) +{ + if( dir == NULL ) + return( 0 ); + + switch( dir->class_type ) + { + case MCPU_INCLUDE_SYSTEM_EXPLICIT: + case MCPU_INCLUDE_SYSTEM_CONFIG_LANGUAGE: + case MCPU_INCLUDE_SYSTEM_CONFIG: + case MCPU_INCLUDE_AFTER_EXPLICIT: + case MCPU_INCLUDE_AFTER_CONFIG: + return( 1 ); + default: + return( 0 ); + } +} + + +static int +dependency_add_internal( mcpp *cpp, const char *path, int system_header, + int unresolved ) +{ + mcpp_dependency *p; + struct stat st; + int has_identity; + + if( cpp == NULL || path == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + /* + * Resolved dependencies belong to the physical identity domain and are + * deduplicated by st_dev/st_ino whenever possible. -MG dependencies have + * no physical object yet, so they deliberately bypass stat() and occupy a + * separate unresolved identity domain keyed by the exact include operand. + * This prevents an unrelated file in the current directory from turning an + * unresolved <name> into a false physical match. + */ + has_identity = !unresolved && stat(path, &st) == 0; + + for( p = cpp->dependencies_first; p != NULL; p = p->next ) + { + int same = 0; + + if( p->unresolved != unresolved ) + continue; + + if( unresolved ) + same = strcmp(p->path, path) == 0; + else if( has_identity && p->has_identity ) + same = p->device == st.st_dev && p->inode == st.st_ino; + else if( !has_identity && !p->has_identity ) + same = strcmp(p->path, path) == 0; + + if( same ) + { + /* + * A resolved physical file may first be reached through a system path + * and later through a user path; preserve the established registry rule + * that the user reach makes it a user dependency. An unresolved -MG + * entry has no physical provenance, so its first classification remains + * authoritative, matching GNU CPP's missing-header behavior. + */ + if( !unresolved && !system_header ) + p->system_only = 0; + return( 0 ); + } + } + + p = (mcpp_dependency *)calloc( 1, sizeof(*p) ); + if( p == NULL ) + return( -1 ); + + p->path = strdup( path ); + if( p->path == NULL ) + { + free( p ); + return( -1 ); + } + + p->has_identity = has_identity; + p->unresolved = unresolved != 0; + if( has_identity ) + { + p->device = st.st_dev; + p->inode = st.st_ino; + } + p->system_only = system_header != 0; + + if( cpp->dependencies_last != NULL ) + cpp->dependencies_last->next = p; + else + cpp->dependencies_first = p; + cpp->dependencies_last = p; + + return( 0 ); +} + + +static int +dependency_add( mcpp *cpp, const char *path, int system_header ) +{ + return( dependency_add_internal(cpp, path, system_header, 0) ); +} + + +static int +dependency_add_unresolved( mcpp *cpp, const char *path, + int system_header ) +{ + return( dependency_add_internal(cpp, path, system_header, 1) ); +} + + +static void +dependencies_free( mcpp *cpp ) +{ + mcpp_dependency *p; + + if( cpp == NULL ) + return; + + p = cpp->dependencies_first; + while( p != NULL ) + { + mcpp_dependency *next = p->next; + free( p->path ); + free( p ); + p = next; + } + + cpp->dependencies_first = NULL; + cpp->dependencies_last = NULL; +} + +struct mcpp_once_file +{ + dev_t device; + ino_t inode; + mcpp_once_file *next; +}; + + +static int +once_file_identity( const char *filename, dev_t *device, ino_t *inode ) +{ + struct stat st; + + if( filename == NULL || device == NULL || inode == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + if( stat(filename, &st) != 0 ) + return( -1 ); + + *device = st.st_dev; + *inode = st.st_ino; + return( 0 ); +} + + +static int +once_file_seen_identity( const mcpp *cpp, dev_t device, ino_t inode ) +{ + const mcpp_once_file *p; + + if( cpp == NULL ) + return( 0 ); + + for( p = cpp->once_files; p != NULL; p = p->next ) + if( p->device == device && p->inode == inode ) + return( 1 ); + + return( 0 ); +} + + +static int +once_file_seen( const mcpp *cpp, const char *filename ) +{ + dev_t device; + ino_t inode; + + if( once_file_identity(filename, &device, &inode) != 0 ) + return( 0 ); + + return( once_file_seen_identity(cpp, device, inode) ); +} + + +static int +once_file_mark( mcpp *cpp, const char *filename ) +{ + mcpp_once_file *entry; + dev_t device; + ino_t inode; + + if( cpp == NULL || filename == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + /* + * A stream such as <stdin> has no stable filesystem identity. In that + * case #pragma once is consumed but there is nothing that can be entered + * in the physical-file registry. + */ + if( once_file_identity(filename, &device, &inode) != 0 ) + return( 0 ); + + if( once_file_seen_identity(cpp, device, inode) ) + return( 0 ); + + entry = (mcpp_once_file *)calloc( 1, sizeof(*entry) ); + if( entry == NULL ) + return( -1 ); + + entry->device = device; + entry->inode = inode; + entry->next = cpp->once_files; + cpp->once_files = entry; + return( 0 ); +} + + +static void +once_files_free( mcpp *cpp ) +{ + mcpp_once_file *p; + + if( cpp == NULL ) + return; + + p = cpp->once_files; + while( p != NULL ) + { + mcpp_once_file *next = p->next; + free( p ); + p = next; + } + cpp->once_files = NULL; +} + + +static enum mcpu_language +current_language( const mcpp *cpp ) +{ + return( cpp->lang_stack[cpp->lang_depth - 1] ); +} + +static int +install_builtin( mcpp *cpp, const char *name, + enum mcpu_macro_builtin builtin ) +{ + mcpu_text text; + int rc; + + mcpu_text_init( &text ); + if( mcpu_text_from_utf8(&text, name) != 0 ) + return( -1 ); + + rc = mcpu_macro_define_builtin( &cpp->macros, text.data, text.length, + builtin ); + mcpu_text_free( &text ); + return( rc ); +} + + +static int +initialize_builtins( mcpp *cpp ) +{ + if( cpp->builtins_initialized ) + return( 0 ); + + if( install_builtin(cpp, "__FILE__", + MCPU_MACRO_BUILTIN_FILE) != 0 || + install_builtin(cpp, "__LINE__", + MCPU_MACRO_BUILTIN_LINE) != 0 || + install_builtin(cpp, "__DATE__", + MCPU_MACRO_BUILTIN_DATE) != 0 || + install_builtin(cpp, "__TIME__", + MCPU_MACRO_BUILTIN_TIME) != 0 || + install_builtin(cpp, "__BASE_FILE__", + MCPU_MACRO_BUILTIN_BASE_FILE) != 0 || + install_builtin(cpp, "__INCLUDE_LEVEL__", + MCPU_MACRO_BUILTIN_INCLUDE_LEVEL) != 0 || + mcpu_predefined_install(&cpp->macros) != 0 ) + return( -1 ); + + cpp->builtins_initialized = 1; + return( 0 ); +} + + +static int +initialize_timestamp( mcpp *cpp ) +{ + static const char *months[] = + { + "Jan", "Feb", "Mar", "Apr", "May", "Jun", + "Jul", "Aug", "Sep", "Oct", "Nov", "Dec" + }; + time_t now; + struct tm *tm; + + now = time( NULL ); + if( now == (time_t)-1 ) + return( -1 ); + + tm = localtime( &now ); + if( tm == NULL || tm->tm_mon < 0 || tm->tm_mon > 11 ) + { + errno = EINVAL; + return( -1 ); + } + + snprintf( cpp->preprocess_date, sizeof(cpp->preprocess_date), + "%s %2d %4d", months[tm->tm_mon], tm->tm_mday, + tm->tm_year + 1900 ); + snprintf( cpp->preprocess_time, sizeof(cpp->preprocess_time), + "%02d:%02d:%02d", tm->tm_hour, tm->tm_min, tm->tm_sec ); + + return( 0 ); +} + + +static void +macro_context( const mcpp *cpp, const char *filename, + unsigned line_number, mcpu_macro_expansion_context *context ) +{ + context->filename = filename; + context->base_filename = cpp->base_filename; + context->line_number = line_number; + context->include_level = cpp->include_depth ? cpp->include_depth - 1 : 0; + context->date = cpp->preprocess_date; + context->time = cpp->preprocess_time; + context->options = MCPP_OPTIONS(cpp); +} + + +static char * +escape_line_filename( const char *filename ) +{ + size_t n; + size_t i; + size_t o = 0; + char *escaped; + + if( filename == NULL ) + { + errno = EINVAL; + return( NULL ); + } + + n = strlen( filename ); + if( n > (SIZE_MAX - 1) / 2 ) + { + errno = EOVERFLOW; + return( NULL ); + } + + escaped = (char *)malloc( n * 2 + 1 ); + if( escaped == NULL ) + return( NULL ); + + for( i = 0; i < n; ++i ) + { + switch( filename[i] ) + { + case '\\': + case '"': + escaped[o++] = '\\'; + escaped[o++] = filename[i]; + break; + case '\n': + escaped[o++] = '\\'; + escaped[o++] = 'n'; + break; + case '\r': + escaped[o++] = '\\'; + escaped[o++] = 'r'; + break; + case '\t': + escaped[o++] = '\\'; + escaped[o++] = 't'; + break; + default: + escaped[o++] = filename[i]; + break; + } + } + + escaped[o] = 0; + return( escaped ); +} + +static int +set_output_position( mcpp *cpp, unsigned line, const char *filename ) +{ + char *copy; + + if( cpp == NULL || filename == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + copy = strdup( filename ); + if( copy == NULL ) + return( -1 ); + + free( cpp->output_filename ); + cpp->output_filename = copy; + cpp->output_line = line; + return( 0 ); +} + +static int +append_line_marker( mcpp *cpp, unsigned line, const char *filename, + int file_change ) +{ + char number[64]; + char *escaped; + char *position_filename; + + escaped = escape_line_filename( filename ); + if( escaped == NULL ) + return( -1 ); + + position_filename = strdup( filename ); + if( position_filename == NULL ) + { + free( escaped ); + return( -1 ); + } + + if( mcpu_text_append_ascii( &cpp->output, "# " ) != 0 ) + { + free( position_filename ); + free( escaped ); + return( -1 ); + } + + snprintf( number, sizeof(number), "%u", line ); + if( mcpu_text_append_ascii( &cpp->output, number ) != 0 || + mcpu_text_append_ascii( &cpp->output, " \"" ) != 0 || + mcpu_text_append_utf8( &cpp->output, escaped ) != 0 || + mcpu_text_append_ascii( &cpp->output, "\"" ) != 0 ) + { + free( position_filename ); + free( escaped ); + return( -1 ); + } + + if( file_change == 1 || file_change == 2 ) + { + if( mcpu_text_append_ascii(&cpp->output, + file_change == 1 ? " 1" : " 2") != 0 ) + { + free( position_filename ); + free( escaped ); + return( -1 ); + } + } + + if( mcpu_text_append_char(&cpp->output, '\n') != 0 ) + { + free( position_filename ); + free( escaped ); + return( -1 ); + } + + free( cpp->output_filename ); + cpp->output_filename = position_filename; + cpp->output_line = line; + free( escaped ); + return( 0 ); +} + +static int +append_output_newlines( mcpp *cpp, unsigned count ) +{ + unsigned i; + + for( i = 0; i < count; ++i ) + { + if( mcpu_text_append_char(&cpp->output, '\n') != 0 ) + return( -1 ); + } + + if( cpp->output_filename != NULL ) + cpp->output_line += count; + + return( 0 ); +} + +/* + * GNU CPP keeps short invisible source gaps as ordinary newlines, but once + * the next visible source position is eight or more lines away it emits a + * fresh line marker instead. The discarded text is therefore forgotten as + * output bytes, but never forgotten as source position: "remove, but + * remember". + */ +static int +sync_output_position( mcpp *cpp, unsigned line, const char *filename ) +{ + if( cpp->output_filename != NULL && + strcmp(cpp->output_filename, filename) == 0 && + line >= cpp->output_line && line - cpp->output_line < 8 ) + return( append_output_newlines(cpp, line - cpp->output_line) ); + + return( append_line_marker(cpp, line, filename, 0) ); +} + +static int +text_has_visible_tokens( const __mpu_char16_t *text, size_t length ) +{ + size_t i; + + for( i = 0; i < length; ++i ) + { + switch( text[i] ) + { + case ' ': + case '\t': + case '\f': + case '\v': + case '\r': + case '\n': + break; + default: + return( 1 ); + } + } + + return( 0 ); +} + +static void +advance_output_position( mcpp *cpp, + const __mpu_char16_t *text, size_t length ) +{ + size_t i; + + if( cpp->output_filename == NULL ) + return; + + for( i = 0; i < length; ++i ) + if( text[i] == '\n' ) + ++cpp->output_line; +} + +static int +append_visible_text( mcpp *cpp, unsigned line, const char *filename, + const __mpu_char16_t *text, size_t length ) +{ + if( !text_has_visible_tokens(text, length) ) + return( 0 ); + + if( sync_output_position(cpp, line, filename) != 0 || + mcpu_text_append(&cpp->output, text, length) != 0 ) + return( -1 ); + + advance_output_position( cpp, text, length ); + return( 0 ); +} + +static int +is_space16( __mpu_char16_t c ) +{ + return( c == ' ' || c == '\t' || c == '\f' || c == '\v' || c == '\r' ); +} + +static int +is_ident16( __mpu_char16_t c ) +{ + return( (c >= 'a' && c <= 'z') || + (c >= 'A' && c <= 'Z') || + (c >= '0' && c <= '9') || c == '_' || c >= 0x80 ); +} + +static int +replacement_quote16( __mpu_char16_t c, enum mcpu_language language ) +{ + if( c == '"' || c == '`' ) + return( 1 ); + + if( c == '\'' && language != MCPU_LANG_DIFF ) + return( 1 ); + + return( 0 ); +} + + +static int +normalize_macro_replacement( const __mpu_char16_t *text, size_t length, + enum mcpu_language language, + mcpu_text *normalized ) +{ + size_t p = 0; + __mpu_char16_t quote = 0; + int escaped = 0; + int pending_space = 0; + + while( p < length ) + { + __mpu_char16_t c = text[p++]; + + if( quote != 0 ) + { + if( mcpu_text_append_char(normalized, c) != 0 ) + return( -1 ); + + if( escaped ) + escaped = 0; + else if( c == '\\' ) + escaped = 1; + else if( c == quote || c == '\n' ) + quote = 0; + + continue; + } + + if( is_space16(c) || c == '\n' ) + { + if( normalized->length != 0 ) + pending_space = 1; + continue; + } + + if( pending_space ) + { + if( mcpu_text_append_char(normalized, ' ') != 0 ) + return( -1 ); + pending_space = 0; + } + + if( mcpu_text_append_char(normalized, c) != 0 ) + return( -1 ); + + if( replacement_quote16(c, language) ) + quote = c; + } + + return( 0 ); +} + + +static void +skip_space16( const __mpu_char16_t *line, size_t length, size_t *pos ) +{ + while( *pos < length && is_space16(line[*pos]) ) + ++*pos; +} + +static int +parse_directive_name( const __mpu_char16_t *line, size_t length, + size_t hash, size_t *name_start, size_t *name_length, + size_t *after_name ) +{ + size_t p = hash + 1; + size_t start; + + skip_space16( line, length, &p ); + start = p; + while( p < length && is_ident16(line[p]) ) + ++p; + + *name_start = start; + *name_length = p - start; + *after_name = p; + + return( *name_length != 0 ); +} + +static int +handle_lang( mcpp *cpp, const char *filename, unsigned line_number, + const __mpu_char16_t *line, size_t length, size_t p, + int spliced ) +{ + size_t start; + size_t end; + size_t q; + enum mcpu_language language; + char *name; + int newline; + + if( spliced ) + { + fprintf( stderr, + "%s:%u: error: #lang must be contained in one physical source line\n", + filename, line_number ); + return( -1 ); + } + + skip_space16( line, length, &p ); + if( p >= length || line[p] != '"' ) + { + fprintf( stderr, + "%s:%u: error: string constant expected after #lang\n", + filename, line_number ); + return( -1 ); + } + + start = ++p; + while( p < length && line[p] != '"' && line[p] != '\n' ) + ++p; + + if( p >= length || line[p] != '"' ) + { + fprintf( stderr, + "%s:%u: error: unterminated #lang string constant\n", + filename, line_number ); + return( -1 ); + } + end = p++; + + if( start == end ) + { + fprintf( stderr, + "%s:%u: error: empty language name in #lang\n", + filename, line_number ); + return( -1 ); + } + + for( q = start; q < end; ++q ) + { + if( line[q] == ' ' || line[q] == '\t' || line[q] == '\f' || + line[q] == '\v' || line[q] == '\r' || line[q] == '\n' ) + { + fprintf( stderr, + "%s:%u: error: #lang language name must be one word without whitespace\n", + filename, line_number ); + return( -1 ); + } + } + + skip_space16( line, length, &p ); + if( p < length && line[p] != '\n' ) + { + fprintf( stderr, + "%s:%u: error: extra text after #lang string constant\n", + filename, line_number ); + return( -1 ); + } + + if( mcpu_language_from_text( line + start, end - start, &language ) != 0 ) + { + name = mcpu_text_to_utf8( line + start, end - start ); + fprintf( stderr, "%s:%u: error: unknown language '%s'\n", + filename, line_number, name ? name : "?" ); + free( name ); + return( -1 ); + } + + if( cpp->lang_depth >= MCPU_CPP_LANG_STACK_SIZE ) + { + fprintf( stderr, "%s:%u: error: #lang stack overflow\n", + filename, line_number ); + return( -1 ); + } + + cpp->lang_stack[cpp->lang_depth++] = language; + + if( cpp->verbose ) + fprintf( stderr, "%s:%u: #lang %s (depth %lu)\n", + filename, line_number, mcpu_language_name(language), + (unsigned long)cpp->lang_depth ); + + newline = length != 0 && line[length - 1] == '\n'; + if( mcpu_text_append_ascii(&cpp->output, "#lang \"") != 0 || + mcpu_text_append(&cpp->output, line + start, end - start) != 0 || + mcpu_text_append_char(&cpp->output, '"') != 0 || + (newline && mcpu_text_append_char(&cpp->output, '\n') != 0) ) + return( -1 ); + + return( 0 ); +} + +static int +handle_endlang( mcpp *cpp, const char *filename, unsigned line_number, + const __mpu_char16_t *line, size_t length ) +{ + if( cpp->lang_depth <= 1 ) + { + fprintf( stderr, "%s:%u: error: unbalanced #endlang\n", + filename, line_number ); + return( -1 ); + } + + --cpp->lang_depth; + + if( cpp->verbose ) + fprintf( stderr, "%s:%u: #endlang -> %s (depth %lu)\n", + filename, line_number, mcpu_language_name(current_language(cpp)), + (unsigned long)cpp->lang_depth ); + + return( mcpu_text_append( &cpp->output, line, length ) ); +} + +static int +parse_include_filename( const __mpu_char16_t *line, size_t length, size_t p, + char **filename, int *quoted ) +{ + __mpu_char16_t open; + __mpu_char16_t close; + size_t start; + size_t end; + + *filename = NULL; + *quoted = 0; + + skip_space16( line, length, &p ); + if( p >= length ) + return( -1 ); + + open = line[p++]; + if( open == '"' ) + { + close = '"'; + *quoted = 1; + } + else if( open == '<' ) + close = '>'; + else + return( -1 ); + + start = p; + while( p < length && line[p] != close && line[p] != '\n' ) + ++p; + end = p; + + if( p >= length || line[p] != close || end == start ) + return( -1 ); + + *filename = mcpu_text_to_utf8( line + start, end - start ); + return( *filename ? 0 : -1 ); +} + + +static size_t +line_content_end( const __mpu_char16_t *line, size_t length ) +{ + size_t end = length; + + if( end != 0 && line[end - 1] == '\n' ) + --end; + + while( end != 0 && is_space16(line[end - 1]) ) + --end; + + return( end ); +} + +static void +free_define_args( const __mpu_char16_t **argnames, + size_t *argname_lengths ) +{ + free( argnames ); + free( argname_lengths ); +} + + +static int +append_define_arg( const __mpu_char16_t ***argnames, + size_t **argname_lengths, + size_t *nargs, size_t *capacity, + const __mpu_char16_t *name, size_t name_length ) +{ + const __mpu_char16_t **new_names; + size_t *new_lengths; + size_t new_capacity; + + if( *nargs == *capacity ) + { + new_capacity = *capacity ? *capacity * 2 : 4; + if( new_capacity < *capacity || + new_capacity > SIZE_MAX / sizeof(**argnames) || + new_capacity > SIZE_MAX / sizeof(**argname_lengths) ) + { + errno = EOVERFLOW; + return( -1 ); + } + + new_names = (const __mpu_char16_t **)realloc( + (void *)*argnames, new_capacity * sizeof(**argnames) ); + if( new_names == NULL ) + return( -1 ); + + *argnames = new_names; + + new_lengths = (size_t *)realloc( *argname_lengths, + new_capacity * sizeof(**argname_lengths) ); + if( new_lengths == NULL ) + return( -1 ); + + *argname_lengths = new_lengths; + *capacity = new_capacity; + } + + (*argnames)[*nargs] = name; + (*argname_lengths)[*nargs] = name_length; + ++*nargs; + + return( 0 ); +} + + +static int +replacement_va_opt_range_valid( const __mpu_char16_t *text, + size_t length, + enum mcpu_language language, + int function_like, int variadic, + int inside_va_opt, + const char *filename, + unsigned line_number ) +{ + size_t p = 0; + __mpu_char16_t quote = 0; + int escaped = 0; + + while( p < length ) + { + __mpu_char16_t c = text[p]; + + if( quote ) + { + if( escaped ) + escaped = 0; + else if( c == '\\' ) + escaped = 1; + else if( c == quote || c == '\n' ) + quote = 0; + + ++p; + continue; + } + + if( c == '"' || c == '`' || + (c == '\'' && language != MCPU_LANG_DIFF) ) + { + quote = c; + ++p; + continue; + } + + if( mcpu_pp_is_identifier_start(c) ) + { + size_t q = p + 1; + + while( q < length && mcpu_pp_is_identifier_char(text[q]) ) + ++q; + + if( mcpu_text_equal_ascii(text + p, q - p, "__VA_OPT__") ) + { + size_t open; + size_t close; + size_t first; + size_t last; + size_t r; + int depth; + __mpu_char16_t inner_quote = 0; + int inner_escaped = 0; + + if( !function_like || !variadic ) + { + fprintf( stderr, + "%s:%u: error: '__VA_OPT__' may appear only in a variadic macro replacement list\n", + filename, line_number ); + return( 0 ); + } + + if( inside_va_opt ) + { + fprintf( stderr, + "%s:%u: error: '__VA_OPT__' may not appear inside another '__VA_OPT__'\n", + filename, line_number ); + return( 0 ); + } + + open = q; + while( open < length && is_space16(text[open]) ) + ++open; + + if( open >= length || text[open] != '(' ) + { + fprintf( stderr, + "%s:%u: error: '__VA_OPT__' must be followed by '('\n", + filename, line_number ); + return( 0 ); + } + + depth = 1; + close = open + 1; + while( close < length && depth != 0 ) + { + __mpu_char16_t d = text[close]; + + if( inner_quote ) + { + if( inner_escaped ) + inner_escaped = 0; + else if( d == '\\' ) + inner_escaped = 1; + else if( d == inner_quote || d == '\n' ) + inner_quote = 0; + ++close; + continue; + } + + if( d == '"' || d == '`' || + (d == '\'' && language != MCPU_LANG_DIFF) ) + { + inner_quote = d; + ++close; + continue; + } + + if( d == '(' ) + ++depth; + else if( d == ')' ) + --depth; + + ++close; + } + + if( depth != 0 ) + { + fprintf( stderr, + "%s:%u: error: unterminated '__VA_OPT__'\n", + filename, line_number ); + return( 0 ); + } + + --close; + first = open + 1; + while( first < close && is_space16(text[first]) ) + ++first; + last = close; + while( last > first && is_space16(text[last - 1]) ) + --last; + + if( first + 1 < last && text[first] == '#' && text[first + 1] == '#' ) + { + fprintf( stderr, + "%s:%u: error: '##' cannot appear at the beginning of '__VA_OPT__'\n", + filename, line_number ); + return( 0 ); + } + + if( last >= first + 2 && text[last - 2] == '#' && text[last - 1] == '#' ) + { + fprintf( stderr, + "%s:%u: error: '##' cannot appear at the end of '__VA_OPT__'\n", + filename, line_number ); + return( 0 ); + } + + if( !replacement_va_opt_range_valid(text + open + 1, + close - open - 1, + language, + function_like, variadic, 1, + filename, line_number) ) + return( 0 ); + + r = close + 1; + p = r; + continue; + } + + p = q; + continue; + } + + ++p; + } + + return( 1 ); +} + + +static int +replacement_va_opt_valid( const __mpu_char16_t *text, + size_t length, + enum mcpu_language language, + int function_like, int variadic, + const char *filename, + unsigned line_number ) +{ + return( replacement_va_opt_range_valid(text, length, + language, + function_like, variadic, 0, + filename, line_number) ); +} + + +static int +replacement_macro_operators_valid( const __mpu_char16_t *text, + size_t length, + enum mcpu_language language, + int function_like, int variadic, + const __mpu_char16_t *const *argnames, + const size_t *argname_lengths, + size_t nargs, + const char *filename, + unsigned line_number ) +{ + size_t p = 0; + __mpu_char16_t quote = 0; + int escaped = 0; + int saw_token = 0; + int paste_at_end = 0; + + while( p < length ) + { + __mpu_char16_t c = text[p]; + + if( quote ) + { + if( escaped ) + escaped = 0; + else if( c == '\\' ) + escaped = 1; + else if( c == quote || c == '\n' ) + quote = 0; + + ++p; + continue; + } + + if( c == '"' || c == '`' || + (c == '\'' && language != MCPU_LANG_DIFF) ) + { + quote = c; + saw_token = 1; + paste_at_end = 0; + ++p; + continue; + } + + if( c == '#' && p + 1 < length && text[p + 1] == '#' ) + { + if( !saw_token ) + { + fprintf( stderr, + "%s:%u: error: '##' cannot appear at the beginning of a macro replacement list\n", + filename, line_number ); + return( 0 ); + } + + paste_at_end = 1; + p += 2; + continue; + } + + if( c == '#' && function_like ) + { + size_t q = p + 1; + size_t start; + size_t i; + int found = 0; + + while( q < length && is_space16(text[q]) ) + ++q; + + if( q >= length || !mcpu_pp_is_identifier_start(text[q]) ) + { + fprintf( stderr, + "%s:%u: error: '#' operator is not followed by a macro argument name\n", + filename, line_number ); + return( 0 ); + } + + start = q++; + while( q < length && mcpu_pp_is_identifier_char(text[q]) ) + ++q; + + for( i = 0; i < nargs; ++i ) + { + if( argname_lengths[i] == q - start && + memcmp(argnames[i], text + start, + (q - start) * sizeof(__mpu_char16_t)) == 0 ) + { + found = 1; + break; + } + } + + if( !found && variadic && + (mcpu_text_equal_ascii(text + start, q - start, "__VA_ARGS__") || + mcpu_text_equal_ascii(text + start, q - start, "__VA_OPT__")) ) + found = 1; + + if( !found ) + { + fprintf( stderr, + "%s:%u: error: '#' operator should be followed by a macro argument name\n", + filename, line_number ); + return( 0 ); + } + + saw_token = 1; + paste_at_end = 0; + p = q; + continue; + } + + if( is_space16(c) ) + { + ++p; + continue; + } + + saw_token = 1; + paste_at_end = 0; + ++p; + } + + if( paste_at_end ) + { + fprintf( stderr, + "%s:%u: error: '##' cannot appear at the end of a macro replacement list\n", + filename, line_number ); + return( 0 ); + } + + return( 1 ); +} + + +static int +handle_define( mcpp *cpp, const char *filename, unsigned line_number, + const __mpu_char16_t *line, size_t length, size_t p, + enum mcpu_macro_origin origin ) +{ + size_t name_start; + size_t name_end; + size_t replacement_start; + size_t replacement_end; + const __mpu_char16_t **argnames = NULL; + size_t *argname_lengths = NULL; + size_t nargs = 0; + size_t capacity = 0; + int function_like = 0; + int variadic = 0; + mcpu_macro *old; + mcpu_text replacement; + int rc = -1; + + mcpu_text_init( &replacement ); + + skip_space16( line, length, &p ); + if( p >= length || !mcpu_pp_is_identifier_start(line[p]) ) + { + fprintf( stderr, "%s:%u: error: macro name expected after #define\n", + filename, line_number ); + return( -1 ); + } + + name_start = p++; + while( p < length && mcpu_pp_is_identifier_char(line[p]) ) + ++p; + name_end = p; + + /*************************************************************** + A function-like definition is recognized only when + the opening parenthesis immediately follows the macro name. + A space changes the definition into an object-like macro whose + replacement happens to begin with '('. + ***************************************************************/ + if( p < length && line[p] == '(' ) + { + size_t i; + + function_like = 1; + ++p; + skip_space16( line, length, &p ); + + if( p + 2 < length && + line[p] == '.' && line[p + 1] == '.' && line[p + 2] == '.' ) + { + variadic = 1; + p += 3; + skip_space16( line, length, &p ); + if( p >= length || line[p] != ')' ) + { + fprintf( stderr, + "%s:%u: error: badly punctuated parameter list in #define\n", + filename, line_number ); + goto done; + } + } + else + { + while( p < length && line[p] != ')' ) + { + size_t arg_start; + size_t arg_end; + + if( !mcpu_pp_is_identifier_start(line[p]) ) + { + fprintf( stderr, + "%s:%u: error: invalid macro parameter name in #define\n", + filename, line_number ); + goto done; + } + + arg_start = p++; + while( p < length && mcpu_pp_is_identifier_char(line[p]) ) + ++p; + arg_end = p; + + if( mcpu_text_equal_ascii(line + arg_start, arg_end - arg_start, + "__VA_ARGS__") ) + { + fprintf( stderr, + "%s:%u: error: '__VA_ARGS__' cannot be used as a macro parameter name\n", + filename, line_number ); + goto done; + } + + for( i = 0; i < nargs; ++i ) + { + if( argname_lengths[i] == arg_end - arg_start && + memcmp( argnames[i], line + arg_start, + (arg_end - arg_start) * sizeof(__mpu_char16_t) ) == 0 ) + { + char *arg = mcpu_text_to_utf8( line + arg_start, + arg_end - arg_start ); + fprintf( stderr, + "%s:%u: error: duplicate argument name '%s' in #define\n", + filename, line_number, arg ? arg : "?" ); + free( arg ); + goto done; + } + } + + if( append_define_arg(&argnames, &argname_lengths, + &nargs, &capacity, + line + arg_start, arg_end - arg_start) != 0 ) + goto done; + + skip_space16( line, length, &p ); + if( p >= length ) + { + fprintf( stderr, + "%s:%u: error: unterminated parameter list in #define\n", + filename, line_number ); + goto done; + } + + if( line[p] == ',' ) + { + ++p; + skip_space16( line, length, &p ); + + if( p + 2 < length && + line[p] == '.' && line[p + 1] == '.' && line[p + 2] == '.' ) + { + variadic = 1; + p += 3; + skip_space16( line, length, &p ); + if( p >= length || line[p] != ')' ) + { + fprintf( stderr, + "%s:%u: error: badly punctuated parameter list in #define\n", + filename, line_number ); + goto done; + } + break; + } + + if( p >= length || !mcpu_pp_is_identifier_start(line[p]) ) + { + fprintf( stderr, + "%s:%u: error: badly punctuated parameter list in #define\n", + filename, line_number ); + goto done; + } + continue; + } + + if( line[p] != ')' ) + { + fprintf( stderr, + "%s:%u: error: badly punctuated parameter list in #define\n", + filename, line_number ); + goto done; + } + } + } + + if( p >= length || line[p] != ')' ) + { + fprintf( stderr, + "%s:%u: error: unterminated parameter list in #define\n", + filename, line_number ); + goto done; + } + + ++p; + } + + skip_space16( line, length, &p ); + replacement_start = p; + replacement_end = line_content_end( line, length ); + if( replacement_end < replacement_start ) + replacement_end = replacement_start; + + if( normalize_macro_replacement(line + replacement_start, + replacement_end - replacement_start, + current_language(cpp), &replacement) != 0 ) + goto done; + + if( !replacement_va_opt_valid( + replacement.data, replacement.length, + current_language(cpp), function_like, variadic, + filename, line_number) || + !replacement_macro_operators_valid( + replacement.data, replacement.length, + current_language(cpp), function_like, variadic, + argnames, argname_lengths, nargs, + filename, line_number) ) + goto done; + + old = mcpu_macro_find( &cpp->macros, + line + name_start, name_end - name_start ); + if( old && + !mcpu_macro_definition_equal(old, + function_like ? (int)nargs : -1, + function_like ? variadic : 0, + argnames, argname_lengths, + replacement.data, + replacement.length) ) + { + char *name = mcpu_text_to_utf8( line + name_start, + name_end - name_start ); + if( mcpp_diagnostic_warning(MCPP_OPTIONS(cpp), filename, line_number, + "macro '%s' redefined", + name ? name : "?") != 0 ) + { + free( name ); + goto done; + } + free( name ); + } + + if( function_like ) + rc = mcpu_macro_define_function(&cpp->macros, + line + name_start, name_end - name_start, + argnames, argname_lengths, nargs, variadic, + replacement.data, replacement.length, + origin); + else + rc = mcpu_macro_define_object(&cpp->macros, + line + name_start, name_end - name_start, + replacement.data, replacement.length, + origin); + +done: + mcpu_text_free( &replacement ); + free_define_args( argnames, argname_lengths ); + return( rc ); +} + + +static int +handle_undef( mcpp *cpp, const char *filename, unsigned line_number, + const __mpu_char16_t *line, size_t length, size_t p ) +{ + size_t name_start; + size_t name_end; + + skip_space16( line, length, &p ); + if( p >= length || !mcpu_pp_is_identifier_start(line[p]) ) + { + fprintf( stderr, "%s:%u: error: macro name expected after #undef\n", + filename, line_number ); + return( -1 ); + } + + name_start = p++; + while( p < length && mcpu_pp_is_identifier_char(line[p]) ) + ++p; + name_end = p; + + skip_space16( line, length, &p ); + if( p < length && line[p] != '\n' ) + { + fprintf( stderr, "%s:%u: error: extra tokens after #undef\n", + filename, line_number ); + return( -1 ); + } + + if( mcpu_macro_undef(&cpp->macros, + line + name_start, name_end - name_start) < 0 ) + return( -1 ); + + return( 0 ); +} + + +static int +command_line_text( const mcpp *cpp, const mcpp_option_action *action, + mcpu_text *text ) +{ + if( mcpu_text_from_utf8(text, action->argument) == 0 ) + return( 0 ); + + if( errno == EILSEQ ) + fprintf( stderr, + "%s: %s: invalid UTF-8 or non-UCS-2 character\n", + MCPP_OPTIONS(cpp)->progname, action->option ); + else + fprintf( stderr, "%s: %s: %s\n", + MCPP_OPTIONS(cpp)->progname, action->option, strerror(errno) ); + + return( -1 ); +} + + +static size_t +command_line_define_declarator_end( const mcpu_text *text, size_t equal ) +{ + size_t p; + + if( text == NULL || text->length == 0 || equal == 0 ) + return( 0 ); + + if( !mcpu_pp_is_identifier_start(text->data[0]) ) + return( equal ); + + p = 1; + while( p < equal && mcpu_pp_is_identifier_char(text->data[p]) ) + ++p; + + if( p < equal && text->data[p] == '(' ) + { + ++p; + while( p < equal && text->data[p] != ')' ) + ++p; + if( p < equal && text->data[p] == ')' ) + ++p; + } + + return( p ); +} + + +static int +prepare_command_line_define( mcpp *cpp, const mcpp_option_action *action, + const mcpu_text *text, mcpu_text *prepared ) +{ + mcpu_text line; + mcpu_pp_lexer_state lexer; + size_t equal; + size_t declarator_end; + int has_equal; + int rc = -1; + + mcpu_text_init( &line ); + + equal = 0; + has_equal = 0; + while( equal < text->length ) + { + if( text->data[equal] == '=' ) + { + has_equal = 1; + break; + } + ++equal; + } + + declarator_end = command_line_define_declarator_end( + text, has_equal ? equal : text->length ); + + if( mcpu_text_append(&line, text->data, declarator_end) != 0 ) + goto done; + + if( has_equal ) + { + if( mcpu_text_append_char(&line, ' ') != 0 || + mcpu_text_append(&line, text->data + equal + 1, + text->length - equal - 1) != 0 ) + goto done; + } + else + { + if( mcpu_text_append_ascii(&line, " 1") != 0 ) + goto done; + } + + mcpu_pp_lexer_init( &lexer ); + if( mcpu_pp_prepare_line(&lexer, line.data, line.length, + current_language(cpp), prepared) != 0 ) + goto done; + + if( lexer.in_block_comment ) + { + fprintf( stderr, "%s: %s: unterminated comment\n", + MCPP_OPTIONS(cpp)->progname, action->option ); + goto done; + } + + rc = 0; + +done: + mcpu_text_free( &line ); + return( rc ); +} + + +static int +prepare_command_line_undef( mcpp *cpp, const mcpp_option_action *action, + const mcpu_text *text, mcpu_text *prepared ) +{ + mcpu_text line; + mcpu_pp_lexer_state lexer; + size_t p; + int rc = -1; + + mcpu_text_init( &line ); + + if( text->length != 0 && mcpu_pp_is_identifier_start(text->data[0]) ) + { + p = 1; + while( p < text->length && mcpu_pp_is_identifier_char(text->data[p]) ) + ++p; + } + else + p = text->length; + + if( mcpu_text_append(&line, text->data, p) != 0 ) + goto done; + + mcpu_pp_lexer_init( &lexer ); + if( mcpu_pp_prepare_line(&lexer, line.data, line.length, + current_language(cpp), prepared) != 0 ) + goto done; + + if( lexer.in_block_comment ) + { + fprintf( stderr, "%s: %s: unterminated comment\n", + MCPP_OPTIONS(cpp)->progname, action->option ); + goto done; + } + + rc = 0; + +done: + mcpu_text_free( &line ); + return( rc ); +} + + +static int +apply_command_line_define( mcpp *cpp, const mcpp_option_action *action ) +{ + mcpu_text text; + mcpu_text prepared; + int rc; + + mcpu_text_init( &text ); + mcpu_text_init( &prepared ); + + if( command_line_text(cpp, action, &text) != 0 || + prepare_command_line_define(cpp, action, &text, &prepared) != 0 ) + goto fail; + + rc = handle_define( cpp, "<command-line>", 0, + prepared.data, prepared.length, 0, + MCPU_MACRO_ORIGIN_COMMAND_LINE ); + mcpu_text_free( &prepared ); + mcpu_text_free( &text ); + return( rc ); + +fail: + mcpu_text_free( &prepared ); + mcpu_text_free( &text ); + return( -1 ); +} + + +static int +apply_command_line_undef( mcpp *cpp, const mcpp_option_action *action ) +{ + mcpu_text text; + mcpu_text prepared; + int rc; + + mcpu_text_init( &text ); + mcpu_text_init( &prepared ); + + if( command_line_text(cpp, action, &text) != 0 || + prepare_command_line_undef(cpp, action, &text, &prepared) != 0 ) + { + mcpu_text_free( &prepared ); + mcpu_text_free( &text ); + return( -1 ); + } + + rc = handle_undef( cpp, "<command-line>", 0, + prepared.data, prepared.length, 0 ); + mcpu_text_free( &prepared ); + mcpu_text_free( &text ); + return( rc ); +} + + +static int +apply_command_line_macros( mcpp *cpp ) +{ + size_t i; + + if( cpp->command_line_macros_applied || MCPP_OPTIONS(cpp) == NULL ) + return( 0 ); + + for( i = 0; i < MCPP_OPTIONS(cpp)->action_count; ++i ) + { + const mcpp_option_action *action = &MCPP_OPTIONS(cpp)->actions[i]; + + switch( action->kind ) + { + case MCPP_ACTION_DEFINE: + if( apply_command_line_define(cpp, action) != 0 ) + return( -1 ); + break; + case MCPP_ACTION_UNDEF: + if( apply_command_line_undef(cpp, action) != 0 ) + return( -1 ); + break; + default: + break; + } + } + + cpp->command_line_macros_applied = 1; + return( 0 ); +} + + +enum mcpp_conditional_directive +{ + MCPP_COND_NONE = 0, + MCPP_COND_IF, + MCPP_COND_IFDEF, + MCPP_COND_IFNDEF, + MCPP_COND_ELIF, + MCPP_COND_ELSE, + MCPP_COND_ENDIF +}; + +static int +diagnostic_directive( const __mpu_char16_t *name, size_t length, + int *is_error, const char **directive_name ) +{ + if( mcpu_text_equal_ascii(name, length, "error") ) + { + *is_error = 1; + *directive_name = "error"; + return( 1 ); + } + + if( mcpu_text_equal_ascii(name, length, "warning") ) + { + *is_error = 0; + *directive_name = "warning"; + return( 1 ); + } + + return( 0 ); +} + +static int +append_diagnostic_text( mcpu_text *message, + const __mpu_char16_t *line, size_t length, + size_t p ) +{ + int pending_space = 0; + __mpu_char16_t quote = 0; + int escaped = 0; + + skip_space16( line, length, &p ); + + while( p < length && line[p] != '\n' ) + { + __mpu_char16_t c = line[p++]; + + if( quote ) + { + if( mcpu_text_append_char(message, c) != 0 ) + return( -1 ); + + if( escaped ) + escaped = 0; + else if( c == '\\' ) + escaped = 1; + else if( c == quote ) + quote = 0; + continue; + } + + if( c == ' ' || c == '\t' || c == '\f' || c == '\v' || c == '\r' ) + { + if( message->length != 0 ) + pending_space = 1; + continue; + } + + if( pending_space ) + { + if( mcpu_text_append_char(message, ' ') != 0 ) + return( -1 ); + pending_space = 0; + } + + if( c == '"' || c == '\'' || c == '`' ) + quote = c; + + if( mcpu_text_append_char(message, c) != 0 ) + return( -1 ); + } + + return( 0 ); +} + +static int +handle_diagnostic( mcpp *cpp, const char *filename, unsigned line_number, + const __mpu_char16_t *line, size_t length, size_t p, + int is_error, const char *directive_name ) +{ + mcpu_text message; + char *utf8; + + mcpu_text_init( &message ); + if( append_diagnostic_text(&message, line, length, p) != 0 ) + { + mcpu_text_free( &message ); + return( -1 ); + } + + utf8 = mcpu_text_to_utf8( message.data, message.length ); + mcpu_text_free( &message ); + if( utf8 == NULL ) + return( -1 ); + + if( is_error ) + { + fprintf( stderr, "%s:%u: error: #%s%s%s\n", + filename, line_number, directive_name, + *utf8 ? " " : "", utf8 ); + free( utf8 ); + return( -1 ); + } + + { + int rc = mcpp_diagnostic_warning(MCPP_OPTIONS(cpp), filename, line_number, + "#%s%s%s", directive_name, + *utf8 ? " " : "", utf8); + free( utf8 ); + return( rc ); + } +} + +static enum mcpp_conditional_directive +conditional_directive( const __mpu_char16_t *name, size_t length ) +{ + if( mcpu_text_equal_ascii(name, length, "if") ) + return( MCPP_COND_IF ); + + if( mcpu_text_equal_ascii(name, length, "ifdef") ) + return( MCPP_COND_IFDEF ); + + if( mcpu_text_equal_ascii(name, length, "ifndef") ) + return( MCPP_COND_IFNDEF ); + + if( mcpu_text_equal_ascii(name, length, "elif") ) + return( MCPP_COND_ELIF ); + + if( mcpu_text_equal_ascii(name, length, "else") ) + return( MCPP_COND_ELSE ); + + if( mcpu_text_equal_ascii(name, length, "endif") ) + return( MCPP_COND_ENDIF ); + + return( MCPP_COND_NONE ); +} + +static int +conditional_active( const mcpp *cpp ) +{ + if( cpp->conditional_depth == 0 ) + return( 1 ); + + return( cpp->conditional_stack[cpp->conditional_depth - 1].active ); +} + +static int +conditional_push( mcpp *cpp, const char *filename, unsigned line_number, + int parent_active, int condition ) +{ + mcpp_conditional *frame; + + if( cpp->conditional_depth >= MCPU_CPP_CONDITIONAL_STACK_SIZE ) + { + fprintf( stderr, "%s:%u: error: conditional nesting exceeds %d levels\n", + filename, line_number, MCPU_CPP_CONDITIONAL_STACK_SIZE ); + return( -1 ); + } + + frame = &cpp->conditional_stack[cpp->conditional_depth++]; + frame->filename = filename; + frame->line_number = line_number; + frame->parent_active = parent_active; + frame->active = parent_active && condition; + frame->branch_succeeded = frame->active; + frame->seen_else = 0; + + return( 0 ); +} + +static int +conditional_identifier( mcpp *cpp, const char *filename, + unsigned line_number, + const __mpu_char16_t *line, size_t length, + size_t p, int negate, int parent_active ) +{ + size_t start; + size_t end; + int defined; + + if( !parent_active ) + return( conditional_push(cpp, filename, line_number, 0, 0) ); + + skip_space16( line, length, &p ); + if( p >= length || !mcpu_pp_is_identifier_start(line[p]) ) + { + fprintf( stderr, "%s:%u: error: identifier expected after #%s\n", + filename, line_number, negate ? "ifndef" : "ifdef" ); + return( -1 ); + } + + start = p++; + while( p < length && mcpu_pp_is_identifier_char(line[p]) ) + ++p; + end = p; + + skip_space16( line, length, &p ); + if( p < length && line[p] != '\n' ) + { + fprintf( stderr, "%s:%u: error: extra tokens after #%s\n", + filename, line_number, negate ? "ifndef" : "ifdef" ); + return( -1 ); + } + + defined = mcpu_macro_find( &cpp->macros, line + start, end - start ) != NULL; + if( negate ) defined = !defined; + + return( conditional_push(cpp, filename, line_number, 1, defined) ); +} + +static int +conditional_if( mcpp *cpp, const char *filename, unsigned line_number, + const __mpu_char16_t *line, size_t length, size_t p, + int parent_active ) +{ + mcpu_macro_expansion_context context; + int value = 0; + + if( parent_active ) + { + macro_context( cpp, filename, line_number, &context ); + if( mcpp_eval_if_expression(&cpp->macros, line + p, length - p, + current_language(cpp), &context, + filename, line_number, &value) != 0 ) + return( -1 ); + } + + return( conditional_push(cpp, filename, line_number, + parent_active, value) ); +} + +static int +conditional_elif( mcpp *cpp, const char *filename, unsigned line_number, + const __mpu_char16_t *line, size_t length, size_t p, + size_t conditional_base ) +{ + mcpp_conditional *frame; + mcpu_macro_expansion_context context; + int value = 0; + + if( cpp->conditional_depth <= conditional_base ) + { + fprintf( stderr, "%s:%u: error: '#elif' not within a conditional\n", + filename, line_number ); + return( -1 ); + } + + frame = &cpp->conditional_stack[cpp->conditional_depth - 1]; + if( frame->seen_else ) + { + fprintf( stderr, "%s:%u: error: '#elif' after '#else'\n", + filename, line_number ); + return( -1 ); + } + + if( !frame->parent_active || frame->branch_succeeded ) + { + frame->active = 0; + return( 0 ); + } + + macro_context( cpp, filename, line_number, &context ); + if( mcpp_eval_if_expression(&cpp->macros, line + p, length - p, + current_language(cpp), &context, + filename, line_number, &value) != 0 ) + return( -1 ); + + frame->active = value != 0; + if( frame->active ) + frame->branch_succeeded = 1; + + return( 0 ); +} + +static int +conditional_else( mcpp *cpp, const char *filename, unsigned line_number, + size_t conditional_base ) +{ + mcpp_conditional *frame; + + if( cpp->conditional_depth <= conditional_base ) + { + fprintf( stderr, "%s:%u: error: '#else' not within a conditional\n", + filename, line_number ); + return( -1 ); + } + + frame = &cpp->conditional_stack[cpp->conditional_depth - 1]; + if( frame->seen_else ) + { + fprintf( stderr, "%s:%u: error: '#else' after '#else'\n", + filename, line_number ); + return( -1 ); + } + + frame->seen_else = 1; + frame->active = frame->parent_active && !frame->branch_succeeded; + if( frame->active ) + frame->branch_succeeded = 1; + + return( 0 ); +} + +static int +conditional_endif( mcpp *cpp, const char *filename, unsigned line_number, + size_t conditional_base ) +{ + if( cpp->conditional_depth <= conditional_base ) + { + fprintf( stderr, "%s:%u: error: unbalanced '#endif'\n", + filename, line_number ); + return( -1 ); + } + + --cpp->conditional_depth; + return( 0 ); +} + +static int +handle_conditional( mcpp *cpp, enum mcpp_conditional_directive directive, + const char *filename, unsigned line_number, + const __mpu_char16_t *line, size_t length, size_t p, + size_t conditional_base ) +{ + int parent_active = conditional_active( cpp ); + + switch( directive ) + { + case MCPP_COND_IF: + return( conditional_if(cpp, filename, line_number, + line, length, p, parent_active) ); + + case MCPP_COND_IFDEF: + return( conditional_identifier(cpp, filename, line_number, + line, length, p, 0, parent_active) ); + + case MCPP_COND_IFNDEF: + return( conditional_identifier(cpp, filename, line_number, + line, length, p, 1, parent_active) ); + + case MCPP_COND_ELIF: + return( conditional_elif(cpp, filename, line_number, + line, length, p, conditional_base) ); + + case MCPP_COND_ELSE: + return( conditional_else(cpp, filename, line_number, + conditional_base) ); + + case MCPP_COND_ENDIF: + return( conditional_endif(cpp, filename, line_number, + conditional_base) ); + + default: + return( 0 ); + } +} + +static int +parse_line_filename( const __mpu_char16_t *text, size_t length, + size_t *pos, char **filename ) +{ + mcpu_text decoded; + size_t p = *pos; + + *filename = NULL; + if( p >= length || text[p] != '"' ) + return( 0 ); + + ++p; + mcpu_text_init( &decoded ); + + while( p < length && text[p] != '\n' ) + { + __mpu_char16_t c = text[p++]; + + if( c == '"' ) + { + *filename = mcpu_text_to_utf8( decoded.data, decoded.length ); + mcpu_text_free( &decoded ); + if( *filename == NULL ) + return( -1 ); + *pos = p; + return( 1 ); + } + + if( c == '\\' ) + { + unsigned value; + int digit; + unsigned count; + + if( p >= length || text[p] == '\n' ) + goto bad; + + c = text[p++]; + switch( c ) + { + case 'a': c = '\a'; break; + case 'b': c = '\b'; break; + case 'f': c = '\f'; break; + case 'n': c = '\n'; break; + case 'r': c = '\r'; break; + case 't': c = '\t'; break; + case 'v': c = '\v'; break; + case '\\': c = '\\'; break; + case '\'': c = '\''; break; + case '"': c = '"'; break; + case '?': c = '?'; break; + + case 'x': + case 'X': + value = 0; + count = 0; + while( p < length ) + { + __mpu_char16_t h = text[p]; + + if( h >= '0' && h <= '9' ) digit = h - '0'; + else if( h >= 'a' && h <= 'f' ) digit = h - 'a' + 10; + else if( h >= 'A' && h <= 'F' ) digit = h - 'A' + 10; + else break; + + if( value > 0xffffU / 16U ) + goto bad; + value = value * 16U + (unsigned)digit; + if( value > 0xffffU ) + goto bad; + ++p; + ++count; + } + if( count == 0 || value == 0 || + (value >= 0xd800U && value <= 0xdfffU) ) + goto bad; + c = (__mpu_char16_t)value; + break; + + default: + if( c >= '0' && c <= '7' ) + { + value = c - '0'; + for( count = 1; count < 3 && p < length; ++count ) + { + c = text[p]; + if( c < '0' || c > '7' ) break; + value = value * 8U + (unsigned)(c - '0'); + ++p; + } + if( value == 0 || value > 0xffffU || + (value >= 0xd800U && value <= 0xdfffU) ) + goto bad; + c = (__mpu_char16_t)value; + } + break; + } + } + + if( mcpu_text_append_char(&decoded, c) != 0 ) + goto fail; + } + +bad: + errno = EINVAL; +fail: + mcpu_text_free( &decoded ); + return( -1 ); +} + +static int +handle_line( mcpp *cpp, const char *filename, unsigned line_number, + unsigned next_physical_line, + const __mpu_char16_t *line, size_t length, size_t p, + char **nominal_filename, int64_t *line_delta ) +{ + mcpu_macro_expansion_context context; + mcpu_text expanded; + const __mpu_char16_t *text; + size_t text_length; + size_t end; + uintmax_t value = 0; + char *new_filename = NULL; + int parsed_filename; + + macro_context( cpp, *nominal_filename, line_number, &context ); + mcpu_text_init( &expanded ); + + if( mcpu_macro_expand(&cpp->macros, line + p, length - p, + current_language(cpp), &context, + &expanded) != 0 ) + goto fail; + + text = expanded.data; + text_length = expanded.length; + end = line_content_end( text, text_length ); + p = 0; + skip_space16( text, end, &p ); + + if( p >= end || text[p] < '0' || text[p] > '9' ) + { + fprintf( stderr, "%s:%u: error: invalid #line directive\n", + filename, line_number ); + goto fail; + } + + while( p < end && text[p] >= '0' && text[p] <= '9' ) + { + unsigned digit = text[p++] - '0'; + + if( value > (UINT_MAX - digit) / 10U ) + { + fprintf( stderr, "%s:%u: error: line number out of range in #line directive\n", + filename, line_number ); + goto fail; + } + value = value * 10U + digit; + } + + skip_space16( text, end, &p ); + parsed_filename = parse_line_filename( text, end, &p, &new_filename ); + if( parsed_filename < 0 ) + { + fprintf( stderr, "%s:%u: error: invalid filename in #line directive\n", + filename, line_number ); + goto fail; + } + + skip_space16( text, end, &p ); + if( p != end ) + { + fprintf( stderr, "%s:%u: error: garbage at end of #line directive\n", + filename, line_number ); + goto fail; + } + + if( parsed_filename > 0 ) + { + free( *nominal_filename ); + *nominal_filename = new_filename; + new_filename = NULL; + } + + *line_delta = (int64_t)value - (int64_t)next_physical_line; + + if( append_line_marker(cpp, (unsigned)value, *nominal_filename, 0) != 0 ) + goto fail; + + mcpu_text_free( &expanded ); + return( 0 ); + +fail: + free( new_filename ); + mcpu_text_free( &expanded ); + return( -1 ); +} + +static int process_file( mcpp *cpp, const char *filename, + const mcpu_include_dir *include_dir, + int system_header ); + +static int +pragma_once_directive( const __mpu_char16_t *line, size_t length, size_t p ) +{ + size_t end = line_content_end( line, length ); + + skip_space16( line, end, &p ); + if( p + 4 > end || + !mcpu_text_equal_ascii(line + p, 4, "once") ) + return( 0 ); + + p += 4; + skip_space16( line, end, &p ); + return( p == end ); +} + +static int +handle_include( mcpp *cpp, const char *source_filename, + const char *nominal_filename, + unsigned line_number, unsigned resume_line_number, + const __mpu_char16_t *line, size_t length, size_t p, + int include_next, const mcpu_include_dir *current_dir, + int source_is_system ) +{ + char *requested = NULL; + char *found = NULL; + const mcpu_include_dir *found_dir = NULL; + int quoted; + int rc = -1; + mcpu_macro_expansion_context context; + + macro_context( cpp, nominal_filename, line_number, &context ); + + if( parse_include_filename( line, length, p, &requested, "ed ) != 0 ) + { + mcpu_text expanded; + + mcpu_text_init( &expanded ); + if( mcpu_macro_expand(&cpp->macros, line + p, length - p, + current_language(cpp), &context, + &expanded) != 0 || + parse_include_filename(expanded.data, expanded.length, 0, + &requested, "ed) != 0 ) + { + fprintf( stderr, + "%s:%u: error: #%s does not expand to \"file\" or <file>\n", + nominal_filename, line_number, + include_next ? "include_next" : "include" ); + mcpu_text_free( &expanded ); + goto done; + } + mcpu_text_free( &expanded ); + } + + found = mcpu_include_find( &cpp->include_paths, + source_filename, + requested, + quoted, + current_language(cpp), + include_next, + include_next ? current_dir : NULL, + &found_dir ); + if( found == NULL ) + { + if( MCPP_OPTIONS(cpp) != NULL && + MCPP_OPTIONS(cpp)->print_deps_missing_files ) + { + int dependency_is_system = source_is_system || !quoted; + + if( dependency_add_unresolved(cpp, requested, + dependency_is_system) != 0 ) + goto done; + + if( cpp->verbose ) + fprintf( stderr, "%s:%u: %s %s -> <generated> [%s]\n", + nominal_filename, line_number, + include_next ? "include_next" : "include", + requested, mcpu_language_name(current_language(cpp)) ); + + rc = 0; + goto done; + } + + fprintf( stderr, "%s:%u: error: cannot find %s file '%s'\n", + nominal_filename, line_number, + include_next ? "include_next" : "include", requested ); + goto done; + } + + if( cpp->verbose ) + fprintf( stderr, "%s:%u: %s %s -> %s [%s]\n", + nominal_filename, line_number, + include_next ? "include_next" : "include", + requested, found, mcpu_language_name(current_language(cpp)) ); + + { + int dependency_is_system = source_is_system || include_dir_is_system(found_dir); + + if( dependency_add(cpp, found, dependency_is_system) != 0 ) + goto done; + + if( once_file_seen(cpp, found) ) + { + rc = 0; + goto done; + } + + if( sync_output_position(cpp, line_number, nominal_filename) != 0 || + append_line_marker(cpp, 1, found, 1) != 0 ) + goto done; + if( process_file( cpp, found, found_dir, dependency_is_system ) != 0 ) + goto done; + } + if( append_line_marker( cpp, resume_line_number, nominal_filename, 2 ) != 0 ) + goto done; + + rc = 0; + +done: + free( requested ); + free( found ); + return( rc ); +} + +static int +process_source( mcpp *cpp, mcpu_source *source, + const mcpu_include_dir *include_dir, + int source_is_system ) +{ + mcpu_pp_lexer_state lexer; + mcpu_text prepared; + mcpu_text expanded; + const __mpu_char16_t *line; + size_t length; + unsigned line_number; + unsigned next_line_number; + int spliced; + size_t conditional_base; + int status; + mcpu_macro_expansion_context context; + const char *filename = source->filename; + char *nominal_filename; + int64_t line_delta = 0; + + if( cpp->include_depth >= MCPU_CPP_INCLUDE_STACK_SIZE ) + { + fprintf( stderr, "%s: error: include nesting exceeds %d files\n", + filename, MCPU_CPP_INCLUDE_STACK_SIZE ); + return( -1 ); + } + + nominal_filename = strdup( filename ); + if( nominal_filename == NULL ) + return( -1 ); + + ++cpp->include_depth; + conditional_base = cpp->conditional_depth; + mcpu_pp_lexer_init( &lexer ); + mcpu_text_init( &prepared ); + mcpu_text_init( &expanded ); + + while( (status = mcpu_source_next_line( source, &line, &length, + &line_number, + &next_line_number, &spliced )) > 0 ) + { + int64_t reported = (int64_t)line_number + line_delta; + int64_t reported_next = (int64_t)next_line_number + line_delta; + unsigned logical_line; + unsigned logical_next_line; + size_t hash = 0; + int directive; + + if( reported < 0 || reported > UINT_MAX || + reported_next < 0 || reported_next > UINT_MAX ) + { + fprintf( stderr, "%s:%u: error: logical line number out of range\n", + filename, line_number ); + goto fail; + } + + logical_line = (unsigned)reported; + logical_next_line = (unsigned)reported_next; + + mcpu_pp_lexer_set_diagnostics(&lexer, MCPP_OPTIONS(cpp), + nominal_filename, logical_line, + source->splice_offsets, + source->splice_count); + + if( mcpu_pp_prepare_line(&lexer, line, length, + current_language(cpp), &prepared) != 0 ) + goto fail; + + line = prepared.data; + length = prepared.length; + + if( !mcpu_source_valid_ucs2(line, length) ) + { + fprintf( stderr, "%s:%u: error: character outside UCS-2\n", + nominal_filename, logical_line ); + goto fail; + } + + directive = mcpu_pp_find_directive( line, length, &hash ); + if( directive < 0 ) + goto fail; + + if( directive ) + { + size_t name_start; + size_t name_length; + size_t after_name; + + if( parse_directive_name( line, length, hash, + &name_start, &name_length, + &after_name ) ) + { + enum mcpp_conditional_directive conditional; + + conditional = conditional_directive( line + name_start, name_length ); + if( conditional != MCPP_COND_NONE ) + { + if( handle_conditional(cpp, conditional, source->filename, + line_number, line, length, after_name, + conditional_base) != 0 ) + goto fail; + continue; + } + + if( !conditional_active(cpp) ) + continue; + + { + int diagnostic_error; + const char *diagnostic_name; + + if( diagnostic_directive(line + name_start, name_length, + &diagnostic_error, &diagnostic_name) ) + { + if( handle_diagnostic(cpp, nominal_filename, logical_line, + line, length, after_name, + diagnostic_error, diagnostic_name) != 0 ) + goto fail; + + continue; + } + } + + if( mcpu_text_equal_ascii(line + name_start, name_length, "line") ) + { + if( handle_line( cpp, nominal_filename, logical_line, + next_line_number, line, length, after_name, + &nominal_filename, &line_delta ) != 0 ) + goto fail; + continue; + } + + if( mcpu_text_equal_ascii(line + name_start, name_length, "pragma") ) + { + if( pragma_once_directive(line, length, after_name) ) + { + if( once_file_mark(cpp, source->filename) != 0 ) + { + fprintf( stderr, + "%s:%u: error: cannot record #pragma once: %s\n", + nominal_filename, logical_line, strerror(errno) ); + goto fail; + } + + continue; + } + } + + if( mcpu_text_equal_ascii(line + name_start, name_length, "include") || + mcpu_text_equal_ascii(line + name_start, name_length, + "include_next") ) + { + int include_next = + mcpu_text_equal_ascii( line + name_start, name_length, + "include_next" ); + + if( handle_include( cpp, source->filename, nominal_filename, + logical_line, logical_next_line, + line, length, after_name, include_next, + include_dir, source_is_system ) != 0 ) + goto fail; + continue; + } + + if( mcpu_text_equal_ascii(line + name_start, name_length, "define") ) + { + if( handle_define( cpp, nominal_filename, logical_line, + line, length, after_name, + MCPU_MACRO_ORIGIN_SOURCE ) != 0 ) + goto fail; + + if( MCPP_OPTIONS(cpp) != NULL && + MCPP_OPTIONS(cpp)->dump_macros == MCPP_DUMP_DEFINITIONS && + append_visible_text(cpp, logical_line, nominal_filename, + line, length) != 0 ) + goto fail; + + continue; + } + + if( mcpu_text_equal_ascii(line + name_start, name_length, "undef") ) + { + if( handle_undef( cpp, nominal_filename, logical_line, + line, length, after_name ) != 0 ) + goto fail; + continue; + } + + if( mcpu_text_equal_ascii(line + name_start, name_length, "lang") ) + { + { + size_t output_start; + + if( sync_output_position(cpp, logical_line, nominal_filename) != 0 ) + goto fail; + output_start = cpp->output.length; + if( handle_lang(cpp, nominal_filename, logical_line, + line, length, after_name, spliced) != 0 ) + goto fail; + advance_output_position( cpp, cpp->output.data + output_start, + cpp->output.length - output_start ); + } + continue; + } + + if( mcpu_text_equal_ascii( line + name_start, + name_length, "endlang" ) ) + { + { + size_t output_start; + + if( sync_output_position(cpp, logical_line, nominal_filename) != 0 ) + goto fail; + output_start = cpp->output.length; + if( handle_endlang(cpp, nominal_filename, logical_line, + line, length) != 0 ) + goto fail; + advance_output_position( cpp, cpp->output.data + output_start, + cpp->output.length - output_start ); + } + continue; + } + } + } + + if( !conditional_active(cpp) ) + continue; + + macro_context( cpp, nominal_filename, logical_line, &context ); + if( mcpu_macro_expand(&cpp->macros, line, length, + current_language(cpp), &context, + &expanded) != 0 || + append_visible_text(cpp, logical_line, nominal_filename, + expanded.data, expanded.length) != 0 ) + goto fail; + } + + if( status < 0 ) + goto fail; + + if( lexer.in_block_comment ) + { + fprintf( stderr, "%s: error: unterminated comment at end of file\n", + nominal_filename ); + goto fail; + } + + if( cpp->conditional_depth != conditional_base ) + { + mcpp_conditional *frame = + &cpp->conditional_stack[cpp->conditional_depth - 1]; + + fprintf( stderr, "%s:%u: error: unterminated conditional directive\n", + frame->filename, frame->line_number ); + goto fail; + } + + --cpp->include_depth; + mcpu_text_free( &prepared ); + mcpu_text_free( &expanded ); + free( nominal_filename ); + return( 0 ); + +fail: + cpp->conditional_depth = conditional_base; + --cpp->include_depth; + mcpu_text_free( &prepared ); + mcpu_text_free( &expanded ); + free( nominal_filename ); + return( -1 ); +} + +static int +process_file( mcpp *cpp, const char *filename, + const mcpu_include_dir *include_dir, + int system_header ) +{ + mcpu_source source; + int rc; + + mcpu_source_init( &source ); + if( mcpu_source_open( &source, filename ) != 0 ) + { + fprintf( stderr, "%s: %s\n", filename, strerror(errno) ); + mcpu_source_free( &source ); + return( -1 ); + } + + rc = process_source( cpp, &source, include_dir, system_header ); + mcpu_source_free( &source ); + return( rc ); +} + +static int +process_stream( mcpp *cpp, FILE *stream, const char *name ) +{ + mcpu_source source; + int rc; + + mcpu_source_init( &source ); + if( mcpu_source_open_stream( &source, stream, name ) != 0 ) + { + fprintf( stderr, "%s: %s\n", name, strerror(errno) ); + mcpu_source_free( &source ); + return( -1 ); + } + + rc = process_source( cpp, &source, NULL, 0 ); + mcpu_source_free( &source ); + return( rc ); +} + +void +mcpp_init( mcpp *cpp ) +{ + if( cpp == NULL ) + return; + + cpp->options_data = NULL; + mcpu_config_init( &cpp->config ); + mcpp_runtime_paths_init( &cpp->runtime_paths ); + mcpu_include_paths_init( &cpp->include_paths ); + cpp->lang_depth = 1; + cpp->lang_stack[0] = MCPU_LANG_0; + cpp->include_depth = 0; + cpp->conditional_depth = 0; + cpp->base_filename = NULL; + cpp->preprocess_date[0] = 0; + cpp->preprocess_time[0] = 0; + cpp->builtins_initialized = 0; + cpp->command_line_macros_applied = 0; + mcpu_macro_table_init( &cpp->macros ); + cpp->once_files = NULL; + cpp->dependencies_first = NULL; + cpp->dependencies_last = NULL; + cpp->verbose = 0; + cpp->output_filename = NULL; + cpp->output_line = 0; + mcpu_text_init( &cpp->output ); +} + +void +mcpp_free( mcpp *cpp ) +{ + if( cpp == NULL ) + return; + + mcpu_config_free( &cpp->config ); + mcpp_runtime_paths_free( &cpp->runtime_paths ); + mcpu_include_paths_free( &cpp->include_paths ); + free( cpp->base_filename ); + cpp->base_filename = NULL; + mcpu_macro_table_free( &cpp->macros ); + once_files_free( cpp ); + dependencies_free( cpp ); + free( cpp->output_filename ); + cpp->output_filename = NULL; + cpp->output_line = 0; + mcpu_text_free( &cpp->output ); +} + +int +mcpp_load_config( mcpp *cpp, const char *explicit_filename, + int no_config ) +{ + const char *home; + + if( cpp == NULL ) + return( -1 ); + + /* + * The physical MCPU runtime root is not a configuration-file value. The + * sibling <root>/include directory is the lowest-priority default and + * therefore remains active even when --no-config suppresses all + * configuration-file reads. A subsequently read configuration file may + * replace it, including with an empty value that disables the configured + * system include tree. + */ + if( mcpu_config_set( &cpp->config, + "MCPU_CPP_SYSTEM_INCLUDE_PATH", + cpp->runtime_paths.system_include_path ) != 0 ) + return( -1 ); + + if( no_config ) + return( 0 ); + + if( explicit_filename ) + { + if( mcpu_config_read( &cpp->config, explicit_filename, 1 ) != 0 ) + { + fprintf( stderr, "%s: cannot read configuration: %s\n", + explicit_filename, strerror(errno) ); + return( -1 ); + } + } + else + { + char *user = NULL; + + if( cpp->runtime_paths.config_file == NULL || + mcpu_config_read( &cpp->config, + cpp->runtime_paths.config_file, 0 ) != 0 ) + return( -1 ); + + if( mcpu_config_read( &cpp->config, + MCPU_CPP_SYSTEM_CONFIG_FILE, 0 ) != 0 ) + return( -1 ); + + home = getenv( "HOME" ); + if( home && *home ) + { + size_t n = strlen(home) + sizeof("/.mcpu/mcpu-cpp.conf"); + user = (char *)malloc( n ); + if( user == NULL ) + return( -1 ); + snprintf( user, n, "%s/.mcpu/mcpu-cpp.conf", home ); + if( mcpu_config_read( &cpp->config, user, 0 ) != 0 ) + { + free( user ); + return( -1 ); + } + free( user ); + } + } + + return( 0 ); +} + + +int +mcpp_prepare( mcpp *cpp ) +{ + if( cpp == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + if( cpp->preprocess_date[0] == 0 && initialize_timestamp(cpp) != 0 ) + return( -1 ); + + if( initialize_builtins(cpp) != 0 ) + return( -1 ); + + return( apply_command_line_macros(cpp) ); +} + + +static void +restore_output_length( mcpp *cpp, size_t length ) +{ + if( cpp == NULL ) + return; + + cpp->output.length = length; + if( cpp->output.data != NULL ) + cpp->output.data[length] = 0; +} + + +static int +process_forced_file( mcpp *cpp, const mcpp_option_action *action, + int discard_output ) +{ + const mcpu_include_dir *found_dir = NULL; + char *found; + char *saved_output_filename = NULL; + size_t output_length; + unsigned saved_output_line = 0; + int system_header; + int rc = -1; + + if( cpp == NULL || action == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + /* + * GNU-style command-line forced files are resolved relative to the current + * working directory first. Passing "." as the synthetic containing source + * gives mcpu_include_find() exactly that quoted-include first step, followed + * by the ordinary configured include chain. The primary input directory is + * deliberately not involved. + */ + found = mcpu_include_find( &cpp->include_paths, ".", action->argument, 1, + current_language(cpp), 0, NULL, &found_dir ); + if( found == NULL ) + { + const char *progname = MCPP_OPTIONS(cpp) != NULL ? + MCPP_OPTIONS(cpp)->progname : "mcpu-cpp"; + + if( MCPP_OPTIONS(cpp) != NULL && + MCPP_OPTIONS(cpp)->print_deps_missing_files ) + { + /* + * GNU -MG also tolerates missing command-line -include/-imacros files. + * With no successful search there is no physical search provenance, so + * the command-line operand enters the unresolved registry as a user + * dependency exactly as written. + */ + if( dependency_add_unresolved(cpp, action->argument, 0) != 0 ) + return( -1 ); + + if( cpp->verbose ) + fprintf( stderr, "%s %s -> <generated> [%s]\n", + action->option, action->argument, + mcpu_language_name(current_language(cpp)) ); + return( 0 ); + } + + fprintf( stderr, "%s: error: %s cannot find file '%s'\n", + progname, action->option, action->argument ); + return( -1 ); + } + + system_header = include_dir_is_system( found_dir ); + if( dependency_add(cpp, found, system_header) != 0 ) + goto done; + + if( once_file_seen(cpp, found) ) + { + rc = 0; + goto done; + } + + output_length = cpp->output.length; + if( discard_output && cpp->output_filename != NULL ) + { + saved_output_filename = strdup( cpp->output_filename ); + if( saved_output_filename == NULL ) + goto done; + saved_output_line = cpp->output_line; + } + + if( cpp->verbose ) + fprintf( stderr, "%s %s -> %s [%s]\n", + action->option, action->argument, found, + mcpu_language_name(current_language(cpp)) ); + + if( append_line_marker(cpp, 1, found, 1) != 0 ) + goto forced_fail; + + /* + * A forced file is conceptually included by the primary source before its + * first line. Reserve the primary-source frame so __INCLUDE_LEVEL__ is 1 + * in the forced file and increases normally in headers included from it. + */ + if( cpp->include_depth >= MCPU_CPP_INCLUDE_STACK_SIZE ) + { + fprintf( stderr, "%s: error: include nesting exceeds %d files\n", + found, MCPU_CPP_INCLUDE_STACK_SIZE ); + goto forced_fail; + } + + ++cpp->include_depth; + rc = process_file( cpp, found, found_dir, system_header ); + --cpp->include_depth; + if( rc != 0 ) + goto forced_fail; + + if( discard_output ) + { + restore_output_length( cpp, output_length ); + if( saved_output_filename != NULL && + set_output_position(cpp, saved_output_line, saved_output_filename) != 0 ) + goto done; + } + else if( append_line_marker(cpp, 1, cpp->base_filename, 2) != 0 ) + { + rc = -1; + goto done; + } + + rc = 0; + goto done; + +forced_fail: + if( discard_output ) + { + restore_output_length( cpp, output_length ); + if( saved_output_filename != NULL ) + (void)set_output_position(cpp, saved_output_line, saved_output_filename); + } + rc = -1; + +done: + free( saved_output_filename ); + free( found ); + return( rc ); +} + + +static int +process_forced_files( mcpp *cpp ) +{ + size_t i; + int pass; + + if( cpp == NULL || MCPP_OPTIONS(cpp) == NULL ) + return( 0 ); + + /* + * Command-line -D/-U actions have already been applied by mcpp_prepare(). + * Process every -imacros file before every -include file while preserving + * command-line order inside each class. + */ + for( pass = 0; pass < 2; ++pass ) + { + enum mcpp_option_action_kind kind = + pass == 0 ? MCPP_ACTION_IMACROS : MCPP_ACTION_INCLUDE; + + for( i = 0; i < MCPP_OPTIONS(cpp)->action_count; ++i ) + { + const mcpp_option_action *action = &MCPP_OPTIONS(cpp)->actions[i]; + + if( action->kind != kind ) + continue; + + if( process_forced_file(cpp, action, kind == MCPP_ACTION_IMACROS) != 0 ) + return( -1 ); + } + } + + return( 0 ); +} + + +int +mcpp_dump_macros( mcpp *cpp, FILE *stream ) +{ + if( cpp == NULL || stream == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + if( mcpp_prepare(cpp) != 0 ) + return( -1 ); + + return( mcpu_macro_dump(&cpp->macros, stream, + MCPP_OPTIONS(cpp) != NULL && + MCPP_OPTIONS(cpp)->dump_only_plus_predefined) ); +} + + +int +mcpp_process( mcpp *cpp, const char *filename ) +{ + if( cpp == NULL || filename == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + free( cpp->base_filename ); + cpp->base_filename = strdup( filename ); + if( cpp->base_filename == NULL || mcpp_prepare(cpp) != 0 ) + return( -1 ); + + if( MCPP_OPTIONS(cpp) != NULL && + MCPP_OPTIONS(cpp)->dump_macros == MCPP_DUMP_DEFINITIONS ) + { + const char *marker = "<built-in>"; + + if( append_line_marker(cpp, 0, filename, 0) != 0 || + mcpu_macro_dump_text(&cpp->macros, &cpp->output, marker) != 0 ) + return( -1 ); + } + + if( dependency_add(cpp, filename, 0) != 0 ) + return( -1 ); + + if( append_line_marker(cpp, 1, filename, 0) != 0 ) + return( -1 ); + + if( process_forced_files(cpp) != 0 ) + return( -1 ); + + if( process_file( cpp, filename, NULL, 0 ) != 0 ) + return( -1 ); + + if( cpp->lang_depth != 1 ) + { + fprintf( stderr, + "%s: error: unbalanced #lang/#endlang at end of translation unit (depth %lu)\n", + filename, (unsigned long)cpp->lang_depth ); + return( -1 ); + } + + return( 0 ); +} + + +int +mcpp_process_stream( mcpp *cpp, FILE *stream, const char *name ) +{ + if( cpp == NULL || stream == NULL || name == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + free( cpp->base_filename ); + cpp->base_filename = strdup( name ); + if( cpp->base_filename == NULL || mcpp_prepare(cpp) != 0 ) + return( -1 ); + + if( MCPP_OPTIONS(cpp) != NULL && + MCPP_OPTIONS(cpp)->dump_macros == MCPP_DUMP_DEFINITIONS ) + { + const char *marker = "<built-in>"; + + if( append_line_marker(cpp, 0, name, 0) != 0 || + mcpu_macro_dump_text(&cpp->macros, &cpp->output, marker) != 0 ) + return( -1 ); + } + + if( dependency_add(cpp, strcmp(name, "<stdin>") == 0 ? "-" : name, 0) != 0 ) + return( -1 ); + + if( append_line_marker(cpp, 1, name, 0) != 0 ) + return( -1 ); + + if( process_forced_files(cpp) != 0 ) + return( -1 ); + + if( process_stream(cpp, stream, name) != 0 ) + return( -1 ); + + if( cpp->lang_depth != 1 ) + { + fprintf( stderr, + "%s: error: unbalanced #lang/#endlang at end of translation unit (depth %lu)\n", + name, (unsigned long)cpp->lang_depth ); + return( -1 ); + } + + return( 0 ); +} + +static int +write_make_escaped( FILE *stream, const char *text ) +{ + const unsigned char *p = (const unsigned char *)text; + + while( *p ) + { + if( *p == '$' ) + { + if( fputs("$$", stream) == EOF ) + return( -1 ); + } + else + { + if( (*p == ' ' || *p == '\t' || *p == '#' || *p == '\\') && + fputc('\\', stream) == EOF ) + return( -1 ); + if( fputc(*p, stream) == EOF ) + return( -1 ); + } + ++p; + } + + return( 0 ); +} + + +static int +write_make_target_quoted( FILE *stream, const char *text ) +{ + const unsigned char *p = (const unsigned char *)text; + size_t slashes = 0; + + while( *p ) + { + unsigned char c = *p++; + + if( c == '\\' ) + { + if( fputc(c, stream) == EOF ) + return( -1 ); + ++slashes; + continue; + } + + if( c == ' ' || c == '\t' ) + { + size_t i; + + for( i = 0; i < slashes; ++i ) + if( fputc('\\', stream) == EOF ) + return( -1 ); + if( fputc('\\', stream) == EOF ) + return( -1 ); + } + else if( c == '#' ) + { + if( fputc('\\', stream) == EOF ) + return( -1 ); + } + else if( c == '$' ) + { + if( fputc('$', stream) == EOF ) + return( -1 ); + } + + slashes = 0; + if( fputc(c, stream) == EOF ) + return( -1 ); + } + + return( 0 ); +} + + +static char * +default_dependency_target( const mcpp *cpp ) +{ + const char *base; + const char *dot; + const char *suffix = ".o"; + size_t stem_length; + size_t suffix_length; + char *target; + + if( cpp == NULL || cpp->base_filename == NULL ) + return( NULL ); + + if( strcmp(cpp->base_filename, "<stdin>") == 0 || + strcmp(cpp->base_filename, "-") == 0 ) + return( strdup("-") ); + + base = strrchr( cpp->base_filename, '/' ); + base = base == NULL ? cpp->base_filename : base + 1; + dot = strrchr( base, '.' ); + stem_length = dot != NULL && dot != base ? (size_t)(dot - base) : strlen(base); + + if( MCPP_OPTIONS(cpp) != NULL && MCPP_OPTIONS(cpp)->object_suffix != NULL ) + suffix = MCPP_OPTIONS(cpp)->object_suffix; + suffix_length = strlen( suffix ); + + target = (char *)malloc( stem_length + suffix_length + 1 ); + if( target == NULL ) + return( NULL ); + + memcpy( target, base, stem_length ); + memcpy( target + stem_length, suffix, suffix_length + 1 ); + return( target ); +} + + +static int +write_dependency_targets( mcpp *cpp, FILE *stream ) +{ + char *target; + size_t i; + int pass; + int written = 0; + + if( cpp == NULL || stream == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + if( MCPP_OPTIONS(cpp) != NULL ) + for( pass = 0; pass < 2; ++pass ) + { + enum mcpp_option_action_kind kind = pass == 0 ? + MCPP_ACTION_DEP_TARGET : MCPP_ACTION_DEP_TARGET_QUOTED; + + for( i = 0; i < MCPP_OPTIONS(cpp)->action_count; ++i ) + { + const mcpp_option_action *action = &MCPP_OPTIONS(cpp)->actions[i]; + + if( action->kind != kind ) + continue; + + if( written && fputc(' ', stream) == EOF ) + return( -1 ); + + if( kind == MCPP_ACTION_DEP_TARGET ) + { + if( fputs(action->argument, stream) == EOF ) + return( -1 ); + } + else if( write_make_target_quoted(stream, action->argument) != 0 ) + return( -1 ); + + written = 1; + } + } + + if( written ) + return( 0 ); + + target = default_dependency_target( cpp ); + if( target == NULL ) + return( -1 ); + + if( write_make_target_quoted(stream, target) != 0 ) + { + free( target ); + return( -1 ); + } + + free( target ); + return( 0 ); +} + + +int +mcpp_write_dependencies( mcpp *cpp, FILE *stream, int include_system ) +{ + mcpp_dependency *p; + + if( cpp == NULL || stream == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + if( write_dependency_targets(cpp, stream) != 0 || fputs(":", stream) == EOF ) + return( -1 ); + + for( p = cpp->dependencies_first; p != NULL; p = p->next ) + { + if( !include_system && p->system_only ) + continue; + + if( fputc(' ', stream) == EOF || write_make_escaped(stream, p->path) != 0 ) + return( -1 ); + } + + if( fputc('\n', stream) == EOF ) + return( -1 ); + + return( ferror(stream) ? -1 : 0 ); +} + + +int +mcpp_write_output( mcpp *cpp, const char *filename ) +{ + char *utf8; + FILE *fp = stdout; + int close_fp = 0; + size_t n; + + if( cpp == NULL ) + return( -1 ); + + utf8 = mcpu_text_to_utf8( cpp->output.data, cpp->output.length ); + if( utf8 == NULL ) + return( -1 ); + + if( filename ) + { + fp = fopen( filename, "wb" ); + if( fp == NULL ) + { + free( utf8 ); + return( -1 ); + } + close_fp = 1; + } + + n = strlen( utf8 ); + if( fwrite( utf8, 1, n, fp ) != n ) + { + if( close_fp ) + fclose( fp ); + free( utf8 ); + return( -1 ); + } + + if( close_fp && fclose(fp) != 0 ) + { + free( utf8 ); + return( -1 ); + } + + free( utf8 ); + return( 0 ); +} diff --git a/src/mcpp-lib.h b/src/mcpp-lib.h new file mode 100644 index 0000000..6974672 --- /dev/null +++ b/src/mcpp-lib.h @@ -0,0 +1,65 @@ +#ifndef __MCPU_CPP_CPP_H__ +#define __MCPU_CPP_CPP_H__ 1 + +#include <defs.h> +#include <mcpp-config-file.h> +#include <mcpp-include-path.h> +#include <mcpp-language.h> +#include <mcpp-macro.h> +#include <mcpp-text.h> + +typedef struct mcpp_conditional mcpp_conditional; +struct mcpp_conditional +{ + const char *filename; + unsigned line_number; + int parent_active; + int active; + int branch_succeeded; + int seen_else; +}; + +typedef struct mcpp_once_file mcpp_once_file; +typedef struct mcpp_dependency mcpp_dependency; + +typedef struct mcpp mcpp; +struct mcpp +{ + mcpp_options *options_data; + mcpu_config config; + mcpp_runtime_paths runtime_paths; + mcpu_include_paths include_paths; + enum mcpu_language lang_stack[MCPU_CPP_LANG_STACK_SIZE]; + size_t lang_depth; + size_t include_depth; + mcpp_conditional conditional_stack[MCPU_CPP_CONDITIONAL_STACK_SIZE]; + size_t conditional_depth; + char *base_filename; + char preprocess_date[16]; + char preprocess_time[16]; + int builtins_initialized; + int command_line_macros_applied; + mcpu_macro_table macros; + mcpp_once_file *once_files; + mcpp_dependency *dependencies_first; + mcpp_dependency *dependencies_last; + int verbose; + char *output_filename; + unsigned output_line; + mcpu_text output; +}; + +void mcpp_init( mcpp *cpp ); +void mcpp_free( mcpp *cpp ); +int mcpp_load_config( mcpp *cpp, const char *explicit_filename, + int no_config ); +int mcpp_prepare( mcpp *cpp ); +int mcpp_process( mcpp *cpp, const char *filename ); +int mcpp_process_stream( mcpp *cpp, FILE *stream, const char *name ); +int mcpp_dump_macros( mcpp *cpp, FILE *stream ); +int mcpp_write_output( mcpp *cpp, const char *filename ); +int mcpp_write_dependencies( mcpp *cpp, FILE *stream, int include_system ); + +#define MCPP_OPTIONS(_cpp_) ((_cpp_)->options_data) + +#endif /* __MCPU_CPP_CPP_H__ */ diff --git a/src/mcpp-macro.c b/src/mcpp-macro.c new file mode 100644 index 0000000..1837548 --- /dev/null +++ b/src/mcpp-macro.c @@ -0,0 +1,2878 @@ +#include <defs.h> + + +typedef struct mcpu_macro_arg mcpu_macro_arg; +struct mcpu_macro_arg +{ + const __mpu_char16_t *raw; + size_t raw_length; + mcpu_text expanded; + int expanded_ready; +}; + + +static unsigned +macro_hash( const __mpu_char16_t *name, size_t length ) +{ + unsigned h = 5381U; + size_t i; + + for( i = 0; i < length; ++i ) + h = ((h << 5) + h) ^ (unsigned)name[i]; + + return( h % MCPU_CPP_MACRO_BUCKETS ); +} + + +static __mpu_char16_t * +dup_text( const __mpu_char16_t *text, size_t length ) +{ + __mpu_char16_t *copy; + + copy = (__mpu_char16_t *)calloc( length + 1, sizeof(__mpu_char16_t) ); + if( copy == NULL ) + return( NULL ); + + if( length != 0 ) + memcpy( copy, text, length * sizeof(__mpu_char16_t) ); + + return( copy ); +} + + +static void +free_argnames( __mpu_char16_t **argnames, size_t *argname_lengths, + int nargs ) +{ + int i; + + if( argnames ) + { + for( i = 0; i < nargs; ++i ) + free( argnames[i] ); + } + + free( argnames ); + free( argname_lengths ); +} + + +static void +free_macro_contents( mcpu_macro *macro ) +{ + if( macro == NULL ) + return; + + free( macro->name ); + free_argnames( macro->argnames, macro->argname_lengths, macro->nargs ); + free( macro->replacement ); +} + + +void +mcpu_macro_table_init( mcpu_macro_table *table ) +{ + if( table ) + memset( table, 0, sizeof(*table) ); +} + + +void +mcpu_macro_table_free( mcpu_macro_table *table ) +{ + unsigned i; + + if( table == NULL ) + return; + + for( i = 0; i < MCPU_CPP_MACRO_BUCKETS; ++i ) + { + mcpu_macro *macro = table->bucket[i]; + + while( macro ) + { + mcpu_macro *next = macro->next; + free_macro_contents( macro ); + free( macro ); + macro = next; + } + + table->bucket[i] = NULL; + } +} + + +mcpu_macro * +mcpu_macro_find( mcpu_macro_table *table, + const __mpu_char16_t *name, size_t name_length ) +{ + mcpu_macro *macro; + unsigned h; + + if( table == NULL || name == NULL || name_length == 0 ) + return( NULL ); + + h = macro_hash( name, name_length ); + for( macro = table->bucket[h]; macro; macro = macro->next ) + { + if( macro->name_length == name_length && + memcmp( macro->name, name, + name_length * sizeof(__mpu_char16_t) ) == 0 ) + return( macro ); + } + + return( NULL ); +} + + +static int +copy_argnames( const __mpu_char16_t *const *argnames, + const size_t *argname_lengths, size_t nargs, + __mpu_char16_t ***copy_names, size_t **copy_lengths ) +{ + __mpu_char16_t **names = NULL; + size_t *lengths = NULL; + size_t i; + + *copy_names = NULL; + *copy_lengths = NULL; + + if( nargs == 0 ) + return( 0 ); + + names = (__mpu_char16_t **)calloc( nargs, sizeof(*names) ); + lengths = (size_t *)calloc( nargs, sizeof(*lengths) ); + if( names == NULL || lengths == NULL ) + goto fail; + + for( i = 0; i < nargs; ++i ) + { + names[i] = dup_text( argnames[i], argname_lengths[i] ); + if( names[i] == NULL ) + goto fail; + lengths[i] = argname_lengths[i]; + } + + *copy_names = names; + *copy_lengths = lengths; + return( 0 ); + +fail: + if( names ) + { + for( i = 0; i < nargs; ++i ) + free( names[i] ); + } + free( names ); + free( lengths ); + return( -1 ); +} + + +static int +macro_install( mcpu_macro_table *table, + const __mpu_char16_t *name, size_t name_length, + int nargs, int variadic, + const __mpu_char16_t *const *argnames, + const size_t *argname_lengths, + const __mpu_char16_t *replacement, + size_t replacement_length, + enum mcpu_macro_builtin builtin, + unsigned builtin_bits, + enum mcpu_macro_origin origin ) +{ + mcpu_macro *macro; + __mpu_char16_t *new_replacement; + __mpu_char16_t **new_argnames = NULL; + size_t *new_argname_lengths = NULL; + unsigned h; + + if( table == NULL || name == NULL || name_length == 0 || + (replacement == NULL && replacement_length != 0) || nargs < -1 || + (variadic && nargs < 0) || + (nargs > 0 && (argnames == NULL || argname_lengths == NULL)) ) + { + errno = EINVAL; + return( -1 ); + } + + new_replacement = dup_text( replacement, replacement_length ); + if( new_replacement == NULL ) + return( -1 ); + + if( nargs >= 0 && + copy_argnames(argnames, argname_lengths, (size_t)nargs, + &new_argnames, &new_argname_lengths) != 0 ) + { + free( new_replacement ); + return( -1 ); + } + + macro = mcpu_macro_find( table, name, name_length ); + if( macro ) + { + free_argnames( macro->argnames, macro->argname_lengths, macro->nargs ); + free( macro->replacement ); + macro->nargs = nargs; + macro->variadic = variadic; + macro->argnames = new_argnames; + macro->argname_lengths = new_argname_lengths; + macro->replacement = new_replacement; + macro->replacement_length = replacement_length; + macro->builtin = builtin; + macro->builtin_bits = builtin_bits; + macro->origin = origin; + return( 0 ); + } + + macro = (mcpu_macro *)calloc( 1, sizeof(*macro) ); + if( macro == NULL ) + { + free_argnames( new_argnames, new_argname_lengths, nargs ); + free( new_replacement ); + return( -1 ); + } + + macro->name = dup_text( name, name_length ); + if( macro->name == NULL ) + { + free_argnames( new_argnames, new_argname_lengths, nargs ); + free( new_replacement ); + free( macro ); + return( -1 ); + } + + macro->name_length = name_length; + macro->nargs = nargs; + macro->variadic = variadic; + macro->argnames = new_argnames; + macro->argname_lengths = new_argname_lengths; + macro->replacement = new_replacement; + macro->replacement_length = replacement_length; + macro->builtin = builtin; + macro->builtin_bits = builtin_bits; + macro->origin = origin; + + h = macro_hash( name, name_length ); + macro->next = table->bucket[h]; + table->bucket[h] = macro; + + return( 0 ); +} + + +int +mcpu_macro_define_object( mcpu_macro_table *table, + const __mpu_char16_t *name, + size_t name_length, + const __mpu_char16_t *replacement, + size_t replacement_length, + enum mcpu_macro_origin origin ) +{ + return( macro_install(table, name, name_length, -1, 0, NULL, NULL, + replacement, replacement_length, + MCPU_MACRO_BUILTIN_NONE, 0, origin) ); +} + + +int +mcpu_macro_define_builtin( mcpu_macro_table *table, + const __mpu_char16_t *name, + size_t name_length, + enum mcpu_macro_builtin builtin ) +{ + if( builtin == MCPU_MACRO_BUILTIN_NONE ) + { + errno = EINVAL; + return( -1 ); + } + + return( macro_install(table, name, name_length, -1, 0, NULL, NULL, + NULL, 0, builtin, 0, MCPU_MACRO_ORIGIN_BUILTIN) ); +} + + +int +mcpu_macro_define_sized_builtin( mcpu_macro_table *table, + const __mpu_char16_t *name, + size_t name_length, + enum mcpu_macro_builtin builtin, + unsigned bits ) +{ + if( builtin < MCPU_MACRO_BUILTIN_REAL_MAX || bits == 0 ) + { + errno = EINVAL; + return( -1 ); + } + + return( macro_install(table, name, name_length, -1, 0, NULL, NULL, + NULL, 0, builtin, bits, MCPU_MACRO_ORIGIN_BUILTIN) ); +} + + +int +mcpu_macro_define_function( mcpu_macro_table *table, + const __mpu_char16_t *name, + size_t name_length, + const __mpu_char16_t *const *argnames, + const size_t *argname_lengths, + size_t nargs, int variadic, + const __mpu_char16_t *replacement, + size_t replacement_length, + enum mcpu_macro_origin origin ) +{ + if( nargs > (size_t)INT_MAX ) + { + errno = EOVERFLOW; + return( -1 ); + } + + return( macro_install(table, name, name_length, (int)nargs, variadic, + argnames, argname_lengths, + replacement, replacement_length, + MCPU_MACRO_BUILTIN_NONE, 0, origin) ); +} + + +int +mcpu_macro_definition_equal( const mcpu_macro *macro, + int nargs, int variadic, + const __mpu_char16_t *const *argnames, + const size_t *argname_lengths, + const __mpu_char16_t *replacement, + size_t replacement_length ) +{ + int i; + + if( macro == NULL || macro->builtin != MCPU_MACRO_BUILTIN_NONE || + macro->nargs != nargs || macro->variadic != variadic || + macro->replacement_length != replacement_length ) + return( 0 ); + + if( replacement_length != 0 && + memcmp( macro->replacement, replacement, + replacement_length * sizeof(__mpu_char16_t) ) != 0 ) + return( 0 ); + + if( nargs < 0 ) + return( 1 ); + + for( i = 0; i < nargs; ++i ) + { + if( macro->argname_lengths[i] != argname_lengths[i] || + memcmp( macro->argnames[i], argnames[i], + argname_lengths[i] * sizeof(__mpu_char16_t) ) != 0 ) + return( 0 ); + } + + return( 1 ); +} + + +int +mcpu_macro_undef( mcpu_macro_table *table, + const __mpu_char16_t *name, size_t name_length ) +{ + mcpu_macro **link; + unsigned h; + + if( table == NULL || name == NULL || name_length == 0 ) + { + errno = EINVAL; + return( -1 ); + } + + h = macro_hash( name, name_length ); + link = &table->bucket[h]; + + while( *link ) + { + mcpu_macro *macro = *link; + + if( macro->name_length == name_length && + memcmp( macro->name, name, + name_length * sizeof(__mpu_char16_t) ) == 0 ) + { + *link = macro->next; + free_macro_contents( macro ); + free( macro ); + return( 1 ); + } + + link = ¯o->next; + } + + return( 0 ); +} + + +static int +is_quote16( __mpu_char16_t c, enum mcpu_language language ) +{ + if( c == '"' || c == '`' ) + return( 1 ); + + if( c == '\'' && language != MCPU_LANG_DIFF ) + return( 1 ); + + return( 0 ); +} + + +static int +is_space16( __mpu_char16_t c ) +{ + return( c == ' ' || c == '\t' || c == '\f' || c == '\v' || + c == '\r' || c == '\n' ); +} + + +static int +macro_argument_index( const mcpu_macro *macro, + const __mpu_char16_t *name, size_t name_length ) +{ + int i; + + for( i = 0; i < macro->nargs; ++i ) + { + if( macro->argname_lengths[i] == name_length && + memcmp( macro->argnames[i], name, + name_length * sizeof(__mpu_char16_t) ) == 0 ) + return( i ); + } + + if( macro->variadic && + mcpu_text_equal_ascii(name, name_length, "__VA_ARGS__") ) + return( macro->nargs ); + + return( -1 ); +} + + +static void +macro_args_free( mcpu_macro_arg *args, size_t nargs ) +{ + size_t i; + + if( args == NULL ) + return; + + for( i = 0; i < nargs; ++i ) + mcpu_text_free( &args[i].expanded ); + + free( args ); +} + + +static int +append_macro_arg( mcpu_macro_arg **args, size_t *nargs, size_t *capacity, + const __mpu_char16_t *raw, size_t raw_length ) +{ + mcpu_macro_arg *new_args; + size_t new_capacity; + + if( *nargs == *capacity ) + { + new_capacity = *capacity ? *capacity * 2 : 4; + if( new_capacity < *capacity || + new_capacity > SIZE_MAX / sizeof(**args) ) + { + errno = EOVERFLOW; + return( -1 ); + } + + new_args = (mcpu_macro_arg *)realloc( *args, + new_capacity * sizeof(**args) ); + if( new_args == NULL ) + return( -1 ); + + *args = new_args; + *capacity = new_capacity; + } + + (*args)[*nargs].raw = raw; + (*args)[*nargs].raw_length = raw_length; + mcpu_text_init( &(*args)[*nargs].expanded ); + (*args)[*nargs].expanded_ready = 0; + ++*nargs; + + return( 0 ); +} + + +static int +parse_macro_args( const __mpu_char16_t *input, size_t length, + size_t open_paren, enum mcpu_language language, + mcpu_macro_arg **args_out, size_t *nargs_out, + size_t *after_call ) +{ + mcpu_macro_arg *args = NULL; + size_t nargs = 0; + size_t capacity = 0; + size_t p = open_paren + 1; + size_t start = p; + int depth = 0; + __mpu_char16_t quote = 0; + int escaped = 0; + + while( p < length ) + { + __mpu_char16_t c = input[p]; + + if( quote ) + { + if( escaped ) + escaped = 0; + else if( c == '\\' ) + escaped = 1; + else if( c == quote ) + quote = 0; + else if( c == '\n' && quote == '\'' ) + quote = 0; + + ++p; + continue; + } + + if( is_quote16(c, language) ) + { + quote = c; + ++p; + continue; + } + + if( c == '(' ) + { + ++depth; + ++p; + continue; + } + + if( c == ')' ) + { + if( depth != 0 ) + { + --depth; + ++p; + continue; + } + + if( append_macro_arg(&args, &nargs, &capacity, + input + start, p - start) != 0 ) + goto fail; + + *args_out = args; + *nargs_out = nargs; + *after_call = p + 1; + return( 0 ); + } + + if( c == ',' && depth == 0 ) + { + if( append_macro_arg(&args, &nargs, &capacity, + input + start, p - start) != 0 ) + goto fail; + start = ++p; + continue; + } + + ++p; + } + + errno = EINVAL; + +fail: + macro_args_free( args, nargs ); + return( -1 ); +} + + +static int +prepare_variadic_args( const mcpu_macro *macro, + const __mpu_char16_t *input, size_t after_call, + mcpu_macro_arg **args_io, size_t *nargs_io ) +{ + mcpu_macro_arg *args = *args_io; + size_t nargs = *nargs_io; + size_t fixed = (size_t)macro->nargs; + size_t i; + + if( !macro->variadic || nargs < fixed ) + return( 0 ); + + if( nargs == fixed ) + { + mcpu_macro_arg *new_args; + + new_args = (mcpu_macro_arg *)realloc( args, + (fixed + 1) * sizeof(*args) ); + if( new_args == NULL ) + return( -1 ); + + args = new_args; + args[fixed].raw = input + after_call - 1; + args[fixed].raw_length = 0; + mcpu_text_init( &args[fixed].expanded ); + args[fixed].expanded_ready = 0; + nargs = fixed + 1; + } + else + { + const __mpu_char16_t *begin = args[fixed].raw; + const __mpu_char16_t *end = args[nargs - 1].raw + + args[nargs - 1].raw_length; + + for( i = fixed + 1; i < nargs; ++i ) + mcpu_text_free( &args[i].expanded ); + + args[fixed].raw = begin; + args[fixed].raw_length = (size_t)(end - begin); + nargs = fixed + 1; + } + + *args_io = args; + *nargs_io = nargs; + return( 0 ); +} + + +static int expand_range( mcpu_macro_table *table, + const __mpu_char16_t *input, size_t length, + enum mcpu_language language, + const mcpu_macro_expansion_context *context, + mcpu_text *output, unsigned depth ); + + +static int +append_quoted_utf8( mcpu_text *output, const char *s ) +{ + mcpu_text text; + size_t i; + int rc = -1; + + if( output == NULL || s == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + mcpu_text_init( &text ); + if( mcpu_text_from_utf8(&text, s) != 0 || + mcpu_text_append_char(output, '"') != 0 ) + goto done; + + for( i = 0; i < text.length; ++i ) + { + switch( text.data[i] ) + { + case '\\': + case '"': + if( mcpu_text_append_char(output, '\\') != 0 || + mcpu_text_append_char(output, text.data[i]) != 0 ) + goto done; + break; + + case '\n': + if( mcpu_text_append_ascii(output, "\\n") != 0 ) + goto done; + break; + + case '\r': + if( mcpu_text_append_ascii(output, "\\r") != 0 ) + goto done; + break; + + case '\t': + if( mcpu_text_append_ascii(output, "\\t") != 0 ) + goto done; + break; + + default: + if( mcpu_text_append_char(output, text.data[i]) != 0 ) + goto done; + break; + } + } + + if( mcpu_text_append_char(output, '"') != 0 ) + goto done; + + rc = 0; + +done: + mcpu_text_free( &text ); + return( rc ); +} + + +static int +append_unsigned( mcpu_text *output, size_t value ) +{ + char buf[64]; + + snprintf( buf, sizeof(buf), "%lu", (unsigned long)value ); + return( mcpu_text_append_ascii(output, buf) ); +} + + +static int +expand_builtin( const mcpu_macro *macro, + const mcpu_macro_expansion_context *context, + mcpu_text *output ) +{ + if( macro == NULL || output == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + if( macro->builtin >= MCPU_MACRO_BUILTIN_FILE && + macro->builtin <= MCPU_MACRO_BUILTIN_INCLUDE_LEVEL && context == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + switch( macro->builtin ) + { + case MCPU_MACRO_BUILTIN_FILE: + return( append_quoted_utf8(output, context->filename ? + context->filename : "") ); + + case MCPU_MACRO_BUILTIN_LINE: + return( append_unsigned(output, context->line_number) ); + + case MCPU_MACRO_BUILTIN_DATE: + return( append_quoted_utf8(output, context->date ? context->date : "") ); + + case MCPU_MACRO_BUILTIN_TIME: + return( append_quoted_utf8(output, context->time ? context->time : "") ); + + case MCPU_MACRO_BUILTIN_BASE_FILE: + return( append_quoted_utf8(output, context->base_filename ? + context->base_filename : "") ); + + case MCPU_MACRO_BUILTIN_INCLUDE_LEVEL: + return( append_unsigned(output, context->include_level) ); + + case MCPU_MACRO_BUILTIN_REAL_MAX: + case MCPU_MACRO_BUILTIN_REAL_MIN: + case MCPU_MACRO_BUILTIN_REAL_EPSILON: + case MCPU_MACRO_BUILTIN_REAL_MAX_EXP: + case MCPU_MACRO_BUILTIN_REAL_MIN_EXP: + case MCPU_MACRO_BUILTIN_REAL_MAX_10_EXP: + case MCPU_MACRO_BUILTIN_REAL_MIN_10_EXP: + return( mcpu_predefined_expand_builtin(macro->builtin, + macro->builtin_bits, + output) ); + + case MCPU_MACRO_BUILTIN_NONE: + default: + errno = EINVAL; + return( -1 ); + } +} + + +static int +ensure_arg_expanded( mcpu_macro_table *table, mcpu_macro_arg *arg, + enum mcpu_language language, + const mcpu_macro_expansion_context *context, + unsigned depth ) +{ + const __mpu_char16_t *raw; + size_t raw_length; + + if( arg->expanded_ready ) + return( 0 ); + + raw = arg->raw; + raw_length = arg->raw_length; + + while( raw_length != 0 && is_space16(*raw) ) + { + ++raw; + --raw_length; + } + while( raw_length != 0 && is_space16(raw[raw_length - 1]) ) + --raw_length; + + if( expand_range(table, raw, raw_length, + language, context, + &arg->expanded, depth + 1) != 0 ) + return( -1 ); + + arg->expanded_ready = 1; + return( 0 ); +} + + +static int +append_stringified_arg( mcpu_text *output, + const __mpu_char16_t *raw, size_t raw_length, + enum mcpu_language language ) +{ + size_t p = 0; + __mpu_char16_t quote = 0; + int escaped = 0; + int pending_space = 0; + + while( raw_length != 0 && is_space16(*raw) ) + { + ++raw; + --raw_length; + } + while( raw_length != 0 && is_space16(raw[raw_length - 1]) ) + --raw_length; + + if( mcpu_text_append_char(output, '"') != 0 ) + return( -1 ); + + while( p < raw_length ) + { + __mpu_char16_t c = raw[p++]; + + if( quote == 0 && is_space16(c) ) + { + pending_space = 1; + continue; + } + + if( pending_space ) + { + if( mcpu_text_append_char(output, ' ') != 0 ) + return( -1 ); + pending_space = 0; + } + + if( c == '"' || (quote != 0 && c == '\\') ) + { + if( mcpu_text_append_char(output, '\\') != 0 ) + return( -1 ); + } + + if( mcpu_text_append_char(output, c) != 0 ) + return( -1 ); + + if( quote != 0 ) + { + if( escaped ) + escaped = 0; + else if( c == '\\' ) + escaped = 1; + else if( c == quote ) + quote = 0; + } + else if( is_quote16(c, language) ) + { + quote = c; + escaped = 0; + } + } + + return( mcpu_text_append_char(output, '"') ); +} + + +/*************************************************************** + Token concatenation works with + preprocessing tokens rather than plain character concatenation. + Parameters adjacent to ## are inserted without prescan, empty + arguments act as placemarkers, pasted tokens are formed first, + and the resulting replacement list is rescanned. + ***************************************************************/ +enum mcpu_macro_piece_kind +{ + MCPU_MACRO_PIECE_TOKEN = 0, + MCPU_MACRO_PIECE_SPACE, + MCPU_MACRO_PIECE_PASTE, + MCPU_MACRO_PIECE_PLACEMARKER +}; + +typedef struct mcpu_macro_piece mcpu_macro_piece; +struct mcpu_macro_piece +{ + enum mcpu_macro_piece_kind kind; + mcpu_text text; +}; + +typedef struct mcpu_macro_piece_list mcpu_macro_piece_list; +struct mcpu_macro_piece_list +{ + mcpu_macro_piece *piece; + size_t count; + size_t capacity; +}; + + +static void +macro_piece_list_init( mcpu_macro_piece_list *list ) +{ + if( list ) + memset( list, 0, sizeof(*list) ); +} + + +static void +macro_piece_list_free( mcpu_macro_piece_list *list ) +{ + size_t i; + + if( list == NULL ) + return; + + for( i = 0; i < list->count; ++i ) + mcpu_text_free( &list->piece[i].text ); + + free( list->piece ); + memset( list, 0, sizeof(*list) ); +} + + +static int +macro_piece_list_reserve( mcpu_macro_piece_list *list, size_t need ) +{ + mcpu_macro_piece *p; + size_t capacity; + + if( need <= list->capacity ) + return( 0 ); + + capacity = list->capacity ? list->capacity : 16; + while( capacity < need ) + { + if( capacity > SIZE_MAX / 2 ) + { + errno = EOVERFLOW; + return( -1 ); + } + capacity *= 2; + } + + if( capacity > SIZE_MAX / sizeof(*p) ) + { + errno = EOVERFLOW; + return( -1 ); + } + + p = (mcpu_macro_piece *)realloc( list->piece, + capacity * sizeof(*p) ); + if( p == NULL ) + return( -1 ); + + list->piece = p; + list->capacity = capacity; + return( 0 ); +} + + +static int +macro_piece_list_append( mcpu_macro_piece_list *list, + enum mcpu_macro_piece_kind kind, + const __mpu_char16_t *text, size_t length ) +{ + mcpu_macro_piece *piece; + + if( macro_piece_list_reserve(list, list->count + 1) != 0 ) + return( -1 ); + + piece = &list->piece[list->count]; + piece->kind = kind; + mcpu_text_init( &piece->text ); + + if( length != 0 && mcpu_text_append(&piece->text, text, length) != 0 ) + { + mcpu_text_free( &piece->text ); + return( -1 ); + } + + ++list->count; + return( 0 ); +} + + +static int +macro_piece_list_append_piece( mcpu_macro_piece_list *list, + const mcpu_macro_piece *piece ) +{ + return( macro_piece_list_append(list, piece->kind, + piece->text.data, piece->text.length) ); +} + + +static void +macro_piece_list_erase( mcpu_macro_piece_list *list, + size_t first, size_t count ) +{ + size_t i; + + if( count == 0 || first >= list->count ) + return; + + if( count > list->count - first ) + count = list->count - first; + + for( i = first; i < first + count; ++i ) + mcpu_text_free( &list->piece[i].text ); + + if( first + count < list->count ) + memmove( list->piece + first, + list->piece + first + count, + (list->count - first - count) * sizeof(*list->piece) ); + + list->count -= count; +} + + +static int +macro_is_digit16( __mpu_char16_t c ) +{ + return( c >= '0' && c <= '9' ); +} + + +static size_t +macro_pp_number_end( const __mpu_char16_t *text, size_t length, size_t p ) +{ + size_t q = p + 1; + + while( q < length ) + { + __mpu_char16_t c = text[q]; + + if( mcpu_pp_is_identifier_char(c) || c == '.' ) + { + ++q; + continue; + } + + if( (c == '+' || c == '-') && q != p ) + { + __mpu_char16_t prev = text[q - 1]; + + if( prev == 'e' || prev == 'E' || prev == 'p' || prev == 'P' ) + { + ++q; + continue; + } + } + + break; + } + + return( q ); +} + + +static size_t +macro_quote_end( const __mpu_char16_t *text, size_t length, size_t p ) +{ + __mpu_char16_t quote = text[p++]; + int escaped = 0; + + while( p < length ) + { + __mpu_char16_t c = text[p++]; + + if( escaped ) + escaped = 0; + else if( c == '\\' ) + escaped = 1; + else if( c == quote || c == '\n' ) + break; + } + + return( p ); +} + + +static size_t +macro_literal_prefix_length( const __mpu_char16_t *text, size_t length, + size_t p, enum mcpu_language language ) +{ + size_t q = p; + + if( q + 2 < length && text[q] == 'u' && text[q + 1] == '8' && + (text[q + 2] == '"' || + (text[q + 2] == '\'' && language != MCPU_LANG_DIFF)) ) + return( 2 ); + + if( q + 1 < length && + (text[q] == 'L' || text[q] == 'u' || text[q] == 'U') && + (text[q + 1] == '"' || + (text[q + 1] == '\'' && language != MCPU_LANG_DIFF)) ) + return( 1 ); + + return( 0 ); +} + + +static int +macro_ascii_at( const __mpu_char16_t *text, size_t length, + size_t p, const char *s ) +{ + size_t i = 0; + + while( s[i] ) + { + if( p + i >= length || + text[p + i] != (__mpu_char16_t)(unsigned char)s[i] ) + return( 0 ); + ++i; + } + + return( 1 ); +} + + +static size_t +macro_punctuator_length( const __mpu_char16_t *text, + size_t length, size_t p ) +{ + static const char *const punctuators[] = + { + "%:%:", + ">>=", "<<=", "...", + "->", "++", "--", "<<", ">>", "<=", ">=", "==", "!=", + "&&", "||", "*=", "/=", "%=", "+=", "-=", "&=", "^=", "|=", + "<:", ":>", "<%", "%>", "%:", "##", + "[", "]", "(", ")", "{", "}", ".", "&", "*", "+", "-", + "~", "!", "/", "%", "<", ">", "^", "|", "?", ":", ";", + "=", ",", "#", + NULL + }; + size_t i; + + for( i = 0; punctuators[i] != NULL; ++i ) + { + size_t n = strlen( punctuators[i] ); + + if( macro_ascii_at(text, length, p, punctuators[i]) ) + return( n ); + } + + return( 0 ); +} + + +static int +macro_piece_list_tokenize( mcpu_macro_piece_list *list, + const __mpu_char16_t *text, size_t length, + enum mcpu_language language, + int recognize_paste ) +{ + size_t p = 0; + + while( p < length ) + { + size_t q; + + if( is_space16(text[p]) ) + { + q = p + 1; + while( q < length && is_space16(text[q]) ) + ++q; + + if( macro_piece_list_append(list, MCPU_MACRO_PIECE_SPACE, + text + p, q - p) != 0 ) + return( -1 ); + p = q; + continue; + } + + if( recognize_paste && p + 1 < length && + text[p] == '#' && text[p + 1] == '#' ) + { + if( macro_piece_list_append(list, MCPU_MACRO_PIECE_PASTE, + NULL, 0) != 0 ) + return( -1 ); + p += 2; + continue; + } + + { + size_t prefix_length; + + prefix_length = macro_literal_prefix_length(text, length, p, language); + if( prefix_length != 0 ) + { + q = macro_quote_end( text, length, p + prefix_length ); + if( macro_piece_list_append(list, MCPU_MACRO_PIECE_TOKEN, + text + p, q - p) != 0 ) + return( -1 ); + p = q; + continue; + } + } + + if( is_quote16(text[p], language) ) + { + q = macro_quote_end( text, length, p ); + if( macro_piece_list_append(list, MCPU_MACRO_PIECE_TOKEN, + text + p, q - p) != 0 ) + return( -1 ); + p = q; + continue; + } + + if( mcpu_pp_is_identifier_start(text[p]) ) + { + q = p + 1; + while( q < length && mcpu_pp_is_identifier_char(text[q]) ) + ++q; + + if( macro_piece_list_append(list, MCPU_MACRO_PIECE_TOKEN, + text + p, q - p) != 0 ) + return( -1 ); + p = q; + continue; + } + + if( macro_is_digit16(text[p]) || + (text[p] == '.' && p + 1 < length && macro_is_digit16(text[p + 1])) ) + { + q = macro_pp_number_end( text, length, p ); + if( macro_piece_list_append(list, MCPU_MACRO_PIECE_TOKEN, + text + p, q - p) != 0 ) + return( -1 ); + p = q; + continue; + } + + q = macro_punctuator_length( text, length, p ); + if( q != 0 ) + { + if( macro_piece_list_append(list, MCPU_MACRO_PIECE_TOKEN, + text + p, q) != 0 ) + return( -1 ); + p += q; + continue; + } + + if( macro_piece_list_append(list, MCPU_MACRO_PIECE_TOKEN, + text + p, 1) != 0 ) + return( -1 ); + ++p; + } + + return( 0 ); +} + + +static int +macro_piece_is_ascii( const mcpu_macro_piece *piece, const char *s ) +{ + if( piece == NULL || piece->kind != MCPU_MACRO_PIECE_TOKEN ) + return( 0 ); + + return( mcpu_text_equal_ascii(piece->text.data, piece->text.length, s) ); +} + + +static int +macro_replacement_has_paste( const mcpu_macro *macro, + enum mcpu_language language ) +{ + size_t p = 0; + __mpu_char16_t quote = 0; + int escaped = 0; + + while( p < macro->replacement_length ) + { + __mpu_char16_t c = macro->replacement[p]; + + if( quote ) + { + if( escaped ) + escaped = 0; + else if( c == '\\' ) + escaped = 1; + else if( c == quote || c == '\n' ) + quote = 0; + ++p; + continue; + } + + if( is_quote16(c, language) ) + { + quote = c; + ++p; + continue; + } + + if( c == '#' && p + 1 < macro->replacement_length && + macro->replacement[p + 1] == '#' ) + return( 1 ); + + ++p; + } + + return( 0 ); +} + + +static int +macro_replacement_has_va_opt( const mcpu_macro *macro, + enum mcpu_language language ) +{ + size_t p = 0; + __mpu_char16_t quote = 0; + int escaped = 0; + + while( p < macro->replacement_length ) + { + __mpu_char16_t c = macro->replacement[p]; + + if( quote ) + { + if( escaped ) + escaped = 0; + else if( c == '\\' ) + escaped = 1; + else if( c == quote || c == '\n' ) + quote = 0; + ++p; + continue; + } + + if( is_quote16(c, language) ) + { + quote = c; + ++p; + continue; + } + + if( mcpu_pp_is_identifier_start(c) ) + { + size_t q = p + 1; + + while( q < macro->replacement_length && + mcpu_pp_is_identifier_char(macro->replacement[q]) ) + ++q; + + if( mcpu_text_equal_ascii(macro->replacement + p, q - p, + "__VA_OPT__") ) + return( 1 ); + + p = q; + continue; + } + + ++p; + } + + return( 0 ); +} + + +static int +macro_piece_parameter_index( const mcpu_macro *macro, + const mcpu_macro_piece *piece ) +{ + if( piece == NULL || piece->kind != MCPU_MACRO_PIECE_TOKEN ) + return( -1 ); + + return( macro_argument_index(macro, piece->text.data, piece->text.length) ); +} + + +static int +macro_piece_parameter_is_pasted( const mcpu_macro_piece_list *replacement, + size_t index ) +{ + size_t p; + + p = index; + while( p != 0 ) + { + --p; + if( replacement->piece[p].kind == MCPU_MACRO_PIECE_SPACE ) + continue; + if( replacement->piece[p].kind == MCPU_MACRO_PIECE_PASTE ) + return( 1 ); + break; + } + + p = index + 1; + while( p < replacement->count ) + { + if( replacement->piece[p].kind == MCPU_MACRO_PIECE_SPACE ) + { + ++p; + continue; + } + if( replacement->piece[p].kind == MCPU_MACRO_PIECE_PASTE ) + return( 1 ); + break; + } + + return( 0 ); +} + + +static int +macro_append_raw_argument( mcpu_macro_piece_list *list, + const mcpu_macro_arg *arg, + enum mcpu_language language ) +{ + const __mpu_char16_t *raw = arg->raw; + size_t raw_length = arg->raw_length; + + while( raw_length != 0 && is_space16(*raw) ) + { + ++raw; + --raw_length; + } + while( raw_length != 0 && is_space16(raw[raw_length - 1]) ) + --raw_length; + + if( raw_length == 0 ) + return( macro_piece_list_append(list, MCPU_MACRO_PIECE_PLACEMARKER, + NULL, 0) ); + + return( macro_piece_list_tokenize(list, raw, raw_length, + language, 0) ); +} + + +static int +macro_append_expanded_argument( mcpu_macro_table *table, + mcpu_macro_piece_list *list, + mcpu_macro_arg *arg, + enum mcpu_language language, + const mcpu_macro_expansion_context *context, + unsigned depth ) +{ + if( ensure_arg_expanded(table, arg, language, context, depth) != 0 ) + return( -1 ); + + return( macro_piece_list_tokenize(list, + arg->expanded.data, + arg->expanded.length, + language, 0) ); +} + + +static int +macro_append_stringified_argument( mcpu_macro_piece_list *list, + const mcpu_macro_arg *arg, + enum mcpu_language language ) +{ + mcpu_text stringified; + int rc; + + mcpu_text_init( &stringified ); + rc = append_stringified_arg( &stringified, + arg->raw, arg->raw_length, + language ); + if( rc == 0 ) + rc = macro_piece_list_append( list, MCPU_MACRO_PIECE_TOKEN, + stringified.data, stringified.length ); + + mcpu_text_free( &stringified ); + return( rc ); +} + + +static int macro_resolve_pastes( + mcpu_macro_piece_list *list, + enum mcpu_language language, + const mcpu_macro_expansion_context *context ); +static int macro_pieces_to_text( + const mcpu_macro_piece_list *list, mcpu_text *text ); + + +static int +macro_va_opt_bounds( const mcpu_macro_piece_list *replacement, + size_t index, size_t limit, + size_t *content_start, size_t *content_end, + size_t *after ) +{ + size_t p; + int depth; + + if( replacement == NULL || index >= limit || + !macro_piece_is_ascii(&replacement->piece[index], "__VA_OPT__") ) + { + errno = EINVAL; + return( -1 ); + } + + p = index + 1; + while( p < limit && replacement->piece[p].kind == MCPU_MACRO_PIECE_SPACE ) + ++p; + + if( p >= limit || !macro_piece_is_ascii(&replacement->piece[p], "(") ) + { + errno = EINVAL; + return( -1 ); + } + + *content_start = ++p; + depth = 1; + while( p < limit ) + { + if( macro_piece_is_ascii(&replacement->piece[p], "(") ) + ++depth; + else if( macro_piece_is_ascii(&replacement->piece[p], ")") ) + { + --depth; + if( depth == 0 ) + { + *content_end = p; + *after = p + 1; + return( 0 ); + } + } + ++p; + } + + errno = EINVAL; + return( -1 ); +} + + +static int +macro_va_opt_is_pasted( const mcpu_macro_piece_list *replacement, + size_t index, size_t after, size_t limit ) +{ + size_t p; + + p = index; + while( p != 0 ) + { + --p; + if( replacement->piece[p].kind == MCPU_MACRO_PIECE_SPACE ) + continue; + if( replacement->piece[p].kind == MCPU_MACRO_PIECE_PASTE ) + return( 1 ); + break; + } + + p = after; + while( p < limit ) + { + if( replacement->piece[p].kind == MCPU_MACRO_PIECE_SPACE ) + { + ++p; + continue; + } + if( replacement->piece[p].kind == MCPU_MACRO_PIECE_PASTE ) + return( 1 ); + break; + } + + return( 0 ); +} + + +static int +macro_piece_list_has_token( const mcpu_macro_piece_list *list ) +{ + size_t i; + + for( i = 0; i < list->count; ++i ) + { + if( list->piece[i].kind == MCPU_MACRO_PIECE_TOKEN ) + return( 1 ); + } + + return( 0 ); +} + + +static int +macro_piece_list_append_range( mcpu_macro_piece_list *dst, + const mcpu_macro_piece_list *src ) +{ + size_t i; + + for( i = 0; i < src->count; ++i ) + { + if( macro_piece_list_append_piece(dst, &src->piece[i]) != 0 ) + return( -1 ); + } + + return( 0 ); +} + + +static int +macro_variadic_has_tokens( mcpu_macro_table *table, + const mcpu_macro *macro, + mcpu_macro_arg *args, + enum mcpu_language language, + const mcpu_macro_expansion_context *context, + unsigned depth ) +{ + mcpu_macro_arg *arg; + size_t i; + + if( !macro->variadic ) + return( 0 ); + + arg = &args[macro->nargs]; + if( ensure_arg_expanded(table, arg, language, context, depth) != 0 ) + return( -1 ); + + for( i = 0; i < arg->expanded.length; ++i ) + { + if( !is_space16(arg->expanded.data[i]) ) + return( 1 ); + } + + return( 0 ); +} + + +static int macro_substitute_piece_range( + mcpu_macro_table *table, + mcpu_macro *macro, + mcpu_macro_arg *args, + const mcpu_macro_piece_list *replacement, + size_t begin, size_t end, + enum mcpu_language language, + const mcpu_macro_expansion_context *context, + unsigned depth, + mcpu_macro_piece_list *substituted ); + + +static int +macro_append_stringified_va_opt( mcpu_macro_table *table, + mcpu_macro *macro, + mcpu_macro_arg *args, + const mcpu_macro_piece_list *replacement, + size_t content_start, size_t content_end, + enum mcpu_language language, + const mcpu_macro_expansion_context *context, + unsigned depth, + mcpu_macro_piece_list *substituted ) +{ + mcpu_macro_piece_list content; + mcpu_text text; + mcpu_text stringified; + int nonempty; + int rc = -1; + + macro_piece_list_init( &content ); + mcpu_text_init( &text ); + mcpu_text_init( &stringified ); + + nonempty = macro_variadic_has_tokens(table, macro, args, + language, context, depth); + if( nonempty < 0 ) + goto done; + + if( nonempty ) + { + if( macro_substitute_piece_range(table, macro, args, replacement, + content_start, content_end, + language, context, depth, + &content) != 0 || + macro_resolve_pastes(&content, language, context) != 0 || + macro_pieces_to_text(&content, &text) != 0 ) + goto done; + } + + if( append_stringified_arg(&stringified, + text.data, text.length, language) != 0 || + macro_piece_list_append(substituted, MCPU_MACRO_PIECE_TOKEN, + stringified.data, stringified.length) != 0 ) + goto done; + + rc = 0; + +done: + mcpu_text_free( &stringified ); + mcpu_text_free( &text ); + macro_piece_list_free( &content ); + return( rc ); +} + + +static int +macro_substitute_piece_range( mcpu_macro_table *table, + mcpu_macro *macro, + mcpu_macro_arg *args, + const mcpu_macro_piece_list *replacement, + size_t begin, size_t end, + enum mcpu_language language, + const mcpu_macro_expansion_context *context, + unsigned depth, + mcpu_macro_piece_list *substituted ) +{ + size_t i = begin; + + while( i < end ) + { + const mcpu_macro_piece *piece = &replacement->piece[i]; + int argno; + + if( piece->kind == MCPU_MACRO_PIECE_SPACE || + piece->kind == MCPU_MACRO_PIECE_PASTE ) + { + if( macro_piece_list_append_piece(substituted, piece) != 0 ) + return( -1 ); + ++i; + continue; + } + + if( macro_piece_is_ascii(piece, "#") ) + { + size_t j = i + 1; + + while( j < end && + replacement->piece[j].kind == MCPU_MACRO_PIECE_SPACE ) + ++j; + + if( j >= end ) + { + errno = EINVAL; + return( -1 ); + } + + argno = macro_piece_parameter_index( macro, &replacement->piece[j] ); + if( argno >= 0 ) + { + if( macro_append_stringified_argument(substituted, + &args[argno], language) != 0 ) + return( -1 ); + i = j + 1; + continue; + } + + if( macro_piece_is_ascii(&replacement->piece[j], "__VA_OPT__") ) + { + size_t content_start; + size_t content_end; + size_t after; + + if( macro_va_opt_bounds(replacement, j, end, + &content_start, &content_end, &after) != 0 || + macro_append_stringified_va_opt(table, macro, args, replacement, + content_start, content_end, + language, context, depth, + substituted) != 0 ) + return( -1 ); + + i = after; + continue; + } + + errno = EINVAL; + return( -1 ); + } + + if( macro_piece_is_ascii(piece, "__VA_OPT__") ) + { + mcpu_macro_piece_list content; + size_t content_start; + size_t content_end; + size_t after; + int nonempty; + int pasted; + int rc = -1; + + macro_piece_list_init( &content ); + if( macro_va_opt_bounds(replacement, i, end, + &content_start, &content_end, &after) != 0 ) + goto va_done; + + nonempty = macro_variadic_has_tokens(table, macro, args, + language, context, depth); + if( nonempty < 0 ) + goto va_done; + + pasted = macro_va_opt_is_pasted(replacement, i, after, end); + if( nonempty ) + { + if( macro_substitute_piece_range(table, macro, args, replacement, + content_start, content_end, + language, context, depth, + &content) != 0 || + macro_resolve_pastes(&content, language, context) != 0 ) + goto va_done; + + if( pasted && !macro_piece_list_has_token(&content) ) + { + if( macro_piece_list_append(substituted, + MCPU_MACRO_PIECE_PLACEMARKER, + NULL, 0) != 0 ) + goto va_done; + } + else if( macro_piece_list_append_range(substituted, &content) != 0 ) + goto va_done; + } + else if( pasted && + macro_piece_list_append(substituted, + MCPU_MACRO_PIECE_PLACEMARKER, + NULL, 0) != 0 ) + goto va_done; + + rc = 0; + +va_done: + macro_piece_list_free( &content ); + if( rc != 0 ) + return( -1 ); + i = after; + continue; + } + + argno = macro_piece_parameter_index( macro, piece ); + if( argno >= 0 ) + { + if( macro_piece_parameter_is_pasted(replacement, i) ) + { + if( macro_append_raw_argument(substituted, + &args[argno], language) != 0 ) + return( -1 ); + } + else + { + if( macro_append_expanded_argument(table, substituted, + &args[argno], language, + context, depth) != 0 ) + return( -1 ); + } + + ++i; + continue; + } + + if( macro_piece_list_append_piece(substituted, piece) != 0 ) + return( -1 ); + ++i; + } + + return( 0 ); +} + + +static int +macro_build_function_pieces( mcpu_macro_table *table, + mcpu_macro *macro, + mcpu_macro_arg *args, + enum mcpu_language language, + const mcpu_macro_expansion_context *context, + unsigned depth, + mcpu_macro_piece_list *substituted ) +{ + mcpu_macro_piece_list replacement; + int rc = -1; + + macro_piece_list_init( &replacement ); + if( macro_piece_list_tokenize(&replacement, + macro->replacement, + macro->replacement_length, + language, 1) != 0 ) + goto done; + + if( macro_substitute_piece_range(table, macro, args, &replacement, + 0, replacement.count, + language, context, depth, + substituted) != 0 ) + goto done; + + rc = 0; + +done: + macro_piece_list_free( &replacement ); + return( rc ); +} + + +static void +macro_normalize_paste_spacing( mcpu_macro_piece_list *list ) +{ + size_t i = 0; + + while( i < list->count ) + { + if( list->piece[i].kind == MCPU_MACRO_PIECE_SPACE && + ((i != 0 && list->piece[i - 1].kind == MCPU_MACRO_PIECE_PASTE) || + (i + 1 < list->count && + list->piece[i + 1].kind == MCPU_MACRO_PIECE_PASTE)) ) + { + macro_piece_list_erase( list, i, 1 ); + if( i != 0 ) + --i; + continue; + } + ++i; + } + + i = 1; + while( i < list->count ) + { + if( list->piece[i - 1].kind == MCPU_MACRO_PIECE_PASTE && + list->piece[i].kind == MCPU_MACRO_PIECE_PASTE ) + { + /*********************************************************** + GNU CPP accepts a run of adjacent ## operators as one paste + operator. Preserve that long-standing behavior. + ***********************************************************/ + macro_piece_list_erase( list, i, 1 ); + continue; + } + ++i; + } +} + + +static int +macro_paste_is_single_token( const mcpu_text *text, + enum mcpu_language language ) +{ + mcpu_macro_piece_list tokens; + int valid; + + macro_piece_list_init( &tokens ); + if( macro_piece_list_tokenize(&tokens, + text->data, text->length, + language, 0) != 0 ) + { + macro_piece_list_free( &tokens ); + return( -1 ); + } + + valid = tokens.count == 1 && + tokens.piece[0].kind == MCPU_MACRO_PIECE_TOKEN; + macro_piece_list_free( &tokens ); + return( valid ); +} + + +static int +macro_warn_invalid_paste( const mcpu_macro_piece *left, + const mcpu_macro_piece *right, + const mcpu_macro_expansion_context *context ) +{ + char *l = mcpu_text_to_utf8( left->text.data, left->text.length ); + char *r = mcpu_text_to_utf8( right->text.data, right->text.length ); + + int rc; + + rc = mcpp_diagnostic_warning( + context ? context->options : NULL, + context && context->filename ? context->filename : "<input>", + context ? context->line_number : 0, + "pasting \"%s\" and \"%s\" does not give a valid preprocessing token", + l ? l : "?", r ? r : "?" ); + + free( l ); + free( r ); + return( rc ); +} + + +static int +macro_resolve_pastes( mcpu_macro_piece_list *list, + enum mcpu_language language, + const mcpu_macro_expansion_context *context ) +{ + size_t i; + + macro_normalize_paste_spacing( list ); + + for( ;; ) + { + for( i = 0; i < list->count; ++i ) + { + if( list->piece[i].kind == MCPU_MACRO_PIECE_PASTE ) + break; + } + + if( i == list->count ) + break; + + if( i == 0 || i + 1 >= list->count || + list->piece[i - 1].kind == MCPU_MACRO_PIECE_SPACE || + list->piece[i + 1].kind == MCPU_MACRO_PIECE_SPACE || + list->piece[i - 1].kind == MCPU_MACRO_PIECE_PASTE || + list->piece[i + 1].kind == MCPU_MACRO_PIECE_PASTE ) + { + errno = EINVAL; + return( -1 ); + } + + if( list->piece[i - 1].kind == MCPU_MACRO_PIECE_PLACEMARKER && + list->piece[i + 1].kind == MCPU_MACRO_PIECE_PLACEMARKER ) + { + macro_piece_list_erase( list, i, 2 ); + continue; + } + + if( list->piece[i - 1].kind == MCPU_MACRO_PIECE_PLACEMARKER ) + { + macro_piece_list_erase( list, i - 1, 2 ); + continue; + } + + if( list->piece[i + 1].kind == MCPU_MACRO_PIECE_PLACEMARKER ) + { + macro_piece_list_erase( list, i, 2 ); + continue; + } + + { + mcpu_text pasted; + int valid; + + mcpu_text_init( &pasted ); + if( mcpu_text_append(&pasted, + list->piece[i - 1].text.data, + list->piece[i - 1].text.length) != 0 || + mcpu_text_append(&pasted, + list->piece[i + 1].text.data, + list->piece[i + 1].text.length) != 0 ) + { + mcpu_text_free( &pasted ); + return( -1 ); + } + + valid = macro_paste_is_single_token( &pasted, language ); + if( valid < 0 ) + { + mcpu_text_free( &pasted ); + return( -1 ); + } + + if( !valid ) + { + /*********************************************************** + GNU CPP documents this case as a diagnostic followed by + emission of the two original tokens. Whitespace between + them is unspecified, so keep them adjacent and discard ##. + ***********************************************************/ + if( macro_warn_invalid_paste(&list->piece[i - 1], + &list->piece[i + 1], context) != 0 ) + { + mcpu_text_free( &pasted ); + return( -1 ); + } + mcpu_text_free( &pasted ); + macro_piece_list_erase( list, i, 1 ); + continue; + } + + mcpu_text_free( &list->piece[i - 1].text ); + list->piece[i - 1].text = pasted; + list->piece[i - 1].kind = MCPU_MACRO_PIECE_TOKEN; + macro_piece_list_erase( list, i, 2 ); + } + } + + i = 0; + while( i < list->count ) + { + if( list->piece[i].kind == MCPU_MACRO_PIECE_PLACEMARKER ) + { + macro_piece_list_erase( list, i, 1 ); + continue; + } + ++i; + } + + return( 0 ); +} + + +static int +macro_pieces_to_text( const mcpu_macro_piece_list *list, mcpu_text *text ) +{ + size_t i; + int previous_space = 0; + + for( i = 0; i < list->count; ++i ) + { + if( list->piece[i].kind == MCPU_MACRO_PIECE_PASTE || + list->piece[i].kind == MCPU_MACRO_PIECE_PLACEMARKER ) + { + errno = EINVAL; + return( -1 ); + } + + if( list->piece[i].kind == MCPU_MACRO_PIECE_SPACE ) + { + /*********************************************************** + Removal of an empty parameter, placemarker or __VA_OPT__ + fragment can make two replacement-list space pieces + adjacent. They are one logical whitespace separator. + + Preserve the text of a single space piece: it may come from + inside an actual macro argument, whose formatting is not part + of replacement-list normalization. + ***********************************************************/ + if( previous_space ) + continue; + previous_space = 1; + } + else + previous_space = 0; + + if( mcpu_text_append(text, + list->piece[i].text.data, + list->piece[i].text.length) != 0 ) + return( -1 ); + } + + return( 0 ); +} + + +static int +expand_function_macro_with_paste( mcpu_macro_table *table, + mcpu_macro *macro, + mcpu_macro_arg *args, + enum mcpu_language language, + const mcpu_macro_expansion_context *context, + mcpu_text *output, unsigned depth ) +{ + mcpu_macro_piece_list pieces; + mcpu_text substituted; + int rc = -1; + + macro_piece_list_init( &pieces ); + mcpu_text_init( &substituted ); + + if( macro_build_function_pieces(table, macro, args, + language, context, depth, + &pieces) != 0 || + macro_resolve_pastes(&pieces, language, context) != 0 || + macro_pieces_to_text(&pieces, &substituted) != 0 ) + goto done; + + macro->expanding = 1; + rc = expand_range( table, substituted.data, substituted.length, + language, context, output, depth + 1 ); + macro->expanding = 0; + +done: + macro->expanding = 0; + mcpu_text_free( &substituted ); + macro_piece_list_free( &pieces ); + return( rc ); +} + + +static int +expand_object_macro_with_paste( mcpu_macro_table *table, + mcpu_macro *macro, + enum mcpu_language language, + const mcpu_macro_expansion_context *context, + mcpu_text *output, unsigned depth ) +{ + mcpu_macro_piece_list pieces; + mcpu_text substituted; + int rc = -1; + + macro_piece_list_init( &pieces ); + mcpu_text_init( &substituted ); + + if( macro_piece_list_tokenize(&pieces, + macro->replacement, + macro->replacement_length, + language, 1) != 0 || + macro_resolve_pastes(&pieces, language, context) != 0 || + macro_pieces_to_text(&pieces, &substituted) != 0 ) + goto done; + + macro->expanding = 1; + rc = expand_range( table, substituted.data, substituted.length, + language, context, output, depth + 1 ); + macro->expanding = 0; + +done: + macro->expanding = 0; + mcpu_text_free( &substituted ); + macro_piece_list_free( &pieces ); + return( rc ); +} + + +static int +expand_function_macro( mcpu_macro_table *table, mcpu_macro *macro, + mcpu_macro_arg *args, size_t nargs, + enum mcpu_language language, + const mcpu_macro_expansion_context *context, + mcpu_text *output, unsigned depth ) +{ + mcpu_text substituted; + size_t p = 0; + size_t i; + int rc = -1; + + if( !macro->variadic && macro->nargs == 0 && nargs == 1 ) + { + int all_space = 1; + + for( i = 0; i < args[0].raw_length; ++i ) + { + if( !is_space16(args[0].raw[i]) ) + { + all_space = 0; + break; + } + } + + if( all_space ) + nargs = 0; + } + + if( nargs < (size_t)macro->nargs + (macro->variadic ? 1U : 0U) ) + { + char *name = mcpu_text_to_utf8( macro->name, macro->name_length ); + fprintf( stderr, "%s:%u: error: macro '%s' used with too few arguments\n", + context && context->filename ? context->filename : "<input>", + context ? context->line_number : 0, + name ? name : "?" ); + free( name ); + errno = EINVAL; + return( -1 ); + } + + if( !macro->variadic && nargs > (size_t)macro->nargs ) + { + char *name = mcpu_text_to_utf8( macro->name, macro->name_length ); + fprintf( stderr, "%s:%u: error: macro '%s' used with too many arguments\n", + context && context->filename ? context->filename : "<input>", + context ? context->line_number : 0, + name ? name : "?" ); + free( name ); + errno = EINVAL; + return( -1 ); + } + + if( macro_replacement_has_paste(macro, language) || + macro_replacement_has_va_opt(macro, language) ) + return( expand_function_macro_with_paste(table, macro, args, + language, context, + output, depth) ); + + mcpu_text_init( &substituted ); + + while( p < macro->replacement_length ) + { + if( is_quote16(macro->replacement[p], language) ) + { + __mpu_char16_t quote = macro->replacement[p]; + int escaped = 0; + int first = 1; + + do + { + __mpu_char16_t c = macro->replacement[p++]; + + if( mcpu_text_append_char(&substituted, c) != 0 ) + goto substituted_fail; + + if( first ) + first = 0; + else if( escaped ) + escaped = 0; + else if( c == '\\' ) + escaped = 1; + else if( c == quote ) + break; + else if( c == '\n' ) + break; + } + while( p < macro->replacement_length ); + + continue; + } + + if( macro->replacement[p] == '#' ) + { + size_t q = p + 1; + size_t start; + int argno; + + if( q < macro->replacement_length && macro->replacement[q] == '#' ) + { + errno = EINVAL; + goto substituted_fail; + } + + while( q < macro->replacement_length && is_space16(macro->replacement[q]) ) + ++q; + + if( q >= macro->replacement_length || + !mcpu_pp_is_identifier_start(macro->replacement[q]) ) + { + errno = EINVAL; + goto substituted_fail; + } + + start = q++; + while( q < macro->replacement_length && + mcpu_pp_is_identifier_char(macro->replacement[q]) ) + ++q; + + argno = macro_argument_index( macro, + macro->replacement + start, q - start ); + if( argno < 0 ) + { + errno = EINVAL; + goto substituted_fail; + } + + if( append_stringified_arg(&substituted, + args[argno].raw, + args[argno].raw_length, + language) != 0 ) + goto substituted_fail; + + p = q; + continue; + } + + if( mcpu_pp_is_identifier_start(macro->replacement[p]) ) + { + size_t start = p++; + int argno; + + while( p < macro->replacement_length && + mcpu_pp_is_identifier_char(macro->replacement[p]) ) + ++p; + + argno = macro_argument_index( macro, + macro->replacement + start, p - start ); + if( argno >= 0 ) + { + /*********************************************************** + Ordinary parameter occurrences use + the macro-expanded actual argument. Stringified occurrences + above deliberately use the raw spelling and therefore do not + force this expansion. + ***********************************************************/ + if( ensure_arg_expanded(table, &args[argno], language, + context, depth) != 0 ) + goto substituted_fail; + + if( mcpu_text_append(&substituted, + args[argno].expanded.data, + args[argno].expanded.length) != 0 ) + goto substituted_fail; + } + else if( mcpu_text_append(&substituted, + macro->replacement + start, p - start) != 0 ) + goto substituted_fail; + + continue; + } + + if( mcpu_text_append_char(&substituted, macro->replacement[p++]) != 0 ) + goto substituted_fail; + } + + macro->expanding = 1; + rc = expand_range( table, substituted.data, substituted.length, + language, context, + output, depth + 1 ); + +substituted_fail: + mcpu_text_free( &substituted ); + + macro->expanding = 0; + return( rc ); +} + + +static int +expand_range( mcpu_macro_table *table, + const __mpu_char16_t *input, size_t length, + enum mcpu_language language, + const mcpu_macro_expansion_context *context, + mcpu_text *output, unsigned depth ) +{ + size_t p = 0; + + if( depth > MCPU_CPP_MACRO_EXPANSION_LIMIT ) + { + errno = ELOOP; + return( -1 ); + } + + while( p < length ) + { + if( is_quote16(input[p], language) ) + { + __mpu_char16_t quote = input[p]; + int escaped = 0; + int first = 1; + + do + { + __mpu_char16_t c = input[p++]; + + if( mcpu_text_append_char(output, c) != 0 ) + return( -1 ); + + if( first ) + first = 0; + else if( escaped ) + escaped = 0; + else if( c == '\\' ) + escaped = 1; + else if( c == quote ) + break; + else if( c == '\n' ) + break; + } + while( p < length ); + + continue; + } + + if( mcpu_pp_is_identifier_start(input[p]) ) + { + size_t start = p++; + mcpu_macro *macro; + + while( p < length && mcpu_pp_is_identifier_char(input[p]) ) + ++p; + + macro = mcpu_macro_find( table, input + start, p - start ); + if( macro && !macro->expanding ) + { + if( macro->builtin != MCPU_MACRO_BUILTIN_NONE ) + { + if( expand_builtin(macro, context, output) != 0 ) + return( -1 ); + continue; + } + + if( macro->nargs < 0 ) + { + int rc; + + if( macro_replacement_has_paste(macro, language) ) + rc = expand_object_macro_with_paste(table, macro, + language, context, + output, depth); + else + { + macro->expanding = 1; + rc = expand_range( table, + macro->replacement, + macro->replacement_length, + language, context, + output, depth + 1 ); + macro->expanding = 0; + } + + if( rc != 0 ) + return( -1 ); + continue; + } + else + { + size_t q = p; + + while( q < length && is_space16(input[q]) ) + ++q; + + if( q < length && input[q] == '(' ) + { + mcpu_macro_arg *args = NULL; + size_t nargs = 0; + size_t after_call; + int rc; + + if( parse_macro_args(input, length, q, language, + &args, &nargs, &after_call) != 0 ) + { + char *name = mcpu_text_to_utf8( macro->name, + macro->name_length ); + fprintf( stderr, + "%s:%u: error: unterminated argument list for macro '%s'\n", + context && context->filename ? context->filename : "<input>", + context ? context->line_number : 0, + name ? name : "?" ); + free( name ); + return( -1 ); + } + + if( prepare_variadic_args(macro, input, after_call, + &args, &nargs) != 0 ) + { + macro_args_free( args, nargs ); + return( -1 ); + } + + rc = expand_function_macro( table, macro, args, nargs, + language, context, + output, depth ); + macro_args_free( args, nargs ); + if( rc != 0 ) + return( -1 ); + + p = after_call; + continue; + } + } + } + + if( mcpu_text_append(output, input + start, p - start) != 0 ) + return( -1 ); + + continue; + } + + if( mcpu_text_append_char(output, input[p++]) != 0 ) + return( -1 ); + } + + return( 0 ); +} + + + +static int +macro_name_compare( const void *a, const void *b ) +{ + const mcpu_macro *ma = *(const mcpu_macro * const *)a; + const mcpu_macro *mb = *(const mcpu_macro * const *)b; + size_t n; + size_t i; + + n = ma->name_length < mb->name_length ? ma->name_length : mb->name_length; + for( i = 0; i < n; ++i ) + { + if( ma->name[i] < mb->name[i] ) + return( -1 ); + if( ma->name[i] > mb->name[i] ) + return( 1 ); + } + + if( ma->name_length < mb->name_length ) + return( -1 ); + if( ma->name_length > mb->name_length ) + return( 1 ); + return( 0 ); +} + + +int +mcpu_macro_dump( const mcpu_macro_table *table, FILE *stream, + int include_predefined ) +{ + mcpu_macro **list; + size_t predefined_count = 0; + size_t other_count = 0; + size_t count; + size_t i; + size_t k = 0; + size_t other_begin; + + if( table == NULL || stream == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + for( i = 0; i < MCPU_CPP_MACRO_BUCKETS; ++i ) + { + const mcpu_macro *macro; + for( macro = table->bucket[i]; macro; macro = macro->next ) + { + /* Context-dependent special macros are not useful in a static dump. */ + if( macro->builtin != MCPU_MACRO_BUILTIN_NONE && + macro->builtin < MCPU_MACRO_BUILTIN_REAL_MAX ) + continue; + + if( macro->origin == MCPU_MACRO_ORIGIN_BUILTIN ) + ++predefined_count; + else + ++other_count; + } + } + + count = other_count + (include_predefined ? predefined_count : 0); + list = count ? (mcpu_macro **)calloc( count, sizeof(*list) ) : NULL; + if( count != 0 && list == NULL ) + return( -1 ); + + if( include_predefined ) + { + for( i = 0; i < MCPU_CPP_MACRO_BUCKETS; ++i ) + { + mcpu_macro *macro; + for( macro = table->bucket[i]; macro; macro = macro->next ) + { + if( (macro->builtin == MCPU_MACRO_BUILTIN_NONE || + macro->builtin >= MCPU_MACRO_BUILTIN_REAL_MAX) && + macro->origin == MCPU_MACRO_ORIGIN_BUILTIN ) + list[k++] = macro; + } + } + } + + other_begin = k; + for( i = 0; i < MCPU_CPP_MACRO_BUCKETS; ++i ) + { + mcpu_macro *macro; + for( macro = table->bucket[i]; macro; macro = macro->next ) + { + if( (macro->builtin == MCPU_MACRO_BUILTIN_NONE || + macro->builtin >= MCPU_MACRO_BUILTIN_REAL_MAX) && + macro->origin != MCPU_MACRO_ORIGIN_BUILTIN ) + list[k++] = macro; + } + } + + if( include_predefined && predefined_count > 1 ) + qsort( list, predefined_count, sizeof(*list), macro_name_compare ); + if( other_count > 1 ) + qsort( list + other_begin, other_count, sizeof(*list), macro_name_compare ); + + for( i = 0; i < count; ++i ) + { + mcpu_macro *macro = list[i]; + char *name = mcpu_text_to_utf8( macro->name, macro->name_length ); + char *replacement = NULL; + mcpu_text generated; + size_t replacement_length; + int j; + + mcpu_text_init( &generated ); + if( macro->builtin == MCPU_MACRO_BUILTIN_NONE ) + { + replacement = mcpu_text_to_utf8( macro->replacement, + macro->replacement_length ); + replacement_length = macro->replacement_length; + } + else + { + if( expand_builtin(macro, NULL, &generated) != 0 ) + { + free( name ); + mcpu_text_free( &generated ); + free( list ); + return( -1 ); + } + replacement = mcpu_text_to_utf8( generated.data, generated.length ); + replacement_length = generated.length; + } + + if( name == NULL || replacement == NULL ) + { + free( name ); + free( replacement ); + mcpu_text_free( &generated ); + free( list ); + return( -1 ); + } + + if( fprintf(stream, "#define %s", name) < 0 ) + goto io_error; + + if( macro->nargs >= 0 ) + { + if( fputc('(', stream) == EOF ) + goto io_error; + for( j = 0; j < macro->nargs; ++j ) + { + char *arg = mcpu_text_to_utf8( macro->argnames[j], + macro->argname_lengths[j] ); + if( arg == NULL ) + goto io_error; + if( j != 0 && fputc(',', stream) == EOF ) + { + free( arg ); + goto io_error; + } + if( fputs(arg, stream) == EOF ) + { + free( arg ); + goto io_error; + } + free( arg ); + } + if( macro->variadic ) + { + if( macro->nargs != 0 && fputc(',', stream) == EOF ) + goto io_error; + if( fputs("...", stream) == EOF ) + goto io_error; + } + if( fputc(')', stream) == EOF ) + goto io_error; + } + + if( replacement_length != 0 ) + { + if( fputc(' ', stream) == EOF || fputs(replacement, stream) == EOF ) + goto io_error; + } + + if( fputc('\n', stream) == EOF ) + goto io_error; + + free( name ); + free( replacement ); + mcpu_text_free( &generated ); + continue; + +io_error: + free( name ); + free( replacement ); + mcpu_text_free( &generated ); + free( list ); + return( -1 ); + } + + free( list ); + return( ferror(stream) ? -1 : 0 ); +} + + +int +mcpu_macro_dump_text( const mcpu_macro_table *table, mcpu_text *output, + const char *line_filename ) +{ + mcpu_macro **list; + size_t count = 0; + size_t i; + size_t k = 0; + + if( table == NULL || output == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + for( i = 0; i < MCPU_CPP_MACRO_BUCKETS; ++i ) + { + const mcpu_macro *macro; + + for( macro = table->bucket[i]; macro; macro = macro->next ) + if( macro->builtin == MCPU_MACRO_BUILTIN_NONE || + macro->builtin >= MCPU_MACRO_BUILTIN_REAL_MAX ) + ++count; + } + + list = count ? (mcpu_macro **)calloc( count, sizeof(*list) ) : NULL; + if( count != 0 && list == NULL ) + return( -1 ); + + for( i = 0; i < MCPU_CPP_MACRO_BUCKETS; ++i ) + { + mcpu_macro *macro; + + for( macro = table->bucket[i]; macro; macro = macro->next ) + if( macro->builtin == MCPU_MACRO_BUILTIN_NONE || + macro->builtin >= MCPU_MACRO_BUILTIN_REAL_MAX ) + list[k++] = macro; + } + + if( count > 1 ) + qsort( list, count, sizeof(*list), macro_name_compare ); + + for( i = 0; i < count; ++i ) + { + mcpu_macro *macro = list[i]; + mcpu_text generated; + const __mpu_char16_t *replacement; + size_t replacement_length; + int j; + + mcpu_text_init( &generated ); + + if( line_filename != NULL ) + { + const char *marker = macro->origin == MCPU_MACRO_ORIGIN_COMMAND_LINE ? + "<command-line>" : "<built-in>"; + + if( mcpu_text_append_ascii(output, "# 0 \"" ) != 0 || + mcpu_text_append_utf8(output, marker) != 0 || + mcpu_text_append_ascii(output, "\"\n") != 0 ) + goto fail; + } + + if( mcpu_text_append_ascii(output, "#define ") != 0 || + mcpu_text_append(output, macro->name, macro->name_length) != 0 ) + goto fail; + + if( macro->nargs >= 0 ) + { + if( mcpu_text_append_char(output, '(') != 0 ) + goto fail; + + for( j = 0; j < macro->nargs; ++j ) + { + if( j != 0 && mcpu_text_append_char(output, ',') != 0 ) + goto fail; + if( mcpu_text_append(output, macro->argnames[j], + macro->argname_lengths[j]) != 0 ) + goto fail; + } + + if( macro->variadic ) + { + if( macro->nargs != 0 && mcpu_text_append_char(output, ',') != 0 ) + goto fail; + if( mcpu_text_append_ascii(output, "...") != 0 ) + goto fail; + } + + if( mcpu_text_append_char(output, ')') != 0 ) + goto fail; + } + + if( macro->builtin == MCPU_MACRO_BUILTIN_NONE ) + { + replacement = macro->replacement; + replacement_length = macro->replacement_length; + } + else + { + if( expand_builtin(macro, NULL, &generated) != 0 ) + goto fail; + replacement = generated.data; + replacement_length = generated.length; + } + + if( replacement_length != 0 && + (mcpu_text_append_char(output, ' ') != 0 || + mcpu_text_append(output, replacement, replacement_length) != 0) ) + goto fail; + + if( mcpu_text_append_char(output, '\n') != 0 ) + goto fail; + + mcpu_text_free( &generated ); + continue; + +fail: + mcpu_text_free( &generated ); + free( list ); + return( -1 ); + } + + free( list ); + return( 0 ); +} + + +int +mcpu_macro_expand( mcpu_macro_table *table, + const __mpu_char16_t *input, size_t length, + enum mcpu_language language, + const mcpu_macro_expansion_context *context, + mcpu_text *output ) +{ + if( table == NULL || input == NULL || output == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + mcpu_text_free( output ); + mcpu_text_init( output ); + + return( expand_range(table, input, length, language, + context, output, 0) ); +} diff --git a/src/mcpp-macro.h b/src/mcpp-macro.h new file mode 100644 index 0000000..719972c --- /dev/null +++ b/src/mcpp-macro.h @@ -0,0 +1,126 @@ +#ifndef __MCPU_CPP_MACRO_H__ +#define __MCPU_CPP_MACRO_H__ 1 + +#include <defs.h> +#include <mcpp-language.h> +#include <mcpp-text.h> + +#define MCPU_CPP_MACRO_BUCKETS 257 +#define MCPU_CPP_MACRO_EXPANSION_LIMIT 400 + +enum mcpu_macro_origin +{ + MCPU_MACRO_ORIGIN_SOURCE = 0, + MCPU_MACRO_ORIGIN_BUILTIN, + MCPU_MACRO_ORIGIN_COMMAND_LINE +}; + + +enum mcpu_macro_builtin +{ + MCPU_MACRO_BUILTIN_NONE = 0, + MCPU_MACRO_BUILTIN_FILE, + MCPU_MACRO_BUILTIN_LINE, + MCPU_MACRO_BUILTIN_DATE, + MCPU_MACRO_BUILTIN_TIME, + MCPU_MACRO_BUILTIN_BASE_FILE, + MCPU_MACRO_BUILTIN_INCLUDE_LEVEL, + MCPU_MACRO_BUILTIN_REAL_MAX, + MCPU_MACRO_BUILTIN_REAL_MIN, + MCPU_MACRO_BUILTIN_REAL_EPSILON, + MCPU_MACRO_BUILTIN_REAL_MAX_EXP, + MCPU_MACRO_BUILTIN_REAL_MIN_EXP, + MCPU_MACRO_BUILTIN_REAL_MAX_10_EXP, + MCPU_MACRO_BUILTIN_REAL_MIN_10_EXP +}; + +typedef struct mcpp_options mcpp_options; +typedef struct mcpu_macro_expansion_context mcpu_macro_expansion_context; +struct mcpu_macro_expansion_context +{ + const char *filename; + const char *base_filename; + unsigned line_number; + size_t include_level; + const char *date; + const char *time; + const mcpp_options *options; +}; + +typedef struct mcpu_macro mcpu_macro; +struct mcpu_macro +{ + __mpu_char16_t *name; + size_t name_length; + + int nargs; + int variadic; + __mpu_char16_t **argnames; + size_t *argname_lengths; + + __mpu_char16_t *replacement; + size_t replacement_length; + + enum mcpu_macro_builtin builtin; + enum mcpu_macro_origin origin; + unsigned builtin_bits; + int expanding; + mcpu_macro *next; +}; + +typedef struct mcpu_macro_table mcpu_macro_table; +struct mcpu_macro_table +{ + mcpu_macro *bucket[MCPU_CPP_MACRO_BUCKETS]; +}; + +void mcpu_macro_table_init( mcpu_macro_table *table ); +void mcpu_macro_table_free( mcpu_macro_table *table ); +mcpu_macro *mcpu_macro_find( mcpu_macro_table *table, + const __mpu_char16_t *name, + size_t name_length ); +int mcpu_macro_define_object( mcpu_macro_table *table, + const __mpu_char16_t *name, + size_t name_length, + const __mpu_char16_t *replacement, + size_t replacement_length, + enum mcpu_macro_origin origin ); +int mcpu_macro_define_builtin( mcpu_macro_table *table, + const __mpu_char16_t *name, + size_t name_length, + enum mcpu_macro_builtin builtin ); +int mcpu_macro_define_sized_builtin( mcpu_macro_table *table, + const __mpu_char16_t *name, + size_t name_length, + enum mcpu_macro_builtin builtin, + unsigned bits ); +int mcpu_macro_define_function( mcpu_macro_table *table, + const __mpu_char16_t *name, + size_t name_length, + const __mpu_char16_t *const *argnames, + const size_t *argname_lengths, + size_t nargs, int variadic, + const __mpu_char16_t *replacement, + size_t replacement_length, + enum mcpu_macro_origin origin ); +int mcpu_macro_definition_equal( const mcpu_macro *macro, + int nargs, int variadic, + const __mpu_char16_t *const *argnames, + const size_t *argname_lengths, + const __mpu_char16_t *replacement, + size_t replacement_length ); +int mcpu_macro_undef( mcpu_macro_table *table, + const __mpu_char16_t *name, + size_t name_length ); +int mcpu_macro_dump( const mcpu_macro_table *table, FILE *stream, + int include_predefined ); +int mcpu_macro_dump_text( const mcpu_macro_table *table, mcpu_text *output, + const char *line_filename ); +int mcpu_macro_expand( mcpu_macro_table *table, + const __mpu_char16_t *input, + size_t length, + enum mcpu_language language, + const mcpu_macro_expansion_context *context, + mcpu_text *output ); + +#endif /* __MCPU_CPP_MACRO_H__ */ diff --git a/src/mcpp-options.c b/src/mcpp-options.c new file mode 100644 index 0000000..62fafc6 --- /dev/null +++ b/src/mcpp-options.c @@ -0,0 +1,365 @@ +#include <defs.h> + +static const char * +option_argument( mcpp_options *opts, int argc, char **argv, int *i, + const char *attached, const char *option ) +{ + if( attached != NULL && *attached != 0 ) + return( attached ); + + if( *i + 1 >= argc ) + { + fprintf( stderr, "%s: argument missing after '%s'\n", + opts->progname, option ); + return( NULL ); + } + + ++*i; + return( argv[*i] ); +} + +static int +append_action( mcpp_options *opts, enum mcpp_option_action_kind kind, + const char *option, const char *argument ) +{ + if( opts->action_count >= opts->action_capacity ) + { + errno = EOVERFLOW; + return( -1 ); + } + + opts->actions[opts->action_count].kind = kind; + opts->actions[opts->action_count].option = option; + opts->actions[opts->action_count].argument = argument; + ++opts->action_count; + + return( 0 ); +} + +static int +set_filename( mcpp_options *opts, const char *name ) +{ + if( opts->in_fname == NULL ) + { + opts->in_fname = strcmp(name, "-") == 0 ? "" : name; + return( 0 ); + } + + if( opts->out_fname == NULL ) + { + opts->out_fname = strcmp(name, "-") == 0 ? "" : name; + return( 0 ); + } + + fprintf( stderr, "%s: too many filenames\n", opts->progname ); + return( -1 ); +} + +static int +parse_dump_option( mcpp_options *opts, const char *arg ) +{ + const char *p = arg + 2; + + if( *p == 0 ) + return( -1 ); + + while( *p ) + { + switch( *p++ ) + { + case 'M': + opts->dump_macros = MCPP_DUMP_ONLY; + opts->no_output = 1; + while( *p == 'P' ) + { + opts->dump_only_plus_predefined = 1; + ++p; + } + break; + case 'D': + opts->dump_macros = MCPP_DUMP_DEFINITIONS; + break; + default: + return( -1 ); + } + } + + return( 0 ); +} + +void +mcpp_options_init( mcpp_options *opts, int argc ) +{ + if( opts == NULL ) + return; + + memset( opts, 0, sizeof(*opts) ); + opts->progname = "mcpu-cpp"; + opts->object_suffix = ".o"; + opts->action_capacity = argc > 0 ? (size_t)argc * 2 : 2; + opts->actions = (mcpp_option_action *)calloc( opts->action_capacity, + sizeof(*opts->actions) ); +} + +void +mcpp_options_free( mcpp_options *opts ) +{ + if( opts == NULL ) + return; + + free( opts->actions ); + opts->actions = NULL; + opts->action_count = 0; + opts->action_capacity = 0; +} + +int +mcpp_handle_options( mcpp_options *opts, int argc, char **argv ) +{ + size_t j; + int i; + + if( opts == NULL || argv == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + for( i = 0; i < argc; ++i ) + { + const char *arg = argv[i]; + const char *value; + + if( arg[0] != '-' || arg[1] == 0 ) + { + if( set_filename(opts, arg) != 0 ) return( -1 ); + continue; + } + + if( strcmp(arg, "--help") == 0 ) opts->help = 1; + else if( strcmp(arg, "--version") == 0 ) opts->version = 1; + else if( strcmp(arg, "--no-config") == 0 ) opts->no_config = 1; + else if( strcmp(arg, "--config-file") == 0 ) + { + value = option_argument(opts, argc, argv, &i, NULL, arg); + if( value == NULL ) return( -1 ); + opts->config_fname = value; + } + else if( strcmp(arg, "--object-suffix") == 0 ) + { + value = option_argument(opts, argc, argv, &i, NULL, arg); + if( value == NULL ) return( -1 ); + opts->object_suffix = value; + } + else if( strcmp(arg, "-dconfig") == 0 ) opts->dump_config = 1; + else if( strcmp(arg, "-dsearch-dirs") == 0 ) opts->dump_search_dirs = 1; + else if( arg[0] == '-' && arg[1] == 'd' && arg[2] != 0 ) + { + if( parse_dump_option(opts, arg) != 0 ) goto unknown; + } + else if( strcmp(arg, "-v") == 0 || strcmp(arg, "--verbose") == 0 ) opts->verbose = 1; + else if( strcmp(arg, "-nostdinc") == 0 ) opts->no_standard_includes = 1; + else if( strcmp(arg, "-nostdinc++") == 0 ) opts->no_standard_kplusplus_includes = 1; + else if( strcmp(arg, "-E") == 0 ) { } + else if( strcmp(arg, "-w") == 0 ) opts->inhibit_warnings = 1; + else if( strcmp(arg, "-Wcomment") == 0 || strcmp(arg, "-Wcomments") == 0 ) + { + opts->warn_comments = 1; + opts->warn_comments_explicit = 1; + } + else if( strcmp(arg, "-Wno-comment") == 0 || strcmp(arg, "-Wno-comments") == 0 ) + { + opts->warn_comments = 0; + opts->warn_comments_explicit = 1; + } + else if( strcmp(arg, "-Wall") == 0 ) + { + if( !opts->warn_comments_explicit ) + opts->warn_comments = 1; + } + else if( strcmp(arg, "-Werror") == 0 ) opts->warnings_are_errors = 1; + else if( strcmp(arg, "-Wno-error") == 0 ) opts->warnings_are_errors = 0; + else if( strcmp(arg, "-lint") == 0 ) opts->for_lint = 1; + else if( strcmp(arg, "-M") == 0 ) + { + opts->print_deps = 2; + opts->inhibit_output = 1; + } + else if( strcmp(arg, "-MM") == 0 ) + { + opts->print_deps = 1; + opts->inhibit_output = 1; + } + else if( strcmp(arg, "-MD") == 0 ) + { + opts->print_deps = 2; + opts->inhibit_output = 0; + } + else if( strcmp(arg, "-MMD") == 0 ) + { + opts->print_deps = 1; + opts->inhibit_output = 0; + } + else if( arg[0] == '-' && arg[1] == 'M' && arg[2] == 'F' ) + { + value = option_argument(opts, argc, argv, &i, arg + 3, "-MF"); + if( value == NULL ) return( -1 ); + opts->deps_file = value; + } + else if( arg[0] == '-' && arg[1] == 'M' && + (arg[2] == 'T' || arg[2] == 'Q') ) + { + enum mcpp_option_action_kind kind = arg[2] == 'T' ? + MCPP_ACTION_DEP_TARGET : MCPP_ACTION_DEP_TARGET_QUOTED; + const char *option = arg[2] == 'T' ? "-MT" : "-MQ"; + + value = option_argument(opts, argc, argv, &i, arg + 3, option); + if( value == NULL || append_action(opts, kind, option, value) != 0 ) + return( -1 ); + } + else if( strcmp(arg, "-MG") == 0 ) opts->print_deps_missing_files = 1; + else if( strcmp(arg, "-o") == 0 ) + { + if( opts->out_fname != NULL ) + { + fprintf(stderr, "%s: output filename specified twice\n", opts->progname); + return( -1 ); + } + value = option_argument(opts, argc, argv, &i, NULL, arg); + if( value == NULL ) return( -1 ); + opts->out_fname = strcmp(value, "-") == 0 ? "" : value; + } + else if( arg[1] == 'D' ) + { + value = option_argument(opts, argc, argv, &i, arg + 2, "-D"); + if( value == NULL || append_action(opts, MCPP_ACTION_DEFINE, "-D", value) != 0 ) return( -1 ); + } + else if( arg[1] == 'U' ) + { + value = option_argument(opts, argc, argv, &i, arg + 2, "-U"); + if( value == NULL || append_action(opts, MCPP_ACTION_UNDEF, "-U", value) != 0 ) return( -1 ); + } + else if( arg[1] == 'I' && arg[2] == '-' && arg[3] == 0 ) goto unknown; + else if( arg[1] == 'I' ) + { + value = option_argument(opts, argc, argv, &i, arg + 2, "-I"); + if( value == NULL || append_action(opts, MCPP_ACTION_INCLUDE_USER, "-I", value) != 0 ) return( -1 ); + } + else if( strcmp(arg, "-include") == 0 || strcmp(arg, "-imacros") == 0 || + strcmp(arg, "-isystem") == 0 || strcmp(arg, "-idirafter") == 0 ) + { + enum mcpp_option_action_kind kind = MCPP_ACTION_INCLUDE; + if( strcmp(arg, "-imacros") == 0 ) kind = MCPP_ACTION_IMACROS; + else if( strcmp(arg, "-isystem") == 0 ) kind = MCPP_ACTION_INCLUDE_SYSTEM; + else if( strcmp(arg, "-idirafter") == 0 ) kind = MCPP_ACTION_INCLUDE_AFTER; + value = option_argument(opts, argc, argv, &i, NULL, arg); + if( value == NULL || append_action(opts, kind, arg, value) != 0 ) return( -1 ); + } + else + { +unknown: + fprintf( stderr, "%s: unknown option '%s'\n", opts->progname, arg ); + return( -1 ); + } + } + + if( opts->in_fname == NULL ) opts->in_fname = ""; + if( opts->out_fname == NULL ) opts->out_fname = ""; + + if( opts->deps_file != NULL && opts->print_deps == 0 ) + { + fprintf( stderr, + "%s: -MF requires one of -M, -MM, -MD or -MMD\n", + opts->progname ); + return( -1 ); + } + + if( opts->print_deps == 0 ) + { + for( j = 0; j < opts->action_count; ++j ) + if( opts->actions[j].kind == MCPP_ACTION_DEP_TARGET || + opts->actions[j].kind == MCPP_ACTION_DEP_TARGET_QUOTED ) + { + fprintf( stderr, + "%s: -MT/-MQ require one of -M, -MM, -MD or -MMD\n", + opts->progname ); + return( -1 ); + } + } + + if( opts->print_deps_missing_files && + (opts->print_deps == 0 || !opts->inhibit_output) ) + { + fprintf( stderr, "%s: -MG may only be used with -M or -MM\n", + opts->progname ); + return( -1 ); + } + + return( 0 ); +} + +static const char * +action_name( enum mcpp_option_action_kind kind ) +{ + switch( kind ) + { + case MCPP_ACTION_DEFINE: return( "define" ); + case MCPP_ACTION_UNDEF: return( "undef" ); + case MCPP_ACTION_INCLUDE: return( "include" ); + case MCPP_ACTION_IMACROS: return( "imacros" ); + case MCPP_ACTION_INCLUDE_USER: return( "include-user" ); + case MCPP_ACTION_INCLUDE_SYSTEM: return( "include-system" ); + case MCPP_ACTION_INCLUDE_AFTER: return( "include-after" ); + case MCPP_ACTION_DEP_TARGET: return( "dep-target" ); + case MCPP_ACTION_DEP_TARGET_QUOTED: return( "dep-target-quoted" ); + } + return( "unknown" ); +} + +int +mcpp_options_dump( const mcpp_options *opts, FILE *stream ) +{ + size_t i; + + if( opts == NULL || stream == NULL ) return( -1 ); + + fprintf(stream, "input=%s\n", *opts->in_fname ? opts->in_fname : "<stdin>"); + fprintf(stream, "output=%s\n", *opts->out_fname ? opts->out_fname : "<stdout>"); + fprintf(stream, "verbose=%d\n", opts->verbose); + fprintf(stream, "nostdinc=%d\n", opts->no_standard_includes); + fprintf(stream, "dump_config=%d\n", opts->dump_config); + fprintf(stream, "dump_search_dirs=%d\n", opts->dump_search_dirs); + fprintf(stream, "dump_macros=%d\n", (int)opts->dump_macros); + fprintf(stream, "actions=%lu\n", (unsigned long)opts->action_count); + for( i = 0; i < opts->action_count; ++i ) + fprintf(stream, "action[%lu]=%s:%s\n", (unsigned long)i, + action_name(opts->actions[i].kind), opts->actions[i].argument); + return( 0 ); +} + +int +mcpp_options_has_unimplemented( const mcpp_options *opts ) +{ + size_t i; + if( opts == NULL ) return( 0 ); + if( opts->print_deps && opts->inhibit_output && + opts->dump_macros != MCPP_DUMP_NONE ) + return( 1 ); + if( opts->print_deps && !opts->inhibit_output && + opts->dump_macros == MCPP_DUMP_ONLY ) + return( 1 ); + if( opts->no_standard_kplusplus_includes || opts->for_lint ) + return( 1 ); + for( i = 0; i < opts->action_count; ++i ) + if( opts->actions[i].kind != MCPP_ACTION_DEFINE && + opts->actions[i].kind != MCPP_ACTION_UNDEF && + opts->actions[i].kind != MCPP_ACTION_INCLUDE && + opts->actions[i].kind != MCPP_ACTION_IMACROS && + opts->actions[i].kind != MCPP_ACTION_INCLUDE_USER && + opts->actions[i].kind != MCPP_ACTION_INCLUDE_SYSTEM && + opts->actions[i].kind != MCPP_ACTION_INCLUDE_AFTER && + opts->actions[i].kind != MCPP_ACTION_DEP_TARGET && + opts->actions[i].kind != MCPP_ACTION_DEP_TARGET_QUOTED ) + return( 1 ); + return( 0 ); +} diff --git a/src/mcpp-options.h b/src/mcpp-options.h new file mode 100644 index 0000000..910db92 --- /dev/null +++ b/src/mcpp-options.h @@ -0,0 +1,74 @@ +#ifndef __MCPU_CPP_OPTIONS_H__ +#define __MCPU_CPP_OPTIONS_H__ 1 + +enum mcpp_dump_macros +{ + MCPP_DUMP_NONE = 0, + MCPP_DUMP_ONLY, + MCPP_DUMP_DEFINITIONS +}; + +enum mcpp_option_action_kind +{ + MCPP_ACTION_DEFINE = 0, + MCPP_ACTION_UNDEF, + MCPP_ACTION_INCLUDE, + MCPP_ACTION_IMACROS, + MCPP_ACTION_INCLUDE_USER, + MCPP_ACTION_INCLUDE_SYSTEM, + MCPP_ACTION_INCLUDE_AFTER, + MCPP_ACTION_DEP_TARGET, + MCPP_ACTION_DEP_TARGET_QUOTED +}; + +typedef struct mcpp_option_action mcpp_option_action; +struct mcpp_option_action +{ + enum mcpp_option_action_kind kind; + const char *option; + const char *argument; +}; + +typedef struct mcpp_options mcpp_options; +struct mcpp_options +{ + const char *progname; + const char *in_fname; + const char *out_fname; + const char *config_fname; + const char *deps_file; + const char *deps_target; + const char *object_suffix; + + int help; + int version; + int verbose; + int no_config; + int dump_config; + int dump_search_dirs; + int no_standard_includes; + int no_standard_kplusplus_includes; + int inhibit_output; + int no_output; + int print_deps; + int print_deps_missing_files; + int inhibit_warnings; + int warn_comments; + int warn_comments_explicit; + int warnings_are_errors; + int for_lint; + enum mcpp_dump_macros dump_macros; + int dump_only_plus_predefined; + + mcpp_option_action *actions; + size_t action_count; + size_t action_capacity; +}; + +void mcpp_options_init( mcpp_options *opts, int argc ); +void mcpp_options_free( mcpp_options *opts ); +int mcpp_handle_options( mcpp_options *opts, int argc, char **argv ); +int mcpp_options_dump( const mcpp_options *opts, FILE *stream ); +int mcpp_options_has_unimplemented( const mcpp_options *opts ); + +#endif diff --git a/src/mcpp-predefined.c b/src/mcpp-predefined.c new file mode 100644 index 0000000..556232c --- /dev/null +++ b/src/mcpp-predefined.c @@ -0,0 +1,627 @@ +#include <defs.h> + + +static int +define_utf8( mcpu_macro_table *table, const char *name, + const char *replacement ) +{ + mcpu_text n; + mcpu_text r; + int rc = -1; + + mcpu_text_init( &n ); + mcpu_text_init( &r ); + + if( mcpu_text_from_utf8(&n, name) != 0 || + mcpu_text_from_utf8(&r, replacement ? replacement : "") != 0 ) + goto done; + + rc = mcpu_macro_define_object( table, n.data, n.length, + r.data, r.length, + MCPU_MACRO_ORIGIN_BUILTIN ); + +done: + mcpu_text_free( &n ); + mcpu_text_free( &r ); + return( rc ); +} + + +static int +define_unsigned( mcpu_macro_table *table, const char *name, + unsigned long value ) +{ + char text[64]; + + snprintf( text, sizeof(text), "%lu", value ); + return( define_utf8(table, name, text) ); +} + + +static int +define_quoted( mcpu_macro_table *table, const char *name, + const char *value ) +{ + size_t n; + char *text; + size_t i; + size_t o = 0; + int rc; + + if( value == NULL ) + value = ""; + + n = strlen( value ); + if( n > (SIZE_MAX - 3) / 2 ) + { + errno = EOVERFLOW; + return( -1 ); + } + + text = (char *)malloc( n * 2 + 3 ); + if( text == NULL ) + return( -1 ); + + text[o++] = '"'; + for( i = 0; i < n; ++i ) + { + if( value[i] == '\\' || value[i] == '"' ) + text[o++] = '\\'; + text[o++] = value[i]; + } + text[o++] = '"'; + text[o] = 0; + + rc = define_utf8( table, name, text ); + free( text ); + return( rc ); +} + + +static int +define_from_macro( mcpu_macro_table *table, const char *name, + const char *source_name ) +{ + mcpu_text source; + mcpu_text target; + mcpu_macro *macro; + int rc = -1; + + mcpu_text_init( &source ); + mcpu_text_init( &target ); + + if( mcpu_text_from_utf8(&source, source_name) != 0 || + mcpu_text_from_utf8(&target, name) != 0 ) + goto done; + + macro = mcpu_macro_find( table, source.data, source.length ); + if( macro == NULL || macro->builtin != MCPU_MACRO_BUILTIN_NONE ) + { + errno = EINVAL; + goto done; + } + + rc = mcpu_macro_define_object( table, target.data, target.length, + macro->replacement, + macro->replacement_length, + MCPU_MACRO_ORIGIN_BUILTIN ); + +done: + mcpu_text_free( &source ); + mcpu_text_free( &target ); + return( rc ); +} + + +static int +define_sized_builtin_utf8( mcpu_macro_table *table, const char *name, + enum mcpu_macro_builtin builtin, unsigned bits ) +{ + mcpu_text n; + int rc; + + mcpu_text_init( &n ); + if( mcpu_text_from_utf8(&n, name) != 0 ) + return( -1 ); + + rc = mcpu_macro_define_sized_builtin( table, n.data, n.length, + builtin, bits ); + mcpu_text_free( &n ); + return( rc ); +} + + +static unsigned long +mcpp_uint_decimal_digs( unsigned bits ) +{ + uint64_t n; + + if( bits < 1 ) + return( 0 ); + + /* + * Number of decimal digits in 2^bits - 1. This follows the integer-only + * scheme used by LibMPU _int_digs(), but does not include a terminating + * NUL. The tighter upper approximation keeps the result exact for every + * integer width supported by the current LibMPU family through 65536 bits. + * + * log10(2) < 301029995664 / 1000000000000 + */ + n = ((uint64_t)bits * UINT64_C(301029995664) + + UINT64_C(999999999999)) / UINT64_C(1000000000000); + + return( (unsigned long)n ); +} + + +static unsigned long +mcpp_int_decimal_digs( unsigned bits ) +{ + if( bits < 2 ) + return( 0 ); + + /* Signed maximum is 2^(bits - 1) - 1; sign is not part of DECIMAL_DIG. */ + return( mcpp_uint_decimal_digs(bits - 1) ); +} + + +static int +define_fixed_integer_family( mcpu_macro_table *table, unsigned bits ) +{ + size_t nb = (size_t)bits / 8; + mpu_int *value = NULL; + char *text = NULL; + char name[96]; + size_t text_size; + unsigned long int_digs; + unsigned long uint_digs; + int rc = -1; + + if( bits == 0 || (bits % 8) != 0 ) + { + errno = EINVAL; + return( -1 ); + } + + snprintf( name, sizeof(name), "__INT%u_TYPE__", bits ); + { + char type[64]; + snprintf( type, sizeof(type), "int%u", bits ); + if( define_utf8(table, name, type) != 0 ) + return( -1 ); + } + + snprintf( name, sizeof(name), "__UINT%u_TYPE__", bits ); + { + char type[64]; + snprintf( type, sizeof(type), "uint%u", bits ); + if( define_utf8(table, name, type) != 0 ) + return( -1 ); + } + + snprintf( name, sizeof(name), "__INT%u_WIDTH__", bits ); + if( define_unsigned(table, name, bits) != 0 ) + return( -1 ); + snprintf( name, sizeof(name), "__UINT%u_WIDTH__", bits ); + if( define_unsigned(table, name, bits) != 0 ) + return( -1 ); + + /* + * TYPE/WIDTH/SIZEOF and decimal-digit metadata remain available for the + * complete LibMPU integer family through NB_I_MAX * 8. Only textual MAX + * values are capped at 256 bits so -dMP stays compact and useful. + */ + snprintf( name, sizeof(name), "__SIZEOF_INT%u__", bits ); + if( define_unsigned(table, name, bits / 8) != 0 ) + return( -1 ); + snprintf( name, sizeof(name), "__SIZEOF_UINT%u__", bits ); + if( define_unsigned(table, name, bits / 8) != 0 ) + return( -1 ); + + int_digs = mcpp_int_decimal_digs( bits ); + uint_digs = mcpp_uint_decimal_digs( bits ); + if( int_digs == 0 || uint_digs == 0 ) + { + errno = EINVAL; + return( -1 ); + } + snprintf( name, sizeof(name), "__INT%u_DECIMAL_DIG__", bits ); + if( define_unsigned(table, name, int_digs) != 0 ) + return( -1 ); + snprintf( name, sizeof(name), "__UINT%u_DECIMAL_DIG__", bits ); + if( define_unsigned(table, name, uint_digs) != 0 ) + return( -1 ); + + if( bits > 256 ) + return( 0 ); + + value = (mpu_int *)malloc( nb ); + if( value == NULL ) + return( -1 ); + + text_size = (size_t)bits + 3; + text = (char *)calloc( text_size, 1 ); + if( text == NULL ) + goto done; + + memset( value, 0xff, nb ); + iuitoa( (__mpu_char8_t *)text, value, RADIX_HEX, LOWERCASE, (int)nb ); + + snprintf( name, sizeof(name), "__UINT%u_MAX__", bits ); + if( define_utf8(table, name, text) != 0 ) + goto done; + + /* + * iuitoa() is the canonical LibMPU conversion. For the signed maximum + * the all-ones unsigned bit pattern has only its top bit cleared. + */ + if( strlen(text) < 3 || text[0] != '0' || text[1] != 'x' ) + { + errno = EINVAL; + goto done; + } + text[2] = '7'; + + snprintf( name, sizeof(name), "__INT%u_MAX__", bits ); + if( define_utf8(table, name, text) != 0 ) + goto done; + + rc = 0; + +done: + free( text ); + free( value ); + return( rc ); +} + +typedef void (*real_value_fn)( mpu_real *, unsigned int, int ); + +static int +append_real_value( mcpu_text *output, unsigned bits, real_value_fn fn ) +{ + int nb = (int)(bits / 8); + int max_string; + mpu_real *value; + char *text; + int rc = -1; + + max_string = _real_max_string( nb ); + if( max_string <= 0 ) + { + errno = EINVAL; + return( -1 ); + } + + value = (mpu_real *)calloc( (size_t)nb, 1 ); + text = (char *)calloc( (size_t)max_string + 4096, 1 ); + if( value == NULL || text == NULL ) + goto done; + + fn( value, 0, nb ); + real_to_ascii( (__mpu_char8_t *)text, value, + _real_mant_digs(nb), 'e', 0, 0, nb ); + if( text[0] == 0 ) + { + errno = EINVAL; + goto done; + } + + rc = mcpu_text_append_ascii( output, text ); + +done: + free( text ); + free( value ); + return( rc ); +} + + +static void +real_epsilon_wrapper( mpu_real *value, unsigned int sign, int nb ) +{ + (void)sign; + _real_epsilon( value, nb ); +} + + +static int +append_real_exponent( mcpu_text *output, unsigned bits, + void (*fn)(mpu_int *, int, int) ) +{ + int nb_r = (int)(bits / 8); + int nb_e = _sizeof_exp( nb_r ); + mpu_int *value; + char *text; + size_t text_size; + int rc = -1; + + if( nb_e <= 0 ) + { + errno = EINVAL; + return( -1 ); + } + + value = (mpu_int *)calloc( (size_t)nb_e, 1 ); + text_size = (size_t)nb_e * 8 + 3; + text = (char *)calloc( text_size, 1 ); + if( value == NULL || text == NULL ) + goto done; + + fn( value, nb_e, nb_r ); + iitoa( (__mpu_char8_t *)text, value, RADIX_DEC, LOWERCASE, nb_e ); + if( text[0] == 0 ) + { + errno = EINVAL; + goto done; + } + + rc = mcpu_text_append_ascii( output, text ); + +done: + free( text ); + free( value ); + return( rc ); +} + + +static int +define_real_family( mcpu_macro_table *table, unsigned bits ) +{ + char name[96]; + char type[64]; + int nb = (int)(bits / 8); + + /* + * Real and Complex language types exist only through the configured + * MPU_REAL_IO_LIMIT. Complex WIDTH is the width parameter in the language + * type name (complex128 -> 128), while SIZEOF is twice the Real storage. + */ + if( bits > MCPU_CPP_MPU_REAL_IO_LIMIT ) + return( 0 ); + + snprintf( name, sizeof(name), "__REAL%u_TYPE__", bits ); + snprintf( type, sizeof(type), "real%u", bits ); + if( define_utf8(table, name, type) != 0 ) + return( -1 ); + + snprintf( name, sizeof(name), "__COMPLEX%u_TYPE__", bits ); + snprintf( type, sizeof(type), "complex%u", bits ); + if( define_utf8(table, name, type) != 0 ) + return( -1 ); + + snprintf( name, sizeof(name), "__REAL%u_WIDTH__", bits ); + if( define_unsigned(table, name, bits) != 0 ) + return( -1 ); + snprintf( name, sizeof(name), "__COMPLEX%u_WIDTH__", bits ); + if( define_unsigned(table, name, bits) != 0 ) + return( -1 ); + + snprintf( name, sizeof(name), "__SIZEOF_REAL%u__", bits ); + if( define_unsigned(table, name, bits / 8) != 0 ) + return( -1 ); + snprintf( name, sizeof(name), "__SIZEOF_COMPLEX%u__", bits ); + if( define_unsigned(table, name, bits / 4) != 0 ) + return( -1 ); + + /* + * These are compact structural/text-conversion properties. They are + * available for every configured Real type through MPU_REAL_IO_LIMIT. + * _real_max_string() returns a character count, not a byte count. + */ + { + int exp_size = _sizeof_exp( nb ); + int max_strlen = _real_max_string( nb ); + + if( exp_size <= 0 || max_strlen <= 0 ) + { + errno = EINVAL; + return( -1 ); + } + + snprintf( name, sizeof(name), "__SIZEOF_REAL%u_EXP__", bits ); + if( define_unsigned(table, name, (unsigned long)exp_size) != 0 ) + return( -1 ); + + snprintf( name, sizeof(name), "__REAL%u_MAX_STRLEN__", bits ); + if( define_unsigned(table, name, (unsigned long)max_strlen) != 0 ) + return( -1 ); + } + + /* + * Precision metadata is compact and is useful for every configured Real + * width. Only large textual numeric values are capped at 256 bits. + */ + snprintf( name, sizeof(name), "__REAL%u_MANT_DIG__", bits ); + if( define_unsigned(table, name, (unsigned long)_real_mant_digs(nb)) != 0 ) + return( -1 ); + + snprintf( name, sizeof(name), "__REAL%u_DECIMAL_DIG__", bits ); + if( define_unsigned(table, name, (unsigned long)_real_digs(nb)) != 0 ) + return( -1 ); + + if( bits > 256 ) + return( 0 ); + + snprintf( name, sizeof(name), "__REAL%u_MAX__", bits ); + if( define_sized_builtin_utf8(table, name, + MCPU_MACRO_BUILTIN_REAL_MAX, bits) != 0 ) + return( -1 ); + + snprintf( name, sizeof(name), "__REAL%u_MIN__", bits ); + if( define_sized_builtin_utf8(table, name, + MCPU_MACRO_BUILTIN_REAL_MIN, bits) != 0 ) + return( -1 ); + + snprintf( name, sizeof(name), "__REAL%u_EPSILON__", bits ); + if( define_sized_builtin_utf8(table, name, + MCPU_MACRO_BUILTIN_REAL_EPSILON, bits) != 0 ) + return( -1 ); + + snprintf( name, sizeof(name), "__REAL%u_MAX_EXP__", bits ); + if( define_sized_builtin_utf8(table, name, + MCPU_MACRO_BUILTIN_REAL_MAX_EXP, bits) != 0 ) + return( -1 ); + snprintf( name, sizeof(name), "__REAL%u_MIN_EXP__", bits ); + if( define_sized_builtin_utf8(table, name, + MCPU_MACRO_BUILTIN_REAL_MIN_EXP, bits) != 0 ) + return( -1 ); + snprintf( name, sizeof(name), "__REAL%u_MAX_10_EXP__", bits ); + if( define_sized_builtin_utf8(table, name, + MCPU_MACRO_BUILTIN_REAL_MAX_10_EXP, bits) != 0 ) + return( -1 ); + snprintf( name, sizeof(name), "__REAL%u_MIN_10_EXP__", bits ); + if( define_sized_builtin_utf8(table, name, + MCPU_MACRO_BUILTIN_REAL_MIN_10_EXP, bits) != 0 ) + return( -1 ); + + return( 0 ); +} + +int +mcpu_predefined_expand_builtin( enum mcpu_macro_builtin builtin, + unsigned bits, + mcpu_text *output ) +{ + if( output == NULL || bits == 0 || bits > 256 || + bits > MCPU_CPP_MPU_REAL_IO_LIMIT ) + { + errno = EINVAL; + return( -1 ); + } + + switch( builtin ) + { + case MCPU_MACRO_BUILTIN_REAL_MAX: + return( append_real_value(output, bits, _real_max) ); + case MCPU_MACRO_BUILTIN_REAL_MIN: + return( append_real_value(output, bits, _real_min) ); + case MCPU_MACRO_BUILTIN_REAL_EPSILON: + return( append_real_value(output, bits, real_epsilon_wrapper) ); + case MCPU_MACRO_BUILTIN_REAL_MAX_EXP: + return( append_real_exponent(output, bits, _real_max_exp) ); + case MCPU_MACRO_BUILTIN_REAL_MIN_EXP: + return( append_real_exponent(output, bits, _real_min_exp) ); + case MCPU_MACRO_BUILTIN_REAL_MAX_10_EXP: + return( append_real_exponent(output, bits, _real_max_10_exp) ); + case MCPU_MACRO_BUILTIN_REAL_MIN_10_EXP: + return( append_real_exponent(output, bits, _real_min_10_exp) ); + default: + errno = EINVAL; + return( -1 ); + } +} + + +int +mcpu_predefined_install( mcpu_macro_table *table ) +{ + unsigned bits; + char type[64]; + char ref[96]; + + if( table == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + if( define_utf8(table, "_ARCH_MCPU", "1") != 0 || + define_quoted(table, "__MCPU_CPP_VERSION__", PACKAGE_VERSION) != 0 || + define_utf8(table, "__REGISTER_PREFIX__", "") != 0 || + define_utf8(table, "__LOCAL_LABEL_PREFIX__", "") != 0 || + define_utf8(table, "__USER_LABEL_PREFIX__", "") != 0 || + define_utf8(table, "__IMMEDIATE_PREFIX__", "") != 0 || + define_utf8(table, "__ORDER_LITTLE_ENDIAN__", "1234") != 0 || + define_utf8(table, "__ORDER_BIG_ENDIAN__", "4321") != 0 || + define_utf8(table, "__ORDER_PDP_ENDIAN__", "3412") != 0 ) + return( -1 ); + +#if MCPU_CPP_MPU_BYTE_ORDER == 1234 + if( define_utf8(table, "__MCPU_BYTE_ORDER__", "__ORDER_LITTLE_ENDIAN__") != 0 ) + return( -1 ); +#else + if( define_utf8(table, "__MCPU_BYTE_ORDER__", "__ORDER_BIG_ENDIAN__") != 0 ) + return( -1 ); +#endif + if( define_utf8(table, "__BYTE_ORDER__", "__MCPU_BYTE_ORDER__") != 0 ) + return( -1 ); + +#if MCPU_CPP_MPU_WORD_ORDER == 1234 + if( define_utf8(table, "__MCPU_WORD_ORDER__", "__ORDER_LITTLE_ENDIAN__") != 0 ) + return( -1 ); +#else + if( define_utf8(table, "__MCPU_WORD_ORDER__", "__ORDER_BIG_ENDIAN__") != 0 ) + return( -1 ); +#endif + + if( define_unsigned(table, "__MCPU_MACHINE_REGISTER_WIDTH__", + MCPU_CPP_MPU_REGISTER_WIDTH) != 0 || + define_unsigned(table, "__MCPU_REAL_IO_LIMIT__", + MCPU_CPP_MPU_REAL_IO_LIMIT) != 0 || + define_unsigned(table, "__MCPU_MATH_FN_LIMIT__", + MCPU_CPP_MPU_MATH_FN_LIMIT) != 0 || + define_unsigned(table, "__MCPU_INT_MAX_WIDTH__", + MCPU_CPP_MPU_INT_MAX_WIDTH) != 0 || + define_unsigned(table, "__MCPU_REAL_MAX_WIDTH__", + MCPU_CPP_MPU_REAL_IO_LIMIT) != 0 || + define_unsigned(table, "__MCPU_COMPLEX_MAX_WIDTH__", + MCPU_CPP_MPU_REAL_IO_LIMIT) != 0 || + define_unsigned(table, "__SIZEOF_POINTER__", 8) != 0 || + define_unsigned(table, "__MCPU_POINTER_WIDTH__", 64) != 0 || + define_unsigned(table, "__MCPU_SIZEOF_SIZE__", MCPU_CPP_SIZEOF_SIZE_T) != 0 || + define_unsigned(table, "__MCPU_SIZE_WIDTH__", MCPU_CPP_SIZE_WIDTH) != 0 || + define_unsigned(table, "__SIZEOF_PTRDIFF__", 8) != 0 || + define_unsigned(table, "__PTRDIFF_WIDTH__", 64) != 0 || + define_utf8(table, "__CHAR8_TYPE__", "char8") != 0 || + define_utf8(table, "__CHAR16_TYPE__", "char16") != 0 || + define_unsigned(table, "__CHAR8_WIDTH__", 8) != 0 || + define_unsigned(table, "__CHAR16_WIDTH__", 16) != 0 || + define_unsigned(table, "__SIZEOF_CHAR8__", 1) != 0 || + define_unsigned(table, "__SIZEOF_CHAR16__", 2) != 0 || + define_utf8(table, "__CHAR8_MAX__", "0xff") != 0 || + define_utf8(table, "__CHAR16_MAX__", "0xffff") != 0 ) + return( -1 ); + + for( bits = 8; bits != 0 && bits <= MCPU_CPP_MPU_INT_MAX_WIDTH; bits *= 2 ) + { + if( define_fixed_integer_family(table, bits) != 0 ) + return( -1 ); + } + + snprintf( type, sizeof(type), "uint%u", (unsigned)MCPU_CPP_SIZE_WIDTH ); + if( define_utf8(table, "__MCPU_SIZE_TYPE__", type) != 0 ) + return( -1 ); + snprintf( ref, sizeof(ref), "__UINT%u_MAX__", (unsigned)MCPU_CPP_SIZE_WIDTH ); + if( define_from_macro(table, "__MCPU_SIZE_MAX__", ref) != 0 ) + return( -1 ); + + snprintf( type, sizeof(type), "int%u", (unsigned)MCPU_CPP_SSIZE_WIDTH ); + if( define_utf8(table, "__MCPU_SSIZE_TYPE__", type) != 0 || + define_unsigned(table, "__MCPU_SSIZE_WIDTH__", MCPU_CPP_SSIZE_WIDTH) != 0 || + define_unsigned(table, "__MCPU_SIZEOF_SSIZE__", MCPU_CPP_SIZEOF_SSIZE_T) != 0 ) + return( -1 ); + if( MCPU_CPP_SSIZE_WIDTH <= 256 ) + { + snprintf( ref, sizeof(ref), "__INT%u_MAX__", (unsigned)MCPU_CPP_SSIZE_WIDTH ); + if( define_from_macro(table, "__MCPU_SSIZE_MAX__", ref) != 0 ) + return( -1 ); + } + + if( define_utf8(table, "__PTRDIFF_TYPE__", "int64") != 0 || + define_unsigned(table, "__PTRDIFF_WIDTH__", 64) != 0 || + define_utf8(table, "__PTRDIFF_MAX__", "0x7fffffffffffffff") != 0 || + define_utf8(table, "__INTPTR_TYPE__", "int64") != 0 || + define_utf8(table, "__UINTPTR_TYPE__", "uint64") != 0 || + define_unsigned(table, "__INTPTR_WIDTH__", 64) != 0 || + define_unsigned(table, "__UINTPTR_WIDTH__", 64) != 0 || + define_utf8(table, "__INTPTR_MAX__", "0x7fffffffffffffff") != 0 || + define_utf8(table, "__UINTPTR_MAX__", "0xffffffffffffffff") != 0 ) + return( -1 ); + + for( bits = 32; bits != 0 && bits <= MCPU_CPP_MPU_REAL_IO_LIMIT; bits *= 2 ) + { + if( define_real_family(table, bits) != 0 ) + return( -1 ); + } + + return( 0 ); +} diff --git a/src/mcpp-predefined.h b/src/mcpp-predefined.h new file mode 100644 index 0000000..4010ae0 --- /dev/null +++ b/src/mcpp-predefined.h @@ -0,0 +1,12 @@ +#ifndef __MCPU_CPP_PREDEFINED_H__ +#define __MCPU_CPP_PREDEFINED_H__ 1 + +#include <defs.h> +#include <mcpp-macro.h> + +int mcpu_predefined_install( mcpu_macro_table *table ); +int mcpu_predefined_expand_builtin( enum mcpu_macro_builtin builtin, + unsigned bits, + mcpu_text *output ); + +#endif /* __MCPU_CPP_PREDEFINED_H__ */ diff --git a/src/mcpp-runtime.c b/src/mcpp-runtime.c new file mode 100644 index 0000000..f498b20 --- /dev/null +++ b/src/mcpp-runtime.c @@ -0,0 +1,268 @@ +#include <defs.h> + +#include <unistd.h> + + +static char * +path_dirname( const char *path ) +{ + const char *slash; + size_t length; + char *result; + + if( path == NULL || *path == 0 ) + { + errno = EINVAL; + return( NULL ); + } + + slash = strrchr( path, '/' ); + if( slash == NULL ) + { + errno = EINVAL; + return( NULL ); + } + + if( slash == path ) + length = 1; + else + length = (size_t)(slash - path); + + result = (char *)malloc( length + 1 ); + if( result == NULL ) + return( NULL ); + + memcpy( result, path, length ); + result[length] = 0; + return( result ); +} + + +static char * +path_join( const char *directory, const char *name ) +{ + size_t directory_length; + size_t name_length; + int need_slash; + char *result; + + if( directory == NULL || name == NULL ) + { + errno = EINVAL; + return( NULL ); + } + + directory_length = strlen( directory ); + name_length = strlen( name ); + need_slash = directory_length != 0 && directory[directory_length - 1] != '/'; + + if( directory_length > SIZE_MAX - name_length - (size_t)need_slash - 1 ) + { + errno = EOVERFLOW; + return( NULL ); + } + + result = (char *)malloc( directory_length + name_length + + (size_t)need_slash + 1 ); + if( result == NULL ) + return( NULL ); + + memcpy( result, directory, directory_length ); + if( need_slash ) + result[directory_length++] = '/'; + memcpy( result + directory_length, name, name_length ); + result[directory_length + name_length] = 0; + return( result ); +} + + +static int +resolve_proc_self_exe( char path[PATH_MAX] ) +{ + ssize_t length; + + length = readlink( "/proc/self/exe", path, PATH_MAX - 1 ); + if( length < 0 ) + return( -1 ); + + if( length >= PATH_MAX - 1 ) + { + errno = ENAMETOOLONG; + return( -1 ); + } + + path[length] = 0; + return( 0 ); +} + + +static int +resolve_candidate( const char *candidate, char path[PATH_MAX] ) +{ + if( candidate == NULL || *candidate == 0 ) + { + errno = EINVAL; + return( -1 ); + } + + if( realpath(candidate, path) == NULL ) + return( -1 ); + + if( access(path, X_OK) != 0 ) + return( -1 ); + + return( 0 ); +} + + +static int +resolve_argv0( const char *argv0, char path[PATH_MAX] ) +{ + const char *search; + const char *entry; + + if( argv0 == NULL || *argv0 == 0 ) + { + errno = EINVAL; + return( -1 ); + } + + if( strchr(argv0, '/') != NULL ) + return( resolve_candidate(argv0, path) ); + + search = getenv( "PATH" ); + if( search == NULL ) + { + errno = ENOENT; + return( -1 ); + } + + entry = search; + for( ;; ) + { + const char *colon = strchr( entry, ':' ); + size_t length = colon ? (size_t)(colon - entry) : strlen(entry); + const char *directory = entry; + char directory_buffer[PATH_MAX]; + char *candidate; + int rc; + + if( length == 0 ) + directory = "."; + else + { + if( length >= sizeof(directory_buffer) ) + { + errno = ENAMETOOLONG; + return( -1 ); + } + memcpy( directory_buffer, entry, length ); + directory_buffer[length] = 0; + directory = directory_buffer; + } + + candidate = path_join( directory, argv0 ); + if( candidate == NULL ) + return( -1 ); + rc = resolve_candidate( candidate, path ); + free( candidate ); + if( rc == 0 ) + return( 0 ); + + if( colon == NULL ) + break; + entry = colon + 1; + } + + errno = ENOENT; + return( -1 ); +} + + +void +mcpp_runtime_paths_init( mcpp_runtime_paths *paths ) +{ + if( paths == NULL ) + return; + + paths->executable_path = NULL; + paths->executable_dir = NULL; + paths->installation_root = NULL; + paths->config_file = NULL; + paths->system_include_path = NULL; +} + + +void +mcpp_runtime_paths_free( mcpp_runtime_paths *paths ) +{ + if( paths == NULL ) + return; + + free( paths->executable_path ); + free( paths->executable_dir ); + free( paths->installation_root ); + free( paths->config_file ); + free( paths->system_include_path ); + mcpp_runtime_paths_init( paths ); +} + + +int +mcpp_runtime_paths_discover( mcpp_runtime_paths *paths, const char *argv0 ) +{ + char resolved[PATH_MAX]; + char *executable_path = NULL; + char *executable_dir = NULL; + char *installation_root = NULL; + char *config_file = NULL; + char *system_include_path = NULL; + + if( paths == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + if( resolve_proc_self_exe(resolved) != 0 ) + { + if( resolve_argv0(argv0, resolved) != 0 ) + return( -1 ); + } + + executable_path = strdup( resolved ); + if( executable_path == NULL ) + goto fail; + + executable_dir = path_dirname( executable_path ); + if( executable_dir == NULL ) + goto fail; + + installation_root = path_dirname( executable_dir ); + if( installation_root == NULL ) + goto fail; + + config_file = path_join( installation_root, "etc/mcpu-cpp.conf" ); + if( config_file == NULL ) + goto fail; + + system_include_path = path_join( installation_root, "include" ); + if( system_include_path == NULL ) + goto fail; + + mcpp_runtime_paths_free( paths ); + paths->executable_path = executable_path; + paths->executable_dir = executable_dir; + paths->installation_root = installation_root; + paths->config_file = config_file; + paths->system_include_path = system_include_path; + return( 0 ); + +fail: + free( executable_path ); + free( executable_dir ); + free( installation_root ); + free( config_file ); + free( system_include_path ); + return( -1 ); +} diff --git a/src/mcpp-runtime.h b/src/mcpp-runtime.h new file mode 100644 index 0000000..75d1ddf --- /dev/null +++ b/src/mcpp-runtime.h @@ -0,0 +1,19 @@ +#ifndef __MCPU_CPP_RUNTIME_H__ +#define __MCPU_CPP_RUNTIME_H__ 1 + +typedef struct mcpp_runtime_paths mcpp_runtime_paths; +struct mcpp_runtime_paths +{ + char *executable_path; + char *executable_dir; + char *installation_root; + char *config_file; + char *system_include_path; +}; + +void mcpp_runtime_paths_init( mcpp_runtime_paths *paths ); +void mcpp_runtime_paths_free( mcpp_runtime_paths *paths ); +int mcpp_runtime_paths_discover( mcpp_runtime_paths *paths, + const char *argv0 ); + +#endif /* __MCPU_CPP_RUNTIME_H__ */ diff --git a/src/mcpp-semantic.c b/src/mcpp-semantic.c new file mode 100644 index 0000000..3063139 --- /dev/null +++ b/src/mcpp-semantic.c @@ -0,0 +1,349 @@ +#include <defs.h> + +#define MCPP_INTEGER_BITS 64U + +static __mpu_int64_t +mcpp_semantic_signed( mcpp_integer value ) +{ + if( value.value <= (__mpu_uint64_t)INT64_MAX ) + return( (__mpu_int64_t)value.value ); + + return( -1 - (__mpu_int64_t)(UINT64_MAX - value.value) ); +} + +static __mpu_uint64_t +mcpp_semantic_unsigned_from_signed( __mpu_int64_t value ) +{ + if( value >= 0 ) + return( (__mpu_uint64_t)value ); + + return( UINT64_MAX - (__mpu_uint64_t)(-(value + 1)) ); +} + +static unsigned +mcpp_semantic_shift_count( mcpp_integer value, int *negative ) +{ + __mpu_uint64_t magnitude; + + *negative = 0; + if( !value.unsignedp && mcpp_semantic_signed(value) < 0 ) + { + *negative = 1; + magnitude = (__mpu_uint64_t)0 - value.value; + } + else + magnitude = value.value; + + if( magnitude >= MCPP_INTEGER_BITS ) + return( MCPP_INTEGER_BITS ); + + return( (unsigned)magnitude ); +} + +static mcpp_integer +mcpp_semantic_shift_left_count( mcpp_integer value, unsigned shift ) +{ + mcpp_integer result; + + result.unsignedp = value.unsignedp; + if( shift >= MCPP_INTEGER_BITS ) + result.value = 0; + else + result.value = value.value << shift; + + return( result ); +} + +static mcpp_integer +mcpp_semantic_shift_right_count( mcpp_integer value, unsigned shift ) +{ + mcpp_integer result; + int negative; + + result.unsignedp = value.unsignedp; + negative = !value.unsignedp && mcpp_semantic_signed(value) < 0; + + if( shift >= MCPP_INTEGER_BITS ) + { + result.value = negative ? UINT64_MAX : 0; + return( result ); + } + + if( shift == 0 || value.unsignedp || !negative ) + { + result.value = value.value >> shift; + return( result ); + } + + result.value = (value.value >> shift) | + (UINT64_MAX << (MCPP_INTEGER_BITS - shift)); + return( result ); +} + +void +mcpp_semantic_context_init( mcpp_semantic_context *context, + const char *filename, + unsigned line_number, + const mcpp_options *options ) +{ + context->filename = filename; + context->line_number = line_number; + context->options = options; + context->failed = 0; +} + +void +mcpp_semantic_error( mcpp_semantic_context *context, const char *message ) +{ + if( context->failed ) + return; + + fprintf( stderr, "%s:%u: error: %s\n", + context->filename, context->line_number, message ); + context->failed = 1; +} + +void +mcpp_semantic_warning( mcpp_semantic_context *context, + const char *message ) +{ + if( mcpp_diagnostic_warning(context->options, + context->filename, context->line_number, + "%s", message) != 0 ) + context->failed = 1; +} + +int +mcpp_semantic_failed( const mcpp_semantic_context *context ) +{ + return( context->failed ); +} + +mcpp_integer +mcpp_semantic_make( __mpu_uint64_t value, int unsignedp ) +{ + mcpp_integer result; + + result.value = value; + result.unsignedp = unsignedp ? 1 : 0; + return( result ); +} + +int +mcpp_semantic_true( mcpp_integer value ) +{ + return( value.value != 0 ); +} + +mcpp_integer +mcpp_semantic_neg( mcpp_integer value ) +{ + value.value = (__mpu_uint64_t)0 - value.value; + return( value ); +} + +mcpp_integer +mcpp_semantic_not( mcpp_integer value ) +{ + return( mcpp_semantic_make(!mcpp_semantic_true(value), 0) ); +} + +mcpp_integer +mcpp_semantic_compl( mcpp_integer value ) +{ + value.value = ~value.value; + return( value ); +} + +mcpp_integer +mcpp_semantic_mul( mcpp_integer left, mcpp_integer right ) +{ + return( mcpp_semantic_make(left.value * right.value, + left.unsignedp || right.unsignedp) ); +} + +mcpp_integer +mcpp_semantic_div( mcpp_semantic_context *context, + mcpp_integer left, mcpp_integer right, int evaluate ) +{ + int unsignedp = left.unsignedp || right.unsignedp; + + if( right.value == 0 ) + { + if( evaluate ) + mcpp_semantic_error( context, "division by zero in #if expression" ); + return( mcpp_semantic_make(0, unsignedp) ); + } + + if( unsignedp ) + return( mcpp_semantic_make(left.value / right.value, 1) ); + + if( mcpp_semantic_signed(left) == INT64_MIN && + mcpp_semantic_signed(right) == -1 ) + return( mcpp_semantic_make((__mpu_uint64_t)1 << 63, 0) ); + + return( mcpp_semantic_make( + mcpp_semantic_unsigned_from_signed( + mcpp_semantic_signed(left) / mcpp_semantic_signed(right)), 0) ); +} + +mcpp_integer +mcpp_semantic_mod( mcpp_semantic_context *context, + mcpp_integer left, mcpp_integer right, int evaluate ) +{ + int unsignedp = left.unsignedp || right.unsignedp; + + if( right.value == 0 ) + { + if( evaluate ) + mcpp_semantic_error( context, "division by zero in #if expression" ); + return( mcpp_semantic_make(0, unsignedp) ); + } + + if( unsignedp ) + return( mcpp_semantic_make(left.value % right.value, 1) ); + + if( mcpp_semantic_signed(left) == INT64_MIN && + mcpp_semantic_signed(right) == -1 ) + return( mcpp_semantic_make(0, 0) ); + + return( mcpp_semantic_make( + mcpp_semantic_unsigned_from_signed( + mcpp_semantic_signed(left) % mcpp_semantic_signed(right)), 0) ); +} + +mcpp_integer +mcpp_semantic_add( mcpp_integer left, mcpp_integer right ) +{ + return( mcpp_semantic_make(left.value + right.value, + left.unsignedp || right.unsignedp) ); +} + +mcpp_integer +mcpp_semantic_sub( mcpp_integer left, mcpp_integer right ) +{ + return( mcpp_semantic_make(left.value - right.value, + left.unsignedp || right.unsignedp) ); +} + +mcpp_integer +mcpp_semantic_lshift( mcpp_integer left, mcpp_integer right ) +{ + unsigned shift; + int negative; + + shift = mcpp_semantic_shift_count( right, &negative ); + if( negative ) + return( mcpp_semantic_shift_right_count(left, shift) ); + + return( mcpp_semantic_shift_left_count(left, shift) ); +} + +mcpp_integer +mcpp_semantic_rshift( mcpp_integer left, mcpp_integer right ) +{ + unsigned shift; + int negative; + + shift = mcpp_semantic_shift_count( right, &negative ); + if( negative ) + return( mcpp_semantic_shift_left_count(left, shift) ); + + return( mcpp_semantic_shift_right_count(left, shift) ); +} + +mcpp_integer +mcpp_semantic_equal( mcpp_integer left, mcpp_integer right ) +{ + int unsignedp = left.unsignedp || right.unsignedp; + int value = unsignedp ? left.value == right.value : + mcpp_semantic_signed(left) == mcpp_semantic_signed(right); + return( mcpp_semantic_make((__mpu_uint64_t)value, 0) ); +} + +mcpp_integer +mcpp_semantic_notequal( mcpp_integer left, mcpp_integer right ) +{ + mcpp_integer result = mcpp_semantic_equal( left, right ); + result.value = !result.value; + return( result ); +} + +mcpp_integer +mcpp_semantic_less( mcpp_integer left, mcpp_integer right ) +{ + int unsignedp = left.unsignedp || right.unsignedp; + int value = unsignedp ? left.value < right.value : + mcpp_semantic_signed(left) < mcpp_semantic_signed(right); + return( mcpp_semantic_make((__mpu_uint64_t)value, 0) ); +} + +mcpp_integer +mcpp_semantic_greater( mcpp_integer left, mcpp_integer right ) +{ + int unsignedp = left.unsignedp || right.unsignedp; + int value = unsignedp ? left.value > right.value : + mcpp_semantic_signed(left) > mcpp_semantic_signed(right); + return( mcpp_semantic_make((__mpu_uint64_t)value, 0) ); +} + +mcpp_integer +mcpp_semantic_leq( mcpp_integer left, mcpp_integer right ) +{ + int unsignedp = left.unsignedp || right.unsignedp; + int value = unsignedp ? left.value <= right.value : + mcpp_semantic_signed(left) <= mcpp_semantic_signed(right); + return( mcpp_semantic_make((__mpu_uint64_t)value, 0) ); +} + +mcpp_integer +mcpp_semantic_geq( mcpp_integer left, mcpp_integer right ) +{ + int unsignedp = left.unsignedp || right.unsignedp; + int value = unsignedp ? left.value >= right.value : + mcpp_semantic_signed(left) >= mcpp_semantic_signed(right); + return( mcpp_semantic_make((__mpu_uint64_t)value, 0) ); +} + +mcpp_integer +mcpp_semantic_bitand( mcpp_integer left, mcpp_integer right ) +{ + return( mcpp_semantic_make(left.value & right.value, + left.unsignedp || right.unsignedp) ); +} + +mcpp_integer +mcpp_semantic_bitxor( mcpp_integer left, mcpp_integer right ) +{ + return( mcpp_semantic_make(left.value ^ right.value, + left.unsignedp || right.unsignedp) ); +} + +mcpp_integer +mcpp_semantic_bitor( mcpp_integer left, mcpp_integer right ) +{ + return( mcpp_semantic_make(left.value | right.value, + left.unsignedp || right.unsignedp) ); +} + +mcpp_integer +mcpp_semantic_and( mcpp_integer left, mcpp_integer right ) +{ + return( mcpp_semantic_make(mcpp_semantic_true(left) && mcpp_semantic_true(right), 0) ); +} + +mcpp_integer +mcpp_semantic_or( mcpp_integer left, mcpp_integer right ) +{ + return( mcpp_semantic_make(mcpp_semantic_true(left) || mcpp_semantic_true(right), 0) ); +} + +mcpp_integer +mcpp_semantic_conditional( mcpp_integer condition, + mcpp_integer yes, mcpp_integer no ) +{ + mcpp_integer result = mcpp_semantic_true(condition) ? yes : no; + + result.unsignedp = yes.unsignedp || no.unsignedp; + return( result ); +} diff --git a/src/mcpp-semantic.h b/src/mcpp-semantic.h new file mode 100644 index 0000000..efa793e --- /dev/null +++ b/src/mcpp-semantic.h @@ -0,0 +1,67 @@ +#ifndef __MCPU_CPP_SEMANTIC_H__ +#define __MCPU_CPP_SEMANTIC_H__ 1 + +#include <defs.h> + +typedef struct mcpp_options mcpp_options; +typedef struct mcpp_integer mcpp_integer; +struct mcpp_integer +{ + __mpu_uint64_t value; + int unsignedp; +}; + +typedef struct mcpp_semantic_context mcpp_semantic_context; +struct mcpp_semantic_context +{ + const char *filename; + unsigned line_number; + const mcpp_options *options; + int failed; +}; + +void mcpp_semantic_context_init( mcpp_semantic_context *context, + const char *filename, + unsigned line_number, + const mcpp_options *options ); +void mcpp_semantic_error( mcpp_semantic_context *context, + const char *message ); +void mcpp_semantic_warning( mcpp_semantic_context *context, + const char *message ); +int mcpp_semantic_failed( const mcpp_semantic_context *context ); + +mcpp_integer mcpp_semantic_make( __mpu_uint64_t value, int unsignedp ); +int mcpp_semantic_true( mcpp_integer value ); + +mcpp_integer mcpp_semantic_neg( mcpp_integer value ); +mcpp_integer mcpp_semantic_not( mcpp_integer value ); +mcpp_integer mcpp_semantic_compl( mcpp_integer value ); + +mcpp_integer mcpp_semantic_mul( mcpp_integer left, mcpp_integer right ); +mcpp_integer mcpp_semantic_div( mcpp_semantic_context *context, + mcpp_integer left, mcpp_integer right, + int evaluate ); +mcpp_integer mcpp_semantic_mod( mcpp_semantic_context *context, + mcpp_integer left, mcpp_integer right, + int evaluate ); +mcpp_integer mcpp_semantic_add( mcpp_integer left, mcpp_integer right ); +mcpp_integer mcpp_semantic_sub( mcpp_integer left, mcpp_integer right ); +mcpp_integer mcpp_semantic_lshift( mcpp_integer left, mcpp_integer right ); +mcpp_integer mcpp_semantic_rshift( mcpp_integer left, mcpp_integer right ); + +mcpp_integer mcpp_semantic_equal( mcpp_integer left, mcpp_integer right ); +mcpp_integer mcpp_semantic_notequal( mcpp_integer left, mcpp_integer right ); +mcpp_integer mcpp_semantic_less( mcpp_integer left, mcpp_integer right ); +mcpp_integer mcpp_semantic_greater( mcpp_integer left, mcpp_integer right ); +mcpp_integer mcpp_semantic_leq( mcpp_integer left, mcpp_integer right ); +mcpp_integer mcpp_semantic_geq( mcpp_integer left, mcpp_integer right ); +mcpp_integer mcpp_semantic_bitand( mcpp_integer left, mcpp_integer right ); +mcpp_integer mcpp_semantic_bitxor( mcpp_integer left, mcpp_integer right ); +mcpp_integer mcpp_semantic_bitor( mcpp_integer left, mcpp_integer right ); +mcpp_integer mcpp_semantic_and( mcpp_integer left, mcpp_integer right ); +mcpp_integer mcpp_semantic_or( mcpp_integer left, mcpp_integer right ); +mcpp_integer mcpp_semantic_conditional( mcpp_integer condition, + mcpp_integer yes, + mcpp_integer no ); + +#endif /* __MCPU_CPP_SEMANTIC_H__ */ diff --git a/src/mcpp-source.c b/src/mcpp-source.c new file mode 100644 index 0000000..c40e370 --- /dev/null +++ b/src/mcpp-source.c @@ -0,0 +1,332 @@ +#include <defs.h> + +static char * +read_stream_bytes( FILE *fp, size_t *file_length ) +{ + char *data = NULL; + size_t length = 0; + size_t capacity = 0; + + if( fp == NULL || file_length == NULL ) + { + errno = EINVAL; + return( NULL ); + } + + *file_length = 0; + + for( ;; ) + { + size_t n; + + if( capacity - length < 4096 ) + { + size_t new_capacity = capacity ? capacity * 2 : 8192; + char *q = (char *)realloc( data, new_capacity + 1 ); + if( q == NULL ) + { + free( data ); + return( NULL ); + } + data = q; + capacity = new_capacity; + } + + n = fread( data + length, 1, capacity - length, fp ); + length += n; + + if( n == 0 ) + { + if( ferror(fp) ) + { + free( data ); + return( NULL ); + } + break; + } + } + + if( data == NULL ) + { + data = (char *)calloc( 1, 1 ); + if( data == NULL ) + return( NULL ); + } + else + data[length] = 0; + + *file_length = length; + return( data ); +} + +static int +normalize_newlines( mcpu_text *text ) +{ + size_t r; + size_t w = 0; + + if( text == NULL || text->data == NULL ) + return( 0 ); + + for( r = 0; r < text->length; ++r ) + { + if( text->data[r] == '\r' ) + { + if( r + 1 < text->length && text->data[r + 1] == '\n' ) + ++r; + text->data[w++] = '\n'; + } + else + text->data[w++] = text->data[r]; + } + + text->length = w; + text->data[w] = 0; + + return( 0 ); +} + +void +mcpu_source_init( mcpu_source *source ) +{ + if( source == NULL ) + return; + + source->filename = NULL; + mcpu_text_init( &source->text ); + mcpu_text_init( &source->logical_line ); + source->splice_offsets = NULL; + source->splice_count = 0; + source->splice_capacity = 0; + source->offset = 0; + source->line = 1; +} + +void +mcpu_source_free( mcpu_source *source ) +{ + if( source == NULL ) + return; + + free( source->filename ); + source->filename = NULL; + mcpu_text_free( &source->text ); + mcpu_text_free( &source->logical_line ); + free( source->splice_offsets ); + source->splice_offsets = NULL; + source->splice_count = 0; + source->splice_capacity = 0; + source->offset = 0; + source->line = 1; +} + +static int +source_from_bytes( mcpu_source *source, char *bytes, size_t byte_length, + const char *name ) +{ + const __mpu_char8_t *input = (const __mpu_char8_t *)bytes; + const __mpu_char8_t *p; + + if( memchr( bytes, 0, byte_length ) != NULL ) + { + fprintf( stderr, "%s: NUL character is not allowed in source text\n", name ); + return( -1 ); + } + + if( byte_length >= 3 && + (unsigned char)bytes[0] == 0xef && + (unsigned char)bytes[1] == 0xbb && + (unsigned char)bytes[2] == 0xbf ) + input += 3; + + if( !mpu_utf8valid(input) ) + { + fprintf( stderr, "%s: invalid UTF-8\n", name ); + return( -1 ); + } + + p = input; + while( *p != 0 ) + { + __mpu_char32_t value; + const __mpu_char8_t *next; + __mpu_char16_t c; + + next = mpu_utf8get( p, &value ); + if( next == NULL ) + { + fprintf( stderr, "%s: invalid UTF-8\n", name ); + return( -1 ); + } + + if( value <= 0xffff ) + c = (__mpu_char16_t)value; + else + c = MCPU_CPP_NON_UCS2_SENTINEL; + + if( mcpu_text_append_char(&source->text, c) != 0 ) + return( -1 ); + + p = next; + } + + if( normalize_newlines(&source->text) != 0 ) + return( -1 ); + + source->filename = strdup( name ); + if( source->filename == NULL ) + return( -1 ); + + return( 0 ); +} + +int +mcpu_source_valid_ucs2( const __mpu_char16_t *text, size_t length ) +{ + size_t i; + + if( text == NULL && length != 0 ) + return( 0 ); + + for( i = 0; i < length; ++i ) + if( text[i] == MCPU_CPP_NON_UCS2_SENTINEL ) + return( 0 ); + + return( 1 ); +} + +int +mcpu_source_open_stream( mcpu_source *source, FILE *stream, const char *name ) +{ + char *bytes; + size_t byte_length; + int rc; + + if( source == NULL || stream == NULL || name == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + mcpu_source_free( source ); + mcpu_source_init( source ); + + bytes = read_stream_bytes( stream, &byte_length ); + if( bytes == NULL ) + return( -1 ); + + rc = source_from_bytes( source, bytes, byte_length, name ); + free( bytes ); + return( rc ); +} + +int +mcpu_source_open( mcpu_source *source, const char *filename ) +{ + FILE *fp; + int rc; + + if( source == NULL || filename == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + fp = fopen( filename, "rb" ); + if( fp == NULL ) + return( -1 ); + + rc = mcpu_source_open_stream( source, fp, filename ); + if( fclose(fp) != 0 && rc == 0 ) + rc = -1; + + return( rc ); +} + +static int +append_splice_offset( mcpu_source *source, size_t offset ) +{ + size_t *offsets; + size_t capacity; + + if( source->splice_count < source->splice_capacity ) + { + source->splice_offsets[source->splice_count++] = offset; + return( 0 ); + } + + capacity = source->splice_capacity ? source->splice_capacity * 2 : 8; + offsets = (size_t *)realloc( source->splice_offsets, + capacity * sizeof(*offsets) ); + if( offsets == NULL ) + return( -1 ); + + source->splice_offsets = offsets; + source->splice_capacity = capacity; + source->splice_offsets[source->splice_count++] = offset; + return( 0 ); +} + + +int +mcpu_source_next_line( mcpu_source *source, + const __mpu_char16_t **line, + size_t *length, + unsigned *line_number, + unsigned *next_line_number, + int *spliced ) +{ + unsigned first_line; + + if( source == NULL || line == NULL || length == NULL || + line_number == NULL || next_line_number == NULL || spliced == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + if( source->offset >= source->text.length ) + return( 0 ); + + mcpu_text_free( &source->logical_line ); + mcpu_text_init( &source->logical_line ); + source->splice_count = 0; + first_line = source->line; + *spliced = 0; + + while( source->offset < source->text.length ) + { + __mpu_char16_t c = source->text.data[source->offset++]; + + /* + * Backslash-newline deletion is preprocessing phase 2 and therefore + * happens before comments, directives and macro recognition. + */ + if( c == '\\' && source->offset < source->text.length && + source->text.data[source->offset] == '\n' ) + { + if( append_splice_offset(source, source->logical_line.length) != 0 ) + return( -1 ); + ++source->offset; + ++source->line; + *spliced = 1; + continue; + } + + if( mcpu_text_append_char( &source->logical_line, c ) != 0 ) + return( -1 ); + + if( c == '\n' ) + { + ++source->line; + break; + } + } + + *line = source->logical_line.data; + *length = source->logical_line.length; + *line_number = first_line; + *next_line_number = source->line; + + return( 1 ); +} diff --git a/src/mcpp-source.h b/src/mcpp-source.h new file mode 100644 index 0000000..9a7973a --- /dev/null +++ b/src/mcpp-source.h @@ -0,0 +1,35 @@ +#ifndef __MCPU_CPP_SOURCE_H__ +#define __MCPU_CPP_SOURCE_H__ 1 + +#include <defs.h> +#include <mcpp-text.h> + +#define MCPU_CPP_NON_UCS2_SENTINEL 0xd800 + +typedef struct mcpu_source mcpu_source; +struct mcpu_source +{ + char *filename; + mcpu_text text; + mcpu_text logical_line; + size_t *splice_offsets; + size_t splice_count; + size_t splice_capacity; + size_t offset; + unsigned line; +}; + +void mcpu_source_init( mcpu_source *source ); +void mcpu_source_free( mcpu_source *source ); +int mcpu_source_open( mcpu_source *source, const char *filename ); +int mcpu_source_open_stream( mcpu_source *source, FILE *stream, + const char *name ); +int mcpu_source_valid_ucs2( const __mpu_char16_t *text, size_t length ); +int mcpu_source_next_line( mcpu_source *source, + const __mpu_char16_t **line, + size_t *length, + unsigned *line_number, + unsigned *next_line_number, + int *spliced ); + +#endif /* __MCPU_CPP_SOURCE_H__ */ diff --git a/src/mcpp-text.c b/src/mcpp-text.c new file mode 100644 index 0000000..febbd3b --- /dev/null +++ b/src/mcpp-text.c @@ -0,0 +1,237 @@ +#include <defs.h> + +void +mcpu_text_init( mcpu_text *text ) +{ + if( text == NULL ) + return; + + text->data = NULL; + text->length = 0; + text->capacity = 0; +} + +void +mcpu_text_free( mcpu_text *text ) +{ + if( text == NULL ) + return; + + free( text->data ); + text->data = NULL; + text->length = 0; + text->capacity = 0; +} + +int +mcpu_text_reserve( mcpu_text *text, size_t need ) +{ + __mpu_char16_t *p; + size_t capacity; + + if( text == NULL ) + return( -1 ); + + if( need <= text->capacity ) + return( 0 ); + + capacity = text->capacity ? text->capacity : 256; + while( capacity < need ) + { + if( capacity > (SIZE_MAX / 2) ) + { + errno = EOVERFLOW; + return( -1 ); + } + capacity *= 2; + } + + if( capacity > (SIZE_MAX / sizeof(__mpu_char16_t)) ) + { + errno = EOVERFLOW; + return( -1 ); + } + + p = (__mpu_char16_t *)realloc( text->data, + capacity * sizeof(__mpu_char16_t) ); + if( p == NULL ) + return( -1 ); + + text->data = p; + text->capacity = capacity; + + return( 0 ); +} + +int +mcpu_text_append( mcpu_text *text, + const __mpu_char16_t *data, size_t length ) +{ + if( text == NULL || (data == NULL && length != 0) ) + { + errno = EINVAL; + return( -1 ); + } + + if( length > SIZE_MAX - text->length - 1 ) + { + errno = EOVERFLOW; + return( -1 ); + } + + if( mcpu_text_reserve( text, text->length + length + 1 ) != 0 ) + return( -1 ); + + if( length != 0 ) + memcpy( text->data + text->length, data, + length * sizeof(__mpu_char16_t) ); + + text->length += length; + text->data[text->length] = 0; + + return( 0 ); +} + +int +mcpu_text_append_char( mcpu_text *text, __mpu_char16_t ch ) +{ + return( mcpu_text_append( text, &ch, 1 ) ); +} + +int +mcpu_text_append_ascii( mcpu_text *text, const char *s ) +{ + if( text == NULL || s == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + while( *s ) + { + unsigned char c = (unsigned char)*s++; + + if( c > 0x7f ) + { + errno = EILSEQ; + return( -1 ); + } + + if( mcpu_text_append_char( text, (__mpu_char16_t)c ) != 0 ) + return( -1 ); + } + + return( 0 ); +} + +int +mcpu_text_append_utf8( mcpu_text *text, const char *s ) +{ + __mpu_char16_t *u; + __mpu_size_t n; + int rc; + + if( text == NULL || s == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + n = mpu_utf8_to_ucs2( NULL, (const __mpu_char8_t *)s, 0 ); + if( n == (__mpu_size_t)-1 ) + return( -1 ); + + u = (__mpu_char16_t *)calloc( (size_t)n + 1, sizeof(__mpu_char16_t) ); + if( u == NULL ) + return( -1 ); + + if( mpu_utf8_to_ucs2( u, (const __mpu_char8_t *)s, + (size_t)n + 1 ) == (__mpu_size_t)-1 ) + { + free( u ); + return( -1 ); + } + + rc = mcpu_text_append( text, u, (size_t)n ); + free( u ); + + return( rc ); +} + +int +mcpu_text_from_utf8( mcpu_text *text, const char *s ) +{ + if( text == NULL || s == NULL ) + { + errno = EINVAL; + return( -1 ); + } + + mcpu_text_free( text ); + mcpu_text_init( text ); + + return( mcpu_text_append_utf8( text, s ) ); +} + +char * +mcpu_text_to_utf8( const __mpu_char16_t *data, size_t length ) +{ + __mpu_char16_t *tmp; + __mpu_char8_t *out; + __mpu_size_t n; + + if( data == NULL && length != 0 ) + { + errno = EINVAL; + return( NULL ); + } + + tmp = (__mpu_char16_t *)calloc( length + 1, sizeof(__mpu_char16_t) ); + if( tmp == NULL ) + return( NULL ); + + if( length != 0 ) + memcpy( tmp, data, length * sizeof(__mpu_char16_t) ); + + n = mpu_ucs2_to_utf8( NULL, tmp, 0 ); + if( n == (__mpu_size_t)-1 ) + { + free( tmp ); + return( NULL ); + } + + out = (__mpu_char8_t *)calloc( (size_t)n + 1, 1 ); + if( out == NULL ) + { + free( tmp ); + return( NULL ); + } + + if( mpu_ucs2_to_utf8( out, tmp, (size_t)n + 1 ) == (__mpu_size_t)-1 ) + { + free( out ); + free( tmp ); + return( NULL ); + } + + free( tmp ); + return( (char *)out ); +} + +int +mcpu_text_equal_ascii( const __mpu_char16_t *data, + size_t length, const char *s ) +{ + size_t i; + + if( data == NULL || s == NULL ) + return( 0 ); + + for( i = 0; i < length && s[i]; ++i ) + { + if( data[i] != (__mpu_char16_t)(unsigned char)s[i] ) + return( 0 ); + } + + return( i == length && s[i] == 0 ); +} diff --git a/src/mcpp-text.h b/src/mcpp-text.h new file mode 100644 index 0000000..dce651c --- /dev/null +++ b/src/mcpp-text.h @@ -0,0 +1,27 @@ +#ifndef __MCPU_CPP_TEXT_H__ +#define __MCPU_CPP_TEXT_H__ 1 + +#include <defs.h> + +typedef struct mcpu_text mcpu_text; +struct mcpu_text +{ + __mpu_char16_t *data; + size_t length; + size_t capacity; +}; + +void mcpu_text_init( mcpu_text *text ); +void mcpu_text_free( mcpu_text *text ); +int mcpu_text_reserve( mcpu_text *text, size_t need ); +int mcpu_text_append( mcpu_text *text, + const __mpu_char16_t *data, size_t length ); +int mcpu_text_append_char( mcpu_text *text, __mpu_char16_t ch ); +int mcpu_text_append_ascii( mcpu_text *text, const char *s ); +int mcpu_text_append_utf8( mcpu_text *text, const char *s ); +int mcpu_text_from_utf8( mcpu_text *text, const char *s ); +char *mcpu_text_to_utf8( const __mpu_char16_t *data, size_t length ); +int mcpu_text_equal_ascii( const __mpu_char16_t *data, + size_t length, const char *s ); + +#endif /* __MCPU_CPP_TEXT_H__ */ diff --git a/tests/Makefile.am b/tests/Makefile.am new file mode 100644 index 0000000..ade9bae --- /dev/null +++ b/tests/Makefile.am @@ -0,0 +1,101 @@ + +TESTS = \ + t0001-utf8.sh \ + t0002-lang.sh \ + t0003-include.sh \ + t0004-config.sh \ + t0005-errors.sh \ + t0006-path-order.sh \ + t0007-language-path.sh \ + t0008-comments.sh \ + t0009-text-scanner.sh \ + t0010-language-names.sh \ + t0011-base-include-path.sh \ + t0012-user-config-path.sh \ + t0013-include-precedence.sh \ + t0014-preprocessing-phases.sh \ + t0015-object-macros.sh \ + t0016-macro-include.sh \ + t0017-macro-errors.sh \ + t0018-include-lexing.sh \ + t0019-recursive-macro.sh \ + t0020-function-macros.sh \ + t0021-function-macro-errors.sh \ + t0022-predefined-macros.sh \ + t0023-predefined-redefine.sh \ + t0024-abi-predefined.sh \ + t0025-dump-macros.sh \ + t0026-dump-config.sh \ + t0027-system-language-path.sh \ + t0028-predefined-ranges.sh \ + t0029-predefined-abi-names.sh \ + t0030-integer-decimal-digits.sh \ + t0031-stringification.sh \ + t0032-command-line.sh \ + t0033-dump-definitions.sh \ + t0034-conditionals.sh \ + t0035-line-control.sh \ + t0036-lang-string.sh \ + t0037-token-concatenation.sh \ + t0038-ucs2-identifiers.sh \ + t0039-command-line-macros.sh \ + t0040-zubr-expression.sh \ + t0041-integer-width-suffix.sh \ + t0042-diagnostics.sh \ + t0043-include-next.sh \ + t0044-configured-search-order.sh \ + t0045-search-dirs-verbose.sh \ + t0046-pragma-once.sh \ + t0047-dependencies.sh \ + t0048-no-config-defaults.sh \ + t0049-relocatable-root.sh \ + t0050-dependency-side-effects.sh \ + t0051-dump-macro-origin.sh \ + t0052-warning-control.sh \ + t0053-interface-cleanup.sh \ + t0054-inhibit-warnings.sh \ + t0055-forced-files.sh \ + t0056-dependency-targets.sh \ + t0057-missing-generated-dependencies.sh \ + t0058-variadic-macros.sh \ + t0059-va-opt.sh \ + t0060-macro-whitespace.sh \ + t0061-output-line-compaction.sh + +noinst_PROGRAMS = mcpp-options-test + +mcpp_options_test_SOURCES = \ + mcpp-options-test.c \ + ../src/mcpp-options.c + +mcpp_options_test_CPPFLAGS = -I$(top_srcdir)/src $(LIBMPUIO_CFLAGS) +mcpp_options_test_LDADD = $(LIBMPUIO_LIBS) + +EXTRA_DIST = $(TESTS) \ + data/utf8.c \ + data/lang.c \ + data/include-main.c \ + data/include/one.h \ + data/include/bridge.h \ + data/system/system.h \ + data/config.conf \ + data/config-main.c \ + data/unbalanced.c \ + data/nonucs2.c \ + data/order-main.c \ + data/user/order.h \ + data/system/order.h \ + data/lang-path-main.c \ + data/lang-as/lang.h \ + data/lang-diff/lang.h \ + data/base-path-main.c \ + data/lang-base/base.h \ + data/lang-base/diff/explicit.h \ + data/common-precedence/same.h \ + data/lang-as-precedence/same.h + +TESTS_ENVIRONMENT = MCPU_CPP='$(abs_top_builddir)/src/mcpu-cpp'; export MCPU_CPP; \ + MCPP_OPTIONS_TEST='$(abs_top_builddir)/tests/mcpp-options-test'; export MCPP_OPTIONS_TEST; + +distclean-local: + -rm -rf $(DEPDIR) diff --git a/tests/data/base-path-main.c b/tests/data/base-path-main.c new file mode 100644 index 0000000..6ca1ffb --- /dev/null +++ b/tests/data/base-path-main.c @@ -0,0 +1 @@ +#include <base.h> diff --git a/tests/data/common-precedence/same.h b/tests/data/common-precedence/same.h new file mode 100644 index 0000000..8a7b2c8 --- /dev/null +++ b/tests/data/common-precedence/same.h @@ -0,0 +1 @@ +int from_common_include_root; diff --git a/tests/data/config-main.c b/tests/data/config-main.c new file mode 100644 index 0000000..892eb5e --- /dev/null +++ b/tests/data/config-main.c @@ -0,0 +1,2 @@ +#include "one.h" +#include <system.h> diff --git a/tests/data/config.conf b/tests/data/config.conf new file mode 100644 index 0000000..e395802 --- /dev/null +++ b/tests/data/config.conf @@ -0,0 +1,2 @@ +MCPU_CPP_INCLUDE_PATH = ./include; +MCPU_CPP_SYSTEM_INCLUDE_PATH = ./system; diff --git a/tests/data/include-main.c b/tests/data/include-main.c new file mode 100644 index 0000000..93bca59 --- /dev/null +++ b/tests/data/include-main.c @@ -0,0 +1,6 @@ +#include "one.h" +#include <system.h> +#lang "diff" +#include "bridge.h" +y' = 1; +#endlang diff --git a/tests/data/include/bridge.h b/tests/data/include/bridge.h new file mode 100644 index 0000000..a878df1 --- /dev/null +++ b/tests/data/include/bridge.h @@ -0,0 +1,4 @@ +/* Entered while current language is diff. */ +#endlang +int bridge_c; +#lang "diff" diff --git a/tests/data/include/one.h b/tests/data/include/one.h new file mode 100644 index 0000000..6200094 --- /dev/null +++ b/tests/data/include/one.h @@ -0,0 +1 @@ +int from_one; diff --git a/tests/data/lang-as-precedence/same.h b/tests/data/lang-as-precedence/same.h new file mode 100644 index 0000000..f139324 --- /dev/null +++ b/tests/data/lang-as-precedence/same.h @@ -0,0 +1 @@ +int from_as_specific_include_path; diff --git a/tests/data/lang-as/lang.h b/tests/data/lang-as/lang.h new file mode 100644 index 0000000..f1f9283 --- /dev/null +++ b/tests/data/lang-as/lang.h @@ -0,0 +1 @@ +MCPU_AS_LANGUAGE_PATH from_as_language_path; diff --git a/tests/data/lang-base/base.h b/tests/data/lang-base/base.h new file mode 100644 index 0000000..1d8a123 --- /dev/null +++ b/tests/data/lang-base/base.h @@ -0,0 +1 @@ +int from_base_language_include_root; diff --git a/tests/data/lang-base/diff/explicit.h b/tests/data/lang-base/diff/explicit.h new file mode 100644 index 0000000..fa9157f --- /dev/null +++ b/tests/data/lang-base/diff/explicit.h @@ -0,0 +1 @@ +int from_explicit_diff_subdirectory; diff --git a/tests/data/lang-diff/lang.h b/tests/data/lang-diff/lang.h new file mode 100644 index 0000000..54603aa --- /dev/null +++ b/tests/data/lang-diff/lang.h @@ -0,0 +1 @@ +y' = from_diff_language_path; diff --git a/tests/data/lang-path-main.c b/tests/data/lang-path-main.c new file mode 100644 index 0000000..864bc07 --- /dev/null +++ b/tests/data/lang-path-main.c @@ -0,0 +1,6 @@ +#lang "as" +#include "lang.h" +#endlang +#lang "diff" +#include "lang.h" +#endlang diff --git a/tests/data/lang.c b/tests/data/lang.c new file mode 100644 index 0000000..75cdc70 --- /dev/null +++ b/tests/data/lang.c @@ -0,0 +1,9 @@ +int before; +#lang "diff" +y' = omega; +#lang "alg" +f = x + y; +#endlang +z' = y; +#endlang +int after; diff --git a/tests/data/nonucs2.c b/tests/data/nonucs2.c new file mode 100644 index 0000000..d8cc27c --- /dev/null +++ b/tests/data/nonucs2.c @@ -0,0 +1 @@ +int x; /* 😀 */ diff --git a/tests/data/order-main.c b/tests/data/order-main.c new file mode 100644 index 0000000..04cac0c --- /dev/null +++ b/tests/data/order-main.c @@ -0,0 +1 @@ +#include <order.h> diff --git a/tests/data/system/order.h b/tests/data/system/order.h new file mode 100644 index 0000000..8e7f307 --- /dev/null +++ b/tests/data/system/order.h @@ -0,0 +1 @@ +int system_order; diff --git a/tests/data/system/system.h b/tests/data/system/system.h new file mode 100644 index 0000000..c94d2ab --- /dev/null +++ b/tests/data/system/system.h @@ -0,0 +1 @@ +int from_system; diff --git a/tests/data/unbalanced.c b/tests/data/unbalanced.c new file mode 100644 index 0000000..1bf56c1 --- /dev/null +++ b/tests/data/unbalanced.c @@ -0,0 +1,2 @@ +#lang "diff" +y' = 1; diff --git a/tests/data/user/order.h b/tests/data/user/order.h new file mode 100644 index 0000000..f6313ca --- /dev/null +++ b/tests/data/user/order.h @@ -0,0 +1 @@ +int user_order; diff --git a/tests/data/utf8.c b/tests/data/utf8.c new file mode 100644 index 0000000..d5153ea --- /dev/null +++ b/tests/data/utf8.c @@ -0,0 +1,2 @@ +/* UTF-8 source, UCS-2 internal representation. */ +const char *message = "Привет, MCPU"; diff --git a/tests/mcpp-options-test.c b/tests/mcpp-options-test.c new file mode 100644 index 0000000..a0e8311 --- /dev/null +++ b/tests/mcpp-options-test.c @@ -0,0 +1,24 @@ +#include <defs.h> + +int +main( int argc, char **argv ) +{ + mcpp_options options; + const char *p; + int rc = 1; + + mcpp_options_init( &options, argc ); + if( options.actions == NULL ) + return( 1 ); + + p = argv[0] + strlen( argv[0] ); + while( p != argv[0] && p[-1] != '/' ) --p; + options.progname = p; + + if( mcpp_handle_options(&options, argc - 1, argv + 1) == 0 && + mcpp_options_dump(&options, stdout) == 0 ) + rc = 0; + + mcpp_options_free( &options ); + return( rc ); +} diff --git a/tests/t0001-utf8.sh b/tests/t0001-utf8.sh new file mode 100755 index 0000000..6726cfe --- /dev/null +++ b/tests/t0001-utf8.sh @@ -0,0 +1,6 @@ +#!/bin/sh +set -eu +out="${TMPDIR-/tmp}/mcpu-cpp-utf8-$$.out" +trap 'rm -f "$out"' EXIT HUP INT TERM +"$MCPU_CPP" --no-config "$srcdir/data/utf8.c" -o "$out" +grep 'Привет, MCPU' "$out" >/dev/null diff --git a/tests/t0002-lang.sh b/tests/t0002-lang.sh new file mode 100755 index 0000000..1d0102e --- /dev/null +++ b/tests/t0002-lang.sh @@ -0,0 +1,8 @@ +#!/bin/sh +set -eu +out="${TMPDIR-/tmp}/mcpu-cpp-lang-$$.out" +trap 'rm -f "$out"' EXIT HUP INT TERM +"$MCPU_CPP" --no-config "$srcdir/data/lang.c" -o "$out" +grep '#lang "diff"' "$out" >/dev/null +grep '#lang "alg"' "$out" >/dev/null +grep '#endlang' "$out" >/dev/null diff --git a/tests/t0003-include.sh b/tests/t0003-include.sh new file mode 100755 index 0000000..b8ddcec --- /dev/null +++ b/tests/t0003-include.sh @@ -0,0 +1,11 @@ +#!/bin/sh +set -eu +out="${TMPDIR-/tmp}/mcpu-cpp-include-$$.out" +trap 'rm -f "$out"' EXIT HUP INT TERM +"$MCPU_CPP" --no-config -I "$srcdir/data/include" \ + -isystem "$srcdir/data/system" \ + "$srcdir/data/include-main.c" -o "$out" +grep 'from_one' "$out" >/dev/null +grep 'from_system' "$out" >/dev/null +grep 'bridge_c' "$out" >/dev/null +grep "y' = 1" "$out" >/dev/null diff --git a/tests/t0004-config.sh b/tests/t0004-config.sh new file mode 100755 index 0000000..d025d99 --- /dev/null +++ b/tests/t0004-config.sh @@ -0,0 +1,12 @@ +#!/bin/sh +set -eu +out="${TMPDIR-/tmp}/mcpu-cpp-config-$$.out" +conf="${TMPDIR-/tmp}/mcpu-cpp-config-$$.conf" +trap 'rm -f "$out" "$conf"' EXIT HUP INT TERM +cat > "$conf" <<EOT +MCPU_CPP_INCLUDE_PATH = $srcdir/data/include; +MCPU_CPP_SYSTEM_INCLUDE_PATH = $srcdir/data/system; +EOT +"$MCPU_CPP" --config-file "$conf" "$srcdir/data/config-main.c" -o "$out" +grep 'from_one' "$out" >/dev/null +grep 'from_system' "$out" >/dev/null diff --git a/tests/t0005-errors.sh b/tests/t0005-errors.sh new file mode 100755 index 0000000..458b4c8 --- /dev/null +++ b/tests/t0005-errors.sh @@ -0,0 +1,22 @@ +#!/bin/sh +set -eu +out="${TMPDIR-/tmp}/mcpu-cpp-error-$$.out" +err="${TMPDIR-/tmp}/mcpu-cpp-error-$$.err" +bad="${TMPDIR-/tmp}/mcpu-cpp-nonucs2-$$.c" +trap 'rm -f "$out" "$err" "$bad"' EXIT HUP INT TERM + +if "$MCPU_CPP" --no-config "$srcdir/data/unbalanced.c" -o "$out" 2>"$err"; then + exit 1 +fi +grep 'unbalanced #lang/#endlang' "$err" >/dev/null + +# A valid UTF-8 scalar outside UCS-2 is harmless inside a comment. +"$MCPU_CPP" --no-config "$srcdir/data/nonucs2.c" -o "$out" +grep '^int x;$' "$out" >/dev/null + +# The same scalar in program text is rejected after comment removal. +printf 'int x = \360\237\230\200;\n' > "$bad" +if "$MCPU_CPP" --no-config "$bad" -o "$out" 2>"$err"; then + exit 1 +fi +grep 'character outside UCS-2' "$err" >/dev/null diff --git a/tests/t0006-path-order.sh b/tests/t0006-path-order.sh new file mode 100755 index 0000000..0b9f76e --- /dev/null +++ b/tests/t0006-path-order.sh @@ -0,0 +1,13 @@ +#!/bin/sh +set -eu +out="${TMPDIR-/tmp}/mcpu-cpp-order-$$.out" +trap 'rm -f "$out"' EXIT HUP INT TERM +# Deliberately put -isystem first: -I must still have higher semantic priority. +"$MCPU_CPP" --no-config \ + -isystem "$srcdir/data/system" \ + -I "$srcdir/data/user" \ + "$srcdir/data/order-main.c" -o "$out" +grep 'user_order' "$out" >/dev/null +if grep 'system_order' "$out" >/dev/null; then + exit 1 +fi diff --git a/tests/t0007-language-path.sh b/tests/t0007-language-path.sh new file mode 100755 index 0000000..6a75797 --- /dev/null +++ b/tests/t0007-language-path.sh @@ -0,0 +1,12 @@ +#!/bin/sh +set -eu +out="${TMPDIR-/tmp}/mcpu-cpp-langpath-$$.out" +conf="${TMPDIR-/tmp}/mcpu-cpp-langpath-$$.conf" +trap 'rm -f "$out" "$conf"' EXIT HUP INT TERM +cat > "$conf" <<EOT +MCPU_CPP_AS_INCLUDE_PATH = $srcdir/data/lang-as; +MCPU_CPP_DIFF_INCLUDE_PATH = $srcdir/data/lang-diff; +EOT +"$MCPU_CPP" --config-file "$conf" "$srcdir/data/lang-path-main.c" -o "$out" +grep 'from_as_language_path' "$out" >/dev/null +grep 'from_diff_language_path' "$out" >/dev/null diff --git a/tests/t0008-comments.sh b/tests/t0008-comments.sh new file mode 100755 index 0000000..26386ba --- /dev/null +++ b/tests/t0008-comments.sh @@ -0,0 +1,41 @@ +#!/bin/sh +set -eu +src="${TMPDIR-/tmp}/mcpu-cpp-comments-$$.c" +out="${TMPDIR-/tmp}/mcpu-cpp-comments-$$.out" +trap 'rm -f "$src" "$out"' EXIT HUP INT TERM +cat > "$src" <<'EOT' +/* +#lang "diff" +*/ +int c_side; +/* comment before directive */ #lang "diff" +y' = 1; +#endlang + // one-line comment + /* + multi-line comment + */ +LEFT/**/RIGHT +int block_tail; /* block tail */ +int line_tail; // line tail +int multi_tail; /* + block tail + */ +EOT +"$MCPU_CPP" --no-config "$src" -o "$out" +grep 'int c_side' "$out" >/dev/null +grep "y' = 1" "$out" >/dev/null +grep '^LEFT RIGHT$' "$out" >/dev/null + +# Comment-only lines must be really empty; no synthetic or leading spaces remain. +if grep '^[[:blank:]][[:blank:]]*$' "$out" >/dev/null; then + exit 1 +fi + +# A comment ending a non-empty line must not leave trailing whitespace. +grep '^int block_tail;$' "$out" >/dev/null +grep '^int line_tail;$' "$out" >/dev/null +grep '^int multi_tail;$' "$out" >/dev/null +if grep '[[:blank:]]$' "$out" >/dev/null; then + exit 1 +fi diff --git a/tests/t0009-text-scanner.sh b/tests/t0009-text-scanner.sh new file mode 100755 index 0000000..773968f --- /dev/null +++ b/tests/t0009-text-scanner.sh @@ -0,0 +1,37 @@ +#!/bin/sh +set -eu +src="${TMPDIR-/tmp}/mcpu-cpp-scan-$$.c" +out="${TMPDIR-/tmp}/mcpu-cpp-scan-$$.out" +err="${TMPDIR-/tmp}/mcpu-cpp-scan-$$.err" +bom="${TMPDIR-/tmp}/mcpu-cpp-bom-$$.c" +nul="${TMPDIR-/tmp}/mcpu-cpp-nul-$$.c" +trap 'rm -f "$src" "$out" "$err" "$bom" "$nul"' EXIT HUP INT TERM + +cat > "$src" <<'EOT' +const char *a = "/* not a comment"; +#lang "diff" +y' = 1; +const char *b = "*/ still not a comment"; +#endlang +const char *c = "// not a comment"; +/* real comment starts here +#lang "alg" +*/ +int c_side; +EOT +"$MCPU_CPP" --no-config "$src" -o "$out" +grep '#lang "diff"' "$out" >/dev/null +grep "y' = 1" "$out" >/dev/null +grep 'int c_side' "$out" >/dev/null + +# UTF-8 BOM is accepted and removed at the external text boundary. +printf '\357\273\277int bom_ok;\n' > "$bom" +"$MCPU_CPP" --no-config "$bom" -o "$out" +grep 'int bom_ok' "$out" >/dev/null + +# Embedded NUL is not text and must never be silently truncated. +printf 'int before;\000int after;\n' > "$nul" +if "$MCPU_CPP" --no-config "$nul" -o "$out" 2>"$err"; then + exit 1 +fi +grep 'NUL character is not allowed' "$err" >/dev/null diff --git a/tests/t0010-language-names.sh b/tests/t0010-language-names.sh new file mode 100755 index 0000000..83d4325 --- /dev/null +++ b/tests/t0010-language-names.sh @@ -0,0 +1,38 @@ +#!/bin/sh +set -eu +dir="${TMPDIR-/tmp}/mcpu-cpp-languages-$$" +mkdir -p "$dir" +trap 'rm -rf "$dir"' EXIT HUP INT TERM + +cat > "$dir/ok.c" <<'EOT' +#lang "diff" +#endlang +#lang "DIFT" +#endlang +#lang "Alg" +#endlang +#lang "aS" +#endlang +#lang "AvM" +#endlang +#lang "acs" +#endlang +EOT +"$MCPU_CPP" --no-config "$dir/ok.c" -o "$dir/ok.out" +grep '^#lang "diff"$' "$dir/ok.out" >/dev/null +grep '^#lang "DIFT"$' "$dir/ok.out" >/dev/null +grep '^#lang "Alg"$' "$dir/ok.out" >/dev/null +grep '^#lang "aS"$' "$dir/ok.out" >/dev/null +grep '^#lang "AvM"$' "$dir/ok.out" >/dev/null +grep '^#lang "acs"$' "$dir/ok.out" >/dev/null + +for lang in 0 c vasm unknown +do + printf '#lang "%s"\n#endlang\n' "$lang" > "$dir/bad.c" + if "$MCPU_CPP" --no-config "$dir/bad.c" -o "$dir/bad.out" 2> "$dir/bad.err" + then + echo "#lang \"$lang\" unexpectedly accepted" >&2 + exit 1 + fi + grep "unknown language '$lang'" "$dir/bad.err" >/dev/null +done diff --git a/tests/t0011-base-include-path.sh b/tests/t0011-base-include-path.sh new file mode 100755 index 0000000..f862e88 --- /dev/null +++ b/tests/t0011-base-include-path.sh @@ -0,0 +1,19 @@ +#!/bin/sh +set -eu +dir="${TMPDIR-/tmp}/mcpu-cpp-basepath-$$" +mkdir -p "$dir" +trap 'rm -rf "$dir"' EXIT HUP INT TERM +cat > "$dir/config" <<EOT +MCPU_CPP_INCLUDE_PATH = $srcdir/data/lang-base; +EOT +"$MCPU_CPP" --config-file "$dir/config" "$srcdir/data/base-path-main.c" -o "$dir/base.out" +grep 'from_base_language_include_root' "$dir/base.out" >/dev/null +cat > "$dir/as.c" <<'EOT' +#lang "as" +#include <base.h> +#include <diff/explicit.h> +#endlang +EOT +"$MCPU_CPP" --config-file "$dir/config" "$dir/as.c" -o "$dir/as.out" +grep 'from_base_language_include_root' "$dir/as.out" >/dev/null +grep 'from_explicit_diff_subdirectory' "$dir/as.out" >/dev/null diff --git a/tests/t0012-user-config-path.sh b/tests/t0012-user-config-path.sh new file mode 100755 index 0000000..54b6e64 --- /dev/null +++ b/tests/t0012-user-config-path.sh @@ -0,0 +1,26 @@ +#!/bin/sh +set -eu +dir="${TMPDIR-/tmp}/mcpu-cpp-userconf-$$" +mkdir -p "$dir/home/.mcpu" "$dir/include" "$dir/system/as" +trap 'rm -rf "$dir"' EXIT HUP INT TERM +cat > "$dir/home/.mcpu/mcpu-cpp.conf" <<EOT +MCPU_CPP_INCLUDE_PATH = $dir/include; +MCPU_CPP_SYSTEM_INCLUDE_PATH = $dir/system; +EOT +cat > "$dir/include/userconf.h" <<'EOT' +int from_user_mcpu_configuration; +EOT +cat > "$dir/system/as/system-userconf.h" <<'EOT' +int from_user_system_root; +EOT +cat > "$dir/main.c" <<'EOT' +#include <userconf.h> +#lang "as" +#include <system-userconf.h> +#endlang +EOT +HOME="$dir/home" "$MCPU_CPP" "$dir/main.c" -o "$dir/out" +grep 'from_user_mcpu_configuration' "$dir/out" >/dev/null +grep 'from_user_system_root' "$dir/out" >/dev/null +HOME="$dir/home" "$MCPU_CPP" -dconfig > "$dir/config.out" +grep "^MCPU_CPP_SYSTEM_INCLUDE_PATH = $dir/system;$" "$dir/config.out" >/dev/null diff --git a/tests/t0013-include-precedence.sh b/tests/t0013-include-precedence.sh new file mode 100755 index 0000000..0bfc1e5 --- /dev/null +++ b/tests/t0013-include-precedence.sh @@ -0,0 +1,18 @@ +#!/bin/sh +set -eu +dir="${TMPDIR-/tmp}/mcpu-cpp-precedence-$$" +mkdir -p "$dir" +trap 'rm -rf "$dir"' EXIT HUP INT TERM +cat > "$dir/config" <<EOT +MCPU_CPP_INCLUDE_PATH = $srcdir/data/common-precedence; +MCPU_CPP_AS_INCLUDE_PATH = $srcdir/data/lang-as-precedence; +EOT +cat > "$dir/main.c" <<'EOT' +#include <same.h> +#lang "as" +#include <same.h> +#endlang +EOT +"$MCPU_CPP" --config-file "$dir/config" "$dir/main.c" -o "$dir/out" +grep 'from_common_include_root' "$dir/out" >/dev/null +grep 'from_as_specific_include_path' "$dir/out" >/dev/null diff --git a/tests/t0014-preprocessing-phases.sh b/tests/t0014-preprocessing-phases.sh new file mode 100755 index 0000000..5cc98c8 --- /dev/null +++ b/tests/t0014-preprocessing-phases.sh @@ -0,0 +1,21 @@ +#!/bin/sh +set -eu +src="${TMPDIR-/tmp}/mcpu-cpp-phases-$$.c" +out="${TMPDIR-/tmp}/mcpu-cpp-phases-$$.out" +trap 'rm -f "$src" "$out"' EXIT HUP INT TERM +cat > "$src" <<'EOT' +#defi\ +ne VALUE 10\ +20 +/* comment */ VALUE // trailing comment +const char *s = "VALUE /* not a comment */"; +EOT +"$MCPU_CPP" --no-config "$src" -o "$out" +grep '1020' "$out" >/dev/null +grep 'VALUE /\* not a comment \*/' "$out" >/dev/null +if grep 'trailing comment' "$out" >/dev/null; then + exit 1 +fi +if grep '/\* comment \*/' "$out" >/dev/null; then + exit 1 +fi diff --git a/tests/t0015-object-macros.sh b/tests/t0015-object-macros.sh new file mode 100755 index 0000000..a59bd23 --- /dev/null +++ b/tests/t0015-object-macros.sh @@ -0,0 +1,27 @@ +#!/bin/sh +set -eu +src="${TMPDIR-/tmp}/mcpu-cpp-macro-$$.c" +out="${TMPDIR-/tmp}/mcpu-cpp-macro-$$.out" +trap 'rm -f "$src" "$out"' EXIT HUP INT TERM +cat > "$src" <<'EOT' +#define A B +#define B 42 +#define EMPTY +A +x EMPTY y +"A B" +#define SELF SELF +SELF +#undef B +A +#lang "diff" +y' = A; +#endlang +EOT +"$MCPU_CPP" --no-config "$src" -o "$out" +grep '^42$' "$out" >/dev/null +grep '^x y$' "$out" >/dev/null +grep '^"A B"$' "$out" >/dev/null +grep '^SELF$' "$out" >/dev/null +grep '^B$' "$out" >/dev/null +grep "y' = B;" "$out" >/dev/null diff --git a/tests/t0016-macro-include.sh b/tests/t0016-macro-include.sh new file mode 100755 index 0000000..0c1b56e --- /dev/null +++ b/tests/t0016-macro-include.sh @@ -0,0 +1,16 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-minclude-$$" +mkdir -p "$base/inc" +trap 'rm -rf "$base"' EXIT HUP INT TERM +cat > "$base/main.c" <<'EOT' +#define HEADER <macro.h> +#include HEADER +VALUE +EOT +cat > "$base/inc/macro.h" <<'EOT' +#define VALUE 77 +EOT +"$MCPU_CPP" --no-config -I "$base/inc" "$base/main.c" -o "$base/out" +grep '^77$' "$base/out" >/dev/null +grep "^# 3 \"$base/main.c\" 2$" "$base/out" >/dev/null diff --git a/tests/t0017-macro-errors.sh b/tests/t0017-macro-errors.sh new file mode 100755 index 0000000..f54f74c --- /dev/null +++ b/tests/t0017-macro-errors.sh @@ -0,0 +1,13 @@ +#!/bin/sh +set -eu +src="${TMPDIR-/tmp}/mcpu-cpp-macro-errors-$$.c" +out="${TMPDIR-/tmp}/mcpu-cpp-macro-errors-$$.out" +err="${TMPDIR-/tmp}/mcpu-cpp-macro-errors-$$.err" +trap 'rm -f "$src" "$out" "$err"' EXIT HUP INT TERM +cat > "$src" <<'EOT' +#define +EOT +if "$MCPU_CPP" --no-config "$src" -o "$out" 2>"$err"; then + exit 1 +fi +grep 'macro name expected after #define' "$err" >/dev/null diff --git a/tests/t0018-include-lexing.sh b/tests/t0018-include-lexing.sh new file mode 100755 index 0000000..ed1561d --- /dev/null +++ b/tests/t0018-include-lexing.sh @@ -0,0 +1,17 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-include-lex-$$" +mkdir -p "$base/inc/x" +trap 'rm -rf "$base"' EXIT HUP INT TERM +cat > "$base/main.c" <<'EOT' +#include \ +<x/*y> +AFTER +EOT +cat > "$base/inc/x/*y" <<'EOT' +INSIDE +EOT +"$MCPU_CPP" --no-config -I "$base/inc" "$base/main.c" -o "$base/out" +grep '^INSIDE$' "$base/out" >/dev/null +grep '^AFTER$' "$base/out" >/dev/null +grep "^# 3 \"$base/main.c\" 2$" "$base/out" >/dev/null diff --git a/tests/t0019-recursive-macro.sh b/tests/t0019-recursive-macro.sh new file mode 100755 index 0000000..ffba144 --- /dev/null +++ b/tests/t0019-recursive-macro.sh @@ -0,0 +1,13 @@ +#!/bin/sh +set -eu +src="${TMPDIR-/tmp}/mcpu-cpp-recursive-$$.c" +out="${TMPDIR-/tmp}/mcpu-cpp-recursive-$$.out" +trap 'rm -f "$src" "$out"' EXIT HUP INT TERM +cat > "$src" <<'EOT' +int foo; +#define foo (4 + foo) +f = foo; +EOT +"$MCPU_CPP" --no-config "$src" -o "$out" +grep '^int foo;$' "$out" >/dev/null +grep '^f = (4 + foo);$' "$out" >/dev/null diff --git a/tests/t0020-function-macros.sh b/tests/t0020-function-macros.sh new file mode 100755 index 0000000..aba3fc5 --- /dev/null +++ b/tests/t0020-function-macros.sh @@ -0,0 +1,42 @@ +#!/bin/sh +set -eu +src="${TMPDIR-/tmp}/mcpu-cpp-fmacro-$$.c" +out="${TMPDIR-/tmp}/mcpu-cpp-fmacro-$$.out" +trap 'rm -f "$src" "$out"' EXIT HUP INT TERM +cat > "$src" <<'EOT' +#define min(X, Y) ((X) < (Y) ? (X) : (Y)) +#define A 7 +#define ZERO() zero-text +#define INNER(X) [X] +#define OUTER(X, Y) INNER(X) + INNER(Y) +#define OBJECT (argument) +#define WRAP(X) <X> +#define TWICE(X) (X) + \ +(X) +min(1, 2) +min(A, 3) +min(min(a, b), c) +OUTER(A, min(4, 5)) +ZERO () +ZERO +OBJECT +WRAP("a,b") +TWICE(3) +"min(A, 3)" +#lang "diff" +#define DERIV(X) X' + X +DERIV(y) +#endlang +EOT +"$MCPU_CPP" --no-config "$src" -o "$out" +grep '^((1) < (2) ? (1) : (2))$' "$out" >/dev/null +grep '^((7) < (3) ? (7) : (3))$' "$out" >/dev/null +grep '^((((a) < (b) ? (a) : (b))) < (c) ? (((a) < (b) ? (a) : (b))) : (c))$' "$out" >/dev/null +grep '^\[7\] + \[((4) < (5) ? (4) : (5))\]$' "$out" >/dev/null +grep '^zero-text$' "$out" >/dev/null +grep '^ZERO$' "$out" >/dev/null +grep '^(argument)$' "$out" >/dev/null +grep '^<"a,b">$' "$out" >/dev/null +grep '^(3) + (3)$' "$out" >/dev/null +grep '^"min(A, 3)"$' "$out" >/dev/null +grep "^y' + y$" "$out" >/dev/null diff --git a/tests/t0021-function-macro-errors.sh b/tests/t0021-function-macro-errors.sh new file mode 100755 index 0000000..2353b35 --- /dev/null +++ b/tests/t0021-function-macro-errors.sh @@ -0,0 +1,89 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-fmacro-errors-$$" +trap 'rm -f "$base".*' EXIT HUP INT TERM + +cat > "$base.dup.c" <<'EOT' +#define F(X, X) X +EOT +if "$MCPU_CPP" --no-config "$base.dup.c" -o "$base.out" 2>"$base.err"; then + echo "duplicate parameter name accepted" >&2 + exit 1 +fi +grep "duplicate argument name 'X'" "$base.err" >/dev/null + +cat > "$base.few.c" <<'EOT' +#define F(X, Y) X + Y +F(1) +EOT +if "$MCPU_CPP" --no-config "$base.few.c" -o "$base.out" 2>"$base.err"; then + echo "too few macro arguments accepted" >&2 + exit 1 +fi +grep "used with too few arguments" "$base.err" >/dev/null + +cat > "$base.many.c" <<'EOT' +#define F(X) X +F(1, 2) +EOT +if "$MCPU_CPP" --no-config "$base.many.c" -o "$base.out" 2>"$base.err"; then + echo "too many macro arguments accepted" >&2 + exit 1 +fi +grep "used with too many arguments" "$base.err" >/dev/null + +cat > "$base.unterm.c" <<'EOT' +#define F(X) X +F(1 +EOT +if "$MCPU_CPP" --no-config "$base.unterm.c" -o "$base.out" 2>"$base.err"; then + echo "unterminated macro call accepted" >&2 + exit 1 +fi +grep "unterminated argument list" "$base.err" >/dev/null + +cat > "$base.concat-first.c" <<'EOT' +#define CAT(X, Y) ## X +EOT +if "$MCPU_CPP" --no-config "$base.concat-first.c" -o "$base.out" 2>"$base.err"; then + echo "leading ## in replacement list accepted" >&2 + exit 1 +fi +grep "'##' cannot appear at the beginning of a macro replacement list" "$base.err" >/dev/null + +cat > "$base.concat-last.c" <<'EOT' +#define CAT(X, Y) Y ## +EOT +if "$MCPU_CPP" --no-config "$base.concat-last.c" -o "$base.out" 2>"$base.err"; then + echo "trailing ## in replacement list accepted" >&2 + exit 1 +fi +grep "'##' cannot appear at the end of a macro replacement list" "$base.err" >/dev/null + +cat > "$base.sharp-name.c" <<'EOT' +#define S(X) #Y +EOT +if "$MCPU_CPP" --no-config "$base.sharp-name.c" -o "$base.out" 2>"$base.err"; then + echo "stringification of a non-parameter accepted" >&2 + exit 1 +fi +grep "'#' operator should be followed by a macro argument name" "$base.err" >/dev/null + +cat > "$base.sharp-end.c" <<'EOT' +#define S(X) # +EOT +if "$MCPU_CPP" --no-config "$base.sharp-end.c" -o "$base.out" 2>"$base.err"; then + echo "unterminated stringification operator accepted" >&2 + exit 1 +fi +grep "'#' operator is not followed by a macro argument name" "$base.err" >/dev/null + +cat > "$base.bracket.c" <<'EOT' +#define F(X) X +F(array[x = y, x + 1]) +EOT +if "$MCPU_CPP" --no-config "$base.bracket.c" -o "$base.out" 2>"$base.err"; then + echo "comma in brackets incorrectly protected macro argument" >&2 + exit 1 +fi +grep "used with too many arguments" "$base.err" >/dev/null diff --git a/tests/t0022-predefined-macros.sh b/tests/t0022-predefined-macros.sh new file mode 100755 index 0000000..960e1e8 --- /dev/null +++ b/tests/t0022-predefined-macros.sh @@ -0,0 +1,56 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-predef-$$" +mkdir -p "$base" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +cat > "$base/main.c" <<'EOT' +root_file = __FILE__; +root_base = __BASE_FILE__; +root_level = __INCLUDE_LEVEL__; +#define HERE() __LINE__ +root_line = HERE(); +cpp_version = __MCPU_CPP_VERSION__; +old_version_name = __VERSION__; +date = __DATE__; +time = __TIME__; +#include "one.h" +after_line = __LINE__; +EOT + +cat > "$base/one.h" <<'EOT' +one_file = __FILE__; +one_base = __BASE_FILE__; +one_level = __INCLUDE_LEVEL__; +one_line = __LINE__; +#include "two.h" +EOT + +cat > "$base/two.h" <<'EOT' +two_file = __FILE__; +two_base = __BASE_FILE__; +two_level = __INCLUDE_LEVEL__; +two_line = __LINE__; +EOT + +"$MCPU_CPP" --no-config "$base/main.c" -o "$base/out" + +grep -F "root_file = \"$base/main.c\";" "$base/out" >/dev/null +grep -F "root_base = \"$base/main.c\";" "$base/out" >/dev/null +grep '^root_level = 0;$' "$base/out" >/dev/null +grep '^root_line = 5;$' "$base/out" >/dev/null +grep '^cpp_version = "1\.0\.2";$' "$base/out" >/dev/null +grep '^old_version_name = __VERSION__;$' "$base/out" >/dev/null +grep '^date = "[A-Z][a-z][a-z] [0-9] [0-9][0-9][0-9][0-9]";$\|^date = "[A-Z][a-z][a-z] [12][0-9] [0-9][0-9][0-9][0-9]";$\|^date = "[A-Z][a-z][a-z] 3[01] [0-9][0-9][0-9][0-9]";$' "$base/out" >/dev/null +grep '^time = "[0-2][0-9]:[0-5][0-9]:[0-5][0-9]";$' "$base/out" >/dev/null + +grep -F "one_file = \"$base/one.h\";" "$base/out" >/dev/null +grep -F "one_base = \"$base/main.c\";" "$base/out" >/dev/null +grep '^one_level = 1;$' "$base/out" >/dev/null +grep '^one_line = 4;$' "$base/out" >/dev/null + +grep -F "two_file = \"$base/two.h\";" "$base/out" >/dev/null +grep -F "two_base = \"$base/main.c\";" "$base/out" >/dev/null +grep '^two_level = 2;$' "$base/out" >/dev/null +grep '^two_line = 4;$' "$base/out" >/dev/null +grep '^after_line = 11;$' "$base/out" >/dev/null diff --git a/tests/t0023-predefined-redefine.sh b/tests/t0023-predefined-redefine.sh new file mode 100755 index 0000000..b30a1d9 --- /dev/null +++ b/tests/t0023-predefined-redefine.sh @@ -0,0 +1,20 @@ +#!/bin/sh +set -eu +src="${TMPDIR-/tmp}/mcpu-cpp-predef-redef-$$.c" +out="${TMPDIR-/tmp}/mcpu-cpp-predef-redef-$$.out" +err="${TMPDIR-/tmp}/mcpu-cpp-predef-redef-$$.err" +trap 'rm -f "$src" "$out" "$err"' EXIT HUP INT TERM +cat > "$src" <<'EOT' +before = __LINE__; +#define __LINE__ 77 +after = __LINE__; +#undef __LINE__ +literal = __LINE__; +"__FILE__ __LINE__ __MCPU_CPP_VERSION__" +EOT +"$MCPU_CPP" --no-config "$src" -o "$out" 2>"$err" +grep '^before = 1;$' "$out" >/dev/null +grep '^literal = __LINE__;$' "$out" >/dev/null +grep '^after = 77;$' "$out" >/dev/null +grep '^"__FILE__ __LINE__ __MCPU_CPP_VERSION__"$' "$out" >/dev/null +grep "warning: macro '__LINE__' redefined" "$err" >/dev/null diff --git a/tests/t0024-abi-predefined.sh b/tests/t0024-abi-predefined.sh new file mode 100755 index 0000000..984f0a2 --- /dev/null +++ b/tests/t0024-abi-predefined.sh @@ -0,0 +1,191 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-abi-predef-$$" +mkdir -p "$base" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +cat > "$base/main.c" <<'EOT' +arch = _ARCH_MCPU; +cpp_version = __MCPU_CPP_VERSION__; +old_version = __VERSION__; +byte_order = __BYTE_ORDER__; +mcpu_byte_order = __MCPU_BYTE_ORDER__; +word_order = __MCPU_WORD_ORDER__; +old_float_word_order = __FLOAT_WORD_ORDER__; +register_width = __MCPU_MACHINE_REGISTER_WIDTH__; +real_io_limit = __MCPU_REAL_IO_LIMIT__; +math_fn_limit = __MCPU_MATH_FN_LIMIT__; +pointer_width = __MCPU_POINTER_WIDTH__; +int_max_width = __MCPU_INT_MAX_WIDTH__; +real_max_width = __MCPU_REAL_MAX_WIDTH__; +complex_max_width = __MCPU_COMPLEX_MAX_WIDTH__; +size_type = __MCPU_SIZE_TYPE__; +size_width = __MCPU_SIZE_WIDTH__; +size_sizeof = __MCPU_SIZEOF_SIZE__; +size_max = __MCPU_SIZE_MAX__; +ssize_type = __MCPU_SSIZE_TYPE__; +ssize_width = __MCPU_SSIZE_WIDTH__; +ssize_sizeof = __MCPU_SIZEOF_SSIZE__; +ssize_max = __MCPU_SSIZE_MAX__; +ptrdiff_type = __PTRDIFF_TYPE__; +ptrdiff_width = __PTRDIFF_WIDTH__; +ptrdiff_max = __PTRDIFF_MAX__; +sizeof_ptrdiff = __SIZEOF_PTRDIFF__; +old_sizeof_ptrdiff_t = __SIZEOF_PTRDIFF_T__; +intptr_type = __INTPTR_TYPE__; +uintptr_type = __UINTPTR_TYPE__; +intptr_width = __INTPTR_WIDTH__; +uintptr_width = __UINTPTR_WIDTH__; +intptr_max = __INTPTR_MAX__; +uintptr_max = __UINTPTR_MAX__; +char8_type = __CHAR8_TYPE__; +char16_type = __CHAR16_TYPE__; +char8_width = __CHAR8_WIDTH__; +char16_width = __CHAR16_WIDTH__; +sizeof_char8 = __SIZEOF_CHAR8__; +sizeof_char16 = __SIZEOF_CHAR16__; +register_prefix = [__REGISTER_PREFIX__]; +local_label_prefix = [__LOCAL_LABEL_PREFIX__]; +user_label_prefix = [__USER_LABEL_PREFIX__]; +immediate_prefix = [__IMMEDIATE_PREFIX__]; +int128_type = __INT128_TYPE__; +uint128_type = __UINT128_TYPE__; +int128_width = __INT128_WIDTH__; +int128_decimal = __INT128_DECIMAL_DIG__; +uint128_decimal = __UINT128_DECIMAL_DIG__; +sizeof_int128 = __SIZEOF_INT128__; +sizeof_uint128 = __SIZEOF_UINT128__; +int128_max = __INT128_MAX__; +int128_min_must_not_exist = __INT128_MIN__; +uint128_max = __UINT128_MAX__; +int1024_type = __INT1024_TYPE__; +int1024_width = __INT1024_WIDTH__; +int1024_max_must_not_exist = __INT1024_MAX__; +int1024_decimal = __INT1024_DECIMAL_DIG__; +sizeof_int1024 = __SIZEOF_INT1024__; +real128_type = __REAL128_TYPE__; +complex128_type = __COMPLEX128_TYPE__; +real128_width = __REAL128_WIDTH__; +complex128_width = __COMPLEX128_WIDTH__; +sizeof_real128 = __SIZEOF_REAL128__; +sizeof_real128_exp = __SIZEOF_REAL128_EXP__; +real128_max_strlen = __REAL128_MAX_STRLEN__; +sizeof_complex128 = __SIZEOF_COMPLEX128__; +real128_mant = __REAL128_MANT_DIG__; +real128_decimal = __REAL128_DECIMAL_DIG__; +real128_old_dig_must_not_exist = __REAL128_DIG__; +real128_max = __REAL128_MAX__; +real128_min = __REAL128_MIN__; +real128_epsilon = __REAL128_EPSILON__; +real128_max_exp = __REAL128_MAX_EXP__; +real128_min_exp = __REAL128_MIN_EXP__; +real1024_type = __REAL1024_TYPE__; +real1024_width = __REAL1024_WIDTH__; +real1024_max_must_not_exist = __REAL1024_MAX__; +real1024_mant = __REAL1024_MANT_DIG__; +real1024_decimal = __REAL1024_DECIMAL_DIG__; +sizeof_real1024 = __SIZEOF_REAL1024__; +sizeof_real1024_exp = __SIZEOF_REAL1024_EXP__; +real1024_max_strlen = __REAL1024_MAX_STRLEN__; +complex1024_type = __COMPLEX1024_TYPE__; +complex1024_width = __COMPLEX1024_WIDTH__; +sizeof_complex1024 = __SIZEOF_COMPLEX1024__; +int65536_type = __INT65536_TYPE__; +int65536_width = __INT65536_WIDTH__; +sizeof_int65536 = __SIZEOF_INT65536__; +char_type_must_not_exist = __CHAR_TYPE__; +wchar_type_must_not_exist = __WCHAR_TYPE__; +EOT + +"$MCPU_CPP" --no-config "$base/main.c" -o "$base/out" + +grep '^arch = 1;$' "$base/out" >/dev/null +grep '^cpp_version = "1\.0\.2";$' "$base/out" >/dev/null +grep '^old_version = __VERSION__;$' "$base/out" >/dev/null +grep '^byte_order = \(1234\|4321\);$' "$base/out" >/dev/null +grep '^mcpu_byte_order = \(1234\|4321\);$' "$base/out" >/dev/null +grep '^word_order = \(1234\|4321\);$' "$base/out" >/dev/null +grep '^old_float_word_order = __FLOAT_WORD_ORDER__;$' "$base/out" >/dev/null +grep '^register_width = 64;$' "$base/out" >/dev/null +grep '^pointer_width = 64;$' "$base/out" >/dev/null +grep '^int_max_width = [1-9][0-9]*;$' "$base/out" >/dev/null +grep '^real_max_width = [1-9][0-9]*;$' "$base/out" >/dev/null +grep '^complex_max_width = [1-9][0-9]*;$' "$base/out" >/dev/null +grep '^size_type = uint[0-9][0-9]*;$' "$base/out" >/dev/null +grep '^size_width = [1-9][0-9]*;$' "$base/out" >/dev/null +grep '^size_sizeof = [1-9][0-9]*;$' "$base/out" >/dev/null +grep '^size_max = 0x[0-9a-f][0-9a-f]*;$' "$base/out" >/dev/null +grep '^ssize_type = int[0-9][0-9]*;$' "$base/out" >/dev/null +grep '^ssize_width = [1-9][0-9]*;$' "$base/out" >/dev/null +grep '^ssize_sizeof = [1-9][0-9]*;$' "$base/out" >/dev/null +grep '^ssize_max = 0x[0-9a-f][0-9a-f]*;$' "$base/out" >/dev/null +grep '^ptrdiff_type = int64;$' "$base/out" >/dev/null +grep '^ptrdiff_width = 64;$' "$base/out" >/dev/null +grep '^ptrdiff_max = 0x7fffffffffffffff;$' "$base/out" >/dev/null +grep '^sizeof_ptrdiff = 8;$' "$base/out" >/dev/null +grep '^old_sizeof_ptrdiff_t = __SIZEOF_PTRDIFF_T__;$' "$base/out" >/dev/null +grep '^intptr_type = int64;$' "$base/out" >/dev/null +grep '^uintptr_type = uint64;$' "$base/out" >/dev/null +grep '^intptr_width = 64;$' "$base/out" >/dev/null +grep '^uintptr_width = 64;$' "$base/out" >/dev/null +grep '^intptr_max = 0x7fffffffffffffff;$' "$base/out" >/dev/null +grep '^uintptr_max = 0xffffffffffffffff;$' "$base/out" >/dev/null +grep '^char8_type = char8;$' "$base/out" >/dev/null +grep '^char16_type = char16;$' "$base/out" >/dev/null +grep '^char8_width = 8;$' "$base/out" >/dev/null +grep '^char16_width = 16;$' "$base/out" >/dev/null +grep '^sizeof_char8 = 1;$' "$base/out" >/dev/null +grep '^sizeof_char16 = 2;$' "$base/out" >/dev/null +grep '^register_prefix = \[\];$' "$base/out" >/dev/null +grep '^local_label_prefix = \[\];$' "$base/out" >/dev/null +grep '^user_label_prefix = \[\];$' "$base/out" >/dev/null +grep '^immediate_prefix = \[\];$' "$base/out" >/dev/null +grep '^int128_type = int128;$' "$base/out" >/dev/null +grep '^uint128_type = uint128;$' "$base/out" >/dev/null +grep '^int128_width = 128;$' "$base/out" >/dev/null +grep '^int128_decimal = 39;$' "$base/out" >/dev/null +grep '^uint128_decimal = 39;$' "$base/out" >/dev/null +grep '^sizeof_int128 = 16;$' "$base/out" >/dev/null +grep '^sizeof_uint128 = 16;$' "$base/out" >/dev/null +grep '^sizeof_real128_exp = [1-9][0-9]*;$' "$base/out" >/dev/null +grep '^real128_max_strlen = [1-9][0-9]*;$' "$base/out" >/dev/null +grep '^int128_max = 0x7fffffffffffffffffffffffffffffff;$' "$base/out" >/dev/null +grep '^int128_min_must_not_exist = __INT128_MIN__;$' "$base/out" >/dev/null +grep '^uint128_max = 0xffffffffffffffffffffffffffffffff;$' "$base/out" >/dev/null +grep '^int1024_type = int1024;$' "$base/out" >/dev/null +grep '^int1024_width = 1024;$' "$base/out" >/dev/null +grep '^int1024_max_must_not_exist = __INT1024_MAX__;$' "$base/out" >/dev/null +grep '^int1024_decimal = 308;$' "$base/out" >/dev/null +grep '^sizeof_int1024 = 128;$' "$base/out" >/dev/null +grep '^real128_type = real128;$' "$base/out" >/dev/null +grep '^complex128_type = complex128;$' "$base/out" >/dev/null +grep '^real128_width = 128;$' "$base/out" >/dev/null +grep '^complex128_width = 128;$' "$base/out" >/dev/null +grep '^sizeof_real128 = 16;$' "$base/out" >/dev/null +grep '^sizeof_complex128 = 32;$' "$base/out" >/dev/null +grep '^real128_mant = [1-9][0-9]*;$' "$base/out" >/dev/null +grep '^real128_decimal = [1-9][0-9]*;$' "$base/out" >/dev/null +grep '^real128_old_dig_must_not_exist = __REAL128_DIG__;$' "$base/out" >/dev/null +grep '^real128_max = [0-9][0-9.]*e+[0-9][0-9]*;$' "$base/out" >/dev/null +grep '^real128_min = [0-9][0-9.]*e-[0-9][0-9]*;$' "$base/out" >/dev/null +grep '^real128_epsilon = [0-9][0-9.]*e-[0-9][0-9]*;$' "$base/out" >/dev/null +grep '^real128_max_exp = [1-9][0-9]*;$' "$base/out" >/dev/null +grep '^real128_min_exp = -[1-9][0-9]*;$' "$base/out" >/dev/null +grep '^real1024_type = real1024;$' "$base/out" >/dev/null +grep '^real1024_width = 1024;$' "$base/out" >/dev/null +grep '^real1024_max_must_not_exist = __REAL1024_MAX__;$' "$base/out" >/dev/null +grep '^real1024_mant = [1-9][0-9]*;$' "$base/out" >/dev/null +grep '^real1024_decimal = [1-9][0-9]*;$' "$base/out" >/dev/null +grep '^sizeof_real1024 = 128;$' "$base/out" >/dev/null +grep '^sizeof_real1024_exp = [1-9][0-9]*;$' "$base/out" >/dev/null +grep '^real1024_max_strlen = [1-9][0-9]*;$' "$base/out" >/dev/null +grep '^complex1024_type = complex1024;$' "$base/out" >/dev/null +grep '^complex1024_width = 1024;$' "$base/out" >/dev/null +grep '^sizeof_complex1024 = 256;$' "$base/out" >/dev/null +grep '^int65536_type = int65536;$' "$base/out" >/dev/null +grep '^int65536_width = 65536;$' "$base/out" >/dev/null +grep '^sizeof_int65536 = 8192;$' "$base/out" >/dev/null +grep '^real_io_limit = [1-9][0-9]*;$' "$base/out" >/dev/null +grep '^math_fn_limit = [1-9][0-9]*;$' "$base/out" >/dev/null +grep '^char_type_must_not_exist = __CHAR_TYPE__;$' "$base/out" >/dev/null +grep '^wchar_type_must_not_exist = __WCHAR_TYPE__;$' "$base/out" >/dev/null diff --git a/tests/t0025-dump-macros.sh b/tests/t0025-dump-macros.sh new file mode 100755 index 0000000..ea0ede5 --- /dev/null +++ b/tests/t0025-dump-macros.sh @@ -0,0 +1,99 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-dm-$$" +mkdir -p "$base" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +"$MCPU_CPP" --no-config -dMP < /dev/null > "$base/macros" + +test -s "$base/macros" +LC_ALL=C sort -c "$base/macros" +grep '^#define _ARCH_MCPU 1$' "$base/macros" >/dev/null +grep '^#define __MCPU_CPP_VERSION__ "1\.0\.2"$' "$base/macros" >/dev/null +grep '^#define __MCPU_MACHINE_REGISTER_WIDTH__ 64$' "$base/macros" >/dev/null +grep '^#define __MCPU_REAL_IO_LIMIT__ [0-9][0-9]*$' "$base/macros" >/dev/null +grep '^#define __MCPU_MATH_FN_LIMIT__ [0-9][0-9]*$' "$base/macros" >/dev/null +grep '^#define __MCPU_BYTE_ORDER__ __ORDER_\(LITTLE\|BIG\)_ENDIAN__$' "$base/macros" >/dev/null +grep '^#define __MCPU_WORD_ORDER__ __ORDER_\(LITTLE\|BIG\)_ENDIAN__$' "$base/macros" >/dev/null +grep '^#define __BYTE_ORDER__ __MCPU_BYTE_ORDER__$' "$base/macros" >/dev/null +grep '^#define __MCPU_INT_MAX_WIDTH__ [1-9][0-9]*$' "$base/macros" >/dev/null +grep '^#define __MCPU_REAL_MAX_WIDTH__ [1-9][0-9]*$' "$base/macros" >/dev/null +grep '^#define __MCPU_COMPLEX_MAX_WIDTH__ [1-9][0-9]*$' "$base/macros" >/dev/null +grep '^#define __REGISTER_PREFIX__$' "$base/macros" >/dev/null +grep '^#define __USER_LABEL_PREFIX__$' "$base/macros" >/dev/null +grep '^#define __MCPU_SIZE_TYPE__ uint[0-9][0-9]*$' "$base/macros" >/dev/null +grep '^#define __MCPU_SIZE_WIDTH__ [1-9][0-9]*$' "$base/macros" >/dev/null +grep '^#define __MCPU_SIZEOF_SIZE__ [1-9][0-9]*$' "$base/macros" >/dev/null +grep '^#define __MCPU_SIZE_MAX__ 0x[0-9a-f][0-9a-f]*$' "$base/macros" >/dev/null +grep '^#define __PTRDIFF_TYPE__ int64$' "$base/macros" >/dev/null +grep '^#define __PTRDIFF_WIDTH__ 64$' "$base/macros" >/dev/null +grep '^#define __PTRDIFF_MAX__ 0x7fffffffffffffff$' "$base/macros" >/dev/null +grep '^#define __SIZEOF_PTRDIFF__ 8$' "$base/macros" >/dev/null +grep '^#define __INTPTR_TYPE__ int64$' "$base/macros" >/dev/null +grep '^#define __INTPTR_WIDTH__ 64$' "$base/macros" >/dev/null +grep '^#define __INTPTR_MAX__ 0x7fffffffffffffff$' "$base/macros" >/dev/null +grep '^#define __MCPU_SSIZE_TYPE__ int[0-9][0-9]*$' "$base/macros" >/dev/null +grep '^#define __MCPU_SSIZE_WIDTH__ [1-9][0-9]*$' "$base/macros" >/dev/null +grep '^#define __MCPU_SIZEOF_SSIZE__ [1-9][0-9]*$' "$base/macros" >/dev/null +grep '^#define __MCPU_SSIZE_MAX__ 0x[0-9a-f][0-9a-f]*$' "$base/macros" >/dev/null +grep '^#define __UINTPTR_TYPE__ uint64$' "$base/macros" >/dev/null +grep '^#define __UINTPTR_WIDTH__ 64$' "$base/macros" >/dev/null +grep '^#define __UINTPTR_MAX__ 0xffffffffffffffff$' "$base/macros" >/dev/null +grep '^#define __CHAR8_TYPE__ char8$' "$base/macros" >/dev/null +grep '^#define __CHAR16_TYPE__ char16$' "$base/macros" >/dev/null +grep '^#define __SIZEOF_CHAR8__ 1$' "$base/macros" >/dev/null +grep '^#define __SIZEOF_CHAR16__ 2$' "$base/macros" >/dev/null +grep '^#define __INT128_DECIMAL_DIG__ 39$' "$base/macros" >/dev/null +grep '^#define __UINT128_DECIMAL_DIG__ 39$' "$base/macros" >/dev/null +grep '^#define __REAL128_DECIMAL_DIG__ [1-9][0-9]*$' "$base/macros" >/dev/null +grep '^#define __REAL128_MANT_DIG__ [1-9][0-9]*$' "$base/macros" >/dev/null +grep '^#define __SIZEOF_REAL128_EXP__ 4$' "$base/macros" >/dev/null +grep '^#define __REAL128_MAX_STRLEN__ 60$' "$base/macros" >/dev/null +grep '^#define __REAL128_MAX__ [0-9][0-9.]*e+[0-9][0-9]*$' "$base/macros" >/dev/null +grep '^#define __INT1024_TYPE__ int1024$' "$base/macros" >/dev/null +grep '^#define __INT1024_WIDTH__ 1024$' "$base/macros" >/dev/null +grep '^#define __SIZEOF_INT1024__ 128$' "$base/macros" >/dev/null +grep '^#define __INT1024_DECIMAL_DIG__ 308$' "$base/macros" >/dev/null +grep '^#define __UINT1024_DECIMAL_DIG__ 309$' "$base/macros" >/dev/null +grep '^#define __INT65536_TYPE__ int65536$' "$base/macros" >/dev/null +grep '^#define __INT65536_WIDTH__ 65536$' "$base/macros" >/dev/null +grep '^#define __SIZEOF_INT65536__ 8192$' "$base/macros" >/dev/null +grep '^#define __REAL1024_TYPE__ real1024$' "$base/macros" >/dev/null +grep '^#define __REAL1024_WIDTH__ 1024$' "$base/macros" >/dev/null +grep '^#define __SIZEOF_REAL1024__ 128$' "$base/macros" >/dev/null +grep '^#define __REAL1024_DECIMAL_DIG__ [1-9][0-9]*$' "$base/macros" >/dev/null +grep '^#define __REAL1024_MANT_DIG__ [1-9][0-9]*$' "$base/macros" >/dev/null +grep '^#define __COMPLEX128_TYPE__ complex128$' "$base/macros" >/dev/null +grep '^#define __COMPLEX128_WIDTH__ 128$' "$base/macros" >/dev/null +grep '^#define __SIZEOF_COMPLEX128__ 32$' "$base/macros" >/dev/null +real_limit=`sed -n 's/^#define __MCPU_REAL_IO_LIMIT__ \([0-9][0-9]*\)$/\1/p' "$base/macros"` +test -n "$real_limit" +real_size=`expr "$real_limit" / 8` +complex_size=`expr "$real_limit" / 4` +grep "^#define __REAL${real_limit}_TYPE__ real${real_limit}$" "$base/macros" >/dev/null +grep "^#define __REAL${real_limit}_WIDTH__ ${real_limit}$" "$base/macros" >/dev/null +grep "^#define __SIZEOF_REAL${real_limit}__ ${real_size}$" "$base/macros" >/dev/null +grep "^#define __SIZEOF_REAL${real_limit}_EXP__ [1-9][0-9]*$" "$base/macros" >/dev/null +grep "^#define __REAL${real_limit}_MAX_STRLEN__ [1-9][0-9]*$" "$base/macros" >/dev/null +grep "^#define __REAL${real_limit}_DECIMAL_DIG__ [1-9][0-9]*$" "$base/macros" >/dev/null +grep "^#define __REAL${real_limit}_MANT_DIG__ [1-9][0-9]*$" "$base/macros" >/dev/null +grep "^#define __COMPLEX${real_limit}_TYPE__ complex${real_limit}$" "$base/macros" >/dev/null +grep "^#define __COMPLEX${real_limit}_WIDTH__ ${real_limit}$" "$base/macros" >/dev/null +grep "^#define __SIZEOF_COMPLEX${real_limit}__ ${complex_size}$" "$base/macros" >/dev/null + +for name in \ + __VERSION__ __MPU_MACHINE_REGISTER_WIDTH__ __MPU_REAL_IO_LIMIT__ \ + __MPU_MATH_FN_LIMIT__ __INT128_MIN__ __INT1024_MAX__ \ + __REAL128_DIG__ __REAL1024_MAX__ __FLOAT_WORD_ORDER__ \ + __SIZE_TYPE__ __SIZE_WIDTH__ __SIZE_MAX__ __SIZEOF_SIZE_T__ \ + __MCPU_SIZEOF_SSIZE_T__ __SIZEOF_CHAR8_T__ __SIZEOF_CHAR16_T__ \ + __SIZEOF_PTRDIFF_T__; do + if grep "^#define $name\>" "$base/macros" >/dev/null; then exit 1; fi +done + +if grep '^#define __FILE__' "$base/macros" >/dev/null; then exit 1; fi +if grep '^#define __LINE__' "$base/macros" >/dev/null; then exit 1; fi +if grep '^#define __DATE__' "$base/macros" >/dev/null; then exit 1; fi +if grep '^#define __TIME__' "$base/macros" >/dev/null; then exit 1; fi +if grep '^#define __BASE_FILE__' "$base/macros" >/dev/null; then exit 1; fi +if grep '^#define __INCLUDE_LEVEL__' "$base/macros" >/dev/null; then exit 1; fi diff --git a/tests/t0026-dump-config.sh b/tests/t0026-dump-config.sh new file mode 100755 index 0000000..0778f28 --- /dev/null +++ b/tests/t0026-dump-config.sh @@ -0,0 +1,22 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-dconfig-$$" +mkdir -p "$base" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +cat > "$base/test.conf" <<'EOT' +ZETA = /zeta; +ROOT = /opt/mcpu; +ALPHA = $ROOT/include; +MCPU_CPP_INCLUDE_PATH = $ROOT/include/mcpu; +EOT + +"$MCPU_CPP" --config-file "$base/test.conf" -dconfig > "$base/config" + +test -s "$base/config" +LC_ALL=C sort -c "$base/config" +grep '^ALPHA = /opt/mcpu/include;$' "$base/config" >/dev/null +grep '^MCPU_CPP_INCLUDE_PATH = /opt/mcpu/include/mcpu;$' "$base/config" >/dev/null +grep '^ROOT = /opt/mcpu;$' "$base/config" >/dev/null +grep '^ZETA = /zeta;$' "$base/config" >/dev/null +grep '^MCPU_CPP_SYSTEM_INCLUDE_PATH = ' "$base/config" >/dev/null diff --git a/tests/t0027-system-language-path.sh b/tests/t0027-system-language-path.sh new file mode 100755 index 0000000..0947c07 --- /dev/null +++ b/tests/t0027-system-language-path.sh @@ -0,0 +1,44 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-system-lang-$$" +mkdir -p "$base/system/diff" "$base/system/as" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +cat > "$base/test.conf" <<EOT +MCPU_CPP_SYSTEM_INCLUDE_PATH = $base/system; +EOT + +cat > "$base/system/common.h" <<'EOT' +common_system_header +EOT +cat > "$base/system/diff/lang.h" <<'EOT' +diff_system_header +EOT +cat > "$base/system/as/lang.h" <<'EOT' +as_system_header +EOT +cat > "$base/system/diff/explicit.h" <<'EOT' +explicit_diff_system_header +EOT + +cat > "$base/main.c" <<'EOT' +#include <common.h> +#lang "diff" +#include <lang.h> +#endlang +#lang "as" +#include <lang.h> +#include <diff/explicit.h> +#endlang +EOT + +"$MCPU_CPP" --config-file "$base/test.conf" "$base/main.c" -o "$base/out" + +grep '^common_system_header$' "$base/out" >/dev/null +grep '^diff_system_header$' "$base/out" >/dev/null +grep '^as_system_header$' "$base/out" >/dev/null +grep '^explicit_diff_system_header$' "$base/out" >/dev/null + +if "$MCPU_CPP" --config-file "$base/test.conf" -nostdinc "$base/main.c" -o "$base/no" 2>/dev/null; then + exit 1 +fi diff --git a/tests/t0028-predefined-ranges.sh b/tests/t0028-predefined-ranges.sh new file mode 100755 index 0000000..72693a2 --- /dev/null +++ b/tests/t0028-predefined-ranges.sh @@ -0,0 +1,95 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-predef-ranges-$$" +mkdir -p "$base" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +"$MCPU_CPP" --no-config -dMP < /dev/null > "$base/macros" + +# Integer families: derive the largest advertised width, then verify every +# power-of-two family from int8 through that LibMPU limit. +int_limit=`sed -n 's/^#define __INT\([0-9][0-9]*\)_TYPE__ int[0-9][0-9]*$/\1/p' "$base/macros" | sort -n | tail -1` +test -n "$int_limit" +grep "^#define __MCPU_INT_MAX_WIDTH__ ${int_limit}$" "$base/macros" >/dev/null + +bits=8 +while test "$bits" -le "$int_limit"; do + bytes=`expr "$bits" / 8` + grep "^#define __INT${bits}_TYPE__ int${bits}$" "$base/macros" >/dev/null + grep "^#define __UINT${bits}_TYPE__ uint${bits}$" "$base/macros" >/dev/null + grep "^#define __INT${bits}_WIDTH__ ${bits}$" "$base/macros" >/dev/null + grep "^#define __UINT${bits}_WIDTH__ ${bits}$" "$base/macros" >/dev/null + grep "^#define __SIZEOF_INT${bits}__ ${bytes}$" "$base/macros" >/dev/null + grep "^#define __SIZEOF_UINT${bits}__ ${bytes}$" "$base/macros" >/dev/null + + grep "^#define __INT${bits}_DECIMAL_DIG__ [1-9][0-9]*$" "$base/macros" >/dev/null + grep "^#define __UINT${bits}_DECIMAL_DIG__ [1-9][0-9]*$" "$base/macros" >/dev/null + + if test "$bits" -le 256; then + grep "^#define __INT${bits}_MAX__ 0x[0-9a-f][0-9a-f]*$" "$base/macros" >/dev/null + grep "^#define __UINT${bits}_MAX__ 0x[0-9a-f][0-9a-f]*$" "$base/macros" >/dev/null + else + for name in "__INT${bits}_MAX__" "__UINT${bits}_MAX__"; do + if grep "^#define ${name}\\>" "$base/macros" >/dev/null; then exit 1; fi + done + fi + + if grep "^#define __INT${bits}_MIN__\\>" "$base/macros" >/dev/null; then exit 1; fi + bits=`expr "$bits" \* 2` +done + +# Real/Complex families: their complete structural metadata extends through +# MPU_REAL_IO_LIMIT. Complex WIDTH is the language parameter, not storage bits. +real_limit=`sed -n 's/^#define __MCPU_REAL_IO_LIMIT__ \([0-9][0-9]*\)$/\1/p' "$base/macros"` +test -n "$real_limit" +grep "^#define __MCPU_REAL_MAX_WIDTH__ ${real_limit}$" "$base/macros" >/dev/null +grep "^#define __MCPU_COMPLEX_MAX_WIDTH__ ${real_limit}$" "$base/macros" >/dev/null + +bits=32 +while test "$bits" -le "$real_limit"; do + real_bytes=`expr "$bits" / 8` + complex_bytes=`expr "$bits" / 4` + grep "^#define __REAL${bits}_TYPE__ real${bits}$" "$base/macros" >/dev/null + grep "^#define __REAL${bits}_WIDTH__ ${bits}$" "$base/macros" >/dev/null + grep "^#define __SIZEOF_REAL${bits}__ ${real_bytes}$" "$base/macros" >/dev/null + case "$bits" in + 32) exp_bytes=1; max_strlen=20 ;; + 64) exp_bytes=2; max_strlen=40 ;; + 128) exp_bytes=4; max_strlen=60 ;; + 256) exp_bytes=4; max_strlen=80 ;; + 512) exp_bytes=8; max_strlen=160 ;; + 1024) exp_bytes=8; max_strlen=320 ;; + 2048) exp_bytes=16; max_strlen=640 ;; + 4096) exp_bytes=16; max_strlen=1280 ;; + 8192) exp_bytes=32; max_strlen=2560 ;; + 16384) exp_bytes=32; max_strlen=5120 ;; + 32768) exp_bytes=64; max_strlen=10240 ;; + 65536) exp_bytes=64; max_strlen=20480 ;; + *) exit 1 ;; + esac + grep "^#define __SIZEOF_REAL${bits}_EXP__ ${exp_bytes}$" "$base/macros" >/dev/null + grep "^#define __REAL${bits}_MAX_STRLEN__ ${max_strlen}$" "$base/macros" >/dev/null + grep "^#define __COMPLEX${bits}_TYPE__ complex${bits}$" "$base/macros" >/dev/null + grep "^#define __COMPLEX${bits}_WIDTH__ ${bits}$" "$base/macros" >/dev/null + grep "^#define __SIZEOF_COMPLEX${bits}__ ${complex_bytes}$" "$base/macros" >/dev/null + + grep "^#define __REAL${bits}_DECIMAL_DIG__ [1-9][0-9]*$" "$base/macros" >/dev/null + grep "^#define __REAL${bits}_MANT_DIG__ [1-9][0-9]*$" "$base/macros" >/dev/null + + if test "$bits" -le 256; then + grep "^#define __REAL${bits}_MAX__ " "$base/macros" >/dev/null + grep "^#define __REAL${bits}_MIN__ " "$base/macros" >/dev/null + grep "^#define __REAL${bits}_EPSILON__ " "$base/macros" >/dev/null + grep "^#define __REAL${bits}_MAX_EXP__ " "$base/macros" >/dev/null + grep "^#define __REAL${bits}_MIN_EXP__ " "$base/macros" >/dev/null + grep "^#define __REAL${bits}_MAX_10_EXP__ " "$base/macros" >/dev/null + grep "^#define __REAL${bits}_MIN_10_EXP__ " "$base/macros" >/dev/null + else + for suffix in MAX MIN EPSILON MAX_EXP MIN_EXP MAX_10_EXP MIN_10_EXP; do + if grep "^#define __REAL${bits}_${suffix}__\\>" "$base/macros" >/dev/null; then exit 1; fi + done + fi + + if grep "^#define __REAL${bits}_DIG__\\>" "$base/macros" >/dev/null; then exit 1; fi + bits=`expr "$bits" \* 2` +done diff --git a/tests/t0029-predefined-abi-names.sh b/tests/t0029-predefined-abi-names.sh new file mode 100755 index 0000000..37b2f55 --- /dev/null +++ b/tests/t0029-predefined-abi-names.sh @@ -0,0 +1,38 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-predef-names-$$" +mkdir -p "$base" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +"$MCPU_CPP" --no-config -dMP < /dev/null > "$base/macros" + +grep '^#define __MCPU_SIZE_TYPE__ uint[0-9][0-9]*$' "$base/macros" >/dev/null +grep '^#define __MCPU_SIZE_WIDTH__ [1-9][0-9]*$' "$base/macros" >/dev/null +grep '^#define __MCPU_SIZEOF_SIZE__ [1-9][0-9]*$' "$base/macros" >/dev/null +grep '^#define __MCPU_SIZE_MAX__ 0x[0-9a-f][0-9a-f]*$' "$base/macros" >/dev/null +grep '^#define __MCPU_SSIZE_TYPE__ int[0-9][0-9]*$' "$base/macros" >/dev/null +grep '^#define __MCPU_SSIZE_WIDTH__ [1-9][0-9]*$' "$base/macros" >/dev/null +grep '^#define __MCPU_SIZEOF_SSIZE__ [1-9][0-9]*$' "$base/macros" >/dev/null +grep '^#define __MCPU_SSIZE_MAX__ 0x[0-9a-f][0-9a-f]*$' "$base/macros" >/dev/null +grep '^#define __SIZEOF_PTRDIFF__ 8$' "$base/macros" >/dev/null +grep '^#define __BYTE_ORDER__ __MCPU_BYTE_ORDER__$' "$base/macros" >/dev/null +grep '^#define __MCPU_BYTE_ORDER__ __ORDER_\(LITTLE\|BIG\)_ENDIAN__$' "$base/macros" >/dev/null +grep '^#define __MCPU_WORD_ORDER__ __ORDER_\(LITTLE\|BIG\)_ENDIAN__$' "$base/macros" >/dev/null +grep '^#define __MCPU_INT_MAX_WIDTH__ [1-9][0-9]*$' "$base/macros" >/dev/null +grep '^#define __MCPU_REAL_MAX_WIDTH__ [1-9][0-9]*$' "$base/macros" >/dev/null +grep '^#define __MCPU_COMPLEX_MAX_WIDTH__ [1-9][0-9]*$' "$base/macros" >/dev/null + +grep '^#define __SIZEOF_CHAR8__ 1$' "$base/macros" >/dev/null +grep '^#define __SIZEOF_CHAR16__ 2$' "$base/macros" >/dev/null + +real_limit=`sed -n 's/^#define __MCPU_REAL_IO_LIMIT__ \([0-9][0-9]*\)$/\1/p' "$base/macros"` +test -n "$real_limit" +grep "^#define __SIZEOF_REAL${real_limit}_EXP__ [1-9][0-9]*$" "$base/macros" >/dev/null +grep "^#define __REAL${real_limit}_MAX_STRLEN__ [1-9][0-9]*$" "$base/macros" >/dev/null + +for name in \ + __SIZE_TYPE__ __SIZE_WIDTH__ __SIZE_MAX__ __SIZEOF_SIZE_T__ \ + __MCPU_SIZEOF_SSIZE_T__ __SIZEOF_CHAR8_T__ __SIZEOF_CHAR16_T__ \ + __SIZEOF_PTRDIFF_T__ __FLOAT_WORD_ORDER__; do + if grep "^#define ${name}\\>" "$base/macros" >/dev/null; then exit 1; fi +done diff --git a/tests/t0030-integer-decimal-digits.sh b/tests/t0030-integer-decimal-digits.sh new file mode 100755 index 0000000..08f7fe2 --- /dev/null +++ b/tests/t0030-integer-decimal-digits.sh @@ -0,0 +1,33 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-int-digs-$$" +mkdir -p "$base" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +"$MCPU_CPP" --no-config -dMP < /dev/null > "$base/macros" + +check_digs() +{ + bits="$1" + int_digs="$2" + uint_digs="$3" + + grep "^#define __INT${bits}_DECIMAL_DIG__ ${int_digs}$" "$base/macros" >/dev/null + grep "^#define __UINT${bits}_DECIMAL_DIG__ ${uint_digs}$" "$base/macros" >/dev/null +} + +# Decimal digits of the numeric maxima only: no sign and no terminating NUL. +check_digs 8 3 3 +check_digs 16 5 5 +check_digs 32 10 10 +check_digs 64 19 20 +check_digs 128 39 39 +check_digs 256 77 78 +check_digs 512 154 155 +check_digs 1024 308 309 +check_digs 2048 617 617 +check_digs 4096 1233 1234 +check_digs 8192 2466 2467 +check_digs 16384 4932 4933 +check_digs 32768 9864 9865 +check_digs 65536 19729 19729 diff --git a/tests/t0031-stringification.sh b/tests/t0031-stringification.sh new file mode 100755 index 0000000..213d787 --- /dev/null +++ b/tests/t0031-stringification.sh @@ -0,0 +1,50 @@ +#!/bin/sh +set -eu +src="${TMPDIR-/tmp}/mcpu-cpp-stringify-$$.c" +out="${TMPDIR-/tmp}/mcpu-cpp-stringify-$$.out" +dump="${TMPDIR-/tmp}/mcpu-cpp-stringify-$$.dump" +trap 'rm -f "$src" "$out" "$dump"' EXIT HUP INT TERM + +cat > "$src" <<'EOT' +#define A 7 +#define STR(X) #X +#define WSTR(X) # X +#define XSTR(X) STR(X) +#define MIX(X) #X | X +#define BAD(X, Y) X + Y +#define HASH(X) "#" X +#define EMPTY(X) <X> +STR(A) +WSTR(alpha) +XSTR(A) +STR( a + b ) +STR(Привет мир) +STR("a\\b") +STR() +STR(BAD(1)) +MIX(A) +HASH(A) +EMPTY() +#lang "diff" +#define DSTR(X) #X +DSTR(y' + z) +#endlang +EOT + +"$MCPU_CPP" --no-config "$src" -o "$out" + +grep -Fx '"A"' "$out" >/dev/null +grep -Fx '"alpha"' "$out" >/dev/null +grep -Fx '"7"' "$out" >/dev/null +grep -Fx '"a + b"' "$out" >/dev/null +grep -Fx '"Привет мир"' "$out" >/dev/null +grep -Fx '"\"a\\\\b\""' "$out" >/dev/null +grep -Fx '""' "$out" >/dev/null +grep -Fx '"BAD(1)"' "$out" >/dev/null +grep -Fx '"A" | 7' "$out" >/dev/null +grep -Fx '"#" 7' "$out" >/dev/null +grep -Fx '<>' "$out" >/dev/null +grep -Fx '"y'"'"' + z"' "$out" >/dev/null + +"$MCPU_CPP" --no-config -dM "$src" > "$dump" +grep -Fx '#define STR(X) #X' "$dump" >/dev/null diff --git a/tests/t0032-command-line.sh b/tests/t0032-command-line.sh new file mode 100755 index 0000000..e0deb91 --- /dev/null +++ b/tests/t0032-command-line.sh @@ -0,0 +1,119 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-options-$$" +mkdir -p "$base" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +check() +{ + name=$1 + shift + "$MCPP_OPTIONS_TEST" "$@" > "$base/$name" +} + +check pipe +cat > "$base/pipe.expected" <<'EOT' +input=<stdin> +output=<stdout> +verbose=0 +nostdinc=0 +dump_config=0 +dump_search_dirs=0 +dump_macros=0 +actions=0 +EOT +cmp "$base/pipe.expected" "$base/pipe" + +check files input.S output.s +sed -n '1,2p' "$base/files" > "$base/files.io" +printf '%s\n' 'input=input.S' 'output=output.s' > "$base/files.expected" +cmp "$base/files.expected" "$base/files.io" + +check dash - - +sed -n '1,2p' "$base/dash" > "$base/dash.io" +printf '%s\n' 'input=<stdin>' 'output=<stdout>' > "$base/dash.expected" +cmp "$base/dash.expected" "$base/dash.io" + +check outfile input.S -o result.s +sed -n '1,2p' "$base/outfile" > "$base/outfile.io" +printf '%s\n' 'input=input.S' 'output=result.s' > "$base/outfile.expected" +cmp "$base/outfile.expected" "$base/outfile.io" + +check options -v -nostdinc -Iinc1 -I inc2 -isystem sys -idirafter after input.S - +grep '^verbose=1$' "$base/options" >/dev/null +grep '^nostdinc=1$' "$base/options" >/dev/null +grep '^actions=4$' "$base/options" >/dev/null +grep '^action\[0\]=include-user:inc1$' "$base/options" >/dev/null +grep '^action\[1\]=include-user:inc2$' "$base/options" >/dev/null +grep '^action\[2\]=include-system:sys$' "$base/options" >/dev/null +grep '^action\[3\]=include-after:after$' "$base/options" >/dev/null + +check ordered -DA=1 -UA -DA=2 +sed -n '/^actions=/,$p' "$base/ordered" > "$base/ordered.actions" +cat > "$base/ordered.expected" <<'EOT' +actions=3 +action[0]=define:A=1 +action[1]=undef:A +action[2]=define:A=2 +EOT +cmp "$base/ordered.expected" "$base/ordered.actions" + +check searchdirs -dsearch-dirs +grep '^dump_search_dirs=1$' "$base/searchdirs" >/dev/null + +check dump -dMP input.S -o macros.txt +grep '^dump_macros=1$' "$base/dump" >/dev/null + +check retained -include force.h -imacros macros.h -DA=1 -UOLD -M -MG -w -Wall -Werror input.S output.s +grep '^input=input.S$' "$base/retained" >/dev/null +grep '^output=output.s$' "$base/retained" >/dev/null +grep '^action\[[0-9][0-9]*\]=include:force.h$' "$base/retained" >/dev/null +grep '^action\[[0-9][0-9]*\]=imacros:macros.h$' "$base/retained" >/dev/null +grep '^action\[[0-9][0-9]*\]=define:A=1$' "$base/retained" >/dev/null +grep '^action\[[0-9][0-9]*\]=undef:OLD$' "$base/retained" >/dev/null + +check dumpD -dD input.S +grep '^dump_macros=2$' "$base/dumpD" >/dev/null + +check md -MD input.S +grep '^input=input.S$' "$base/md" >/dev/null +check mmd -MMD input.S +grep '^input=input.S$' "$base/mmd" >/dev/null +check mf -MD -MF deps.mk input.S +grep '^input=input.S$' "$base/mf" >/dev/null +check mfattached -MMD -MFdeps.mk input.S +grep '^input=input.S$' "$base/mfattached" >/dev/null +check mt -M -MT target.o input.S +grep '^action\[[0-9][0-9]*\]=dep-target:target.o$' "$base/mt" >/dev/null +check mtattached -MM -MTtarget.o input.S +grep '^action\[[0-9][0-9]*\]=dep-target:target.o$' "$base/mtattached" >/dev/null +check mq -MD -MQ '$(OBJDIR)/target.o' input.S +grep '^action\[[0-9][0-9]*\]=dep-target-quoted:$(OBJDIR)/target.o$' "$base/mq" >/dev/null +check mqattached -MMD '-MQ$(OBJDIR)/target.o' input.S +grep '^action\[[0-9][0-9]*\]=dep-target-quoted:$(OBJDIR)/target.o$' "$base/mqattached" >/dev/null +if "$MCPP_OPTIONS_TEST" a b c > /dev/null 2>&1; then exit 1; fi +if "$MCPP_OPTIONS_TEST" -o > /dev/null 2>&1; then exit 1; fi +if "$MCPP_OPTIONS_TEST" -isystem > /dev/null 2>&1; then exit 1; fi +if "$MCPP_OPTIONS_TEST" --bad-option > /dev/null 2>&1; then exit 1; fi +if "$MCPP_OPTIONS_TEST" -MF deps.mk input.S > /dev/null 2>&1; then exit 1; fi +if "$MCPP_OPTIONS_TEST" -MT target.o input.S > /dev/null 2>&1; then exit 1; fi +if "$MCPP_OPTIONS_TEST" -MQ target.o input.S > /dev/null 2>&1; then exit 1; fi +if "$MCPP_OPTIONS_TEST" -M -MT > /dev/null 2>&1; then exit 1; fi +if "$MCPP_OPTIONS_TEST" -M -MQ > /dev/null 2>&1; then exit 1; fi + +# Unsupported legacy assertion options remain rejected. +if "$MCPP_OPTIONS_TEST" -Aquestion input.S > /dev/null 2>&1; then exit 1; fi +if "$MCPP_OPTIONS_TEST" -dMA input.S > /dev/null 2>&1; then exit 1; fi + +# Removed compatibility options are not recognized by mcpu-cpp. +for opt in \ + -x -X -out-unix-mode -I- -H -C -P '-$' -r -R -undef -u -dN -g3 -dark \ + -Wimport -Wno-import -pedantic-ANSI-C -pedantic-ANSI-C-errors \ + -lang-k -lang-k++ -lang-k-k++-comments -lang-asm -+ \ + -iprefix -iwithprefix -iwithprefixbefore +do + if "$MCPP_OPTIONS_TEST" "$opt" input.S > /dev/null 2>&1; then + echo "removed option unexpectedly accepted: $opt" >&2 + exit 1 + fi +done diff --git a/tests/t0033-dump-definitions.sh b/tests/t0033-dump-definitions.sh new file mode 100755 index 0000000..2637e53 --- /dev/null +++ b/tests/t0033-dump-definitions.sh @@ -0,0 +1,51 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-dD-$$" +mkdir -p "$base/inc" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +cat > "$base/inc/defs.h" <<'EOT' +#ifndef MCPP_TEST_DEFS_H +#define MCPP_TEST_DEFS_H 1 +#define FROM_HEADER 17 /* header definition */ +#endif /* MCPP_TEST_DEFS_H */ +EOT + +cat > "$base/main.c" <<'EOT' +#include <defs.h> +#define LOCAL_VALUE 25 // local definition +LOCAL_VALUE + FROM_HEADER +EOT + +"$MCPU_CPP" --no-config -I"$base/inc" -dD "$base/main.c" -o "$base/out" + +# -dD starts with an input marker, emits predefined definitions with +# <built-in> markers, keeps source #define directives, and still emits the +# ordinary preprocessed result. +grep '^# 0 "'"$base/main.c"'"$' "$base/out" >/dev/null +grep '^# 0 "<built-in>"$' "$base/out" >/dev/null +grep '^#define __MCPU_CPP_VERSION__ "1.0.2"$' "$base/out" >/dev/null +grep '^#define FROM_HEADER 17$' "$base/out" >/dev/null +grep '^#define LOCAL_VALUE 25$' "$base/out" >/dev/null +if grep '^[[:blank:]]*#[[:blank:]]*\(if\|ifdef\|ifndef\|elif\|else\|endif\)\>' "$base/out" >/dev/null; then + exit 1 +fi +grep '^25 + 17$' "$base/out" >/dev/null +if grep '[[:blank:]]$' "$base/out" >/dev/null; then + exit 1 +fi + +# Every predefined definition in the initial block has its own built-in marker. +awk ' + /^# 1 / { exit } + /^#define / { + if( previous != "# 0 \"<built-in>\"") exit 1 + } + { previous = $0 } +' "$base/out" + +# Without -dD the same source definitions are removed from normal output. +"$MCPU_CPP" --no-config -I"$base/inc" "$base/main.c" -o "$base/normal" +if grep '^#define LOCAL_VALUE 25$' "$base/normal" >/dev/null; then + exit 1 +fi diff --git a/tests/t0034-conditionals.sh b/tests/t0034-conditionals.sh new file mode 100755 index 0000000..85f3d66 --- /dev/null +++ b/tests/t0034-conditionals.sh @@ -0,0 +1,136 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-conditionals-$$" +mkdir -p "$base/inc" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +cat > "$base/inc/guard.h" <<'EOT' +#ifndef MCPP_TEST_GUARD_H +#define MCPP_TEST_GUARD_H 1 +#define HEADER_VALUE 17 +header_line +#endif +EOT + +cat > "$base/main.c" <<'EOT' +#define A 1 +#define B 2 +#define F(x) ((x) + 1) + +#include <guard.h> +#include <guard.h> + +#if A + B * 3 == 7 && defined(A) && !defined(NO_SUCH_MACRO) +if_line +#else +bad_if +#endif + +#if 0 +#define BAD_SKIPPED 1 +bad_elif_0 +#elif F(B) == 3 +elif_line +#else +bad_elif_else +#endif + +#ifdef A +ifdef_line +#else +bad_ifdef +#endif + +#ifndef NO_SUCH_MACRO +ifndef_line +#endif + +#if 0 +# if 1 / 0 +bad_nested +# endif +#else +nested_skip_line +#endif + +#if 0 && (1 / 0) +bad_short_and +#endif + +#if 1 || (1 / 0) +short_or_line +#endif + +#if 1 ? 2 : 1 / 0 +ternary_true_line +#endif + +#if 0 ? 1 / 0 : 3 +ternary_false_line +#endif + +#if UNKNOWN_IDENTIFIER +bad_unknown +#else +unknown_zero_line +#endif +EOT + +"$MCPU_CPP" --no-config -I"$base/inc" -dD "$base/main.c" -o "$base/out" + +for word in \ + header_line if_line elif_line ifdef_line ifndef_line nested_skip_line \ + short_or_line ternary_true_line ternary_false_line unknown_zero_line; do + test "`grep -c "^${word}$" "$base/out"`" -eq 1 +done + +for word in \ + bad_if bad_elif_0 bad_elif_else bad_ifdef bad_nested bad_short_and bad_unknown; do + if grep "^${word}$" "$base/out" >/dev/null; then exit 1; fi +done + +# Conditional-control directives are consumed by the preprocessor and never +# survive in the normal output, including -dD output. +if grep '^[[:blank:]]*#[[:blank:]]*\(if\|ifdef\|ifndef\|elif\|else\|endif\)\>' "$base/out" >/dev/null; then + exit 1 +fi + +# Active definitions are retained by -dD, skipped ones are not. +grep '^#define A 1$' "$base/out" >/dev/null +grep '^#define HEADER_VALUE 17$' "$base/out" >/dev/null +if grep '^#define BAD_' "$base/out" >/dev/null; then exit 1; fi + +cat > "$base/bad-endif.c" <<'EOT' +#endif +EOT +if "$MCPU_CPP" --no-config "$base/bad-endif.c" -o "$base/bad-out" 2>/dev/null; then + exit 1 +fi + +cat > "$base/bad-else.c" <<'EOT' +#if 1 +#else +#else +#endif +EOT +if "$MCPU_CPP" --no-config "$base/bad-else.c" -o "$base/bad-out" 2>/dev/null; then + exit 1 +fi + +cat > "$base/bad-elif.c" <<'EOT' +#if 0 +#else +#elif 1 +#endif +EOT +if "$MCPU_CPP" --no-config "$base/bad-elif.c" -o "$base/bad-out" 2>/dev/null; then + exit 1 +fi + +cat > "$base/unterminated.c" <<'EOT' +#if 1 +unterminated +EOT +if "$MCPU_CPP" --no-config "$base/unterminated.c" -o "$base/bad-out" 2>/dev/null; then + exit 1 +fi diff --git a/tests/t0035-line-control.sh b/tests/t0035-line-control.sh new file mode 100755 index 0000000..b892015 --- /dev/null +++ b/tests/t0035-line-control.sh @@ -0,0 +1,41 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-line-control-$$" +mkdir -p "$base/inc" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +cat > "$base/inc/one.h" <<'EOT' +int header_line = __LINE__; +EOT + +cat > "$base/main.c" <<'EOT' +int before = __LINE__; +#line 62 "main.y" +int after = __LINE__; +const char *file_after = __FILE__; +#include "one.h" +int returned = __LINE__; +#define LINE_NUMBER 90 +#line LINE_NUMBER "macro-line.y" +int macro_line = __LINE__; +EOT + +"$MCPU_CPP" --no-config -I"$base/inc" "$base/main.c" -o "$base/out" + +# Generated line information uses GNU linemarker syntax, not input #line syntax. +if grep '^#line[[:blank:]]' "$base/out" >/dev/null; then exit 1; fi + +grep "^# 1 \"$base/main.c\"$" "$base/out" >/dev/null +grep '^# 62 "main.y"$' "$base/out" >/dev/null +grep '^int after = 62;$' "$base/out" >/dev/null +grep '^const char \*file_after = "main.y";$' "$base/out" >/dev/null + +# Flag 1 means entering an included file; flag 2 means returning to its includer. +grep "^# 1 \"$base/inc/one.h\" 1$" "$base/out" >/dev/null +grep '^# 65 "main.y" 2$' "$base/out" >/dev/null +grep '^int returned = 65;$' "$base/out" >/dev/null + +# Arguments of #line are macro-expanded before interpretation. +grep '^# 90 "macro-line.y"$' "$base/out" >/dev/null +grep '^int macro_line = 90;$' "$base/out" >/dev/null + diff --git a/tests/t0036-lang-string.sh b/tests/t0036-lang-string.sh new file mode 100755 index 0000000..8e40022 --- /dev/null +++ b/tests/t0036-lang-string.sh @@ -0,0 +1,120 @@ +#!/bin/sh +set -eu +dir="${TMPDIR-/tmp}/mcpu-cpp-lang-string-$$" +mkdir -p "$dir" +trap 'rm -rf "$dir"' EXIT HUP INT TERM + +cat > "$dir/normalize.c" <<'EOT' + # lang "DiFf" +#endlang + #lang "dIfT" +#endlang +EOT +"$MCPU_CPP" --no-config "$dir/normalize.c" -o "$dir/normalize.out" +grep '^#lang "DiFf"$' "$dir/normalize.out" >/dev/null +grep '^#lang "dIfT"$' "$dir/normalize.out" >/dev/null +if grep '^[[:space:]][[:space:]]*#lang' "$dir/normalize.out" >/dev/null +then + echo '#lang leading whitespace was not normalized' >&2 + exit 1 +fi + +cat > "$dir/case.c" <<'EOT' +#lang "diff" +#endlang +#lang "Diff" +#endlang +#lang "DIFF" +#endlang +#lang "dIfF" +y' = 1; +#endlang +#lang "dIfT" +#endlang +#lang "ALG" +#endlang +#lang "As" +#endlang +#lang "aVm" +#endlang +#lang "aCs" +#endlang +EOT +"$MCPU_CPP" --no-config "$dir/case.c" -o "$dir/case.out" +for spelling in diff Diff DIFF dIfF dIfT ALG As aVm aCs +do + grep "^#lang \"$spelling\"$" "$dir/case.out" >/dev/null +done +grep "^y' = 1;$" "$dir/case.out" >/dev/null + +check_bad() +{ + name=$1 + pattern=$2 + if "$MCPU_CPP" --no-config "$dir/$name.c" -o "$dir/$name.out" 2> "$dir/$name.err" + then + echo "$name unexpectedly accepted" >&2 + exit 1 + fi + grep -F "$pattern" "$dir/$name.err" >/dev/null +} + +cat > "$dir/unquoted.c" <<'EOT' +#lang diff +EOT +check_bad unquoted 'string constant expected after #lang' + +cat > "$dir/empty.c" <<'EOT' +#lang "" +EOT +check_bad empty 'empty language name in #lang' + +cat > "$dir/leading-space.c" <<'EOT' +#lang " diff" +EOT +check_bad leading-space 'one word without whitespace' + +cat > "$dir/trailing-space.c" <<'EOT' +#lang "diff " +EOT +check_bad trailing-space 'one word without whitespace' + +cat > "$dir/internal-space.c" <<'EOT' +#lang "di ff" +EOT +check_bad internal-space 'one word without whitespace' + +printf '#lang "di\tff"\n' > "$dir/internal-tab.c" +check_bad internal-tab 'one word without whitespace' + +cat > "$dir/unknown.c" <<'EOT' +#lang "cplusplus" +EOT +check_bad unknown "unknown language 'cplusplus'" + +cat > "$dir/zero.c" <<'EOT' +#lang "0" +EOT +check_bad zero "unknown language '0'" + +cat > "$dir/extra.c" <<'EOT' +#lang "diff" extra +EOT +check_bad extra 'extra text after #lang string constant' + +cat > "$dir/multiline.c" <<'EOT' +#lang "di +ff" +EOT +check_bad multiline 'unterminated #lang string constant' + +cat > "$dir/spliced.c" <<'EOT' +#lang "di\ +ff" +EOT +check_bad spliced 'one physical source line' + +cat > "$dir/no-escape.c" <<'EOT' +#lang "d\iff" +EOT +check_bad no-escape "unknown language 'd\\iff'" diff --git a/tests/t0037-token-concatenation.sh b/tests/t0037-token-concatenation.sh new file mode 100755 index 0000000..f2d2c3f --- /dev/null +++ b/tests/t0037-token-concatenation.sh @@ -0,0 +1,122 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-concat-$$" +mkdir -p "$base" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +cat > "$base/main.c" <<'EOT' +#define A left +#define B right +#define AB raw_result +#define leftright expanded_result +#define CAT(X, Y) X ## Y +#define XCAT(X, Y) CAT(X, Y) +#define AFTERX(X) X_ ## X +#define XAFTERX(X) AFTERX(X) +#define TABLESIZE 1024 +#define BUFSIZE TABLESIZE +#define MADE 91 +#define MK(A, B) A ## B +#define L(X) X ## tail +#define R(X) head ## X +#define CHAIN(A, B, C) A ## B ## C +#define OBJECT obj ## ect +#define object object_result +#define COMMAND(NAME) #NAME | NAME ## _command +#define quit_command command_result +#define CALL(NAME) NAME ## _fn(1) +#define run_fn(X) called_result +#define ONE1 1 +#define PCAT(A, B) A ## B +#define RUN(A, B) A ## ## B +#define ab adjacent_result +#define TEXT "a ## b" +#define COMMENT_CAT(A, B) A /* left */ ## /* right */ B +#define DIRECT_EMPTY +#define xDIRECT_EMPTY direct_empty_raw +#define EXPAND_EMPTY(A, B) CAT(A, B) + +CAT(A, B) +XCAT(A, B) +AFTERX(BUFSIZE) +XAFTERX(BUFSIZE) +MK(MA, DE) +CAT(x y, z) +CAT(x, y z) +L() +R() +CHAIN(x, , z) +OBJECT +CAT(1.5, e3) +CAT(+, =) +CAT(L, "wide") +CAT(u8, "utf8") +COMMAND(quit) +CALL(run) +RUN(a, b) +CAT(x, DIRECT_EMPTY) +EXPAND_EMPTY(x, DIRECT_EMPTY) +TEXT +COMMENT_CAT(c, d) +CAT(e /* argument comment */, f) +#if PCAT(ONE, 1) +conditional_result +#endif +EOT + +"$MCPU_CPP" --no-config "$base/main.c" -o "$base/out" + +grep -Fx 'raw_result' "$base/out" >/dev/null +grep -Fx 'expanded_result' "$base/out" >/dev/null +grep -Fx 'X_BUFSIZE' "$base/out" >/dev/null +grep -Fx 'X_1024' "$base/out" >/dev/null +grep -Fx '91' "$base/out" >/dev/null +grep -Fx 'x yz' "$base/out" >/dev/null +grep -Fx 'xy z' "$base/out" >/dev/null +grep -Fx 'tail' "$base/out" >/dev/null +grep -Fx 'head' "$base/out" >/dev/null +grep -Fx 'xz' "$base/out" >/dev/null +grep -Fx 'object_result' "$base/out" >/dev/null +grep -Fx '1.5e3' "$base/out" >/dev/null +grep -Fx '+=' "$base/out" >/dev/null +grep -Fx 'L"wide"' "$base/out" >/dev/null +grep -Fx 'u8"utf8"' "$base/out" >/dev/null +grep -Fx '"quit" | command_result' "$base/out" >/dev/null +grep -Fx 'called_result' "$base/out" >/dev/null +grep -Fx 'adjacent_result' "$base/out" >/dev/null +grep -Fx 'direct_empty_raw' "$base/out" >/dev/null +grep -Fx 'x' "$base/out" >/dev/null +grep -Fx '"a ## b"' "$base/out" >/dev/null +grep -Fx 'cd' "$base/out" >/dev/null +grep -Fx 'ef' "$base/out" >/dev/null +grep -Fx 'conditional_result' "$base/out" >/dev/null + +"$MCPU_CPP" --no-config -dM "$base/main.c" > "$base/dump" +grep -Fx '#define CAT(X,Y) X ## Y' "$base/dump" >/dev/null +grep -Fx '#define COMMAND(NAME) #NAME | NAME ## _command' "$base/dump" >/dev/null + +cat > "$base/invalid.c" <<'EOT' +#define BAD(X) X ## + +BAD(foo) +EOT +"$MCPU_CPP" --no-config "$base/invalid.c" -o "$base/invalid.out" 2>"$base/invalid.err" +grep 'pasting "foo" and "+" does not give a valid preprocessing token' "$base/invalid.err" >/dev/null +grep 'foo' "$base/invalid.out" | tr -d '[:space:]' | grep -Fx 'foo+' >/dev/null + +cat > "$base/first.c" <<'EOT' +#define FIRST(X) ## X +EOT +if "$MCPU_CPP" --no-config "$base/first.c" -o "$base/first.out" 2>"$base/first.err"; then + echo "leading ## accepted" >&2 + exit 1 +fi +grep "'##' cannot appear at the beginning of a macro replacement list" "$base/first.err" >/dev/null + +cat > "$base/last.c" <<'EOT' +#define LAST(X) X ## +EOT +if "$MCPU_CPP" --no-config "$base/last.c" -o "$base/last.out" 2>"$base/last.err"; then + echo "trailing ## accepted" >&2 + exit 1 +fi +grep "'##' cannot appear at the end of a macro replacement list" "$base/last.err" >/dev/null diff --git a/tests/t0038-ucs2-identifiers.sh b/tests/t0038-ucs2-identifiers.sh new file mode 100755 index 0000000..d1ae5db --- /dev/null +++ b/tests/t0038-ucs2-identifiers.sh @@ -0,0 +1,122 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-ucs2-identifiers-$$" +mkdir -p "$base" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +# The source file is UTF-8 externally. mcpu-cpp decodes it to strict UCS-2, +# and identifier classification is then performed with the locale-independent +# Unicode XID_Start/XID_Continue predicates provided by LibMPUIO. +cat > "$base/main.c" <<'EOT' +#define АНДРЕЙ 11 +#define résumé 12 +#define Élève 13 +#define ΩΜΕΓΑ 14 +#define переменная2 15 +#define a١ 16 +#define é 17 +#define СУММА(лево, право) лево + право +#define СТРОКА(имя) #имя +#define СКЛЕИТЬ(лево, право) лево ## право +#define АНДРЕЙ2 18 +#define ВРЕМЕННЫЙ 19 +#undef ВРЕМЕННЫЙ +#define VALUE$OLD 20 +#define ДОЛЛАР$ИМЯ 21 +#define DOLLAR_PARAM(x$old) x$old +#define CAT_DOLLAR(a,b) a ## b +#define LEFT$RIGHT 22 + +АНДРЕЙ +résumé +Élève +ΩΜΕΓΑ +переменная2 +a١ +é +СУММА(АНДРЕЙ, résumé) +СТРОКА(АНДРЕЙ) +СКЛЕИТЬ(АНДР, ЕЙ) +СКЛЕИТЬ(АНДРЕЙ, 2) +андрей +ВРЕМЕННЫЙ +VALUE$OLD +ДОЛЛАР$ИМЯ +DOLLAR_PARAM(23) +CAT_DOLLAR(LEFT$, RIGHT) + +#ifdef АНДРЕЙ +ifdef_ok +#endif + +#ifndef ВРЕМЕННЫЙ +ifndef_ok +#endif + +#if defined(résumé) && defined(ΩΜΕΓΑ) +defined_ok +#endif + +#ifdef VALUE$OLD +dollar_ifdef_ok +#endif + +#if defined(ДОЛЛАР$ИМЯ) +dollar_defined_ok +#endif +EOT + +"$MCPU_CPP" --no-config "$base/main.c" -o "$base/out" + +grep -Fx '11' "$base/out" >/dev/null +grep -Fx '12' "$base/out" >/dev/null +grep -Fx '13' "$base/out" >/dev/null +grep -Fx '14' "$base/out" >/dev/null +grep -Fx '15' "$base/out" >/dev/null +grep -Fx '16' "$base/out" >/dev/null +grep -Fx '17' "$base/out" >/dev/null +grep -Fx '11 + 12' "$base/out" >/dev/null +grep -Fx '"АНДРЕЙ"' "$base/out" >/dev/null +# Raw operands of ## are concatenated, then the result is rescanned. +grep -Fx '11' "$base/out" >/dev/null +grep -Fx '18' "$base/out" >/dev/null +grep -Fx 'андрей' "$base/out" >/dev/null +grep -Fx 'ВРЕМЕННЫЙ' "$base/out" >/dev/null +grep -Fx 'ifdef_ok' "$base/out" >/dev/null +grep -Fx 'ifndef_ok' "$base/out" >/dev/null +grep -Fx 'defined_ok' "$base/out" >/dev/null +grep -Fx '20' "$base/out" >/dev/null +grep -Fx '21' "$base/out" >/dev/null +grep -Fx '23' "$base/out" >/dev/null +grep -Fx '22' "$base/out" >/dev/null +grep -Fx 'dollar_ifdef_ok' "$base/out" >/dev/null +grep -Fx 'dollar_defined_ok' "$base/out" >/dev/null + +# XID_Continue admits combining marks after an identifier start. The same +# combining mark is not XID_Start and therefore cannot begin a macro name. +printf '#define \314\201bad 1\n' > "$base/bad-start.c" +if "$MCPU_CPP" --no-config "$base/bad-start.c" -o "$base/bad-start.out" 2>"$base/bad-start.err"; then + echo "combining mark accepted as identifier start" >&2 + exit 1 +fi + +grep 'macro name expected after #define' "$base/bad-start.err" >/dev/null + +# A decimal digit is XID_Continue but not XID_Start. Non-ASCII digits are +# therefore legal after the first character but cannot start an identifier. +printf '#define \331\241bad 1\n' > "$base/bad-digit-start.c" +if "$MCPU_CPP" --no-config "$base/bad-digit-start.c" -o "$base/bad-digit-start.out" 2>"$base/bad-digit-start.err"; then + echo "non-ASCII digit accepted as identifier start" >&2 + exit 1 +fi + +grep 'macro name expected after #define' "$base/bad-digit-start.err" >/dev/null + + +# Dollar is an identifier-continuation extension, never an identifier start. +printf '#define $BAD 1\n' > "$base/bad-dollar-start.c" +if "$MCPU_CPP" --no-config "$base/bad-dollar-start.c" -o "$base/bad-dollar-start.out" 2>"$base/bad-dollar-start.err"; then + echo "dollar accepted as identifier start" >&2 + exit 1 +fi +grep 'macro name expected after #define' "$base/bad-dollar-start.err" >/dev/null diff --git a/tests/t0039-command-line-macros.sh b/tests/t0039-command-line-macros.sh new file mode 100755 index 0000000..94a5ec3 --- /dev/null +++ b/tests/t0039-command-line-macros.sh @@ -0,0 +1,174 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-command-line-macros-$$" +mkdir -p "$base" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +cat > "$base/main.c" <<'EOT' +OBJECT +DEFAULT +EMPTY marker +СУММА(10, 7) +АНДРЕЙ +résumé +ПОРЯДОК +СТРОКА(АНДРЕЙ) +СКЛЕИТЬ(АНД,РЕЙ) +#ifdef __MCPU_CPP_VERSION__ +BUILTIN_PRESENT +#else +BUILTIN_REMOVED +#endif +#if defined(КОМАНДА) && КОМАНДА == 31 +CONDITIONAL_OK +#else +CONDITIONAL_BAD +#endif +EOT + +"$MCPU_CPP" --no-config \ + -D OBJECT=17 \ + -DDEFAULT \ + -DEMPTY= \ + '-DСУММА(лево,право)=лево + право' \ + -DАНДРЕЙ=23 \ + -Drésumé=29 \ + -DПОРЯДОК=1 -U ПОРЯДОК -DПОРЯДОК=2 \ + '-DСТРОКА(x)=#x' \ + '-DСКЛЕИТЬ(a,b)=a ## b' \ + -DАНДРЕЙ=23 \ + -U__MCPU_CPP_VERSION__ \ + -DКОМАНДА=31 \ + "$base/main.c" -o "$base/out" + +grep '^17$' "$base/out" >/dev/null +grep '^1$' "$base/out" >/dev/null +grep '^ marker$' "$base/out" >/dev/null +grep '^10 + 7$' "$base/out" >/dev/null +test "`grep -c '^23$' "$base/out"`" -eq 2 +grep '^29$' "$base/out" >/dev/null +grep '^2$' "$base/out" >/dev/null +grep '^"АНДРЕЙ"$' "$base/out" >/dev/null +grep '^BUILTIN_REMOVED$' "$base/out" >/dev/null +if grep '^BUILTIN_PRESENT$' "$base/out" >/dev/null; then exit 1; fi +grep '^CONDITIONAL_OK$' "$base/out" >/dev/null +if grep '^CONDITIONAL_BAD$' "$base/out" >/dev/null; then exit 1; fi + +# A function-like -D without '=' has the GNU meaning "replacement is 1". +printf '%s\n' 'F(99)' > "$base/default-func.c" +"$MCPU_CPP" --no-config '-DF(x)' "$base/default-func.c" -o "$base/default-func.out" +grep '^1$' "$base/default-func.out" >/dev/null + +# Unicode XID continuation works on the command line as it does in source. +combining=$(printf 'e\314\201') +printf '%s\n' "$combining" > "$base/combining.c" +"$MCPU_CPP" --no-config "-D$combining=37" "$base/combining.c" -o "$base/combining.out" +grep '^37$' "$base/combining.out" >/dev/null + +arabic_three=$(printf '\331\243') +name="ИМЯ$arabic_three" +printf '%s\n' "$name" > "$base/digit.c" +"$MCPU_CPP" --no-config "-D$name=43" "$base/digit.c" -o "$base/digit.out" +grep '^43$' "$base/digit.out" >/dev/null + +# Command-line definitions pass through the same phase-3 comment cleanup as +# source #define directives. +printf '%s\n' 'COMMENTED' > "$base/comment.c" +"$MCPU_CPP" --no-config '-DCOMMENTED=left/* comment */right' \ + "$base/comment.c" -o "$base/comment.out" +grep '^left right$' "$base/comment.out" >/dev/null + +# -dM observes the final command-line macro state. +printf '%s\n' '' > "$base/empty.c" +"$MCPU_CPP" --no-config -dM -DПОРЯДОК=1 -UПОРЯДОК -DПОРЯДОК=2 \ + "$base/empty.c" -o "$base/macros" +grep '^#define ПОРЯДОК 2$' "$base/macros" >/dev/null + +# '$' is accepted after the first identifier character. Single quotes protect +# it from shell parameter expansion; without quotes a backslash can do the same. +"$MCPU_CPP" --no-config -dD '-DАНДРЕЙ$_Y=62' -DОЛЬГА\$OLD=37 \ + "$base/empty.c" -o "$base/dollar-command-line" +grep '^#define АНДРЕЙ$_Y 62$' "$base/dollar-command-line" >/dev/null +grep '^#define ОЛЬГА$OLD 37$' "$base/dollar-command-line" >/dev/null + +# '$' is not an identifier-start character. +if "$MCPU_CPP" --no-config '-D$BAD=1' "$base/empty.c" -o "$base/bad" \ + >"$base/dollar-start.stdout" 2>"$base/dollar-start.stderr"; then + exit 1 +fi +grep 'macro name expected after #define' "$base/dollar-start.stderr" >/dev/null + +# For -D/-U an invalid tail after a valid identifier is discarded rather than +# becoming part of the replacement list. The value after '=' is preserved. +"$MCPU_CPP" --no-config -dD '-DАНДРЕЙ@XYZ=62' '-DABC+TAIL=17' \ + '-UREMOVE@TAIL' -DREMOVE=99 "$base/empty.c" -o "$base/invalid-tail" +grep '^#define АНДРЕЙ 62$' "$base/invalid-tail" >/dev/null +grep '^#define ABC 17$' "$base/invalid-tail" >/dev/null +grep '^#define REMOVE 99$' "$base/invalid-tail" >/dev/null +if grep '^#define АНДРЕЙ @XYZ 62$' "$base/invalid-tail" >/dev/null; then + exit 1 +fi + +# Explicit '=' in an -U tail is just another invalid suffix and is discarded. +"$MCPU_CPP" --no-config -DNAME=1 '-UNAME=bad' -dM "$base/empty.c" -o "$base/undef-tail" +if grep '^#define NAME ' "$base/undef-tail" >/dev/null; then + exit 1 +fi + +# In -dD output command-line definitions have their own GNU-style origin +# marker and must not be reported as predefined/builtin macros. +"$MCPU_CPP" --no-config -dD -DАНДРЕЙ_К=62 -DОЛЬГА=37 \ + "$base/empty.c" -o "$base/dD-command-line" +awk ' + /^#define АНДРЕЙ_К 62$/ { + if( previous != "# 0 \"<command-line>\"") exit 1 + andrey = 1 + } + /^#define ОЛЬГА 37$/ { + if( previous != "# 0 \"<command-line>\"") exit 1 + olga = 1 + } + /^#define __MCPU_CPP_VERSION__ / { + if( previous != "# 0 \"<built-in>\"") exit 1 + builtin = 1 + } + { previous = $0 } + END { if( !andrey || !olga || !builtin ) exit 1 } +' "$base/dD-command-line" + +# An unterminated block comment in a command-line definition is diagnosed. +if "$MCPU_CPP" --no-config '-DBROKEN=left/*' "$base/empty.c" -o "$base/bad" \ + >"$base/comment-bad.stdout" 2>"$base/comment-bad.stderr"; then + exit 1 +fi +grep -- '-D: unterminated comment' "$base/comment-bad.stderr" >/dev/null + +# Malformed macro names are rejected by the ordinary #define/#undef parsers. +if "$MCPU_CPP" --no-config -D1ABC=2 "$base/empty.c" -o "$base/bad" \ + >"$base/bad.stdout" 2>"$base/bad.stderr"; then + exit 1 +fi +grep 'macro name expected after #define' "$base/bad.stderr" >/dev/null + + +# Only -D/-U payloads are decoded. Invalid UTF-8 and valid supplementary-plane +# UTF-8 are rejected because the internal text model is strict UCS-2. +bad_utf8=$(printf '\377') +if "$MCPU_CPP" --no-config "-D${bad_utf8}=1" "$base/empty.c" -o "$base/bad" \ + >"$base/utf8.stdout" 2>"$base/utf8.stderr"; then + exit 1 +fi +grep -- '-D: invalid UTF-8 or non-UCS-2 character' "$base/utf8.stderr" >/dev/null + +non_ucs2=$(printf '\360\237\230\200') +if "$MCPU_CPP" --no-config "-D${non_ucs2}=1" "$base/empty.c" -o "$base/bad" \ + >"$base/nonucs2.stdout" 2>"$base/nonucs2.stderr"; then + exit 1 +fi +grep -- '-D: invalid UTF-8 or non-UCS-2 character' "$base/nonucs2.stderr" >/dev/null + +if "$MCPU_CPP" --no-config "-U${bad_utf8}" "$base/empty.c" -o "$base/bad" \ + >"$base/utf8u.stdout" 2>"$base/utf8u.stderr"; then + exit 1 +fi +grep -- '-U: invalid UTF-8 or non-UCS-2 character' "$base/utf8u.stderr" >/dev/null diff --git a/tests/t0040-zubr-expression.sh b/tests/t0040-zubr-expression.sh new file mode 100755 index 0000000..d95c629 --- /dev/null +++ b/tests/t0040-zubr-expression.sh @@ -0,0 +1,79 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-zubr-expression-$$" +mkdir -p "$base" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +cat > "$base/main.c" <<'EOT' +#if 0b1010 == 10 && 077 == 63 && 0x2a == 42 +radix_ok +#else +bad_radix +#endif + +#if 9223372036854775807 == 0x7fffffffffffffff +int64_ok +#else +bad_int64 +#endif + +#if 0xffffffffffffffffU > 9223372036854775807 +uint64_ok +#else +bad_uint64 +#endif + +#if 'A' == 65 && 'Я' == 0x42f && '\x42f' == 0x42f && '\101' == 65 +ucs2_char_ok +#else +bad_ucs2_char +#endif + +#if -1 < 0 && (8 << -1) == 4 && (4 >> -1) == 8 +signed_shift_ok +#else +bad_signed_shift +#endif + +#if 0 && (1 / 0) +bad_short_and +#endif + +#if 1 || (1 / 0) +short_or_ok +#endif + +#if 1 ? 7 : 1 / 0 +ternary_ok +#endif +EOT + +"$MCPU_CPP" --no-config "$base/main.c" -o "$base/out" + +for word in \ + radix_ok int64_ok uint64_ok ucs2_char_ok signed_shift_ok short_or_ok ternary_ok; do + test "`grep -c "^${word}$" "$base/out"`" -eq 1 +done + +for word in \ + bad_radix bad_int64 bad_uint64 bad_ucs2_char bad_signed_shift bad_short_and; do + if grep "^${word}$" "$base/out" >/dev/null; then exit 1; fi +done + +cat > "$base/bad-float.c" <<'EOT' +#if 1.0 +bad +#endif +EOT +if "$MCPU_CPP" --no-config "$base/bad-float.c" -o "$base/bad-out" 2>/dev/null; then + exit 1 +fi + +cat > "$base/bad-inc.c" <<'EOT' +#if 1++2 +bad +#endif +EOT +if "$MCPU_CPP" --no-config "$base/bad-inc.c" -o "$base/bad-out" 2>/dev/null; then + exit 1 +fi diff --git a/tests/t0041-integer-width-suffix.sh b/tests/t0041-integer-width-suffix.sh new file mode 100755 index 0000000..a59483f --- /dev/null +++ b/tests/t0041-integer-width-suffix.sh @@ -0,0 +1,95 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-integer-width-suffix-$$" +mkdir -p "$base" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +cat > "$base/main.c" <<'EOT' +#if 0x7fz8 == 127 +z8_positive +#endif +#if 0x80z8 == -128 +z8_sign_extend +#endif +#if 0xffz8 == -1 +z8_minus_one +#endif +#if 0xffz8u == 255 +z8_zero_extend +#endif +#if 0x1ffz8 == -1 && 0x1ffz8U == 255 +z8_truncate +#endif +#if 0x8000z16 == -32768 && 0xffffz16U == 65535 +z16_ok +#endif +#if 0x80000000z32 < 0 && 0xffffffffz32u == 4294967295U +z32_ok +#endif +#if 0xffffffffffffffffz64 == -1 +z64_signed +#endif +#if 0xffffffffffffffffZ064U > 0 +z64_unsigned_leading_zero_width +#endif +#if 0x7fz8 + 1 == 128 +operations_are_64_bit +#endif +#if (~0xffz8u) == 0xffffffffffffff00U +compl_is_64_bit +#endif +#if (0x80z8 >> 1) == -64 && (0x80z8u >> 1) == 64 +shift_uses_promoted_value +#endif +#if 1z24 == 1 +invalid_width_ignored +#endif +#if 1z024U == 1U +invalid_width_u_preserved +#endif +#if 1U == 1 && 1u == 1 +plain_u_ok +#endif +EOT + +"$MCPU_CPP" --no-config "$base/main.c" -o "$base/out" 2>"$base/err" + +for word in \ + z8_positive z8_sign_extend z8_minus_one z8_zero_extend z8_truncate \ + z16_ok z32_ok z64_signed z64_unsigned_leading_zero_width \ + operations_are_64_bit compl_is_64_bit shift_uses_promoted_value \ + invalid_width_ignored invalid_width_u_preserved plain_u_ok; do + test "`grep -c "^${word}$" "$base/out"`" -eq 1 +done + +test "`grep -c 'warning: invalid zNNN integer-width suffix; suffix ignored' "$base/err"`" -eq 2 + +for bad in \ + '1z128' \ + '1z256u' \ + '1z32undefined' \ + '1z32ufoo' \ + '1z32$foo' \ + '1L' \ + '1LL' \ + '18446744073709551616'; do + cat > "$base/bad.c" <<EOT +#if $bad +bad +#endif +EOT + if "$MCPU_CPP" --no-config "$base/bad.c" -o "$base/bad-out" 2>"$base/bad-err"; then + echo "accepted invalid integer constant: $bad" >&2 + exit 1 + fi +done + +cat > "$base/wide.c" <<'EOT' +#if 1z128 +bad +#endif +EOT +if "$MCPU_CPP" --no-config "$base/wide.c" -o "$base/bad-out" 2>"$base/wide.err"; then + exit 1 +fi +grep 'error: integer constants wider than 64 bits are not allowed in conditional directives' "$base/wide.err" >/dev/null diff --git a/tests/t0042-diagnostics.sh b/tests/t0042-diagnostics.sh new file mode 100755 index 0000000..9510332 --- /dev/null +++ b/tests/t0042-diagnostics.sh @@ -0,0 +1,68 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-diagnostics-$$" +mkdir -p "$base" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +cat > "$base/warning.c" <<'EOT' +#define MESSAGE expanded_message +before +#warning MESSAGE is not expanded +#warning one /* removed comment */ two +#warning "spaces inside string" +#warning Привет мир +#if 0 +#warning hidden warning +#error hidden error +#endif +after +EOT + +"$MCPU_CPP" --no-config "$base/warning.c" -o "$base/warning.out" 2>"$base/warning.err" + +grep '^before$' "$base/warning.out" >/dev/null +grep '^after$' "$base/warning.out" >/dev/null +! grep '#warning' "$base/warning.out" >/dev/null + +grep 'warning: #warning MESSAGE is not expanded$' "$base/warning.err" >/dev/null +grep 'warning: #warning one two$' "$base/warning.err" >/dev/null +grep 'warning: #warning "spaces inside string"$' "$base/warning.err" >/dev/null +grep 'warning: #warning Привет мир$' "$base/warning.err" >/dev/null +! grep 'expanded_message' "$base/warning.err" >/dev/null +! grep 'hidden warning' "$base/warning.err" >/dev/null +! grep 'hidden error' "$base/warning.err" >/dev/null + +"$MCPU_CPP" --no-config -dD "$base/warning.c" -o "$base/warning-dd.out" 2>"$base/warning-dd.err" +grep '^#define MESSAGE expanded_message$' "$base/warning-dd.out" >/dev/null +! grep '#warning' "$base/warning-dd.out" >/dev/null +grep 'warning: #warning MESSAGE is not expanded$' "$base/warning-dd.err" >/dev/null + +cat > "$base/line.c" <<'EOT' +#line 62 "generated.mc" +#warning logical location +ok +EOT +"$MCPU_CPP" --no-config "$base/line.c" -o "$base/line.out" 2>"$base/line.err" +grep '^generated\.mc:62: warning: #warning logical location$' "$base/line.err" >/dev/null +grep '^ok$' "$base/line.out" >/dev/null + +cat > "$base/error.c" <<'EOT' +#define WHY replacement +#error WHY failed +never +EOT +if "$MCPU_CPP" --no-config "$base/error.c" -o "$base/error.out" 2>"$base/error.err"; then + echo '#error did not stop preprocessing' >&2 + exit 1 +fi +grep 'error: #error WHY failed$' "$base/error.err" >/dev/null +! grep 'replacement' "$base/error.err" >/dev/null + + +cat > "$base/empty.c" <<'EOT' +#warning +ok +EOT +"$MCPU_CPP" --no-config "$base/empty.c" -o "$base/empty.out" 2>"$base/empty.err" +grep 'warning: #warning$' "$base/empty.err" >/dev/null +grep '^ok$' "$base/empty.out" >/dev/null diff --git a/tests/t0043-include-next.sh b/tests/t0043-include-next.sh new file mode 100755 index 0000000..a2d86c3 --- /dev/null +++ b/tests/t0043-include-next.sh @@ -0,0 +1,87 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-include-next-$$" +mkdir -p "$base/wrapper" "$base/system" "$base/after" "$base/local" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +cat > "$base/config" <<EOT +MCPU_CPP_SYSTEM_INCLUDE_PATH = $base/system; +EOT + +cat > "$base/main.c" <<'EOT' +#include <wrapped.h> +#include "local/local.h" +#if 0 +#include_next <never.h> +#endif +EOT + +cat > "$base/wrapper/wrapped.h" <<'EOT' +wrapper_begin +#line 500 "virtual/wrapped.h" +#define NEXT_WRAPPED <wrapped.h> +#include_next NEXT_WRAPPED +wrapper_end +EOT + +cat > "$base/system/wrapped.h" <<'EOT' +system_begin +#include_next "wrapped.h" +system_end +EOT + +cat > "$base/after/wrapped.h" <<'EOT' +after_header +EOT + +cat > "$base/local/local.h" <<'EOT' +local_header +#include_next <from_chain.h> +EOT + +cat > "$base/wrapper/from_chain.h" <<'EOT' +from_chain_wrapper +EOT + +"$MCPU_CPP" --config-file "$base/config" \ + -isystem "$base/wrapper" -idirafter "$base/after" \ + "$base/main.c" -o "$base/out" + +# The wrapper header found through -isystem continues into the configured +# system include root, then that header continues into -idirafter. The quote +# form of #include_next must not re-search the current source directory. +awk ' + /wrapper_begin/ { a = NR } + /system_begin/ { b = NR } + /after_header/ { c = NR } + /system_end/ { d = NR } + /wrapper_end/ { e = NR } + END { exit !(a && a < b && b < c && c < d && d < e) } +' "$base/out" + +# A header found by ordinary quoted-source-directory lookup has no configured +# search entry to skip, so include_next begins with the configured chain. +grep '^local_header$' "$base/out" >/dev/null +grep '^from_chain_wrapper$' "$base/out" >/dev/null + +# Macro-expanded include_next operands use the same include operand parser. +grep '^system_begin$' "$base/out" >/dev/null + +# Inactive conditional branches must not execute include_next. +! grep 'never.h' "$base/out" >/dev/null + +# If no later search entry contains the file, include_next is an error rather +# than falling back to the directory that supplied the current header. +cat > "$base/wrapper/last.h" <<'EOT' +#include_next <last.h> +EOT +cat > "$base/last-main.c" <<'EOT' +#include <last.h> +EOT +if "$MCPU_CPP" --config-file "$base/config" \ + -isystem "$base/wrapper" "$base/last-main.c" \ + -o "$base/last.out" 2>"$base/last.err"; then + echo '#include_next unexpectedly found the current header again' >&2 + exit 1 +fi +grep "cannot find include_next file 'last.h'" "$base/last.err" >/dev/null diff --git a/tests/t0044-configured-search-order.sh b/tests/t0044-configured-search-order.sh new file mode 100755 index 0000000..980790f --- /dev/null +++ b/tests/t0044-configured-search-order.sh @@ -0,0 +1,114 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-config-order-$$" +mkdir -p \ + "$base/cmd-I" \ + "$base/cmd-isystem" \ + "$base/config-as" \ + "$base/config-user" \ + "$base/system/as" \ + "$base/cmd-after" \ + "$base/config-after" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +cat > "$base/config" <<EOT +MCPU_CPP_AS_INCLUDE_PATH = $base/config-as; +MCPU_CPP_INCLUDE_PATH = $base/config-user; +MCPU_CPP_SYSTEM_INCLUDE_PATH = $base/system; +MCPU_CPP_AFTER_INCLUDE_PATH = $base/config-after; +EOT + +cat > "$base/main.c" <<'EOT' +#lang "as" +#include <same.h> +#endlang +EOT + +cat > "$base/cmd-I/same.h" <<'EOT' +from_explicit_I +#include_next <same.h> +EOT +cat > "$base/cmd-isystem/same.h" <<'EOT' +from_explicit_isystem +#include_next <same.h> +EOT +cat > "$base/config-as/same.h" <<'EOT' +from_config_language +#include_next <same.h> +EOT +cat > "$base/config-user/same.h" <<'EOT' +from_config_user +#include_next <same.h> +EOT +cat > "$base/system/as/same.h" <<'EOT' +from_system_language +#include_next <same.h> +EOT +cat > "$base/system/same.h" <<'EOT' +from_system_root +#include_next <same.h> +EOT +cat > "$base/cmd-after/same.h" <<'EOT' +from_explicit_idirafter +#include_next <same.h> +EOT +cat > "$base/config-after/same.h" <<'EOT' +from_config_after +EOT + +# Deliberately scramble command-line option order. Semantic classes, not +# argv order, define precedence. +"$MCPU_CPP" --config-file "$base/config" \ + -idirafter "$base/cmd-after" \ + -isystem "$base/cmd-isystem" \ + -I "$base/cmd-I" \ + "$base/main.c" -o "$base/out" + +awk ' + /from_explicit_I/ { a = NR } + /from_explicit_isystem/ { b = NR } + /from_config_language/ { c = NR } + /from_config_user/ { d = NR } + /from_system_language/ { e = NR } + /from_system_root/ { f = NR } + /from_explicit_idirafter/ { g = NR } + /from_config_after/ { h = NR } + END { exit !(a && a < b && b < c && c < d && d < e && e < f && f < g && g < h) } +' "$base/out" + +# An empty MCPU_CPP_SYSTEM_INCLUDE_PATH disables the complete configured +# system tree, including the automatically derived <lang> subdirectory. +cat > "$base/no-system.conf" <<EOT +MCPU_CPP_SYSTEM_INCLUDE_PATH = ; +EOT +cat > "$base/system-only.c" <<'EOT' +#lang "as" +#include <system-only.h> +#endlang +EOT +cat > "$base/system/as/system-only.h" <<'EOT' +system_only_header +EOT +if "$MCPU_CPP" --config-file "$base/no-system.conf" \ + "$base/system-only.c" -o "$base/no-system.out" 2>"$base/no-system.err"; then + echo 'empty MCPU_CPP_SYSTEM_INCLUDE_PATH did not disable system includes' >&2 + exit 1 +fi + +# -nostdinc suppresses the configured system tree for one invocation but does +# not suppress explicit -isystem or either after class. +cat > "$base/nostdinc.c" <<'EOT' +#lang "as" +#include <nostdinc.h> +#endlang +EOT +cat > "$base/system/as/nostdinc.h" <<'EOT' +wrong_configured_system +EOT +cat > "$base/cmd-after/nostdinc.h" <<'EOT' +from_after_with_nostdinc +EOT +"$MCPU_CPP" --config-file "$base/config" -nostdinc \ + -idirafter "$base/cmd-after" "$base/nostdinc.c" -o "$base/nostdinc.out" +grep '^from_after_with_nostdinc$' "$base/nostdinc.out" >/dev/null +! grep '^wrong_configured_system$' "$base/nostdinc.out" >/dev/null diff --git a/tests/t0045-search-dirs-verbose.sh b/tests/t0045-search-dirs-verbose.sh new file mode 100755 index 0000000..03a6393 --- /dev/null +++ b/tests/t0045-search-dirs-verbose.sh @@ -0,0 +1,89 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-search-dirs-$$" +mkdir -p \ + "$base/cmd-user" \ + "$base/cmd-system" \ + "$base/config-as" \ + "$base/config-user" \ + "$base/system" \ + "$base/cmd-after" \ + "$base/config-after" +trap 'rm -rf "$base"' EXIT HUP INT TERM +runtime_root=$(cd "$(dirname "$MCPU_CPP")/.." && pwd -P) + +cat > "$base/test.conf" <<EOT +MCPU_CPP_INCLUDE_PATH = $base/old-user; +MCPU_CPP_AS_INCLUDE_PATH = $base/config-as; +MCPU_CPP_INCLUDE_PATH = $base/config-user; +MCPU_CPP_SYSTEM_INCLUDE_PATH = $base/system; +MCPU_CPP_AFTER_INCLUDE_PATH = $base/config-after; +EOT + +"$MCPU_CPP" \ + --config-file "$base/test.conf" \ + -I "$base/cmd-user" \ + -isystem "$base/cmd-system" \ + -idirafter "$base/cmd-after" \ + -v -dsearch-dirs \ + > "$base/search.out" 2> "$base/verbose.out" + +cat > "$base/search.expected" <<EOT +search: $base/cmd-user +search: $base/cmd-system +search: $base/config-as +search: $base/config-user +search: $base/system/diff +search: $base/system/dift +search: $base/system/alg +search: $base/system/as +search: $base/system/avm +search: $base/system/acs +search: $base/system +search: $base/cmd-after +search: $base/config-after +EOT +cmp "$base/search.expected" "$base/search.out" + +# -v prints one effective value, not every assignment encountered while +# reading configuration files. +test "$(grep -c '^config: MCPU_CPP_INCLUDE_PATH=' "$base/verbose.out")" -eq 1 +grep "^config: MCPU_CPP_INCLUDE_PATH=$base/config-user$" "$base/verbose.out" >/dev/null +if grep "$base/old-user" "$base/verbose.out" >/dev/null; then + echo 'verbose output contains an overridden configuration value' >&2 + exit 1 +fi +grep "^config: MCPU_CPP_AS_INCLUDE_PATH=$base/config-as$" "$base/verbose.out" >/dev/null +grep "^config: MCPU_CPP_SYSTEM_INCLUDE_PATH=$base/system$" "$base/verbose.out" >/dev/null +grep "^config: MCPU_CPP_AFTER_INCLUDE_PATH=$base/config-after$" "$base/verbose.out" >/dev/null + +# Even when an explicitly selected configuration file is empty, -v and +# -dsearch-dirs must expose the runtime-derived installation system root. +: > "$base/empty.conf" +"$MCPU_CPP" --config-file "$base/empty.conf" -v -dsearch-dirs \ + > "$base/default-search.out" 2> "$base/default-verbose.out" +grep '^config: MCPU_CPP_SYSTEM_INCLUDE_PATH=' "$base/default-verbose.out" >/dev/null +grep "^search: $runtime_root/include$" "$base/default-search.out" >/dev/null + +# -nostdinc removes both the generated <lang> system directories and the +# configured system root from the directory dump, but keeps explicit -isystem. +"$MCPU_CPP" \ + --config-file "$base/test.conf" \ + -isystem "$base/cmd-system" \ + -nostdinc -dsearch-dirs > "$base/nostdinc.out" +grep "^search: $base/cmd-system$" "$base/nostdinc.out" >/dev/null +if grep "^search: $base/system" "$base/nostdinc.out" >/dev/null; then + echo '-nostdinc left configured system directories in -dsearch-dirs' >&2 + exit 1 +fi + +# A dump action stops preprocessing. An input name may be present but need +# not exist. +"$MCPU_CPP" --config-file "$base/test.conf" -dsearch-dirs \ + "$base/does-not-exist.c" > /dev/null + +if "$MCPU_CPP" --config-file "$base/test.conf" -dsearch-dirs -o "$base/out" \ + > /dev/null 2>&1; then + echo '-dsearch-dirs accepted an output file' >&2 + exit 1 +fi diff --git a/tests/t0046-pragma-once.sh b/tests/t0046-pragma-once.sh new file mode 100755 index 0000000..9429dd7 --- /dev/null +++ b/tests/t0046-pragma-once.sh @@ -0,0 +1,67 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-pragma-once-$$" +mkdir -p "$base" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +cat > "$base/once.h" <<'EOT' +#pragma once +ONCE_BODY +EOT +ln -s once.h "$base/once-link.h" +ln "$base/once.h" "$base/once-hard.h" + +cat > "$base/self.h" <<'EOT' +#pragma once +SELF_BEGIN +#include "self.h" +SELF_END +EOT + +cat > "$base/inactive.h" <<'EOT' +#if 0 +#pragma once +#endif +INACTIVE_BODY +EOT + +cat > "$base/main.c" <<'EOT' +#include "once.h" +#include "once.h" +#include "once-link.h" +#include "once-hard.h" +#include "self.h" +#include "inactive.h" +#include "inactive.h" +#pragma pack(push, 1) +EOT + +"$MCPU_CPP" --no-config "$base/main.c" -o "$base/out" + +# The same physical file is processed only once, even through a symbolic link +# and a hard link. #pragma once itself is consumed by the preprocessor. +test "$(grep -c '^ONCE_BODY$' "$base/out")" -eq 1 +! grep '^#pragma once$' "$base/out" >/dev/null + +# Marking happens as soon as the active directive is seen, so a file may safely +# include itself after #pragma once without recursive inclusion. +test "$(grep -c '^SELF_BEGIN$' "$base/out")" -eq 1 +test "$(grep -c '^SELF_END$' "$base/out")" -eq 1 + +# An inactive #pragma once has no effect. +test "$(grep -c '^INACTIVE_BODY$' "$base/out")" -eq 2 + +# Other pragmas are not consumed; they remain available to the later compiler. +grep '^#pragma pack(push, 1)$' "$base/out" >/dev/null + +# -dD does not make #pragma once reappear; unrelated pragmas are still kept. +"$MCPU_CPP" --no-config -dD "$base/main.c" -o "$base/dd.out" +! grep '^#pragma once$' "$base/dd.out" >/dev/null +grep '^#pragma pack(push, 1)$' "$base/dd.out" >/dev/null + +# A non-file input stream has no physical identity. The directive is simply +# consumed and must not turn stdin preprocessing into an error. +printf '#pragma once\nSTDIN_BODY\n' | \ + "$MCPU_CPP" --no-config > "$base/stdin.out" +grep '^STDIN_BODY$' "$base/stdin.out" >/dev/null +! grep '^#pragma once$' "$base/stdin.out" >/dev/null diff --git a/tests/t0047-dependencies.sh b/tests/t0047-dependencies.sh new file mode 100755 index 0000000..87ecb63 --- /dev/null +++ b/tests/t0047-dependencies.sh @@ -0,0 +1,140 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-deps-$$" +mkdir -p "$base/user" "$base/sys" "$base/cfg-user" "$base/cfg-sys" "$base/after" "$base/cfg-after" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +cat > "$base/config.conf" <<EOT +MCPU_CPP_INCLUDE_PATH = $base/cfg-user; +MCPU_CPP_SYSTEM_INCLUDE_PATH = $base/cfg-sys; +MCPU_CPP_AFTER_INCLUDE_PATH = $base/cfg-after; +EOT + +cat > "$base/main.c" <<'EOT' +#include "local.h" +#include "local-alias.h" +#include <user.h> +#include <cfg-user.h> +#include "sys-explicit.h" +#include <sys-config.h> +#include <after-explicit.h> +#include <after-config.h> +#include <same.h> +#include <shared.h> +SHOULD_NOT_APPEAR_IN_DEP_OUTPUT +EOT + +cat > "$base/local.h" <<'EOT' +#pragma once +#line 100 "logical-local.h" +LOCAL_BODY +EOT +ln "$base/local.h" "$base/local-alias.h" + +cat > "$base/user/user.h" <<'EOT' +USER_HEADER +EOT +cat > "$base/cfg-user/cfg-user.h" <<'EOT' +CONFIG_USER_HEADER +EOT +cat > "$base/user/shared.h" <<'EOT' +SHARED_USER_HEADER +EOT +cat > "$base/user/transitive-only.h" <<'EOT' +TRANSITIVE_USER_HEADER +EOT + +cat > "$base/sys/sys-explicit.h" <<'EOT' +#include <transitive-only.h> +#include <shared.h> +#include "sys-child.h" +EOT +cat > "$base/sys/sys-child.h" <<'EOT' +SYSTEM_CHILD +EOT +cat > "$base/cfg-sys/sys-config.h" <<'EOT' +SYSTEM_CONFIG +EOT +cat > "$base/after/after-explicit.h" <<'EOT' +AFTER_EXPLICIT +EOT +cat > "$base/cfg-after/after-config.h" <<'EOT' +AFTER_CONFIG +EOT + +cat > "$base/user/same.h" <<'EOT' +USER_WRAPPER +#include_next <same.h> +EOT +cat > "$base/sys/same.h" <<'EOT' +SYSTEM_WRAPPED +EOT + +"$MCPU_CPP" --config-file "$base/config.conf" \ + -I "$base/user" -isystem "$base/sys" -idirafter "$base/after" \ + -M "$base/main.c" > "$base/M.out" + +# -M emits only a make rule and includes every physical dependency. +grep "^main\\.o:" "$base/M.out" >/dev/null +! grep 'SHOULD_NOT_APPEAR_IN_DEP_OUTPUT' "$base/M.out" >/dev/null +for f in \ + "$base/main.c" "$base/local.h" "$base/user/user.h" \ + "$base/cfg-user/cfg-user.h" "$base/sys/sys-explicit.h" \ + "$base/user/transitive-only.h" "$base/user/shared.h" \ + "$base/sys/sys-child.h" "$base/cfg-sys/sys-config.h" \ + "$base/after/after-explicit.h" "$base/cfg-after/after-config.h" \ + "$base/user/same.h" "$base/sys/same.h" +do + grep -F "$f" "$base/M.out" >/dev/null || { + echo "-M omitted dependency: $f" >&2 + exit 1 + } +done + +# local-alias.h is a hard link to local.h and must not create a duplicate +# physical dependency. The logical #line name is not a dependency either. +! grep -F "$base/local-alias.h" "$base/M.out" >/dev/null +! grep -F 'logical-local.h' "$base/M.out" >/dev/null + +"$MCPU_CPP" --config-file "$base/config.conf" \ + -I "$base/user" -isystem "$base/sys" -idirafter "$base/after" \ + -MM "$base/main.c" > "$base/MM.out" + +grep "^main\\.o:" "$base/MM.out" >/dev/null +for f in "$base/main.c" "$base/local.h" "$base/user/user.h" \ + "$base/cfg-user/cfg-user.h" "$base/user/same.h" "$base/user/shared.h" +do + grep -F "$f" "$base/MM.out" >/dev/null || { + echo "-MM omitted user dependency: $f" >&2 + exit 1 + } +done + +# Quoting style does not define system-ness. sys-explicit.h was included +# with quotes but found in -isystem, and all descendants reachable only from +# that system header are excluded. -idirafter/config AFTER are system class. +for f in "$base/sys/sys-explicit.h" "$base/user/transitive-only.h" \ + "$base/sys/sys-child.h" "$base/cfg-sys/sys-config.h" \ + "$base/after/after-explicit.h" "$base/cfg-after/after-config.h" \ + "$base/sys/same.h" +do + if grep -F "$f" "$base/MM.out" >/dev/null; then + echo "-MM kept system dependency: $f" >&2 + exit 1 + fi +done + +# The same physical shared.h is first reached transitively from a system +# header, then directly from main. The direct user inclusion makes it a user +# dependency, so -MM must retain it (checked above). + +# Normal mcpu-cpp output-file syntax remains usable for dependency output. +"$MCPU_CPP" --config-file "$base/config.conf" -I "$base/user" \ + -isystem "$base/sys" -idirafter "$base/after" \ + -MM "$base/main.c" -o "$base/deps.mk" +test -s "$base/deps.mk" +grep '^main\.o:' "$base/deps.mk" >/dev/null + +# stdin follows GNU cpp's useful default target convention. +printf 'int x;\n' | "$MCPU_CPP" --no-config -M > "$base/stdin.dep" +grep '^-: -$' "$base/stdin.dep" >/dev/null diff --git a/tests/t0048-no-config-defaults.sh b/tests/t0048-no-config-defaults.sh new file mode 100755 index 0000000..a6d2c75 --- /dev/null +++ b/tests/t0048-no-config-defaults.sh @@ -0,0 +1,73 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-no-config-defaults-$$" +mkdir -p "$base/home/.mcpu" "$base/I" "$base/isystem" "$base/after" +trap 'rm -rf "$base"' EXIT HUP INT TERM +runtime_root=$(cd "$(dirname "$MCPU_CPP")/.." && pwd -P) + +cat > "$base/home/.mcpu/mcpu-cpp.conf" <<EOT +MCPU_CPP_INCLUDE_PATH = $base/forbidden-user; +MCPU_CPP_SYSTEM_INCLUDE_PATH = $base/forbidden-system; +MCPU_CPP_AFTER_INCLUDE_PATH = $base/forbidden-after; +EOT + +# --no-config skips every configuration file, but the runtime-derived installation +# default remains the effective system root. +HOME="$base/home" "$MCPU_CPP" --no-config -dsearch-dirs > "$base/search.out" +grep "^search: $runtime_root/include/diff$" "$base/search.out" >/dev/null +grep "^search: $runtime_root/include/dift$" "$base/search.out" >/dev/null +grep "^search: $runtime_root/include/alg$" "$base/search.out" >/dev/null +grep "^search: $runtime_root/include/as$" "$base/search.out" >/dev/null +grep "^search: $runtime_root/include/avm$" "$base/search.out" >/dev/null +grep "^search: $runtime_root/include/acs$" "$base/search.out" >/dev/null +grep "^search: $runtime_root/include$" "$base/search.out" >/dev/null +if grep "$base/forbidden" "$base/search.out" >/dev/null; then + echo '--no-config read the user configuration file' >&2 + exit 1 +fi + +# The same runtime-derived value is visible through -dconfig and -v. +HOME="$base/home" "$MCPU_CPP" --no-config -dconfig > "$base/config.out" +grep "^MCPU_CPP_SYSTEM_INCLUDE_PATH = $runtime_root/include;$" "$base/config.out" >/dev/null +if grep "$base/forbidden" "$base/config.out" >/dev/null; then + echo '--no-config leaked user configuration into -dconfig' >&2 + exit 1 +fi + +HOME="$base/home" "$MCPU_CPP" --no-config -v -dsearch-dirs \ + > "$base/verbose-search.out" 2> "$base/verbose.out" +grep "^config: MCPU_CPP_SYSTEM_INCLUDE_PATH=$runtime_root/include$" "$base/verbose.out" >/dev/null +if grep "$base/forbidden" "$base/verbose.out" >/dev/null; then + echo '--no-config leaked user configuration into -v' >&2 + exit 1 +fi + +# Explicit command-line classes retain their normative priority around the +# runtime-derived system root. +HOME="$base/home" "$MCPU_CPP" --no-config \ + -I "$base/I" -isystem "$base/isystem" -idirafter "$base/after" \ + -dsearch-dirs > "$base/ordered.out" +first=$(sed -n '1p' "$base/ordered.out") +second=$(sed -n '2p' "$base/ordered.out") +last=$(tail -n 1 "$base/ordered.out") +test "$first" = "search: $base/I" +test "$second" = "search: $base/isystem" +test "$last" = "search: $base/after" +grep "^search: $runtime_root/include$" "$base/ordered.out" >/dev/null + +# -nostdinc, not --no-config, is the switch that suppresses the runtime-derived +# standard-system include tree. Explicit classes remain visible. +HOME="$base/home" "$MCPU_CPP" --no-config -nostdinc \ + -I "$base/I" -isystem "$base/isystem" -idirafter "$base/after" \ + -dsearch-dirs > "$base/nostdinc.out" +cat > "$base/nostdinc.expected" <<EOT +search: $base/I +search: $base/isystem +search: $base/after +EOT +cmp "$base/nostdinc.expected" "$base/nostdinc.out" + +# With no explicit directories, --no-config -nostdinc has no search entries. +HOME="$base/home" "$MCPU_CPP" --no-config -nostdinc -dsearch-dirs \ + > "$base/empty.out" +test ! -s "$base/empty.out" diff --git a/tests/t0049-relocatable-root.sh b/tests/t0049-relocatable-root.sh new file mode 100755 index 0000000..d64a9b8 --- /dev/null +++ b/tests/t0049-relocatable-root.sh @@ -0,0 +1,55 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-relocatable-root-$$" +root_a="$base/mcpu-a" +root_b="$base/mcpu-b" +mkdir -p "$root_a/bin" "$root_a/etc" "$root_a/include/diff" "$base/home" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +cp "$MCPU_CPP" "$root_a/bin/mcpu-cpp" +chmod +x "$root_a/bin/mcpu-cpp" + +cat > "$root_a/include/diff/relocated.h" <<'EOT' +int relocated_header = 37; +EOT + +cat > "$root_a/etc/mcpu-cpp.conf" <<'EOT' +MCPU_CPP_RELOCATION_SENTINEL = runtime-config; +EOT + +# The copied executable first derives its defaults from its new physical root. +HOME="$base/home" "$root_a/bin/mcpu-cpp" --no-config -dsearch-dirs \ + > "$base/root-a.search" +grep "^search: $root_a/include/diff$" "$base/root-a.search" >/dev/null +grep "^search: $root_a/include$" "$base/root-a.search" >/dev/null + +# Move the complete MCPU tree without reconfiguring or rebuilding anything. +mv "$root_a" "$root_b" + +HOME="$base/home" "$root_b/bin/mcpu-cpp" --no-config -dsearch-dirs \ + > "$base/root-b.search" +grep "^search: $root_b/include/diff$" "$base/root-b.search" >/dev/null +grep "^search: $root_b/include$" "$base/root-b.search" >/dev/null +if grep "$root_a" "$base/root-b.search" >/dev/null; then + echo 'relocated mcpu-cpp retained its previous installation root' >&2 + exit 1 +fi + +cat > "$base/main.c" <<'EOT' +#lang "diff" +#include <relocated.h> +#endlang +EOT +HOME="$base/home" "$root_b/bin/mcpu-cpp" --no-config \ + "$base/main.c" -o "$base/main.out" +grep '^int relocated_header = 37;$' "$base/main.out" >/dev/null + +# The packaged configuration is found relative to the relocated root as well. +HOME="$base/home" "$root_b/bin/mcpu-cpp" -dconfig > "$base/config.out" +grep '^MCPU_CPP_RELOCATION_SENTINEL = runtime-config;$' "$base/config.out" >/dev/null + +# Invoking through a symbolic link must still resolve the physical executable +# and therefore the same relocated MCPU root. +ln -s "$root_b/bin/mcpu-cpp" "$base/mcpu-cpp" +HOME="$base/home" "$base/mcpu-cpp" --no-config -dconfig > "$base/link.out" +grep "^MCPU_CPP_SYSTEM_INCLUDE_PATH = $root_b/include;$" "$base/link.out" >/dev/null diff --git a/tests/t0050-dependency-side-effects.sh b/tests/t0050-dependency-side-effects.sh new file mode 100755 index 0000000..67cbcf6 --- /dev/null +++ b/tests/t0050-dependency-side-effects.sh @@ -0,0 +1,136 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-dep-side-effects-$$" +mkdir -p "$base/src" "$base/user" "$base/sys" "$base/out" "$base/stdin" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +cat > "$base/src/main.c" <<'EOT' +#include "user.h" +#include <system.h> +MAIN_BODY +EOT +cat > "$base/user/user.h" <<'EOT' +USER_BODY +EOT +cat > "$base/sys/system.h" <<'EOT' +#include "system-child.h" +SYSTEM_BODY +EOT +cat > "$base/sys/system-child.h" <<'EOT' +SYSTEM_CHILD_BODY +EOT + +# -MD keeps preprocessing output and writes basename.d in the current +# directory when no -o/-MF is present. The input directory is not copied to +# the default dependency filename. +( + cd "$base" + "$MCPU_CPP" --no-config -I "$base/user" -isystem "$base/sys" \ + -MD "$base/src/main.c" > md.out +) +test -s "$base/md.out" +grep 'MAIN_BODY' "$base/md.out" >/dev/null +test -s "$base/main.d" +grep '^main\.o:' "$base/main.d" >/dev/null +for f in "$base/src/main.c" "$base/user/user.h" \ + "$base/sys/system.h" "$base/sys/system-child.h" +do + grep -F "$f" "$base/main.d" >/dev/null || { + echo "-MD omitted dependency: $f" >&2 + exit 1 + } +done + +# -MMD has the same side-effect behavior but applies the already established +# -MM user-dependency filter. +rm -f "$base/main.d" +( + cd "$base" + "$MCPU_CPP" --no-config -I "$base/user" -isystem "$base/sys" \ + -MMD "$base/src/main.c" > mmd.out +) +test -s "$base/mmd.out" +grep 'MAIN_BODY' "$base/mmd.out" >/dev/null +test -s "$base/main.d" +grep -F "$base/src/main.c" "$base/main.d" >/dev/null +grep -F "$base/user/user.h" "$base/main.d" >/dev/null +! grep -F "$base/sys/system.h" "$base/main.d" >/dev/null +! grep -F "$base/sys/system-child.h" "$base/main.d" >/dev/null + +# With ordinary -o, preprocessing goes to that file and the automatic +# dependency filename is obtained by replacing its suffix with .d. +rm -f "$base/main.d" "$base/out/result.i" "$base/out/result.d" +"$MCPU_CPP" --no-config -I "$base/user" -isystem "$base/sys" \ + -MD "$base/src/main.c" -o "$base/out/result.i" +test -s "$base/out/result.i" +grep 'MAIN_BODY' "$base/out/result.i" >/dev/null +test -s "$base/out/result.d" +grep '^main\.o:' "$base/out/result.d" >/dev/null +test ! -e "$base/main.d" + +# -MF overrides automatic dependency naming while leaving normal preprocessing +# output semantics untouched. +rm -f "$base/custom.d" "$base/main.d" +( + cd "$base" + "$MCPU_CPP" --no-config -I "$base/user" -isystem "$base/sys" \ + -MD -MF custom.d "$base/src/main.c" > mf.out +) +test -s "$base/mf.out" +grep 'MAIN_BODY' "$base/mf.out" >/dev/null +test -s "$base/custom.d" +test ! -e "$base/main.d" + +# Attached -MFfile spelling is accepted as by GNU cpp. +rm -f "$base/attached.d" +( + cd "$base" + "$MCPU_CPP" --no-config -I "$base/user" -isystem "$base/sys" \ + -MMD -MFattached.d "$base/src/main.c" > attached.out +) +test -s "$base/attached.d" +! grep -F "$base/sys/system.h" "$base/attached.d" >/dev/null + +# -MF also redirects the already implemented dependency-only -M/-MM modes. +rm -f "$base/only.d" "$base/only.out" +"$MCPU_CPP" --no-config -I "$base/user" -isystem "$base/sys" \ + -M -MF "$base/only.d" "$base/src/main.c" > "$base/only.out" +test ! -s "$base/only.out" +test -s "$base/only.d" +grep -F "$base/sys/system.h" "$base/only.d" >/dev/null + +rm -f "$base/only-user.d" "$base/only-user.out" +"$MCPU_CPP" --no-config -I "$base/user" -isystem "$base/sys" \ + -MM -MF "$base/only-user.d" "$base/src/main.c" > "$base/only-user.out" +test ! -s "$base/only-user.out" +test -s "$base/only-user.d" +! grep -F "$base/sys/system.h" "$base/only-user.d" >/dev/null + +# -MF - means dependency stdout. Put normal preprocessing in a regular -o +# file so the two streams are easy to verify independently. +rm -f "$base/out/stdout.i" "$base/out/stdout.d" +"$MCPU_CPP" --no-config -I "$base/user" -isystem "$base/sys" \ + -MMD -MF - "$base/src/main.c" -o "$base/out/stdout.i" \ + > "$base/mf-dash.out" +test -s "$base/out/stdout.i" +grep 'MAIN_BODY' "$base/out/stdout.i" >/dev/null +grep '^main\.o:' "$base/mf-dash.out" >/dev/null +test ! -e "$base/out/stdout.d" + +# stdin follows the existing '-' target convention and gets '-.d' when -MD +# needs an automatic dependency filename. +( + cd "$base/stdin" + printf 'STDIN_BODY\n' | "$MCPU_CPP" --no-config -MD > stdout.i +) +test -s "$base/stdin/stdout.i" +grep 'STDIN_BODY' "$base/stdin/stdout.i" >/dev/null +test -s "$base/stdin/-.d" +grep '^-: -$' "$base/stdin/-.d" >/dev/null + +# -MF alone is a command-line error: there is no dependency mode to redirect. +if "$MCPU_CPP" --no-config -MF "$base/orphan.d" "$base/src/main.c" \ + > /dev/null 2>&1; then + echo "-MF without a dependency mode unexpectedly succeeded" >&2 + exit 1 +fi diff --git a/tests/t0051-dump-macro-origin.sh b/tests/t0051-dump-macro-origin.sh new file mode 100755 index 0000000..1a2c504 --- /dev/null +++ b/tests/t0051-dump-macro-origin.sh @@ -0,0 +1,63 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-dump-origin-$$" +mkdir -p "$base" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +cat > "$base/main.c" <<'EOT' +#define Z_USER 62 +#define A_USER 37 +#define ФУНКЦИЯ(x) x + 1 +#undef __MCPU_CPP_VERSION__ +EOT + +# -dM contains only ordinary final-state macros. Source and command-line +# definitions are ordinary; built-ins are not. +"$MCPU_CPP" --no-config -dM -DCMD_USER=91 "$base/main.c" > "$base/dM" +grep -Fx '#define A_USER 37' "$base/dM" >/dev/null +grep -Fx '#define CMD_USER 91' "$base/dM" >/dev/null +grep -Fx '#define Z_USER 62' "$base/dM" >/dev/null +grep -Fx '#define ФУНКЦИЯ(x) x + 1' "$base/dM" >/dev/null +if grep '^#define _ARCH_MCPU ' "$base/dM" >/dev/null; then exit 1; fi +if grep '^#define __MCPU_' "$base/dM" >/dev/null; then exit 1; fi +LC_ALL=C sort -c "$base/dM" + +# Empty input has no ordinary definitions. +"$MCPU_CPP" --no-config -dM < /dev/null > "$base/empty-dM" +test ! -s "$base/empty-dM" + +# -dMP contains predefined macros first and ordinary macros second. Both +# groups are deterministic and the #undef above removes the version builtin. +"$MCPU_CPP" --no-config -dMP -DCMD_USER=91 "$base/main.c" > "$base/dMP" +grep -Fx '#define _ARCH_MCPU 1' "$base/dMP" >/dev/null +grep -Fx '#define A_USER 37' "$base/dMP" >/dev/null +grep -Fx '#define CMD_USER 91' "$base/dMP" >/dev/null +grep -Fx '#define Z_USER 62' "$base/dMP" >/dev/null +grep -Fx '#define ФУНКЦИЯ(x) x + 1' "$base/dMP" >/dev/null +if grep '^#define __MCPU_CPP_VERSION__ ' "$base/dMP" >/dev/null; then exit 1; fi + +predefined_line=`grep -n '^#define _ARCH_MCPU 1$' "$base/dMP" | sed 's/:.*//'` +ordinary_line=`grep -n '^#define A_USER 37$' "$base/dMP" | sed 's/:.*//'` +test "$predefined_line" -lt "$ordinary_line" + +# Verify ordering inside the ordinary tail explicitly. +tail -n +"$ordinary_line" "$base/dMP" > "$base/ordinary-tail" +LC_ALL=C sort -c "$base/ordinary-tail" + +# A former builtin name, if explicitly redefined after #undef, is an ordinary +# macro because origin follows the current definition rather than the spelling. +cat > "$base/redefine.c" <<'EOT' +#undef _ARCH_MCPU +#define _ARCH_MCPU 777 +#define LOCAL 1 +EOT +"$MCPU_CPP" --no-config -dM "$base/redefine.c" > "$base/redefine-dM" +grep -Fx '#define _ARCH_MCPU 777' "$base/redefine-dM" >/dev/null +"$MCPU_CPP" --no-config -dMP "$base/redefine.c" > "$base/redefine-dMP" +grep -Fx '#define _ARCH_MCPU 777' "$base/redefine-dMP" >/dev/null +test "`grep -c '^#define _ARCH_MCPU ' "$base/redefine-dMP"`" -eq 1 + +# Context-dependent special builtins remain outside the static dump. +for name in __FILE__ __LINE__ __DATE__ __TIME__ __BASE_FILE__ __INCLUDE_LEVEL__; do + if grep "^#define ${name}\\>" "$base/dMP" >/dev/null; then exit 1; fi +done diff --git a/tests/t0052-warning-control.sh b/tests/t0052-warning-control.sh new file mode 100755 index 0000000..8214d96 --- /dev/null +++ b/tests/t0052-warning-control.sh @@ -0,0 +1,108 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-warning-control-$$" +mkdir -p "$base" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +cat > "$base/comments.c" <<'EOT' +/* outer /* nested */ +// continued \ +comment +ok +EOT + +# Comment warnings are optional and disabled by default. +"$MCPU_CPP" --no-config "$base/comments.c" -o "$base/default.out" 2>"$base/default.err" +test ! -s "$base/default.err" +grep '^ok$' "$base/default.out" >/dev/null + +# -Wcomment and -Wcomments are synonyms. +for option in -Wcomment -Wcomments -Wall; do + "$MCPU_CPP" --no-config "$option" "$base/comments.c" \ + -o "$base/comment.out" 2>"$base/comment.err" + test "`grep -c 'warning: \"/\*\" within comment' "$base/comment.err"`" -eq 1 + test "`grep -c 'warning: multi-line comment' "$base/comment.err"`" -eq 1 +done + +# A specific -Wno-comment is stronger than the -Wall group regardless of order. +for options in '-Wall -Wno-comment' '-Wno-comment -Wall' \ + '-Wall -Wno-comments' '-Wno-comments -Wall'; do + # shellcheck disable=SC2086 + "$MCPU_CPP" --no-config $options "$base/comments.c" \ + -o "$base/no-comment.out" 2>"$base/no-comment.err" + test ! -s "$base/no-comment.err" +done + +# Same-specificity comment options use the last occurrence. +"$MCPU_CPP" --no-config -Wno-comment -Wcomment "$base/comments.c" \ + -o "$base/comment-last.out" 2>"$base/comment-last.err" +grep 'warning: multi-line comment' "$base/comment-last.err" >/dev/null + +"$MCPU_CPP" --no-config -Wcomment -Wno-comment "$base/comments.c" \ + -o "$base/no-comment-last.out" 2>"$base/no-comment-last.err" +test ! -s "$base/no-comment-last.err" + +# -Werror does not enable optional comment diagnostics by itself. +"$MCPU_CPP" --no-config -Werror "$base/comments.c" \ + -o "$base/werror-only.out" 2>"$base/werror-only.err" +test ! -s "$base/werror-only.err" + +# When comment diagnostics are enabled, -Werror promotes them to errors. +if "$MCPU_CPP" --no-config -Wcomment -Werror "$base/comments.c" \ + -o "$base/comment-error.out" 2>"$base/comment-error.err"; then + echo '-Wcomment -Werror did not fail' >&2 + exit 1 +fi +grep 'error: "/\*" within comment' "$base/comment-error.err" >/dev/null + +# -Werror/-Wno-error have equal specificity; the last occurrence wins. +cat > "$base/directive.c" <<'EOT' +#warning requested warning +ok +EOT +"$MCPU_CPP" --no-config -Werror -Wno-error "$base/directive.c" \ + -o "$base/no-error.out" 2>"$base/no-error.err" +grep 'warning: #warning requested warning$' "$base/no-error.err" >/dev/null + +if "$MCPU_CPP" --no-config -Wno-error -Werror "$base/directive.c" \ + -o "$base/error-last.out" 2>"$base/error-last.err"; then + echo '-Wno-error -Werror did not fail' >&2 + exit 1 +fi +grep 'error: #warning requested warning$' "$base/error-last.err" >/dev/null + +# Existing always-issued warnings are also promoted by -Werror. +cat > "$base/width.c" <<'EOT' +#if 1z24 == 1 +ok +#endif +EOT +if "$MCPU_CPP" --no-config -Werror "$base/width.c" \ + -o "$base/width.out" 2>"$base/width.err"; then + echo 'zNNN warning was not promoted by -Werror' >&2 + exit 1 +fi +grep 'error: invalid zNNN integer-width suffix; suffix ignored$' "$base/width.err" >/dev/null + +cat > "$base/redefine.c" <<'EOT' +#define VALUE 1 +#define VALUE 2 +EOT +if "$MCPU_CPP" --no-config -Werror "$base/redefine.c" \ + -o "$base/redefine.out" 2>"$base/redefine.err"; then + echo 'macro redefinition warning was not promoted by -Werror' >&2 + exit 1 +fi +grep "error: macro 'VALUE' redefined$" "$base/redefine.err" >/dev/null + +cat > "$base/paste.c" <<'EOT' +#define BAD(X) X ## + +BAD(foo) +EOT +if "$MCPU_CPP" --no-config -Werror "$base/paste.c" \ + -o "$base/paste.out" 2>"$base/paste.err"; then + echo 'invalid paste warning was not promoted by -Werror' >&2 + exit 1 +fi +grep 'error: pasting "foo" and "+" does not give a valid preprocessing token$' \ + "$base/paste.err" >/dev/null diff --git a/tests/t0053-interface-cleanup.sh b/tests/t0053-interface-cleanup.sh new file mode 100755 index 0000000..80d5410 --- /dev/null +++ b/tests/t0053-interface-cleanup.sh @@ -0,0 +1,34 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-interface-cleanup-$$" +mkdir -p "$base/wrapper" "$base/after" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +# --version identifies the independent MCPU languages preprocessor. +"$MCPU_CPP" --version > "$base/version" +printf '%s\n' 'mcpu-cpp 1.0.2' 'MCPU languages preprocessor' > "$base/version.expected" +cmp "$base/version.expected" "$base/version" + +# --help contains the current public interface, including --object-suffix, +# and puts the stdin/stdout notes after the option list. +"$MCPU_CPP" --help > "$base/help" +grep '^mcpu-cpp 1\.0\.2$' "$base/help" >/dev/null +grep '^MCPU languages preprocessor$' "$base/help" >/dev/null +grep '^ --object-suffix SFX set object suffix used for dependency targets$' "$base/help" >/dev/null +help_version_line=$(grep -n '^ --version ' "$base/help" | cut -d: -f1) +input_note_line=$(grep -n "^If input is omitted or '-', read standard input\.$" "$base/help" | cut -d: -f1) +test "$input_note_line" -gt "$help_version_line" + +grep "^If output is omitted or '-', write standard output\.$" "$base/help" >/dev/null + +# --object-suffix requires an argument and changes the default dependency target. +cat > "$base/object.c" <<'SRC' +int x; +SRC +if "$MCPU_CPP" --no-config --object-suffix > /dev/null 2> "$base/object.err"; then + echo '--object-suffix unexpectedly accepted without an argument' >&2 + exit 1 +fi +grep "argument missing after '--object-suffix'" "$base/object.err" >/dev/null +"$MCPU_CPP" --no-config --object-suffix .obj -M "$base/object.c" > "$base/object.d" +grep '^object\.obj:' "$base/object.d" >/dev/null diff --git a/tests/t0054-inhibit-warnings.sh b/tests/t0054-inhibit-warnings.sh new file mode 100755 index 0000000..339d098 --- /dev/null +++ b/tests/t0054-inhibit-warnings.sh @@ -0,0 +1,81 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-inhibit-warnings-$$" +mkdir -p "$base" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +cat > "$base/directive.c" <<'EOT' +#warning requested warning +ok +EOT + +# -w suppresses #warning completely. +"$MCPU_CPP" --no-config -w "$base/directive.c" \ + -o "$base/directive.out" 2>"$base/directive.err" +test ! -s "$base/directive.err" +grep '^ok$' "$base/directive.out" >/dev/null + +# -w is a global warning gate, so command-line order relative to -Werror +# does not matter and there is no warning left to promote. +for options in '-w -Werror' '-Werror -w'; do + # shellcheck disable=SC2086 + "$MCPU_CPP" --no-config $options "$base/directive.c" \ + -o "$base/werror.out" 2>"$base/werror.err" + test ! -s "$base/werror.err" +done + +cat > "$base/width.c" <<'EOT' +#if 1z24 == 1 +ok +#endif +EOT +"$MCPU_CPP" --no-config -w "$base/width.c" \ + -o "$base/width.out" 2>"$base/width.err" +test ! -s "$base/width.err" +grep '^ok$' "$base/width.out" >/dev/null + +cat > "$base/redefine.c" <<'EOT' +#define VALUE 1 +#define VALUE 2 +VALUE +EOT +"$MCPU_CPP" --no-config -w "$base/redefine.c" \ + -o "$base/redefine.out" 2>"$base/redefine.err" +test ! -s "$base/redefine.err" +grep '^2$' "$base/redefine.out" >/dev/null + +cat > "$base/paste.c" <<'EOT' +#define BAD(X) X ## + +BAD(foo) +EOT +"$MCPU_CPP" --no-config -w "$base/paste.c" \ + -o "$base/paste.out" 2>"$base/paste.err" +test ! -s "$base/paste.err" + +cat > "$base/comments.c" <<'EOT' +/* outer /* nested */ +// continued \ +comment +ok +EOT + +# A specific warning class may be enabled, but -w still suppresses its output. +# Command-line order does not matter. +for options in '-w -Wcomment' '-Wcomment -w' '-w -Wall' '-Wall -w'; do + # shellcheck disable=SC2086 + "$MCPU_CPP" --no-config $options "$base/comments.c" \ + -o "$base/comments.out" 2>"$base/comments.err" + test ! -s "$base/comments.err" + grep '^ok$' "$base/comments.out" >/dev/null +done + +# Errors are not warnings and must remain visible/fatal under -w. +cat > "$base/error.c" <<'EOT' +#error requested error +EOT +if "$MCPU_CPP" --no-config -w "$base/error.c" \ + -o "$base/error.out" 2>"$base/error.err"; then + echo '-w suppressed a real preprocessing error' >&2 + exit 1 +fi +grep 'error: #error requested error$' "$base/error.err" >/dev/null diff --git a/tests/t0055-forced-files.sh b/tests/t0055-forced-files.sh new file mode 100755 index 0000000..454fa55 --- /dev/null +++ b/tests/t0055-forced-files.sh @@ -0,0 +1,242 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-forced-files-$$" +mkdir -p "$base/project/src" "$base/project/user" "$base/project/sys" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +cat > "$base/project/defs1.h" <<'EOT' +IMACROS1_BODY_MUST_NOT_APPEAR +#if CLI_VALUE != 9 +#error command-line macros were not applied before -imacros +#endif +#if __INCLUDE_LEVEL__ != 1 +#error -imacros must run at include level 1 +#endif +#define ORDER_VALUE 1 +#define FROM_IMACROS1 11 +#include "imacros-child.h" +#lang "as" +#define FROM_AS_LANG 13 +#endlang +EOT + +cat > "$base/project/imacros-child.h" <<'EOT' +IMACROS_CHILD_BODY_MUST_NOT_APPEAR +#if __INCLUDE_LEVEL__ != 2 +#error nested include from -imacros must run at include level 2 +#endif +#define FROM_IMACROS_CHILD 12 +EOT + +cat > "$base/project/defs2.h" <<'EOT' +IMACROS2_BODY_MUST_NOT_APPEAR +#if ORDER_VALUE != 1 +#error -imacros command-line order is broken +#endif +#if FROM_IMACROS_CHILD != 12 || FROM_AS_LANG != 13 +#error -imacros did not retain preprocessing state +#endif +#undef ORDER_VALUE +#define ORDER_VALUE 2 +EOT + +cat > "$base/project/a.h" <<'EOT' +#if ORDER_VALUE != 2 +#error all -imacros files must precede all -include files +#endif +INCLUDE_A_BODY +FORCED_LEVEL __INCLUDE_LEVEL__ +FORCED_BASE __BASE_FILE__ +#undef ORDER_VALUE +#define ORDER_VALUE 3 +EOT + +cat > "$base/project/b.h" <<'EOT' +#if ORDER_VALUE != 3 +#error -include command-line order is broken +#endif +INCLUDE_B_BODY +#undef ORDER_VALUE +#define ORDER_VALUE 4 +EOT + +cat > "$base/project/src/main.c" <<'EOT' +#if ORDER_VALUE != 4 +#error forced files did not complete before primary input +#endif +MAIN_BODY +MAIN_LEVEL __INCLUDE_LEVEL__ +FROM_IMACROS1 FROM_IMACROS_CHILD FROM_AS_LANG CLI_VALUE +EOT + +# Deliberately interleave -include and -imacros. The semantic order is all +# command-line -D/-U first, then all -imacros in their own argv order, then all +# -include in their own argv order, and only then the primary input. +( + cd "$base/project" + "$MCPU_CPP" --no-config \ + -include a.h -DCLI_VALUE=9 -imacros defs1.h \ + -include b.h -imacros defs2.h \ + src/main.c -o "$base/order.out" +) + +! grep 'IMACROS1_BODY_MUST_NOT_APPEAR' "$base/order.out" >/dev/null +! grep 'IMACROS2_BODY_MUST_NOT_APPEAR' "$base/order.out" >/dev/null +! grep 'IMACROS_CHILD_BODY_MUST_NOT_APPEAR' "$base/order.out" >/dev/null +! grep '^#lang ' "$base/order.out" >/dev/null +! grep '^#endlang' "$base/order.out" >/dev/null + +awk ' + /^INCLUDE_A_BODY$/ { a = NR } + /^INCLUDE_B_BODY$/ { b = NR } + /^MAIN_BODY$/ { c = NR } + END { exit !(a && a < b && b < c) } +' "$base/order.out" +grep '^11 12 13 9$' "$base/order.out" >/dev/null +grep '^FORCED_LEVEL 1$' "$base/order.out" >/dev/null +grep '^FORCED_BASE "src/main.c"$' "$base/order.out" >/dev/null +grep '^MAIN_LEVEL 0$' "$base/order.out" >/dev/null +# Each visible forced include returns to the primary file in the line-marker +# stack before another forced include or the primary body continues. +test `grep -c '^# 1 "src/main.c" 2$' "$base/order.out"` -eq 2 + +# A forced command-line file is searched from the current working directory, +# not from the physical directory of the primary input. +cat > "$base/project/cwd-first.h" <<'EOT' +CWD_FORCED_HEADER +EOT +cat > "$base/project/src/cwd-first.h" <<'EOT' +PRIMARY_DIRECTORY_MUST_NOT_WIN +EOT +cat > "$base/project/src/cwd-main.c" <<'EOT' +CWD_MAIN +EOT +( + cd "$base/project" + "$MCPU_CPP" --no-config -include cwd-first.h src/cwd-main.c \ + -o "$base/cwd.out" +) +grep '^CWD_FORCED_HEADER$' "$base/cwd.out" >/dev/null +! grep 'PRIMARY_DIRECTORY_MUST_NOT_WIN' "$base/cwd.out" >/dev/null + +# After CWD, the ordinary quoted include chain is used. A forced file found +# through -I retains its physical provenance, so quoted includes inside it are +# resolved relative to that file's own directory. +cat > "$base/project/user/forced-user.h" <<'EOT' +FORCED_USER_BEGIN +#include "forced-child.h" +FORCED_USER_END +EOT +cat > "$base/project/user/forced-child.h" <<'EOT' +FORCED_USER_CHILD +EOT +( + cd "$base/project" + "$MCPU_CPP" --no-config -I user -include forced-user.h src/cwd-main.c \ + -o "$base/user.out" +) +grep '^FORCED_USER_BEGIN$' "$base/user.out" >/dev/null +grep '^FORCED_USER_CHILD$' "$base/user.out" >/dev/null +grep '^FORCED_USER_END$' "$base/user.out" >/dev/null + +# Search-chain provenance of a forced file is retained for #include_next. +cat > "$base/project/user/wrapped.h" <<'EOT' +FORCED_WRAPPER +#include_next <wrapped.h> +EOT +cat > "$base/project/sys/wrapped.h" <<'EOT' +FORCED_INCLUDE_NEXT +EOT +( + cd "$base/project" + "$MCPU_CPP" --no-config -I user -isystem sys -include wrapped.h \ + src/cwd-main.c -o "$base/next.out" +) +grep '^FORCED_WRAPPER$' "$base/next.out" >/dev/null +grep '^FORCED_INCLUDE_NEXT$' "$base/next.out" >/dev/null + +# -dD does not make normal output from -imacros visible; macro state still +# reaches the primary input. +cat > "$base/project/only-macros.h" <<'EOT' +IMACROS_DD_BODY_MUST_NOT_APPEAR +#define DD_VALUE 77 +EOT +cat > "$base/project/src/dd-main.c" <<'EOT' +DD_VALUE +EOT +( + cd "$base/project" + "$MCPU_CPP" --no-config -dD -imacros only-macros.h src/dd-main.c \ + -o "$base/dd.out" +) +! grep 'IMACROS_DD_BODY_MUST_NOT_APPEAR' "$base/dd.out" >/dev/null +! grep '^#define DD_VALUE 77$' "$base/dd.out" >/dev/null +grep '^77$' "$base/dd.out" >/dev/null + +# Forced files participate in physical dependency tracking. The primary input +# remains the first dependency even though forced files are preprocessed first. +( + cd "$base/project" + "$MCPU_CPP" --no-config -M -DCLI_VALUE=9 \ + -imacros defs1.h -imacros defs2.h -include a.h -include b.h src/main.c \ + > "$base/deps.out" +) +grep '^main\.o:' "$base/deps.out" >/dev/null +first_dep=`sed 's/^[^:]*: //' "$base/deps.out" | awk '{print $1}'` +test "$first_dep" = 'src/main.c' || { + echo 'primary input is not the first forced-file dependency' >&2 + exit 1 +} +for f in defs1.h imacros-child.h defs2.h a.h b.h; do + grep -E "(^|[[:space:]])(\./)?$f([[:space:]]|$)" "$base/deps.out" >/dev/null || { + echo "forced dependency missing: $f" >&2 + exit 1 + } +done + +# A forced file found through -isystem has system dependency class and is +# therefore omitted by -MM while the primary user source remains. +cat > "$base/project/sys/system-forced.h" <<'EOT' +SYSTEM_FORCED_BODY +EOT +( + cd "$base/project" + "$MCPU_CPP" --no-config -MM -isystem sys -include system-forced.h \ + src/cwd-main.c > "$base/mm.out" +) +grep -E '(^|[[:space:]])src/cwd-main\.c([[:space:]]|$)' "$base/mm.out" >/dev/null +! grep -F "$base/project/sys/system-forced.h" "$base/mm.out" >/dev/null + +# Public help advertises both implemented forced-file options. +"$MCPU_CPP" --help > "$base/help" +grep '^ -imacros FILE ' "$base/help" >/dev/null +grep '^ -include FILE ' "$base/help" >/dev/null + +# Both option families require an argument. +for opt in -include -imacros; do + if "$MCPU_CPP" --no-config "$opt" > /dev/null 2>"$base/missing-arg.err"; then + echo "$opt accepted a missing argument" >&2 + exit 1 + fi + grep -- "argument missing after '$opt'" "$base/missing-arg.err" >/dev/null +done + +# Missing forced files are preprocessing errors for both option families. +for opt in -include -imacros; do + if ( + cd "$base/project" + "$MCPU_CPP" --no-config "$opt" no-such-forced-file.h src/cwd-main.c \ + -o "$base/missing.out" 2>"$base/missing.err" + ); then + echo "$opt accepted a missing forced file" >&2 + exit 1 + fi + grep -- "$opt cannot find file 'no-such-forced-file.h'" "$base/missing.err" >/dev/null +done + +# The same forced-file pipeline also applies when the primary source is stdin. +printf 'DD_VALUE\n' | ( + cd "$base/project" + "$MCPU_CPP" --no-config -imacros only-macros.h - +) > "$base/stdin.out" +grep '^77$' "$base/stdin.out" >/dev/null diff --git a/tests/t0056-dependency-targets.sh b/tests/t0056-dependency-targets.sh new file mode 100755 index 0000000..ea5f4b9 --- /dev/null +++ b/tests/t0056-dependency-targets.sh @@ -0,0 +1,103 @@ +#!/bin/sh +set -eu + +base="${TMPDIR-/tmp}/mcpu-cpp-dependency-targets-$$" +mkdir -p "$base" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +cd "$base" + +cat > main.c <<'EOT' +#include "local.h" +int main_value; +EOT + +cat > local.h <<'EOT' +#define LOCAL_VALUE 1 +EOT + +# The established default target remains make-quoted and uses the configured +# object suffix only when no explicit -MT/-MQ target is present. +"$MCPU_CPP" --no-config -M main.c > default.d +grep '^main\.o: main\.c \./local\.h$' default.d >/dev/null + +"$MCPU_CPP" --no-config --object-suffix .obj -M main.c > suffix.d +grep '^main\.obj: main\.c \./local\.h$' suffix.d >/dev/null + +# -MT replaces the default target and is emitted exactly as supplied. +"$MCPU_CPP" --no-config -M -MT build/main.o main.c > mt.d +grep '^build/main\.o: main\.c \./local\.h$' mt.d >/dev/null + +"$MCPU_CPP" --no-config -M -MTattached.o main.c > mt-attached.d +grep '^attached\.o: main\.c \./local\.h$' mt-attached.d >/dev/null + +"$MCPU_CPP" --no-config -M -MT 'raw one raw#two$' main.c > mt-raw.d +grep '^raw one raw#two\$: main\.c \./local\.h$' mt-raw.d >/dev/null + +# -MQ applies Make quoting to the target. +"$MCPU_CPP" --no-config -M -MQ '$(OBJDIR)/main.o' main.c > mq.d +grep '^\$\$(OBJDIR)/main\.o: main\.c \./local\.h$' mq.d >/dev/null + +"$MCPU_CPP" --no-config -M '-MQ$(OBJDIR)/attached.o' main.c > mq-attached.d +grep '^\$\$(OBJDIR)/attached\.o: main\.c \./local\.h$' mq-attached.d >/dev/null + +"$MCPU_CPP" --no-config -M -MQ 'dir name/#x$.o' main.c > mq-special.d +grep '^dir\\ name/\\#x\$\$\.o: main\.c \./local\.h$' mq-special.d >/dev/null + +"$MCPU_CPP" --no-config -M -MQ 'path\name.o' main.c > mq-backslash.d +grep -F 'path\name.o: main.c ./local.h' mq-backslash.d >/dev/null + +# Multiple targets produce one rule. As in GNU CPP, all -MT targets precede +# all -MQ targets; command-line order is preserved inside each class. +"$MCPU_CPP" --no-config -M \ + -MQ '$(DIR)/quoted1.o' \ + -MT raw1.o \ + -MQ 'quoted two.o' \ + -MT 'raw2.o raw3.o' \ + main.c > multiple.d +grep '^raw1\.o raw2\.o raw3\.o \$\$(DIR)/quoted1\.o quoted\\ two\.o: main\.c \./local\.h$' multiple.d >/dev/null + +# Explicit targets override the automatic target even when an object suffix was +# requested. +"$MCPU_CPP" --no-config --object-suffix .obj -M -MT explicit.target main.c > explicit.d +grep '^explicit\.target: main\.c \./local\.h$' explicit.d >/dev/null + +# -MT/-MQ work in side-effect dependency modes and with -MF. +"$MCPU_CPP" --no-config -MD -MF md.d -MT md-target.o main.c > md.i +grep '^md-target\.o: main\.c \./local\.h$' md.d >/dev/null +grep '^int main_value;$' md.i >/dev/null + +"$MCPU_CPP" --no-config -MMD -MF mmd.d -MQ '$(OUT)/mmd.o' main.c > mmd.i +grep '^\$\$(OUT)/mmd\.o: main\.c \./local\.h$' mmd.d >/dev/null +grep '^int main_value;$' mmd.i >/dev/null + +# Both options require an active dependency-generation mode. +if "$MCPU_CPP" --no-config -MT orphan.o main.c > orphan.out 2> orphan.err; then + echo "-MT without dependency generation unexpectedly succeeded" >&2 + exit 1 +fi +grep -- '-MT/-MQ require one of -M, -MM, -MD or -MMD' orphan.err >/dev/null + +if "$MCPU_CPP" --no-config -MQ orphan.o main.c > orphan-q.out 2> orphan-q.err; then + echo "-MQ without dependency generation unexpectedly succeeded" >&2 + exit 1 +fi +grep -- '-MT/-MQ require one of -M, -MM, -MD or -MMD' orphan-q.err >/dev/null + +# Missing arguments remain command-line errors. +if "$MCPU_CPP" --no-config -M -MT > missing-mt.out 2> missing-mt.err; then + echo "-MT without an argument unexpectedly succeeded" >&2 + exit 1 +fi +grep "argument missing after '-MT'" missing-mt.err >/dev/null + +if "$MCPU_CPP" --no-config -M -MQ > missing-mq.out 2> missing-mq.err; then + echo "-MQ without an argument unexpectedly succeeded" >&2 + exit 1 +fi +grep "argument missing after '-MQ'" missing-mq.err >/dev/null + +# Public help documents both options. +"$MCPU_CPP" --help > help.out +grep '^ -MT TARGET set unquoted make dependency target$' help.out >/dev/null +grep '^ -MQ TARGET set make-quoted dependency target$' help.out >/dev/null diff --git a/tests/t0057-missing-generated-dependencies.sh b/tests/t0057-missing-generated-dependencies.sh new file mode 100755 index 0000000..7ad5001 --- /dev/null +++ b/tests/t0057-missing-generated-dependencies.sh @@ -0,0 +1,166 @@ +#!/bin/sh +set -eu + +base="${TMPDIR-/tmp}/mcpu-cpp-missing-generated-deps-$$" +mkdir -p "$base/user" "$base/sys" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +cd "$base" + +cat > main.c <<'EOT' +#include "generated-user.h" +#include <generated-system.h> +#define GENERATED_MACRO_HEADER "generated-macro.h" +#include GENERATED_MACRO_HEADER +EOT + +# Without -MG, the first unresolved include remains a hard error. +if "$MCPU_CPP" --no-config -M main.c > no-mg.out 2> no-mg.err; then + echo "missing include unexpectedly succeeded without -MG" >&2 + exit 1 +fi +grep "cannot find include file 'generated-user.h'" no-mg.err >/dev/null + +# -M -MG records unresolved operands exactly as written and does not diagnose +# them as missing files. Macro-expanded include operands use their expanded +# filename. +"$MCPU_CPP" --no-config -M -MG main.c > M.d 2> M.err +test ! -s M.err +grep '^main\.o:' M.d >/dev/null +for f in main.c generated-user.h generated-system.h generated-macro.h +do + grep -F "$f" M.d >/dev/null || { + echo "-M -MG omitted unresolved dependency: $f" >&2 + exit 1 + } +done + +# -MM retains unresolved quoted/user includes but filters unresolved angle +# includes as system-class dependencies. +"$MCPU_CPP" --no-config -MM -MG main.c > MM.d +grep -F 'generated-user.h' MM.d >/dev/null +grep -F 'generated-macro.h' MM.d >/dev/null +if grep -F 'generated-system.h' MM.d >/dev/null; then + echo "-MM -MG kept unresolved angle/system dependency" >&2 + exit 1 +fi + +# Missing dependencies reached from a real system header inherit system +# context, even when the nested directive uses quotes. User-header quoted +# descendants remain user dependencies. +cat > sys/system-parent.h <<'EOT' +#include "generated-from-system.h" +EOT +cat > user/user-parent.h <<'EOT' +#include "generated-from-user.h" +EOT +cat > nested.c <<'EOT' +#include <system-parent.h> +#include <user-parent.h> +EOT + +"$MCPU_CPP" --no-config -MM -MG -I user -isystem sys nested.c > nested-MM.d +grep -F 'user/user-parent.h' nested-MM.d >/dev/null +grep -F 'generated-from-user.h' nested-MM.d >/dev/null +if grep -F 'sys/system-parent.h' nested-MM.d >/dev/null || + grep -F 'generated-from-system.h' nested-MM.d >/dev/null; then + echo "-MM -MG kept system-context dependency" >&2 + exit 1 +fi + +"$MCPU_CPP" --no-config -M -MG -I user -isystem sys nested.c > nested-M.d +grep -F 'sys/system-parent.h' nested-M.d >/dev/null +grep -F 'generated-from-system.h' nested-M.d >/dev/null + +# Unresolved dependency identity is textual and separate from physical +# st_dev/st_ino identity. shadow.h exists in CWD, but <shadow.h> is not found +# through the angle search chain under --no-config; a later quoted ./shadow.h +# resolves that physical file. Both registry entries must survive under -M. +cat > shadow.h <<'EOT' +#define SHADOW_VALUE 1 +EOT +cat > identity.c <<'EOT' +#include <shadow.h> +#include "./shadow.h" +EOT +"$MCPU_CPP" --no-config -M -MG identity.c > identity-M.d +grep '^identity\.o: identity\.c shadow\.h \./\./shadow\.h$' identity-M.d >/dev/null || { + echo "unresolved and physical dependency identities were incorrectly merged" >&2 + cat identity-M.d >&2 + exit 1 +} + +# In -MM the unresolved angle entry is filtered, while the resolved quoted +# physical dependency remains. +"$MCPU_CPP" --no-config -MM -MG identity.c > identity-MM.d +grep -F '././shadow.h' identity-MM.d >/dev/null +case "$(cat identity-MM.d)" in + *' shadow.h '*) + echo "-MM kept the unresolved angle dependency" >&2 + exit 1 + ;; +esac + +# Repeated unresolved operands are deduplicated textually. Their first +# classification is retained, matching GNU CPP: angle-first stays system-only, +# while quote-first stays visible to -MM. +cat > first-system.c <<'EOT' +#include <same-missing.h> +#include "same-missing.h" +EOT +"$MCPU_CPP" --no-config -MM -MG first-system.c > first-system.d +if grep -F 'same-missing.h' first-system.d >/dev/null; then + echo "angle-first unresolved dependency lost its first classification" >&2 + exit 1 +fi + +cat > first-user.c <<'EOT' +#include "same-missing.h" +#include <same-missing.h> +EOT +"$MCPU_CPP" --no-config -MM -MG first-user.c > first-user.d +grep -F 'same-missing.h' first-user.d >/dev/null + +# Different unresolved spellings are distinct because no physical identity is +# available to canonicalize them. +cat > spelling.c <<'EOT' +#include "generated.h" +#include "./generated.h" +EOT +"$MCPU_CPP" --no-config -M -MG spelling.c > spelling.d +grep -F ' generated.h' spelling.d >/dev/null +grep -F ' ./generated.h' spelling.d >/dev/null + +# -MG also applies to missing command-line forced files. They enter the same +# unresolved registry as user dependencies and preserve the command-line +# operand; no synthetic path is prepended. +: > forced.c +"$MCPU_CPP" --no-config -M -MG \ + -imacros missing-macros.h -include missing-include.h forced.c > forced-M.d +grep -F 'missing-macros.h' forced-M.d >/dev/null +grep -F 'missing-include.h' forced-M.d >/dev/null + +"$MCPU_CPP" --no-config -MM -MG \ + -imacros missing-macros.h -include missing-include.h forced.c > forced-MM.d +grep -F 'missing-macros.h' forced-MM.d >/dev/null +grep -F 'missing-include.h' forced-MM.d >/dev/null + +# -MG is dependency-only: GNU permits it with -M/-MM, not side-effect -MD/-MMD +# and not without dependency generation. +for opts in '-MG' '-MD -MG' '-MMD -MG' +do + if "$MCPU_CPP" --no-config $opts forced.c > invalid.out 2> invalid.err; then + echo "$opts unexpectedly accepted" >&2 + exit 1 + fi + grep -- '-MG may only be used with -M or -MM' invalid.err >/dev/null +done + +# Explicit dependency targets continue to compose with -MG. +"$MCPU_CPP" --no-config -M -MG -MT generated-target.o main.c > target.d +grep '^generated-target\.o:' target.d >/dev/null +grep -F 'generated-user.h' target.d >/dev/null + +# Public help exposes the implemented option. +"$MCPU_CPP" --help > help.out +grep '^ -MG treat missing headers as generated dependencies$' help.out >/dev/null diff --git a/tests/t0058-variadic-macros.sh b/tests/t0058-variadic-macros.sh new file mode 100755 index 0000000..584e980 --- /dev/null +++ b/tests/t0058-variadic-macros.sh @@ -0,0 +1,135 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-variadic-$$" +mkdir -p "$base" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +cat > "$base/main.c" <<'EOT' +#define A 7 +#define SUM(X, Y) X + Y +#define V(...) <__VA_ARGS__> +#define F(first, ...) first | __VA_ARGS__ +#define STRV(...) #__VA_ARGS__ +#define LEFT(...) pre ## __VA_ARGS__ +#define RIGHT(...) __VA_ARGS__ ## post +#define CALL(FN, ...) FN(__VA_ARGS__) +#define WRAP(...) V(__VA_ARGS__) +#define ID(X) X +#define РУССКИЙ(первый, ...) первый : __VA_ARGS__ + +V(alpha, beta, gamma) +V() +F(one, two, three) +F(one) +F(one,) +F(A, SUM(1, 2)) +STRV(A, b + c) +LEFT(fix) +LEFT() +RIGHT(fix) +RIGHT() +CALL(ID, A) +WRAP(A, SUM(3, 4)) +РУССКИЙ(один, два, три) +EOT + +"$MCPU_CPP" --no-config "$base/main.c" -o "$base/out" + +grep -Fx '<alpha, beta, gamma>' "$base/out" >/dev/null +grep -Fx '<>' "$base/out" >/dev/null +grep -Fx 'one | two, three' "$base/out" >/dev/null +# Both an omitted variadic tail and an explicitly empty one expand to no tokens. +test "$(grep -Fxc 'one | ' "$base/out")" -eq 2 +grep -Fx '7 | 1 + 2' "$base/out" >/dev/null +grep -Fx '"A, b + c"' "$base/out" >/dev/null +grep -Fx 'prefix' "$base/out" >/dev/null +grep -Fx 'pre' "$base/out" >/dev/null +grep -Fx 'fixpost' "$base/out" >/dev/null +grep -Fx 'post' "$base/out" >/dev/null +grep -Fx '7' "$base/out" >/dev/null +grep -Fx '<7, 3 + 4>' "$base/out" >/dev/null +grep -Fx 'один : два, три' "$base/out" >/dev/null + +# The variable argument is preserved as such when macro definitions are dumped. +"$MCPU_CPP" --no-config -dM "$base/main.c" > "$base/dump" +grep -Fx '#define V(...) <__VA_ARGS__>' "$base/dump" >/dev/null +grep -Fx '#define F(first,...) first | __VA_ARGS__' "$base/dump" >/dev/null +grep -Fx '#define STRV(...) #__VA_ARGS__' "$base/dump" >/dev/null +grep -Fx '#define РУССКИЙ(первый,...) первый : __VA_ARGS__' "$base/dump" >/dev/null + + +# Variadic-ness is part of a macro definition. Repeating the same variadic +# definition is harmless, while changing a non-variadic definition into a +# variadic one is a real redefinition and therefore participates in -Werror. +cat > "$base/redefine-same.c" <<'EOT' +#define SAME(x, ...) x | __VA_ARGS__ +#define SAME(x, ...) x | __VA_ARGS__ +SAME(1, 2) +EOT +"$MCPU_CPP" --no-config "$base/redefine-same.c" \ + -o "$base/redefine-same.out" 2>"$base/redefine-same.err" +test ! -s "$base/redefine-same.err" +grep -Fx '1 | 2' "$base/redefine-same.out" >/dev/null + +cat > "$base/redefine-kind.c" <<'EOT' +#define CHANGE(x) x +#define CHANGE(x, ...) x | __VA_ARGS__ +EOT +if "$MCPU_CPP" --no-config -Werror "$base/redefine-kind.c" \ + -o "$base/redefine-kind.out" 2>"$base/redefine-kind.err"; then + echo 'variadic/non-variadic redefinition escaped -Werror' >&2 + exit 1 +fi +grep "error: macro 'CHANGE' redefined" "$base/redefine-kind.err" >/dev/null + +# A variadic macro may have fixed parameters, but those fixed parameters are +# still required. An empty spelling counts as an argument, just as it does +# for ordinary function-like macros. +cat > "$base/few.c" <<'EOT' +#define NEEDS_TWO(a, b, ...) a + b + __VA_ARGS__ +NEEDS_TWO(1) +EOT +if "$MCPU_CPP" --no-config "$base/few.c" -o "$base/few.out" 2>"$base/few.err"; then + echo 'variadic macro accepted too few fixed arguments' >&2 + exit 1 +fi +grep "macro 'NEEDS_TWO' used with too few arguments" "$base/few.err" >/dev/null + +# The old GNU named variadic form is deliberately outside the 0.0.46 +# contract. Only the C99-style ... + __VA_ARGS__ form is accepted. +cat > "$base/named.c" <<'EOT' +#define OLD(args...) args +EOT +if "$MCPU_CPP" --no-config "$base/named.c" -o "$base/named.out" 2>"$base/named.err"; then + echo 'GNU named variadic macro form unexpectedly accepted' >&2 + exit 1 +fi +grep 'badly punctuated parameter list in #define' "$base/named.err" >/dev/null + +cat > "$base/malformed.c" <<'EOT' +#define BAD(..., x) x +EOT +if "$MCPU_CPP" --no-config "$base/malformed.c" -o "$base/malformed.out" 2>"$base/malformed.err"; then + echo 'parameters after ... unexpectedly accepted' >&2 + exit 1 +fi +grep 'badly punctuated parameter list in #define' "$base/malformed.err" >/dev/null + +cat > "$base/reserved.c" <<'EOT' +#define BAD(__VA_ARGS__) __VA_ARGS__ +EOT +if "$MCPU_CPP" --no-config "$base/reserved.c" -o "$base/reserved.out" 2>"$base/reserved.err"; then + echo '__VA_ARGS__ unexpectedly accepted as an ordinary parameter name' >&2 + exit 1 +fi +grep "'__VA_ARGS__' cannot be used as a macro parameter name" "$base/reserved.err" >/dev/null + +# A replacement using #__VA_ARGS__ is valid only for a variadic macro. +cat > "$base/not-variadic.c" <<'EOT' +#define BAD(x) #__VA_ARGS__ +EOT +if "$MCPU_CPP" --no-config "$base/not-variadic.c" -o "$base/not-variadic.out" 2>"$base/not-variadic.err"; then + echo '#__VA_ARGS__ unexpectedly accepted in a non-variadic macro' >&2 + exit 1 +fi +grep "'#' operator should be followed by a macro argument name" "$base/not-variadic.err" >/dev/null diff --git a/tests/t0059-va-opt.sh b/tests/t0059-va-opt.sh new file mode 100755 index 0000000..052df6c --- /dev/null +++ b/tests/t0059-va-opt.sh @@ -0,0 +1,150 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-va-opt-$$" +mkdir -p "$base" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +cat > "$base/main.c" <<'EOT' +#define EMPTY +#define X 123 +#define xy RESCANNED +#define HAS(...) __VA_OPT__(yes) +#define COMMA(format, ...) call(format __VA_OPT__(,) __VA_ARGS__) +#define STR(...) #__VA_OPT__(__VA_ARGS__) +#define STRFIX(a, ...) #__VA_OPT__(a __VA_ARGS__) +#define STRPASTE(a, ...) #__VA_OPT__(a ## z) +#define LEFT(...) pre ## __VA_OPT__(__VA_ARGS__) +#define RIGHT(...) __VA_OPT__(__VA_ARGS__) ## post +#define INNER(a, ...) __VA_OPT__(a ## y) +#define BALANCED(...) __VA_OPT__((a, (b, c))) +#define EMPTY_LEFT(...) x ## __VA_OPT__() +#define EMPTY_RIGHT(...) __VA_OPT__() ## y +#define TEXT(...) "__VA_OPT__(not syntax)" __VA_OPT__(ok) + +[HAS()] +[HAS(EMPTY)] +[HAS(token)] +COMMA("zero") +COMMA("two", X, 7) +STR() +STR(X) +STRFIX(X, y) +STRPASTE(X, y) +LEFT() +LEFT(X) +RIGHT() +RIGHT(X) +INNER(x, token) +BALANCED(token) +EMPTY_LEFT(token) +EMPTY_RIGHT(token) +TEXT() +TEXT(token) +EOT + +"$MCPU_CPP" --no-config "$base/main.c" -o "$base/out" +sed '/^#/d; /^[[:space:]]*$/d; s/[[:space:]]//g' "$base/out" > "$base/norm" + +cat > "$base/expected" <<'EOT' +[] +[] +[yes] +call("zero") +call("two",123,7) +"" +"123" +"123y" +"Xz" +pre +pre123 +post +123post +RESCANNED +(a,(b,c)) +x +y +"__VA_OPT__(notsyntax)" +"__VA_OPT__(notsyntax)"ok +EOT + +diff -u "$base/expected" "$base/norm" + +# __VA_OPT__ remains visible in macro dumps as part of the replacement list. +"$MCPU_CPP" --no-config -dM "$base/main.c" > "$base/dump" +grep -Fx '#define HAS(...) __VA_OPT__(yes)' "$base/dump" >/dev/null +grep -Fx '#define COMMA(format,...) call(format __VA_OPT__(,) __VA_ARGS__)' "$base/dump" >/dev/null +grep -Fx '#define STR(...) #__VA_OPT__(__VA_ARGS__)' "$base/dump" >/dev/null + +# __VA_OPT__ is structural syntax, not a general identifier-like extension. +cat > "$base/nonvariadic.c" <<'EOT' +#define BAD(x) __VA_OPT__(x) +EOT +if "$MCPU_CPP" --no-config "$base/nonvariadic.c" -o "$base/nonvariadic.out" 2>"$base/nonvariadic.err"; then + echo '__VA_OPT__ unexpectedly accepted in a non-variadic macro' >&2 + exit 1 +fi +grep "'__VA_OPT__' may appear only in a variadic macro replacement list" "$base/nonvariadic.err" >/dev/null + +cat > "$base/object.c" <<'EOT' +#define BAD __VA_OPT__(x) +EOT +if "$MCPU_CPP" --no-config "$base/object.c" -o "$base/object.out" 2>"$base/object.err"; then + echo '__VA_OPT__ unexpectedly accepted in an object-like macro' >&2 + exit 1 +fi +grep "'__VA_OPT__' may appear only in a variadic macro replacement list" "$base/object.err" >/dev/null + +cat > "$base/nested.c" <<'EOT' +#define BAD(...) __VA_OPT__(a __VA_OPT__(b)) +EOT +if "$MCPU_CPP" --no-config "$base/nested.c" -o "$base/nested.out" 2>"$base/nested.err"; then + echo 'nested __VA_OPT__ unexpectedly accepted' >&2 + exit 1 +fi +grep "'__VA_OPT__' may not appear inside another '__VA_OPT__'" "$base/nested.err" >/dev/null + +cat > "$base/no-open.c" <<'EOT' +#define BAD(...) __VA_OPT__ x +EOT +if "$MCPU_CPP" --no-config "$base/no-open.c" -o "$base/no-open.out" 2>"$base/no-open.err"; then + echo '__VA_OPT__ without opening parenthesis unexpectedly accepted' >&2 + exit 1 +fi +grep "'__VA_OPT__' must be followed by '('" "$base/no-open.err" >/dev/null + +cat > "$base/unterminated.c" <<'EOT' +#define BAD(...) __VA_OPT__((x) +EOT +if "$MCPU_CPP" --no-config "$base/unterminated.c" -o "$base/unterminated.out" 2>"$base/unterminated.err"; then + echo 'unterminated __VA_OPT__ unexpectedly accepted' >&2 + exit 1 +fi +grep "unterminated '__VA_OPT__'" "$base/unterminated.err" >/dev/null + +cat > "$base/paste-first.c" <<'EOT' +#define BAD(...) __VA_OPT__(## x) +EOT +if "$MCPU_CPP" --no-config "$base/paste-first.c" -o "$base/paste-first.out" 2>"$base/paste-first.err"; then + echo 'leading ## inside __VA_OPT__ unexpectedly accepted' >&2 + exit 1 +fi +grep "'##' cannot appear at the beginning of '__VA_OPT__'" "$base/paste-first.err" >/dev/null + +cat > "$base/paste-last.c" <<'EOT' +#define BAD(...) __VA_OPT__(x ##) +EOT +if "$MCPU_CPP" --no-config "$base/paste-last.c" -o "$base/paste-last.out" 2>"$base/paste-last.err"; then + echo 'trailing ## inside __VA_OPT__ unexpectedly accepted' >&2 + exit 1 +fi +grep "'##' cannot appear at the end of '__VA_OPT__'" "$base/paste-last.err" >/dev/null + +# The historical GNU comma-swallow extension remains deliberately absent. +# __VA_OPT__ is the supported way to make a separator conditional. +cat > "$base/no-gnu-comma.c" <<'EOT' +#define OLD(format, ...) call(format, ## __VA_ARGS__) +OLD("x") +EOT +"$MCPU_CPP" --no-config "$base/no-gnu-comma.c" -o "$base/no-gnu-comma.out" 2>"$base/no-gnu-comma.err" +sed '/^#/d; /^[[:space:]]*$/d; s/[[:space:]]//g' "$base/no-gnu-comma.out" > "$base/no-gnu-comma.norm" +grep -Fx 'call("x",)' "$base/no-gnu-comma.norm" >/dev/null diff --git a/tests/t0060-macro-whitespace.sh b/tests/t0060-macro-whitespace.sh new file mode 100755 index 0000000..8f870ba --- /dev/null +++ b/tests/t0060-macro-whitespace.sh @@ -0,0 +1,72 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-macro-whitespace-$$" +mkdir -p "$base" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +cat > "$base/main.c" <<'EOT' +#define MULTI(fmt, ...) \ + do \ + { \ + output(fmt __VA_OPT__(,) __VA_ARGS__); \ + done(); \ + } \ + while( 0 ) + +#define TEXT "left right" +#define OPS + + - > < < +#define ID(x) x +#define V(...) __VA_OPT__(__VA_ARGS__) + +void +f( int x ) +{ + MULTI("x=%d", x); + MULTI("hello"); + ID(alpha + beta); + V(gamma + delta); + const char *s = TEXT; + OPS +} +EOT + +"$MCPU_CPP" --no-config "$base/main.c" -o "$base/out" +sed '/^#/d; /^[[:space:]]*$/d' "$base/out" > "$base/norm" + +cat > "$base/expected" <<'EOT' +void +f( int x ) +{ + do { output("x=%d" , x); done(); } while( 0 ); + do { output("hello" ); done(); } while( 0 ); + alpha + beta; + gamma + delta; + const char *s = "left right"; + + + - > < < +} +EOT + +diff -u "$base/expected" "$base/norm" + +# Macro dumps use the same normalized replacement-list whitespace. +"$MCPU_CPP" --no-config -dM "$base/main.c" > "$base/dump" +grep -Fx '#define MULTI(fmt,...) do { output(fmt __VA_OPT__(,) __VA_ARGS__); done(); } while( 0 )' "$base/dump" >/dev/null +grep -Fx '#define TEXT "left right"' "$base/dump" >/dev/null +grep -Fx '#define OPS + + - > < <' "$base/dump" >/dev/null +grep -Fx '#define ID(x) x' "$base/dump" >/dev/null +grep -Fx '#define V(...) __VA_OPT__(__VA_ARGS__)' "$base/dump" >/dev/null + +# Ordinary source text that is not replacement-list formatting stays untouched. +cat > "$base/plain.c" <<'EOT' +int plain = 1; +#define ID(x) x +#define V(...) __VA_OPT__(__VA_ARGS__) +ID(a + b) +EOT +"$MCPU_CPP" --no-config "$base/plain.c" -o "$base/plain.out" +sed '/^#/d; /^[[:space:]]*$/d' "$base/plain.out" > "$base/plain.norm" +cat > "$base/plain.expected" <<'EOT' +int plain = 1; +a + b +EOT +diff -u "$base/plain.expected" "$base/plain.norm" diff --git a/tests/t0061-output-line-compaction.sh b/tests/t0061-output-line-compaction.sh new file mode 100755 index 0000000..e40b1ee --- /dev/null +++ b/tests/t0061-output-line-compaction.sh @@ -0,0 +1,131 @@ +#!/bin/sh +set -eu +base="${TMPDIR-/tmp}/mcpu-cpp-output-line-compaction-$$" +mkdir -p "$base" +trap 'rm -rf "$base"' EXIT HUP INT TERM + +# Seven invisible lines stay as seven ordinary newlines. The next visible +# source line is line 9, and no corrective marker is needed. +cat > "$base/gap7.c" <<'EOT' +int a; +// one +#define A 1 +#if 0 +hidden +#endif +/* six */ + +int b; +EOT +"$MCPU_CPP" --no-config "$base/gap7.c" -o "$base/gap7.out" +if grep -F "# 9 \"$base/gap7.c\"" "$base/gap7.out" >/dev/null; then + echo "unexpected line marker for seven-line invisible gap" >&2 + exit 1 +fi +gap7_count=`awk ' + $0 == "int a;" { inside = 1; next } + $0 == "int b;" { print count; exit } + inside { ++count } +' "$base/gap7.out"` +test "$gap7_count" -eq 7 + +# Eight invisible lines cross the GNU CPP threshold. They disappear from the +# byte stream and are represented by a line marker for the next visible line. +cat > "$base/gap8.c" <<'EOT' +int a; +// one +#define A 1 +#if 0 +hidden +#endif +/* six */ +// seven + +int b; +EOT +"$MCPU_CPP" --no-config "$base/gap8.c" -o "$base/gap8.out" +grep -F "# 10 \"$base/gap8.c\"" "$base/gap8.out" >/dev/null +awk ' + /^# 10 "/ { getline; if( $0 == "int b;" ) ok = 1 } + END { exit ok ? 0 : 1 } +' "$base/gap8.out" + +# A completely invisible header has no synthetic end-of-header marker. Only +# the structural enter/return markers survive. A long invisible gap in the +# parent is then represented by the parent's next visible source position. +cat > "$base/silent.h" <<'EOT' +#ifndef SILENT_H +#define SILENT_H 1 +#define H1 1 +#define H2 2 +#define H3 3 +#define H4 4 +#define H5 5 +#define H6 6 +#define H7 7 +#define H8 8 +#define H9 9 +#endif +EOT +cat > "$base/include.c" <<'EOT' +// leading comment +#include "silent.h" +// one +#define P1 1 +#if 0 +hidden +#endif +// six +// seven + +int visible; +EOT +"$MCPU_CPP" --no-config -I"$base" "$base/include.c" -o "$base/include.out" +grep -F "# 1 \"$base/silent.h\" 1" "$base/include.out" >/dev/null +grep -F "# 3 \"$base/include.c\" 2" "$base/include.out" >/dev/null +grep -F "# 11 \"$base/include.c\"" "$base/include.out" >/dev/null +if grep -E "^# (2|3|4|5|6|7|8|9|10|11|12) \"$base/silent.h\"" "$base/include.out" >/dev/null; then + echo "silent header acquired a synthetic progress marker" >&2 + exit 1 +fi +awk ' + /\/silent\.h" 1$/ { getline; if( $0 ~ /\/include\.c" 2$/ ) adjacent = 1 } + /^# 11 "/ { getline; if( $0 == "int visible;" ) visible = 1 } + END { exit adjacent && visible ? 0 : 1 } +' "$base/include.out" + +# The threshold is based on source position, not on why the lines are +# invisible. A long run of comments alone is compacted in exactly the same +# way as directives or inactive conditional text. +cat > "$base/comments.c" <<'EOT' +int before; +// 1 +// 2 +// 3 +// 4 +// 5 +// 6 +// 7 +// 8 +int after; +EOT +"$MCPU_CPP" --no-config "$base/comments.c" -o "$base/comments.out" +grep -F "# 10 \"$base/comments.c\"" "$base/comments.out" >/dev/null + +# __LINE__ observes source coordinates, not the compacted output layout. +cat > "$base/line.c" <<'EOT' +int a = __LINE__; +// 1 +// 2 +// 3 +// 4 +// 5 +// 6 +// 7 +// 8 +int b = __LINE__; +EOT +"$MCPU_CPP" --no-config "$base/line.c" -o "$base/line.out" +grep -Fx 'int a = 1;' "$base/line.out" >/dev/null +grep -Fx 'int b = 10;' "$base/line.out" >/dev/null +grep -F "# 10 \"$base/line.c\"" "$base/line.out" >/dev/null |
