summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
-rw-r--r--.gitignore67
-rw-r--r--AUTHORS4
-rw-r--r--ChangeLog381
-rw-r--r--Makefile.am20
-rw-r--r--NEWS748
-rw-r--r--README480
-rw-r--r--acsite.m4797
-rwxr-xr-xauto-clean42
-rwxr-xr-xbootstrap106
-rw-r--r--configure.ac190
-rw-r--r--doc/Makefile.am3
-rw-r--r--doc/mcpu-cpp-en.md2169
-rw-r--r--doc/mcpu-cpp-ru.md2167
-rw-r--r--etc/Makefile.am4
-rw-r--r--etc/mcpu-cpp.conf.in41
-rw-r--r--m4/README4
-rw-r--r--man/Makefile.am6
-rw-r--r--man/mcpu-cpp.1432
-rw-r--r--man/ru/Makefile.am8
-rw-r--r--man/ru/mcpu-cpp.1428
-rw-r--r--src/Makefile.am64
-rw-r--r--src/defs.h42
-rw-r--r--src/main.c412
-rw-r--r--src/mcpp-config-file.c512
-rw-r--r--src/mcpp-config-file.h30
-rw-r--r--src/mcpp-diagnostic.c28
-rw-r--r--src/mcpp-diagnostic.h11
-rw-r--r--src/mcpp-expr.h10
-rw-r--r--src/mcpp-expr.zubr723
-rw-r--r--src/mcpp-expression.c178
-rw-r--r--src/mcpp-expression.h15
-rw-r--r--src/mcpp-include-path.c467
-rw-r--r--src/mcpp-include-path.h61
-rw-r--r--src/mcpp-language.c108
-rw-r--r--src/mcpp-language.h24
-rw-r--r--src/mcpp-lexer.c371
-rw-r--r--src/mcpp-lexer.h38
-rw-r--r--src/mcpp-lib.c3443
-rw-r--r--src/mcpp-lib.h65
-rw-r--r--src/mcpp-macro.c2878
-rw-r--r--src/mcpp-macro.h126
-rw-r--r--src/mcpp-options.c365
-rw-r--r--src/mcpp-options.h74
-rw-r--r--src/mcpp-predefined.c627
-rw-r--r--src/mcpp-predefined.h12
-rw-r--r--src/mcpp-runtime.c268
-rw-r--r--src/mcpp-runtime.h19
-rw-r--r--src/mcpp-semantic.c349
-rw-r--r--src/mcpp-semantic.h67
-rw-r--r--src/mcpp-source.c332
-rw-r--r--src/mcpp-source.h35
-rw-r--r--src/mcpp-text.c237
-rw-r--r--src/mcpp-text.h27
-rw-r--r--tests/Makefile.am101
-rw-r--r--tests/data/base-path-main.c1
-rw-r--r--tests/data/common-precedence/same.h1
-rw-r--r--tests/data/config-main.c2
-rw-r--r--tests/data/config.conf2
-rw-r--r--tests/data/include-main.c6
-rw-r--r--tests/data/include/bridge.h4
-rw-r--r--tests/data/include/one.h1
-rw-r--r--tests/data/lang-as-precedence/same.h1
-rw-r--r--tests/data/lang-as/lang.h1
-rw-r--r--tests/data/lang-base/base.h1
-rw-r--r--tests/data/lang-base/diff/explicit.h1
-rw-r--r--tests/data/lang-diff/lang.h1
-rw-r--r--tests/data/lang-path-main.c6
-rw-r--r--tests/data/lang.c9
-rw-r--r--tests/data/nonucs2.c1
-rw-r--r--tests/data/order-main.c1
-rw-r--r--tests/data/system/order.h1
-rw-r--r--tests/data/system/system.h1
-rw-r--r--tests/data/unbalanced.c2
-rw-r--r--tests/data/user/order.h1
-rw-r--r--tests/data/utf8.c2
-rw-r--r--tests/mcpp-options-test.c24
-rwxr-xr-xtests/t0001-utf8.sh6
-rwxr-xr-xtests/t0002-lang.sh8
-rwxr-xr-xtests/t0003-include.sh11
-rwxr-xr-xtests/t0004-config.sh12
-rwxr-xr-xtests/t0005-errors.sh22
-rwxr-xr-xtests/t0006-path-order.sh13
-rwxr-xr-xtests/t0007-language-path.sh12
-rwxr-xr-xtests/t0008-comments.sh41
-rwxr-xr-xtests/t0009-text-scanner.sh37
-rwxr-xr-xtests/t0010-language-names.sh38
-rwxr-xr-xtests/t0011-base-include-path.sh19
-rwxr-xr-xtests/t0012-user-config-path.sh26
-rwxr-xr-xtests/t0013-include-precedence.sh18
-rwxr-xr-xtests/t0014-preprocessing-phases.sh21
-rwxr-xr-xtests/t0015-object-macros.sh27
-rwxr-xr-xtests/t0016-macro-include.sh16
-rwxr-xr-xtests/t0017-macro-errors.sh13
-rwxr-xr-xtests/t0018-include-lexing.sh17
-rwxr-xr-xtests/t0019-recursive-macro.sh13
-rwxr-xr-xtests/t0020-function-macros.sh42
-rwxr-xr-xtests/t0021-function-macro-errors.sh89
-rwxr-xr-xtests/t0022-predefined-macros.sh56
-rwxr-xr-xtests/t0023-predefined-redefine.sh20
-rwxr-xr-xtests/t0024-abi-predefined.sh191
-rwxr-xr-xtests/t0025-dump-macros.sh99
-rwxr-xr-xtests/t0026-dump-config.sh22
-rwxr-xr-xtests/t0027-system-language-path.sh44
-rwxr-xr-xtests/t0028-predefined-ranges.sh95
-rwxr-xr-xtests/t0029-predefined-abi-names.sh38
-rwxr-xr-xtests/t0030-integer-decimal-digits.sh33
-rwxr-xr-xtests/t0031-stringification.sh50
-rwxr-xr-xtests/t0032-command-line.sh119
-rwxr-xr-xtests/t0033-dump-definitions.sh51
-rwxr-xr-xtests/t0034-conditionals.sh136
-rwxr-xr-xtests/t0035-line-control.sh41
-rwxr-xr-xtests/t0036-lang-string.sh120
-rwxr-xr-xtests/t0037-token-concatenation.sh122
-rwxr-xr-xtests/t0038-ucs2-identifiers.sh122
-rwxr-xr-xtests/t0039-command-line-macros.sh174
-rwxr-xr-xtests/t0040-zubr-expression.sh79
-rwxr-xr-xtests/t0041-integer-width-suffix.sh95
-rwxr-xr-xtests/t0042-diagnostics.sh68
-rwxr-xr-xtests/t0043-include-next.sh87
-rwxr-xr-xtests/t0044-configured-search-order.sh114
-rwxr-xr-xtests/t0045-search-dirs-verbose.sh89
-rwxr-xr-xtests/t0046-pragma-once.sh67
-rwxr-xr-xtests/t0047-dependencies.sh140
-rwxr-xr-xtests/t0048-no-config-defaults.sh73
-rwxr-xr-xtests/t0049-relocatable-root.sh55
-rwxr-xr-xtests/t0050-dependency-side-effects.sh136
-rwxr-xr-xtests/t0051-dump-macro-origin.sh63
-rwxr-xr-xtests/t0052-warning-control.sh108
-rwxr-xr-xtests/t0053-interface-cleanup.sh34
-rwxr-xr-xtests/t0054-inhibit-warnings.sh81
-rwxr-xr-xtests/t0055-forced-files.sh242
-rwxr-xr-xtests/t0056-dependency-targets.sh103
-rwxr-xr-xtests/t0057-missing-generated-dependencies.sh166
-rwxr-xr-xtests/t0058-variadic-macros.sh135
-rwxr-xr-xtests/t0059-va-opt.sh150
-rwxr-xr-xtests/t0060-macro-whitespace.sh72
-rwxr-xr-xtests/t0061-output-line-compaction.sh131
137 files changed, 24608 insertions, 0 deletions
diff --git a/.gitignore b/.gitignore
new file mode 100644
index 0000000..d2ad001
--- /dev/null
+++ b/.gitignore
@@ -0,0 +1,67 @@
+
+# Files generated by ./bootstrap
+
+aclocal.m4
+m4/*.m4
+config.h.in
+configure
+Makefile.in
+src/Makefile.in
+tests/Makefile.in
+doc/Makefile.in
+man/Makefile.in
+man/ru/Makefile.in
+etc/Makefile.in
+compile
+config.guess
+config.sub
+depcomp
+install-sh
+missing
+test-driver
+src/mcpp-expr.c
+
+
+# Autotools/cache/configure state:
+
+autom4te.cache
+config.h
+config.log
+config.status
+stamp-h1
+Makefile
+src/Makefile
+tests/Makefile
+doc/Makefile
+man/Makefile
+man/ru/Makefile
+etc/Makefile
+etc/mcpu-cpp.conf
+
+
+# Build products:
+
+.deps
+src/.deps
+tests/.deps
+src/mcpu-cpp
+tests/mcpp-options-test
+tests/*.log
+tests/*.trs
+tests/test-suite.log
+z.output
+
+
+# Distribution products:
+
+mcpu-cpp-*.tar.gz
+mcpu-cpp-*.tar.xz
+
+
+# Editor/backup files:
+
+*~
+*.swp
+*.swo
+
+*.o
diff --git a/AUTHORS b/AUTHORS
new file mode 100644
index 0000000..9e9895f
--- /dev/null
+++ b/AUTHORS
@@ -0,0 +1,4 @@
+
+Authors of mcpu-cpp (in chronological order of initial contribution)
+
+Andrey V.Kosteltsev Main author
diff --git a/ChangeLog b/ChangeLog
new file mode 100644
index 0000000..c902a5c
--- /dev/null
+++ b/ChangeLog
@@ -0,0 +1,381 @@
+2026-10-01 Andrey V. Kosteltsev
+
+ * mcpu-cpp 1.0.2: implement LibMPU, LibMPUIO congiguring check
+ using libmpu.m4, libmpuio.m4.
+
+2026-09-30 Andrey V. Kosteltsev
+
+ * mcpu-cpp 1.0.1: integrate English and Russian mcpu-cpp(1) manual
+ pages into the Automake build and install them through the standard
+ system man directory.
+
+2026-09-30 Andrey V. Kosteltsev
+
+ * mcpu-cpp 1.0.0: first public release of the preprocessor.
+
+2026-09-30 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.51: remove Russian preprocessing-directive aliases while
+ retaining full Unicode identifier/text support. Update the synchronized
+ manuals with a GNU-compatible-features section and remove predecessor
+ references. Adjust only alias-related and version-sensitive tests.
+
+2026-09-29 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.50: replace the single normative manual with synchronized
+ Russian and English editions, update the manual to the complete 0.0.49
+ feature set, add a LibMPU/LibMPUIO-style developer bootstrap that
+ regenerates the ZUBR parser and Autotools files, add a .gitignore for
+ reproducible/generated Git-tree files, and keep release archives and the
+ existing ordinary build model self-contained and unchanged.
+
+2026-09-29 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.49: compact invisible source-line gaps using GNU CPP's
+ 0..7-newline versus 8+-linemarker policy while preserving logical source
+ coordinates, include enter/return markers, __LINE__, diagnostics and #line.
+
+2026-09-29 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.48: normalize whitespace belonging to macro replacement lists
+ without reformatting ordinary source text or actual macro arguments; keep
+ literal contents and preprocessing-token boundaries intact.
+
+2026-09-29 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.47: implement standard __VA_OPT__(pp-tokens) with expanded
+ variadic-emptiness semantics, balanced parentheses, #/##/placemarker/rescan
+ integration, and explicitly omit historical GNU comma swallowing.
+
+2026-09-29 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.46: add C99-style variadic function-like macros with ...
+ and __VA_ARGS__; integrate the variable argument with ordinary prescan,
+ stringification, token concatenation, placemarkers, rescanning and macro
+ dumps; intentionally leave GNU named args... and __VA_OPT__ unsupported,
+ document the contract and add dedicated regression coverage.
+
+2026-09-29 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.45: implement GNU-like -MG for -M/-MM; integrate missing
+ generated headers and forced files with the ordered dependency registry
+ through a separate unresolved textual identity domain, preserve physical
+ st_dev/st_ino deduplication for real files and GNU-like user/system
+ filtering, document the model and add dedicated regression coverage.
+
+2026-09-29 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.44: implement GNU-like -MT/-MQ dependency targets with
+ repeated and attached forms, exact -MT output, Make-quoted -MQ output and
+ multiple targets per rule; document the 0.0.43 -include/-imacros forced-
+ file contract and the new dependency-target behavior in doc/mcpu-cpp.md,
+ and add dedicated regression coverage.
+
+2026-09-29 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.42: implement -w as the global warning-output gate; suppress
+ every warning class before -Werror promotion regardless of command-line
+ order, leave real errors unaffected, expose -w in --help, document the
+ warning policy, and add regression coverage for all existing warning
+ sources.
+
+2026-09-29 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.41: clean the public command-line interface, remove
+ obsolete compatibility options, document --object-suffix, add selected
+ Russian directive aliases (#определить, #строка, #включить,
+ #включить_следующий, #отменить, #управление, #язык), rename the product
+ description to "MCPU languages preprocessor", and remove predecessor-
+ oriented wording from comments and documentation.
+
+2026-09-29 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.40: implement warning control with -Wcomment/-Wcomments,
+ -Wno-comment/-Wno-comments, -Wall, -Werror and -Wno-error; preserve
+ specific-option priority over -Wall, promote every emitted warning via
+ one diagnostic policy, add comment-lexing diagnostics and regression
+ coverage, and document the warning model as a separate normative section.
+
+2026-09-29 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.39: split final macro dumping by macro origin; make -dM
+ report only source/include and command-line macros, implement -dMP as
+ predefined macros first followed by ordinary macros, preserve #undef
+ final-state semantics and deterministic per-group ordering, leave -dD
+ unchanged, and add documentation/regression coverage.
+
+2026-09-29 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.38: add -MD/-MMD side-effect dependency generation and
+ -MF dependency-output selection; derive default .d names from the input
+ basename or ordinary -o output, preserve the established -M/-MM graph
+ and system classification, and add documentation/regression coverage.
+
+2026-09-29 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.37: replace the versioned, compile-time-bound installation
+ root with a relocatable `$libdir/mcpu` ecosystem tree; derive the runtime
+ root from the physical executable path, derive the packaged config and
+ default system include tree from that root, retain configuration overrides,
+ add physical relocation regression coverage, and document relocatability as
+ a general MCPU ecosystem principle for cpp/as/ld/run and future libraries.
+
+2026-09-29 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.36: fix --no-config so it suppresses only configuration-file
+ reads while retaining the compiled MCPU_CPP_SYSTEM_INCLUDE_PATH default;
+ keep -nostdinc as the mechanism that removes the standard-system include
+ tree; add regression coverage for -dsearch-dirs, -dconfig, -v and explicit
+ include classes, and document the corrected configuration contract.
+
+2026-09-29 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.35: implement basic -M/-MM dependency generation directly
+ from the existing include pipeline; classify system dependencies by real
+ include provenance and propagate system context transitively; deduplicate
+ physical files by device/inode identity; keep logical #line names out of
+ make rules; add documentation and regression coverage.
+
+2026-09-28 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.33: align configure output with the LibMPU/LibMPUIO/ZUBR
+ package family by adding the common headline and AC_MSG_CFG_PART section
+ headings while preserving the detailed final summary; normalize every
+ textual occurrence of the author given name to Andrey.
+
+2026-09-28 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.32: make -v report one effective include-configuration
+ snapshot instead of logging every assignment from every configuration
+ file; add -dsearch-dirs and regression coverage for semantic search-order
+ dumping, compiled system defaults and -nostdinc; document both interfaces.
+
+2026-09-28 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.31: introduce versioned MCPU installation under
+ $libdir/mcpu-0.0.31, install a public $bindir/mcpu-cpp symlink, move
+ the packaged configuration into the versioned tree, treat /etc/mcpu as
+ an optional administrator override, keep the unversioned per-user config
+ at $HOME/.mcpu/mcpu-cpp.conf with highest priority, and restore
+ MCPU_CPP_SYSTEM_INCLUDE_PATH as a completely replaceable system-header
+ root. Document the configuration hierarchy, sandbox use case and
+ #include_next interaction as normative contracts.
+
+2026-09-28 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.30: make command-line include classes override configured
+ defaults and define the complete normative search chain. Rename the
+ configured system root to MCPU_CPP_SYSTEM_INCLUDE_DIR, derive fixed
+ language subdirectories internally, allow an empty root to disable the
+ configured system tree, keep AFTER language-independent, extend
+ #include_next across every class, add full-order regression coverage and
+ document the wrapper-header/search contract in a dedicated section.
+
+2026-09-28 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.29: implement GNU-style #include_next. Preserve the
+ physical include search entry that supplied each nested header and resume
+ searching strictly after it across user, explicit-system, configured-system
+ and idirafter classes. #include_next does not re-search the current source
+ directory and treats quoted/angle operands identically; macro-expanded
+ operands remain supported. Add dedicated wrapper-header regression tests.
+
+2026-09-28 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.28: implement #error/#warning diagnostics with logical
+ source locations, no macro expansion of diagnostic text, whitespace
+ normalization outside quoted tokens, conditional suppression and the
+ historical Russian aliases #ошибка/#предупреждение. Add regression
+ coverage and document the diagnostic-directive contract.
+
+2026-09-28 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.27: implement zNNN/ZNNN integer-width suffixes with an
+ optional U/u in #if expressions. Parse NNN as decimal, normalize valid
+ 8/16/32/64-bit literals by sign or zero extension to the fixed 64-bit
+ evaluator, reject widths above 64, warn and ignore invalid smaller widths,
+ reject C L/LL suffixes and malformed suffix continuations. Remove the
+ obsolete assertion -A action and -dMA modifier. Expand the manual with
+ the normative conversion model and add dedicated regressions.
+
+2026-09-28 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.26: replace the temporary hand-written #if parser with
+ mcpp-expr.zubr generated by ZUBR 4.1.0; retain the current defined()/
+ macro-expansion pipeline, add an UCS-2 lexer using LibMPU iatoui() for
+ numeric conversion, move fixed 64-bit expression semantics into
+ mcpp-semantic.c/h, preserve short-circuit and historical shift behavior,
+ and add dedicated regression coverage.
+
+2026-09-28 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.25: allow `$` as a preprocessing-identifier continuation
+ character but never as identifier start; parse -D macro declarators
+ separately from replacement values so invalid suffixes are silently
+ discarded instead of entering the replacement list; apply matching -U
+ prefix handling; document shell quoting and add regressions.
+
+2026-09-28 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.24: fix -dD provenance for command-line macros. Track
+ builtin/source/command-line macro origin and emit # 0 "<command-line>"
+ for definitions installed through -D while preserving # 0 "<built-in>"
+ for predefined/builtin definitions. Add regression coverage.
+
+2026-09-28 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.23: implement -D/-U command-line macros. Decode only
+ their payloads from UTF-8 to strict UCS-2, reuse the ordinary #define/
+ #undef parser and macro engine, preserve action order after builtin/
+ predefined installation, support Unicode XID identifiers and function-
+ like macros, reject invalid UTF-8/non-UCS-2 payloads, and add dedicated
+ regression coverage.
+
+2026-09-28 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.22: replace the ASCII-only preprocessing-identifier
+ predicates with LibMPUIO 1.0.4 Unicode 18.0.0 XID_Start/XID_Continue
+ classification. Keep underscore as an explicit language identifier
+ character, preserve case-sensitive macro names, require the UCS-2 ctype
+ API at configure time, and add regression coverage for Cyrillic, Latin,
+ Greek, combining marks, non-ASCII digits, macro parameters, #if/defined,
+ ## rescanning.
+
+2026-09-27 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.21: implement GNU-compatible ## token concatenation for
+ object-like and function-like macros; use raw arguments next to ##, keep
+ empty arguments as placemarkers, validate the pasted preprocessing token,
+ rescan the resulting replacement list, preserve # stringification
+ interaction, diagnose invalid/end-position uses, and add regressions.
+
+2026-09-27 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.20: keep the quoted #lang implementation contract unchanged,
+ strengthen its case-insensitive regression coverage, and synchronize every
+ package-version regression expectation with the release version.
+
+2026-09-27 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.19: require a quoted #lang language name; validate it
+ case-insensitively against the internal language table only, reject empty
+ or whitespace-containing names and multi-physical-line forms, preserve the
+ original spelling inside quotes, normalize external whitespace in output,
+ and add dedicated regression coverage.
+
+2026-09-27 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.18: implement conditional-compilation control flow;
+ consume #if/#ifdef/#ifndef/#elif/#else/#endif instead of emitting them,
+ skip inactive branches, evaluate #if expressions with defined and the
+ historical operator precedence. Restore GNU-compatible line control:
+ generated line information uses # N "file" linemarkers, include entry and
+ return use flags 1/2, and input #line updates __LINE__/__FILE__ instead of
+ passing through to output. Add dedicated regression coverage.
+
+2026-09-27 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.17: finish comment whitespace cleanup. Remove blanks left
+ before end-of-line comments, including multi-line block comments, while
+ preserving the separator required by comments between adjacent tokens.
+
+2026-09-27 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.16: implement -dD output; keep source #define directives
+ and emit predefined definitions with built-in markers; remove whitespace
+ left on comment-only lines; defer the UCS-2 range check until after
+ comment removal and add dedicated regressions.
+
+2026-09-25 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.13: implement function-macro stringification (#);
+ preserve raw arguments for stringification, lazily expand ordinary
+ argument occurrences, handle empty actual arguments correctly, keep ##
+ rejected, and add dedicated stringification/dump/error regressions.
+
+2026-09-25 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.12: correct INT/UINT DECIMAL_DIG semantics by computing
+ decimal digits from the known language type width, excluding signs and
+ terminating NUL characters; add exact full-range regression coverage.
+
+2026-09-25 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.11: publish integer decimal-digit metadata for every
+ LibMPU integer width and Real decimal/mantissa precision for every Real
+ width through MPU_REAL_IO_LIMIT; replace FLOAT_WORD_ORDER with explicit
+ MCPU byte/word-order macros, add maximum-width capability macros, and
+ rename SIZEOF_PTRDIFF_T to SIZEOF_PTRDIFF.
+
+2026-09-25 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.10: move configured size/ssize metadata fully into the
+ MCPU namespace; use char8/char16-native SIZEOF names; add Real exponent
+ storage size and maximum textual character-count metadata for every
+ configured Real width through MPU_REAL_IO_LIMIT; extend -dM regressions.
+
+2026-09-25 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.9: correct ptrdiff to signed int64; add the MCPU ssize
+ profile; extend TYPE/WIDTH/SIZEOF metadata through NB_I_MAX * 8 for
+ integers and MPU_REAL_IO_LIMIT for Real/Complex; define Complex WIDTH
+ as the language parameter while keeping double storage size; add
+ exhaustive predefined-range regression coverage.
+
+2026-09-25 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.8: refine the predefined MCPU ABI contract: rename the
+ preprocessor/version and LibMPU-profile macros, use char8/char16,
+ cap textual numeric predefines at 256 bits, add LibMPU-derived decimal
+ precision macros, fix pointer/ptrdiff width at 64 bits, and derive
+ active-language subdirectories from configured system include roots.
+
+2026-09-25 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.7: establish the first serious predefined ABI/environment
+ layer using the LibMPU/LibMPUIO configure model and GNU GCC probes;
+ add MCPU/fixed-width integer/Real/Complex/type-size/endian predefines,
+ lazy LibMPU-generated Real limits, and the -dM/-dconfig inspection
+ options.
+
+2026-09-25 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.6: implement the neutral predefined-macro mechanism: __FILE__,
+ __LINE__, __BASE_FILE__, __INCLUDE_LEVEL__, __DATE__, __TIME__ and __VERSION__;
+ preserve include nesting, fixed startup timestamp and non-rescanned special
+ expansion semantics.
+
+2026-09-25 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.5: move project-owned M4 logic to acsite.m4, reserve m4/
+ for external macros, document the future explicit `make parser` Bison
+ workflow, and implement function-like macro definitions/calls and argument
+ expansion.
+
+2026-09-25 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.4: implement the first macro-engine layer, add
+ backslash-newline/comment preprocessing phases, object-like #define/#undef,
+ recursive object-macro expansion and computed #include;
+ add doc/mcpu-cpp.md as the implementation-order manual.
+
+2026-09-25 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.3: make MCPU_CPP_INCLUDE_PATH a common include root
+ visible in every language state, keep switched-language include paths
+ as additional higher-priority configured user paths, and move the
+ configuration hierarchy to /etc/mcpu/mcpu-cpp.conf and
+ $HOME/.mcpu/mcpu-cpp.conf.
+
+2026-09-25 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.2: restore LANG_0 as the initial base language state, replace
+ historical vasm with as, remove the temporary C-specific include path,
+ define MCPU_CPP_INCLUDE_PATH as the include root of language 0, tighten
+ the canonical #lang names, and add
+ the public `make tests` target.
+
+2026-09-24 Andrey V. Kosteltsev
+
+ * mcpu-cpp 0.0.1: initial implementation based on the architecture and
+ semantics required by the MCPU project with a UTF-8/UCS-2 text model.
diff --git a/Makefile.am b/Makefile.am
new file mode 100644
index 0000000..6b4ac3b
--- /dev/null
+++ b/Makefile.am
@@ -0,0 +1,20 @@
+
+ACLOCAL_AMFLAGS = -I m4
+
+SUBDIRS = src tests doc man etc
+
+EXTRA_DIST = AUTHORS \
+ LICENSE \
+ README \
+ NEWS \
+ ChangeLog \
+ acsite.m4 \
+ bootstrap \
+ m4/README
+
+.PHONY: tests
+tests: all
+ $(MAKE) $(AM_MAKEFLAGS) -C tests check-TESTS
+
+distclean-local:
+ -rm -rf $(top_srcdir)/autom4te.cache
diff --git a/NEWS b/NEWS
new file mode 100644
index 0000000..c445f6d
--- /dev/null
+++ b/NEWS
@@ -0,0 +1,748 @@
+mcpu-cpp 1.0.2
+===============
+
+Implemented LibMPU, LibMPUIO congiguring check using libmpu.m4, libmpuio.m4.
+
+
+mcpu-cpp 1.0.1
+===============
+
+Integrate the mcpu-cpp(1) manual pages.
+
+* Add English man/mcpu-cpp.1 and Russian man/ru/mcpu-cpp.1.
+* Install the pages through Automake's standard man directory hierarchy.
+
+mcpu-cpp 1.0.0
+===============
+
+Remove Russian preprocessing-directive aliases and prepare the public interface
+for another verification cycle before 1.0.0.
+
+* Accept only canonical English preprocessing-directive names; Unicode support
+ for identifiers and ordinary source text is unchanged.
+* Remove Russian-alias regression cases without changing the underlying macro,
+ conditional, include, diagnostic, language-stack, or output engines.
+* Retitle manual section 10 as "GNU-compatible features" and summarize the
+ documented behaviors intentionally aligned with GNU CPP.
+* Remove references to the predecessor preprocessor from both normative manuals.
+
+mcpu-cpp 0.0.50
+===============
+
+Refresh and bilingualize the normative manual and add a developer bootstrap.
+
+* Replace the single doc/mcpu-cpp.md with synchronized Russian and English
+ manuals: doc/mcpu-cpp-ru.md and doc/mcpu-cpp-en.md.
+* Bring the manual up to the 0.0.49 feature set, fix stale future-work text and
+ conditional-expression subsection numbering, and document the current
+ intentionally unsupported legacy features.
+* Add a root bootstrap script, modeled after the LibMPU/LibMPUIO bootstrap
+ workflow, that regenerates the ZUBR parser and Autotools-generated files.
+* Add a root .gitignore describing files that a developer Git tree may omit;
+ keep release archives self-contained and the ordinary release build unchanged.
+
+mcpu-cpp 0.0.49
+===============
+
+Compact invisible output lines while preserving exact source coordinates.
+
+* Track the next visible logical source position instead of materializing every
+ consumed directive/comment/blank source line in .E output.
+* Match GNU CPP's 0..7 newline versus 8+ corrective-linemarker boundary.
+* Preserve include entry/return markers and avoid synthetic progress markers
+ for completely silent included files.
+* Keep __LINE__, diagnostics, #line and logical coordinates independent of the
+ compacted physical output representation.
+
+mcpu-cpp 0.0.48
+===============
+
+Normalize whitespace that belongs to macro replacement lists.
+
+* Canonicalize replacement-list whitespace to one ASCII space after line
+ splicing while preserving ordinary source formatting and actual-argument
+ whitespace.
+* Preserve literal contents and preprocessing-token boundaries around #, ##,
+ __VA_ARGS__, __VA_OPT__ and placemarkers.
+
+mcpu-cpp 0.0.47
+===============
+
+Add the standard __VA_OPT__(pp-tokens) mechanism.
+
+* Decide variadic emptiness after normal macro substitution.
+* Support balanced parentheses, stringification, token concatenation,
+ placemarkers and rescan integration.
+* Intentionally omit GNU named variadics and the historical comma-swallow
+ extension , ## __VA_ARGS__.
+
+mcpu-cpp 0.0.46
+===============
+
+Add C99-style variadic function-like macros.
+
+* Accept ... as the final macro parameter and substitute __VA_ARGS__.
+* Preserve ordinary prescan, raw stringification and token-paste semantics for
+ the complete variadic tail, including empty-tail placemarkers.
+* Keep the old GNU named form args... and __VA_OPT__ outside this release.
+* Preserve variadic definitions in macro dumps and document the exact contract.
+* Add dedicated regression coverage for expansion, #, ##, dumps and errors.
+
+mcpu-cpp 0.0.45
+===============
+
+Add GNU-like missing-generated dependency handling.
+
+* Implement -MG for dependency-only -M/-MM modes.
+* Record unresolved includes in the existing ordered dependency registry while
+ keeping their textual identity separate from physical st_dev/st_ino identity.
+* Preserve GNU-like user/system filtering for missing quoted/angle includes and
+ system-header context; tolerate missing -include/-imacros operands under -MG.
+* Keep -MG invalid with -MD/-MMD or without -M/-MM.
+* Document the unresolved-dependency model and add dedicated regression coverage.
+
+mcpu-cpp 0.0.44
+===============
+
+Add explicit dependency targets and complete the forced-file documentation.
+
+* Implement GNU-like -MT TARGET and -MQ TARGET, including attached forms.
+* Allow repeated -MT/-MQ options and multiple targets in one dependency rule.
+* Keep -MT literal while -MQ applies Make quoting; suppress the default target
+ whenever an explicit target is present.
+* Document -include/-imacros and -MT/-MQ in doc/mcpu-cpp.md.
+* Add dedicated dependency-target regression coverage.
+
+mcpu-cpp 0.0.41
+===============
+
+Clean the public interface and formalize selected Russian directive aliases.
+
+* Remove obsolete compatibility options from the command-line parser so they
+ are diagnosed as unknown options instead of remaining dormant handlers.
+* Keep -E as the intentional compiler-driver compatibility no-op.
+* Document --object-suffix in --help and require its argument.
+* Add Russian aliases for define, line, include, include_next, undef, pragma
+ and lang; normalize non-once #управление output to #pragma.
+* Rename the product description to "MCPU languages preprocessor" and remove
+ predecessor-oriented wording from source comments and documentation.
+* Add interface-cleanup and Russian-alias regression coverage.
+
+mcpu-cpp 0.0.40
+===============
+
+Add explicit warning-control semantics.
+
+* Implement -Wcomment/-Wcomments and negative forms for nested block-comment
+ openers and backslash-newline continuation of // comments.
+* Make -Wall enable the optional comment-warning class while preserving the
+ priority of an explicit -Wno-comment regardless of option order.
+* Implement -Werror/-Wno-error for every warning actually emitted by mcpu-cpp,
+ without making -Werror enable optional warning classes.
+* Route #warning, zNNN width warnings, macro redefinition and invalid ## paste
+ through the common warning policy.
+* Add dedicated regression coverage and a normative documentation section.
+
+mcpu-cpp 0.0.39
+===============
+
+Split final macro dumping into ordinary and predefined views.
+
+* Make -dM dump only source/include and command-line macros.
+* Implement -dMP as predefined macros first, followed by ordinary macros.
+* Preserve deterministic name sorting inside each group and final #undef state.
+* Keep context-dependent special macros out of the static dump.
+* Leave -dD semantics unchanged.
+* Add dedicated regression coverage and document the macro-origin contract.
+
+mcpu-cpp 0.0.38
+===============
+
+Extend dependency generation with side-effect dependency files.
+
+* Add GNU-like -MD and -MMD without suppressing normal preprocessing output.
+* Add -MF FILE, including attached -MFfile spelling and -MF - for stdout.
+* Derive the default .d filename from the input basename, or from ordinary -o
+ output when present.
+* Reuse the existing -M/-MM dependency graph and system-header classification.
+* Add dedicated regression coverage and document the dependency-output rules.
+
+mcpu-cpp 0.0.37
+===============
+
+Make the MCPU installation tree relocatable and shared by the future toolchain.
+
+* Install under `$libdir/mcpu` rather than `$libdir/mcpu-VERSION`.
+* Resolve the real executable at run time and derive `<root>/include` and
+ `<root>/etc/mcpu-cpp.conf` from the physical MCPU root.
+* Keep `/etc/mcpu/mcpu-cpp.conf` and `$HOME/.mcpu/mcpu-cpp.conf` as higher
+ priority overrides, including full replacement of MCPU_CPP_SYSTEM_INCLUDE_PATH.
+* Remove the absolute installation default from the packaged config and runtime
+ binary contract.
+* Add a regression that physically moves a complete MCPU tree and verifies the
+ new root, include search, packaged config and symlink invocation.
+* Document one relocatable `bin/etc/include/lib` root as a general MCPU
+ ecosystem principle for mcpu-cpp, mcpu-as, mcpu-ld and mcpu-run.
+
+mcpu-cpp 0.0.36
+===============
+
+Correct --no-config so it suppresses configuration-file reads without
+discarding the compiled installation defaults.
+
+* Keep the compiled MCPU_CPP_SYSTEM_INCLUDE_PATH active under --no-config.
+* Make -dsearch-dirs, -dconfig and -v expose that built-in default even when
+ every configuration file is ignored.
+* Preserve explicit -I/-isystem/-idirafter ordering around the compiled system
+ tree and keep -nostdinc as the switch that suppresses standard-system paths.
+* Add dedicated regression coverage and document the distinction between
+ compiled defaults, configuration files and -nostdinc.
+
+mcpu-cpp 0.0.35
+===============
+
+Implement the first dependency-generation stage.
+
+* Add -M make-rule output including system headers.
+* Add -MM user-dependency output excluding system headers and descendants
+ reachable only from system-header contexts.
+* Collect dependencies directly from the existing include/#include_next
+ pipeline and deduplicate physical files by device/inode identity.
+* Keep #line logical filenames out of dependency output and preserve a file in
+ -MM when it is also reached through a user include context.
+* Add dedicated dependency-generation regression coverage and documentation.
+
+mcpu-cpp 0.0.33
+===============
+
+Align the configure-time console presentation with the LibMPU/LibMPUIO/ZUBR
+package family and normalize the author's given name spelling.
+
+* Add the shared-style mcpu-cpp configure headline and AC_MSG_CFG_PART section
+ headings to make configure output follow the same visual structure as the
+ other MCPU/LibMPU projects.
+* Keep the existing detailed configuration summary as its own final section.
+* Normalize every textual occurrence of the author given name to `Andrey`
+ throughout the maintained project tree.
+
+mcpu-cpp 0.0.32
+===============
+
+Polish include-path diagnostics after the 0.0.31 installation/search contract.
+
+* Make -v print each effective include configuration variable once, after all
+ configuration layers have been resolved.
+* Add -dsearch-dirs to print effective include search directories in semantic
+ priority order and stop without preprocessing.
+* Document the distinction between effective configuration values and the
+ expanded global search-directory chain.
+
+mcpu-cpp 0.0.31
+===============
+
+Introduce the versioned MCPU installation/configuration contract.
+
+* Install mcpu-cpp under `$libdir/mcpu-0.0.31/bin` and its packaged
+ configuration under `$libdir/mcpu-0.0.31/etc`.
+* Install `$bindir/mcpu-cpp` as a symbolic link to the versioned executable.
+* Do not install or create `/etc/mcpu`; treat `/etc/mcpu/mcpu-cpp.conf` only as
+ an optional distributor/administrator override.
+* Load configuration in increasing priority: compiled versioned defaults,
+ versioned packaged config, optional `/etc` override, then the unversioned
+ `$HOME/.mcpu/mcpu-cpp.conf` user override.
+* Restore `MCPU_CPP_SYSTEM_INCLUDE_PATH` as the replaceable effective system
+ include root, defaulting to `$libdir/mcpu-0.0.31/include`. Derive
+ `<root>/<lang>` internally; no per-language system variables exist.
+* Let a higher-priority config replace the system tree completely or disable it
+ with an empty value.
+* Keep the normative include and #include_next search order established in
+ 0.0.30, now using the effective `MCPU_CPP_SYSTEM_INCLUDE_PATH`.
+* Document the versioned tree, configuration precedence, sandbox workflow and
+ wrapper-header rationale as normative interfaces.
+
+mcpu-cpp 0.0.30
+===============
+
+Define and enforce the normative include-search contract discovered by manual
+wrapper-header testing.
+
+* Give explicit command-line -I, -isystem and -idirafter directories priority
+ over persistent configuration classes.
+* Search configured user language paths before MCPU_CPP_INCLUDE_PATH.
+* Replace MCPU_CPP_SYSTEM_INCLUDE_PATH with the singular
+ MCPU_CPP_SYSTEM_INCLUDE_DIR system-tree root. Derive the fixed <root>/<lang>
+ directory internally and then search <root>; no per-language system config
+ variables exist.
+* Allow an empty MCPU_CPP_SYSTEM_INCLUDE_DIR to disable the configured system
+ tree. -nostdinc suppresses the same configured system classes for one run
+ without suppressing explicit -isystem.
+* Keep explicit -idirafter before MCPU_CPP_AFTER_INCLUDE_PATH and never derive
+ language-specific AFTER directories.
+* Make #include_next traverse this exact effective chain by physical provenance.
+* Add a dedicated manual-style regression that places the same header in all
+ eight search classes and verifies the complete #include_next order.
+* Add a separate normative documentation section explaining the search order,
+ configurable user paths, fixed system subdirectory names and wrapper-header
+ purpose.
+
+mcpu-cpp 0.0.29
+===============
+
+Implement GNU-style #include_next wrapper-header search.
+
+* Track the exact configured include-directory entry that supplied each nested
+ header and continue #include_next strictly after that entry.
+* Preserve the semantic search order across -I, explicit -isystem, configured
+ system directories and -idirafter directories.
+* Do not re-search the current source-file directory for #include_next; the
+ directive intentionally does not distinguish between "file" and <file>.
+* Support macro-expanded #include_next operands through the ordinary include
+ operand parser and macro-expansion path.
+* Keep physical include provenance independent of logical #line filenames.
+* Add regression coverage for the wrapper -> configured-system -> idirafter
+ chain, quoted-source origin, macro operands, inactive branches and failure
+ when no later header exists.
+
+mcpu-cpp 0.0.28
+===============
+
+Implement #error and #warning diagnostics.
+
+* Active #error emits a source-location error diagnostic and stops preprocessing.
+* Active #warning emits a source-location warning and preprocessing continues.
+* Diagnostic directive arguments are not macro-expanded, following the directive contract; whitespace between tokens is folded while quoted text is kept.
+* Diagnostic directives are consumed and never copied to normal or -dD output.
+* Inactive conditional branches suppress both directives completely.
+* Historical Russian aliases #ошибка and #предупреждение are recognized.
+
+mcpu-cpp 0.0.27
+===============
+
+Define the MCPU integer-width suffix contract for conditional expressions and
+remove the obsolete assertion command-line heritage.
+
+* Accept `zNNN`/`ZNNN` and optional trailing `U`/`u` on integer constants.
+ NNN is always decimal, including forms with leading zeroes.
+* `z8`, `z16`, `z32` and `z64` normalize the low N bits by sign extension;
+ an optional U/u selects zero extension. All later expression evaluation is
+ still exactly 64-bit and does not retain the source width.
+* Widths above 64 are errors in conditional directives. Invalid widths up to
+ 64 are ignored with a warning; a trailing U/u remains effective.
+* Remove C-style L/LL integer suffix acceptance from the #if lexer.
+* Diagnose a suffix that runs into an identifier instead of splitting it into
+ unrelated tokens.
+* Remove the obsolete `-A` assertion option, the assertion action kind and the
+ `-dMA` assertion-dump modifier.
+* Document the complete normalization and 64-bit evaluation model.
+* Add regression coverage for width suffixes, sign/zero extension, truncation,
+ 64-bit post-normalization operations, errors and removed assertion options.
+
+mcpu-cpp 0.0.26
+===============
+
+Use the ZUBR grammar architecture for #if expression parsing.
+
+* Add mcpp-expr.zubr with the grammar and UCS-2 lexical analyzer; generated
+ mcpp-expr.c uses the mcpp_zubr_* prefix from ZUBR 4.1.0.
+* Keep the established defined() preprocessing and macro-expansion front end.
+* Move fixed-width #if arithmetic to mcpp-semantic.c/h; expression integers
+ use LibMPU 64-bit types rather than host long/intmax_t.
+* Parse binary/octal/decimal/hexadecimal integer lexemes through LibMPU
+ iatoui() after validating/copying the ASCII subset from UCS-2.
+* Character units are UCS-2 __mpu_uint16_t values; multi-character
+ packing is retained up to the 64-bit expression width.
+* Preserve the defined precedence, negative-shift direction and short-circuit
+ behavior for &&, || and ?:.
+* Ship the generated C source; ZUBR is needed only when the .zubr grammar is
+ regenerated. The obsolete code-page -T option is not used.
+* Add a dedicated ZUBR expression regression.
+
+mcpu-cpp 0.0.25
+===============
+
+Refine preprocessing identifiers and command-line macro declarator parsing.
+
+* Permit `$` after the first preprocessing-identifier character, while keeping
+ `$` forbidden as an identifier start.
+* Apply the same `$` continuation rule throughout macro names, parameters,
+ conditionals, expansion, # stringification and ## rescanning.
+* Parse the -D declarator separately from its replacement value so an invalid
+ tail before '=' is discarded instead of becoming replacement text.
+* Apply the same valid-identifier-prefix rule to -U command-line actions.
+* Document shell quoting explicitly: single quotes already protect `$`; an
+ unquoted or double-quoted `$` must be escaped when literal text is intended.
+* Extend Unicode/command-line macro regressions for `$`, invalid tails, -U and
+ forbidden `$` at identifier start.
+
+mcpu-cpp 0.0.24
+===============
+
+Correct -dD origin markers for command-line macro definitions.
+
+* Macros installed by -D are tagged as command-line definitions internally.
+* -dD emits `# 0 "<command-line>"` before command-line definitions instead of
+ incorrectly reporting them as `# 0 "<built-in>"`.
+* Predefined/builtin macros retain the `# 0 "<built-in>"` marker.
+* Added a regression covering Unicode -D names and both marker classes.
+
+mcpu-cpp 0.0.23
+===============
+
+Implement command-line macro definition and undefinition.
+
+* -DNAME defines NAME as 1; -DNAME=VALUE defines VALUE, including an empty
+ replacement after an explicit '='.
+* Function-like command-line macros use the same parser and macro engine as
+ source #define directives, including # stringification and ## concatenation.
+* -D and -U payloads alone are decoded from UTF-8 to strict UCS-2; filenames,
+ include paths and all other command-line arguments remain byte strings.
+* Unicode command-line macro names use the same XID_Start/XID_Continue rules
+ as source identifiers. Invalid UTF-8 and non-UCS-2 characters are rejected.
+* -D/-U actions are applied in command-line order after predefined/builtin
+ macros are installed, so -U may remove a predefined macro and later -D may
+ define it again.
+* Added dedicated regression coverage for ASCII and Unicode command-line
+ macros, order, function macros, #/##, conditionals and invalid encodings.
+
+mcpu-cpp 0.0.22
+===============
+
+Use LibMPUIO 1.0.4 strict UCS-2 Unicode 18.0.0 XID classification for
+preprocessing identifiers.
+
+* Identifier start is `_` or Unicode XID_Start.
+* Identifier continuation is `_` or Unicode XID_Continue.
+* Macro names and parameters may therefore use UCS-2 letters from supported
+ scripts, combining marks in continuation positions, and non-ASCII decimal
+ digits in continuation positions.
+* The same identifier rules are used by #define/#undef, #ifdef/#ifndef,
+ defined(), ordinary macro expansion, stringification, token concatenation
+ rescanning.
+* Classification is locale-independent and inherits LibMPUIO strict UCS-2
+ semantics; surrogate code units are never identifiers.
+* Configure now verifies that the required LibMPUIO UCS-2 ctype API is present.
+
+mcpu-cpp 0.0.21
+===============
+
+Implement GNU-compatible token concatenation (##).
+
+* ## now pastes preprocessing tokens in both object-like and function-like
+ macro replacement lists.
+* A formal parameter adjacent to ## uses its raw, unexpanded actual argument;
+ the completed replacement list is rescanned afterwards. This preserves the
+ standard two-level CAT/XCAT expansion technique.
+* Empty pasted arguments use placemarker semantics, and multi-token arguments
+ paste only the token adjacent to ##.
+* Pasting can form identifiers, preprocessing numbers, multi-character
+ punctuators and prefixed string/character tokens.
+* An invalid paste is diagnosed and the original two tokens are emitted; ## at
+ either end of a replacement list is rejected when the macro is defined.
+* Stringification (#) and concatenation (##) may be used in the same macro.
+* Added dedicated token-concatenation regression coverage and updated macro
+ error tests.
+
+mcpu-cpp 0.0.20
+===============
+
+Corrective release of the quoted #lang contract introduced in 0.0.19.
+
+* The #lang implementation contract is unchanged: exactly one quoted language
+ name, no whitespace inside the quotes, no escape processing, same-physical-line
+ termination, ASCII case-insensitive validation against the internal language
+ table only, original spelling preserved, and normalized external whitespace.
+* Regression coverage explicitly checks the accepted spellings "diff", "Diff",
+ "DIFF" and "dIfF", as well as all other internal languages.
+* All package-version regression expectations are synchronized with 0.0.20;
+ this fixes the stale-version test failures seen in the first 0.0.19 test run.
+
+mcpu-cpp 0.0.19
+===============
+
+Tighten and normalize the #lang directive contract.
+
+* #lang now requires a quoted language name, for example `#lang "diff"`;
+ the unquoted form is rejected.
+* The quoted text must be one non-empty word without whitespace, must close on
+ the same physical source line, and may be followed only by whitespace.
+* Escape sequences are not interpreted inside the #lang string.
+* Language lookup is ASCII case-insensitive but is restricted to the internal
+ language table: diff, dift, alg, as, avm and ACS.
+* Output preserves the spelling inside the quotes while normalizing external
+ whitespace to exactly `#lang "name"`.
+* Updated existing language tests and added a dedicated #lang string-contract
+ regression.
+
+mcpu-cpp 0.0.18
+===============
+
+Implement conditional-compilation control flow.
+
+* #if, #ifdef, #ifndef, #elif, #else and #endif are consumed by the
+ preprocessor and never copied to normal or -dD output.
+* Inactive branches are skipped without executing definitions, includes or
+ ordinary macro expansion; nested conditionals remain balanced.
+* #if expressions follow the expr.zubr precedence model, including
+ defined, unknown identifiers as zero, integer operators, ?: and short-circuit
+ evaluation.
+* Include files keep independent conditional-stack boundaries and diagnose
+ unbalanced or unterminated groups.
+* Generated source-location information now uses GNU linemarkers (`# N "file"`)
+ instead of emitting `#line`; include entry/return markers carry flags 1 and 2.
+* Input #line directives are consumed, macro-expanded and update __LINE__ and
+ __FILE__; the resulting location is emitted as a GNU linemarker.
+* Added regression coverage for include guards, nested groups, all conditional
+ directives, defined, macro-expanded expressions, malformed groups and line
+ control.
+
+mcpu-cpp 0.0.17
+===============
+
+Finish comment whitespace cleanup.
+
+* Removing a // or /* ... */ comment at the end of a non-empty source line no
+ longer leaves a trailing blank before the newline.
+* The same rule applies when a block comment begins after program text and
+ continues onto following physical lines.
+* Inline comments between tokens still preserve a separating blank where it is
+ required to prevent accidental token concatenation.
+
+mcpu-cpp 0.0.16
+===============
+
+Implement -dD and correct comment/UCS-2 phase behavior.
+
+* -dD preserves normal preprocessed output, emits predefined macro definitions
+ with <built-in> markers, and keeps encountered #define directives.
+* Comment-only lines are left truly empty after // or /* ... */ removal while
+ inline comments still preserve token separation.
+* Valid UTF-8 scalars outside UCS-2 are accepted inside comments and rejected
+ only if they remain in program text after comment removal.
+* Added regression coverage for -dD, comment-only whitespace and non-UCS-2
+ characters inside versus outside comments.
+
+mcpu-cpp 0.0.13
+===============
+
+Implement macro stringification (#).
+
+* Function-like macro replacement lists now support #parameter stringification.
+* Stringification uses the raw, unexpanded actual argument, trims leading and
+ trailing whitespace, folds internal whitespace outside quoted tokens, and
+ escapes quotes/backslashes in the generated string literal.
+* Ordinary argument expansion is now lazy, matching the established raw-vs-expanded
+ argument semantics and preserving nested calls of the same macro.
+* Empty actual arguments are represented correctly; zero-argument macros keep
+ their established behavior.
+* Invalid # uses are diagnosed at definition time. ## remains deliberately
+ rejected for the next separate token-concatenation port.
+* Added regression coverage for raw-vs-expanded stringification, whitespace,
+ quoted text, empty arguments, diff apostrophe semantics, diagnostics and -dM.
+
+mcpu-cpp 0.0.12
+===============
+
+Correct integer DECIMAL_DIG semantics.
+
+* Replaced direct use of LibMPU _int_digs() for predefined integer precision
+ metadata with mcpu-cpp integer-only helpers based on the known type width.
+* __INT<bits>_DECIMAL_DIG__ now counts decimal digits of INT<bits>_MAX only;
+ it excludes both the sign and any terminating NUL.
+* __UINT<bits>_DECIMAL_DIG__ now counts decimal digits of UINT<bits>_MAX only;
+ it excludes any terminating NUL.
+* Added exact regression values for every integer family width from 8 through
+ 65536 bits.
+
+mcpu-cpp 0.0.11
+===============
+
+Complete predefined precision metadata and MCPU byte/word-order naming.
+
+* Added __INT<bits>_DECIMAL_DIG__ and __UINT<bits>_DECIMAL_DIG__ from LibMPU
+ _int_digs() for every configured integer width through NB_I_MAX * 8; MAX
+ values remain capped at 256 bits.
+* Added __REAL<bits>_DECIMAL_DIG__ and __REAL<bits>_MANT_DIG__ for every Real
+ width through MPU_REAL_IO_LIMIT; large numeric MAX/MIN/EPSILON/exponent
+ values remain capped at 256 bits.
+* Replaced __FLOAT_WORD_ORDER__ with __MCPU_BYTE_ORDER__ and
+ __MCPU_WORD_ORDER__; __BYTE_ORDER__ remains as an alias to
+ __MCPU_BYTE_ORDER__.
+* Added __MCPU_INT_MAX_WIDTH__, __MCPU_REAL_MAX_WIDTH__ and
+ __MCPU_COMPLEX_MAX_WIDTH__ capability macros.
+* Renamed __SIZEOF_PTRDIFF_T__ to language-native __SIZEOF_PTRDIFF__.
+* Extended predefined ABI/range regression coverage for the complete integer
+ and Real families.
+
+mcpu-cpp 0.0.10
+===============
+
+Predefined ABI naming cleanup and complete Real text-buffer metadata.
+
+* Replaced the C-style public size family with __MCPU_SIZE_TYPE__, WIDTH,
+ SIZEOF_SIZE and MAX; renamed __MCPU_SIZEOF_SSIZE_T__ to
+ __MCPU_SIZEOF_SSIZE__.
+* Renamed __SIZEOF_CHAR8_T__/__SIZEOF_CHAR16_T__ to the language-native
+ __SIZEOF_CHAR8__/__SIZEOF_CHAR16__.
+* Added __SIZEOF_REAL<bits>_EXP__ from LibMPU _sizeof_exp() for every Real
+ width through MPU_REAL_IO_LIMIT.
+* Added __REAL<bits>_MAX_STRLEN__ from _real_max_string() for every Real width;
+ the value is a character count, not a byte count.
+* Extended -dM regression coverage to verify the new names and the largest
+ configured Real family.
+
+mcpu-cpp 0.0.9
+==============
+
+Complete structural predefined metadata across LibMPU type ranges.
+
+* Corrected MCPU ptrdiff to signed int64 with max 0x7fffffffffffffff.
+* Added the MCPU-specific signed-size family: __MCPU_SSIZE_TYPE__, WIDTH,
+ SIZEOF and MAX from the configured LibMPU __mpu_ssize_t profile.
+* Integer TYPE/WIDTH/SIZEOF families now extend through NB_I_MAX * 8;
+ numeric MAX/DECIMAL_DIG values remain capped at 256 bits.
+* Real/Complex TYPE/WIDTH/SIZEOF families now extend through
+ MPU_REAL_IO_LIMIT; numeric Real characteristics remain capped at 256 bits.
+* Corrected __COMPLEX<bits>_WIDTH__ to the language type parameter <bits>;
+ __SIZEOF_COMPLEX<bits>__ remains twice the corresponding Real storage.
+* Added exhaustive -dM regression coverage across all advertised integer,
+ Real and Complex structural families.
+
+mcpu-cpp 0.0.8
+==============
+
+Predefined ABI refinement and derived system language include paths.
+
+* Renamed __VERSION__ to __MCPU_CPP_VERSION__ and report the mcpu-cpp version.
+* Renamed LibMPU-profile predefines to the __MCPU_* namespace.
+* Character type spellings are now char8 and char16.
+* MCPU pointer and ptrdiff contracts are fixed at 64 bits independently of host;
+ ptrdiff is currently uint64 by ABI decision.
+* Integer/Real textual value constants are emitted only through 256 bits.
+ Wider families retain only type and width macros.
+* Removed computed integer MIN predefines.
+* Added integer DECIMAL_DIG values from _int_digs().
+* Replaced REAL<bits>_DIG with REAL<bits>_DECIMAL_DIG from _real_digs();
+ MANT_DIG comes from _real_mant_digs().
+* Configured system include roots automatically add active-language
+ subdirectories such as /usr/include/mcpu/diff without extra config variables.
+* Added regression coverage for the refined predefined environment and derived
+ system language include search.
+
+mcpu-cpp 0.0.7
+==============
+
+Predefined ABI/environment layer and inspection tools.
+
+* GNU GCC is now an explicit mandatory build compiler.
+* Ported the LibMPU/LibMPUIO GCC type-size, endian and machine-register probes
+ into mcpu-cpp acsite.m4 under project-specific macro names.
+* Configure reads and verifies the installed LibMPU real-I/O/math limits,
+ byte/word order, addressable-unit width, size_t and ptrdiff_t profile.
+* Added _ARCH_MCPU and GCC-compatible byte-order, pointer, size_t/ptrdiff_t and
+ assembler-prefix predefined macros.
+* Added fixed-width int/uint predefined families through 65536 bits.
+* Added Real/Complex type families through 65536 bits; Real numeric limits are
+ generated with LibMPU and real_to_ascii up to MPU_REAL_IO_LIMIT.
+* Large Real numeric predefined values are generated lazily so ordinary
+ preprocessing remains fast.
+* __VERSION__ is now the future high-level compiler version placeholder 0.0.0.
+* Added `-dM` for deterministic macro-table dumps and `-dconfig` for effective
+ configuration dumps; neither requires an input file.
+* Added ABI, dump-macro and dump-config regression tests.
+
+mcpu-cpp 0.0.6
+==============
+
+Implement predefined macros.
+
+* Added the dynamic predefined macros __FILE__, __LINE__, __BASE_FILE__,
+ __INCLUDE_LEVEL__, __DATE__, __TIME__ and __VERSION__.
+* __FILE__/__LINE__/__INCLUDE_LEVEL__ follow nested #include processing;
+ __BASE_FILE__ remains the original translation-unit input file.
+* __DATE__ and __TIME__ are fixed once at preprocessing startup, matching the
+ translation-unit timestamp model.
+* __VERSION__ expands to the mcpu-cpp package version, as preprocessor used its own
+ package version for the corresponding macro.
+* Predefined expansions are emitted without rescanning, following the established
+ special-symbol behavior.
+* Built-ins share the normal macro table and can therefore be undefined or
+ explicitly redefined.
+* K-specific legacy built-ins (__STDK__, __K__, __kxLab__, and related names)
+ are not carried into the MCPU language family.
+
+mcpu-cpp 0.0.5
+==============
+
+Extend the macro engine.
+
+* Moved project-owned Autoconf macros from m4/ to the root acsite.m4; m4/ is
+ reserved for external/vendor macros.
+* Documented the future developer-only Bison workflow: make parser, make tests,
+ then make dist. Normal release builds do not require Bison.
+* Added function-like #define macros with zero or more formal parameters.
+* Added nested argument parsing with defined parenthesis/comma rules.
+* Actual arguments are macro-expanded before ordinary substitution, including
+ nested calls such as min(min(a,b),c).
+* Added diagnostics for duplicate parameters, malformed definitions, incomplete
+ calls and wrong argument counts.
+* # and ## are explicitly rejected until their semantics are implemented, avoiding silent partial preprocessing.
+
+mcpu-cpp 0.0.4
+==============
+
+Implement the first macro-processing layer.
+
+* Added doc/mcpu-cpp.md as a living implementation-order manual as the normative MCPU preprocessing manual.
+* Added the preprocessing phase needed before directive and macro processing:
+ backslash-newline splicing and comment removal with preserved source lines.
+* Added object-like #define and #undef.
+* Added recursive/cascaded object-macro expansion with recursion blocking.
+* Macro names are not expanded inside quoted strings; the historical diff
+ apostrophe rule remains in force.
+* Added computed #include through macro expansion.
+* Preserved literal #include <...> lexical behavior for comment-like text.
+* Began reusing macro regression cases; the self-referential macro
+ case is now part of the mcpu-cpp test suite.
+
+mcpu-cpp 0.0.3
+==============
+
+Include-root and configuration-layout correction.
+
+* MCPU_CPP_INCLUDE_PATH is now a common user include root visible in every
+ language state.
+* Active language-specific include directories are additional search paths
+ and precede the common configured root.
+* Explicit names such as <diff/a.h> can therefore be used through the common
+ root regardless of the current #lang state.
+* The system configuration moved to /etc/mcpu/mcpu-cpp.conf.
+* The per-user configuration moved to $HOME/.mcpu/mcpu-cpp.conf.
+
+mcpu-cpp 0.0.2
+==============
+
+Language-contract cleanup before macro-engine development.
+
+* Establish initial language state 0.
+* Language 0 is active at startup and cannot be named by #lang.
+* Canonical #lang set is: diff, dift, alg, as, avm, ACS.
+* Historical vasm was replaced by as for the real MCPU assembler.
+* MCPU_CPP_VASM_INCLUDE_PATH became MCPU_CPP_AS_INCLUDE_PATH.
+* Removed the temporary MCPU_CPP_C_INCLUDE_PATH. The root
+ MCPU_CPP_INCLUDE_PATH is the main include directory of the unnamed base
+ base MCPU language (state 0) and is selected only in that state.
+* Added the public `make tests` target.
+
+mcpu-cpp 0.0.1
+==============
+
+Initial architectural release.
+
+* New UTF-8 external / UCS-2 internal source model with strict NUL rejection.
+* No legacy code-page machinery.
+* Hand-written preprocessing scanner; flex is not used.
+* Scanner honors quoted strings and the diff apostrophe rule.
+* Source/include stack with #include support.
+* #lang/#endlang language stack defined for the MCPU language stack.
+* Configuration-file and command-line include search paths.
+* Initial language set: c, diff, dift, alg, vasm, avm, ACS.
diff --git a/README b/README
new file mode 100644
index 0000000..247af4e
--- /dev/null
+++ b/README
@@ -0,0 +1,480 @@
+mcpu-cpp
+========
+
+mcpu-cpp is the preprocessor/front-end source manager for MCPU programming
+languages. It is an independent component of the LibMPU/LibMPUIO/LibMCPU
+software platform and provides preprocessing, language selection, include-path
+management, dependency generation and preprocessing diagnostics for the MCPU
+toolchain.
+
+Text model
+----------
+
+External text files are UTF-8. Internally mcpu-cpp uses strict UCS-2 through
+LibMPUIO. UTF-8 scalar values which cannot be represented by UCS-2 are
+rejected. Legacy external code pages are not supported; the external representation is UTF-8.
+
+Lexical analysis
+----------------
+
+mcpu-cpp does not use flex. Its preprocessing scanners are hand-written and
+operate on the internal UCS-2 text model. The `#if` expression grammar and its
+expression lexer are generated by ZUBR 4.1.0.
+
+Parser architecture
+-------------------
+
+The generated `src/mcpp-expr.c` is shipped in release archives, so ordinary
+builds do not require ZUBR. In the developer Git tree the generated parser
+may be omitted; the root `./bootstrap` script regenerates it from
+`src/mcpp-expr.zubr` with ZUBR 4.1.0 before regenerating the Autotools files.
+
+Documentation
+-------------
+
+The normative preprocessing manual is maintained in two synchronized forms:
+
+ doc/mcpu-cpp-en.md English
+ doc/mcpu-cpp-ru.md Russian
+
+Both documents describe the same public contract and are distributed together.
+
+Configuration
+-------------
+
+mcpu-cpp reads UTF-8 configuration files using a simple `NAME = value;`
+syntax. Shell-style `$NAME` and `${NAME}` references are expanded from values
+already defined in the effective configuration and then from the process
+environment.
+
+MCPU uses one relocatable installation tree rather than a versioned directory
+per tool. With `--prefix=/usr --libdir=/usr/lib64`, `make install` creates:
+
+```text
+/usr/lib64/mcpu/
+ bin/mcpu-cpp
+ etc/mcpu-cpp.conf
+ include/
+ lib/
+```
+
+`/usr/bin/mcpu-cpp` is a public symbolic link to
+`../lib64/mcpu/bin/mcpu-cpp`. The absolute `/usr/lib64/mcpu` path is an
+install-time choice only; it is not embedded as the runtime MCPU root.
+
+On Linux, mcpu-cpp resolves its real executable through `/proc/self/exe`, takes
+the parent of the executable directory as the MCPU runtime root, and derives
+`<root>/etc/mcpu-cpp.conf` and `<root>/include` from it. A fallback based on
+`argv[0]`, `PATH`, and `realpath(3)` is used only when `/proc/self/exe` cannot
+be read. Consequently a complete MCPU tree may be copied or moved without
+rebuilding mcpu-cpp.
+
+This is the intended ecosystem-wide layout for future `mcpu-as`, `mcpu-ld`,
+`mcpu-run`, libraries and CRT components as well: tool versions do not define
+separate roots; a coherent MCPU environment is identified by one physical
+runtime tree.
+
+Configuration layers are applied in this order:
+
+```text
+runtime-derived defaults
+<runtime-root>/etc/mcpu-cpp.conf
+/etc/mcpu/mcpu-cpp.conf optional
+$HOME/.mcpu/mcpu-cpp.conf optional, highest priority
+```
+
+The packaged config deliberately does not store an absolute default system
+include path. Before reading any config file mcpu-cpp sets
+`MCPU_CPP_SYSTEM_INCLUDE_PATH=<runtime-root>/include`; any higher-priority
+configuration may replace that value or set it empty.
+
+`--config-file FILE` reads only FILE on top of the runtime-derived defaults.
+`--no-config` reads no configuration files at all but keeps those runtime
+defaults. Use `-nostdinc` when the effective standard-system include tree
+itself must be suppressed for one invocation.
+
+Installation does not create `/etc/mcpu`; that directory is reserved for an
+optional distributor or system-administrator override. The per-user
+configuration is deliberately not versioned.
+
+Recognized path variables are:
+
+ MCPU_CPP_INCLUDE_PATH
+ MCPU_CPP_DIFF_INCLUDE_PATH
+ MCPU_CPP_DIFT_INCLUDE_PATH
+ MCPU_CPP_ALG_INCLUDE_PATH
+ MCPU_CPP_AS_INCLUDE_PATH
+ MCPU_CPP_AVM_INCLUDE_PATH
+ MCPU_CPP_ACS_INCLUDE_PATH
+ MCPU_CPP_SYSTEM_INCLUDE_PATH
+ MCPU_CPP_AFTER_INCLUDE_PATH
+
+User and AFTER path lists use the host PATH separator (`:` on UNIX systems).
+`MCPU_CPP_SYSTEM_INCLUDE_PATH` is different: it is one replaceable root of the
+MCPU system-header tree. Its runtime-derived default is
+`<runtime-root>/include`. When a switched language is active mcpu-cpp
+searches `<root>/<lang>` and then `<root>`. A higher-priority configuration can
+replace the root completely for a developer/tester sandbox, or set it to an
+empty value to disable the configured system tree.
+
+Command line
+------------
+
+ mcpu-cpp [options] [input [output]]
+
+Important options:
+
+ -o FILE write output to FILE instead of stdout
+ -D NAME[=VALUE] define a command-line macro
+ -U NAME undefine a command-line macro
+ -imacros FILE preprocess FILE for macro state; discard its output
+ -include FILE preprocess FILE before the primary input
+ -I DIR, -IDIR add a user include directory
+ -isystem DIR add an explicit system include directory
+ -idirafter DIR add a directory searched after system directories
+ -nostdinc suppress the effective standard-system include tree
+ -dM dump non-predefined macros to stdout
+ -dMP dump predefined macros first, then other macros
+ -dD preserve #define directives in normal output
+ -dconfig dump effective configuration variables to stdout
+ -dsearch-dirs dump effective include search directories and exit
+ -M output make dependencies including system headers
+ -MM output make dependencies excluding system headers
+ -MD write dependencies and keep preprocessing output
+ -MMD like -MD but exclude system headers
+ -MF FILE write dependencies to FILE ('-' means stdout)
+ -MT TARGET set unquoted make dependency target
+ -MQ TARGET set make-quoted dependency target
+ --object-suffix SFX set object suffix used for dependency targets
+ -w suppress all warnings
+ -Wcomment[s] warn about nested /* and multi-line // comments
+ -Wno-comment[s] disable comment warnings even under -Wall
+ -Wall enable all optional warning classes
+ -Werror promote every emitted warning to an error
+ -Wno-error keep emitted warnings as warnings
+ --config-file FILE use only FILE as the configuration file
+ --no-config do not read config files; keep runtime-derived defaults
+ -v, --verbose print configuration and include activity
+ --help print help
+ --version print version
+
+Include search
+--------------
+
+For `#include "file"`, the physical directory containing the current source
+file is searched first. For `#include <file>`, that first step is omitted.
+The remaining include search order is normative:
+
+```text
+explicit -I
+explicit -isystem
+MCPU_CPP_<LANG>_INCLUDE_PATH
+MCPU_CPP_INCLUDE_PATH
+MCPU_CPP_SYSTEM_INCLUDE_PATH/<lang>
+MCPU_CPP_SYSTEM_INCLUDE_PATH
+explicit -idirafter
+MCPU_CPP_AFTER_INCLUDE_PATH
+```
+
+Command-line include directories therefore override persistent configuration.
+The language-specific user paths are freely configurable. The system path is
+a single effective root; mcpu-cpp derives the fixed `<root>/<lang>` directory
+itself, so there are intentionally no
+`MCPU_CPP_SYSTEM_<LANG>_INCLUDE_PATH` variables. Replacing
+`MCPU_CPP_SYSTEM_INCLUDE_PATH` replaces the complete installed system-header
+tree rather than adding another directory. An empty effective value disables
+the configured system tree. Neither `-idirafter` nor
+`MCPU_CPP_AFTER_INCLUDE_PATH` acquires automatic language subdirectories.
+
+`#include_next` is intended for wrapper headers. It remembers the exact
+physical entry of this effective search chain that supplied the current header
+and resumes at the following entry. This allows an explicit wrapper to alter
+policy and then continue into a sandbox or installed system tree without
+copying the original header or hard-coding its absolute pathname. The
+`"file"` and `<file>` forms are equivalent for `#include_next`, and logical
+names established by `#line` do not affect physical search provenance.
+
+`#pragma once` marks the current physical file as processed for the remainder
+of the preprocessing run. Identity is the filesystem device/inode pair, not
+the pathname spelling, so the same file reached through another relative name,
+a symbolic link or a hard link is skipped. The exact active directive is
+consumed by mcpu-cpp; an inactive `#pragma once` has no effect. Other pragmas
+remain in output for later compiler stages.
+
+Command-line forced files
+-------------------------
+
+`-imacros FILE` and `-include FILE` use one deterministic preprocessing
+pipeline. Predefined macros are installed first, all command-line `-D`/`-U`
+actions are then applied in their own command-line order, every `-imacros`
+file is processed in command-line order, every `-include` file is processed in
+command-line order, and only then does preprocessing enter the primary input.
+The relative placement of `-imacros` and `-include` options in argv therefore
+does not interleave the two classes: every `-imacros` always precedes every
+`-include`.
+
+An `-imacros` file goes through the ordinary preprocessing machinery, including
+`#include`, macro definition/undefinition, conditional directives, `#lang`,
+`#pragma once`, diagnostics and dependency tracking. Its normal preprocessing
+output, including output from headers reached from that file, is discarded.
+The resulting macro table and other preprocessing state remain available to
+subsequent forced files and to the primary input. `-include` uses the same
+preprocessing machinery but retains its normal output, as if the header had
+been included immediately before the primary source.
+
+The operand of either forced-file option has GNU-style command-line search
+semantics. An absolute path is used directly. A relative path is searched
+first in the current working directory and then through the ordinary quoted
+include chain shown above. The physical directory of the primary input does
+not receive the special first priority that a literal `#include "file"` inside
+that source would receive. Once a forced file has been found, quoted includes
+inside it are resolved normally relative to that file's physical directory,
+and `#include_next` retains the search-chain provenance of the entry that found
+it.
+
+Forced files are ordinary physical dependencies. Files reached through CWD or
+user include classes remain user dependencies; files found through `-isystem`,
+the configured system tree, `-idirafter` or configured AFTER paths retain
+system dependency class for `-MM`/`-MMD`.
+
+
+Dependency generation
+---------------------
+
+`-M` preprocesses the translation unit and writes one make rule instead of
+normal preprocessor output. The rule contains the main source file and every
+physical header actually reached through the ordinary `#include` /
+`#include_next` pipeline, including system headers. Repeated spellings,
+symbolic links and hard links to the same physical file are recorded once.
+Logical names established by `#line` are never dependency names.
+
+`-MM` uses the same traversal but omits headers found through `-isystem`, the
+configured MCPU system tree, `-idirafter` and configured AFTER paths, and also
+omits headers reachable only as descendants of such a system header. Quote
+versus angle include spelling does not decide whether a dependency is system.
+If the same physical file is also reached directly through a user include
+context, it remains a user dependency.
+
+The default target is the input basename with its suffix replaced by the
+configured object suffix (normally `.o`). In dependency-only `-M`/`-MM` mode,
+`-MF FILE` selects the make-rule destination; without `-MF`, the existing
+mcpu-cpp `-o FILE` destination remains available.
+
+`-MD` and `-MMD` generate dependencies as a side effect and do **not** suppress
+normal preprocessing output. `-MD` includes system headers like `-M`; `-MMD`
+applies the same user-only filter as `-MM`. Without `-MF`, the dependency file
+is `<input-basename>.d` in the current directory, or is derived from ordinary
+`-o` output by replacing its suffix with `.d`. `-MF FILE` overrides that
+automatic name, and `-MF -` sends the dependency rule to stdout.
+
+Options `-MT` and `-MQ` are intentionally left for the next dependency stage.
+
+#lang contract
+--------------
+
+The initial language state is `0`. It is the base MCPU language state and is
+installed before preprocessing begins. `0` is deliberately not a parameter of
+`#lang`.
+
+`#lang` requires exactly one quoted language name:
+
+ #lang "diff"
+
+The text inside the quotes is not escape-decoded. It must be a non-empty
+single word without whitespace, and the closing quote must occur on the same
+physical source line. After the closing quote only whitespace is allowed.
+The accepted names come only from the internal language table: `diff`, `dift`,
+`alg`, `as`, `avm`, and `ACS`. Matching is ASCII case-insensitive, so for
+example `"diff"`, `"Diff"`, `"DIFF"` and `"dIfF"` all select the same
+language. Unsupported names such as `"0"`, `"c"` and `"vasm"` are errors.
+
+External whitespace is normalized when `#lang` is copied to output. For
+example:
+
+ # lang "DiFf"
+
+is emitted as:
+
+ #lang "DiFf"
+
+The spelling inside the quotes is preserved as written.
+
+`#lang` and `#endlang` form one translation-unit-wide stack. The stack is not
+reset at an `#include` boundary. Therefore a `#lang` in one file and its
+matching `#endlang` in another included file are intentionally legal, exactly
+as required by the macro-expansion contract.
+
+Both directives are passed through to output so the later language dispatcher
+and parser layer can observe the same language boundaries. `#lang` is emitted
+in the normalized form described above.
+
+Macro operators
+---------------
+
+Function-macro stringification (`#`) is supported.
+The operator uses the raw, unexpanded actual argument, removes leading and
+trailing whitespace, folds internal whitespace outside quoted tokens to one
+space, and escapes double quotes and backslashes as required by the resulting
+quoted string. `##` token concatenation is also supported and rescans the concatenated token
+through the ordinary macro-expansion path.
+
+The completed predefined ABI/environment layer from 0.0.12 is preserved.
+INT/UINT DECIMAL_DIG counts only decimal digits of the corresponding numeric
+maximum, without a sign or terminating NUL. Integer decimal precision and
+Real DECIMAL_DIG/MANT_DIG metadata cover every configured type width;
+large MAX/MIN/EPSILON textual values remain capped at 256 bits.
+
+`__MCPU_CPP_VERSION__` is the mcpu-cpp package version. `_ARCH_MCPU` identifies
+the target. The LibMPU profile macros use the MCPU namespace:
+`__MCPU_MACHINE_REGISTER_WIDTH__`, `__MCPU_REAL_IO_LIMIT__` and
+`__MCPU_MATH_FN_LIMIT__`.
+
+MCPU pointers are always 64 bits independently of the host. `intptr` and
+`ptrdiff` are signed `int64`; `uintptr` is `uint64`:
+
+ __INTPTR_TYPE__ int64
+ __INTPTR_WIDTH__ 64
+ __INTPTR_MAX__ 0x7fffffffffffffff
+ __UINTPTR_TYPE__ uint64
+ __UINTPTR_WIDTH__ 64
+ __UINTPTR_MAX__ 0xffffffffffffffff
+ __PTRDIFF_TYPE__ int64
+ __PTRDIFF_WIDTH__ 64
+ __PTRDIFF_MAX__ 0x7fffffffffffffff
+
+The configured LibMPU size and signed-size types are exposed only in the MCPU
+namespace. For a 64-bit profile the public contract is:
+
+ __MCPU_SIZE_TYPE__ uint64
+ __MCPU_SIZE_WIDTH__ 64
+ __MCPU_SIZEOF_SIZE__ 8
+ __MCPU_SIZE_MAX__ 0xffffffffffffffff
+ __MCPU_SSIZE_TYPE__ int64
+ __MCPU_SSIZE_WIDTH__ 64
+ __MCPU_SIZEOF_SSIZE__ 8
+ __MCPU_SSIZE_MAX__ 0x7fffffffffffffff
+
+Integer TYPE/WIDTH/SIZEOF metadata is generated for every power-of-two LibMPU
+integer family through `NB_I_MAX * 8`. Real and Complex TYPE/WIDTH/SIZEOF
+metadata is generated through the configured `MPU_REAL_IO_LIMIT`. Complex
+WIDTH is the language type parameter: `complex128` has WIDTH 128 but SIZEOF 32
+bytes, because it stores two real128 components.
+
+Every available Real family also exports two compact conversion/layout
+properties through `MPU_REAL_IO_LIMIT`: `__SIZEOF_REAL<bits>_EXP__` comes from
+LibMPU `_sizeof_exp()`, while `__REAL<bits>_MAX_STRLEN__` comes from
+`_real_max_string()`. MAX_STRLEN is a number of characters, not bytes; a
+zero-terminated `char8` or `char16` buffer therefore needs at least
+`MAX_STRLEN + 1` elements.
+
+Large numeric textual values remain deliberately capped at 256 bits. Integer
+MAX and Real MAX/MIN/EPSILON/exponent-value macros are not emitted above that
+width. Integer DECIMAL_DIG and Real DECIMAL_DIG/MANT_DIG remain available for
+the complete configured families. Integer MIN expressions are never
+predefined. This keeps `-dMP` compact while preserving useful precision,
+structural and text-buffer metadata for large LibMPU types.
+
+The future language uses fixed-width names. Character types are `char8` and
+`char16`; their sizes are `__SIZEOF_CHAR8__` and `__SIZEOF_CHAR16__`. Ordinary
+C `char`, `short`, `int`, `long`, `wchar_t`, and the C-style `char*_t` names are
+not part of the target language type model.
+
+The effective `MCPU_CPP_SYSTEM_INCLUDE_PATH` automatically contributes a
+language-specific subdirectory for each active switched language. For example,
+with the runtime-derived root and `#lang "diff"`,
+`<runtime-root>/include/diff` is searched before
+`<runtime-root>/include`. These derived directories need not exist and
+require no extra configuration variables.
+
+`-dM` dumps only the current non-predefined macro table in deterministic
+`#define` form. `-dMP` first dumps active predefined macros and then the
+non-predefined macros; each group is sorted by name.
+`-dconfig` dumps the effective configuration variables in sorted `NAME = value;`
+form. `-v` prints the include-related effective configuration values once,
+after all configuration layers have been resolved, and then preserves the usual
+include/language runtime trace. `-dsearch-dirs` prints `search: DIR` lines for
+the effective global include directories in semantic priority order and exits
+without preprocessing. The dynamic source-directory step used by quoted
+includes is not part of that global dump. None of these dump actions requires
+an input file.
+
+The dynamic source macros `__FILE__`, `__LINE__`, `__BASE_FILE__`,
+`__INCLUDE_LEVEL__`, `__DATE__` and `__TIME__` use one translation-unit
+source stack and timestamp. Stringification, concatenation and conditional
+compilation are part of the current preprocessing contract.
+
+Parser generation
+-----------------
+
+The #if expression parser is generated by ZUBR 4.1.0 from
+`src/mcpp-expr.zubr`. The grammar also contains the UCS-2 lexical analyzer.
+The generated source `src/mcpp-expr.c` is included in release archives, so a
+normal build from a release archive does not require ZUBR.
+
+The conditional-expression evaluator is deliberately 64-bit only. Integer
+literals may use `U`/`u`, or the MCPU width suffix `zNNN[Uu]` / `ZNNN[Uu]`.
+For valid widths up to 64, the low N bits are taken and then sign-extended
+(`zNNN`) or zero-extended (`zNNNu`) to 64 bits; all later operations remain
+64-bit and the original width is forgotten. Widths above 64 are rejected in
+conditional directives. Invalid widths not exceeding 64 produce a warning
+and the width suffix is ignored. C `L`/`LL` integer suffixes are not accepted.
+
+`#error` and `#warning` are implemented as diagnostic directives. Their
+arguments are not macro-expanded. Outside quoted tokens, whitespace sequences
+are folded to one space for the diagnostic text. `#error` stops preprocessing;
+`#warning` continues. Both directives disappear from normal and `-dD` output,
+and both are ignored in inactive conditional branches.
+
+Warning control follows a compact GNU-like model. `-Wcomment` and `-Wcomments`
+enable warnings for `/*` inside an existing block comment and for a
+backslash-newline continuing a `//` comment; `-Wall` currently enables this
+optional warning class. The specific `-Wno-comment`/`-Wno-comments` setting
+overrides `-Wall` regardless of command-line order. `-w` globally suppresses
+all warnings, including `#warning`, invalid `zNNN` widths, macro redefinition,
+invalid `##` paste and enabled comment diagnostics. It does not suppress
+errors. `-Werror` promotes every warning that is actually emitted to an error
+and unsuccessful preprocessing; because `-w` prevents warning emission first,
+`-w -Werror` and `-Werror -w` are equivalent and successful when no independent
+error occurs. Likewise, warning classes may be enabled by `-Wcomment` or
+`-Wall`, but remain silent under `-w` regardless of option order. `-Wno-error`
+restores ordinary warning severity. `-Werror` does not by itself enable
+optional warning classes.
+
+When `mcpp-expr.zubr` is changed, the ordinary Automake `.zubr.c` rule
+regenerates the C source with:
+
+ zubr -vl -s -Bmcpp_ -o mcpp-expr.c mcpp-expr.zubr
+
+
+Developer bootstrap
+-------------------
+
+A Git checkout may omit files that are regenerated mechanically, including
+`configure`, `Makefile.in`, Automake helper scripts, `config.h.in`, `aclocal.m4`
+and `src/mcpp-expr.c`. Regenerate them with:
+
+```text
+./bootstrap
+```
+
+`./bootstrap --target-dest-dir=DIR` follows the LibMPU/LibMPUIO convention for
+using Autoconf macro/header directories from a target ROOTFS. Release archives
+remain self-contained and do not require bootstrap before `configure`.
+
+Build
+-----
+
+mcpu-cpp uses Autoconf/Automake and obtains LibMPUIO compilation and linker
+flags from `mpuio-config`, following the conventions of LibMPU and LibMPUIO.
+No pkg-config or flex dependency is introduced. The 1.0.2 release is
+developed and tested against LibMPU 1.0.25 and LibMPUIO 1.0.4. The LibMPUIO
+UCS-2 ctype API is required. ZUBR 4.1.0 is the parser-regeneration tool; it is
+not required for a normal build from a release archive containing the generated
+`src/mcpp-expr.c`.
+
+A normal build is:
+
+ ./configure --prefix=/usr --libdir=/usr/lib64
+ make
+ make tests
+ make install
diff --git a/acsite.m4 b/acsite.m4
new file mode 100644
index 0000000..8b4c011
--- /dev/null
+++ b/acsite.m4
@@ -0,0 +1,797 @@
+dnl ============================================================
+dnl Force configure to run under /bin/bash
+dnl ============================================================
+m4_define([AC_MCPU_CPP_REQUIRE_BASH], [dnl
+m4_divert_push([M4SH-SANITIZE])dnl
+
+if test ! -x /bin/bash; then
+ echo "configure: error: /bin/bash is required" >&2
+ exit 1
+fi
+
+CONFIG_SHELL=/bin/bash
+export CONFIG_SHELL
+
+m4_divert_pop([M4SH-SANITIZE])dnl
+])dnl
+
+
+dnl ============================================================
+dnl Support for Configuration Headers
+dnl
+dnl configure.ac:
+dnl AC_MCPU_CPP_HEADLINE(<short-name>, <long-name>,
+dnl <vers-var>, <copyright>)
+dnl
+dnl NOTE:
+dnl ====
+dnl See the paragraph 8.3.3 of Autoconf Documentation at:
+dnl
+dnl https://www.gnu.org/software/autoconf/manual/autoconf.html#Diversion-support
+dnl ============================================================
+m4_define([AC_MCPU_CPP_HEADLINE], [dnl
+m4_divert_push([M4SH-INIT])dnl
+{
+ if test ".`echo dummy [$]@ | grep help`" = .; then
+
+ ####### нахождение escape последовательностей для
+ ####### обозначения начала и конца выделяемого текста
+ ####### Use a `Quadrigaph'. '@<:@' gives you [ and '@:>@' gives you ] :
+ TB=`echo -n -e '\033@<:@1m'`
+ TN=`echo -n -e '\033@<:@0m'`
+
+ ####### получение короткого номера версии продукта
+ ####### из AC_INIT()
+ $3="AC_PACKAGE_VERSION"
+ AC_SUBST($3)
+
+ ####### печать заголовка
+ echo "Configuring:"
+ echo ""
+ echo "${TB}$1${TN} ($2), Version ${TB}${$3}${TN}"
+ echo "$4"
+ echo ""
+ fi
+}
+m4_divert_pop([M4SH-INIT])dnl
+])dnl
+
+
+dnl ============================================================
+dnl Display Configuration Headers
+dnl
+dnl configure.ac:
+dnl AC_MSG_CFG_PART(<text>)
+dnl ============================================================
+AC_DEFUN([AC_MSG_CFG_PART],[dnl
+ AC_MSG_RESULT()
+ AC_MSG_RESULT([${TB}$1:${TN}])
+])dnl
+
+
+dnl ============================================================
+dnl GCC is a required part of the mcpu-cpp build environment.
+dnl ============================================================
+AC_DEFUN([MCPU_CPP_REQUIRE_GCC], [dnl
+ AC_MSG_CHECKING([whether $CC is GNU GCC])
+ AC_COMPILE_IFELSE([
+ AC_LANG_SOURCE([[
+#if !defined(__GNUC__) || defined(__clang__) || defined(__INTEL_COMPILER) || defined(__INTEL_LLVM_COMPILER)
+#error mcpu-cpp requires GNU GCC
+#endif
+int main( void ) { return 0; }
+ ]])
+ ], [
+ AC_MSG_RESULT([yes])
+ ], [
+ AC_MSG_RESULT([no])
+ AC_MSG_ERROR([mcpu-cpp requires GNU GCC])
+ ])
+])
+dnl ============================================================
+dnl Test for WIDTH of MACHINE REGISTER using GCC predefines:
+dnl ==================
+dnl
+dnl configure.ac:
+dnl MCPU_CPP_GCC_REGISTER_WIDTH
+dnl
+dnl ============================================================
+AC_DEFUN([MCPU_CPP_GCC_REGISTER_WIDTH],
+[dnl
+AC_MSG_CHECKING(for CPP predefined macro __INT_FAST32_WIDTH__ )
+AC_CACHE_VAL(ac_cv_int_fast32_width,
+[ac_cv_int_fast32_width=`$CC -dM -E - < /dev/null | grep __INT_FAST32_WIDTH__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_int_fast32_width" != ""; then
+ AC_MSG_RESULT($ac_cv_int_fast32_width)
+else
+ AC_MSG_RESULT(not defined)
+fi
+
+AC_MSG_CHECKING(for CPP predefined macro __INT_FAST64_WIDTH__ )
+AC_CACHE_VAL(ac_cv_int_fast64_width,
+[ac_cv_int_fast64_width=`$CC -dM -E - < /dev/null | grep __INT_FAST64_WIDTH__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_int_fast64_width" != ""; then
+ AC_MSG_RESULT($ac_cv_int_fast64_width)
+else
+ AC_MSG_RESULT(not defined)
+fi
+
+AC_MSG_CHECKING(for MACHINE REGISTER WIDTH )
+AC_CACHE_VAL(ac_cv_machine_register_width,
+[dnl
+if test "$ac_cv_int_fast64_width" != "" -a "$ac_cv_int_fast32_width" != "" ; then
+ if test "$ac_cv_int_fast64_width" -gt "$ac_cv_int_fast32_width" ; then
+ ac_cv_machine_register_width=$ac_cv_int_fast32_width
+ else
+ ac_cv_machine_register_width=$ac_cv_int_fast64_width
+ fi
+elif test "$ac_cv_int_fast64_width" != "" ; then
+ ac_cv_machine_register_width=$ac_cv_int_fast64_width
+elif test "$ac_cv_int_fast32_width" != "" ; then
+ ac_cv_machine_register_width=$ac_cv_int_fast32_width
+else
+ ac_cv_machine_register_width=32
+fi
+])dnl
+if test "$ac_cv_machine_register_width" != ""; then
+ MACHINE_REGISTER_WIDTH=$ac_cv_machine_register_width
+ AC_MSG_RESULT($ac_cv_machine_register_width)
+else
+ MACHINE_REGISTER_WIDTH=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(MACHINE_REGISTER_WIDTH)dnl
+AC_DEFINE_UNQUOTED([MACHINE_REGISTER_WIDTH], [$ac_cv_machine_register_width], [The size of Machine Register in bits.])dnl
+])
+
+
+dnl ============================================================
+dnl Test for GCC Types:
+dnl ==================
+dnl
+dnl configure.ac:
+dnl MCPU_CPP_GCC_TYPES
+dnl
+dnl ============================================================
+AC_DEFUN([MCPU_CPP_GCC_TYPES],
+[dnl
+AC_MSG_CHECKING(for CPP predefined macro __CHAR16_TYPE__ )
+AC_CACHE_VAL(ac_cv_char16_type,
+[ac_cv_char16_type=`$CC -dM -E - < /dev/null | grep __CHAR16_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_char16_type" != ""; then
+ GCC_CHAR16_TYPE=$ac_cv_char16_type
+ AC_MSG_RESULT($ac_cv_char16_type)
+else
+ GCC_CHAR16_TYPE=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_CHAR16_TYPE)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __CHAR32_TYPE__ )
+AC_CACHE_VAL(ac_cv_char32_type,
+[ac_cv_char32_type=`$CC -dM -E - < /dev/null | grep __CHAR32_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_char32_type" != ""; then
+ GCC_CHAR32_TYPE=$ac_cv_char32_type
+ AC_MSG_RESULT($ac_cv_char32_type)
+else
+ GCC_CHAR32_TYPE=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_CHAR32_TYPE)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __WCHAR_TYPE__ )
+AC_CACHE_VAL(ac_cv_wchar_type,
+[ac_cv_wchar_type=`$CC -dM -E - < /dev/null | grep __WCHAR_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_wchar_type" != ""; then
+ GCC_WCHAR_TYPE=$ac_cv_wchar_type
+ AC_MSG_RESULT($ac_cv_wchar_type)
+else
+ GCC_WCHAR_TYPE=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_WCHAR_TYPE)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __INT8_TYPE__ )
+AC_CACHE_VAL(ac_cv_int8_type,
+[ac_cv_int8_type=`$CC -dM -E - < /dev/null | grep __INT8_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_int8_type" != ""; then
+ GCC_INT8_TYPE=$ac_cv_int8_type
+ AC_MSG_RESULT($ac_cv_int8_type)
+else
+ GCC_INT8_TYPE=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_INT8_TYPE)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __UINT8_TYPE__ )
+AC_CACHE_VAL(ac_cv_uint8_type,
+[ac_cv_uint8_type=`$CC -dM -E - < /dev/null | grep __UINT8_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_uint8_type" != ""; then
+ GCC_UINT8_TYPE=$ac_cv_uint8_type
+ AC_MSG_RESULT($ac_cv_uint8_type)
+else
+ GCC_UINT8_TYPE=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_UINT8_TYPE)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __INT16_TYPE__ )
+AC_CACHE_VAL(ac_cv_int16_type,
+[ac_cv_int16_type=`$CC -dM -E - < /dev/null | grep __INT16_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_int16_type" != ""; then
+ GCC_INT16_TYPE=$ac_cv_int16_type
+ AC_MSG_RESULT($ac_cv_int16_type)
+else
+ GCC_INT16_TYPE=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_INT16_TYPE)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __UINT16_TYPE__ )
+AC_CACHE_VAL(ac_cv_uint16_type,
+[ac_cv_uint16_type=`$CC -dM -E - < /dev/null | grep __UINT16_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_uint16_type" != ""; then
+ GCC_UINT16_TYPE=$ac_cv_uint16_type
+ AC_MSG_RESULT($ac_cv_uint16_type)
+else
+ GCC_UINT16_TYPE=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_UINT16_TYPE)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __INT32_TYPE__ )
+AC_CACHE_VAL(ac_cv_int32_type,
+[ac_cv_int32_type=`$CC -dM -E - < /dev/null | grep __INT32_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_int32_type" != ""; then
+ GCC_INT32_TYPE=$ac_cv_int32_type
+ AC_MSG_RESULT($ac_cv_int32_type)
+else
+ GCC_INT32_TYPE=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_INT32_TYPE)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __UINT32_TYPE__ )
+AC_CACHE_VAL(ac_cv_uint32_type,
+[ac_cv_uint32_type=`$CC -dM -E - < /dev/null | grep __UINT32_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+AC_CACHE_VAL(ac_cv_uint32_const_suffix,
+[ac_cv_uint32_const_suffix=`$CC -dM -E - < /dev/null | grep __UINT32_C | cut -f5- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_uint32_type" != ""; then
+ GCC_UINT32_TYPE=$ac_cv_uint32_type
+ AC_MSG_RESULT($ac_cv_uint32_type)
+else
+ GCC_UINT32_TYPE=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_UINT32_TYPE)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __INT64_TYPE__ )
+AC_CACHE_VAL(ac_cv_int64_type,
+[ac_cv_int64_type=`$CC -dM -E - < /dev/null | grep __INT64_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_int64_type" != ""; then
+ GCC_INT64_TYPE=$ac_cv_int64_type
+ AC_MSG_RESULT($ac_cv_int64_type)
+else
+ GCC_INT64_TYPE=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_INT64_TYPE)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __UINT64_TYPE__ )
+AC_CACHE_VAL(ac_cv_uint64_type,
+[ac_cv_uint64_type=`$CC -dM -E - < /dev/null | grep __UINT64_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+AC_CACHE_VAL(ac_cv_uint64_const_suffix,
+[ac_cv_uint64_const_suffix=`$CC -dM -E - < /dev/null | grep __UINT64_C | cut -f5- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_uint64_type" != ""; then
+ GCC_UINT64_TYPE=$ac_cv_uint64_type
+ AC_MSG_RESULT($ac_cv_uint64_type)
+else
+ GCC_UINT64_TYPE=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_UINT64_TYPE)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __WINT_TYPE__ )
+AC_CACHE_VAL(ac_cv_wint_type,
+[ac_cv_wint_type=`$CC -dM -E - < /dev/null | grep __WINT_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_wint_type" != ""; then
+ GCC_WINT_TYPE=$ac_cv_wint_type
+ AC_MSG_RESULT($ac_cv_wint_type)
+else
+ GCC_WINT_TYPE=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_WINT_TYPE)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __INTMAX_TYPE__ )
+AC_CACHE_VAL(ac_cv_intmax_type,
+[ac_cv_intmax_type=`$CC -dM -E - < /dev/null | grep __INTMAX_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_intmax_type" != ""; then
+ GCC_INTMAX_TYPE=$ac_cv_intmax_type
+ AC_MSG_RESULT($ac_cv_intmax_type)
+else
+ GCC_INTMAX_TYPE=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_INTMAX_TYPE)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __UINTMAX_TYPE__ )
+AC_CACHE_VAL(ac_cv_uintmax_type,
+[ac_cv_uintmax_type=`$CC -dM -E - < /dev/null | grep __UINTMAX_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_uintmax_type" != ""; then
+ GCC_UINTMAX_TYPE=$ac_cv_uintmax_type
+ AC_MSG_RESULT($ac_cv_uintmax_type)
+else
+ GCC_UINTMAX_TYPE=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_UINTMAX_TYPE)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __SIZE_TYPE__ )
+AC_CACHE_VAL(ac_cv_size_type,
+[ac_cv_size_type=`$CC -dM -E - < /dev/null | grep __SIZE_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_size_type" != ""; then
+ GCC_SIZE_TYPE=$ac_cv_size_type
+ AC_MSG_RESULT($ac_cv_size_type)
+else
+ GCC_SIZE_TYPE=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_SIZE_TYPE)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __INTPTR_TYPE__ )
+AC_CACHE_VAL(ac_cv_intptr_type,
+[ac_cv_intptr_type=`$CC -dM -E - < /dev/null | grep __INTPTR_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_intptr_type" != ""; then
+ GCC_INTPTR_TYPE=$ac_cv_intptr_type
+ AC_MSG_RESULT($ac_cv_intptr_type)
+else
+ GCC_INTPTR_TYPE=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_INTPTR_TYPE)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __UINTPTR_TYPE__ )
+AC_CACHE_VAL(ac_cv_uintptr_type,
+[ac_cv_uintptr_type=`$CC -dM -E - < /dev/null | grep __UINTPTR_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_uintptr_type" != ""; then
+ GCC_UINTPTR_TYPE=$ac_cv_uintptr_type
+ AC_MSG_RESULT($ac_cv_uintptr_type)
+else
+ GCC_UINTPTR_TYPE=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_UINTPTR_TYPE)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __PTRDIFF_TYPE__ )
+AC_CACHE_VAL(ac_cv_ptrdiff_type,
+[ac_cv_ptrdiff_type=`$CC -dM -E - < /dev/null | grep __PTRDIFF_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_ptrdiff_type" != ""; then
+ GCC_PTRDIFF_TYPE=$ac_cv_ptrdiff_type
+ AC_MSG_RESULT($ac_cv_ptrdiff_type)
+else
+ GCC_PTRDIFF_TYPE=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_PTRDIFF_TYPE)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __SIG_ATOMIC_TYPE__ )
+AC_CACHE_VAL(ac_cv_sig_atomic_type,
+[ac_cv_sig_atomic_type=`$CC -dM -E - < /dev/null | grep __SIG_ATOMIC_TYPE__ | cut -f3- -d' ' | tr '\n' ' ' | tr -s ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_sig_atomic_type" != ""; then
+ GCC_SIG_ATOMIC_TYPE=$ac_cv_sig_atomic_type
+ AC_MSG_RESULT($ac_cv_sig_atomic_type)
+else
+ GCC_SIG_ATOMIC_TYPE=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_SIG_ATOMIC_TYPE)dnl
+])
+
+
+dnl ============================================================
+dnl Test for GCC Sizeof Types:
+dnl =========================
+dnl
+dnl configure.ac:
+dnl MCPU_CPP_GCC_SIZEOF_TYPES
+dnl
+dnl ============================================================
+AC_DEFUN([MCPU_CPP_GCC_SIZEOF_TYPES],
+[dnl
+AC_MSG_CHECKING(for CPP predefined macro __SIZEOF_WCHAR_T__ )
+AC_CACHE_VAL(ac_cv_sizeof_wchar_t,
+[ac_cv_sizeof_wchar_t=`$CC -dM -E - < /dev/null | grep __SIZEOF_WCHAR_T__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_sizeof_wchar_t" != ""; then
+ GCC_SIZEOF_WCHAR_T=$ac_cv_sizeof_wchar_t
+ AC_MSG_RESULT($ac_cv_sizeof_wchar_t)
+else
+ GCC_SIZEOF_WCHAR_T=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_SIZEOF_WCHAR_T)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __SIZEOF_SHORT__ )
+AC_CACHE_VAL(ac_cv_sizeof_short,
+[ac_cv_sizeof_short=`$CC -dM -E - < /dev/null | grep __SIZEOF_SHORT__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_sizeof_short" != ""; then
+ GCC_SIZEOF_SHORT=$ac_cv_sizeof_short
+ AC_MSG_RESULT($ac_cv_sizeof_short)
+else
+ GCC_SIZEOF_SHORT=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_SIZEOF_SHORT)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __SIZEOF_INT__ )
+AC_CACHE_VAL(ac_cv_sizeof_int,
+[ac_cv_sizeof_int=`$CC -dM -E - < /dev/null | grep __SIZEOF_INT__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_sizeof_int" != ""; then
+ GCC_SIZEOF_INT=$ac_cv_sizeof_int
+ AC_MSG_RESULT($ac_cv_sizeof_int)
+else
+ GCC_SIZEOF_INT=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_SIZEOF_INT)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __SIZEOF_WINT_T__ )
+AC_CACHE_VAL(ac_cv_sizeof_wint_t,
+[ac_cv_sizeof_wint_t=`$CC -dM -E - < /dev/null | grep __SIZEOF_WINT_T__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_sizeof_wint_t" != ""; then
+ GCC_SIZEOF_WINT_T=$ac_cv_sizeof_wint_t
+ AC_MSG_RESULT($ac_cv_sizeof_wint_t)
+else
+ GCC_SIZEOF_WINT_T=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_SIZEOF_WINT_T)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __SIZEOF_LONG__ )
+AC_CACHE_VAL(ac_cv_sizeof_long,
+[ac_cv_sizeof_long=`$CC -dM -E - < /dev/null | grep __SIZEOF_LONG__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_sizeof_long" != ""; then
+ GCC_SIZEOF_LONG=$ac_cv_sizeof_long
+ AC_MSG_RESULT($ac_cv_sizeof_long)
+else
+ GCC_SIZEOF_LONG=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_SIZEOF_LONG)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __SIZEOF_LONG_LONG__ )
+AC_CACHE_VAL(ac_cv_sizeof_long_long,
+[ac_cv_sizeof_long_long=`$CC -dM -E - < /dev/null | grep __SIZEOF_LONG_LONG__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_sizeof_long_long" != ""; then
+ GCC_SIZEOF_LONG_LONG=$ac_cv_sizeof_long_long
+ AC_MSG_RESULT($ac_cv_sizeof_long_long)
+else
+ GCC_SIZEOF_LONG_LONG=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_SIZEOF_LONG_LONG)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __SIZEOF_FLOAT__ )
+AC_CACHE_VAL(ac_cv_sizeof_float,
+[ac_cv_sizeof_float=`$CC -dM -E - < /dev/null | grep __SIZEOF_FLOAT__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_sizeof_float" != ""; then
+ GCC_SIZEOF_FLOAT=$ac_cv_sizeof_float
+ AC_MSG_RESULT($ac_cv_sizeof_float)
+else
+ GCC_SIZEOF_FLOAT=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_SIZEOF_FLOAT)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __SIZEOF_DOUBLE__ )
+AC_CACHE_VAL(ac_cv_sizeof_double,
+[ac_cv_sizeof_double=`$CC -dM -E - < /dev/null | grep __SIZEOF_DOUBLE__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_sizeof_double" != ""; then
+ GCC_SIZEOF_DOUBLE=$ac_cv_sizeof_double
+ AC_MSG_RESULT($ac_cv_sizeof_double)
+else
+ GCC_SIZEOF_DOUBLE=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_SIZEOF_DOUBLE)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __SIZEOF_LONG_DOUBLE__ )
+AC_CACHE_VAL(ac_cv_sizeof_long_double,
+[ac_cv_sizeof_long_double=`$CC -dM -E - < /dev/null | grep __SIZEOF_LONG_DOUBLE__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_sizeof_long_double" != ""; then
+ GCC_SIZEOF_LONG_DOUBLE=$ac_cv_sizeof_long_double
+ AC_MSG_RESULT($ac_cv_sizeof_long_double)
+else
+ GCC_SIZEOF_LONG_DOUBLE=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_SIZEOF_LONG_DOUBLE)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __SIZEOF_SIZE_T__ )
+AC_CACHE_VAL(ac_cv_sizeof_size_t,
+[ac_cv_sizeof_size_t=`$CC -dM -E - < /dev/null | grep __SIZEOF_SIZE_T__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_sizeof_size_t" != ""; then
+ GCC_SIZEOF_SIZE_T=$ac_cv_sizeof_size_t
+ AC_MSG_RESULT($ac_cv_sizeof_size_t)
+else
+ GCC_SIZEOF_SIZE_T=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_SIZEOF_SIZE_T)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __SIZEOF_POINTER__ )
+AC_CACHE_VAL(ac_cv_sizeof_pointer,
+[ac_cv_sizeof_pointer=`$CC -dM -E - < /dev/null | grep __SIZEOF_POINTER__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_sizeof_pointer" != ""; then
+ GCC_SIZEOF_POINTER=$ac_cv_sizeof_pointer
+ AC_MSG_RESULT($ac_cv_sizeof_pointer)
+else
+ GCC_SIZEOF_POINTER=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_SIZEOF_POINTER)dnl
+
+AC_MSG_CHECKING(for CPP predefined macro __SIZEOF_PTRDIFF_T__ )
+AC_CACHE_VAL(ac_cv_sizeof_ptrdiff_t,
+[ac_cv_sizeof_ptrdiff_t=`$CC -dM -E - < /dev/null | grep __SIZEOF_PTRDIFF_T__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_sizeof_ptrdiff_t" != ""; then
+ GCC_SIZEOF_PTRDIFF_T=$ac_cv_sizeof_ptrdiff_t
+ AC_MSG_RESULT($ac_cv_sizeof_ptrdiff_t)
+else
+ GCC_SIZEOF_PTRDIFF_T=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_SIZEOF_PTRDIFF_T)dnl
+])
+
+
+dnl ============================================================
+dnl Test for GCC Width of Types:
+dnl ===========================
+dnl
+dnl configure.ac:
+dnl MCPU_CPP_GCC_WIDTH_OF_TYPES
+dnl
+dnl ============================================================
+AC_DEFUN([MCPU_CPP_GCC_WIDTH_OF_TYPES],
+[dnl
+AC_MSG_CHECKING(for CPP predefined macro __CHAR_BIT__ )
+AC_CACHE_VAL(ac_cv_char_width,
+[ac_cv_char_width=`$CC -dM -E - < /dev/null | grep __CHAR_BIT__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_char_width" != ""; then
+ GCC_CHAR_WIDTH=$ac_cv_char_width
+ AC_MSG_RESULT($ac_cv_char_width)
+else
+ GCC_CHAR_WIDTH=
+ AC_MSG_RESULT(not defined)
+fi
+AC_SUBST(GCC_CHAR_WIDTH)dnl
+])
+
+
+dnl ============================================================
+dnl Test for GCC Byte Order:
+dnl =======================
+dnl
+dnl configure.ac:
+dnl MCPU_CPP_GCC_BYTE_ORDER
+dnl
+dnl ============================================================
+AC_DEFUN([MCPU_CPP_GCC_BYTE_ORDER],
+[dnl
+AC_MSG_CHECKING(for CPP predefined macro __BYTE_ORDER__ )
+AC_CACHE_VAL(ac_cv_order_little_endian,
+[ac_cv_order_little_endian=`$CC -dM -E - < /dev/null | grep __ORDER_LITTLE_ENDIAN__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+AC_CACHE_VAL(ac_cv_order_big_endian,
+[ac_cv_order_big_endian=`$CC -dM -E - < /dev/null | grep __ORDER_BIG_ENDIAN__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+AC_CACHE_VAL(ac_cv_byte_order,
+[ac_cv_byte_order=`$CC -dM -E - < /dev/null | grep __BYTE_ORDER__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_byte_order" = "__ORDER_LITTLE_ENDIAN__"; then
+ GCC_BYTE_ORDER=1234
+ GCC_BYTE_ORDER_LITTLE_ENDIAN=1
+ GCC_BYTE_ORDER_BIG_ENDIAN=0
+ AC_MSG_RESULT(Little-endian)
+else
+ GCC_BYTE_ORDER=4321
+ GCC_BYTE_ORDER_LITTLE_ENDIAN=0
+ GCC_BYTE_ORDER_BIG_ENDIAN=1
+ AC_MSG_RESULT(Big-endian)
+fi
+AC_SUBST(GCC_BYTE_ORDER)dnl
+AC_SUBST(GCC_BYTE_ORDER_LITTLE_ENDIAN)dnl
+AC_SUBST(GCC_BYTE_ORDER_BIG_ENDIAN)dnl
+])
+
+
+dnl ============================================================
+dnl Test for GCC Float Word Order:
+dnl =============================
+dnl
+dnl configure.ac:
+dnl MCPU_CPP_GCC_FLOAT_WORD_ORDER
+dnl
+dnl ============================================================
+AC_DEFUN([MCPU_CPP_GCC_FLOAT_WORD_ORDER],
+[dnl
+AC_MSG_CHECKING(for CPP predefined macro __FLOAT_WORD_ORDER__ )
+AC_CACHE_VAL(ac_cv_order_little_endian,
+[ac_cv_order_little_endian=`$CC -dM -E - < /dev/null | grep __ORDER_LITTLE_ENDIAN__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+AC_CACHE_VAL(ac_cv_order_big_endian,
+[ac_cv_order_big_endian=`$CC -dM -E - < /dev/null | grep __ORDER_BIG_ENDIAN__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+AC_CACHE_VAL(ac_cv_float_word_order,
+[ac_cv_float_word_order=`$CC -dM -E - < /dev/null | grep __FLOAT_WORD_ORDER__ | cut -f3 -d' ' | tr '\n' ' ' | sed 's,^[ \t]*,,;s,[ \t]*$,,'`])dnl
+if test "$ac_cv_float_word_order" = "__ORDER_LITTLE_ENDIAN__"; then
+ GCC_FLOAT_WORD_ORDER=1234
+ GCC_FLOAT_WORD_ORDER_LITTLE_ENDIAN=1
+ GCC_FLOAT_WORD_ORDER_BIG_ENDIAN=0
+ AC_MSG_RESULT(Little-endian)
+else
+ GCC_FLOAT_WORD_ORDER=4321
+ GCC_FLOAT_WORD_ORDER_LITTLE_ENDIAN=0
+ GCC_FLOAT_WORD_ORDER_BIG_ENDIAN=1
+ AC_MSG_RESULT(Big-endian)
+fi
+AC_SUBST(GCC_FLOAT_WORD_ORDER)dnl
+AC_SUBST(GCC_FLOAT_WORD_ORDER_LITTLE_ENDIAN)dnl
+AC_SUBST(GCC_FLOAT_WORD_ORDER_BIG_ENDIAN)dnl
+])
+
+
+dnl =======================================================================
+dnl AC_GCC_PROGRAMMING_MODEL
+dnl
+dnl Determine programming model corresponding to following list:
+dnl
+dnl PROGRAMMING MODELS:
+dnl ------------------
+dnl
+dnl PM: | 1 | 2 | 3 | 4 | 5 | 6
+dnl -----------+-------+-------+-----------+-------+-----------+--------
+dnl | LP32 | ILP32 | LP64 | ILP64 | LLP64 | SILP64
+dnl Data type | | | (I32LP64) | | (IL32P64) |
+dnl ===========|=======|=======|===========|=======|===========|========
+dnl char | 8 | 8 | 8 | 8 | 8 | 8
+dnl -----------+-------+-------+-----------+-------+-----------+--------
+dnl short | 16 | 16 | 16 | 16 | 16 | 64
+dnl -----------+-------+-------+-----------+-------+-----------+--------
+dnl int | 16 | 32 | 32 | 64 | 32 | 64
+dnl -----------+-------+-------+-----------+-------+-----------+--------
+dnl long | 32 | 32 | 64 | 64 | 32 | 64
+dnl -----------+-------+-------+-----------+-------+-----------+--------
+dnl long long | 64
+dnl -----------+-------+-------+-----------+-------+-----------+--------
+dnl pointer | 32 | 32 | 64 | 64 | 64 | 64
+dnl -----------+-------+-------+-----------+-------+-----------+--------
+dnl ptrdiff_t | 32 | 32 | 64 | 64 | 64 | 64
+dnl ===========+=======+=======+===========+=======+===========+========
+dnl CPU: | | i686 | x86_64 | | |
+dnl
+dnl Many 64-bit platforms today use an LP64 model (including Solaris,
+dnl AIX, HP-UX, Linux, macOS, BSD, and IBM z/OS). Microsoft Windows
+dnl uses an LLP64 model. The disadvantage of the LP64 model is that
+dnl storing a long into an int may overflow. On the other hand,
+dnl converting a pointer to a long will “work” in LP64. In the LLP64
+dnl model, the reverse is true. These are not problems which affect
+dnl fully standard-compliant code, but code is often written with
+dnl implicit assumptions about the widths of data types. C code
+dnl should prefer (u)intptr_t instead of long when casting pointers
+dnl into integer objects.
+dnl
+
+
+dnl ============================================================
+dnl Read the ABI constants exported by the configured LibMPU.
+dnl mpuio-config supplies the include path and LibMPUIO includes
+dnl <libmpu.h>, so the values below are the exact constants used
+dnl by the installed LibMPU rather than independent guesses.
+dnl ============================================================
+AC_DEFUN([MCPU_CPP_CHECK_LIBMPU_ABI], [dnl
+ AC_REQUIRE([MCPU_CPP_GCC_REGISTER_WIDTH])dnl
+ AC_REQUIRE([MCPU_CPP_GCC_TYPES])dnl
+ AC_REQUIRE([MCPU_CPP_GCC_SIZEOF_TYPES])dnl
+ AC_REQUIRE([MCPU_CPP_GCC_WIDTH_OF_TYPES])dnl
+ AC_REQUIRE([MCPU_CPP_GCC_BYTE_ORDER])dnl
+
+ mcpu_cpp_save_CPPFLAGS="$CPPFLAGS"
+ mcpu_cpp_save_CFLAGS="$CFLAGS"
+ CFLAGS="$CFLAGS $MPUIO_CFLAGS"
+
+ AC_MSG_CHECKING([LibMPU real I/O limit])
+ AC_COMPUTE_INT([MCPU_CPP_MPU_REAL_IO_LIMIT], [MPU_REAL_IO_LIMIT],
+ [[#include <libmpuio.h>]],
+ [AC_MSG_ERROR([cannot determine MPU_REAL_IO_LIMIT])])
+ AC_MSG_RESULT([$MCPU_CPP_MPU_REAL_IO_LIMIT])
+
+ AC_MSG_CHECKING([LibMPU math-function limit])
+ AC_COMPUTE_INT([MCPU_CPP_MPU_MATH_FN_LIMIT], [MPU_MATH_FN_LIMIT],
+ [[#include <libmpuio.h>]],
+ [AC_MSG_ERROR([cannot determine MPU_MATH_FN_LIMIT])])
+ AC_MSG_RESULT([$MCPU_CPP_MPU_MATH_FN_LIMIT])
+
+ AC_MSG_CHECKING([LibMPU integer maximum width])
+ AC_COMPUTE_INT([MCPU_CPP_MPU_INT_MAX_WIDTH], [NB_I_MAX * BITS_PER_UNIT_T],
+ [[#include <libmpuio.h>]],
+ [AC_MSG_ERROR([cannot determine NB_I_MAX * BITS_PER_UNIT_T])])
+ AC_MSG_RESULT([$MCPU_CPP_MPU_INT_MAX_WIDTH])
+
+ AC_MSG_CHECKING([LibMPU byte order])
+ AC_COMPUTE_INT([MCPU_CPP_MPU_BYTE_ORDER], [MPU_BYTE_ORDER],
+ [[#include <libmpuio.h>]],
+ [AC_MSG_ERROR([cannot determine MPU_BYTE_ORDER])])
+ AC_MSG_RESULT([$MCPU_CPP_MPU_BYTE_ORDER])
+
+ AC_MSG_CHECKING([LibMPU word order])
+ AC_COMPUTE_INT([MCPU_CPP_MPU_WORD_ORDER], [MPU_WORD_ORDER],
+ [[#include <libmpuio.h>]],
+ [AC_MSG_ERROR([cannot determine MPU_WORD_ORDER])])
+ AC_MSG_RESULT([$MCPU_CPP_MPU_WORD_ORDER])
+
+ AC_MSG_CHECKING([LibMPU machine-register width])
+ AC_COMPUTE_INT([MCPU_CPP_MPU_REGISTER_WIDTH], [BITS_PER_MACHINE_REGISTER],
+ [[#include <libmpuio.h>]],
+ [AC_MSG_ERROR([cannot determine BITS_PER_MACHINE_REGISTER])])
+ AC_MSG_RESULT([$MCPU_CPP_MPU_REGISTER_WIDTH])
+
+ AC_MSG_CHECKING([LibMPU addressable-unit width])
+ AC_COMPUTE_INT([MCPU_CPP_MPU_UNIT_WIDTH], [BITS_PER_UNIT_T],
+ [[#include <libmpuio.h>]],
+ [AC_MSG_ERROR([cannot determine BITS_PER_UNIT_T])])
+ AC_MSG_RESULT([$MCPU_CPP_MPU_UNIT_WIDTH])
+
+ AC_MSG_CHECKING([sizeof(__mpu_size_t)])
+ AC_COMPUTE_INT([MCPU_CPP_SIZEOF_SIZE_T], [sizeof(__mpu_size_t)],
+ [[#include <libmpuio.h>]],
+ [AC_MSG_ERROR([cannot determine sizeof(__mpu_size_t)])])
+ AC_MSG_RESULT([$MCPU_CPP_SIZEOF_SIZE_T])
+
+ AC_MSG_CHECKING([sizeof(__mpu_ssize_t)])
+ AC_COMPUTE_INT([MCPU_CPP_SIZEOF_SSIZE_T], [sizeof(__mpu_ssize_t)],
+ [[#include <libmpuio.h>]],
+ [AC_MSG_ERROR([cannot determine sizeof(__mpu_ssize_t)])])
+ AC_MSG_RESULT([$MCPU_CPP_SIZEOF_SSIZE_T])
+
+ CPPFLAGS="$mcpu_cpp_save_CPPFLAGS"
+ CFLAGS="$mcpu_cpp_save_CFLAGS"
+
+ MCPU_CPP_SIZE_WIDTH=`expr "$MCPU_CPP_SIZEOF_SIZE_T" \* "$MCPU_CPP_MPU_UNIT_WIDTH"`
+ MCPU_CPP_SSIZE_WIDTH=`expr "$MCPU_CPP_SIZEOF_SSIZE_T" \* "$MCPU_CPP_MPU_UNIT_WIDTH"`
+
+ case "$MCPU_CPP_SIZE_WIDTH" in
+ 8|16|32|64|128|256|512|1024|2048|4096|8192|16384|32768|65536) ;;
+ *) AC_MSG_ERROR([unsupported __mpu_size_t width: $MCPU_CPP_SIZE_WIDTH]) ;;
+ esac
+ case "$MCPU_CPP_SSIZE_WIDTH" in
+ 8|16|32|64|128|256|512|1024|2048|4096|8192|16384|32768|65536) ;;
+ *) AC_MSG_ERROR([unsupported __mpu_ssize_t width: $MCPU_CPP_SSIZE_WIDTH]) ;;
+ esac
+
+ if test "x$MCPU_CPP_MPU_BYTE_ORDER" != "x$GCC_BYTE_ORDER"; then
+ AC_MSG_ERROR([LibMPU byte order disagrees with GCC target byte order])
+ fi
+ if test "x$MCPU_CPP_MPU_REGISTER_WIDTH" != "x$MACHINE_REGISTER_WIDTH"; then
+ AC_MSG_ERROR([LibMPU machine-register width disagrees with GCC target])
+ fi
+
+ AC_DEFINE_UNQUOTED([MCPU_CPP_MPU_REAL_IO_LIMIT], [$MCPU_CPP_MPU_REAL_IO_LIMIT],
+ [LibMPU real I/O limit used by mcpu-cpp.])
+ AC_DEFINE_UNQUOTED([MCPU_CPP_MPU_MATH_FN_LIMIT], [$MCPU_CPP_MPU_MATH_FN_LIMIT],
+ [LibMPU math-function limit used by mcpu-cpp.])
+ AC_DEFINE_UNQUOTED([MCPU_CPP_MPU_INT_MAX_WIDTH], [$MCPU_CPP_MPU_INT_MAX_WIDTH],
+ [Maximum LibMPU integer width in bits used by mcpu-cpp.])
+ AC_DEFINE_UNQUOTED([MCPU_CPP_MPU_BYTE_ORDER], [$MCPU_CPP_MPU_BYTE_ORDER],
+ [LibMPU byte order used by mcpu-cpp.])
+ AC_DEFINE_UNQUOTED([MCPU_CPP_MPU_WORD_ORDER], [$MCPU_CPP_MPU_WORD_ORDER],
+ [LibMPU word order used by mcpu-cpp.])
+ AC_DEFINE_UNQUOTED([MCPU_CPP_MPU_REGISTER_WIDTH], [$MCPU_CPP_MPU_REGISTER_WIDTH],
+ [LibMPU machine-register width used by mcpu-cpp.])
+ AC_DEFINE_UNQUOTED([MCPU_CPP_MPU_UNIT_WIDTH], [$MCPU_CPP_MPU_UNIT_WIDTH],
+ [LibMPU addressable-unit width used by mcpu-cpp.])
+ AC_DEFINE_UNQUOTED([MCPU_CPP_SIZEOF_SIZE_T], [$MCPU_CPP_SIZEOF_SIZE_T],
+ [Size of __mpu_size_t in bytes.])
+ AC_DEFINE_UNQUOTED([MCPU_CPP_SIZE_WIDTH], [$MCPU_CPP_SIZE_WIDTH],
+ [Width of __mpu_size_t in bits.])
+ AC_DEFINE_UNQUOTED([MCPU_CPP_SIZEOF_SSIZE_T], [$MCPU_CPP_SIZEOF_SSIZE_T],
+ [Size of __mpu_ssize_t in bytes.])
+ AC_DEFINE_UNQUOTED([MCPU_CPP_SSIZE_WIDTH], [$MCPU_CPP_SSIZE_WIDTH],
+ [Width of __mpu_ssize_t in bits.])
+])
diff --git a/auto-clean b/auto-clean
new file mode 100755
index 0000000..346660a
--- /dev/null
+++ b/auto-clean
@@ -0,0 +1,42 @@
+#!/bin/bash
+
+CWD=`pwd`
+
+program=`basename $0`
+
+usage() {
+ cat << EOF
+
+Usage: $program [options]
+
+Options:
+ -h,--help Display this message.
+
+EOF
+}
+
+if [ -f "${CWD}/Makefile" ] ; then
+ make distclean
+fi
+
+gitignore='.gitignore'
+
+if [ -f "$gitignore" ] ; then
+ while read ln; do
+ line=`echo "${ln}" | sed 's,^[ \t],,' | sed 's,[ \t]$,,'`
+ if [ "x$line" != "x" -a "${line:0:1}" != "#" ] ; then
+ if `echo "${line}" | grep -q '\*~$'` ; then
+ find "`dirname "${line}"`" -type f -iname '*~' -print0 | while IFS= read -r -d '' file ; do
+ rm -f "$file"
+ done
+ elif `echo "${line}" | grep -q '\*'` ; then
+ find "`dirname "${line}"`" -type f -iname "`basename "${line}"`" -print0 | while IFS= read -r -d '' file ; do
+ rm -f "$file"
+ done
+ else
+ if [ -d "${line}" ] ; then rm -rf "${line}" ; fi
+ if [ -f "${line}" ] ; then rm -f "${line}" ; fi
+ fi
+ fi
+ done < ${CWD}/${gitignore}
+fi
diff --git a/bootstrap b/bootstrap
new file mode 100755
index 0000000..0557f2d
--- /dev/null
+++ b/bootstrap
@@ -0,0 +1,106 @@
+#!/bin/bash
+
+CWD=`pwd`
+program=`basename $0`
+
+usage() {
+ cat << EOF_USAGE
+
+Usage: $program [options]
+
+Options:
+ -h,--help Display this message.
+ -d,--target-dest-dir=DIR The target ROOTFS directory
+ [default: DIR=/].
+
+The script regenerates files that do not need to be stored in Git:
+ src/mcpp-expr.c from src/mcpp-expr.zubr using ZUBR 4.1.0;
+ aclocal.m4 and m4 macros using aclocal;
+ config.h.in using autoheader;
+ Makefile.in and helper files using automake;
+ configure using autoconf.
+
+EOF_USAGE
+}
+
+TARGET_DEST_DIR=/
+ACDIR=usr/share/aclocal
+INCDIR=usr/include
+SYSTEM_ACDIR=
+SYSTEM_INCDIR=
+
+while [ 0 ] ; do
+ if [ "$1" = "-h" -o "$1" = "--help" ] ; then
+ usage
+ exit 0
+ elif [ "$1" = "-d" -o "$1" = "--target-dest-dir" ] ; then
+ if [ "$2" = "" ] ; then
+ echo -e "\n${program}: ERROR: --target-dest-dir is not specified.\n"
+ usage
+ exit 1
+ fi
+ TARGET_DEST_DIR="$2"
+ shift 2
+ elif [[ $1 == --target-dest-dir=* ]] ; then
+ TARGET_DEST_DIR="`echo $1 | cut -f2 -d'='`"
+ shift 1
+ else
+ if [ "$1" != "" ] ; then
+ echo -e "\n${program}: ERROR: Unknown argument: $1.\n"
+ usage
+ exit 1
+ fi
+ break
+ fi
+done
+
+if [ ! -d "${TARGET_DEST_DIR}" ] ; then
+ echo -e "\n${program}: ERROR: --target-dest-dir is not a directory.\n"
+ usage
+ exit 1
+fi
+
+# Absolute path:
+if [ "${TARGET_DEST_DIR:0:1}" != "/" ] ; then
+ TARGET_DEST_DIR=${CWD}/${TARGET_DEST_DIR}
+fi
+
+# Remove last '/' char except for the root itself:
+if [ "${TARGET_DEST_DIR}" != "/" -a "${TARGET_DEST_DIR: -1}" = "/" ] ; then
+ len=${#TARGET_DEST_DIR}
+ let "len = len - 1"
+ tmp="${TARGET_DEST_DIR:0:$len}"
+ TARGET_DEST_DIR=${tmp}
+fi
+
+if [ "${TARGET_DEST_DIR}" = "/" ] ; then
+ SYSTEM_ACDIR=/${ACDIR}
+ SYSTEM_INCDIR=/${INCDIR}
+else
+ SYSTEM_ACDIR="${TARGET_DEST_DIR}/${ACDIR}"
+ SYSTEM_INCDIR="${TARGET_DEST_DIR}/${INCDIR}"
+fi
+
+if ! command -v zubr >/dev/null 2>&1 ; then
+ echo -e "\n${program}: ERROR: ZUBR 4.1.0 is required to regenerate src/mcpp-expr.c.\n"
+ exit 1
+fi
+
+(
+ cd "$CWD/src" || exit 1
+ echo -e "\n${program}: producing './src/mcpu-expr.c'"
+ zubr -vl -s -Bmcpp_ -o mcpp-expr.c mcpp-expr.zubr
+) || exit 1
+
+aclocal --install -I m4 --force --system-acdir=${SYSTEM_ACDIR} || exit 1
+autoheader --include=${SYSTEM_INCDIR} || exit 1
+automake --gnu --add-missing --copy --force-missing || exit 1
+autoconf --force || exit 1
+
+################################################################
+# Remove cache and backup files:
+#
+rm -rf autom4te.cache src/z.output *~ src/*~ tests/*~ doc/*~ etc/*~
+#
+# End of Cleanup.
+################################################################
diff --git a/configure.ac b/configure.ac
new file mode 100644
index 0000000..87f5503
--- /dev/null
+++ b/configure.ac
@@ -0,0 +1,190 @@
+AC_PREREQ([2.69])
+
+m4_include([acsite.m4])
+
+AC_INIT([mcpu-cpp], [1.0.2], [], [mcpu-cpp])
+
+AC_MCPU_CPP_REQUIRE_BASH
+
+AC_MCPU_CPP_HEADLINE([mcpu-cpp], [MCPU Languages Preprocessor], [MCPU_CPP_VERSION],dnl
+[Copyright (c) 1998-2026 Andrey V.Kosteltsev])dnl
+
+AC_MSG_CFG_PART(Getting the canonical system type)
+AC_CANONICAL_HOST
+
+AC_MSG_CFG_PART(Init Automake environment)
+AC_CONFIG_SRCDIR([src/main.c])
+AC_CONFIG_HEADERS([config.h])
+AC_CONFIG_MACRO_DIR([m4])
+AM_INIT_AUTOMAKE([foreign dist-xz no-dist-gzip subdir-objects])
+
+dnl With the project convention --prefix=/usr, system configuration belongs
+dnl in /etc rather than /usr/etc unless the caller explicitly selects another
+dnl sysconfdir.
+if test "x$sysconfdir" = 'x${prefix}/etc'; then
+ if test "x$prefix" = xNONE || test "x$prefix" = x/usr; then
+ sysconfdir=/etc
+ fi
+fi
+
+AC_MSG_CFG_PART(Test for GNU Compiler Collection)
+AC_USE_SYSTEM_EXTENSIONS
+AC_PROG_CC
+AC_PROG_LN_S
+MCPU_CPP_REQUIRE_GCC
+
+AC_MSG_CFG_PART(Test for LibMPUIO)
+#MCPU_CPP_CHECK_LIBMPUIO
+dnl ============================================================
+dnl Check for LibMPU
+dnl ============================================================
+AC_CHECK_LIBMPU(
+ [1.0.25],
+ [yes],
+ [
+ AC_DEFINE(
+ [HAVE_LIBMPU],
+ [1],
+ [Define to 1 if LibMPU is available.]
+ )
+ LIBMPU_VERSION=`$LIBMPU_CONFIG --version 2>/dev/null`
+ AC_SUBST([LIBMPU_VERSION])
+ ],
+ [
+ AC_MSG_ERROR([LibMPU version 1.0.25 or later is required.])
+ ]
+)
+
+dnl ============================================================
+dnl Check for LibMPUIO
+dnl ============================================================
+AC_CHECK_LIBMPUIO(
+ [1.0.4],
+ [yes],
+ [
+ AC_DEFINE(
+ [HAVE_LIBMPUIO],
+ [1],
+ [Define to 1 if LibMPUIO is available.]
+ )
+ LIBMPUIO_VERSION=`$LIBMPUIO_CONFIG --version 2>/dev/null`
+ AC_SUBST([LIBMPUIO_VERSION])
+ ],
+ [
+ AC_MSG_ERROR([LibMPUIO version 1.0.4 or later is required.])
+ ]
+)
+
+AC_MSG_CFG_PART(Test for LibMPU ABI)
+MCPU_CPP_CHECK_LIBMPU_ABI
+
+AC_MSG_CFG_PART(Define MCPU installation layout)
+
+mcpu_cpp_prefix="$prefix"
+if test "x$mcpu_cpp_prefix" = xNONE; then
+ mcpu_cpp_prefix="$ac_default_prefix"
+fi
+
+mcpu_cpp_exec_prefix="$exec_prefix"
+if test "x$mcpu_cpp_exec_prefix" = xNONE; then
+ mcpu_cpp_exec_prefix="$mcpu_cpp_prefix"
+fi
+
+mcpu_cpp_save_prefix="$prefix"
+mcpu_cpp_save_exec_prefix="$exec_prefix"
+prefix="$mcpu_cpp_prefix"
+exec_prefix="$mcpu_cpp_exec_prefix"
+eval mcpu_cpp_bindir="\"$bindir\""
+eval mcpu_cpp_libdir="\"$libdir\""
+eval mcpu_cpp_sysconfdir="\"$sysconfdir\""
+prefix="$mcpu_cpp_save_prefix"
+exec_prefix="$mcpu_cpp_save_exec_prefix"
+
+mcpu_cpp_root="$mcpu_cpp_libdir/mcpu"
+mcpu_cpp_private_bindir="$mcpu_cpp_root/bin"
+mcpu_cpp_private_etcdir="$mcpu_cpp_root/etc"
+mcpu_cpp_private_includedir="$mcpu_cpp_root/include"
+
+MCPU_CPP_ROOT="$mcpu_cpp_root"
+MCPU_CPP_BINDIR="$mcpu_cpp_private_bindir"
+MCPU_CPP_ETCDIR="$mcpu_cpp_private_etcdir"
+MCPU_CPP_INCLUDEDIR="$mcpu_cpp_private_includedir"
+AC_SUBST([MCPU_CPP_ROOT])
+AC_SUBST([MCPU_CPP_BINDIR])
+AC_SUBST([MCPU_CPP_ETCDIR])
+AC_SUBST([MCPU_CPP_INCLUDEDIR])
+
+mcpu_cpp_public_link_target="$mcpu_cpp_private_bindir/mcpu-cpp"
+case "$mcpu_cpp_bindir:$mcpu_cpp_private_bindir" in
+ "$mcpu_cpp_exec_prefix"/*:"$mcpu_cpp_exec_prefix"/*)
+ mcpu_cpp_bindir_rel=${mcpu_cpp_bindir#"$mcpu_cpp_exec_prefix/"}
+ mcpu_cpp_target_rel=${mcpu_cpp_private_bindir#"$mcpu_cpp_exec_prefix/"}
+ mcpu_cpp_up=
+ mcpu_cpp_rest=$mcpu_cpp_bindir_rel
+ while :; do
+ mcpu_cpp_up="../$mcpu_cpp_up"
+ case "$mcpu_cpp_rest" in
+ */*) mcpu_cpp_rest=${mcpu_cpp_rest#*/} ;;
+ *) break ;;
+ esac
+ done
+ mcpu_cpp_public_link_target="$mcpu_cpp_up$mcpu_cpp_target_rel/mcpu-cpp"
+ ;;
+esac
+MCPU_CPP_PUBLIC_LINK_TARGET="$mcpu_cpp_public_link_target"
+AC_SUBST([MCPU_CPP_PUBLIC_LINK_TARGET])
+
+AC_DEFINE_UNQUOTED([MCPU_CPP_SYSTEM_CONFIG_FILE],
+ ["$mcpu_cpp_sysconfdir/mcpu/mcpu-cpp.conf"],
+ [Optional system-wide mcpu-cpp configuration override.])
+
+
+AC_MSG_CFG_PART(OUTPUT Substitutions)
+AC_CONFIG_FILES([
+ Makefile
+ src/Makefile
+ tests/Makefile
+ doc/Makefile
+ man/Makefile
+ man/ru/Makefile
+ etc/Makefile
+ etc/mcpu-cpp.conf
+])
+AC_OUTPUT
+
+AC_MSG_CFG_PART(Configuration summary)
+AC_MSG_NOTICE([mcpu-cpp configuration:])
+AC_MSG_NOTICE([ version : $PACKAGE_VERSION])
+AC_MSG_NOTICE([ host : $host])
+AC_MSG_NOTICE([ prefix : $prefix])
+AC_MSG_NOTICE([ install MCPU root : $mcpu_cpp_root])
+AC_MSG_NOTICE([ runtime config : <runtime-root>/etc/mcpu-cpp.conf])
+AC_MSG_NOTICE([ system override : $mcpu_cpp_sysconfdir/mcpu/mcpu-cpp.conf])
+AC_MSG_NOTICE([ runtime MCPU include : <runtime-root>/include])
+AC_MSG_NOTICE([ public command : $mcpu_cpp_bindir/mcpu-cpp])
+AC_MSG_NOTICE([ public symlink : $mcpu_cpp_public_link_target])
+AC_MSG_NOTICE([ mpuio-config : $LIBMPUIO_CONFIG])
+AC_MSG_NOTICE([ LibMPU version : $LIBMPU_VERSION])
+AC_MSG_NOTICE([ LibMPUIO version : $LIBMPUIO_VERSION])
+AC_MSG_NOTICE([ LibMPUIO CFLAGS : $LIBMPUIO_CFLAGS])
+AC_MSG_NOTICE([ LibMPUIO LDFLAGS : $LIBMPUIO_LDFLAGS])
+AC_MSG_NOTICE([ LibMPUIO LIBS : $LIBMPUIO_LIBS])
+AC_MSG_NOTICE([ LibMPU real I/O limit : $MCPU_CPP_MPU_REAL_IO_LIMIT])
+AC_MSG_NOTICE([ LibMPU math fn limit : $MCPU_CPP_MPU_MATH_FN_LIMIT])
+AC_MSG_NOTICE([ LibMPU int max width : $MCPU_CPP_MPU_INT_MAX_WIDTH])
+AC_MSG_NOTICE([ LibMPU byte order : $MCPU_CPP_MPU_BYTE_ORDER])
+AC_MSG_NOTICE([ LibMPU word order : $MCPU_CPP_MPU_WORD_ORDER])
+AC_MSG_NOTICE([ machine reg width : $MCPU_CPP_MPU_REGISTER_WIDTH])
+AC_MSG_NOTICE([ size_t width : $MCPU_CPP_SIZE_WIDTH])
+AC_MSG_NOTICE([ ssize_t width : $MCPU_CPP_SSIZE_WIDTH])
+
+if test -f "config.h"; then
+ echo ""
+ echo "Now you can run:"
+ echo " \`${TB}make${TN}' to compile,"
+ echo " \`${TB}make install${TN}' to make and install ${TB}mcpu-cpp${TN},"
+ echo " \`${TB}make dist${TN}' to create distributable tarballs, or"
+ echo " \`${TB}make distclean${TN}' to clean before configure for another target."
+ echo "Enjoy."
+ echo ""
+fi
diff --git a/doc/Makefile.am b/doc/Makefile.am
new file mode 100644
index 0000000..7f354f3
--- /dev/null
+++ b/doc/Makefile.am
@@ -0,0 +1,3 @@
+EXTRA_DIST = \
+ mcpu-cpp-ru.md \
+ mcpu-cpp-en.md
diff --git a/doc/mcpu-cpp-en.md b/doc/mcpu-cpp-en.md
new file mode 100644
index 0000000..5429456
--- /dev/null
+++ b/doc/mcpu-cpp-en.md
@@ -0,0 +1,2169 @@
+# mcpu-cpp
+
+`mcpu-cpp` is the preprocessor for MCPU programming languages. It is an
+independent component of the LibMPU/LibMPUIO/LibMCPU ecosystem and is not tied
+to the name of any single language: the active language is selected with the
+`#lang` directive.
+
+This document defines the normative behavior of `mcpu-cpp`: its text model,
+directives, macro engine, include pipeline, configuration, diagnostics, and
+dependency generation for MCPU tools.
+
+## 1. Text model
+
+External source files and configuration files are encoded in UTF-8. The UTF-8
+must be valid. For source programs, the check that characters belong to the
+UCS-2 range is performed after comments have been removed. Therefore a valid
+Unicode scalar value above `U+FFFF` is permitted inside a comment, but remains
+an error in program text. After this stage, source text is processed as a
+sequence of `__mpu_char16_t` values. An input UTF-8 BOM is accepted and
+removed. An embedded NUL in a source file is forbidden.
+
+`CRLF` and `CR` line endings are normalized to `LF`.
+
+## 2. Actions performed independently of directives
+
+`mcpu-cpp` performs several transformations before directives are parsed.
+
+### 2.1. Backslash-newline
+
+A `\\` immediately followed by a newline is removed before comments,
+directives, and macros are recognized. For example,
+
+```text
+#defi\
+ne FOO 10\
+20
+```
+
+is equivalent to the logical line
+
+```text
+#define FOO 1020
+```
+
+Physical line numbers continue to contribute to the current source position.
+Unless the user changes that position with `#line`, those physical positions
+are the ones reflected in generated line markers.
+
+### 2.2. Comments
+
+`/* ... */` and `// ...` comments are removed before subsequent processing.
+Where needed to keep adjacent tokens separate, a whitespace separator is
+preserved. If a comment terminates a nonempty line, neither a synthetic
+separator nor whitespace that immediately preceded the comment is retained
+after the comment is removed: the line ends at its last significant
+character. The same rule applies to a multi-line comment that starts after
+program text. If comment removal leaves a line containing only whitespace, the
+line becomes genuinely empty. A comment between two tokens still leaves the
+separator required to keep the tokens from being joined. Newlines are
+preserved so source coordinates are not destroyed.
+
+Comments are not recognized inside string or character constants. In the
+`diff` language, an apostrophe is not treated as the beginning of a character
+constant because it is used in derivative notation.
+
+Within a literal `#include <...>` operand, `/*` and `//` sequences are treated
+as part of the file name.
+
+## 3. Directives and the output stream
+
+A directive begins with `#` when only whitespace or comments precede it on the
+logical line. Whitespace is permitted between `#` and the directive name.
+
+Source-position service information in the output stream uses GNU **line
+markers**:
+
+```text
+# line-number "file-name" [flags]
+```
+
+This is not the input directive `#line`. Entering an included file adds flag
+`1` to the line marker, and returning to the file that contained the
+`#include` adds flag `2`. These values have the same meaning as in GNU CPP:
+`1` means entering a new file and `2` means returning to the previous file.
+Flag `2` is not a nesting count or include level.
+
+For example:
+
+```text
+# 1 "main.c"
+# 1 "defs.h" 1
+...
+# 2 "main.c" 2
+```
+
+The input directive
+
+```text
+#line 62 "main.y"
+```
+
+is not copied to the output stream. It changes the logical values of
+`__LINE__` and `__FILE__` for subsequent text and is represented in output by
+a line marker:
+
+```text
+# 62 "main.y"
+```
+
+The arguments of `#line` undergo macro expansion according to the line-control
+model. If an `#include` follows such a `#line`, the return marker receives flag
+`2`, for example `# 65 "main.y" 2`. A name installed by `#line` becomes the
+logical name used by `__FILE__` and line markers; it does not change the
+directory used to resolve a quoted `#include`.
+
+Preprocessor directives use canonical English names only. Unicode remains fully supported in identifiers, strings, comments, and other user text.
+
+## 4. Header files
+
+The following forms are supported:
+
+```text
+#include "file"
+#include <file>
+#include_next "file"
+#include_next <file>
+#pragma once
+```
+
+For ordinary `#include "file"`, the directory of the **physical** current
+source file is always checked first. A logical name established by `#line`
+does not affect this step. For `#include <file>`, the directory of the current
+file is not checked.
+
+### 4.1. Relocatable MCPU root as an ecosystem-wide principle
+
+Starting with release 0.0.37, the MCPU installation directory **does not
+contain a version number of a particular tool** and is not an absolute runtime
+constant compiled into the binary. A version belongs to `mcpu-cpp`,
+`mcpu-as`, `mcpu-ld`, `mcpu-run`, or a library; it does not define the root of
+the shared MCPU environment.
+
+For a typical configuration:
+
+```text
+./configure --prefix=/usr --libdir=/usr/lib64
+```
+
+`make install` creates:
+
+```text
+/usr/lib64/mcpu/
+├── bin/
+│ └── mcpu-cpp
+├── etc/
+│ └── mcpu-cpp.conf
+├── include/
+│ ├── diff/
+│ ├── dift/
+│ ├── alg/
+│ ├── as/
+│ ├── avm/
+│ └── acs/
+└── lib/ # common directory for future MCPU libraries
+```
+
+The public program name lives in `$bindir`:
+
+```text
+/usr/bin/mcpu-cpp -> ../lib64/mcpu/bin/mcpu-cpp
+```
+
+The absolute `/usr/lib64/mcpu` path is **not part of the MCPU-CPP runtime
+ABI**. It is only the configure-time installation location selected by
+`make install`.
+
+On every normal invocation, MCPU-CPP determines the actual path of its own
+executable through Linux `/proc/self/exe`. The public-command symlink does not
+interfere with this: `/proc/self/exe` names the binary that is actually being
+executed. If `/proc/self/exe` is unavailable, a fallback resolves `argv[0]`
+through `PATH` and `realpath(3)`; there is no fallback to a compiled-in
+configure-time installation root.
+
+For an executable
+
+```text
+<root>/bin/mcpu-cpp
+```
+
+the runtime root is derived as:
+
+```text
+executable = <root>/bin/mcpu-cpp
+executable dir = <root>/bin
+MCPU runtime root = <root>
+```
+
+and the following paths are derived from it automatically:
+
+```text
+<root>/etc/mcpu-cpp.conf
+<root>/include
+```
+
+Therefore the whole tree can be physically moved, for example from
+
+```text
+/usr/lib64/mcpu/
+```
+
+to
+
+```text
+/opt/mcpu-test/
+```
+
+or
+
+```text
+$HOME/devel/mcpu-next/
+```
+
+and `<new-root>/bin/mcpu-cpp` immediately starts using
+`<new-root>/etc/mcpu-cpp.conf` and `<new-root>/include` without being
+reconfigured. The old absolute path is retained neither in runtime defaults
+nor in the installed `mcpu-cpp.conf`.
+
+This is not a preprocessor-specific trick; it is a **general MCPU ecosystem
+principle**. Future `mcpu-as`, `mcpu-ld`, `mcpu-run`, libraries, CRT, and other
+components are expected to share one relocatable root:
+
+```text
+<root>/bin
+<root>/etc
+<root>/include
+<root>/lib
+```
+
+Their own versions may differ. Consistency of a particular MCPU environment is
+defined by all components residing in one runtime tree, not by matching
+version suffixes in directory names.
+
+### 4.2. Runtime defaults, configuration layers, and the system include root
+
+Before reading any configuration file, MCPU-CPP creates the runtime-derived
+value:
+
+```text
+MCPU_CPP_SYSTEM_INCLUDE_PATH = <runtime-root>/include
+```
+
+Configuration layers are then applied in order of increasing priority:
+
+```text
+runtime-derived defaults
+ ↓
+<runtime-root>/etc/mcpu-cpp.conf
+ ↓
+/etc/mcpu/mcpu-cpp.conf
+ ↓
+$HOME/.mcpu/mcpu-cpp.conf
+```
+
+`<runtime-root>/etc/mcpu-cpp.conf` is installed with MCPU-CPP, but deliberately
+does not contain an absolute default `MCPU_CPP_SYSTEM_INCLUDE_PATH`: otherwise
+moving the tree would restore the old path. `/etc/mcpu/mcpu-cpp.conf` is an
+optional machine-wide override; `make install` does not create `/etc/mcpu`.
+`$HOME/.mcpu/mcpu-cpp.conf` is also optional, is not versioned, and has the
+highest configuration priority.
+
+If one variable is defined more than once, the last definition wins, including
+an empty definition. Therefore `MCPU_CPP_SYSTEM_INCLUDE_PATH` remains a fully
+replaceable system root. For example:
+
+```text
+MCPU_CPP_SYSTEM_INCLUDE_PATH = $HOME/mcpu-next/include;
+```
+
+completely replaces the runtime-derived `<runtime-root>/include`. For active
+`#lang "as"`, the following locations are then checked:
+
+```text
+$HOME/mcpu-next/include/as
+$HOME/mcpu-next/include
+```
+
+Standard language subdirectories are always derived by the preprocessor from
+one root; there are no variables named
+`MCPU_CPP_SYSTEM_<LANG>_INCLUDE_PATH`.
+
+An empty effective value:
+
+```text
+MCPU_CPP_SYSTEM_INCLUDE_PATH = ;
+```
+
+removes the configured system stage entirely. A higher-priority configuration
+file may later enable it again with a nonempty value.
+
+`--config-file FILE` applies an explicitly selected file on top of the
+runtime-derived default. `--no-config` disables **only configuration-file
+reading**: `<runtime-root>/etc/mcpu-cpp.conf`, `/etc/mcpu/mcpu-cpp.conf`, and
+`$HOME/.mcpu/mcpu-cpp.conf` are not read, but `<runtime-root>/include` remains
+the standard system root. Only `-nostdinc` removes the effective standard
+system tree from include search for one invocation; an explicit `-isystem`
+still remains a command-line directory.
+
+### 4.3. Normative include-file search order
+
+Search order is part of the MCPU-CPP contract. Explicit command-line
+parameters have priority over persistent configuration. After the optional
+directory of the current physical file, the effective chain is strictly:
+
+```text
+explicit -I
+ ↓
+explicit -isystem
+ ↓
+MCPU_CPP_<LANG>_INCLUDE_PATH
+ ↓
+MCPU_CPP_INCLUDE_PATH
+ ↓
+MCPU_CPP_SYSTEM_INCLUDE_PATH/<lang>
+ ↓
+MCPU_CPP_SYSTEM_INCLUDE_PATH
+ ↓
+explicit -idirafter
+ ↓
+MCPU_CPP_AFTER_INCLUDE_PATH
+```
+
+Entries that are absent or do not contain the requested file are skipped.
+
+`MCPU_CPP_<LANG>_INCLUDE_PATH` denotes user-configurable language-specific path
+lists:
+
+```text
+MCPU_CPP_DIFF_INCLUDE_PATH
+MCPU_CPP_DIFT_INCLUDE_PATH
+MCPU_CPP_ALG_INCLUDE_PATH
+MCPU_CPP_AS_INCLUDE_PATH
+MCPU_CPP_AVM_INCLUDE_PATH
+MCPU_CPP_ACS_INCLUDE_PATH
+```
+
+The user fully controls the names and locations of these directories.
+`MCPU_CPP_INCLUDE_PATH` is a common user path list visible in every language
+state.
+
+`-idirafter` and `MCPU_CPP_AFTER_INCLUDE_PATH` form a common fallback area.
+MCPU-CPP does not automatically derive `<lang>` subdirectories for them. The
+user controls their internal layout and may, for example, write:
+
+```text
+#include <vendor/device.h>
+```
+
+Priority is determined by the semantic class, not by the relative appearance
+of different classes in argv or configuration. Within one class, insertion
+order is preserved.
+
+### 4.4. `#include_next` and wrapper headers
+
+`#include_next` is intended primarily for wrapper headers. It allows a local
+header to precede a system header, adjust local policy, and then continue the
+search for a same-named header along the normative chain without copying the
+system file or using an absolute name.
+
+For example:
+
+```text
+mcpu-cpp -isystem $HOME/mcpu-wrapper ...
+```
+
+with `$HOME/mcpu-wrapper/math.h`:
+
+```text
+#ifndef SOME_SYSTEM_MACRO
+#define SOME_SYSTEM_MACRO temporary_value
+#define REMOVE_SOME_SYSTEM_MACRO 1
+#endif
+
+#include_next <math.h>
+
+#ifdef REMOVE_SOME_SYSTEM_MACRO
+#undef SOME_SYSTEM_MACRO
+#undef REMOVE_SOME_SYSTEM_MACRO
+#endif
+```
+
+If the home configuration also specifies:
+
+```text
+MCPU_CPP_SYSTEM_INCLUDE_PATH = $HOME/mcpu-next/include;
+```
+
+a wrapper found through `-isystem` continues `#include_next` through the
+configured user paths, then through `$HOME/mcpu-next/include/<lang>` and
+`$HOME/mcpu-next/include`. The old system tree at the original installation
+location does not participate. This is the intended way for a system developer
+or tester to work in a private sandbox.
+
+MCPU-CPP stores the exact **physical element of the effective search chain**
+from which the current header was found. `#include_next` starts at the next
+element. The `"file"` and `<file>` forms of `#include_next` are equivalent; the
+directory of the current file is not checked again. If the current file was
+found by ordinary quoted search relative to its containing file and therefore
+has no search-chain provenance, `#include_next` starts at the first element of
+the configured chain.
+
+The operand may be produced by macro expansion. A logical name installed by
+`#line` does not affect physical provenance. If no suitable file exists after
+the current entry, preprocessing fails.
+
+### 4.5. `#pragma once`
+
+An active
+
+```text
+#pragma once
+```
+
+directive marks the **physical file** as already processed during the current
+MCPU-CPP invocation. A later attempt to include the same physical file skips
+its contents. The directive itself is consumed by the preprocessor and is not
+copied to output, including in `-dD` mode.
+
+Identity is determined by the file-system `st_dev`/`st_ino` pair, not by the
+path string. Therefore the same file cannot bypass `#pragma once` by being
+reached as `./file.h`, through a symbolic link, or through another hard-link
+name. A logical name installed by `#line` also has no effect on this physical
+identity.
+
+The mark takes effect immediately when the active directive is processed.
+Therefore a header may include itself after `#pragma once`: the repeated
+include is skipped and recursion does not occur. A directive in an inactive
+conditional branch has no effect.
+
+MCPU-CPP recognizes only the exact `#pragma once` form, with optional
+whitespace. Other `#pragma` directives are not interpreted by the preprocessor
+and are preserved for later compiler stages; for example, `#pragma pack(...)`
+continues to be passed through to output.
+
+`#pragma once` supplements, but does not modify, the normative
+`#include`/`#include_next` search chain. The ordinary search mechanism first
+finds a physical file, then the `once` registry decides whether its contents
+must be processed.
+
+### 4.6. Forced files: `-imacros FILE` and `-include FILE`
+
+The command-line options
+
+```text
+-imacros FILE
+-include FILE
+```
+
+process a file before the primary input. They use the ordinary preprocessing
+engine, not a separate simplified parser.
+
+The normative start-of-translation-unit order is:
+
+```text
+predefined macros
+ -> -D/-U in command-line order
+ -> all -imacros in command-line order
+ -> all -include in command-line order
+ -> primary input
+```
+
+Thus the relative interleaving of `-imacros` and `-include` in `argv` does not
+interleave the two groups: **all** `-imacros` files are always processed before
+**all** `-include` files.
+
+`-imacros FILE` fully preprocesses the file. Its `#define`/`#undef`,
+conditional directives, `#lang`/`#endlang`, `#include`, `#include_next`,
+`#pragma once`, and diagnostics have normal semantics. However, all normal
+preprocessing output from this forced file, including line markers and text
+from nested headers, is discarded. The resulting macro-table state and other
+preprocessing state are retained for later forced files and for the primary
+input.
+
+`-include FILE` uses the same machinery, but preserves normal output, as if the
+located header had been included immediately before the primary source. A
+forced include is a real include boundary: inside it `__INCLUDE_LEVEL__ == 1`,
+inside a header that it includes the level is `2`, and the primary input
+remains at level `0`. `__BASE_FILE__` inside forced files remains the name of
+the primary input.
+
+An absolute forced-file operand is used directly. A relative operand is first
+searched for in the **current working directory**, then along the ordinary
+include chain:
+
+```text
+explicit -I
+explicit -isystem
+MCPU_CPP_<LANG>_INCLUDE_PATH
+MCPU_CPP_INCLUDE_PATH
+MCPU_CPP_SYSTEM_INCLUDE_PATH/<lang>
+MCPU_CPP_SYSTEM_INCLUDE_PATH
+explicit -idirafter
+MCPU_CPP_AFTER_INCLUDE_PATH
+```
+
+The directory of the primary input receives no special priority when resolving
+an `-imacros`/`-include` operand. Once a forced file is found, ordinary quoted
+`#include "file"` inside it is again resolved relative to the physical
+directory of that forced file. If the forced file was found through an element
+of the include chain, its provenance is retained and `#include_next` continues
+at the next chain element.
+
+Forced files and the headers actually reached from them participate in the
+ordinary physical dependency registry. Their user/system classification is
+derived from the same search provenance, so `-MM`/`-MMD` filter system forced
+headers exactly as they filter ordinary system headers. A missing forced file
+is an error.
+
+### 4.7. Dependency generation: `-M`, `-MM`, `-MG`, `-MD`, `-MMD`, `-MF`, `-MT`, `-MQ`
+
+`-M` and `-MM` use **the same include-pipeline pass** as ordinary preprocessing.
+There is no second, independent header search. The dependency graph therefore
+inherits the normative search order, `#include_next`, macro-expanded include
+operands, conditional compilation, and `#pragma once` automatically.
+
+`-M` suppresses normal preprocessing output and emits one Make rule:
+
+```make
+file.o: file.c header1.h header2.h
+```
+
+The list contains the primary source and all physical headers actually reached,
+including system headers. A physical file appears only once. Identity is
+`st_dev + st_ino`, so an alternate relative spelling, symbolic link, or hard
+link does not create a duplicate dependency. A logical name established by
+`#line` is only a source name and never enters the dependency list.
+
+`-MM` builds the same graph but removes system dependencies. System context
+includes headers found through explicit `-isystem`, the configured system tree
+`MCPU_CPP_SYSTEM_INCLUDE_PATH/<lang>` / `MCPU_CPP_SYSTEM_INCLUDE_PATH`, explicit
+`-idirafter`, and `MCPU_CPP_AFTER_INCLUDE_PATH`, as well as the entire branch of
+headers included directly or indirectly from such a system header. The syntax
+`#include "file"` versus `#include <file>` does not by itself determine whether
+a dependency is a system dependency. If one physical file is reached from a
+system branch and is later included directly from user context, it remains a
+user dependency and is present in `-MM` output.
+
+The default target is derived from the basename of the primary source: its
+suffix is replaced with the object suffix (`.o` by default). Paths and the
+default target are Make-quoted. For stdin, the GNU-like form is `-: -`.
+
+`-MD` and `-MMD` use the same dependency graph but, unlike `-M` and `-MM`,
+**do not suppress normal preprocessing output**. `-MD` includes system headers
+like `-M`; `-MMD` applies the user-only filter of `-MM`. One pass can therefore
+produce both preprocessed text and a side-effect dependency file.
+
+If `-MF` is not specified, side-effect mode chooses the `.d` file name
+automatically:
+
+* without `-o`, the input basename loses its suffix and receives `.d`; input
+ pathname directories are not copied into the dependency-file name;
+* with an ordinary `-o FILE`, the output-file suffix is replaced by `.d`;
+* stdin uses `-.d`.
+
+`-MF FILE` overrides the automatic dependency-file name. `-MF -` means stdout.
+`-MF` also works with dependency-only `-M`/`-MM`; in that case it takes
+priority over the ordinary destination of the Make rule. `-MF` by itself,
+without one of `-M`, `-MM`, `-MD`, or `-MMD`, is an error.
+
+The semantics intentionally follow GNU CPP: `-MD`/`-MMD` do not accept their
+own argument; `-MF` is a separate dependency-output option.
+
+`-MT TARGET` replaces the automatic target with `TARGET` **exactly as supplied**.
+No Make quoting is performed. Thus one `-MT` argument may contain multiple
+targets separated by spaces:
+
+```text
+-MT 'obj/a.o obj/a.pic.o'
+```
+
+and repeated `-MT` options also append targets to the same rule:
+
+```text
+-MT obj/a.o -MT obj/a.pic.o
+```
+
+`-MQ TARGET` has the same target-selection semantics but quotes characters that
+are special to Make. For example:
+
+```text
+-MQ '$(OBJDIR)/foo.o'
+```
+
+produces the left-hand side:
+
+```make
+$$(OBJDIR)/foo.o:
+```
+
+Both separate arguments (`-MT TARGET`, `-MQ TARGET`) and attached forms
+(`-MTTARGET`, `-MQTARGET`) are supported.
+
+If at least one `-MT` or `-MQ` is present, the automatic default target is not
+emitted. In particular, `--object-suffix` affects only the automatic target and
+does not rewrite explicit targets. When no explicit target is present, the
+default target is Make-quoted as with `-MQ`.
+
+Repeated and mixed `-MT`/`-MQ` options are allowed. As in GNU CPP, all `-MT`
+targets are emitted first in their command-line order, followed by all `-MQ`
+targets in their command-line order. All of them form the left-hand side of
+**one** dependency rule.
+
+`-MT` and `-MQ` are meaningful only with one of `-M`, `-MM`, `-MD`, or `-MMD`.
+Using either without dependency generation is a command-line error.
+
+`-MG` changes only the handling of **missing** include files in dependency-only
+`-M` and `-MM` modes. Without `-MG`, an unresolved `#include` remains an error.
+With `-M -MG` or `-MM -MG`, a missing header is treated as a future generated
+file: preprocessing does not fail and the directive operand is added to the
+dependency rule **exactly as obtained after macro expansion**, without
+prepending a guessed include directory. For example:
+
+```text
+#include "generated.h"
+```
+
+adds `generated.h` under `-M -MG`, even when that file does not yet exist. A
+macro-expanded include behaves the same way: the dependency receives the
+expanded name. `-MG` is valid only with `-M` or `-MM`; combinations with
+`-MD`/`-MMD`, or use without dependency-only mode, are command-line errors.
+
+Unresolved dependencies are integrated into **the same ordered dependency
+registry** as physical files, but occupy a separate identity domain. For a
+found file, the registry still uses `st_dev/st_ino` and physical provenance.
+For a missing file there is no such information, so an `-MG` entry performs no
+`stat()` and is deduplicated by the exact include-operand text. This is
+essential: an identically named file in the current working directory must not
+turn an unresolved `<name>` into a false physical match when angle search did
+not find that file. Different unresolved spellings, such as `generated.h` and
+`./generated.h`, are distinct dependencies.
+
+For `-MM`, an unresolved dependency gets its user/system class from search
+context: a missing `<file>` is system-class and a missing `"file"` is
+user-class when the including unit is not itself a system header; any missing
+include reached from a system header remains system-class. If the same
+unresolved operand appears more than once, the classification of its first
+occurrence is retained, matching GNU CPP. Physical dependencies keep the
+existing rule that if the same inode is later reached from user context, it is
+no longer system-only.
+
+`-MG` also applies to missing command-line forced files `-include FILE` and
+`-imacros FILE`: their operand enters the unresolved registry as a user
+dependency without a synthetic search prefix. When the file exists,
+`-include`/`-imacros` continue to use the normal physical dependency registry
+and search provenance.
+
+
+## 5. Language switching
+
+The preprocessor starts in language state `0`. This is the unnamed primary
+C-like language and it is not a valid argument of `#lang`.
+
+Supported languages are:
+
+| Name | Purpose |
+|---|---|
+| `diff` | differential equations |
+| `dift` | difference equations |
+| `alg` | algebraic equations |
+| `as` | MCPU assembler (`mcpu-as`) |
+| `avm` | analog-computer schemes |
+| `ACS` | block diagrams of automatic-control systems |
+
+`#lang` must be followed by a string constant containing one nonempty word:
+
+```text
+#lang "diff"
+```
+
+The name is checked only against the internal language list above and is
+compared case-insensitively in ASCII. Thus `"diff"`, `"Diff"`, `"DIFF"`, and
+`"dIfF"` select the same language. The spelling inside the quotes is preserved
+in output.
+
+Whitespace inside the string constant is forbidden: `" diff"`, `"diff "`, and
+`"di ff"` are errors. Escape sequences are not interpreted inside this
+constant. The closing quote must occur on the same physical source line. Only
+whitespace is permitted between the closing quote and the end of the line.
+
+Whitespace outside the string is normalized. For example:
+
+```text
+ # lang "DiFf"
+```
+
+becomes:
+
+```text
+#lang "DiFf"
+```
+
+`#lang` pushes a new language onto the language stack; `#endlang` restores the
+previous language. The stack is not reset by `#include`, so a language block
+may begin and end in different files. `#lang` and `#endlang` remain in the
+output stream for the later frontend dispatcher; `#lang` is emitted in
+normalized form.
+
+## 6. Object-like macro definitions
+
+Starting with 0.0.4, object-like macros are supported:
+
+```text
+#define BUFFER_SIZE 1024
+#define NAME value
+#define EMPTY
+```
+
+The `#define` directive itself is not copied to normal output. In ordinary
+text, a macro identifier is replaced with its replacement list. That
+replacement is rescanned for macro names, so cascaded expansion works:
+
+```text
+#define A B
+#define B 10
+A
+```
+
+produces `10`.
+
+While a particular macro is being expanded, that macro is temporarily
+disabled. Therefore self-referential and mutually recursive definitions do not
+cause infinite recursion.
+
+Macro names are not expanded inside string or character constants. For the
+`diff` language, an apostrophe keeps its language-specific meaning and does
+not protect following text as a C character constant.
+
+A multi-line definition using backslash-newline is supported because splicing
+occurs before `#define` is parsed.
+
+### 6.1. `#undef`
+
+```text
+#undef NAME
+```
+
+removes an object-like macro definition. Undefining a nonexistent macro is not
+an error.
+
+### 6.2. Computed `#include`
+
+An `#include` argument that does not begin directly with `"` or `<` is first
+macro-expanded. Therefore both of these are valid:
+
+```text
+#define HEADER <diff/model.h>
+#include HEADER
+```
+
+and
+
+```text
+#define HEADER "local.h"
+#include HEADER
+```
+
+The expansion result must have the form `"file"` or `<file>`.
+
+## 7. Function-like macros
+
+The macro engine supports function-like macros:
+
+```text
+#define identifier( argument-list ) replacement
+```
+
+The opening parenthesis in the definition must follow the macro name
+**immediately**. Thus
+
+```text
+#define F(X) X
+```
+
+defines a function-like macro, while
+
+```text
+#define F (X)
+```
+
+defines an object-like macro with replacement `(X)`.
+
+At a use site, whitespace is permitted between a function-like macro name and
+the opening parenthesis. If `(` does not follow, the identifier is not a call
+of that macro and remains in output.
+
+For an ordinary function-like macro, the number of actual arguments must equal
+the number of formal parameters. For a variadic macro, every fixed argument
+must be present, while the variadic tail may contain any number of arguments,
+including an empty tail. Nested parentheses are tracked while parsing actual
+arguments; a comma inside them does not separate arguments. Square brackets do
+not have this property; this is part of the adopted macro-expansion semantics.
+
+For example:
+
+```text
+#define min(X, Y) ((X) < (Y) ? (X) : (Y))
+min(1, 2)
+```
+
+produces:
+
+```text
+((1) < (2) ? (1) : (2))
+```
+
+Before substitution, an ordinary actual argument itself undergoes macro
+expansion. Cascaded and nested calls therefore work naturally:
+
+```text
+#define A 7
+#define min(X, Y) ((X) < (Y) ? (X) : (Y))
+min(min(A, 3), 10)
+```
+
+A formal parameter may occur any number of times in the replacement list. An
+expression with side effects in an actual argument may therefore be evaluated
+multiple times by the later compiler; the preprocessor does not attempt to
+repair such source code.
+
+Macros with no formal parameters are supported:
+
+```text
+#define READY() 1
+```
+
+They expand only when called as `READY()` (whitespace between the name and `(`
+is permitted at a use site); the standalone identifier `READY` does not
+expand.
+
+Formal parameter names must be distinct. An unterminated parameter list,
+invalid punctuation, and too few or too many actual arguments are errors.
+
+### 7.1. Stringification `#`
+
+The stringification operator (`#`) is supported for parameters of
+function-like macros:
+
+```text
+#define STR(X) #X
+STR(alpha + beta)
+```
+
+produces:
+
+```text
+"alpha + beta"
+```
+
+Stringification uses the **raw actual argument before macro expansion**. Thus:
+
+```text
+#define A 7
+#define STR(X) #X
+#define XSTR(X) STR(X)
+
+STR(A) -> "A"
+XSTR(A) -> "7"
+```
+
+Leading and trailing whitespace in the argument is removed. Internal
+whitespace sequences are collapsed to one space except inside string/character
+tokens of the active language. Double quotes and backslashes inside quoted
+tokens are escaped so that the result remains one valid string constant.
+
+In a function-like replacement list, `#` must refer, directly or after
+whitespace, to a formal parameter name. Inside a quoted token, `#` is not an
+operator. An empty actual argument is allowed and stringifies as `""`.
+
+### 7.2. Token concatenation `##`
+
+Starting with 0.0.21, token concatenation (`##`) is supported with semantics
+aligned with GNU CPP and the macro engine's `collect_expansion()` /
+`macroexpand()` model. The operator combines two adjacent preprocessing tokens
+into one token, after which the resulting replacement list is rescanned for
+macro expansion.
+
+For example:
+
+```text
+#define CAT(A, B) A ## B
+CAT(foo, bar)
+```
+
+produces `foobar`. Concatenation may form an identifier, preprocessing number,
+or multi-character punctuator. For example:
+
+```text
+CAT(1.5, e3) -> 1.5e3
+CAT(+, =) -> +=
+```
+
+When a formal parameter is directly adjacent to `##`, its actual argument is
+substituted **without preliminary macro expansion**. This is the same raw
+argument principle used by stringification. To expand first and concatenate
+second, use the normal two-level GNU CPP pattern:
+
+```text
+#define AFTERX(X) X_ ## X
+#define XAFTERX(X) AFTERX(X)
+#define TABLESIZE 1024
+#define BUFSIZE TABLESIZE
+
+AFTERX(BUFSIZE) -> X_BUFSIZE
+XAFTERX(BUFSIZE) -> X_1024
+```
+
+An empty actual argument adjacent to `##` behaves as a placemarker: it adds no
+token, and concatenation on that side leaves the remaining operand unchanged.
+If an actual argument contains multiple preprocessing tokens, only the edge
+token directly adjacent to `##` is concatenated; the others are preserved and
+participate in the subsequent rescan.
+
+`#` and `##` may be used in the same function-like macro, for example:
+
+```text
+#define COMMAND(NAME) #NAME | NAME ## _command
+```
+
+Here `#NAME` uses the raw spelling of the argument for stringification, while
+`NAME ## _command` uses the same raw argument for concatenation.
+
+Inside a quoted token, `##` is not an operator. Comments have already become
+whitespace by the time macro expansion occurs, so comments cannot be created
+by concatenating `/` and `*`. Whitespace may originally appear between `##`
+and its operands; it does not participate in the concatenation.
+
+If the two operands do not form one valid preprocessing token, a diagnostic is
+issued and the original tokens are retained; whether whitespace appears
+between them after that diagnostic is not part of the contract. `##` at the
+beginning or end of a replacement list is a macro-definition error.
+
+### 7.3. Variadic macros: `...` and `__VA_ARGS__`
+
+Starting with 0.0.46, variadic function-like macros are supported in the modern
+C99-compatible form:
+
+```text
+#define LOG(...) output(__VA_ARGS__)
+#define LOGF(format, ...) output(format, __VA_ARGS__)
+```
+
+The `...` marker may be the only parameter or the final element after one or
+more fixed parameters. The old GNU extension with a named variadic parameter,
+
+```text
+#define LOG(args...) ...
+```
+
+is intentionally not supported in 0.0.46. `__VA_OPT__` was also not part of
+that particular release.
+
+At invocation, every token after the last fixed parameter, including commas
+that separate those tokens, forms one logical variable argument and is
+substituted for `__VA_ARGS__`. In an ordinary position, that variable argument
+undergoes macro expansion before substitution, just like an ordinary actual
+argument:
+
+```text
+#define A 7
+#define V(...) <__VA_ARGS__>
+#define F(first, ...) first | __VA_ARGS__
+
+V(A, 2, 3) -> <7, 2, 3>
+F(1, A, 3) -> 1 | 7, 3
+```
+
+The variadic tail may be empty. Both
+
+```text
+F(1)
+F(1,)
+```
+
+are valid and substitute an empty `__VA_ARGS__`. This does **not** imply that a
+comma written explicitly in the replacement list is removed automatically.
+For example, with
+
+```text
+#define E(format, ...) output(format, __VA_ARGS__)
+```
+
+`E("ok")` leaves the comma before the empty tail. The historical GNU
+`, ## __VA_ARGS__` comma-swallowing behavior is deliberately outside the 0.0.46
+contract and remains unsupported; modern code should use `__VA_OPT__(,)`.
+
+`__VA_ARGS__` participates in the existing `#` and `##` semantics as a real
+macro parameter. Stringification uses the raw spelling of the whole variadic
+tail:
+
+```text
+#define STRV(...) #__VA_ARGS__
+STRV(A, b + c) -> "A, b + c"
+```
+
+When adjacent to `##`, the variadic argument is likewise substituted without
+prescan; the ordinary placemarker, token-concatenation, and rescan rules then
+apply. For example:
+
+```text
+#define L(...) pre ## __VA_ARGS__
+#define R(...) __VA_ARGS__ ## post
+
+L(fix) -> prefix
+R(fix) -> fixpost
+```
+
+If the variadic argument contains multiple preprocessing tokens, only the edge
+token immediately adjacent to `##` is concatenated and the remaining tokens
+are preserved, exactly as for an ordinary parameter. An empty variadic tail
+next to `##` behaves as a placemarker.
+
+The name `__VA_ARGS__` is reserved for the variable argument and is not
+accepted as an ordinary formal parameter name. `#__VA_ARGS__` is valid only in
+a variadic macro. Dump modes preserve the variadic form of the definition, for
+example:
+
+```text
+#define F(first,...) first | __VA_ARGS__
+```
+
+### 7.4. `__VA_OPT__`
+
+Starting with 0.0.47, variadic macros support the standard conditional fragment
+`__VA_OPT__(pp-tokens)`. If the variable argument contains no preprocessing
+tokens after normal macro substitution, the entire `__VA_OPT__(...)` expands
+to an empty sequence. If the variable argument is nonempty, the parenthesized
+contents participate in the replacement list:
+
+```text
+#define DEBUG(format, ...) \
+ fprintf(stderr, format __VA_OPT__(,) __VA_ARGS__)
+
+DEBUG("ready") -> fprintf(stderr, "ready")
+DEBUG("x=%d", x) -> fprintf(stderr, "x=%d", x)
+```
+
+Emptiness is decided **after expansion of the variable argument**, not from its
+raw spelling. Therefore a macro that itself expands to an empty sequence does
+not activate `__VA_OPT__`:
+
+```text
+#define EMPTY
+#define HAS(...) [__VA_OPT__(yes)]
+
+HAS() -> []
+HAS(EMPTY) -> []
+HAS(token) -> [yes]
+```
+
+The contents of `__VA_OPT__` may contain balanced nested parentheses. The
+closing `)` of the `__VA_OPT__` construct is found with nesting taken into
+account. A nested `__VA_OPT__` inside another `__VA_OPT__` is deliberately
+forbidden.
+
+`__VA_OPT__` is integrated with the existing rules for parameter substitution,
+stringification, token concatenation, placemarkers, and rescan. For example:
+
+```text
+#define X 123
+#define S(...) #__VA_OPT__(__VA_ARGS__)
+#define L(...) pre ## __VA_OPT__(__VA_ARGS__)
+
+S() -> ""
+S(X) -> "123"
+L() -> pre
+L(X) -> pre123
+```
+
+With `#__VA_OPT__(...)`, parameter substitution inside the fragment happens
+first, including prescan of ordinary parameters, but arbitrary macro names in
+the fragment are not additionally rescanned before stringification. Thus:
+
+```text
+#define X 123
+#define S(a, ...) #__VA_OPT__(a X)
+
+S(X, y) -> "123 X"
+```
+
+If a parameter inside `__VA_OPT__` participates directly in an internal `##`,
+prescan is suppressed for that parameter in the usual way; the paste is
+performed before later rescan. An outer `##` adjacent to `__VA_OPT__` receives
+the edge token of the already prepared fragment. An empty `__VA_OPT__` result
+next to `##` behaves as a placemarker.
+
+`__VA_OPT__` is valid only in the replacement list of a variadic function-like
+macro and must immediately introduce a parenthesized fragment. `##` cannot be
+the first or last preprocessing token inside that fragment.
+
+The historical GNU extension
+
+```text
+, ## __VA_ARGS__
+```
+
+is intentionally **not implemented** by `mcpu-cpp`. Use the modern
+`__VA_OPT__(,)` form for a conditional comma. The old GNU named variadic
+parameter form `args...` also remains unsupported.
+
+### 7.5. Whitespace normalization in replacement lists
+
+Starting with 0.0.48, `mcpu-cpp` does not carry alignment whitespace from a
+multi-line macro definition into the expansion result. After `\\` + newline
+has been removed, a whitespace sequence belonging to the replacement list
+itself is canonicalized to one ASCII space. This is particularly important for
+definitions whose backslashes are visually aligned in one column:
+
+```text
+#define TRACE(x) \
+ do \
+ { \
+ output(x); \
+ done(); \
+ } \
+ while( 0 )
+```
+
+Such a definition expands to the compact replacement:
+
+```text
+do { output(x); done(); } while( 0 )
+```
+
+rather than preserving dozens of spaces before each former physical-line
+boundary.
+
+Normalization applies **only to whitespace belonging to the replacement
+list**. `mcpu-cpp` is not a source formatter: whitespace in ordinary input text
+is preserved. Whitespace inside an actual macro argument is likewise not
+reformatted merely because the argument is substituted into a macro:
+
+```text
+#define ID(x) x
+
+ID(a + b) -> a + b
+```
+
+String and character literal contents are preserved verbatim, so:
+
+```text
+#define S "left right"
+```
+
+still contains five spaces inside the string.
+
+The presence of whitespace between preprocessing tokens is preserved as one
+space. This prevents accidental retokenization such as turning `+ +` into
+`++`, `- >` into `->`, or `< <` into `<<`. The `#` and `##` operators,
+placemarkers, `__VA_ARGS__`, `__VA_OPT__`, and later rescan keep their existing
+rules; the policy changes only the amount of ordinary replacement-list
+whitespace.
+
+Dump modes (`-dM`, `-dD`) show the same canonical replacement-list form stored
+in the internal macro table.
+
+### 7.6. Invisible-line compaction and line markers
+
+Starting with 0.0.49, `mcpu-cpp` uses the same model as GNU CPP for vertical
+whitespace: **remove it, but do not forget it**. Source lines that produce no
+output preprocessing token after preprocessing need not remain as physical
+blank lines in the `.E` output, but their source position still contributes to
+line markers and to `__LINE__`.
+
+Why a line is invisible does not matter. It may be a consumed directive, an
+inactive `#if` branch, a single-line or multi-line comment, an ordinary blank
+line, or any mixture of these. The emitter compares its current output source
+position with the position of the next line that will actually be emitted.
+
+If the next position is fewer than eight lines away, the gap is represented by
+ordinary newlines. If the distance is eight lines or greater, the long run of
+blank lines is replaced by a corrective line marker:
+
+```text
+# N "file"
+```
+
+and the next content line immediately belongs to source line `N`. The behavior
+therefore matches the GNU CPP boundary: gaps 0 through 7 use newlines; a gap of
+8 or more uses a line marker.
+
+Structural enter/return markers for included files keep their ordinary
+meaning:
+
+```text
+# 1 "header.h" 1
+# 4 "source.c" 2
+```
+
+If an included file produces no output, `mcpu-cpp` does not invent a marker
+reporting how far the preprocessor progressed internally through that header.
+An enter marker may be followed immediately by its return marker. The real
+position is corrected again only when some following content must be emitted.
+
+This optimization changes only the representation of the output stream.
+Source coordinates, `__LINE__`, diagnostics, `#line`, include enter/return
+semantics, and macro processing remain tied to the logical source stream, not
+to the number of physical lines in the compacted `.E` file.
+
+## 8. Predefined macros
+
+Starting with 0.0.6, the historical predefined-macro mechanism was restored in
+the preprocessor. It is treated as a separate ABI/environment layer for the
+future unnamed C-like language. These definitions are not decorative: their
+names and values must match either GNU CPP semantics or an explicitly
+documented MCPU/LibMPU contract.
+
+### 8.1. Dynamic source macros
+
+The following predefined macros are evaluated at the point of use:
+
+| Macro | Expansion |
+|---|---|
+| `__FILE__` | string constant containing the name of the current input file |
+| `__LINE__` | decimal number of the current source line |
+| `__BASE_FILE__` | string constant containing the primary input file name of the translation unit |
+| `__INCLUDE_LEVEL__` | `#include` nesting level; `0` in the primary file |
+| `__DATE__` | preprocessor start date in the form `"Mmm dd yyyy"` |
+| `__TIME__` | preprocessor start time in the form `"hh:mm:ss"` |
+
+`__DATE__` and `__TIME__` share one timestamp for the whole translation unit.
+Their special expansion is emitted without another macro rescan.
+
+These names reside in the ordinary macro table, so `#undef` followed by
+`#define` may deliberately replace a builtin.
+
+### 8.2. Preprocessor version
+
+Starting with 0.0.8, the standalone preprocessor does not define GCC's
+`__VERSION__`. That name belongs to a compiler environment, which does not yet
+exist for the future high-level language. The version of `mcpu-cpp` has its
+own unambiguous name:
+
+```text
+#define __MCPU_CPP_VERSION__ "1.0.2"
+```
+
+The value is obtained automatically from `PACKAGE_VERSION`. When a compiler
+frontend/driver appears, its version contract will be defined separately and
+will not be mixed with the version of the standalone preprocessor.
+
+### 8.3. ABI sources of truth
+
+`mcpu-cpp` is built only with GNU GCC. During `configure`, the project follows
+the established LibMPU/LibMPUIO `acsite.m4` approach: GCC predefined macros
+describe native type sizes, byte/word order, and machine-register width, while
+the installed `<libmpu.h>` is the final source of truth for LibMPU
+configuration.
+
+In particular, the following values are captured and checked:
+
+```text
+MPU_REAL_IO_LIMIT
+MPU_MATH_FN_LIMIT
+MPU_BYTE_ORDER
+MPU_WORD_ORDER
+BITS_PER_MACHINE_REGISTER
+BITS_PER_UNIT_T
+sizeof(__mpu_size_t)
+sizeof(__mpu_ptrdiff_t)
+```
+
+`configure` additionally verifies that the byte order and
+`BITS_PER_MACHINE_REGISTER` recorded by LibMPU agree with the GCC target used
+to build `mcpu-cpp`. `MPU_WORD_ORDER` is taken directly from the configured
+LibMPU profile and describes word order in the MCPU data environment.
+
+`MPU_REAL_IO_LIMIT` and `MPU_MATH_FN_LIMIT` serve different purposes. For
+example, a library may support Real I/O up to 65536 bits while providing
+mathematical functions only up to 16384 bits. Therefore `MPU_MATH_FN_LIMIT`
+is not used as the limit on existence of Real types.
+
+### 8.4. MCPU architecture and assembler prefixes
+
+The target architecture is identified by:
+
+```text
+#define _ARCH_MCPU 1
+```
+
+MCPU PTR64 is 64 bits wide, so `__SIZEOF_POINTER__`,
+`__MCPU_POINTER_WIDTH__`, `__INTPTR_TYPE__`, `__UINTPTR_TYPE__`, and the
+corresponding width/max macros are defined accordingly.
+
+Assembler-prefix macros follow GNU CPP meaning rather than the first letter of
+a register-view name. `mcpu-as` syntax uses no extra sigil before a register,
+label, or immediate value. The letters `r` and `c` belong to MCPU register
+syntax; they are not a `REGISTER_PREFIX`. Therefore:
+
+```text
+#define __REGISTER_PREFIX__
+#define __LOCAL_LABEL_PREFIX__
+#define __USER_LABEL_PREFIX__
+#define __IMMEDIATE_PREFIX__
+```
+
+all four expand to an empty sequence. `.L...` remains a compiler naming
+convention and is not an assembler-ABI local-label prefix: LOCAL/GLOBAL binding
+is determined by symbol directives.
+
+### 8.5. Byte order and word order
+
+The basic numeric byte-order values are compatible with GNU CPP:
+
+```text
+__ORDER_LITTLE_ENDIAN__
+__ORDER_BIG_ENDIAN__
+__ORDER_PDP_ENDIAN__
+```
+
+The target environment publishes its own MCPU names:
+
+```text
+#define __MCPU_BYTE_ORDER__ __ORDER_LITTLE_ENDIAN__
+#define __MCPU_WORD_ORDER__ __ORDER_LITTLE_ENDIAN__
+#define __BYTE_ORDER__ __MCPU_BYTE_ORDER__
+```
+
+The actual values of `__MCPU_BYTE_ORDER__` and `__MCPU_WORD_ORDER__` come from
+the configured LibMPU profile (`MPU_BYTE_ORDER` and `MPU_WORD_ORDER`). They
+therefore follow the host data representation for which LibMPU was built. This
+does not alter the separate architectural contract for MCPU instruction
+bytecode encoding.
+
+The GNU/C-specific name `__FLOAT_WORD_ORDER__` is not defined because the
+future MCPU language has no `float` type.
+
+LibMPU/MCPU environment parameters are published in the MCPU namespace:
+
+```text
+__MCPU_MACHINE_REGISTER_WIDTH__
+__MCPU_REAL_IO_LIMIT__
+__MCPU_MATH_FN_LIMIT__
+__MCPU_INT_MAX_WIDTH__
+__MCPU_REAL_MAX_WIDTH__
+__MCPU_COMPLEX_MAX_WIDTH__
+```
+
+`__MCPU_INT_MAX_WIDTH__` is `NB_I_MAX * 8`; the Real/Complex maximum width is
+the configured `MPU_REAL_IO_LIMIT`. `__MCPU_MACHINE_REGISTER_WIDTH__` is the
+`BITS_PER_MACHINE_REGISTER` value of the installed LibMPU. Real I/O and math
+limits are deliberately kept separate: `MPU_REAL_IO_LIMIT` controls existence
+of Real/Complex type families and text conversion, while `MPU_MATH_FN_LIMIT`
+controls availability of mathematical functions at a given width.
+
+### 8.6. MCPU size/ssize, `ptrdiff`, and pointers
+
+The future language does not inherit variable-width C names such as `short`,
+`int`, and `long`, and it does not use the C-style name `size_t` as part of its
+own ABI. The unsigned LibMPU size type and signed byte-count/error type are
+published symmetrically in the MCPU namespace. For a 64-bit configured profile,
+for example:
+
+```text
+#define __MCPU_SIZE_TYPE__ uint64
+#define __MCPU_SIZE_WIDTH__ 64
+#define __MCPU_SIZEOF_SIZE__ 8
+#define __MCPU_SIZE_MAX__ 0xffffffffffffffff
+
+#define __MCPU_SSIZE_TYPE__ int64
+#define __MCPU_SSIZE_WIDTH__ 64
+#define __MCPU_SIZEOF_SSIZE__ 8
+#define __MCPU_SSIZE_MAX__ 0x7fffffffffffffff
+```
+
+This is an MCPU-specific family, not an attempt to invent a nonexistent GNU CPP
+`__SSIZE_*` contract.
+
+The MCPU pointer ABI is independent of the host: PTR64 is always 64 bits wide:
+
+```text
+#define __INTPTR_TYPE__ int64
+#define __UINTPTR_TYPE__ uint64
+#define __INTPTR_WIDTH__ 64
+#define __UINTPTR_WIDTH__ 64
+#define __INTPTR_MAX__ 0x7fffffffffffffff
+#define __UINTPTR_MAX__ 0xffffffffffffffff
+#define __SIZEOF_POINTER__ 8
+#define __MCPU_POINTER_WIDTH__ 64
+```
+
+The difference between two MCPU pointers is signed and also fixed independently
+of the host:
+
+```text
+#define __PTRDIFF_TYPE__ int64
+#define __PTRDIFF_WIDTH__ 64
+#define __SIZEOF_PTRDIFF__ 8
+#define __PTRDIFF_MAX__ 0x7fffffffffffffff
+```
+
+Computed MIN expressions such as `(-__PTRDIFF_MAX__ - 1)` are not added to the
+predefined table.
+
+### 8.7. Character types
+
+The future language has no ordinary C `char`. Therefore `__CHAR_TYPE__` and
+`__WCHAR_TYPE__` are not defined. Language types are named without C/C++ `_t`
+suffixes:
+
+```text
+#define __CHAR8_TYPE__ char8
+#define __CHAR16_TYPE__ char16
+#define __CHAR8_WIDTH__ 8
+#define __CHAR16_WIDTH__ 16
+#define __SIZEOF_CHAR8__ 1
+#define __SIZEOF_CHAR16__ 2
+```
+
+These are types of the future language. The implementation of `mcpu-cpp`
+itself continues to use LibMPUIO `__mpu_char16_t` and the strict UCS-2 text
+model internally.
+
+### 8.8. LibMPU integer families
+
+Complete structural metadata for integer families is generated up to the
+actual `NB_I_MAX * 8` of the installed LibMPU rather than stopping at a
+hard-coded final type. For every power-of-two width starting at 8 bits, TYPE,
+WIDTH, and SIZEOF are defined:
+
+```text
+#define __INT1024_TYPE__ int1024
+#define __UINT1024_TYPE__ uint1024
+#define __INT1024_WIDTH__ 1024
+#define __UINT1024_WIDTH__ 1024
+#define __SIZEOF_INT1024__ 128
+#define __SIZEOF_UINT1024__ 128
+```
+
+With the current LibMPU 1.0.25, `NB_I_MAX == 8192`, so the family extends to
+`int65536`/`uint65536`, with `__SIZEOF_INT65536__ == 8192`.
+
+Decimal-digit metadata is defined for **every** permitted integer width:
+
+```text
+__INT<bits>_DECIMAL_DIG__
+__UINT<bits>_DECIMAL_DIG__
+```
+
+The value is computed by `mcpu-cpp` integer-only helpers from the known bit
+width. It is the exact number of decimal digits in the maximum value of the
+type; neither a sign nor a terminating NUL is included in `DECIMAL_DIG`. For
+unsigned values the maximum is `2^bits - 1`; for signed values it is
+`2^(bits-1) - 1`. This differs from LibMPU `_int_digs()`, which estimates a
+string-buffer size and includes room for a terminating NUL.
+
+For example:
+
+```text
+#define __INT64_DECIMAL_DIG__ 19
+#define __UINT64_DECIMAL_DIG__ 20
+#define __INT256_DECIMAL_DIG__ 77
+#define __UINT256_DECIMAL_DIG__ 78
+```
+
+Only the textual maxima themselves are deliberately limited to widths
+`bits <= 256`:
+
+```text
+__INT128_MAX__
+__UINT128_MAX__
+```
+
+Maxima are produced through LibMPU `iuitoa()`. Macros named
+`__INT<bits>_MIN__` are not generated: the predefined table must not contain
+computed expressions such as `(-__INT<bits>_MAX__ - 1)`. For widths above 256
+bits, only MAX is absent; TYPE/WIDTH/SIZEOF/DECIMAL_DIG continue through the
+full `NB_I_MAX * 8` range.
+
+### 8.9. LibMPU Real and Complex families
+
+Real/Complex structural metadata is generated for every power-of-two width
+from 32 bits through the actual configured `MPU_REAL_IO_LIMIT`. TYPE, WIDTH,
+and SIZEOF are published for all of these types.
+
+For Complex, WIDTH denotes the type parameter, not total storage width:
+
+```text
+#define __COMPLEX128_TYPE__ complex128
+#define __COMPLEX128_WIDTH__ 128
+#define __SIZEOF_COMPLEX128__ 32
+```
+
+`complex128` consists of two `real128` components, so its storage size is 32
+bytes. With `MPU_REAL_IO_LIMIT == 65536`, the top of the family is:
+
+```text
+#define __COMPLEX65536_TYPE__ complex65536
+#define __COMPLEX65536_WIDTH__ 65536
+#define __SIZEOF_COMPLEX65536__ 16384
+```
+
+For Real:
+
+```text
+#define __REAL65536_TYPE__ real65536
+#define __REAL65536_WIDTH__ 65536
+#define __SIZEOF_REAL65536__ 8192
+```
+
+Precision metadata is defined for **all** allowed Real widths up to
+`MPU_REAL_IO_LIMIT`. Macro names correspond directly to LibMPU helpers:
+
+```text
+__REAL<bits>_DECIMAL_DIG__ -> _real_digs(bits/8)
+__REAL<bits>_MANT_DIG__ -> _real_mant_digs(bits/8)
+```
+
+`__REAL<bits>_DIG__` is intentionally absent. The `bits <= 256` restriction
+applies only to large textual numeric constants. For widths up to 256 bits,
+the following are also defined:
+
+```text
+__REAL<bits>_MAX__
+__REAL<bits>_MIN__
+__REAL<bits>_EPSILON__
+__REAL<bits>_MAX_EXP__
+__REAL<bits>_MIN_EXP__
+__REAL<bits>_MAX_10_EXP__
+__REAL<bits>_MIN_10_EXP__
+```
+
+For example, with LibMPU 1.0.25, the current `real128` profile gives values of
+the form:
+
+```text
+#define __REAL128_EPSILON__ 2.524354896707237777317531409e-29
+#define __REAL128_MAX__ 4.197157432934775384808581951e+323228496
+#define __REAL128_MIN__ 9.530259619551804292864984035e-323228497
+#define __REAL128_MAX_10_EXP__ 323228496
+#define __REAL128_MAX_EXP__ 1073741823
+#define __REAL128_MIN_10_EXP__ -323228524
+#define __REAL128_MIN_EXP__ -1073741822
+```
+
+MAX/MIN/EPSILON are created by LibMPU itself and converted through
+`real_to_ascii()`. Exponent constants are obtained from LibMPU exponent helpers
+and integer conversion. For widths above 256 bits, these numeric predefines are
+absent, but TYPE/WIDTH/SIZEOF/DECIMAL_DIG/MANT_DIG continue through
+`MPU_REAL_IO_LIMIT`.
+
+For every supported Real type through `MPU_REAL_IO_LIMIT`, two compact
+characteristics are also published:
+
+```text
+#define __SIZEOF_REAL128_EXP__ 4
+#define __REAL128_MAX_STRLEN__ 60
+```
+
+`__SIZEOF_REALxxx_EXP__` is obtained directly from `_sizeof_exp(NB_Rxxx)`.
+`__REALxxx_MAX_STRLEN__` comes from `_real_max_string(NB_Rxxx)` and is the
+maximum **number of characters** in the textual representation, not a byte
+count. A zero-terminated string therefore needs at least
+`__REALxxx_MAX_STRLEN__ + 1` elements: for `char8` that is the same number of
+bytes, while for `char16` the physical byte count is twice as large. These two
+metadata macros are also defined for Real widths above 256 bits because their
+own values remain small.
+
+### 8.10. Macro dumps: `-dM`, `-dMP`
+
+The command:
+
+```text
+mcpu-cpp -dM input.c
+```
+
+prints only final **non-predefined** macros in `#define ...` form. This group
+includes definitions from the primary file and included headers, as well as
+command-line `-D` definitions. MCPU-CPP's own predefined macros are not printed
+by `-dM`. This mode is therefore intended primarily for a compact inspection of
+macro state created by the user program.
+
+The command:
+
+```text
+mcpu-cpp -dMP input.c
+```
+
+adds active MCPU-CPP predefined macros to the same final state. Output contains
+two consecutive groups: predefined macros first, then non-predefined macros.
+Definitions inside each group are sorted deterministically by name. This is
+useful for system development because it exposes the preprocessing ABI and
+architectural properties of the current MCPU environment without mixing them
+with user definitions.
+
+Group membership is determined by macro origin, not by spelling. A macro
+created by `-D` or `#define` is ordinary even if its name looks system-like. If
+a predefined macro is removed with `#undef`, it is not printed. If the user
+then defines the same name again, the new definition belongs to the ordinary
+group and appears in the corresponding part of `-dMP`, and also in `-dM`.
+Thus both modes display the **final macro state**.
+
+Context-dependent `__FILE__`, `__LINE__`, `__DATE__`, `__TIME__`,
+`__BASE_FILE__`, and `__INCLUDE_LEVEL__` are not printed by the static dump.
+Static ABI/architecture predefined macros and computed static Real metadata are
+printed by `-dMP`.
+
+When an input file is supplied, it is fully preprocessed first and the final
+macro state is printed afterward; ordinary preprocessed text is not emitted in
+`-dM`/`-dMP` modes. Without an input file, stdin is used, so empty stdin with
+`-dM` gives an empty dump while `-dMP` provides the active static predefined
+macros of the current MCPU environment.
+
+`-dD` has different semantics and is unaffected by this distinction.
+
+### 8.11. Definition dump: `-dD`
+
+The command:
+
+```text
+mcpu-cpp -dD input.c
+```
+
+preserves ordinary preprocessing output and additionally emits encountered
+`#define` directives. Before primary input starts, static predefined macro
+definitions are printed. Each such definition is preceded by a marker:
+
+```text
+# 0 "<built-in>"
+#define NAME value
+```
+
+and the predefined block itself is preceded by an input-file marker of the form
+`# 0 "input.c"`. Context-dependent `__FILE__`, `__LINE__`, `__DATE__`,
+`__TIME__`, `__BASE_FILE__`, and `__INCLUDE_LEVEL__` are not included in the
+initial built-in block.
+
+### 8.12. Configuration dump: `-dconfig`
+
+The command:
+
+```text
+mcpu-cpp -dconfig
+```
+
+requires no input file and prints the effective configuration-variable layer
+after runtime configuration, optional system override, home user override, or
+a selected `--config-file` have been read, including `$NAME`/`${NAME}`
+expansion. Lines are sorted by name and printed as:
+
+```text
+NAME = value;
+```
+
+This makes it possible to inspect actual include paths without manually
+searching `<runtime-root>/etc/mcpu-cpp.conf`, `/etc/mcpu/mcpu-cpp.conf`, and
+`$HOME/.mcpu/mcpu-cpp.conf`.
+
+### 8.13. Verbose configuration snapshot: `-v`
+
+With `-v`, MCPU-CPP retains its runtime trace for `#lang`, `#include`, and
+`#include_next`, but configuration variables are printed only once, after all
+configuration layers have been read and priority rules applied. Verbose output
+therefore shows only **effective values**; intermediate values from runtime
+root, system, and user configuration are not duplicated.
+
+The configuration block follows include-policy order: language-specific user
+paths, the common user path, the system root, and the AFTER path. A variable
+that is absent from every configuration layer is not printed. The
+runtime-derived default `MCPU_CPP_SYSTEM_INCLUDE_PATH` is a full lowest-priority
+value and is therefore visible under `-v` even when no `mcpu-cpp.conf` exists
+**or all configuration files are disabled with `--no-config`**.
+
+The line form is:
+
+```text
+config: NAME=value
+```
+
+### 8.14. Effective search directories: `-dsearch-dirs`
+
+The command:
+
+```text
+mcpu-cpp -dsearch-dirs
+```
+
+requires no input file, prints the effective global search directories, and
+exits without preprocessing. The format is intentionally simple:
+
+```text
+search: /path/to/directory
+```
+
+Directories are printed in semantic search-class order:
+
+```text
+explicit -I
+explicit -isystem
+configured language-specific user directories
+MCPU_CPP_INCLUDE_PATH
+MCPU_CPP_SYSTEM_INCLUDE_PATH/<lang>
+MCPU_CPP_SYSTEM_INCLUDE_PATH
+explicit -idirafter
+MCPU_CPP_AFTER_INCLUDE_PATH
+```
+
+Language-specific entries are printed for every supported language in their
+canonical order. During a real `#include`, only the directory corresponding to
+the active `#lang` participates. The directory of the current physical file is
+not printed by `-dsearch-dirs`: it exists only dynamically for a particular
+`#include "..."` and changes with the include stack. `--no-config` does not
+remove the runtime-derived system root, so even without configuration files the
+dump still contains `<runtime-root>/include/<lang>` and
+`<runtime-root>/include`. `-nostdinc` removes the effective system `<lang>`
+entries and system root from the dump, but not explicit `-isystem`. A directory
+that does not exist in the file system is still displayed because it remains
+part of the effective search configuration and will simply be skipped during a
+real file search.
+
+`-dsearch-dirs` accounts for `-I`, `-isystem`, `-idirafter`, every
+configuration layer, and the replacement semantics of
+`MCPU_CPP_SYSTEM_INCLUDE_PATH`. Using `-o` with this action is an error.
+
+### 8.15. Conditional compilation
+
+The directives `#if`, `#ifdef`, `#ifndef`, `#elif`, `#else`, and `#endif` are
+processed as preprocessor control directives and are never copied to the
+output stream, including under `-dD`. Inactive branches are skipped without
+executing `#define`, `#undef`, or `#include` directives within them; nested
+conditional groups are still tracked correctly.
+
+An `#if` expression first processes the `defined` operator, then undergoes macro
+expansion, and any remaining identifiers evaluate to `0`. Arithmetic,
+bitwise, comparison, and logical operators are supported, as are `?:` and
+short-circuit semantics for `&&`, `||`, and `?:`.
+
+Starting with 0.0.26, expression syntax is parsed by a parser generated by ZUBR
+4.1.0 from `src/mcpp-expr.zubr`; the same file contains the UCS-2 lexical
+analyzer. `defined` preprocessing and macro expansion take place before the
+parser is entered. Arithmetic semantics live in `mcpp-semantic.c/h` and do not
+depend on the integer sizes of the host system. Generated `mcpp-expr.c` is
+included in releases, so ZUBR is required only when the grammar changes.
+
+#### 8.15.1. The only evaluation width is 64 bits
+
+MCPU-CPP is a preprocessor, not a general-purpose language compiler. All
+integer computation in conditional directives uses only 64-bit arithmetic.
+The preprocessor does not perform arbitrary-width LibMPU arithmetic, floating
+point, or complex-number computation.
+
+When a programmer does not need explicit control over the binary representation
+of a literal, ordinary integer constants with optional `U`/`u` are sufficient.
+For example:
+
+```c
+#if 2 > 1
+#if 0xffffffffffffffffU > 1
+```
+
+A numeric lexeme remains in UCS-2 until classification, after which its ASCII
+portion is passed to LibMPU `iatoui()`. Binary `0b...`, octal `0...`, decimal,
+and hexadecimal `0x...` forms are supported. A value that does not fit in 64
+bits is an error. Old C suffixes `L`, `l`, `LL`, and `ll` are not supported.
+
+#### 8.15.2. Width suffix `zNNN[Uu]`
+
+MCPU-CPP understands the width suffix shared by MCPU languages:
+
+```text
+zNNN
+ZNNN
+zNNNu
+zNNNU
+ZNNNu
+ZNNNU
+```
+
+`NNN` is a nonempty sequence of decimal digits and is **always** interpreted in
+decimal, even with leading zeroes. Thus `z8`, `z08`, and `z008` all denote the
+same width of 8 bits.
+
+In the general MCPU syntax, a valid width must be a power of two from 8 through
+`MPU_REAL_IO_LIMIT`. MCPU-CPP, however, deliberately limits evaluation to 64
+bits:
+
+* `z8`, `z16`, `z32`, `z64`, in either letter case, are valid;
+* `NNN > 64` is immediately an error: conditional preprocessing does not accept
+ numeric constants wider than 64 bits;
+* if `NNN <= 64` but is not a valid power-of-two width, such as `z24`, a warning
+ is issued and the `zNNN` part itself is ignored;
+* a following optional `U`/`u` selects unsigned interpretation and retains that
+ meaning even when an invalid `zNNN` has been ignored.
+
+The numeric preprocessing token must end after the complete suffix. An
+operator or punctuation character begins the next token, so `1z32u+2`,
+`(1z32u)`, and `1z32u==1` are valid. Forms such as `1z32undefined`,
+`1z32ufoo`, and `1z32$foo` are errors and are not artificially split into a
+number followed by a name.
+
+#### 8.15.3. Literal normalization
+
+The width suffix acts **exactly once, while the value of the literal itself is
+formed**. The width is not retained in the semantic value and has no role in
+later operations.
+
+For `VALUEzNNN`, the value is treated as a signed N-bit two's-complement number:
+
+1. retain the low `NNN` bits;
+2. sign-extend the result to 64 bits.
+
+For `VALUEzNNNu`/`VALUEzNNNU`, the low `NNN` bits are retained and then
+zero-extended to 64 bits.
+
+For example:
+
+```text
+0x7fz8 -> 0x000000000000007f -> 127
+0x80z8 -> 0xffffffffffffff80 -> -128
+0xffz8 -> 0xffffffffffffffff -> -1
+0x80z8u -> 0x0000000000000080 -> 128
+0xffz8u -> 0x00000000000000ff -> 255
+0x1ffz8 -> 0xffffffffffffffff -> -1
+0x1ffz8u -> 0x00000000000000ff -> 255
+```
+
+The last two examples are deliberate: `zNNN` specifies the width of the
+**binary representation**, not a mathematical range check. Bits above N are
+discarded before extension.
+
+After this normalization there is no remaining `z8`, `z16`, or `z32` concept
+in the evaluation model. The internal value contains only a 64-bit bit pattern
+and signed/unsigned state.
+
+#### 8.15.4. All subsequent operations are 64-bit
+
+After literal normalization, every arithmetic, bitwise, comparison, and logical
+operation uses 64-bit operands. An operation result is not truncated back to
+the width of the original suffix. Therefore:
+
+```text
+0x7fz8 + 1 -> 128
+0xffz8u + 1 -> 256
+```
+
+not `-128` and `0`. Likewise `~0xffz8u` inverts all 64 bits and gives
+`0xffffffffffffff00`.
+
+For binary operations where signedness matters, the presence of an unsigned
+operand selects 64-bit unsigned interpretation. Comparisons return `0` or `1`.
+Logical `!`, `&&`, and `||` also return signed 64-bit `0` or `1`; short-circuit
+evaluation does not evaluate an unselected operand.
+
+Shifts happen after 64-bit normalization. Right shift of a negative signed
+value is arithmetic; right shift of an unsigned value is logical. For example:
+
+```text
+0x80z8 >> 1 -> -64
+0x80z8u >> 1 -> 64
+```
+
+The historical MCPU-CPP rule for a negative shift count is preserved:
+`A << -N` is equivalent to `A >> N`, and `A >> -N` is equivalent to `A << N`.
+
+Thus `zNNN` does not turn the preprocessor into a compiler with integer
+promotions over multiple widths. It only allows the binary representation of
+the source literal to be stated explicitly; the expression then evaluates in
+one simple 64-bit model.
+
+#### 8.15.5. Character constants
+
+A character unit has type `__mpu_uint16_t`, matching the internal UCS-2
+representation, and is zero-extended to 64 bits before evaluation. Subsequent
+arithmetic is again ordinary 64-bit arithmetic.
+
+Conditional-compilation state is stored on a separate stack; a conditional
+group may not cross an include-file boundary.
+
+### 8.16. Diagnostic directives `#error` and `#warning`
+
+MCPU-CPP supports the standard diagnostic directives:
+
+```text
+#error message
+#warning message
+```
+
+`#error` emits an error diagnostic using the current logical file name and line
+number and immediately terminates preprocessing unsuccessfully. `#warning`
+emits a warning with the same source-location information and preprocessing
+continues. A preceding `#line` therefore affects both diagnostics.
+
+The remainder of the line after the directive name **does not undergo macro
+expansion**. For example:
+
+```c
+#define MESSAGE expanded
+#warning MESSAGE
+```
+
+prints `MESSAGE`, not `expanded`. This distinguishes diagnostic directives from
+`#if` and `#line`, where macro expansion is part of the relevant contract.
+
+Comments are removed by the ordinary preprocessing phase before the directive
+is processed. Leading and trailing whitespace in the message is removed and
+whitespace sequences between preprocessing tokens are collapsed to one space.
+Whitespace inside quotes is preserved. For example:
+
+```c
+#warning one /* comment */ two
+#warning "a b"
+```
+
+produce `one two` and `"a b"`, respectively. Unicode text passes through the
+internal UCS-2 representation and is written to the external diagnostic as
+UTF-8.
+
+Both directives are control directives and are never copied to normal output
+or to `-dD`. In an inactive `#if` branch they are ignored completely, so the
+usual protective pattern behaves as expected:
+
+```c
+#if 0
+#error this error is inactive
+#endif
+```
+
+### 8.17. Warning control: `-Wcomment`, `-Wall`, `-Werror`
+
+MCPU-CPP distinguishes mandatory warnings that are part of established
+preprocessing semantics from optional warning classes enabled by the user.
+Warning control does not alter `-dD`, macro expansion, conditional compilation,
+or include search semantics.
+
+`-Wcomment` and `-Wcomments` are exact aliases and enable two lexical warnings:
+
+* a `/*` sequence seen while already inside an open `/* ... */` comment;
+* backslash-newline inside a `//` comment, causing that single-line comment to
+ continue physically onto the next source line.
+
+This optional class is disabled by default. `-Wall` enables all optional
+MCPU-CPP warning classes; in version 0.0.40 this class is `-Wcomment`.
+`-Wno-comment` and `-Wno-comments` disable it. As in the GNU warning model, a
+more specific setting has priority over a group setting regardless of argument
+order. Therefore both:
+
+```text
+mcpu-cpp -Wall -Wno-comment file.c
+mcpu-cpp -Wno-comment -Wall file.c
+```
+
+leave comment warnings disabled. Between settings of equal specificity, the
+last option wins; for example `-Wno-comment -Wcomment` enables the class.
+
+`-Werror` does not enable any new warning class. It promotes to an error every
+warning that would actually be emitted during that invocation, causing an
+unsuccessful result. This applies both to optional comment warnings and to
+existing mandatory MCPU-CPP warnings, including:
+
+* an active `#warning` directive;
+* an invalid `zNNN` width not exceeding 64 bits;
+* redefinition of a macro with a different replacement list;
+* a `##` result that does not form a single preprocessing token.
+
+For example:
+
+```text
+mcpu-cpp -Wcomment -Werror file.c
+```
+
+turns a detected comment warning into an error. By contrast, `-Werror` alone,
+without `-Wcomment`/`-Wall`, does not cause MCPU-CPP to search for optional
+comment warnings.
+
+`-Wno-error` restores ordinary warning severity. Between `-Werror` and
+`-Wno-error`, which have the same specificity, the last command-line option
+wins. Thus `-Werror -Wno-error` leaves warnings as warnings, while
+`-Wno-error -Werror` promotes them again.
+
+Version 0.0.40 deliberately did not introduce `-Werror=<class>`,
+`-Wno-error=<class>`, `-Wundef`, `-Wunused-macros`, `-Wtraditional`, or other
+compiler-oriented classes. The MCPU-CPP warning interface remains compact and
+is extended only when a class is actually needed by the preprocessing
+language itself.
+
+### 8.18. UCS-2 identifiers
+
+Starting with 0.0.22, preprocessing identifiers are no longer restricted to
+ASCII. Inside `mcpu-cpp`, text is already strict UCS-2, and characters are
+classified by locale-independent LibMPUIO 1.0.4 functions based on Unicode
+18.0.0. The first identifier character must be `_` or have the `XID_Start`
+property; following characters must be `_`, `$`, or have `XID_Continue`.
+`$` is an `mcpu-cpp` extension: it is allowed only after the first character
+and may not start an identifier. One rule is used consistently for macro names
+and parameters, `#undef`, `#ifdef`/`#ifndef`, `defined`, ordinary macro
+expansion, and `#`/`##`. Names remain case-sensitive. Surrogate code units
+`U+D800..U+DFFF` are not valid identifier characters.
+
+For example, all of these are valid:
+
+```c
+#define АНДРЕЙ 1
+#define résumé 2
+#define ΩМЕГА 3
+#define VALUE$OLD 4
+```
+
+`VALUE$OLD` is valid, while `$VALUE` is invalid because `$` is not an
+identifier-start character.
+
+Combining marks and non-ASCII decimal digits may appear in `XID_Continue`
+positions but do not automatically become valid initial characters. Numeric
+constant syntax is unaffected: it follows the rules of the active language,
+not Unicode `isdigit`.
+
+### 8.19. Command-line macros `-D` and `-U`
+
+Starting with 0.0.23, `-D` and `-U` are full preprocessor actions. Supported
+forms are:
+
+```text
+-DNAME
+-DNAME=VALUE
+-D'FUNC(a,b)=a+b'
+-UNAME
+```
+
+`-DNAME` is equivalent to `#define NAME 1`; an `=` with an empty right-hand
+side defines an empty replacement list. Function-like command-line definitions
+use the same macro engine as ordinary `#define`, including parameters, `#`,
+`##`, and subsequent rescanning. `-U` uses the same identifier contract as
+`#undef`. `-D`/`-U` actions are executed in command-line order after predefined
+macros have been installed.
+
+Only the payload of `-D` and `-U` is interpreted as UTF-8 and converted to
+strict UCS-2. File names, `-I`, other pathname arguments, and all other
+command-line arguments remain the original byte strings and undergo no Unicode
+conversion.
+
+Starting with 0.0.25, `$` is allowed inside a macro name, but not in its first
+position. When `$` is passed through a shell, the user must account for shell
+rules: the shell processes `$` **before `mcpu-cpp` starts**. Single quotes
+fully protect `$`, for example:
+
+```sh
+mcpu-cpp '-DАНДРЕЙ$_Y=62' input.c
+```
+
+Without quotes, `$` must be escaped:
+
+```sh
+mcpu-cpp -DАНДРЕЙ\$_Y=62 input.c
+```
+
+or double quotes may be used with escaping:
+
+```sh
+mcpu-cpp -D"АНДРЕЙ\$_Y=62" input.c
+```
+
+The unprotected form:
+
+```sh
+mcpu-cpp -DАНДРЕЙ$_Y=62 input.c
+```
+
+does not pass the spelling literally: `$...` is expanded by the shell first,
+and `mcpu-cpp` receives the already modified `argv`. Inside single quotes, a
+backslash before `$` is unnecessary and would become an ordinary argument
+character.
+
+For command-line `-D`, the left-hand side up to the first `=` is parsed as a
+separate macro declarator. If an invalid tail occurs after a valid name (or a
+completed formal-parameter list of a function-like macro) but before `=`, that
+tail is silently discarded and **never becomes part of the replacement list**.
+For example:
+
+```text
+-D'АНДРЕЙ@XYZ=62'
+```
+
+is equivalent to:
+
+```c
+#define АНДРЕЙ 62
+```
+
+not the invalid `#define АНДРЕЙ @XYZ 62`. The same valid-identifier-prefix rule
+applies to `-U`. If the very first character is not a valid identifier-start
+character, such as `$` or a digit, the definition remains an error.
+
+Under `-dD`, definitions originating from `-D` are marked separately from
+predefined macros:
+
+```text
+# 0 "<command-line>"
+#define NAME value
+```
+
+while predefined macros continue to use `<built-in>`.
+
+### 8.20. Public command-line interface
+
+`mcpu-cpp` supports only the current options documented by `--help`. Obsolete
+compatibility flags do not form a hidden interface and are diagnosed as
+`unknown option`. `-E` is the exception: it is silently accepted and ignored
+because a compiler driver may pass it while invoking a standalone
+preprocessor.
+
+`--object-suffix SUFFIX` selects the object-target suffix used when generating
+Make dependencies; the argument is mandatory.
+
+## 9. Build system and generators
+
+Project-owned Autoconf macros live in the root `acsite.m4`. The `m4/` directory
+is reserved for external/vendor M4 files. This follows the convention used by
+MCPU libraries and keeps project configure code separate from imported macros.
+
+The `#if` expression parser is generated by ZUBR 4.1.0 from
+`src/mcpp-expr.zubr`. Release archives contain both the grammar and the already
+generated `src/mcpp-expr.c`, so an ordinary release build does not require
+ZUBR. After the grammar changes, a developer build uses the normal Automake
+rule:
+
+```text
+zubr -vl -s -Bmcpp_ -o mcpp-expr.c mcpp-expr.zubr
+```
+
+Before a release, generated C must correspond to the grammar, and the full test
+suite plus `make distcheck` must complete without errors.
+
+### 9.1. Developer bootstrap and the Git source tree
+
+Starting with 0.0.50, the root `./bootstrap` script makes it unnecessary to
+store files in Git when they are completely reproducible from source. The
+script first generates `src/mcpp-expr.c` from `src/mcpp-expr.zubr` using ZUBR
+4.1.0, then runs `aclocal`, `autoheader`, `automake`, and `autoconf` in the
+style used by LibMPU and LibMPUIO. `--target-dest-dir=DIR` selects a target
+ROOTFS used for the system Autoconf macro/include directories.
+
+This convention applies specifically to the developer Git tree. **Release
+archives remain self-contained**, exactly as before: they contain `configure`,
+`Makefile.in`, Automake helper scripts, and the generated `src/mcpp-expr.c`.
+Therefore an ordinary release build requires neither a preliminary `bootstrap`
+run nor ZUBR.
+
+The root `.gitignore` lists reproducible bootstrap files and ordinary
+configure/build state. It does not change the existing release/build model; it
+only allows a cleaner Git repository.
+
+## 10. GNU-compatible features
+
+`mcpu-cpp` is an independent MCPU preprocessor, but it intentionally follows
+GNU CPP behavior for a number of well-known operations. Compatibility applies
+to the documented features; it does not imply complete CLI or language
+interchangeability with GCC.
+
+GNU-compatible behavior is used for, in particular:
+
+* object-like and function-like macros, macro rescan, `#`, and `##`;
+* variadic macros `...` / `__VA_ARGS__` and standard `__VA_OPT__`;
+* `#if`, `#ifdef`, `#ifndef`, `#elif`, `#else`, `#endif`, and `defined`;
+* `#include`, `#include_next`, `#pragma once`, `#line`, and GNU linemarkers;
+* compact output mapping: up to seven invisible lines are represented by
+ newlines, while a gap of eight or more uses a corrective linemarker;
+* forced files `-include` / `-imacros` and dependency options `-M`, `-MM`,
+ `-MD`, `-MMD`, `-MF`, `-MT`, `-MQ`, and `-MG`;
+* warning controls `-w`, `-Wall`, `-Werror`, and the supported `-Wcomment` forms.
+
+MCPU-specific facilities, including `#lang` / `#endlang`, the `zNNN` numeric
+suffix, and ABI predefined macros, remain native `mcpu-cpp` extensions.
diff --git a/doc/mcpu-cpp-ru.md b/doc/mcpu-cpp-ru.md
new file mode 100644
index 0000000..f8f762a
--- /dev/null
+++ b/doc/mcpu-cpp-ru.md
@@ -0,0 +1,2167 @@
+# mcpu-cpp
+
+`mcpu-cpp` — препроцессор языков программирования MCPU. Он является
+самостоятельным компонентом экосистемы LibMPU/LibMPUIO/LibMCPU и не привязан
+к названию одного конкретного языка: активный язык выбирается директивой
+`#lang`.
+
+Этот документ задаёт нормативное поведение `mcpu-cpp`: текстовую модель,
+директивы, macro engine, include pipeline, конфигурацию, диагностику и
+генерацию зависимостей для инструментов MCPU.
+
+## 1. Текстовая модель
+
+Внешние исходные файлы и конфигурационные файлы имеют кодировку UTF-8.
+UTF-8 должен быть корректным. Для исходных программ проверка принадлежности
+символов диапазону UCS-2 выполняется после удаления комментариев: поэтому
+корректный Unicode scalar value выше `U+FFFF` допустим внутри комментария, но
+остаётся ошибкой в программном тексте. После этой стадии исходный текст
+обрабатывается как последовательность `__mpu_char16_t`. Входной UTF-8 BOM
+допускается и удаляется. Встроенный NUL в исходном файле запрещён.
+
+Переводы строк `CRLF` и `CR` нормализуются в `LF`.
+
+## 2. Действия, выполняемые независимо от директив
+
+`mcpu-cpp` выполняет несколько преобразований до разбора
+директив.
+
+### 2.1. Backslash-newline
+
+Последовательность `\\` непосредственно перед переводом строки удаляется до
+распознавания комментариев, директив и макросов. Поэтому, например,
+
+```text
+#defi\
+ne FOO 10\
+20
+```
+
+эквивалентно логической строке
+
+```text
+#define FOO 1020
+```
+
+При этом физические номера строк продолжают учитываться при формировании
+текущей позиции; если пользователь не менял её директивой `#line`, они и будут
+видны в генерируемых line marker-ах.
+
+### 2.2. Комментарии
+
+Комментарии `/* ... */` и `// ...` удаляются до последующей обработки. Там,
+где это необходимо для разделения соседних токенов, сохраняется пробельный
+разделитель. Если комментарий завершает непустую строку, после его удаления не
+сохраняются ни синтетический разделитель, ни пробелы, предшествовавшие
+комментарию: строка заканчивается последним значащим символом. Это относится и
+к многострочному комментарию, начавшемуся после программного текста. Если после
+удаления комментария строка вообще не содержит ничего кроме пробелов, она
+становится действительно пустой строкой. При этом комментарий между двумя
+токенами по-прежнему оставляет необходимый разделитель и не склеивает их.
+Переводы строк сохраняются, чтобы не разрушать координаты исходного текста.
+
+Комментарий не распознаётся внутри строковой или символьной константы. Для
+языка `diff` апостроф не считается началом символьной константы, поскольку
+используется в обозначениях производных.
+
+В буквальном аргументе `#include <...>` последовательности `/*` и `//`
+рассматриваются как часть имени файла.
+
+## 3. Директивы и выходной поток
+
+Директива начинается символом `#`, если до него в логической строке находятся
+только пробельные символы или комментарии. Между `#` и именем директивы
+допускаются пробелы.
+
+Служебная информация о позиции в выходном потоке представлена в форме GNU
+**line marker**:
+
+```text
+# номер "имя-файла" [флаги]
+```
+
+Это не входная директива `#line`. При входе во включаемый файл к line marker-у
+добавляется флаг `1`, а при возврате в файл, содержащий `#include`, — флаг `2`.
+Эти значения имеют тот же смысл, что и в GNU CPP: `1` означает вход в новый
+файл, `2` — возврат в предыдущий файл. Флаг `2` не является числом или уровнем
+вложенности.
+
+Например:
+
+```text
+# 1 "main.c"
+# 1 "defs.h" 1
+...
+# 2 "main.c" 2
+```
+
+Входная директива
+
+```text
+#line 62 "main.y"
+```
+
+сама в выходной поток не копируется. Она изменяет логические значения
+`__LINE__` и `__FILE__` для последующего текста, а в выходе представляется
+line marker-ом:
+
+```text
+# 62 "main.y"
+```
+
+Аргументы `#line` предварительно подвергаются macro expansion, как в
+принятой модели line control. Если после такого `#line` происходит `#include`, то после
+возврата marker получает флаг `2`, например `# 65 "main.y" 2`. Имя, заданное
+через `#line`, становится логическим именем для `__FILE__` и line marker-ов; оно
+не меняет каталог, относительно которого ищется quoted `#include`.
+
+Директивы препроцессора имеют только канонические английские имена. Unicode остаётся полностью допустимым в идентификаторах, строках, комментариях и другом пользовательском тексте.
+
+## 4. Заголовочные файлы
+
+Поддерживаются:
+
+```text
+#include "file"
+#include <file>
+#include_next "file"
+#include_next <file>
+#pragma once
+```
+
+Для обычного `#include "file"` первым всегда проверяется каталог **физического**
+текущего исходного файла. Логическое имя, установленное через `#line`, на этот
+шаг не влияет. Для `#include <file>` каталог текущего файла не проверяется.
+
+### 4.1. Перемещаемый корень MCPU как общий принцип экосистемы
+
+Начиная с выпуска 0.0.37 каталог установки MCPU **не содержит версию
+конкретного инструмента** и не является абсолютной runtime-константой,
+зашитой в бинарный файл. Версия относится к самому `mcpu-cpp`, `mcpu-as`,
+`mcpu-ld`, `mcpu-run` или библиотеке, но не определяет корень единой среды
+MCPU.
+
+При типичной конфигурации:
+
+```text
+./configure --prefix=/usr --libdir=/usr/lib64
+```
+
+`make install` создаёт дерево:
+
+```text
+/usr/lib64/mcpu/
+├── bin/
+│ └── mcpu-cpp
+├── etc/
+│ └── mcpu-cpp.conf
+├── include/
+│ ├── diff/
+│ ├── dift/
+│ ├── alg/
+│ ├── as/
+│ ├── avm/
+│ └── acs/
+└── lib/ # общий каталог будущих библиотек MCPU
+```
+
+Публичное имя программы находится в `$bindir`:
+
+```text
+/usr/bin/mcpu-cpp -> ../lib64/mcpu/bin/mcpu-cpp
+```
+
+Абсолютный `/usr/lib64/mcpu` при этом **не является частью runtime ABI
+MCPU-CPP**. Он используется только `make install` как выбранное configure-time
+место размещения файлов.
+
+При каждом обычном запуске MCPU-CPP определяет фактический путь собственного
+исполняемого файла через Linux `/proc/self/exe`. Символическая ссылка публичной
+команды не мешает этому: `/proc/self/exe` указывает на реально выполняемый
+бинарный файл. Если `/proc/self/exe` недоступен, используется резервное
+разрешение `argv[0]` через `PATH` и `realpath(3)`; возврата к зашитому
+configure-time installation root нет.
+
+Для бинарного файла:
+
+```text
+<root>/bin/mcpu-cpp
+```
+
+runtime-корень определяется как:
+
+```text
+executable = <root>/bin/mcpu-cpp
+executable dir = <root>/bin
+MCPU runtime root = <root>
+```
+
+Из него автоматически выводятся:
+
+```text
+<root>/etc/mcpu-cpp.conf
+<root>/include
+```
+
+Следовательно всё дерево можно физически перенести, например из:
+
+```text
+/usr/lib64/mcpu/
+```
+
+в:
+
+```text
+/opt/mcpu-test/
+```
+
+или:
+
+```text
+$HOME/devel/mcpu-next/
+```
+
+и `<new-root>/bin/mcpu-cpp` без переконфигурирования начнёт использовать
+`<new-root>/etc/mcpu-cpp.conf` и `<new-root>/include`. Старый абсолютный путь
+не сохраняется ни в runtime default, ни в штатном `mcpu-cpp.conf`.
+
+Это не частная особенность препроцессора, а **общий принцип экосистемы MCPU**.
+Будущие `mcpu-as`, `mcpu-ld`, `mcpu-run`, библиотеки, CRT и другие компоненты
+должны разделять один перемещаемый корень:
+
+```text
+<root>/bin
+<root>/etc
+<root>/include
+<root>/lib
+```
+
+Их собственные версии могут отличаться, но согласованность конкретной среды
+MCPU определяется тем, что все компоненты находятся в одном runtime tree, а
+не совпадением version suffix в именах каталогов.
+
+### 4.2. Runtime defaults, уровни конфигурации и системный include root
+
+До чтения любого конфигурационного файла MCPU-CPP создаёт runtime-derived
+значение:
+
+```text
+MCPU_CPP_SYSTEM_INCLUDE_PATH = <runtime-root>/include
+```
+
+После этого конфигурационные слои применяются в порядке возрастающего
+приоритета:
+
+```text
+runtime-derived defaults
+ ↓
+<runtime-root>/etc/mcpu-cpp.conf
+ ↓
+/etc/mcpu/mcpu-cpp.conf
+ ↓
+$HOME/.mcpu/mcpu-cpp.conf
+```
+
+`<runtime-root>/etc/mcpu-cpp.conf` устанавливается вместе с MCPU-CPP, но сам
+файл намеренно не содержит абсолютного штатного `MCPU_CPP_SYSTEM_INCLUDE_PATH`:
+иначе перенос всего дерева восстановил бы старый путь. `/etc/mcpu/mcpu-cpp.conf`
+является необязательным machine-wide override: `make install` каталог
+`/etc/mcpu` не создаёт. Домашний `$HOME/.mcpu/mcpu-cpp.conf` также необязателен,
+не версионируется и имеет максимальный config-приоритет.
+
+Если одна переменная определена несколько раз, побеждает последнее
+определение, включая пустое. Поэтому `MCPU_CPP_SYSTEM_INCLUDE_PATH` остаётся
+полностью заменяемым system root. Например:
+
+```text
+MCPU_CPP_SYSTEM_INCLUDE_PATH = $HOME/mcpu-next/include;
+```
+
+полностью заменяет runtime-derived `<runtime-root>/include`. Для активного
+`#lang "as"` тогда проверяются:
+
+```text
+$HOME/mcpu-next/include/as
+$HOME/mcpu-next/include
+```
+
+Штатные language-подкаталоги всегда выводятся самим препроцессором из одного
+root; переменных вида `MCPU_CPP_SYSTEM_<LANG>_INCLUDE_PATH` нет.
+
+Пустое effective значение:
+
+```text
+MCPU_CPP_SYSTEM_INCLUDE_PATH = ;
+```
+
+удаляет configured system stage полностью. Более приоритетный config может
+после этого снова включить его непустым значением.
+
+`--config-file FILE` применяет явно выбранный файл поверх runtime-derived
+default. `--no-config` отключает **только чтение файлов конфигурации**:
+`<runtime-root>/etc/mcpu-cpp.conf`, `/etc/mcpu/mcpu-cpp.conf` и
+`$HOME/.mcpu/mcpu-cpp.conf` не читаются, но `<runtime-root>/include` остаётся
+штатным system root. Только `-nostdinc` удаляет effective standard-system tree
+из include search для конкретного запуска; явно переданный `-isystem` при этом
+остаётся command-line каталогом.
+
+### 4.3. Нормативный порядок поиска include-файлов
+
+Порядок поиска является частью контракта MCPU-CPP. Явно заданные параметры
+командной строки имеют приоритет над persistent configuration. После
+необязательного каталога текущего физического файла эффективная цепочка имеет
+строго следующий вид:
+
+```text
+explicit -I
+ ↓
+explicit -isystem
+ ↓
+MCPU_CPP_<LANG>_INCLUDE_PATH
+ ↓
+MCPU_CPP_INCLUDE_PATH
+ ↓
+MCPU_CPP_SYSTEM_INCLUDE_PATH/<lang>
+ ↓
+MCPU_CPP_SYSTEM_INCLUDE_PATH
+ ↓
+explicit -idirafter
+ ↓
+MCPU_CPP_AFTER_INCLUDE_PATH
+```
+
+Элементы, которых нет или которые не содержат требуемого файла, пропускаются.
+
+`MCPU_CPP_<LANG>_INCLUDE_PATH` — свободно настраиваемые пользователем
+language-specific path-list'ы:
+
+```text
+MCPU_CPP_DIFF_INCLUDE_PATH
+MCPU_CPP_DIFT_INCLUDE_PATH
+MCPU_CPP_ALG_INCLUDE_PATH
+MCPU_CPP_AS_INCLUDE_PATH
+MCPU_CPP_AVM_INCLUDE_PATH
+MCPU_CPP_ACS_INCLUDE_PATH
+```
+
+Пользователь полностью распоряжается именами и расположением этих каталогов.
+`MCPU_CPP_INCLUDE_PATH` — общий пользовательский path-list, видимый во всех
+языковых состояниях.
+
+`-idirafter` и `MCPU_CPP_AFTER_INCLUDE_PATH` являются общим fallback-карманом.
+MCPU-CPP не строит для них автоматических `<lang>`-подкаталогов. Пользователь
+сам организует их внутреннюю структуру и при необходимости пишет, например:
+
+```text
+#include <vendor/device.h>
+```
+
+Именно semantic class, а не порядок появления разных классов в argv/config,
+определяет приоритет. Внутри одного класса сохраняется порядок добавления.
+
+### 4.4. `#include_next` и wrapper headers
+
+`#include_next` предназначен прежде всего для заголовков-обёрток (wrapper
+headers). Он позволяет поставить локальный header раньше системного, изменить
+локальную политику и затем продолжить поиск одноимённого header по нормативной
+цепочке без копирования системного файла и без абсолютного имени.
+
+Например:
+
+```text
+mcpu-cpp -isystem $HOME/mcpu-wrapper ...
+```
+
+и `$HOME/mcpu-wrapper/math.h`:
+
+```text
+#ifndef SOME_SYSTEM_MACRO
+#define SOME_SYSTEM_MACRO temporary_value
+#define REMOVE_SOME_SYSTEM_MACRO 1
+#endif
+
+#include_next <math.h>
+
+#ifdef REMOVE_SOME_SYSTEM_MACRO
+#undef SOME_SYSTEM_MACRO
+#undef REMOVE_SOME_SYSTEM_MACRO
+#endif
+```
+
+Если домашний config одновременно задаёт:
+
+```text
+MCPU_CPP_SYSTEM_INCLUDE_PATH = $HOME/mcpu-next/include;
+```
+
+wrapper найденный через `-isystem` продолжит `#include_next` уже через
+configured user paths, затем через
+`$HOME/mcpu-next/include/<lang>` и `$HOME/mcpu-next/include`; старое system tree исходного места установки при этом не участвует. Именно такой сценарий позволяет
+системному разработчику или тестеру жить в собственной sandbox.
+
+MCPU-CPP хранит конкретный **физический элемент effective search chain**, из
+которого найден текущий header. `#include_next` начинает со следующего элемента.
+Формы `"file"` и `<file>` для `#include_next` эквивалентны; каталог текущего
+файла повторно не проверяется. Если текущий файл найден обычным quoted-поиском
+относительно содержащего файла и не имеет search-chain provenance,
+`#include_next` начинает с первого элемента configured chain.
+
+Операнд может быть получен macro expansion. Логическое имя после `#line` не
+влияет на физический provenance. Если после текущего entry подходящего файла
+нет, preprocessing завершается ошибкой.
+
+### 4.5. `#pragma once`
+
+Активная директива
+
+```text
+#pragma once
+```
+
+помечает **физический файл** как уже обработанный в текущем запуске
+MCPU-CPP. При последующей попытке включить тот же физический файл его
+содержимое повторно не обрабатывается. Сама директива потребляется
+препроцессором и в выходной поток не копируется, в том числе при `-dD`.
+
+Идентичность определяется по паре `st_dev`/`st_ino`, полученной файловой
+системой, а не по строковому имени пути. Поэтому один и тот же файл не может
+обойти `#pragma once`, если он достигнут как `./file.h`, через символическую
+ссылку или через другое жёсткое имя (hard link). Логическое имя после `#line`
+также не влияет на эту физическую идентичность.
+
+Пометка действует сразу в момент обработки активной директивы. Поэтому
+заголовок может после `#pragma once` включить самого себя: повторное включение
+будет пропущено и рекурсия не возникнет. Директива внутри неактивной ветви
+условной компиляции никакого действия не имеет.
+
+MCPU-CPP распознаёт только точную форму `#pragma once` с необязательными
+пробелами. Остальные `#pragma` не интерпретируются препроцессором и сохраняются
+для последующих стадий компиляции; например, `#pragma pack(...)` продолжает
+передаваться в выходной поток.
+
+`#pragma once` дополняет, но не изменяет нормативную search-chain
+`#include`/`#include_next`: сначала обычный механизм поиска находит физический
+файл, затем registry `once` решает, надо ли обрабатывать его содержимое.
+
+### 4.6. Принудительные файлы: `-imacros FILE` и `-include FILE`
+
+Опции командной строки
+
+```text
+-imacros FILE
+-include FILE
+```
+
+обрабатывают файл до главного input. Они используют обычный preprocessing
+engine, а не отдельный облегчённый parser.
+
+Нормативный порядок начала translation unit:
+
+```text
+predefined macros
+ -> -D/-U в порядке командной строки
+ -> все -imacros в порядке командной строки
+ -> все -include в порядке командной строки
+ -> главный input
+```
+
+Таким образом, взаимное расположение `-imacros` и `-include` в `argv` не
+перемешивает эти две группы: **все** `-imacros` всегда выполняются раньше
+**всех** `-include`.
+
+`-imacros FILE` полностью обрабатывает файл: его `#define`/`#undef`,
+условные директивы, `#lang`/`#endlang`, `#include`, `#include_next`,
+`#pragma once` и диагностика имеют обычную семантику. Однако весь normal
+preprocessing output этого forced-файла, включая line markers и текст
+вложенных headers, отбрасывается. Полученное состояние macro table и других
+preprocessing-механизмов сохраняется для последующих forced-файлов и главного
+input.
+
+`-include FILE` использует тот же механизм, но normal output сохраняется, как
+если бы найденный header был включён непосредственно перед главным source.
+Forced include является настоящей include-границей: внутри него
+`__INCLUDE_LEVEL__ == 1`, внутри включённого им header уровень равен `2`, а
+главный input остаётся на уровне `0`. `__BASE_FILE__` внутри forced-файлов
+остаётся именем главного input.
+
+Абсолютный operand forced-файла используется непосредственно. Относительный
+operand ищется сначала в **current working directory**, затем по обычной
+include-chain:
+
+```text
+explicit -I
+explicit -isystem
+MCPU_CPP_<LANG>_INCLUDE_PATH
+MCPU_CPP_INCLUDE_PATH
+MCPU_CPP_SYSTEM_INCLUDE_PATH/<lang>
+MCPU_CPP_SYSTEM_INCLUDE_PATH
+explicit -idirafter
+MCPU_CPP_AFTER_INCLUDE_PATH
+```
+
+Каталог главного input не получает специального приоритета при поиске operand
+`-imacros`/`-include`. После нахождения forced-файла обычный quoted
+`#include "file"` внутри него снова разрешается относительно физического
+каталога этого файла. Если forced-файл найден через элемент include-chain,
+его provenance сохраняется и `#include_next` продолжает поиск со следующего
+элемента цепочки.
+
+Forced-файлы и реально достигнутые из них headers входят в обычный physical
+dependency registry. Их user/system classification определяется тем же
+search provenance, поэтому `-MM`/`-MMD` фильтруют system forced headers так же,
+как обычные system headers. Отсутствующий forced-файл является ошибкой.
+
+
+### 4.7. Генерация зависимостей: `-M`, `-MM`, `-MG`, `-MD`, `-MMD`, `-MF`, `-MT`, `-MQ`
+
+Опции `-M` и `-MM` используют **тот же самый проход include pipeline**, что и
+обычная preprocessing. Отдельного повторного поиска заголовков не выполняется.
+Поэтому dependency graph автоматически наследует нормативный порядок путей,
+`#include_next`, macro-expanded include operands, conditional compilation и
+`#pragma once`.
+
+`-M` подавляет обычный preprocessing output и выводит одно правило Make:
+
+```make
+file.o: file.c header1.h header2.h
+```
+
+В список входят главный source-файл и все реально достигнутые физические
+headers, включая system headers. Один физический файл записывается один раз;
+идентичность определяется как `st_dev + st_ino`, поэтому другое относительное
+имя, symbolic link или hard link не создают дополнительную dependency. Имя,
+назначенное директивой `#line`, является только logical source name и в
+dependency list не попадает.
+
+`-MM` строит тот же граф, но исключает system dependencies. System-контекстом
+считаются headers, найденные через explicit `-isystem`, configured system tree
+`MCPU_CPP_SYSTEM_INCLUDE_PATH/<lang>` / `MCPU_CPP_SYSTEM_INCLUDE_PATH`, explicit
+`-idirafter` и `MCPU_CPP_AFTER_INCLUDE_PATH`, а также вся ветвь headers,
+включённая непосредственно или косвенно из такого system header. Форма
+`#include "file"` или `#include <file>` сама по себе не определяет system-ness.
+Если один и тот же физический файл был достигнут из system-ветви, но затем
+также включён непосредственно из user-контекста, он остаётся пользовательской
+dependency и присутствует в `-MM`.
+
+Default target образуется из basename главного source-файла: его suffix
+заменяется object suffix (`.o` по умолчанию). Пути и target экранируются для
+Make. Для stdin используется GNU-подобная форма `-: -`.
+
+`-MD` и `-MMD` используют тот же dependency graph, но, в отличие от `-M` и
+`-MM`, **не подавляют обычный preprocessing output**. `-MD` включает system
+headers, как `-M`; `-MMD` применяет user-only фильтр `-MM`. Это позволяет одним
+проходом получить и препроцессированный текст, и side-effect dependency file.
+
+Если `-MF` не задан, side-effect режим выбирает имя `.d` автоматически:
+
+* без `-o` из basename входного файла удаляется suffix и добавляется `.d`;
+ каталоги входного pathname в имя dependency-файла не переносятся;
+* при обычном `-o FILE` suffix output-файла заменяется на `.d`;
+* для stdin используется имя `-.d`.
+
+`-MF FILE` переопределяет автоматическое имя dependency-файла. Значение
+`-MF -` означает stdout. `-MF` работает также с dependency-only `-M`/`-MM`;
+в этом случае оно имеет приоритет над обычным destination для make-rule. Само
+по себе `-MF` без одного из `-M`, `-MM`, `-MD`, `-MMD` является ошибкой.
+
+Семантика намеренно следует GNU CPP: `-MD`/`-MMD` не принимают собственный
+аргумент, а `-MF` является отдельной опцией назначения dependency output.
+
+`-MT TARGET` заменяет автоматический target правила строкой `TARGET` **точно
+как она передана**. Make quoting при этом не выполняется. Поэтому один argument
+`-MT` может сам содержать несколько targets, разделённых пробелами:
+
+```text
+-MT 'obj/a.o obj/a.pic.o'
+```
+
+и повторные `-MT` также добавляют targets одного и того же правила:
+
+```text
+-MT obj/a.o -MT obj/a.pic.o
+```
+
+`-MQ TARGET` имеет ту же семантику выбора target, но экранирует специальные
+для Make символы. Например,
+
+```text
+-MQ '$(OBJDIR)/foo.o'
+```
+
+даёт левую часть правила:
+
+```make
+$$(OBJDIR)/foo.o:
+```
+
+Поддерживаются как отдельные arguments (`-MT TARGET`, `-MQ TARGET`), так и
+attached forms (`-MTTARGET`, `-MQTARGET`).
+
+Если задан хотя бы один `-MT` или `-MQ`, автоматический default target не
+выводится. В частности, `--object-suffix` влияет только на автоматический
+target и не переписывает явно заданные targets. Если явных targets нет,
+default target экранируется для Make так же, как при `-MQ`.
+
+Разрешены повторные и смешанные `-MT`/`-MQ`. В соответствии с GNU CPP сначала
+выводятся все `-MT` targets в их command-line order, затем все `-MQ` targets в
+их command-line order. Все они образуют левую часть **одного** dependency
+rule.
+
+`-MT` и `-MQ` имеют смысл только вместе с одним из dependency-generation
+режимов `-M`, `-MM`, `-MD` или `-MMD`. Без такого режима это ошибка командной
+строки.
+
+`-MG` изменяет только обработку **отсутствующих** include-файлов при
+dependency-only режимах `-M` и `-MM`. Без `-MG` неразрешённый `#include` остаётся
+ошибкой. С `-M -MG` или `-MM -MG` отсутствующий header считается будущим
+generated file: preprocessing не завершается ошибкой, а operand директивы
+добавляется в dependency rule **ровно в том виде, который получен после macro
+expansion**, без приписывания предполагаемого include-directory. Например:
+
+```text
+#include "generated.h"
+```
+
+при `-M -MG` добавляет dependency `generated.h`, даже если такого файла ещё нет.
+Macro-expanded include ведёт себя аналогично: dependency получает уже
+развёрнутое имя. `-MG` разрешён только вместе с `-M` или `-MM`; комбинации с
+`-MD`/`-MMD` и использование без dependency-only режима являются ошибкой
+командной строки.
+
+Unresolved dependencies интегрированы в **тот же упорядоченный dependency
+registry**, что и физически найденные файлы, но образуют отдельный identity
+domain. Для найденного файла registry по-прежнему использует `st_dev/st_ino` и
+physical provenance. Для отсутствующего файла этих данных нет, поэтому `-MG`
+entry не выполняет `stat()` и дедуплицируется по точному тексту include operand.
+Это принципиально: наличие одноимённого файла в CWD не должно превращать
+неразрешённый `<name>` в ложное physical совпадение, если angle-search этот файл
+не находил. Разные unresolved spellings (`generated.h` и `./generated.h`)
+считаются разными dependencies.
+
+Для `-MM` unresolved dependency получает user/system class из контекста поиска:
+отсутствующий `<file>` является system-class, отсутствующий `"file"` — user-class,
+если сама включающая единица не является system header; любой missing include,
+достигнутый из system header, остаётся system-class. При повторении одного и того
+же unresolved operand сохраняется классификация его первого появления, что
+соответствует GNU CPP. Физически найденные зависимости сохраняют прежнее правило:
+если один и тот же inode позднее достигается из user-контекста, он перестаёт быть
+system-only.
+
+`-MG` распространяется также на отсутствующие command-line forced files
+`-include FILE` и `-imacros FILE`: их operand заносится в unresolved registry как
+user dependency без синтетического search prefix. При наличии реального файла
+`-include`/`-imacros` продолжают использовать обычный physical dependency
+registry и search provenance.
+
+
+## 5. Переключение языков
+
+Препроцессор запускается в состоянии `0`. Это безымянный основной C-подобный
+язык и он не является допустимым аргументом `#lang`.
+
+Допустимые языки:
+
+| Имя | Назначение |
+|---|---|
+| `diff` | дифференциальные уравнения |
+| `dift` | разностные уравнения |
+| `alg` | алгебраические уравнения |
+| `as` | MCPU assembler (`mcpu-as`) |
+| `avm` | схемы аналоговых вычислительных машин |
+| `ACS` | структурные схемы систем автоматического управления |
+
+После `#lang` обязательна строковая константа с одним непустым словом:
+
+```text
+#lang "diff"
+```
+
+Имя проверяется только по внутреннему списку языков выше и сравнивается без
+учёта ASCII-регистра. Поэтому `"diff"`, `"Diff"`, `"DIFF"` и `"dIfF"`
+эквивалентны при выборе языка. Исходное написание внутри кавычек при этом
+сохраняется в выходном потоке.
+
+Пробелы внутри строковой константы запрещены: `" diff"`, `"diff "` и
+`"di ff"` являются ошибками. Escape-последовательности внутри неё не
+разбираются. Закрывающая кавычка обязана находиться на той же физической строке
+исходного файла. После неё до конца строки допустимы только пробельные символы.
+
+Внешние пробелы директивы нормализуются. Например:
+
+```text
+ # lang "DiFf"
+```
+
+превращается в:
+
+```text
+#lang "DiFf"
+```
+
+`#lang` помещает новый язык в стек, `#endlang` восстанавливает предыдущий.
+Стек не сбрасывается при `#include`, поэтому начало и конец языкового блока
+могут находиться в разных файлах. Директивы `#lang` и `#endlang` сохраняются в
+выходном потоке для последующего frontend dispatcher; `#lang` сохраняется в
+нормализованной форме.
+
+## 6. Простые макроопределения
+
+Начиная с 0.0.4 поддерживаются object-like macros:
+
+```text
+#define BUFFER_SIZE 1024
+#define NAME value
+#define EMPTY
+```
+
+Директива `#define` сама в выходной поток не попадает. В обычном тексте
+идентификатор-макро заменяется его replacement list. Replacement затем снова
+просматривается на макроимена, поэтому допускается каскадное раскрытие:
+
+```text
+#define A B
+#define B 10
+A
+```
+
+даёт `10`.
+
+Во время раскрытия конкретное макро временно блокируется. Поэтому
+самоссылочные и взаимно-рекурсивные определения не вызывают бесконечной
+рекурсии.
+
+Макроимена не раскрываются внутри строковых и символьных констант. Для `diff`
+апостроф сохраняет специальную языковую семантику и не защищает последующий
+текст как C character constant.
+
+Многострочное определение через backslash-newline поддерживается, поскольку
+splice выполняется раньше `#define`.
+
+### 6.1. `#undef`
+
+```text
+#undef NAME
+```
+
+удаляет object-like macro. Отмена несуществующего определения не является
+ошибкой.
+
+### 6.2. Вычисляемый `#include`
+
+Аргумент `#include`, который не начинается непосредственно с `"` или `<`,
+сначала проходит macro expansion. Поэтому допустимо:
+
+```text
+#define HEADER <diff/model.h>
+#include HEADER
+```
+
+или
+
+```text
+#define HEADER "local.h"
+#include HEADER
+```
+
+Результат раскрытия обязан иметь форму `"file"` или `<file>`.
+
+## 7. Макро с аргументами
+
+Macro engine поддерживает
+function-like macros:
+
+```text
+#define идентификатор( список аргументов ) текст
+```
+
+Открывающая скобка в определении должна идти **непосредственно**
+после имени макро. Поэтому
+
+```text
+#define F(X) X
+```
+
+задаёт макро с аргументом, а
+
+```text
+#define F (X)
+```
+
+задаёт простое object-like macro со строкой замены `(X)`.
+
+В месте использования между именем function-like macro и открывающей скобкой
+пробельные символы допустимы. Если `(` не следует, идентификатор не считается
+вызовом данного макро и остаётся в выходном тексте.
+
+Для обычного function-like macro число фактических аргументов должно совпадать
+с числом формальных. Для variadic macro должны присутствовать все фиксированные
+аргументы, а variadic tail может содержать произвольное число аргументов, включая
+пустой tail. При разборе списка фактических аргументов вложенные круглые скобки
+учитываются; запятая внутри них не разделяет аргументы. Квадратные скобки такого
+свойства не имеют — это является частью принятой семантики macro expansion.
+
+Например,
+
+```text
+#define min(X, Y) ((X) < (Y) ? (X) : (Y))
+min(1, 2)
+```
+
+даёт
+
+```text
+((1) < (2) ? (1) : (2))
+```
+
+Перед подстановкой обычный фактический аргумент сам проходит macro expansion.
+Поэтому каскадные и вложенные вызовы работают естественно:
+
+```text
+#define A 7
+#define min(X, Y) ((X) < (Y) ? (X) : (Y))
+min(min(A, 3), 10)
+```
+
+Формальный параметр может встречаться в replacement list произвольное число
+раз. Это означает, что выражение с побочным эффектом в
+фактическом аргументе также может быть вычислено несколько раз уже последующим
+компилятором; препроцессор не пытается исправлять такую программу.
+
+Поддерживаются макро без формальных параметров:
+
+```text
+#define READY() 1
+```
+
+Они раскрываются только как вызов `READY()` (пробел между именем и `(` при
+использовании допустим), но самостоятельный идентификатор `READY` не
+раскрывается.
+
+Имена формальных параметров должны быть различны. Незавершённый список,
+неверная пунктуация, недостаточное или избыточное число фактических аргументов
+диагностируются как ошибки.
+
+### 7.1. Stringification `#`
+
+Поддерживается оператор
+stringification (`#`) для параметров function-like macro:
+
+```text
+#define STR(X) #X
+STR(alpha + beta)
+```
+
+даёт
+
+```text
+"alpha + beta"
+```
+
+Stringification использует **сырой фактический аргумент до macro expansion**.
+Поэтому:
+
+```text
+#define A 7
+#define STR(X) #X
+#define XSTR(X) STR(X)
+
+STR(A) -> "A"
+XSTR(A) -> "7"
+```
+
+Ведущие и завершающие пробелы аргумента удаляются. Последовательности
+пробельных символов внутри аргумента сворачиваются в один пробел, кроме
+пробелов внутри строковых/символьных токенов соответствующего активного
+языка. Двойные кавычки и обратные косые черты внутри quoted tokens экранируются
+так, чтобы результат оставался одной корректной строковой константой.
+
+Оператор `#` в replacement list function-like macro обязан непосредственно или
+через пробельные символы ссылаться на имя формального параметра. Внутри quoted
+token символ `#` оператором не является. Пустой фактический аргумент допустим и
+stringify-ится как `""`.
+
+### 7.2. Token concatenation `##`
+
+Начиная с 0.0.21 поддерживается оператор token concatenation (`##`) в модели,
+согласованной с GNU CPP и механизмом `collect_expansion()` / `macroexpand()`
+macro engine. Оператор объединяет два соседних preprocessing token в
+один token, после чего получившийся replacement list снова проходит macro
+expansion.
+
+Например:
+
+```text
+#define CAT(A, B) A ## B
+CAT(foo, bar)
+```
+
+даёт `foobar`. Склеивание может образовывать identifier, preprocessing number
+или многосимвольный punctuator. Поэтому, например, допустимы:
+
+```text
+CAT(1.5, e3) -> 1.5e3
+CAT(+, =) -> +=
+```
+
+Если формальный параметр непосредственно примыкает к `##`, его фактический
+аргумент подставляется **без предварительного macro expansion**. Это тот же
+raw-argument принцип, который используется для stringification. Для получения
+сначала expansion, а затем concatenation применяется обычный двухуровневый
+приём GNU CPP:
+
+```text
+#define AFTERX(X) X_ ## X
+#define XAFTERX(X) AFTERX(X)
+#define TABLESIZE 1024
+#define BUFSIZE TABLESIZE
+
+AFTERX(BUFSIZE) -> X_BUFSIZE
+XAFTERX(BUFSIZE) -> X_1024
+```
+
+Пустой фактический аргумент рядом с `##` ведёт себя как placemarker: сам по
+себе он не добавляет token, а `##` с такой стороны не изменяет оставшийся
+операнд. Если фактический аргумент содержит несколько preprocessing tokens,
+склеивается только крайний token, непосредственно соседний с `##`; остальные
+tokens сохраняются и затем участвуют в общем rescan.
+
+`#` и `##` могут использоваться в одном function-like macro, например:
+
+```text
+#define COMMAND(NAME) #NAME | NAME ## _command
+```
+
+При этом `#NAME` использует raw spelling аргумента для stringification, а
+`NAME ## _command` — тот же raw argument для concatenation.
+
+`##` внутри quoted token оператором не является. Комментарии к моменту macro
+expansion уже заменены whitespace, поэтому они не могут быть созданы
+склеиванием `/` и `*`. Между `##` и его операндами исходно может находиться
+whitespace; при склеивании он не участвует.
+
+Если два операнда не образуют один допустимый preprocessing token, выдаётся
+диагностика, а сами исходные tokens сохраняются; наличие whitespace между ними
+после такой диагностики не является частью контракта. `##` в начале или в
+конце replacement list является ошибкой определения macro.
+
+### 7.3. Variadic macros: `...` и `__VA_ARGS__`
+
+Начиная с 0.0.46 поддерживаются variadic function-like macros в современной
+C99-совместимой форме:
+
+```text
+#define LOG(...) output(__VA_ARGS__)
+#define LOGF(format, ...) output(format, __VA_ARGS__)
+```
+
+Маркер `...` может быть единственным параметром либо последним элементом после
+одного или нескольких фиксированных параметров. Старое GNU-расширение с
+именованным variadic parameter
+
+```text
+#define LOG(args...) ...
+```
+
+в 0.0.46 намеренно не поддерживается. `__VA_OPT__` также не является частью
+этого релиза.
+
+При вызове все tokens после последнего фиксированного параметра, включая
+разделяющие их запятые, образуют один logical variable argument и подставляются
+вместо `__VA_ARGS__`. В обычной позиции этот variable argument предварительно
+проходит macro expansion так же, как обычный фактический аргумент:
+
+```text
+#define A 7
+#define V(...) <__VA_ARGS__>
+#define F(first, ...) first | __VA_ARGS__
+
+V(A, 2, 3) -> <7, 2, 3>
+F(1, A, 3) -> 1 | 7, 3
+```
+
+Variadic tail может быть пустым. Поэтому оба вызова
+
+```text
+F(1)
+F(1,)
+```
+
+допустимы и подставляют пустой `__VA_ARGS__`. Это **не** означает автоматическое
+удаление запятой, явно записанной в replacement list. Например для
+
+```text
+#define E(format, ...) output(format, __VA_ARGS__)
+```
+
+вызов `E("ok")` оставляет запятую перед пустым tail. Специальная историческая
+GNU-семантика `, ## __VA_ARGS__`, удаляющая такую запятую, в контракт 0.0.46 не
+входит; если она понадобится, её следует вводить отдельным явно документированным
+расширением.
+
+`__VA_ARGS__` участвует в уже существующей семантике `#` и `##` как настоящий
+macro parameter. Stringification использует raw spelling всего variadic tail:
+
+```text
+#define STRV(...) #__VA_ARGS__
+STRV(A, b + c) -> "A, b + c"
+```
+
+При соседстве с `##` variadic argument также подставляется без prescan; затем
+работают обычные правила placemarker, token concatenation и общего rescan.
+Например:
+
+```text
+#define L(...) pre ## __VA_ARGS__
+#define R(...) __VA_ARGS__ ## post
+
+L(fix) -> prefix
+R(fix) -> fixpost
+```
+
+Если variadic argument содержит несколько preprocessing tokens, склеивается
+только крайний token, непосредственно соседний с `##`, а остальные tokens
+сохраняются, как и для обычного параметра. Пустой tail рядом с `##` ведёт себя
+как placemarker.
+
+Имя `__VA_ARGS__` зарезервировано для variable argument и не принимается как
+обычное имя формального параметра. Оператор `#__VA_ARGS__` допустим только в
+variadic macro. Dump-режимы сохраняют variadic форму определения, например:
+
+```text
+#define F(first,...) first | __VA_ARGS__
+```
+
+
+### 7.4. `__VA_OPT__`
+
+Начиная с 0.0.47 variadic macros поддерживают стандартный условный fragment
+`__VA_OPT__(pp-tokens)`. Если variable argument после обычной macro substitution
+не содержит preprocessing tokens, весь `__VA_OPT__(...)` раскрывается в пустую
+последовательность. Если variable argument непуст, содержимое круглых скобок
+участвует в replacement list:
+
+```text
+#define DEBUG(format, ...) \
+ fprintf(stderr, format __VA_OPT__(,) __VA_ARGS__)
+
+DEBUG("ready") -> fprintf(stderr, "ready")
+DEBUG("x=%d", x) -> fprintf(stderr, "x=%d", x)
+```
+
+Решение о непустоте принимается **после expansion variable argument**, а не по
+его исходному spelling. Поэтому macro, который сам раскрывается в пустую
+последовательность, не активирует `__VA_OPT__`:
+
+```text
+#define EMPTY
+#define HAS(...) [__VA_OPT__(yes)]
+
+HAS() -> []
+HAS(EMPTY) -> []
+HAS(token) -> [yes]
+```
+
+Содержимое `__VA_OPT__` может включать сбалансированные вложенные круглые скобки.
+Закрывающая `)` самого `__VA_OPT__` определяется с учётом их вложенности.
+Вложенный `__VA_OPT__` внутри другого `__VA_OPT__` намеренно запрещён.
+
+`__VA_OPT__` интегрирован с существующими правилами parameter substitution,
+stringification, token concatenation, placemarker и rescan. Например:
+
+```text
+#define X 123
+#define S(...) #__VA_OPT__(__VA_ARGS__)
+#define L(...) pre ## __VA_OPT__(__VA_ARGS__)
+
+S() -> ""
+S(X) -> "123"
+L() -> pre
+L(X) -> pre123
+```
+
+При `#__VA_OPT__(...)` сначала выполняется parameter substitution внутри
+fragment, включая prescan обычных параметров, но произвольные macro names самого
+fragment до stringification дополнительно не rescanning-ятся. Поэтому:
+
+```text
+#define X 123
+#define S(a, ...) #__VA_OPT__(a X)
+
+S(X, y) -> "123 X"
+```
+
+Если parameter внутри `__VA_OPT__` непосредственно участвует во внутреннем
+`##`, для него, как обычно, prescan подавляется; paste выполняется до дальнейшего
+rescan. Внешний `##`, соседний с `__VA_OPT__`, получает крайний token уже
+подготовленного fragment. Пустой результат `__VA_OPT__` рядом с `##` ведёт себя
+как placemarker.
+
+`__VA_OPT__` допустим только в replacement list variadic function-like macro и
+должен непосредственно задавать parenthesized fragment. `##` не может быть
+первым или последним preprocessing token внутри самого `__VA_OPT__`.
+
+Историческое GNU-расширение
+
+```text
+, ## __VA_ARGS__
+```
+
+в `mcpu-cpp` намеренно **не реализуется**. Для условной запятой следует
+использовать современную форму `__VA_OPT__(,)`. Старое GNU-расширение с
+именованным variadic parameter `args...` также остаётся неподдерживаемым.
+
+
+
+### 7.5. Нормализация пробелов в replacement list
+
+Начиная с 0.0.48 `mcpu-cpp` не переносит в результат разворачивания
+служебное выравнивание многострочного macro. После удаления `\` + newline
+последовательность пробельных символов, принадлежащая самому replacement list,
+канонизируется в один ASCII-пробел. Это особенно важно для определений, где
+обратные косые черты визуально выровнены в одну колонку:
+
+```text
+#define TRACE(x) \
+ do \
+ { \
+ output(x); \
+ done(); \
+ } \
+ while( 0 )
+```
+
+При разворачивании такое определение выдаёт компактный replacement:
+
+```text
+do { output(x); done(); } while( 0 )
+```
+
+а не сохраняет десятки пробелов перед каждой бывшей границей физической
+строки.
+
+Нормализация относится **только к whitespace самого replacement list**.
+`mcpu-cpp` не является formatter-ом исходной программы: пробелы в обычном
+тексте input сохраняются. Пробелы внутри фактического macro argument также не
+переформатируются только потому, что argument был подставлен в macro:
+
+```text
+#define ID(x) x
+
+ID(a + b) -> a + b
+```
+
+Содержимое string/character literals сохраняется буквально, поэтому:
+
+```text
+#define S "left right"
+```
+
+по-прежнему содержит пять пробелов внутри строки.
+
+Наличие whitespace между preprocessing tokens сохраняется как один пробел.
+Это не позволяет случайно изменить tokenization, например превратить `+ +` в
+`++`, `- >` в `->` или `< <` в `<<`. Операторы `#` и `##`, placemarkers,
+`__VA_ARGS__`, `__VA_OPT__` и последующий rescan продолжают использовать свои
+существующие правила; новая политика меняет только количество обычного
+replacement-list whitespace.
+
+Dump-режимы (`-dM`, `-dD`) показывают ту же каноническую форму replacement
+list, которая хранится во внутренней таблице macro.
+
+
+### 7.6. Компактификация невидимых строк и linemarkers
+
+Начиная с 0.0.49 `mcpu-cpp` использует для вертикального whitespace ту же
+модель, что GNU CPP: **удаляем, но не забываем**. Строки, которые после
+preprocessing не породили ни одного выводимого preprocessing token, не обязаны
+оставаться физическими пустыми строками в `.E`, однако их исходная позиция
+продолжает учитываться при построении linemarkers и значении `__LINE__`.
+
+Причина невидимости не имеет значения. Это могут быть удалённые directives,
+неактивные ветви `#if`, однострочные и многострочные comments, обычные пустые
+строки или их смесь. Emitter сравнивает текущую output source position с
+позицией следующей реально выдаваемой строки.
+
+Если следующая позиция находится менее чем через восемь строк, разрыв
+представляется обычными newline. Если расстояние равно восьми строкам или
+больше, вместо длинной последовательности пустых строк выдаётся корректирующий
+linemarker:
+
+```text
+# N "file"
+```
+
+и следующая содержательная строка сразу относится к source line `N`. Таким
+образом граница поведения совместима с GNU CPP: gaps 0..7 сохраняются через
+newline, gap 8 и больше заменяется linemarker.
+
+Structural markers входа и возврата из include-файла сохраняют обычный смысл:
+
+```text
+# 1 "header.h" 1
+# 4 "source.c" 2
+```
+
+Если included file не породил никакого output, `mcpu-cpp` не создаёт
+искусственный marker, сообщающий, до какой внутренней строки header дошёл
+препроцессор. После enter-marker сразу может следовать return-marker. Реальная
+позиция снова уточняется только тогда, когда требуется выдать следующий
+содержательный текст.
+
+Эта оптимизация меняет только представление output stream. Source coordinates,
+`__LINE__`, diagnostics, `#line`, include enter/return semantics и обработка
+macro остаются привязаны к исходному логическому потоку, а не к количеству
+физических строк в сжатом `.E`.
+
+
+## 8. Предопределённые макро
+
+Начиная с 0.0.6 был перенесён исторический механизм predefined macros из
+препроцессора. Этот механизм оформлен как отдельный ABI/environment layer
+будущего безымянного C-подобного языка. Эти определения не являются
+декоративными: их имена и значения должны соответствовать либо семантике GNU
+CPP, либо явно документированному MCPU/LibMPU contract.
+
+### 8.1. Динамические source macros
+
+Следующие predefined macros вычисляются в точке использования:
+
+| Макро | Раскрытие |
+|---|---|
+| `__FILE__` | строковая константа с именем текущего входного файла |
+| `__LINE__` | десятичный номер текущей строки |
+| `__BASE_FILE__` | строковая константа с именем главного входного файла translation unit |
+| `__INCLUDE_LEVEL__` | уровень вложенности `#include`; для главного файла равен `0` |
+| `__DATE__` | дата запуска препроцессора в форме `"Mmm dd yyyy"` |
+| `__TIME__` | время запуска препроцессора в форме `"hh:mm:ss"` |
+
+`__DATE__` и `__TIME__` получают один timestamp на весь
+translation unit. Специальное раскрытие помещается в output без повторного macro
+rescan.
+
+Эти имена находятся в общей macro table, поэтому `#undef` и последующий
+`#define` могут осознанно заменить builtin.
+
+### 8.2. Версия препроцессора
+
+Начиная с 0.0.8 standalone preprocessor не определяет GCC-имя `__VERSION__`.
+Оно относится к compiler environment, которого для будущего high-level языка
+пока нет. Собственная версия `mcpu-cpp` имеет отдельное однозначное имя:
+
+```text
+#define __MCPU_CPP_VERSION__ "1.0.2"
+```
+
+Значение автоматически берётся из `PACKAGE_VERSION`. Когда появится compiler
+frontend/driver, его version contract будет определён отдельно и не будет
+смешиваться с версией standalone preprocessor.
+
+### 8.3. Источники истины ABI
+
+`mcpu-cpp` собирается только GNU GCC. Во время `configure` проект использует
+проверенные приёмы из `LibMPU`/`LibMPUIO` `acsite.m4`: GCC predefined macros
+определяют native type sizes, byte/word order и machine-register width, а
+установленный `<libmpu.h>` является окончательным источником настроек LibMPU.
+
+В частности, фиксируются и проверяются:
+
+```text
+MPU_REAL_IO_LIMIT
+MPU_MATH_FN_LIMIT
+MPU_BYTE_ORDER
+MPU_WORD_ORDER
+BITS_PER_MACHINE_REGISTER
+BITS_PER_UNIT_T
+sizeof(__mpu_size_t)
+sizeof(__mpu_ptrdiff_t)
+```
+
+`configure` дополнительно проверяет, что byte order и
+`BITS_PER_MACHINE_REGISTER`, записанные в LibMPU, согласованы с GCC target,
+которым собирается `mcpu-cpp`. `MPU_WORD_ORDER` берётся непосредственно из
+configured LibMPU profile и описывает порядок слов MCPU data environment.
+
+Пределы `MPU_REAL_IO_LIMIT` и `MPU_MATH_FN_LIMIT` имеют разные назначения.
+Например, библиотека может иметь Real I/O до 65536 бит и математические функции
+только до 16384 бит. Поэтому `MPU_MATH_FN_LIMIT` не используется как предел
+существования типов Real.
+
+### 8.4. MCPU architecture и assembler prefixes
+
+Целевая архитектура определяется макро:
+
+```text
+#define _ARCH_MCPU 1
+```
+
+MCPU PTR64 имеет ширину 64 бита, поэтому определены `__SIZEOF_POINTER__`,
+`__MCPU_POINTER_WIDTH__`, `__INTPTR_TYPE__`, `__UINTPTR_TYPE__`, соответствующие
+width/max macros.
+
+Смысл assembler-prefix macros согласован с GNU CPP, а не с первой буквой имени
+register view. В синтаксисе `mcpu-as` дополнительного sigil перед register,
+label или immediate нет. `r` и `c` являются частью MCPU register syntax, а не
+`REGISTER_PREFIX`. Поэтому:
+
+```text
+#define __REGISTER_PREFIX__
+#define __LOCAL_LABEL_PREFIX__
+#define __USER_LABEL_PREFIX__
+#define __IMMEDIATE_PREFIX__
+```
+
+все четыре раскрываются в пустую последовательность. `.L...` остаётся
+compiler naming convention и не является assembler ABI local-label prefix:
+LOCAL/GLOBAL binding определяется symbol directives.
+
+### 8.5. Byte order и word order
+
+Базовые числовые значения порядка байт совместимы с GNU CPP:
+
+```text
+__ORDER_LITTLE_ENDIAN__
+__ORDER_BIG_ENDIAN__
+__ORDER_PDP_ENDIAN__
+```
+
+Но целевая среда публикует собственные MCPU names:
+
+```text
+#define __MCPU_BYTE_ORDER__ __ORDER_LITTLE_ENDIAN__
+#define __MCPU_WORD_ORDER__ __ORDER_LITTLE_ENDIAN__
+#define __BYTE_ORDER__ __MCPU_BYTE_ORDER__
+```
+
+Фактические значения `__MCPU_BYTE_ORDER__` и `__MCPU_WORD_ORDER__` получают из
+configured LibMPU profile (`MPU_BYTE_ORDER` и `MPU_WORD_ORDER`). Поэтому они
+следуют host data representation, с которой собрана LibMPU. Это не меняет
+отдельный архитектурный контракт кодировки MCPU instruction bytecode.
+
+GNU/C-specific имя `__FLOAT_WORD_ORDER__` не определяется: типа `float` в
+будущем языке MCPU нет.
+
+Параметры LibMPU/MCPU environment публикуются в MCPU namespace:
+
+```text
+__MCPU_MACHINE_REGISTER_WIDTH__
+__MCPU_REAL_IO_LIMIT__
+__MCPU_MATH_FN_LIMIT__
+__MCPU_INT_MAX_WIDTH__
+__MCPU_REAL_MAX_WIDTH__
+__MCPU_COMPLEX_MAX_WIDTH__
+```
+
+`__MCPU_INT_MAX_WIDTH__` равен `NB_I_MAX * 8`, а Real/Complex maximum width
+равен configured `MPU_REAL_IO_LIMIT`. `__MCPU_MACHINE_REGISTER_WIDTH__` является
+значением `BITS_PER_MACHINE_REGISTER` установленной LibMPU. Пределы Real I/O и
+math functions не смешиваются: `MPU_REAL_IO_LIMIT` определяет существование
+Real/Complex type family и text conversion, а `MPU_MATH_FN_LIMIT` — наличие
+математических функций соответствующей ширины.
+
+### 8.6. MCPU size/ssize, `ptrdiff` и pointers
+
+Будущий язык не наследует variable-width C names `short`, `int`, `long` и
+не использует C-style имя `size_t` как часть собственного ABI. Беззнаковый
+LibMPU size type и знаковый byte-count/error type публикуются симметрично в
+MCPU namespace. Например для 64-bit configured profile:
+
+```text
+#define __MCPU_SIZE_TYPE__ uint64
+#define __MCPU_SIZE_WIDTH__ 64
+#define __MCPU_SIZEOF_SIZE__ 8
+#define __MCPU_SIZE_MAX__ 0xffffffffffffffff
+
+#define __MCPU_SSIZE_TYPE__ int64
+#define __MCPU_SSIZE_WIDTH__ 64
+#define __MCPU_SIZEOF_SSIZE__ 8
+#define __MCPU_SSIZE_MAX__ 0x7fffffffffffffff
+```
+
+Это MCPU-specific family, а не попытка приписать GNU CPP несуществующий
+стандартный `__SSIZE_*` contract.
+
+MCPU pointer ABI от host не зависит: PTR64 всегда имеет ширину 64 бита:
+
+```text
+#define __INTPTR_TYPE__ int64
+#define __UINTPTR_TYPE__ uint64
+#define __INTPTR_WIDTH__ 64
+#define __UINTPTR_WIDTH__ 64
+#define __INTPTR_MAX__ 0x7fffffffffffffff
+#define __UINTPTR_MAX__ 0xffffffffffffffff
+#define __SIZEOF_POINTER__ 8
+#define __MCPU_POINTER_WIDTH__ 64
+```
+
+Разность MCPU pointers является знаковой и также фиксирована независимо от
+host:
+
+```text
+#define __PTRDIFF_TYPE__ int64
+#define __PTRDIFF_WIDTH__ 64
+#define __SIZEOF_PTRDIFF__ 8
+#define __PTRDIFF_MAX__ 0x7fffffffffffffff
+```
+
+Computed MIN expressions вроде `(-__PTRDIFF_MAX__ - 1)` в predefined table не
+создаются.
+
+### 8.7. Character types
+
+Обычного C `char` в будущем языке нет. Поэтому `__CHAR_TYPE__` и
+`__WCHAR_TYPE__` не определяются. Типы языка называются без C/C++ suffix `_t`:
+
+```text
+#define __CHAR8_TYPE__ char8
+#define __CHAR16_TYPE__ char16
+#define __CHAR8_WIDTH__ 8
+#define __CHAR16_WIDTH__ 16
+#define __SIZEOF_CHAR8__ 1
+#define __SIZEOF_CHAR16__ 2
+```
+
+Это типы будущего языка. Внутренняя реализация самого `mcpu-cpp` по-прежнему
+использует LibMPUIO `__mpu_char16_t` и strict UCS-2 text model.
+
+### 8.8. Integer families LibMPU
+
+Полная structural metadata integer families строится не по жёстко записанному
+последнему типу, а до `NB_I_MAX * 8` фактически установленной LibMPU. Для
+каждой power-of-two ширины от 8 бит определяются TYPE, WIDTH и SIZEOF:
+
+```text
+#define __INT1024_TYPE__ int1024
+#define __UINT1024_TYPE__ uint1024
+#define __INT1024_WIDTH__ 1024
+#define __UINT1024_WIDTH__ 1024
+#define __SIZEOF_INT1024__ 128
+#define __SIZEOF_UINT1024__ 128
+```
+
+На текущей LibMPU 1.0.25 `NB_I_MAX == 8192`, поэтому family доходит до
+`int65536`/`uint65536`, а `__SIZEOF_INT65536__ == 8192`.
+
+Decimal-digit metadata определяется для **каждой** разрешённой integer width:
+
+```text
+__INT<bits>_DECIMAL_DIG__
+__UINT<bits>_DECIMAL_DIG__
+```
+
+Значение вычисляется собственными integer-only helpers `mcpu-cpp` из известной
+ширины типа. Оно означает точное число десятичных цифр максимального значения
+соответствующего типа: знак и завершающий NUL в `DECIMAL_DIG` не входят. Для
+unsigned используется максимум `2^bits - 1`, для signed — `2^(bits-1) - 1`.
+Это отличается от LibMPU `_int_digs()`, которая предназначена для оценки
+строкового буфера и включает место для завершающего NUL.
+
+Например:
+
+```text
+#define __INT64_DECIMAL_DIG__ 19
+#define __UINT64_DECIMAL_DIG__ 20
+#define __INT256_DECIMAL_DIG__ 77
+#define __UINT256_DECIMAL_DIG__ 78
+```
+
+Только сами textual maxima намеренно ограничены шириной `bits <= 256`:
+
+```text
+__INT128_MAX__
+__UINT128_MAX__
+```
+
+Максимумы строятся через LibMPU `iuitoa()`. Макро `__INT<bits>_MIN__` не
+создаются: predefined table не должна содержать вычисляемые выражения вида
+`(-__INT<bits>_MAX__ - 1)`. Для widths больше 256 бит отсутствуют только MAX;
+TYPE/WIDTH/SIZEOF/DECIMAL_DIG сохраняются до полного `NB_I_MAX * 8`.
+
+### 8.9. Real и Complex families LibMPU
+
+Real/Complex structural metadata генерируется для каждой power-of-two ширины от
+32 бит до фактического configured `MPU_REAL_IO_LIMIT`. Для всех этих типов
+публикуются TYPE, WIDTH и SIZEOF.
+
+Для Complex WIDTH означает параметр типа, а не суммарную storage width:
+
+```text
+#define __COMPLEX128_TYPE__ complex128
+#define __COMPLEX128_WIDTH__ 128
+#define __SIZEOF_COMPLEX128__ 32
+```
+
+`complex128` состоит из двух компонентов `real128`, поэтому его storage size
+равен 32 байтам. При `MPU_REAL_IO_LIMIT == 65536` верх family имеет вид:
+
+```text
+#define __COMPLEX65536_TYPE__ complex65536
+#define __COMPLEX65536_WIDTH__ 65536
+#define __SIZEOF_COMPLEX65536__ 16384
+```
+
+Для Real соответственно:
+
+```text
+#define __REAL65536_TYPE__ real65536
+#define __REAL65536_WIDTH__ 65536
+#define __SIZEOF_REAL65536__ 8192
+```
+
+Precision metadata определяется для **всех** разрешённых Real widths вплоть
+до `MPU_REAL_IO_LIMIT`. Имена macros согласованы с LibMPU helpers:
+
+```text
+__REAL<bits>_DECIMAL_DIG__ -> _real_digs(bits/8)
+__REAL<bits>_MANT_DIG__ -> _real_mant_digs(bits/8)
+```
+
+`__REAL<bits>_DIG__` намеренно отсутствует. Ограничение `bits <= 256` относится
+только к большим textual numeric constants. Для размеров до 256 бит также
+определяются:
+
+```text
+__REAL<bits>_MAX__
+__REAL<bits>_MIN__
+__REAL<bits>_EPSILON__
+__REAL<bits>_MAX_EXP__
+__REAL<bits>_MIN_EXP__
+__REAL<bits>_MAX_10_EXP__
+__REAL<bits>_MIN_10_EXP__
+```
+
+Например, на LibMPU 1.0.25 для `real128` текущий profile даёт значения вида:
+
+```text
+#define __REAL128_EPSILON__ 2.524354896707237777317531409e-29
+#define __REAL128_MAX__ 4.197157432934775384808581951e+323228496
+#define __REAL128_MIN__ 9.530259619551804292864984035e-323228497
+#define __REAL128_MAX_10_EXP__ 323228496
+#define __REAL128_MAX_EXP__ 1073741823
+#define __REAL128_MIN_10_EXP__ -323228524
+#define __REAL128_MIN_EXP__ -1073741822
+```
+
+MAX/MIN/EPSILON создаются самой LibMPU и преобразуются через
+`real_to_ascii()`. Exponent constants получают значения через LibMPU exponent
+helpers и integer conversion. Для widths больше 256 бит эти numeric predefines отсутствуют, но
+TYPE/WIDTH/SIZEOF/DECIMAL_DIG/MANT_DIG продолжаются до `MPU_REAL_IO_LIMIT`.
+
+Для каждого разрешённого Real type вплоть до `MPU_REAL_IO_LIMIT` также
+публикуются две компактные характеристики:
+
+```text
+#define __SIZEOF_REAL128_EXP__ 4
+#define __REAL128_MAX_STRLEN__ 60
+```
+
+`__SIZEOF_REALxxx_EXP__` непосредственно получает `_sizeof_exp(NB_Rxxx)`.
+`__REALxxx_MAX_STRLEN__` получает `_real_max_string(NB_Rxxx)` и означает
+максимальное **количество символов** текстового представления, а не количество
+байт. Поэтому для zero-terminated строки нужно резервировать не менее
+`__REALxxx_MAX_STRLEN__ + 1` элементов: для `char8` это столько же bytes, а для
+`char16` физический объём в bytes вдвое больше. Эти два metadata-macro
+определяются и для Real widths больше 256, поскольку сами их значения малы.
+
+### 8.10. Dump macros: `-dM`, `-dMP`
+
+Опция:
+
+```text
+mcpu-cpp -dM input.c
+```
+
+печатает только итоговые **непредопределённые** macros в форме `#define ...`.
+К этой группе относятся определения из основного файла и включённых headers, а
+также определения командной строки `-D`. Предопределённые macros самого
+MCPU-CPP в `-dM` не выводятся. Поэтому `-dM` предназначен прежде всего для
+короткой инспекции macro-state, созданного пользовательской программой.
+
+Опция:
+
+```text
+mcpu-cpp -dMP input.c
+```
+
+добавляет к тому же итоговому состоянию активные predefined macros MCPU-CPP.
+Вывод имеет две последовательные группы: сначала все predefined macros, затем
+все непредопределённые macros. Внутри каждой группы определения
+детерминированно сортируются по имени. Такое разделение удобно системному
+разработчику для инспекции preprocessing ABI и архитектурных свойств текущей
+MCPU environment, не смешивая их с пользовательскими определениями.
+
+Принадлежность к группе определяется происхождением macro, а не его именем.
+Macro, заданный через `-D` или `#define`, является обычным даже если его имя
+похоже на системное. Если predefined macro был удалён через `#undef`, он не
+печатается. Если после этого то же имя снова определено пользователем, новое
+определение относится к обычной группе и выводится в её части `-dMP`, а также
+в `-dM`. Тем самым оба режима показывают именно **итоговый macro-state**.
+
+Context-dependent `__FILE__`, `__LINE__`, `__DATE__`, `__TIME__`,
+`__BASE_FILE__` и `__INCLUDE_LEVEL__` в статическом dump не печатаются.
+Статические ABI/architecture predefined macros и вычисляемые static Real
+metadata выводятся в `-dMP`.
+
+Если input file указан, он сначала полностью препроцессируется, после чего
+выводится итоговый macro-state; обычный preprocessed text в режимах `-dM` и
+`-dMP` не выдаётся. Без input file используется stdin, поэтому пустой stdin с
+`-dM` даёт пустой dump, а `-dMP` позволяет получить набор активных static
+predefined macros текущей MCPU environment.
+
+`-dD` имеет другую семантику и этим разделением не затрагивается.
+
+### 8.11. Dump definitions: `-dD`
+
+Опция:
+
+```text
+mcpu-cpp -dD input.c
+```
+
+сохраняет обычный результат препроцессирования и одновременно выводит
+встреченные директивы `#define`. Перед началом основного входного текста
+печатаются статические предопределённые macro definitions. Каждой такой
+дефиниции предшествует marker:
+
+```text
+# 0 "<built-in>"
+#define NAME value
+```
+
+а перед блоком предопределённых macro выводится marker исходного файла вида
+`# 0 "input.c"`. Context-dependent `__FILE__`, `__LINE__`, `__DATE__`,
+`__TIME__`, `__BASE_FILE__` и `__INCLUDE_LEVEL__` в начальный built-in block
+не включаются.
+
+### 8.12. Dump configuration: `-dconfig`
+
+Опция:
+
+```text
+mcpu-cpp -dconfig
+```
+
+не требует input file и выводит effective variables configuration layer после чтения runtime config, необязательного system override, домашнего
+user override или выбранного `--config-file`, включая expansion
+`$NAME`/`${NAME}`. Строки сортируются по имени и печатаются в форме:
+
+```text
+NAME = value;
+```
+
+Это позволяет проверить реальные include paths без ручного поиска
+`<runtime-root>/etc/mcpu-cpp.conf`, `/etc/mcpu/mcpu-cpp.conf` и
+`$HOME/.mcpu/mcpu-cpp.conf`.
+
+### 8.13. Verbose configuration snapshot: `-v`
+
+При `-v` MCPU-CPP сохраняет прежний runtime trace для `#lang`, `#include` и
+`#include_next`, но конфигурационные переменные печатаются только один раз —
+после чтения всех уровней configuration и применения правил приоритета. Поэтому
+в verbose output видны только **effective values**, а промежуточные значения из
+runtime-root, system и user config не дублируются.
+
+Config-блок выводится в порядке include policy: language-specific user paths,
+общий user path, system root и AFTER path. Переменная, отсутствующая во всех
+уровнях configuration, не печатается. Runtime-derived default
+`MCPU_CPP_SYSTEM_INCLUDE_PATH` является полноценным самым нижним значением и
+поэтому виден при `-v`, даже если ни один `mcpu-cpp.conf` не найден **или все
+config-файлы отключены опцией `--no-config`**.
+
+Форма строки:
+
+```text
+config: NAME=value
+```
+
+### 8.14. Effective search directories: `-dsearch-dirs`
+
+Опция:
+
+```text
+mcpu-cpp -dsearch-dirs
+```
+
+не требует input file, печатает effective глобальные каталоги поиска и
+завершает работу без preprocessing. Формат намеренно прост:
+
+```text
+search: /path/to/directory
+```
+
+Каталоги выводятся в семантическом порядке классов поиска:
+
+```text
+explicit -I
+explicit -isystem
+configured language-specific user directories
+MCPU_CPP_INCLUDE_PATH
+MCPU_CPP_SYSTEM_INCLUDE_PATH/<lang>
+MCPU_CPP_SYSTEM_INCLUDE_PATH
+explicit -idirafter
+MCPU_CPP_AFTER_INCLUDE_PATH
+```
+
+Language-specific entries печатаются для всех поддерживаемых языков в их
+каноническом порядке. Во время реального `#include` из этой группы участвует
+только каталог активного `#lang`. Каталог текущего физического файла в
+`-dsearch-dirs` не выводится: он существует только динамически для конкретного
+`#include "..."` и меняется вместе с include stack. `--no-config` не удаляет
+runtime-derived system root, поэтому без конфигурационных файлов dump всё равно
+содержит `<runtime-root>/include/<lang>` и `<runtime-root>/include`. `-nostdinc` удаляет
+из dump effective system `<lang>` entries и system root, но не explicit
+`-isystem`. Не существующий на filesystem каталог всё равно показывается,
+поскольку он является элементом effective search configuration и просто будет
+пропущен при реальном поиске файла.
+
+`-dsearch-dirs` учитывает `-I`, `-isystem`, `-idirafter`, все уровни config и
+replacement-семантику `MCPU_CPP_SYSTEM_INCLUDE_PATH`. Опция `-o` вместе с ним
+является ошибкой.
+
+### 8.15. Условная компиляция
+
+Директивы `#if`, `#ifdef`, `#ifndef`, `#elif`, `#else` и `#endif` обрабатываются
+как управляющие директивы препроцессора и в выходной поток не копируются, в
+том числе при `-dD`. Неактивные ветви пропускаются без выполнения находящихся
+в них `#define`, `#undef` и `#include`; вложенные условные группы при этом
+учитываются корректно.
+
+Выражение `#if` сначала обрабатывает оператор
+`defined`, затем выполняется macro expansion, а оставшиеся идентификаторы
+имеют значение `0`. Поддерживаются арифметические, битовые, сравнительные и
+логические операции, `?:` и short-circuit semantics для `&&`, `||` и `?:`.
+
+Начиная с 0.0.26 синтаксис выражения разбирается parser-ом, генерируемым
+ZUBR 4.1.0 из `src/mcpp-expr.zubr`; в том же файле находится UCS-2 lexical
+analyzer. Предварительная обработка `defined` и macro expansion выполняются до
+входа в parser. Арифметическая семантика вынесена в `mcpp-semantic.c/h` и не
+зависит от размеров целых типов host-системы. Generated `mcpp-expr.c`
+включается в release, поэтому ZUBR требуется только при изменении grammar.
+
+#### 8.15.1. Единственная вычислительная разрядность — 64 бита
+
+MCPU-CPP является препроцессором, а не компилятором языка общего назначения.
+Все целочисленные вычисления в директивах условной компиляции выполняются
+только в 64-разрядной арифметике. Препроцессор не выполняет арифметику LibMPU
+произвольной разрядности, вещественные или комплексные вычисления.
+
+Если программисту не требуется управлять двоичным представлением литерала,
+достаточно обычных целых констант и необязательного `U`/`u`. Например:
+
+```c
+#if 2 > 1
+#if 0xffffffffffffffffU > 1
+```
+
+Числовой lexeme хранится в UCS-2 до классификации, после чего его ASCII-часть
+передаётся LibMPU `iatoui()`. Поддерживаются `0b...`, `0...`, decimal и
+`0x...`. Значение, не помещающееся в 64 бита, является ошибкой. Старые C
+suffixes `L`, `l`, `LL`, `ll` не поддерживаются.
+
+#### 8.15.2. Суффикс разрядности `zNNN[Uu]`
+
+MCPU-CPP понимает общий для MCPU-языков суффикс разрядности:
+
+```text
+zNNN
+ZNNN
+zNNNu
+zNNNU
+ZNNNu
+ZNNNU
+```
+
+`NNN` — непустая последовательность десятичных цифр и **всегда** читается как
+десятичное число, даже если начинается с нулей. Поэтому `z8`, `z08` и `z008`
+задают одну и ту же разрядность 8 бит.
+
+В общем синтаксисе MCPU корректная разрядность должна быть степенью двойки от
+8 до `MPU_REAL_IO_LIMIT`. MCPU-CPP, однако, сознательно ограничен 64-битными
+вычислениями:
+
+* `z8`, `z16`, `z32`, `z64` и варианты регистра допустимы;
+* значение `NNN > 64` немедленно является ошибкой: препроцессор не допускает
+ числовые константы разрядности выше 64 бит в директивах условной компиляции;
+* если `NNN <= 64`, но не задаёт допустимую степень двойки, например `z24`,
+ выводится warning и сам `zNNN` игнорируется;
+* необязательный следующий `U`/`u` задаёт unsigned и сохраняет своё значение
+ даже если некорректный `zNNN` был проигнорирован.
+
+После полного суффикса должна заканчиваться числовая preprocessing token.
+Оператор или punctuation начинает следующий token, поэтому допустимы
+`1z32u+2`, `(1z32u)` и `1z32u==1`. Записи вроде `1z32undefined`, `1z32ufoo` и
+`1z32$foo` являются ошибками и не разбиваются искусственно на число и имя.
+
+#### 8.15.3. Нормализация литерала
+
+Суффикс разрядности действует **только один раз — при формировании значения
+самой константы**. Разрядность не сохраняется в semantic value и не участвует
+в последующих операциях.
+
+Для `VALUEzNNN` значение считается знаковым N-битным числом в дополнительном
+коде:
+
+1. сохраняются младшие `NNN` бит;
+2. результат расширяется со знаком до 64 бит.
+
+Для `VALUEzNNNu`/`VALUEzNNNU` сохраняются младшие `NNN` бит, после чего
+выполняется нулевое расширение до 64 бит.
+
+Например:
+
+```text
+0x7fz8 -> 0x000000000000007f -> 127
+0x80z8 -> 0xffffffffffffff80 -> -128
+0xffz8 -> 0xffffffffffffffff -> -1
+0x80z8u -> 0x0000000000000080 -> 128
+0xffz8u -> 0x00000000000000ff -> 255
+0x1ffz8 -> 0xffffffffffffffff -> -1
+0x1ffz8u -> 0x00000000000000ff -> 255
+```
+
+Последние два примера намеренны: `zNNN` задаёт разрядность **двоичного
+представления**, а не проверку математического диапазона. Биты старше N
+отбрасываются до расширения.
+
+После этой нормализации никакой `z8`, `z16` или `z32` в вычислительной модели
+уже не существует. Внутреннее значение содержит только 64-битный битовый
+образ и признак signed/unsigned.
+
+#### 8.15.4. Все последующие операции — 64-битные
+
+После нормализации все арифметические, побитовые, сравнительные и логические
+операции выполняются над 64-битными операндами. Результат операции не
+усекается обратно до разрядности исходного suffix. Поэтому:
+
+```text
+0x7fz8 + 1 -> 128
+0xffz8u + 1 -> 256
+```
+
+а не `-128` и `0` соответственно. Аналогично `~0xffz8u` инвертирует все 64
+бита и даёт `0xffffffffffffff00`.
+
+Для бинарных операций, где signedness имеет значение, наличие unsigned
+операнда переводит операцию в 64-битную unsigned-интерпретацию. Сравнения
+возвращают `0` или `1`. Логические `!`, `&&`, `||` также возвращают signed
+64-битные `0` или `1`; short-circuit не вычисляет невыбранную часть.
+
+Сдвиги выполняются после 64-битной нормализации. Правый сдвиг signed
+отрицательного значения является арифметическим, unsigned — логическим.
+Например:
+
+```text
+0x80z8 >> 1 -> -64
+0x80z8u >> 1 -> 64
+```
+
+Историческое правило MCPU-CPP для отрицательного счётчика сдвига сохраняется:
+`A << -N` эквивалентно `A >> N`, а `A >> -N` — `A << N`.
+
+Таким образом, `zNNN` не превращает препроцессор в компилятор с системой
+integer promotions разных размеров. Он лишь позволяет явно описать битовый
+образ исходного литерала; затем выражение вычисляется в единственной простой
+64-битной модели.
+
+#### 8.15.5. Символьные константы
+
+Символьная единица имеет тип `__mpu_uint16_t`, соответствующий внутреннему
+UCS-2 представлению, и перед вычислением расширяется нулями до 64 бит.
+Последующая арифметика снова является обычной 64-битной арифметикой.
+
+Состояние условной компиляции хранится в отдельном стеке; условная группа не
+может пересекать границу include-файла.
+
+### 8.16. Диагностические директивы `#error` и `#warning`
+
+MCPU-CPP поддерживает стандартные диагностические директивы:
+
+```text
+#error сообщение
+#warning сообщение
+```
+
+`#error` выдаёт diagnostic уровня error с текущими логическими именем файла и
+номером строки и немедленно завершает preprocessing с ошибкой. `#warning`
+выдаёт warning с той же source-location information, после чего preprocessing
+продолжается. Поэтому предшествующий `#line` влияет на координаты обеих
+диагностик.
+
+Остаток строки после имени директивы
+**не подвергается macro expansion**. Например:
+
+```c
+#define MESSAGE expanded
+#warning MESSAGE
+```
+
+печатает `MESSAGE`, а не `expanded`. Это отличает диагностические директивы от
+`#if` и `#line`, где macro expansion является частью соответствующего
+контракта.
+
+Комментарии удаляются на обычной preprocessing phase до обработки директивы.
+Начальные и конечные пробелы сообщения удаляются, последовательности пробельных
+символов между preprocessing tokens сворачиваются в один пробел. Пробелы внутри
+кавычек сохраняются. Например:
+
+```c
+#warning one /* comment */ two
+#warning "a b"
+```
+
+дают сообщения соответственно `one two` и `"a b"`. Unicode-текст проходит
+через внутреннее UCS-2 представление и выводится во внешнюю диагностику в UTF-8.
+
+Обе директивы являются управляющими и никогда не копируются в обычный выходной
+поток или в `-dD`. В неактивной ветви `#if` они полностью игнорируются, поэтому
+обычная защитная конструкция работает ожидаемо:
+
+```c
+#if 0
+#error this error is inactive
+#endif
+```
+
+### 8.17. Управление предупреждениями: `-Wcomment`, `-Wall`, `-Werror`
+
+MCPU-CPP разделяет обязательные предупреждения, являющиеся частью уже
+зафиксированной preprocessing-семантики, и дополнительные классы предупреждений,
+которые включаются пользователем. Управление предупреждениями не изменяет
+семантику `-dD`, macro expansion, conditional compilation или include search.
+
+Опции `-Wcomment` и `-Wcomments` являются полными синонимами и включают два
+лексических предупреждения:
+
+* последовательность `/*`, встретившуюся внутри уже открытого `/* ... */`
+ комментария;
+* backslash-newline внутри `//` комментария, из-за которого однострочный
+ комментарий физически продолжается на следующую строку.
+
+По умолчанию этот дополнительный класс выключен. `-Wall` включает все
+дополнительные warning classes MCPU-CPP; в версии 0.0.40 таким классом является
+`-Wcomment`. Формы `-Wno-comment` и `-Wno-comments` выключают его. Как в GNU
+warning model, более специфическая настройка имеет приоритет над групповой
+независимо от порядка аргументов. Поэтому обе команды:
+
+```text
+mcpu-cpp -Wall -Wno-comment file.c
+mcpu-cpp -Wno-comment -Wall file.c
+```
+
+оставляют comment warnings выключенными. Между настройками одинаковой
+специфичности действует последнее указание, например `-Wno-comment -Wcomment`
+включает этот класс.
+
+`-Werror` не включает никаких новых warning classes. Он повышает до error любое
+предупреждение, которое в данном запуске действительно было бы выдано, и такой
+запуск завершается неуспешно. Это относится как к дополнительным comment
+warnings, так и к уже существующим обязательным предупреждениям MCPU-CPP:
+
+* активной директиве `#warning`;
+* недопустимой, но не превышающей 64 бита ширине `zNNN`;
+* переопределению macro другим replacement list;
+* результату `##`, не образующему один preprocessing token.
+
+Например:
+
+```text
+mcpu-cpp -Wcomment -Werror file.c
+```
+
+превращает найденный comment warning в error. В то же время один `-Werror` без
+`-Wcomment`/`-Wall` не заставляет MCPU-CPP искать optional comment warnings.
+
+`-Wno-error` возвращает обычную severity warning. Для `-Werror` и `-Wno-error`,
+имеющих одинаковую специфичность, действует последняя опция командной строки.
+Так, `-Werror -Wno-error` оставляет warnings предупреждениями, а
+`-Wno-error -Werror` снова повышает их до errors.
+
+В 0.0.40 намеренно не вводятся `-Werror=<class>`, `-Wno-error=<class>`,
+`-Wundef`, `-Wunused-macros`, `-Wtraditional` и другие компиляторные классы.
+Warning interface MCPU-CPP остаётся компактным и расширяется только тогда, когда
+новый класс действительно нужен самому preprocessing language.
+
+### 8.18. Идентификаторы UCS-2
+
+Начиная с 0.0.22 имена preprocessing identifiers больше не ограничены ASCII.
+Внутри `mcpu-cpp` текст уже представлен строгим UCS-2, а классификация символов
+выполняется locale-independent функциями LibMPUIO 1.0.4, построенными по Unicode
+18.0.0. Первый символ идентификатора должен быть `_` или иметь свойство
+`XID_Start`; последующие символы должны быть `_`, `$` или иметь свойство
+`XID_Continue`. Символ `$` является расширением `mcpu-cpp`: он разрешён только
+после первого символа и не может начинать identifier. Это правило едино для
+имён и параметров macro, `#undef`, `#ifdef`/`#ifndef`, `defined`, обычного macro
+expansion, `#`/`##`. Имена остаются case-sensitive. Surrogate code units
+`U+D800..U+DFFF` не являются допустимыми символами identifiers.
+
+Например, допустимы:
+
+```c
+#define АНДРЕЙ 1
+#define résumé 2
+#define ΩМЕГА 3
+#define VALUE$OLD 4
+```
+
+Например, `VALUE$OLD` допустим, а `$VALUE` недопустим, поскольку `$` не является
+identifier-start character.
+
+Combining marks и не-ASCII decimal digits могут входить в identifier в позициях
+`XID_Continue`, но не становятся автоматически допустимыми первыми символами.
+Синтаксис числовых констант от этого не меняется: его правила остаются правилами
+соответствующего языка, а не Unicode `isdigit`.
+
+### 8.19. Макросы командной строки `-D` и `-U`
+
+Начиная с 0.0.23 опции `-D` и `-U` являются полноценными действиями
+препроцессора. Поддерживаются формы:
+
+```text
+-DNAME
+-DNAME=VALUE
+-D'FUNC(a,b)=a+b'
+-UNAME
+```
+
+`-DNAME` эквивалентна `#define NAME 1`; наличие `=` с пустой правой частью
+задаёт пустой replacement list. Function-like определения используют тот же
+macro engine, что и обычный `#define`, включая параметры, `#`, `##` и
+последующий rescanning. `-U` использует тот же identifier contract, что и
+`#undef`. Действия `-D`/`-U` выполняются в порядке командной строки после
+установки predefined macros.
+
+Только payload опций `-D` и `-U` интерпретируется как UTF-8 и преобразуется в
+строгий UCS-2. Имена файлов, `-I`, другие pathname arguments и остальные
+аргументы командной строки остаются исходными byte strings и не подвергаются
+Unicode-конвертации.
+
+Начиная с 0.0.25 символ `$` разрешён внутри имени macro, но не в первой
+позиции. При передаче `$` из shell пользователь обязан учитывать правила самого
+shell: shell обрабатывает `$` **до запуска `mcpu-cpp`**. Одинарные кавычки уже
+полностью защищают `$`, например:
+
+```sh
+mcpu-cpp '-DАНДРЕЙ$_Y=62' input.c
+```
+
+Без кавычек `$` следует экранировать:
+
+```sh
+mcpu-cpp -DАНДРЕЙ\$_Y=62 input.c
+```
+
+или использовать двойные кавычки с экранированием:
+
+```sh
+mcpu-cpp -D"АНДРЕЙ\$_Y=62" input.c
+```
+
+Вариант без защиты:
+
+```sh
+mcpu-cpp -DАНДРЕЙ$_Y=62 input.c
+```
+
+не передаёт написанное имя буквально: `$...` сначала раскрывается shell и
+`mcpu-cpp` получает уже изменённый `argv`. Внутри одинарных кавычек обратная
+косая черта перед `$` не нужна и стала бы обычным символом аргумента.
+
+Для command-line `-D` левая часть до первого `=` разбирается как отдельный
+macro declarator. Если после допустимого имени (или завершённого списка
+параметров function-like macro) до `=` встречается недопустимый хвост, этот
+хвост молча отбрасывается и **никогда не превращается в replacement list**.
+Например:
+
+```text
+-D'АНДРЕЙ@XYZ=62'
+```
+
+эквивалентно:
+
+```c
+#define АНДРЕЙ 62
+```
+
+а не ошибочной форме `#define АНДРЕЙ @XYZ 62`. Аналогичное правило допустимого
+identifier-prefix применяется к `-U`. Если же первый символ вообще не является
+допустимым identifier-start character (например `$` или цифра), определение
+остаётся ошибочным.
+
+При `-dD` определения, пришедшие через `-D`, маркируются отдельно от
+предопределённых macro:
+
+```text
+# 0 "<command-line>"
+#define NAME value
+```
+
+в то время как predefined macros продолжают использовать `<built-in>`.
+
+### 8.20. Публичный интерфейс командной строки
+
+`mcpu-cpp` поддерживает только актуальные опции, описанные `--help`. Устаревшие
+compatibility-флаги не образуют скрытый интерфейс и диагностируются как
+`unknown option`. Опция `-E` является исключением: она молча принимается и
+игнорируется, поскольку может передаваться compiler driver при запуске
+отдельного препроцессора.
+
+Опция `--object-suffix SUFFIX` задаёт суффикс object target, используемый при
+генерации make-зависимостей; аргумент обязателен.
+
+## 9. Build-system и генераторы
+
+Собственные Autoconf-макросы проекта находятся в корневом `acsite.m4`.
+Каталог `m4/` зарезервирован для внешних/vendor M4-файлов. Такой порядок
+повторяет принятую в библиотеках MCPU схему и не смешивает собственный
+configure-код с импортированными макросами.
+
+Парсер выражений `#if` генерируется ZUBR 4.1.0 из `src/mcpp-expr.zubr`.
+Release archive содержит и грамматику, и уже сгенерированный `src/mcpp-expr.c`,
+поэтому обычная сборка не требует установленного ZUBR. После изменения
+грамматики developer build использует штатное правило Automake:
+
+```text
+zubr -vl -s -Bmcpp_ -o mcpp-expr.c mcpp-expr.zubr
+```
+
+Перед выпуском release generated C должен соответствовать грамматике, полный
+test suite и `make distcheck` должны проходить без ошибок.
+
+### 9.1. Developer bootstrap и Git source tree
+
+Начиная с 0.0.50 корневой скрипт `./bootstrap` позволяет не хранить в Git файлы,
+которые полностью воспроизводятся из исходников. Скрипт сначала генерирует
+`src/mcpp-expr.c` из `src/mcpp-expr.zubr` с помощью ZUBR 4.1.0, затем выполняет
+`aclocal`, `autoheader`, `automake` и `autoconf` в стиле библиотек LibMPU и
+LibMPUIO. Опция `--target-dest-dir=DIR` задаёт target ROOTFS для системных
+Autoconf macro/include directories.
+
+Это правило относится именно к developer Git tree. **Release archive остаётся
+самодостаточным**, как и раньше: он содержит `configure`, `Makefile.in`, helper
+scripts Automake и уже сгенерированный `src/mcpp-expr.c`, поэтому обычная сборка
+релиза не требует предварительного запуска `bootstrap` и не требует ZUBR.
+
+Корневой `.gitignore` перечисляет воспроизводимые bootstrap-файлы и обычный
+configure/build state. Он не меняет существующую release/build model, а только
+позволяет поддерживать более чистый Git repository.
+
+## 10. GNU-compatible features
+
+`mcpu-cpp` является самостоятельным препроцессором MCPU, но для ряда хорошо
+известных операций намеренно повторяет поведение GNU CPP. Совместимость
+относится к документированным возможностям, а не означает полную CLI- или
+языковую взаимозаменяемость с GCC.
+
+В частности, GNU-compatible поведение используется для:
+
+* object-like и function-like macro, повторного macro rescan, `#` и `##`;
+* variadic macro `...` / `__VA_ARGS__` и стандартного `__VA_OPT__`;
+* `#if`, `#ifdef`, `#ifndef`, `#elif`, `#else`, `#endif` и `defined`;
+* `#include`, `#include_next`, `#pragma once`, `#line` и GNU linemarkers;
+* compact output mapping: до семи невидимых строк представляются newline, а
+ разрыв в восемь и более строк — корректирующим linemarker;
+* forced files `-include` / `-imacros` и dependency options `-M`, `-MM`, `-MD`,
+ `-MMD`, `-MF`, `-MT`, `-MQ`, `-MG`;
+* warning controls `-w`, `-Wall`, `-Werror` и поддерживаемых `-Wcomment` forms.
+
+MCPU-specific возможности, включая `#lang` / `#endlang`, числовой суффикс
+`zNNN` и ABI predefined macros, остаются собственными расширениями `mcpu-cpp`.
diff --git a/etc/Makefile.am b/etc/Makefile.am
new file mode 100644
index 0000000..6c8989b
--- /dev/null
+++ b/etc/Makefile.am
@@ -0,0 +1,4 @@
+mcpuprivateetcdir = @MCPU_CPP_ETCDIR@
+mcpuprivateetc_DATA = mcpu-cpp.conf
+
+EXTRA_DIST = mcpu-cpp.conf.in
diff --git a/etc/mcpu-cpp.conf.in b/etc/mcpu-cpp.conf.in
new file mode 100644
index 0000000..1791a59
--- /dev/null
+++ b/etc/mcpu-cpp.conf.in
@@ -0,0 +1,41 @@
+/*
+ * mcpu-cpp configuration.
+ *
+ * External text is UTF-8. Path values may use $NAME or ${NAME}
+ * references. A missing path is harmless; it is simply skipped when
+ * include files are searched.
+ */
+
+/*
+ * Main user include directory. It is available in every language state.
+ * Switched languages add their own language-specific include directories.
+ * Thus, for example, <diff/a.h> may always be named explicitly through
+ * this common include root.
+ */
+MCPU_CPP_INCLUDE_PATH = $HOME/.mcpu/include;
+MCPU_CPP_DIFF_INCLUDE_PATH = $HOME/.mcpu/include/diff;
+MCPU_CPP_DIFT_INCLUDE_PATH = $HOME/.mcpu/include/dift;
+MCPU_CPP_ALG_INCLUDE_PATH = $HOME/.mcpu/include/alg;
+MCPU_CPP_AS_INCLUDE_PATH = $HOME/.mcpu/include/as;
+MCPU_CPP_AVM_INCLUDE_PATH = $HOME/.mcpu/include/avm;
+MCPU_CPP_ACS_INCLUDE_PATH = $HOME/.mcpu/include/acs;
+
+/*
+ * Effective root of the MCPU system include tree.
+ *
+ * The default is not stored as an absolute installation path in this file.
+ * At run time mcpu-cpp locates its real executable, takes the parent of its
+ * executable directory as the MCPU installation root, and uses
+ * <runtime-root>/include. A higher-priority system or per-user configuration
+ * may replace that root completely by defining MCPU_CPP_SYSTEM_INCLUDE_PATH.
+ * Set an empty value in an override configuration to disable the configured
+ * system tree.
+ *
+ * Example override:
+ * MCPU_CPP_SYSTEM_INCLUDE_PATH = $HOME/mcpu-next/include;
+ */
+/*
+ * General fallback path-list searched after explicit -idirafter directories.
+ * No automatic language subdirectories are added here.
+ */
+MCPU_CPP_AFTER_INCLUDE_PATH = ;
diff --git a/m4/README b/m4/README
new file mode 100644
index 0000000..4c2cd60
--- /dev/null
+++ b/m4/README
@@ -0,0 +1,4 @@
+This directory is reserved for external/vendor Autoconf M4 macros.
+
+Project-owned mcpu-cpp macros belong in ../acsite.m4 so locally maintained
+configure logic remains clearly separated from imported macro packages.
diff --git a/man/Makefile.am b/man/Makefile.am
new file mode 100644
index 0000000..19d0e63
--- /dev/null
+++ b/man/Makefile.am
@@ -0,0 +1,6 @@
+
+SUBDIRS = ru
+
+MAN1 = mcpu-cpp.1
+
+dist_man_MANS = $(MAN1)
diff --git a/man/mcpu-cpp.1 b/man/mcpu-cpp.1
new file mode 100644
index 0000000..489de41
--- /dev/null
+++ b/man/mcpu-cpp.1
@@ -0,0 +1,432 @@
+.TH MCPU-CPP 1 "October 2026" "MCPU-CPP 1.0.2" "User Commands"
+.SH NAME
+mcpu-cpp \- preprocessor for MCPU languages
+.SH SYNOPSIS
+.B mcpu-cpp
+.RI [ options ]
+.RI [ input
+.RI [ output ]]
+.SH DESCRIPTION
+.B mcpu-cpp
+is the preprocessor for the MCPU toolchain. It processes source text before
+language-specific frontends, expands macros, evaluates conditional compilation,
+resolves include files, maintains source-location information, and can generate
+Make dependencies.
+.PP
+External text is UTF-8. Text is represented internally as strict UCS-2 through
+LibMPUIO. Preprocessing identifiers are Unicode-aware: the first character
+must be underscore or have the Unicode XID_Start property; following characters
+may also be dollar sign or have XID_Continue. The dollar sign cannot start an
+identifier.
+.PP
+Preprocessor directives use canonical English names only. Unicode remains
+available in identifiers, strings, comments, diagnostics, and ordinary source
+text.
+.PP
+If
+.I input
+is omitted or is
+.BR - ,
+standard input is read. If
+.I output
+is omitted or is
+.BR - ,
+normal preprocessing output is written to standard output.
+.SH TEXT PROCESSING
+Backslash-newline splicing is performed before directive parsing. C block
+comments and C++-style line comments are removed by the preprocessing phase.
+Long invisible source regions are represented compactly while preserving source
+coordinates: short forward gaps are emitted as newlines, while larger gaps use
+corrective GNU-style line markers.
+.SH DIRECTIVES
+The principal supported directives are:
+.TP
+.B #define
+Define an object-like or function-like macro.
+.TP
+.B #undef
+Remove a macro definition.
+.TP
+.BR #if , " #ifdef" , " #ifndef" , " #elif" , " #else" , " #endif"
+Control conditional compilation. The
+.B defined
+operator is supported in
+.B #if
+expressions.
+.TP
+.B #include
+Include a quoted or angle-bracket header. The operand may be produced by macro
+expansion.
+.TP
+.B #include_next
+Continue header lookup after the search-chain element that found the current
+header. It is intended primarily for wrapper headers.
+.TP
+.B #pragma once
+Process a physical header only once. Physical file identity is used rather
+than source spelling.
+.TP
+.B #line
+Set the logical source line and, optionally, logical source file name. Its
+arguments undergo macro expansion; output uses GNU-style line markers.
+.TP
+.B #error
+Emit an error diagnostic and terminate preprocessing unsuccessfully.
+.TP
+.B #warning
+Emit a warning diagnostic and continue unless warning policy promotes it to an
+error.
+.TP
+.B #lang
+Push an MCPU language state. The argument is one quoted language name.
+.TP
+.B #endlang
+Restore the previous MCPU language state.
+.PP
+.B #lang
+and
+.B #endlang
+remain in the output stream for the later frontend dispatcher. The supported
+language names are
+.BR diff ,
+.BR dift ,
+.BR alg ,
+.BR as ,
+.BR avm ,
+and
+.BR ACS .
+Language matching is case-insensitive in ASCII, while the quoted spelling is
+preserved in normalized
+.B #lang
+output.
+.SH MACROS
+Object-like and function-like macros are supported, including recursive rescan,
+stringification with
+.BR # ,
+token concatenation with
+.BR ## ,
+variadic macros using
+.B ...
+and
+.BR __VA_ARGS__ ,
+and standard
+.BR __VA_OPT__ .
+.PP
+Ordinary macro arguments are expanded before substitution except where raw
+arguments are required by stringification or token concatenation. Replacement
+list horizontal whitespace is normalized without changing whitespace inside
+quoted tokens or actual arguments.
+.SH CONDITIONAL EXPRESSIONS
+.B #if
+expressions support integer and character constants, the
+.B defined
+operator, unary, arithmetic, shift, relational, equality, bitwise, logical,
+conditional
+.BR ?: ,
+and comma operators, with short-circuit evaluation for
+.BR && ,
+.BR || ,
+and
+.BR ?: .
+.PP
+Expression evaluation uses one 64-bit model. MCPU also supports the
+.B zNNN
+and
+.B zNNNU
+width suffixes for source integer literals with widths not exceeding 64 bits.
+A negative shift count reverses shift direction according to the MCPU
+preprocessing contract.
+.SH INCLUDE SEARCH
+For
+.BR "#include \"file\"" ,
+the physical directory containing the current source file is searched first.
+This step is omitted for
+.BR "#include <file>" .
+The remaining effective search order is:
+.PP
+.nf
+explicit -I
+explicit -isystem
+MCPU_CPP_<LANG>_INCLUDE_PATH
+MCPU_CPP_INCLUDE_PATH
+MCPU_CPP_SYSTEM_INCLUDE_PATH/<lang>
+MCPU_CPP_SYSTEM_INCLUDE_PATH
+explicit -idirafter
+MCPU_CPP_AFTER_INCLUDE_PATH
+.fi
+.PP
+Within one search class, insertion order is preserved. The logical file name
+set by
+.B #line
+does not alter quoted-header lookup.
+.SH OPTIONS
+.TP
+.BI -o " FILE"
+Write normal preprocessing output to
+.IR FILE .
+.TP
+.BI -D " NAME[=VALUE]"
+Define a command-line macro. If no value is given, the replacement is
+.BR 1 .
+.TP
+.BI -U " NAME"
+Undefine a command-line macro.
+.TP
+.BI -imacros " FILE"
+Preprocess
+.I FILE
+for its macro and preprocessing state, but discard its ordinary output. All
+.B -imacros
+files are processed before all
+.B -include
+files.
+.TP
+.BI -include " FILE"
+Preprocess
+.I FILE
+before the primary input.
+.TP
+.BR "-I DIR" ", " "-IDIR"
+Add a user include directory.
+.TP
+.BI -isystem " DIR"
+Add an explicit system include directory.
+.TP
+.BI -idirafter " DIR"
+Add an include directory searched after the configured system tree.
+.TP
+.B -nostdinc
+Suppress the effective standard-system include tree. Explicit
+.B -isystem
+directories remain active.
+.TP
+.B -dM
+After preprocessing, dump non-predefined macro definitions.
+.TP
+.B -dMP
+Dump predefined macros first, followed by the other macro definitions.
+.TP
+.B -dD
+Preserve
+.B #define
+directives in normal preprocessing output.
+.TP
+.B -dconfig
+Print the effective MCPU-CPP configuration and exit.
+.TP
+.B -dsearch-dirs
+Print the effective include search directories and exit.
+.TP
+.B -M
+Write one Make dependency rule including system headers and suppress normal
+preprocessing output.
+.TP
+.B -MM
+Like
+.BR -M ,
+but omit system dependencies.
+.TP
+.B -MG
+With dependency-only
+.B -M
+or
+.BR -MM ,
+treat missing headers and missing forced files as generated dependencies rather
+than errors. It is not valid with
+.B -MD
+or
+.BR -MMD .
+.TP
+.B -MD
+Generate dependencies including system headers while retaining normal
+preprocessing output.
+.TP
+.B -MMD
+Like
+.BR -MD ,
+but omit system dependencies.
+.TP
+.BI -MF " FILE"
+Write dependencies to
+.IR FILE .
+A file name of
+.B -
+means standard output. This option requires dependency generation.
+.TP
+.BI -MT " TARGET"
+Set an explicit Make dependency target without Make quoting. The option may be
+repeated.
+.TP
+.BI -MQ " TARGET"
+Set an explicit Make dependency target with Make quoting. The option may be
+repeated.
+.TP
+.BI --object-suffix " SFX"
+Set the suffix used for the automatically derived dependency target. The
+default is
+.BR .o .
+.TP
+.B -w
+Suppress all warnings.
+.TP
+.BR -Wcomment , " -Wcomments"
+Enable warnings for nested
+.B /*
+inside a block comment and for backslash-newline inside a
+.B //
+comment.
+.TP
+.BR -Wno-comment , " -Wno-comments"
+Disable comment warnings, including when
+.B -Wall
+is present.
+.TP
+.B -Wall
+Enable all optional MCPU-CPP warning classes.
+.TP
+.B -Werror
+Promote every warning that would be emitted to an error. This does not enable
+new warning classes.
+.TP
+.B -Wno-error
+Keep emitted warnings as warnings.
+.TP
+.BI --config-file " FILE"
+Read only
+.I FILE
+as the explicit configuration layer on top of runtime-derived defaults.
+.TP
+.B --no-config
+Do not read configuration files. Runtime-derived defaults remain active.
+.TP
+.BR -v , " --verbose"
+Print configuration and include activity.
+.TP
+.B --help
+Print command-line help and exit successfully.
+.TP
+.B --version
+Print the program version and exit successfully.
+.TP
+.B -E
+Accepted and ignored for compiler-driver compatibility. It is intentionally
+not listed by
+.BR --help .
+.SH DEPENDENCIES
+Dependency generation uses the same include traversal as ordinary
+preprocessing, including conditional compilation, computed includes,
+.BR #include_next ,
+and
+.BR "#pragma once" .
+Physical dependencies are deduplicated by physical file identity.
+.PP
+With
+.B -M
+or
+.BR -MM ,
+normal preprocessing output is suppressed. With
+.B -MD
+or
+.BR -MMD ,
+normal preprocessing output is retained and a side-effect dependency file is
+written. If
+.B -MF
+is not specified, the dependency file name is derived from the input or
+ordinary
+.B -o
+output name and receives a
+.B .d
+suffix.
+.SH CONFIGURATION
+Before reading configuration files, MCPU-CPP derives
+.B MCPU_CPP_SYSTEM_INCLUDE_PATH
+as
+.IR <runtime-root>/include .
+The runtime root is determined from the actual executable location, normally
+through Linux
+.IR /proc/self/exe .
+.PP
+Configuration layers are applied in increasing priority:
+.PP
+.nf
+<runtime-root>/etc/mcpu-cpp.conf
+/etc/mcpu/mcpu-cpp.conf
+$HOME/.mcpu/mcpu-cpp.conf
+.fi
+.PP
+The user configuration path variables are
+.BR MCPU_CPP_INCLUDE_PATH ,
+.BR MCPU_CPP_AFTER_INCLUDE_PATH ,
+.BR MCPU_CPP_DIFF_INCLUDE_PATH ,
+.BR MCPU_CPP_DIFT_INCLUDE_PATH ,
+.BR MCPU_CPP_ALG_INCLUDE_PATH ,
+.BR MCPU_CPP_AS_INCLUDE_PATH ,
+.BR MCPU_CPP_AVM_INCLUDE_PATH ,
+and
+.BR MCPU_CPP_ACS_INCLUDE_PATH .
+The effective system root is controlled by
+.BR MCPU_CPP_SYSTEM_INCLUDE_PATH .
+A later definition replaces an earlier one, including an empty value.
+.SH SOURCE LOCATIONS
+MCPU-CPP emits GNU-style line markers of the form:
+.PP
+.nf
+# line "file" [flags]
+.fi
+.PP
+Flag 1 denotes entry into an included file and flag 2 denotes return to the
+including file. Input
+.B #line
+directives change the logical values observed by
+.B __LINE__
+and
+.BR __FILE__ .
+The predefined source macros also include
+.B __BASE_FILE__
+and
+.BR __INCLUDE_LEVEL__ .
+.SH GNU-COMPATIBLE FEATURES
+MCPU-CPP is an independent MCPU preprocessor, but intentionally follows GNU CPP
+behavior for documented operations including object-like and function-like
+macros, rescan,
+.BR # ,
+.BR ## ,
+variadic macros,
+.BR __VA_OPT__ ,
+conditional directives,
+.BR #include ,
+.BR #include_next ,
+.BR "#pragma once" ,
+.BR #line ,
+GNU line markers, forced files, Make dependency options, and supported warning
+controls.
+.PP
+MCPU-specific facilities such as
+.BR "#lang / #endlang" ,
+the
+.B zNNN
+integer-literal suffix, and MCPU ABI predefined macros are native extensions.
+.SH FILES
+.TP
+.I <runtime-root>/etc/mcpu-cpp.conf
+Configuration installed with the relocatable MCPU runtime tree.
+.TP
+.I /etc/mcpu/mcpu-cpp.conf
+Optional machine-wide configuration override.
+.TP
+.I $HOME/.mcpu/mcpu-cpp.conf
+Optional per-user configuration override with the highest normal configuration
+priority.
+.SH EXIT STATUS
+.B mcpu-cpp
+returns zero after successful preprocessing or after successful informational
+operations such as
+.B --help
+and
+.BR --version .
+It returns nonzero when command-line processing, configuration, preprocessing,
+diagnostics promoted to errors, or output generation fails.
+.SH SEE ALSO
+.BR libmpu (7),
+.BR libmpuio (7),
+.BR zubr (1)
diff --git a/man/ru/Makefile.am b/man/ru/Makefile.am
new file mode 100644
index 0000000..cf13dfc
--- /dev/null
+++ b/man/ru/Makefile.am
@@ -0,0 +1,8 @@
+
+LANG = ru
+
+mandir = @mandir@/$(LANG)
+
+MAN1 = mcpu-cpp.1
+
+dist_man_MANS = $(MAN1)
diff --git a/man/ru/mcpu-cpp.1 b/man/ru/mcpu-cpp.1
new file mode 100644
index 0000000..c1bf7e3
--- /dev/null
+++ b/man/ru/mcpu-cpp.1
@@ -0,0 +1,428 @@
+.TH MCPU-CPP 1 "Октябрь 2026" "MCPU-CPP 1.0.2" "Команды пользователя"
+.SH ИМЯ
+mcpu-cpp \- препроцессор языков MCPU
+.SH СИНТАКСИС
+.B mcpu-cpp
+.RI [ параметры ]
+.RI [ входной-файл
+.RI [ выходной-файл ]]
+.SH ОПИСАНИЕ
+.B mcpu-cpp
+\- препроцессор инструментария MCPU. Он обрабатывает исходный текст до передачи
+языковым frontend, раскрывает макро, вычисляет условия условной компиляции,
+выполняет поиск include-файлов, поддерживает информацию о позиции в исходном
+тексте и может формировать зависимости Make.
+.PP
+Внешний текст имеет кодировку UTF-8. Внутри текст представлен строгим UCS-2
+средствами LibMPUIO. Идентификаторы препроцессора поддерживают Unicode: первый
+символ должен быть подчёркиванием или иметь свойство Unicode XID_Start;
+последующие символы дополнительно могут быть знаком доллара или иметь
+XID_Continue. Знак доллара не может начинать идентификатор.
+.PP
+Директивы препроцессора имеют только канонические английские имена. Unicode
+полностью сохраняется в идентификаторах, строках, комментариях, диагностике и
+обычном исходном тексте.
+.PP
+Если
+.I входной-файл
+не задан или равен
+.BR - ,
+читается стандартный ввод. Если
+.I выходной-файл
+не задан или равен
+.BR - ,
+обычный результат препроцессирования записывается в стандартный вывод.
+.SH ОБРАБОТКА ТЕКСТА
+Склейка backslash-newline выполняется до разбора директив. Блочные комментарии
+C и однострочные комментарии C++ удаляются фазой препроцессирования. Длинные
+невидимые участки исходного файла представляются компактно с сохранением
+исходных координат: короткие переходы задаются переводами строк, большие
+переходы \- корректирующими GNU-style linemarkers.
+.SH ДИРЕКТИВЫ
+Основные поддерживаемые директивы:
+.TP
+.B #define
+Определяет object-like или function-like macro.
+.TP
+.B #undef
+Удаляет определение макро.
+.TP
+.BR #if , " #ifdef" , " #ifndef" , " #elif" , " #else" , " #endif"
+Управляют условной компиляцией. В выражениях
+.B #if
+поддерживается оператор
+.BR defined .
+.TP
+.B #include
+Включает заголовочный файл в кавычках или угловых скобках. Operand может быть
+получен macro expansion.
+.TP
+.B #include_next
+Продолжает поиск header после элемента search chain, которым был найден текущий
+header. Директива предназначена прежде всего для wrapper headers.
+.TP
+.B #pragma once
+Обрабатывает физический header только один раз. Используется физическая
+идентичность файла, а не написание его имени в исходном тексте.
+.TP
+.B #line
+Задаёт логический номер строки и, необязательно, логическое имя исходного файла.
+Аргументы проходят macro expansion; в выходе используются GNU-style
+linemarkers.
+.TP
+.B #error
+Выдаёт сообщение об ошибке и немедленно завершает препроцессирование неуспешно.
+.TP
+.B #warning
+Выдаёт предупреждение и продолжает работу, если политика предупреждений не
+повышает его до ошибки.
+.TP
+.B #lang
+Помещает состояние языка MCPU в стек. Аргументом является одно имя языка в
+кавычках.
+.TP
+.B #endlang
+Восстанавливает предыдущее состояние языка MCPU.
+.PP
+.B #lang
+и
+.B #endlang
+сохраняются в выходном потоке для последующего frontend dispatcher.
+Поддерживаются имена языков
+.BR diff ,
+.BR dift ,
+.BR alg ,
+.BR as ,
+.BR avm
+и
+.BR ACS .
+Сравнение имени языка выполняется без учёта ASCII-регистра, а написание внутри
+кавычек сохраняется в нормализованном выводе
+.BR #lang .
+.SH МАКРО
+Поддерживаются object-like и function-like macros, повторный macro rescan,
+stringification оператором
+.BR # ,
+token concatenation оператором
+.BR ## ,
+variadic macros с
+.B ...
+и
+.BR __VA_ARGS__ ,
+а также стандартный
+.BR __VA_OPT__ .
+.PP
+Обычные фактические аргументы раскрываются до подстановки, кроме случаев, когда
+stringification или token concatenation требуют сырого аргумента. Горизонтальные
+пробелы replacement list нормализуются без изменения пробелов внутри quoted
+tokens и фактических аргументов.
+.SH УСЛОВНЫЕ ВЫРАЖЕНИЯ
+Выражения
+.B #if
+поддерживают целые и символьные константы, оператор
+.BR defined ,
+унарные, арифметические, shift, relational, equality, bitwise, logical,
+условный оператор
+.B ?:
+и оператор comma. Для
+.BR && ,
+.BR ||
+и
+.B ?:
+используется short-circuit evaluation.
+.PP
+Вычисление выражений выполняется в единой 64-битной модели. MCPU дополнительно
+поддерживает суффиксы разрядности
+.B zNNN
+и
+.B zNNNU
+для исходных целых литералов с разрядностью не более 64 бит. Отрицательное
+значение shift count меняет направление сдвига согласно контракту
+препроцессора MCPU.
+.SH ПОИСК INCLUDE-ФАЙЛОВ
+Для
+.B "#include \"file\""
+сначала проверяется физический каталог текущего исходного файла. Для
+.B "#include <file>"
+этот шаг отсутствует. Остальная effective search chain имеет строгий порядок:
+.PP
+.nf
+explicit -I
+explicit -isystem
+MCPU_CPP_<LANG>_INCLUDE_PATH
+MCPU_CPP_INCLUDE_PATH
+MCPU_CPP_SYSTEM_INCLUDE_PATH/<lang>
+MCPU_CPP_SYSTEM_INCLUDE_PATH
+explicit -idirafter
+MCPU_CPP_AFTER_INCLUDE_PATH
+.fi
+.PP
+Внутри одного search class сохраняется порядок добавления. Логическое имя,
+заданное
+.BR #line ,
+не изменяет поиск quoted header.
+.SH ПАРАМЕТРЫ
+.TP
+.BI -o " ФАЙЛ"
+Записывает обычный результат препроцессирования в
+.IR ФАЙЛ .
+.TP
+.BI -D " ИМЯ[=ЗНАЧЕНИЕ]"
+Определяет макро командной строки.
+.TP
+.BI -U " ИМЯ"
+Отменяет определение макро командной строки.
+.TP
+.BI -imacros " ФАЙЛ"
+Полностью препроцессирует
+.I ФАЙЛ
+для изменения macro/preprocessing state, но отбрасывает его обычный вывод. Все
+.B -imacros
+обрабатываются раньше всех
+.BR -include .
+.TP
+.BI -include " ФАЙЛ"
+Препроцессирует
+.I ФАЙЛ
+до основного входного файла.
+.TP
+.BR "-I КАТАЛОГ" ", " "-IКАТАЛОГ"
+Добавляет пользовательский include-каталог.
+.TP
+.BI -isystem " КАТАЛОГ"
+Добавляет явный системный include-каталог.
+.TP
+.BI -idirafter " КАТАЛОГ"
+Добавляет include-каталог, который ищется после configured system tree.
+.TP
+.B -nostdinc
+Исключает effective standard-system include tree. Явные каталоги
+.B -isystem
+остаются активными.
+.TP
+.B -dM
+После препроцессирования выводит определения непредопределённых макро.
+.TP
+.B -dMP
+Сначала выводит предопределённые макро, затем остальные определения.
+.TP
+.B -dD
+Сохраняет директивы
+.B #define
+в обычном выходном потоке.
+.TP
+.B -dconfig
+Выводит effective configuration MCPU-CPP и завершает работу.
+.TP
+.B -dsearch-dirs
+Выводит effective include search directories и завершает работу.
+.TP
+.B -M
+Выводит одно правило зависимостей Make с системными headers и подавляет обычный
+результат препроцессирования.
+.TP
+.B -MM
+Аналог
+.BR -M ,
+но без системных зависимостей.
+.TP
+.B -MG
+В dependency-only режимах
+.B -M
+или
+.B -MM
+считает отсутствующие headers и forced files генерируемыми зависимостями, а не
+ошибками. С
+.B -MD
+и
+.B -MMD
+не используется.
+.TP
+.B -MD
+Формирует зависимости с системными headers, сохраняя обычный результат
+препроцессирования.
+.TP
+.B -MMD
+Аналог
+.BR -MD ,
+но без системных зависимостей.
+.TP
+.BI -MF " ФАЙЛ"
+Записывает зависимости в
+.IR ФАЙЛ .
+Имя
+.B -
+означает стандартный вывод. Параметр имеет смысл только при генерации
+зависимостей.
+.TP
+.BI -MT " ЦЕЛЬ"
+Задаёт явную цель правила Make без Make quoting. Параметр можно повторять.
+.TP
+.BI -MQ " ЦЕЛЬ"
+Задаёт явную цель правила Make с Make quoting. Параметр можно повторять.
+.TP
+.BI --object-suffix " СУФФИКС"
+Задаёт суффикс автоматически формируемой цели зависимости. По умолчанию
+используется
+.BR .o .
+.TP
+.B -w
+Подавляет все предупреждения.
+.TP
+.BR -Wcomment , " -Wcomments"
+Включает предупреждения о вложенном
+.B /*
+внутри блочного комментария и о backslash-newline внутри комментария
+.BR // .
+.TP
+.BR -Wno-comment , " -Wno-comments"
+Выключает comment warnings, в том числе при наличии
+.BR -Wall .
+.TP
+.B -Wall
+Включает все optional warning classes MCPU-CPP.
+.TP
+.B -Werror
+Повышает каждое реально выдаваемое предупреждение до ошибки, но само не
+включает новые классы предупреждений.
+.TP
+.B -Wno-error
+Оставляет выдаваемые предупреждения предупреждениями.
+.TP
+.BI --config-file " ФАЙЛ"
+Использует только явно выбранный
+.I ФАЙЛ
+как configuration layer поверх runtime-derived defaults.
+.TP
+.B --no-config
+Не читает configuration files. Runtime-derived defaults остаются активными.
+.TP
+.BR -v , " --verbose"
+Выводит информацию об effective configuration и include activity.
+.TP
+.B --help
+Выводит краткую справку по командной строке и успешно завершает работу.
+.TP
+.B --version
+Выводит версию программы и успешно завершает работу.
+.TP
+.B -E
+Принимается и игнорируется для совместимости с compiler drivers. Параметр
+намеренно не показывается в
+.BR --help .
+.SH ЗАВИСИМОСТИ
+Генерация зависимостей использует тот же include traversal, что и обычное
+препроцессирование, включая условную компиляцию, computed includes,
+.BR #include_next
+и
+.BR "#pragma once" .
+Физические зависимости дедуплицируются по физической идентичности файла.
+.PP
+В режимах
+.B -M
+и
+.B -MM
+обычный результат препроцессирования подавляется. В режимах
+.B -MD
+и
+.B -MMD
+он сохраняется, а dependency file формируется как side effect. Если
+.B -MF
+не задан, имя dependency file выводится из имени входного файла или обычного
+выхода
+.B -o
+и получает суффикс
+.BR .d .
+.SH КОНФИГУРАЦИЯ
+До чтения configuration files MCPU-CPP формирует
+.B MCPU_CPP_SYSTEM_INCLUDE_PATH
+как
+.IR <runtime-root>/include .
+Runtime root определяется из фактического расположения исполняемого файла,
+обычно через Linux
+.IR /proc/self/exe .
+.PP
+Configuration layers применяются в порядке возрастания приоритета:
+.PP
+.nf
+<runtime-root>/etc/mcpu-cpp.conf
+/etc/mcpu/mcpu-cpp.conf
+$HOME/.mcpu/mcpu-cpp.conf
+.fi
+.PP
+Пользовательские переменные путей:
+.BR MCPU_CPP_INCLUDE_PATH ,
+.BR MCPU_CPP_AFTER_INCLUDE_PATH ,
+.BR MCPU_CPP_DIFF_INCLUDE_PATH ,
+.BR MCPU_CPP_DIFT_INCLUDE_PATH ,
+.BR MCPU_CPP_ALG_INCLUDE_PATH ,
+.BR MCPU_CPP_AS_INCLUDE_PATH ,
+.BR MCPU_CPP_AVM_INCLUDE_PATH
+и
+.BR MCPU_CPP_ACS_INCLUDE_PATH .
+Effective system root задаётся
+.BR MCPU_CPP_SYSTEM_INCLUDE_PATH .
+Более позднее определение полностью заменяет раннее, включая пустое значение.
+.SH ПОЗИЦИИ В ИСХОДНОМ ТЕКСТЕ
+MCPU-CPP выводит GNU-style linemarkers вида:
+.PP
+.nf
+# line "file" [flags]
+.fi
+.PP
+Флаг 1 означает вход во включённый файл, флаг 2 \- возврат в включающий файл.
+Входная директива
+.B #line
+изменяет логические значения
+.B __LINE__
+и
+.BR __FILE__ .
+К предопределённым source macros также относятся
+.B __BASE_FILE__
+и
+.BR __INCLUDE_LEVEL__ .
+.SH GNU-COMPATIBLE FEATURES
+MCPU-CPP является самостоятельным препроцессором MCPU, но для документированных
+операций намеренно повторяет поведение GNU CPP. К ним относятся object-like и
+function-like macros, rescan,
+.BR # ,
+.BR ## ,
+variadic macros,
+.BR __VA_OPT__ ,
+условные директивы,
+.BR #include ,
+.BR #include_next ,
+.BR "#pragma once" ,
+.BR #line ,
+GNU linemarkers, forced files, параметры генерации зависимостей Make и
+поддерживаемые warning controls.
+.PP
+MCPU-специфичные средства
+.BR "#lang / #endlang" ,
+суффикс целых литералов
+.B zNNN
+и предопределённые макро ABI MCPU являются собственными расширениями.
+.SH ФАЙЛЫ
+.TP
+.I <runtime-root>/etc/mcpu-cpp.conf
+Конфигурация, установленная вместе с перемещаемым runtime tree MCPU.
+.TP
+.I /etc/mcpu/mcpu-cpp.conf
+Необязательная общесистемная configuration override.
+.TP
+.I $HOME/.mcpu/mcpu-cpp.conf
+Необязательная пользовательская configuration override с наивысшим обычным
+приоритетом.
+.SH КОД ЗАВЕРШЕНИЯ
+.B mcpu-cpp
+возвращает ноль после успешного препроцессирования и после успешных
+информационных операций, например
+.B --help
+и
+.BR --version .
+При ошибке командной строки, конфигурации, препроцессирования, диагностике,
+повышенной до ошибки, или ошибке вывода возвращается ненулевое значение.
+.SH СМ. ТАКЖЕ
+.BR libmpu (7),
+.BR libmpuio (7),
+.BR zubr (1)
diff --git a/src/Makefile.am b/src/Makefile.am
new file mode 100644
index 0000000..b5e2ea2
--- /dev/null
+++ b/src/Makefile.am
@@ -0,0 +1,64 @@
+
+mcpuprivatebindir = @MCPU_CPP_BINDIR@
+mcpuprivatebin_PROGRAMS = mcpu-cpp
+
+mcpu_cpp_SOURCES = \
+ main.c \
+ defs.h \
+ mcpp-options.c \
+ mcpp-options.h \
+ mcpp-diagnostic.c \
+ mcpp-diagnostic.h \
+ mcpp-text.c \
+ mcpp-text.h \
+ mcpp-config-file.c \
+ mcpp-config-file.h \
+ mcpp-runtime.c \
+ mcpp-runtime.h \
+ mcpp-language.c \
+ mcpp-language.h \
+ mcpp-include-path.c \
+ mcpp-include-path.h \
+ mcpp-source.c \
+ mcpp-source.h \
+ mcpp-lexer.c \
+ mcpp-lexer.h \
+ mcpp-macro.c \
+ mcpp-macro.h \
+ mcpp-semantic.c \
+ mcpp-semantic.h \
+ mcpp-expr.c \
+ mcpp-expr.h \
+ mcpp-expression.c \
+ mcpp-expression.h \
+ mcpp-predefined.c \
+ mcpp-predefined.h \
+ mcpp-lib.c \
+ mcpp-lib.h
+
+mcpu_cpp_CFLAGS = $(LIBMPUIO_CFLAGS)
+mcpu_cpp_LDFLAGS = $(LIBMPUIO_LDFLAGS)
+mcpu_cpp_LDADD = $(LIBMPUIO_LIBS)
+
+EXTRA_DIST = mcpp-expr.zubr
+
+ZUBR = zubr
+SUFFIXES = .zubr .c
+
+.zubr.c:
+ $(ZUBR) -vl -s -Bmcpp_ -o $@ $<
+
+CLEANFILES = z.output
+
+
+install-exec-hook:
+ $(MKDIR_P) "$(DESTDIR)$(bindir)"
+ rm -f "$(DESTDIR)$(bindir)/mcpu-cpp"
+ $(LN_S) "@MCPU_CPP_PUBLIC_LINK_TARGET@" "$(DESTDIR)$(bindir)/mcpu-cpp"
+
+uninstall-hook:
+ rm -f "$(DESTDIR)$(bindir)/mcpu-cpp"
+
+distclean-local:
+ -rm -rf $(DEPDIR)
+
diff --git a/src/defs.h b/src/defs.h
new file mode 100644
index 0000000..2b22f15
--- /dev/null
+++ b/src/defs.h
@@ -0,0 +1,42 @@
+#ifndef __MCPU_CPP_DEFS_H__
+#define __MCPU_CPP_DEFS_H__ 1
+
+#ifdef HAVE_CONFIG_H
+#include <config.h>
+#endif
+
+#include <errno.h>
+#include <limits.h>
+#include <signal.h>
+#include <stdint.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <time.h>
+
+#include <libmpuio.h>
+
+#define MCPU_CPP_LANG_STACK_SIZE 400
+#define MCPU_CPP_INCLUDE_STACK_SIZE 400
+#define MCPU_CPP_CONDITIONAL_STACK_SIZE 400
+
+#ifndef PATH_MAX
+#define PATH_MAX 4096
+#endif
+
+#include <mcpp-text.h>
+#include <mcpp-config-file.h>
+#include <mcpp-runtime.h>
+#include <mcpp-language.h>
+#include <mcpp-include-path.h>
+#include <mcpp-lexer.h>
+#include <mcpp-macro.h>
+#include <mcpp-semantic.h>
+#include <mcpp-expression.h>
+#include <mcpp-predefined.h>
+#include <mcpp-source.h>
+#include <mcpp-options.h>
+#include <mcpp-diagnostic.h>
+#include <mcpp-lib.h>
+
+#endif /* __MCPU_CPP_DEFS_H__ */
diff --git a/src/main.c b/src/main.c
new file mode 100644
index 0000000..0882904
--- /dev/null
+++ b/src/main.c
@@ -0,0 +1,412 @@
+#include <defs.h>
+
+static void
+welcome( mcpp_options *opts )
+{
+ if( opts == NULL )
+ return;
+
+ printf( "mcpu-cpp %s\n", PACKAGE_VERSION );
+ printf( "MCPU languages preprocessor\n" );
+}
+
+static void
+usage( mcpp_options *opts )
+{
+ printf( "Usage: %s [options] [input [output]]\n\n", opts->progname );
+ printf( "Options:\n" );
+ printf( " -o FILE write output to FILE\n" );
+ printf( " -D NAME[=VALUE] define a command-line macro\n" );
+ printf( " -U NAME undefine a command-line macro\n" );
+ printf( " -imacros FILE preprocess FILE for macro state; discard its output\n" );
+ printf( " -include FILE preprocess FILE before the primary input\n" );
+ printf( " -I DIR, -IDIR add user include directory\n" );
+ printf( " -isystem DIR add explicit system include directory\n" );
+ printf( " -idirafter DIR add include directory searched last\n" );
+ printf( " -nostdinc suppress effective standard-system include tree\n" );
+ printf( " -dM dump non-predefined macros after preprocessing\n" );
+ printf( " -dMP dump predefined macros first, then other macros\n" );
+ printf( " -dD preserve #define directives in normal output\n" );
+ printf( " -dconfig dump effective mcpu-cpp configuration\n" );
+ printf( " -dsearch-dirs dump effective include search directories\n" );
+ printf( " -M output make dependencies including system headers\n" );
+ printf( " -MM output make dependencies excluding system headers\n" );
+ printf( " -MG treat missing headers as generated dependencies\n" );
+ printf( " -MD write dependencies and keep preprocessing output\n" );
+ printf( " -MMD like -MD but exclude system headers\n" );
+ printf( " -MF FILE write dependencies to FILE ('-' means stdout)\n" );
+ printf( " -MT TARGET set unquoted make dependency target\n" );
+ printf( " -MQ TARGET set make-quoted dependency target\n" );
+ printf( " --object-suffix SFX set object suffix used for dependency targets\n" );
+ printf( " -w suppress all warnings\n" );
+ printf( " -Wcomment[s] warn about nested /* and multi-line // comments\n" );
+ printf( " -Wno-comment[s] disable comment warnings even under -Wall\n" );
+ printf( " -Wall enable all optional warning classes\n" );
+ printf( " -Werror promote every emitted warning to an error\n" );
+ printf( " -Wno-error keep emitted warnings as warnings\n" );
+ printf( " --config-file FILE use only FILE as configuration\n" );
+ printf( " --no-config do not read configuration files; keep runtime defaults\n" );
+ printf( " -v, --verbose print configuration/include activity\n" );
+ printf( " --help display this help and exit\n" );
+ printf( " --version display version and exit\n\n" );
+ printf( "If input is omitted or '-', read standard input.\n" );
+ printf( "If output is omitted or '-', write standard output.\n" );
+}
+
+static void
+pipe_closed( int sig )
+{
+ (void)sig;
+ _Exit( 0 );
+}
+
+static int
+apply_include_options( mcpp *cpp, const mcpp_options *opts )
+{
+ size_t i;
+
+ for( i = 0; i < opts->action_count; ++i )
+ {
+ const mcpp_option_action *action = &opts->actions[i];
+ enum mcpu_include_class class_type;
+
+ switch( action->kind )
+ {
+ case MCPP_ACTION_INCLUDE_USER:
+ if( action->argument[0] == 0 )
+ continue;
+ class_type = MCPU_INCLUDE_USER_EXPLICIT;
+ break;
+ case MCPP_ACTION_INCLUDE_SYSTEM:
+ class_type = MCPU_INCLUDE_SYSTEM_EXPLICIT;
+ break;
+ case MCPP_ACTION_INCLUDE_AFTER:
+ class_type = MCPU_INCLUDE_AFTER_EXPLICIT;
+ break;
+ default:
+ continue;
+ }
+
+ if( mcpu_include_paths_add(&cpp->include_paths,
+ action->argument, class_type) != 0 )
+ return( -1 );
+ }
+
+ return( 0 );
+}
+
+static int
+dump_verbose_config( const mcpp *cpp, FILE *stream )
+{
+ enum mcpu_language language;
+ const char *name;
+ const char *value;
+
+ if( cpp == NULL || stream == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ for( language = MCPU_LANG_DIFF;
+ language < MCPU_LANG__COUNT;
+ language = (enum mcpu_language)(language + 1) )
+ {
+ name = mcpu_language_config_variable( language );
+ value = mcpu_config_get( &cpp->config, name );
+ if( value != NULL && fprintf(stream, "config: %s=%s\n", name, value) < 0 )
+ return( -1 );
+ }
+
+ name = "MCPU_CPP_INCLUDE_PATH";
+ value = mcpu_config_get( &cpp->config, name );
+ if( value != NULL && fprintf(stream, "config: %s=%s\n", name, value) < 0 )
+ return( -1 );
+
+ name = "MCPU_CPP_SYSTEM_INCLUDE_PATH";
+ value = mcpu_config_get( &cpp->config, name );
+ if( value != NULL && fprintf(stream, "config: %s=%s\n", name, value) < 0 )
+ return( -1 );
+
+ name = "MCPU_CPP_AFTER_INCLUDE_PATH";
+ value = mcpu_config_get( &cpp->config, name );
+ if( value != NULL && fprintf(stream, "config: %s=%s\n", name, value) < 0 )
+ return( -1 );
+
+ return( ferror(stream) ? -1 : 0 );
+}
+
+
+static FILE *
+open_output( const mcpp_options *opts, int *close_stream )
+{
+ FILE *stream;
+
+ *close_stream = 0;
+ if( opts->out_fname == NULL || opts->out_fname[0] == 0 )
+ return( stdout );
+
+ stream = fopen( opts->out_fname, "wb" );
+ if( stream != NULL )
+ *close_stream = 1;
+
+ return( stream );
+}
+
+
+static char *
+dependency_default_filename( const mcpp_options *opts )
+{
+ const char *source;
+ const char *base;
+ const char *dot;
+ size_t stem_length;
+ char *filename;
+
+ if( opts == NULL )
+ {
+ errno = EINVAL;
+ return( NULL );
+ }
+
+ if( opts->out_fname != NULL && opts->out_fname[0] != 0 )
+ {
+ source = opts->out_fname;
+ base = source;
+ }
+ else if( opts->in_fname != NULL && opts->in_fname[0] != 0 )
+ {
+ base = strrchr( opts->in_fname, '/' );
+ source = base == NULL ? opts->in_fname : base + 1;
+ base = source;
+ }
+ else
+ {
+ source = "-";
+ base = source;
+ }
+
+ dot = strrchr( base, '.' );
+ stem_length = dot != NULL && dot != base ? (size_t)(dot - source) : strlen(source);
+
+ filename = (char *)malloc( stem_length + 3 );
+ if( filename == NULL )
+ return( NULL );
+
+ memcpy( filename, source, stem_length );
+ memcpy( filename + stem_length, ".d", 3 );
+ return( filename );
+}
+
+
+static FILE *
+open_dependency_output( const mcpp_options *opts, int side_effect,
+ char **allocated_name, int *close_stream )
+{
+ const char *name = NULL;
+ FILE *stream;
+
+ if( opts == NULL || allocated_name == NULL || close_stream == NULL )
+ {
+ errno = EINVAL;
+ return( NULL );
+ }
+
+ *allocated_name = NULL;
+ *close_stream = 0;
+
+ if( opts->deps_file != NULL )
+ name = opts->deps_file;
+ else if( side_effect )
+ {
+ *allocated_name = dependency_default_filename( opts );
+ if( *allocated_name == NULL )
+ return( NULL );
+ name = *allocated_name;
+ }
+ else if( opts->out_fname != NULL && opts->out_fname[0] != 0 )
+ name = opts->out_fname;
+
+ if( name == NULL || name[0] == 0 || strcmp(name, "-") == 0 )
+ return( stdout );
+
+ stream = fopen( name, "wb" );
+ if( stream != NULL )
+ *close_stream = 1;
+ else
+ fprintf( stderr, "%s: cannot open dependency output '%s': %s\n",
+ opts->progname, name, strerror(errno) );
+ return( stream );
+}
+
+static int
+run_preprocessor( mcpp *cpp, mcpp_options *opts )
+{
+ FILE *stream;
+ int close_stream;
+ int rc;
+
+ cpp->options_data = opts;
+ cpp->verbose = opts->verbose;
+ cpp->include_paths.no_standard_includes = opts->no_standard_includes;
+
+ if( mcpp_load_config(cpp, opts->config_fname, opts->no_config) != 0 )
+ return( -1 );
+
+ if( cpp->verbose && dump_verbose_config(cpp, stderr) != 0 )
+ return( -1 );
+
+ if( mcpu_include_paths_from_config(&cpp->include_paths, &cpp->config) != 0 ||
+ apply_include_options(cpp, opts) != 0 )
+ return( -1 );
+
+ if( opts->dump_config )
+ {
+ if( opts->out_fname[0] != 0 )
+ {
+ fprintf( stderr, "%s: -o/output file has no meaning with -dconfig\n",
+ opts->progname );
+ return( -1 );
+ }
+ return( mcpu_config_dump(&cpp->config, stdout) );
+ }
+
+ if( opts->dump_search_dirs )
+ {
+ if( opts->out_fname[0] != 0 )
+ {
+ fprintf( stderr, "%s: -o/output file has no meaning with -dsearch-dirs\n",
+ opts->progname );
+ return( -1 );
+ }
+ return( mcpu_include_paths_dump(&cpp->include_paths, stdout) );
+ }
+
+ if( mcpp_options_has_unimplemented(opts) )
+ {
+ fprintf( stderr,
+ "%s: requested option is parsed but its handler is not implemented yet\n",
+ opts->progname );
+ return( -1 );
+ }
+
+ if( opts->in_fname[0] != 0 )
+ rc = mcpp_process( cpp, opts->in_fname );
+ else
+ rc = mcpp_process_stream( cpp, stdin, "<stdin>" );
+ if( rc != 0 )
+ return( -1 );
+
+ if( opts->print_deps && opts->inhibit_output )
+ {
+ char *allocated_name;
+
+ stream = open_dependency_output( opts, 0, &allocated_name, &close_stream );
+ if( stream == NULL )
+ {
+ free( allocated_name );
+ return( -1 );
+ }
+
+ rc = mcpp_write_dependencies( cpp, stream, opts->print_deps == 2 );
+ if( close_stream && fclose(stream) != 0 )
+ rc = -1;
+ free( allocated_name );
+ return( rc );
+ }
+
+ if( opts->dump_macros == MCPP_DUMP_ONLY )
+ {
+ stream = open_output( opts, &close_stream );
+ if( stream == NULL )
+ return( -1 );
+
+ rc = mcpp_dump_macros( cpp, stream );
+ if( close_stream && fclose(stream) != 0 )
+ rc = -1;
+ return( rc );
+ }
+
+ rc = mcpp_write_output(cpp,
+ opts->out_fname[0] != 0 ? opts->out_fname : NULL);
+ if( rc != 0 || !opts->print_deps )
+ return( rc );
+
+ {
+ char *allocated_name;
+
+ stream = open_dependency_output( opts, 1, &allocated_name, &close_stream );
+ if( stream == NULL )
+ {
+ free( allocated_name );
+ return( -1 );
+ }
+
+ rc = mcpp_write_dependencies( cpp, stream, opts->print_deps == 2 );
+ if( close_stream && fclose(stream) != 0 )
+ rc = -1;
+ free( allocated_name );
+ }
+
+ return( rc );
+}
+
+int
+main( int argc, char **argv )
+{
+ mcpp cpp;
+ mcpp_options options;
+ const char *p;
+ int mpu_initialized = 0;
+ int rc = 1;
+
+ mcpp_options_init( &options, argc );
+ if( options.actions == NULL )
+ return( 1 );
+
+ p = argv[0] + strlen( argv[0] );
+ while( p != argv[0] && p[-1] != '/' ) --p;
+ options.progname = p;
+
+#ifdef SIGPIPE
+ signal( SIGPIPE, pipe_closed );
+#endif
+
+ if( mcpp_handle_options(&options, argc - 1, argv + 1) != 0 )
+ goto done_options;
+
+ if( options.version )
+ {
+ welcome( &options );
+ rc = 0;
+ goto done_options;
+ }
+
+ if( options.help )
+ {
+ welcome( &options );
+ usage( &options );
+ rc = 0;
+ goto done_options;
+ }
+
+ __mpu_init();
+ mpu_initialized = 1;
+ mcpp_init( &cpp );
+
+ if( mcpp_runtime_paths_discover(&cpp.runtime_paths, argv[0]) != 0 )
+ {
+ fprintf( stderr, "%s: cannot determine MCPU installation root: %s\n",
+ options.progname, strerror(errno) );
+ }
+ else if( run_preprocessor(&cpp, &options) == 0 )
+ rc = 0;
+
+ mcpp_free( &cpp );
+ if( mpu_initialized )
+ __mpu_free_context();
+
+done_options:
+ mcpp_options_free( &options );
+ return( rc );
+}
diff --git a/src/mcpp-config-file.c b/src/mcpp-config-file.c
new file mode 100644
index 0000000..8022af7
--- /dev/null
+++ b/src/mcpp-config-file.c
@@ -0,0 +1,512 @@
+#include <defs.h>
+
+static char *
+read_bytes( const char *filename, size_t *file_length )
+{
+ FILE *fp;
+ char *data;
+ long size;
+ size_t n;
+
+ if( file_length == NULL )
+ {
+ errno = EINVAL;
+ return( NULL );
+ }
+
+ *file_length = 0;
+
+ fp = fopen( filename, "rb" );
+ if( fp == NULL )
+ return( NULL );
+
+ if( fseek( fp, 0, SEEK_END ) != 0 )
+ {
+ fclose( fp );
+ return( NULL );
+ }
+
+ size = ftell( fp );
+ if( size < 0 || fseek( fp, 0, SEEK_SET ) != 0 )
+ {
+ fclose( fp );
+ return( NULL );
+ }
+
+ data = (char *)calloc( (size_t)size + 1, 1 );
+ if( data == NULL )
+ {
+ fclose( fp );
+ return( NULL );
+ }
+
+ n = fread( data, 1, (size_t)size, fp );
+ if( n != (size_t)size && ferror(fp) )
+ {
+ free( data );
+ fclose( fp );
+ return( NULL );
+ }
+
+ data[n] = 0;
+ fclose( fp );
+ *file_length = n;
+
+ return( data );
+}
+
+static int
+is_space16( __mpu_char16_t c )
+{
+ return( c == ' ' || c == '\t' || c == '\r' || c == '\n' ||
+ c == '\f' || c == '\v' );
+}
+
+static int
+is_name16( __mpu_char16_t c )
+{
+ return( (c >= 'a' && c <= 'z') ||
+ (c >= 'A' && c <= 'Z') ||
+ (c >= '0' && c <= '9') || c == '_' );
+}
+
+static void
+skip_space_and_comments( const __mpu_char16_t *s, size_t n, size_t *pos )
+{
+ size_t p = *pos;
+
+ for( ;; )
+ {
+ while( p < n && is_space16(s[p]) )
+ ++p;
+
+ if( p + 1 < n && s[p] == '/' && s[p + 1] == '*' )
+ {
+ p += 2;
+ while( p + 1 < n && !(s[p] == '*' && s[p + 1] == '/') )
+ ++p;
+ if( p + 1 < n )
+ p += 2;
+ continue;
+ }
+
+ if( p + 1 < n && s[p] == '/' && s[p + 1] == '/' )
+ {
+ p += 2;
+ while( p < n && s[p] != '\n' )
+ ++p;
+ continue;
+ }
+
+ if( p < n && s[p] == '#' )
+ {
+ while( p < n && s[p] != '\n' )
+ ++p;
+ continue;
+ }
+
+ break;
+ }
+
+ *pos = p;
+}
+
+static char *
+expand_value( const mcpu_config *config, const char *value )
+{
+ size_t length = strlen( value );
+ size_t capacity = length + 64;
+ char *out = (char *)calloc( capacity, 1 );
+ size_t i = 0;
+ size_t o = 0;
+
+ if( out == NULL )
+ return( NULL );
+
+ while( i < length )
+ {
+ if( value[i] == '$' )
+ {
+ char name[256];
+ size_t ni = 0;
+ size_t j = i + 1;
+ int braced = 0;
+ const char *replacement;
+
+ if( j < length && value[j] == '{' )
+ {
+ braced = 1;
+ ++j;
+ }
+
+ while( j < length && ni + 1 < sizeof(name) &&
+ ((value[j] >= 'a' && value[j] <= 'z') ||
+ (value[j] >= 'A' && value[j] <= 'Z') ||
+ (value[j] >= '0' && value[j] <= '9') || value[j] == '_') )
+ name[ni++] = value[j++];
+
+ if( braced )
+ {
+ if( j >= length || value[j] != '}' )
+ {
+ free( out );
+ errno = EINVAL;
+ return( NULL );
+ }
+ ++j;
+ }
+
+ if( ni == 0 )
+ {
+ out[o++] = value[i++];
+ continue;
+ }
+
+ name[ni] = 0;
+ replacement = mcpu_config_get( config, name );
+ if( replacement == NULL )
+ replacement = getenv( name );
+ if( replacement == NULL )
+ replacement = "";
+
+ {
+ size_t rl = strlen( replacement );
+ if( o + rl + 1 > capacity )
+ {
+ char *p;
+ while( o + rl + 1 > capacity )
+ capacity *= 2;
+ p = (char *)realloc( out, capacity );
+ if( p == NULL )
+ {
+ free( out );
+ return( NULL );
+ }
+ out = p;
+ }
+ memcpy( out + o, replacement, rl );
+ o += rl;
+ }
+
+ i = j;
+ continue;
+ }
+
+ if( o + 2 > capacity )
+ {
+ char *p;
+ capacity *= 2;
+ p = (char *)realloc( out, capacity );
+ if( p == NULL )
+ {
+ free( out );
+ return( NULL );
+ }
+ out = p;
+ }
+
+ out[o++] = value[i++];
+ }
+
+ out[o] = 0;
+ return( out );
+}
+
+void
+mcpu_config_init( mcpu_config *config )
+{
+ if( config == NULL )
+ return;
+
+ config->first = NULL;
+ config->verbose = 0;
+}
+
+void
+mcpu_config_free( mcpu_config *config )
+{
+ mcpu_config_entry *p;
+
+ if( config == NULL )
+ return;
+
+ p = config->first;
+ while( p )
+ {
+ mcpu_config_entry *next = p->next;
+ free( p->name );
+ free( p->value );
+ free( p );
+ p = next;
+ }
+
+ config->first = NULL;
+}
+
+const char *
+mcpu_config_get( const mcpu_config *config, const char *name )
+{
+ mcpu_config_entry *p;
+
+ if( config == NULL || name == NULL )
+ return( NULL );
+
+ for( p = config->first; p; p = p->next )
+ if( strcmp( p->name, name ) == 0 )
+ return( p->value );
+
+ return( NULL );
+}
+
+int
+mcpu_config_set( mcpu_config *config, const char *name,
+ const char *value )
+{
+ mcpu_config_entry *p;
+ char *new_value;
+
+ if( config == NULL || name == NULL || value == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ new_value = expand_value( config, value );
+ if( new_value == NULL )
+ return( -1 );
+
+ for( p = config->first; p; p = p->next )
+ {
+ if( strcmp( p->name, name ) == 0 )
+ {
+ free( p->value );
+ p->value = new_value;
+ return( 0 );
+ }
+ }
+
+ p = (mcpu_config_entry *)calloc( 1, sizeof(*p) );
+ if( p == NULL )
+ {
+ free( new_value );
+ return( -1 );
+ }
+
+ p->name = strdup( name );
+ if( p->name == NULL )
+ {
+ free( new_value );
+ free( p );
+ return( -1 );
+ }
+
+ p->value = new_value;
+ p->next = config->first;
+ config->first = p;
+
+ return( 0 );
+}
+
+
+static int
+config_entry_compare( const void *a, const void *b )
+{
+ const mcpu_config_entry *ea = *(const mcpu_config_entry * const *)a;
+ const mcpu_config_entry *eb = *(const mcpu_config_entry * const *)b;
+ return( strcmp(ea->name, eb->name) );
+}
+
+
+int
+mcpu_config_dump( const mcpu_config *config, FILE *stream )
+{
+ const mcpu_config_entry *p;
+ mcpu_config_entry **list;
+ size_t count = 0;
+ size_t i = 0;
+
+ if( config == NULL || stream == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ for( p = config->first; p; p = p->next )
+ ++count;
+
+ list = count ? (mcpu_config_entry **)calloc( count, sizeof(*list) ) : NULL;
+ if( count != 0 && list == NULL )
+ return( -1 );
+
+ for( p = config->first; p; p = p->next )
+ list[i++] = (mcpu_config_entry *)p;
+
+ if( count > 1 )
+ qsort( list, count, sizeof(*list), config_entry_compare );
+
+ for( i = 0; i < count; ++i )
+ {
+ if( fprintf(stream, "%s = %s;\n", list[i]->name, list[i]->value) < 0 )
+ {
+ free( list );
+ return( -1 );
+ }
+ }
+
+ free( list );
+ return( ferror(stream) ? -1 : 0 );
+}
+
+
+int
+mcpu_config_read( mcpu_config *config, const char *filename,
+ int missing_is_error )
+{
+ char *bytes;
+ size_t byte_length;
+ mcpu_text text;
+ size_t p = 0;
+ int rc = -1;
+
+ if( config == NULL || filename == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ bytes = read_bytes( filename, &byte_length );
+ if( bytes == NULL )
+ {
+ if( !missing_is_error && errno == ENOENT )
+ return( 0 );
+ return( -1 );
+ }
+
+ mcpu_text_init( &text );
+ {
+ const char *input = bytes;
+
+ if( memchr( bytes, 0, byte_length ) != NULL )
+ {
+ fprintf( stderr, "%s: NUL character is not allowed in configuration text\n",
+ filename );
+ free( bytes );
+ return( -1 );
+ }
+
+ if( byte_length >= 3 &&
+ (unsigned char)bytes[0] == 0xef &&
+ (unsigned char)bytes[1] == 0xbb &&
+ (unsigned char)bytes[2] == 0xbf )
+ input += 3;
+
+ if( mcpu_text_from_utf8( &text, input ) != 0 )
+ {
+ fprintf( stderr, "%s: invalid UTF-8 or non-UCS-2 character\n",
+ filename );
+ free( bytes );
+ return( -1 );
+ }
+ }
+ free( bytes );
+
+ while( p < text.length )
+ {
+ size_t name_start;
+ size_t name_end;
+ size_t value_start;
+ size_t value_end;
+ char *name = NULL;
+ char *value = NULL;
+
+ skip_space_and_comments( text.data, text.length, &p );
+ if( p >= text.length )
+ break;
+
+ name_start = p;
+ while( p < text.length && is_name16(text.data[p]) )
+ ++p;
+ name_end = p;
+
+ if( name_end == name_start )
+ {
+ fprintf( stderr, "%s: malformed configuration variable\n", filename );
+ goto done;
+ }
+
+ skip_space_and_comments( text.data, text.length, &p );
+ if( p >= text.length || text.data[p] != '=' )
+ {
+ fprintf( stderr, "%s: '=' expected after configuration variable\n", filename );
+ goto done;
+ }
+ ++p;
+
+ while( p < text.length && is_space16(text.data[p]) )
+ ++p;
+
+ value_start = p;
+ if( p < text.length && (text.data[p] == '"' || text.data[p] == '\'') )
+ {
+ __mpu_char16_t quote = text.data[p++];
+ value_start = p;
+ while( p < text.length && text.data[p] != quote )
+ {
+ if( text.data[p] == '\\' && p + 1 < text.length )
+ p += 2;
+ else
+ ++p;
+ }
+ if( p >= text.length )
+ {
+ fprintf( stderr, "%s: unterminated quoted configuration value\n", filename );
+ goto done;
+ }
+ value_end = p++;
+ while( p < text.length && is_space16(text.data[p]) )
+ ++p;
+ }
+ else
+ {
+ while( p < text.length && text.data[p] != ';' && text.data[p] != '\n' )
+ ++p;
+ value_end = p;
+ while( value_end > value_start && is_space16(text.data[value_end - 1]) )
+ --value_end;
+ }
+
+ if( p >= text.length || text.data[p] != ';' )
+ {
+ fprintf( stderr, "%s: ';' expected after configuration value\n", filename );
+ goto done;
+ }
+ ++p;
+
+ name = mcpu_text_to_utf8( text.data + name_start, name_end - name_start );
+ value = mcpu_text_to_utf8( text.data + value_start, value_end - value_start );
+ if( name == NULL || value == NULL )
+ {
+ free( name );
+ free( value );
+ goto done;
+ }
+
+ if( mcpu_config_set( config, name, value ) != 0 )
+ {
+ free( name );
+ free( value );
+ goto done;
+ }
+
+ free( name );
+ free( value );
+ name = NULL;
+ value = NULL;
+ }
+
+ rc = 0;
+
+done:
+ mcpu_text_free( &text );
+ return( rc );
+}
diff --git a/src/mcpp-config-file.h b/src/mcpp-config-file.h
new file mode 100644
index 0000000..b2923ab
--- /dev/null
+++ b/src/mcpp-config-file.h
@@ -0,0 +1,30 @@
+#ifndef __MCPU_CPP_CONFIG_FILE_H__
+#define __MCPU_CPP_CONFIG_FILE_H__ 1
+
+#include <defs.h>
+
+typedef struct mcpu_config_entry mcpu_config_entry;
+struct mcpu_config_entry
+{
+ char *name;
+ char *value;
+ mcpu_config_entry *next;
+};
+
+typedef struct mcpu_config mcpu_config;
+struct mcpu_config
+{
+ mcpu_config_entry *first;
+ int verbose;
+};
+
+void mcpu_config_init( mcpu_config *config );
+void mcpu_config_free( mcpu_config *config );
+int mcpu_config_read( mcpu_config *config, const char *filename,
+ int missing_is_error );
+const char *mcpu_config_get( const mcpu_config *config, const char *name );
+int mcpu_config_set( mcpu_config *config, const char *name,
+ const char *value );
+int mcpu_config_dump( const mcpu_config *config, FILE *stream );
+
+#endif /* __MCPU_CPP_CONFIG_FILE_H__ */
diff --git a/src/mcpp-diagnostic.c b/src/mcpp-diagnostic.c
new file mode 100644
index 0000000..76e0c25
--- /dev/null
+++ b/src/mcpp-diagnostic.c
@@ -0,0 +1,28 @@
+#include <defs.h>
+#include <stdarg.h>
+
+int
+mcpp_diagnostic_warning( const mcpp_options *options,
+ const char *filename,
+ unsigned line_number,
+ const char *format, ... )
+{
+ va_list ap;
+ int as_error;
+
+ if( options != NULL && options->inhibit_warnings )
+ return( 0 );
+
+ as_error = options != NULL && options->warnings_are_errors;
+
+ fprintf( stderr, "%s:%u: %s: ",
+ filename != NULL ? filename : "<input>", line_number,
+ as_error ? "error" : "warning" );
+
+ va_start( ap, format );
+ vfprintf( stderr, format, ap );
+ va_end( ap );
+ fputc( '\n', stderr );
+
+ return( as_error ? -1 : 0 );
+}
diff --git a/src/mcpp-diagnostic.h b/src/mcpp-diagnostic.h
new file mode 100644
index 0000000..93cccb1
--- /dev/null
+++ b/src/mcpp-diagnostic.h
@@ -0,0 +1,11 @@
+#ifndef __MCPU_CPP_DIAGNOSTIC_H__
+#define __MCPU_CPP_DIAGNOSTIC_H__ 1
+
+typedef struct mcpp_options mcpp_options;
+
+int mcpp_diagnostic_warning( const mcpp_options *options,
+ const char *filename,
+ unsigned line_number,
+ const char *format, ... );
+
+#endif /* __MCPU_CPP_DIAGNOSTIC_H__ */
diff --git a/src/mcpp-expr.h b/src/mcpp-expr.h
new file mode 100644
index 0000000..6c5a962
--- /dev/null
+++ b/src/mcpp-expr.h
@@ -0,0 +1,10 @@
+#ifndef __MCPU_CPP_EXPR_H__
+#define __MCPU_CPP_EXPR_H__ 1
+
+#include <defs.h>
+
+int mcpp_expr_parse( const __mpu_char16_t *text, size_t length,
+ const char *filename, unsigned line_number,
+ const mcpp_options *options, int *result );
+
+#endif /* __MCPU_CPP_EXPR_H__ */
diff --git a/src/mcpp-expr.zubr b/src/mcpp-expr.zubr
new file mode 100644
index 0000000..1ec5823
--- /dev/null
+++ b/src/mcpp-expr.zubr
@@ -0,0 +1,723 @@
+/***************************************************************
+ MCPP_EXPR.C
+
+ This file containt the grammar & procedure for
+ parse expressions for MCPU-CPP .
+
+ PART OF : MCPU-CPP - MCPU language preproccessor .
+
+ NOTE : NONE .
+
+ Copyright (C) 1998 - 2026 by Andrey V.Kosteltsev.
+ All Rights Reserved.
+ ***************************************************************/
+
+%{
+
+#include <defs.h>
+
+static int mcpp_zubr_lex( void );
+static void mcpp_zubr_error( char *s );
+
+static mcpp_integer expression_value;
+static mcpp_semantic_context semantic_context;
+
+/************************************************************
+ Nonzero means do not evaluate this expression.
+ This is a count, since unevaluated expressions can nest.
+ ************************************************************/
+static int skip_evaluation;
+
+/************************************************************
+ During parsing of an MCPU-CPP expression, LEXPTR points to
+ the next UCS-2 character and LEXEND points one character
+ past the input expression.
+ ************************************************************/
+static const __mpu_char16_t *lexptr;
+static const __mpu_char16_t *lexend;
+
+%}
+
+%union
+{
+ mcpp_integer integer;
+ struct name
+ {
+ const __mpu_char16_t *address;
+ size_t length;
+ } name;
+}
+
+%type <integer> exp exp1 start
+%token <integer> INT CHAR
+%token <name> NAME
+%token <integer> ERROR
+
+%right '?' ':'
+%left ','
+%left OR
+%left AND
+%left '|'
+%left '^'
+%left '&'
+%left EQUAL NOTEQUAL
+%left '<' '>' LEQ GEQ
+%left LSH RSH
+%left '+' '-'
+%left '*' '/' '%'
+%right UNARY
+
+%%
+
+start: exp1
+ {
+ expression_value = $1;
+ }
+ ;
+
+/* Expressions, including the comma operator. */
+exp1: exp
+ | exp1 ',' exp
+ {
+ $$ = $3;
+ }
+ ;
+
+/* Expressions, not including the comma operator. */
+exp: '-' exp %prec UNARY
+ {
+ $$ = mcpp_semantic_neg( $2 );
+ }
+ | '!' exp %prec UNARY
+ {
+ $$ = mcpp_semantic_not( $2 );
+ }
+ | '+' exp %prec UNARY
+ {
+ $$ = $2;
+ }
+ | '~' exp %prec UNARY
+ {
+ $$ = mcpp_semantic_compl( $2 );
+ }
+ | '(' exp1 ')'
+ {
+ $$ = $2;
+ }
+ ;
+
+/* Binary operators in order of decreasing precedence. */
+exp: exp '*' exp
+ {
+ $$ = mcpp_semantic_mul( $1, $3 );
+ }
+ | exp '/' exp
+ {
+ $$ = mcpp_semantic_div( &semantic_context, $1, $3,
+ !skip_evaluation );
+ }
+ | exp '%' exp
+ {
+ $$ = mcpp_semantic_mod( &semantic_context, $1, $3,
+ !skip_evaluation );
+ }
+ | exp '+' exp
+ {
+ $$ = mcpp_semantic_add( $1, $3 );
+ }
+ | exp '-' exp
+ {
+ $$ = mcpp_semantic_sub( $1, $3 );
+ }
+ | exp LSH exp
+ {
+ $$ = mcpp_semantic_lshift( $1, $3 );
+ }
+ | exp RSH exp
+ {
+ $$ = mcpp_semantic_rshift( $1, $3 );
+ }
+ | exp EQUAL exp
+ {
+ $$ = mcpp_semantic_equal( $1, $3 );
+ }
+ | exp NOTEQUAL exp
+ {
+ $$ = mcpp_semantic_notequal( $1, $3 );
+ }
+ | exp LEQ exp
+ {
+ $$ = mcpp_semantic_leq( $1, $3 );
+ }
+ | exp GEQ exp
+ {
+ $$ = mcpp_semantic_geq( $1, $3 );
+ }
+ | exp '<' exp
+ {
+ $$ = mcpp_semantic_less( $1, $3 );
+ }
+ | exp '>' exp
+ {
+ $$ = mcpp_semantic_greater( $1, $3 );
+ }
+ | exp '&' exp
+ {
+ $$ = mcpp_semantic_bitand( $1, $3 );
+ }
+ | exp '^' exp
+ {
+ $$ = mcpp_semantic_bitxor( $1, $3 );
+ }
+ | exp '|' exp
+ {
+ $$ = mcpp_semantic_bitor( $1, $3 );
+ }
+ | exp AND
+ {
+ skip_evaluation += !mcpp_semantic_true( $1 );
+ }
+ exp
+ {
+ skip_evaluation -= !mcpp_semantic_true( $1 );
+ $$ = mcpp_semantic_and( $1, $4 );
+ }
+ | exp OR
+ {
+ skip_evaluation += !!mcpp_semantic_true( $1 );
+ }
+ exp
+ {
+ skip_evaluation -= !!mcpp_semantic_true( $1 );
+ $$ = mcpp_semantic_or( $1, $4 );
+ }
+ | exp '?'
+ {
+ skip_evaluation += !mcpp_semantic_true( $1 );
+ }
+ exp ':'
+ {
+ skip_evaluation += !!mcpp_semantic_true( $1 ) -
+ !mcpp_semantic_true( $1 );
+ }
+ exp
+ {
+ skip_evaluation -= !!mcpp_semantic_true( $1 );
+ $$ = mcpp_semantic_conditional( $1, $4, $7 );
+ }
+ | INT
+ {
+ $$ = $1;
+ }
+ | CHAR
+ {
+ $$ = $1;
+ }
+ | NAME
+ {
+ $$ = mcpp_semantic_make( 0, 0 );
+ }
+ ;
+
+/******* END OF GRAMMAR *******/
+%%
+
+struct token
+{
+ const char *operator;
+ int token;
+};
+
+static struct token tokentab2[] =
+{
+ { "&&", AND },
+ { "||", OR },
+ { "<<", LSH },
+ { ">>", RSH },
+ { "==", EQUAL },
+ { "!=", NOTEQUAL },
+ { "<=", LEQ },
+ { ">=", GEQ },
+ { "++", ERROR },
+ { "--", ERROR },
+ { NULL, ERROR }
+};
+
+static int
+mcpp_expr_space( __mpu_char16_t c )
+{
+ return( c == ' ' || c == '\t' || c == '\r' || c == '\n' ||
+ c == '\f' || c == '\v' );
+}
+
+static int
+mcpp_expr_hex_digit( __mpu_char16_t c )
+{
+ if( c >= '0' && c <= '9' ) return( c - '0' );
+ if( c >= 'a' && c <= 'f' ) return( c - 'a' + 10 );
+ if( c >= 'A' && c <= 'F' ) return( c - 'A' + 10 );
+ return( -1 );
+}
+
+static int
+mcpp_expr_error_token( const char *message )
+{
+ mcpp_zubr_error( (char *)message );
+ return( ERROR );
+}
+
+static __mpu_uint16_t
+mcpp_expr_escape( int *failed )
+{
+ __mpu_char16_t c;
+ __mpu_uint32_t value;
+ int digit;
+ unsigned count;
+
+ *failed = 0;
+ if( lexptr >= lexend )
+ {
+ *failed = 1;
+ mcpp_zubr_error( "incomplete escape sequence in #if expression" );
+ return( 0 );
+ }
+
+ c = *lexptr++;
+ switch( c )
+ {
+ case 'a': return( (__mpu_uint16_t)'\a' );
+ case 'b': return( (__mpu_uint16_t)'\b' );
+ case 'e':
+ case 'E': return( (__mpu_uint16_t)033 );
+ case 'f': return( (__mpu_uint16_t)'\f' );
+ case 'n': return( (__mpu_uint16_t)'\n' );
+ case 'r': return( (__mpu_uint16_t)'\r' );
+ case 't': return( (__mpu_uint16_t)'\t' );
+ case 'v': return( (__mpu_uint16_t)'\v' );
+ case '\\': return( (__mpu_uint16_t)'\\' );
+ case '\'': return( (__mpu_uint16_t)'\'' );
+ case '"': return( (__mpu_uint16_t)'"' );
+ case '?': return( (__mpu_uint16_t)'?' );
+
+ case 'x':
+ case 'X':
+ value = 0;
+ count = 0;
+ while( lexptr < lexend )
+ {
+ digit = mcpp_expr_hex_digit( *lexptr );
+ if( digit < 0 ) break;
+ if( value > 0xffffU >> 4 )
+ {
+ *failed = 1;
+ mcpp_zubr_error( "hex escape sequence out of UCS-2 range" );
+ return( 0 );
+ }
+ value = (value << 4) + (unsigned)digit;
+ ++lexptr;
+ ++count;
+ }
+ if( count == 0 )
+ {
+ *failed = 1;
+ mcpp_zubr_error( "\\x used with no following hex digits" );
+ return( 0 );
+ }
+ return( (__mpu_uint16_t)value );
+
+ default:
+ if( c >= '0' && c <= '7' )
+ {
+ value = c - '0';
+ for( count = 1; count < 6 && lexptr < lexend; ++count )
+ {
+ c = *lexptr;
+ if( c < '0' || c > '7' ) break;
+ if( value > 0xffffU >> 3 )
+ {
+ *failed = 1;
+ mcpp_zubr_error( "octal escape sequence out of UCS-2 range" );
+ return( 0 );
+ }
+ value = (value << 3) + (c - '0');
+ ++lexptr;
+ }
+ if( value > 0xffffU )
+ {
+ *failed = 1;
+ mcpp_zubr_error( "octal escape sequence out of UCS-2 range" );
+ return( 0 );
+ }
+ return( (__mpu_uint16_t)value );
+ }
+ return( (__mpu_uint16_t)c );
+ }
+}
+
+static int
+mcpp_expr_character_constant( void )
+{
+ __mpu_uint64_t result = 0;
+ __mpu_uint16_t value;
+ __mpu_char16_t c;
+ unsigned count = 0;
+ int failed;
+
+ ++lexptr;
+
+ while( lexptr < lexend )
+ {
+ c = *lexptr++;
+ if( c == '\'' )
+ break;
+
+ if( c == '\n' )
+ return( mcpp_expr_error_token(
+ "unterminated character constant in #if expression") );
+
+ if( c == '\\' )
+ {
+ value = mcpp_expr_escape( &failed );
+ if( failed ) return( ERROR );
+ }
+ else
+ value = (__mpu_uint16_t)c;
+
+ ++count;
+ if( count > 4 )
+ return( mcpp_expr_error_token(
+ "character constant too long in #if expression") );
+
+ result = (result << 16) | (__mpu_uint64_t)value;
+ }
+
+ if( lexptr == lexend && (lexptr == 0 || lexptr[-1] != '\'') )
+ return( mcpp_expr_error_token(
+ "unterminated character constant in #if expression") );
+
+ if( count == 0 )
+ return( mcpp_expr_error_token(
+ "empty character constant in #if expression") );
+
+ mcpp_zubr_lval.integer = mcpp_semantic_make( result, 1 );
+ return( CHAR );
+}
+
+static int
+mcpp_expr_valid_integer_width( unsigned width )
+{
+ return( width >= 8 && width <= MPU_REAL_IO_LIMIT &&
+ (width & (width - 1)) == 0 );
+}
+
+static __mpu_uint64_t
+mcpp_expr_apply_integer_width( __mpu_uint64_t value, unsigned width,
+ int unsignedp )
+{
+ __mpu_uint64_t mask;
+ __mpu_uint64_t sign;
+
+ if( width >= 64 )
+ return( value );
+
+ mask = (((__mpu_uint64_t)1 << width) - 1);
+ value &= mask;
+
+ if( unsignedp )
+ return( value );
+
+ sign = ((__mpu_uint64_t)1 << (width - 1));
+ if( value & sign )
+ value |= ~mask;
+
+ return( value );
+}
+
+static int
+mcpp_expr_number( void )
+{
+ const __mpu_char16_t *start = lexptr;
+ const __mpu_char16_t *end;
+ const __mpu_char16_t *number_end;
+ const __mpu_char16_t *p;
+ __mpu_char8_t *ascii;
+ __mpu_uint64_t value = 0;
+ size_t number_length;
+ unsigned base = 10;
+ unsigned width = 0;
+ int digit;
+ int have_digit = 0;
+ int unsignedp = 0;
+ int have_width = 0;
+ int width_over_64 = 0;
+
+ while( lexptr < lexend &&
+ (mcpu_pp_is_identifier_char(*lexptr) || *lexptr == '.') )
+ ++lexptr;
+ end = lexptr;
+
+ for( p = start; p < end; ++p )
+ if( *p == '.' )
+ return( mcpp_expr_error_token(
+ "floating point numbers not allowed in #if expressions") );
+
+ p = start;
+ if( end - p >= 2 && p[0] == '0' &&
+ (p[1] == 'x' || p[1] == 'X') )
+ {
+ base = 16;
+ p += 2;
+ }
+ else if( end - p >= 2 && p[0] == '0' &&
+ (p[1] == 'b' || p[1] == 'B') )
+ {
+ base = 2;
+ p += 2;
+ }
+ else if( end - p > 1 && p[0] == '0' )
+ base = 8;
+
+ for( ; p < end; ++p )
+ {
+ if( *p > 0x7f )
+ break;
+ digit = mcpp_expr_hex_digit( *p );
+ if( digit < 0 || (unsigned)digit >= base )
+ break;
+ have_digit = 1;
+ }
+ number_end = p;
+
+ if( !have_digit )
+ return( mcpp_expr_error_token(
+ "invalid integer constant in #if expression") );
+
+ if( p < end && (*p == 'z' || *p == 'Z') )
+ {
+ have_width = 1;
+ ++p;
+ if( p == end || *p < '0' || *p > '9' )
+ return( mcpp_expr_error_token(
+ "integer width suffix requires decimal digits after z/Z") );
+
+ while( p < end && *p >= '0' && *p <= '9' )
+ {
+ unsigned d = (unsigned)(*p - '0');
+
+ if( !width_over_64 )
+ {
+ if( width > (64U - d) / 10U )
+ width_over_64 = 1;
+ else
+ {
+ width = width * 10U + d;
+ if( width > 64U )
+ width_over_64 = 1;
+ }
+ }
+ ++p;
+ }
+
+ if( p < end && (*p == 'u' || *p == 'U') )
+ {
+ unsignedp = 1;
+ ++p;
+ }
+
+ if( p != end )
+ return( mcpp_expr_error_token(
+ "invalid characters after integer width suffix") );
+
+ if( width_over_64 )
+ return( mcpp_expr_error_token(
+ "integer constants wider than 64 bits are not allowed in conditional directives") );
+
+ if( !mcpp_expr_valid_integer_width(width) )
+ {
+ mcpp_semantic_warning(
+ &semantic_context,
+ "invalid zNNN integer-width suffix; suffix ignored" );
+ have_width = 0;
+ }
+ }
+ else if( p < end && (*p == 'u' || *p == 'U') )
+ {
+ unsignedp = 1;
+ ++p;
+ if( p != end )
+ return( mcpp_expr_error_token(
+ "invalid characters after integer suffix") );
+ }
+ else if( p != end )
+ return( mcpp_expr_error_token(
+ "invalid integer suffix in #if expression") );
+
+ number_length = (size_t)(number_end - start);
+ ascii = (__mpu_char8_t *)malloc( number_length + 1 );
+ if( ascii == NULL )
+ {
+ mcpp_zubr_error( "out of memory while parsing #if expression" );
+ return( ERROR );
+ }
+
+ for( p = start; p < number_end; ++p )
+ {
+ if( *p > 0x7f )
+ {
+ free( ascii );
+ return( mcpp_expr_error_token(
+ "non-ASCII character in integer constant") );
+ }
+ ascii[p - start] = (__mpu_char8_t)*p;
+ }
+ ascii[number_length] = 0;
+
+ __mpu_clo();
+ iatoui( (mpu_int *)&value, ascii, (int)sizeof(value) );
+ free( ascii );
+
+ if( __mpu_gto() )
+ return( mcpp_expr_error_token(
+ "integer constant does not fit in 64 bits") );
+
+ if( have_width )
+ {
+ value = mcpp_expr_apply_integer_width( value, width, unsignedp );
+ mcpp_zubr_lval.integer = mcpp_semantic_make( value, unsignedp );
+ }
+ else
+ mcpp_zubr_lval.integer =
+ mcpp_semantic_make( value,
+ unsignedp || value > (__mpu_uint64_t)INT64_MAX );
+
+ return( INT );
+}
+
+static void
+mcpp_zubr_error( char *s )
+{
+ mcpp_semantic_error( &semantic_context, s );
+ skip_evaluation = 0;
+}
+
+/****************************************************
+ Read one token, getting UCS-2 characters through
+ LEXPTR.
+ ****************************************************/
+static int
+mcpp_zubr_lex( void )
+{
+ const __mpu_char16_t *tokstart;
+ struct token *toktab;
+ __mpu_char16_t c;
+
+retry:
+ while( lexptr < lexend && mcpp_expr_space(*lexptr) )
+ ++lexptr;
+
+ if( lexptr >= lexend )
+ return( 0 );
+
+ tokstart = lexptr;
+ c = *tokstart;
+
+ for( toktab = tokentab2; toktab->operator != NULL; ++toktab )
+ {
+ if( lexend - tokstart >= 2 &&
+ c == (__mpu_char16_t)(unsigned char)toktab->operator[0] &&
+ tokstart[1] == (__mpu_char16_t)(unsigned char)toktab->operator[1] )
+ {
+ lexptr += 2;
+ if( toktab->token == ERROR )
+ return( mcpp_expr_error_token(
+ "increment/decrement operator not allowed in #if expression") );
+ return( toktab->token );
+ }
+ }
+
+ if( c >= '0' && c <= '9' )
+ return( mcpp_expr_number() );
+
+ if( c == '\'' )
+ return( mcpp_expr_character_constant() );
+
+ if( mcpu_pp_is_identifier_start(c) )
+ {
+ ++lexptr;
+ while( lexptr < lexend && mcpu_pp_is_identifier_char(*lexptr) )
+ ++lexptr;
+ mcpp_zubr_lval.name.address = tokstart;
+ mcpp_zubr_lval.name.length = (size_t)(lexptr - tokstart);
+ return( NAME );
+ }
+
+ ++lexptr;
+ switch( c )
+ {
+ case '(':
+ case ')':
+ case '?':
+ case ':':
+ case ',':
+ case '*':
+ case '/':
+ case '%':
+ case '~':
+ case '^':
+ return( (int)c );
+
+ case '+':
+ case '-':
+ return( (int)c );
+
+ case '!':
+ case '<':
+ case '>':
+ case '&':
+ case '|':
+ return( (int)c );
+
+ case '"':
+ case '`':
+ return( mcpp_expr_error_token(
+ "string constants not allowed in #if expressions") );
+
+ default:
+ mcpp_zubr_error( "invalid token in #if expression" );
+ goto retry;
+ }
+}
+
+int
+mcpp_expr_parse( const __mpu_char16_t *text, size_t length,
+ const char *filename, unsigned line_number,
+ const mcpp_options *options, int *result )
+{
+ int rc;
+
+ if( text == NULL || filename == NULL || result == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ mcpp_semantic_context_init( &semantic_context, filename, line_number,
+ options );
+ expression_value = mcpp_semantic_make( 0, 0 );
+ skip_evaluation = 0;
+ lexptr = text;
+ lexend = text + length;
+
+ if( length == 0 )
+ {
+ mcpp_zubr_error( "empty #if expression" );
+ return( -1 );
+ }
+
+ rc = mcpp_zubr_parse();
+ if( rc != 0 || mcpp_semantic_failed(&semantic_context) )
+ return( -1 );
+
+ *result = mcpp_semantic_true( expression_value );
+ return( 0 );
+}
diff --git a/src/mcpp-expression.c b/src/mcpp-expression.c
new file mode 100644
index 0000000..1f9532d
--- /dev/null
+++ b/src/mcpp-expression.c
@@ -0,0 +1,178 @@
+#include <defs.h>
+#include <mcpp-expr.h>
+
+/*
+ * The front end keeps the already established mcpu-cpp preprocessing order:
+ * handle the special defined operator first, expand macros second, then pass
+ * the resulting UCS-2 expression to the ZUBR parser generated from
+ * mcpp-expr.zubr.
+ */
+
+static int
+expr_identifier_start( __mpu_char16_t c )
+{
+ return( mcpu_pp_is_identifier_start(c) );
+}
+
+static int
+expr_identifier_char( __mpu_char16_t c )
+{
+ return( mcpu_pp_is_identifier_char(c) );
+}
+
+static int
+expr_space( __mpu_char16_t c )
+{
+ return( c == ' ' || c == '\t' || c == '\r' || c == '\n' ||
+ c == '\f' || c == '\v' );
+}
+
+static int
+append_defined_value( mcpu_macro_table *macros,
+ const __mpu_char16_t *text, size_t length,
+ size_t *pos, mcpu_text *output,
+ const char *filename, unsigned line_number )
+{
+ size_t p = *pos;
+ size_t start;
+ size_t end;
+ int parenthesized = 0;
+
+ while( p < length && expr_space(text[p]) ) ++p;
+ if( p < length && text[p] == '(' )
+ {
+ parenthesized = 1;
+ ++p;
+ while( p < length && expr_space(text[p]) ) ++p;
+ }
+
+ if( p >= length || !expr_identifier_start(text[p]) )
+ {
+ fprintf( stderr, "%s:%u: error: identifier expected after defined\n",
+ filename, line_number );
+ return( -1 );
+ }
+
+ start = p++;
+ while( p < length && expr_identifier_char(text[p]) ) ++p;
+ end = p;
+
+ while( p < length && expr_space(text[p]) ) ++p;
+ if( parenthesized )
+ {
+ if( p >= length || text[p] != ')' )
+ {
+ fprintf( stderr, "%s:%u: error: missing ')' after defined\n",
+ filename, line_number );
+ return( -1 );
+ }
+ ++p;
+ }
+
+ if( mcpu_text_append_ascii(
+ output, mcpu_macro_find(macros, text + start, end - start) ? "1" : "0") != 0 )
+ return( -1 );
+
+ *pos = p;
+ return( 0 );
+}
+
+static int
+replace_defined_operators( mcpu_macro_table *macros,
+ const __mpu_char16_t *text, size_t length,
+ mcpu_text *output,
+ const char *filename, unsigned line_number )
+{
+ size_t p = 0;
+ __mpu_char16_t quote = 0;
+ int escaped = 0;
+
+ mcpu_text_free( output );
+ mcpu_text_init( output );
+
+ while( p < length )
+ {
+ __mpu_char16_t c = text[p];
+
+ if( quote )
+ {
+ if( mcpu_text_append_char(output, c) != 0 ) return( -1 );
+ if( escaped )
+ escaped = 0;
+ else if( c == '\\' )
+ escaped = 1;
+ else if( c == quote )
+ quote = 0;
+ ++p;
+ continue;
+ }
+
+ if( c == '\'' || c == '"' )
+ {
+ quote = c;
+ if( mcpu_text_append_char(output, c) != 0 ) return( -1 );
+ ++p;
+ continue;
+ }
+
+ if( expr_identifier_start(c) )
+ {
+ size_t start = p++;
+ while( p < length && expr_identifier_char(text[p]) ) ++p;
+
+ if( p - start == 7 &&
+ mcpu_text_equal_ascii(text + start, p - start, "defined") )
+ {
+ if( append_defined_value(macros, text, length, &p, output,
+ filename, line_number) != 0 )
+ return( -1 );
+ }
+ else if( mcpu_text_append(output, text + start, p - start) != 0 )
+ return( -1 );
+ continue;
+ }
+
+ if( mcpu_text_append_char(output, c) != 0 )
+ return( -1 );
+ ++p;
+ }
+
+ return( 0 );
+}
+
+int
+mcpp_eval_if_expression( mcpu_macro_table *macros,
+ const __mpu_char16_t *text, size_t length,
+ enum mcpu_language language,
+ const mcpu_macro_expansion_context *context,
+ const char *filename, unsigned line_number,
+ int *result )
+{
+ mcpu_text defined;
+ mcpu_text expanded;
+ int rc = -1;
+
+ if( macros == NULL || text == NULL || context == NULL ||
+ filename == NULL || result == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ mcpu_text_init( &defined );
+ mcpu_text_init( &expanded );
+
+ if( replace_defined_operators(macros, text, length, &defined,
+ filename, line_number) != 0 ||
+ mcpu_macro_expand(macros, defined.data, defined.length,
+ language, context, &expanded) != 0 )
+ goto done;
+
+ rc = mcpp_expr_parse( expanded.data, expanded.length,
+ filename, line_number, context->options, result );
+
+done:
+ mcpu_text_free( &defined );
+ mcpu_text_free( &expanded );
+ return( rc );
+}
diff --git a/src/mcpp-expression.h b/src/mcpp-expression.h
new file mode 100644
index 0000000..3fc1c96
--- /dev/null
+++ b/src/mcpp-expression.h
@@ -0,0 +1,15 @@
+#ifndef __MCPU_CPP_EXPRESSION_H__
+#define __MCPU_CPP_EXPRESSION_H__ 1
+
+#include <defs.h>
+
+int mcpp_eval_if_expression( mcpu_macro_table *macros,
+ const __mpu_char16_t *text,
+ size_t length,
+ enum mcpu_language language,
+ const mcpu_macro_expansion_context *context,
+ const char *filename,
+ unsigned line_number,
+ int *result );
+
+#endif /* __MCPU_CPP_EXPRESSION_H__ */
diff --git a/src/mcpp-include-path.c b/src/mcpp-include-path.c
new file mode 100644
index 0000000..1b26137
--- /dev/null
+++ b/src/mcpp-include-path.c
@@ -0,0 +1,467 @@
+#include <defs.h>
+
+static char *
+join_path( const char *dir, const char *name )
+{
+ size_t dl;
+ size_t nl;
+ char *p;
+ int slash;
+
+ if( dir == NULL || name == NULL )
+ {
+ errno = EINVAL;
+ return( NULL );
+ }
+
+ dl = strlen( dir );
+ nl = strlen( name );
+ slash = dl != 0 && dir[dl - 1] != '/';
+
+ p = (char *)malloc( dl + (size_t)slash + nl + 1 );
+ if( p == NULL )
+ return( NULL );
+
+ memcpy( p, dir, dl );
+ if( slash )
+ p[dl++] = '/';
+ memcpy( p + dl, name, nl + 1 );
+
+ return( p );
+}
+
+static int
+file_exists( const char *path )
+{
+ FILE *fp = fopen( path, "rb" );
+
+ if( fp == NULL )
+ return( 0 );
+
+ fclose( fp );
+ return( 1 );
+}
+
+static char *
+source_directory( const char *filename )
+{
+ const char *slash;
+ size_t n;
+ char *dir;
+
+ if( filename == NULL )
+ return( NULL );
+
+ slash = strrchr( filename, '/' );
+ if( slash == NULL )
+ return( strdup(".") );
+
+ if( slash == filename )
+ return( strdup("/") );
+
+ n = (size_t)(slash - filename);
+ dir = (char *)malloc( n + 1 );
+ if( dir == NULL )
+ return( NULL );
+
+ memcpy( dir, filename, n );
+ dir[n] = 0;
+
+ return( dir );
+}
+
+void
+mcpu_include_paths_init( mcpu_include_paths *paths )
+{
+ if( paths == NULL )
+ return;
+
+ paths->first = NULL;
+ paths->last = NULL;
+ paths->no_standard_includes = 0;
+}
+
+void
+mcpu_include_paths_free( mcpu_include_paths *paths )
+{
+ mcpu_include_dir *p;
+
+ if( paths == NULL )
+ return;
+
+ p = paths->first;
+ while( p )
+ {
+ mcpu_include_dir *next = p->next;
+ free( p->path );
+ free( p );
+ p = next;
+ }
+
+ paths->first = NULL;
+ paths->last = NULL;
+}
+
+static int
+add_dir( mcpu_include_paths *paths, const char *path,
+ enum mcpu_include_class class_type,
+ int language_specific, enum mcpu_language language )
+{
+ mcpu_include_dir *dir;
+
+ if( paths == NULL || path == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ if( *path == 0 )
+ return( 0 );
+
+ dir = (mcpu_include_dir *)calloc( 1, sizeof(*dir) );
+ if( dir == NULL )
+ return( -1 );
+
+ dir->path = strdup( path );
+ if( dir->path == NULL )
+ {
+ free( dir );
+ return( -1 );
+ }
+
+ dir->class_type = class_type;
+ dir->language = language;
+ dir->language_specific = language_specific;
+
+ if( paths->last )
+ paths->last->next = dir;
+ else
+ paths->first = dir;
+ paths->last = dir;
+
+ return( 0 );
+}
+
+int
+mcpu_include_paths_add( mcpu_include_paths *paths, const char *path,
+ enum mcpu_include_class class_type )
+{
+ return( add_dir( paths, path, class_type, 0, MCPU_LANG_0 ) );
+}
+
+int
+mcpu_include_paths_add_language( mcpu_include_paths *paths,
+ const char *path,
+ enum mcpu_include_class class_type,
+ enum mcpu_language language )
+{
+ return( add_dir( paths, path, class_type, 1, language ) );
+}
+
+int
+mcpu_include_paths_add_list( mcpu_include_paths *paths,
+ const char *list,
+ enum mcpu_include_class class_type )
+{
+ const char *p;
+ const char *start;
+
+ if( paths == NULL || list == NULL )
+ return( 0 );
+
+ p = start = list;
+ for( ;; )
+ {
+ if( *p == ':' || *p == 0 )
+ {
+ size_t n = (size_t)(p - start);
+ if( n != 0 )
+ {
+ char *dir = (char *)malloc( n + 1 );
+ int rc;
+
+ if( dir == NULL )
+ return( -1 );
+ memcpy( dir, start, n );
+ dir[n] = 0;
+ rc = mcpu_include_paths_add( paths, dir, class_type );
+ free( dir );
+ if( rc != 0 )
+ return( -1 );
+ }
+
+ if( *p == 0 )
+ break;
+ start = p + 1;
+ }
+ ++p;
+ }
+
+ return( 0 );
+}
+
+static int
+add_language_list( mcpu_include_paths *paths, const char *list,
+ enum mcpu_include_class class_type,
+ enum mcpu_language language )
+{
+ const char *p;
+ const char *start;
+
+ if( list == NULL )
+ return( 0 );
+
+ p = start = list;
+ for( ;; )
+ {
+ if( *p == ':' || *p == 0 )
+ {
+ size_t n = (size_t)(p - start);
+ if( n != 0 )
+ {
+ char *dir = (char *)malloc( n + 1 );
+ int rc;
+
+ if( dir == NULL )
+ return( -1 );
+ memcpy( dir, start, n );
+ dir[n] = 0;
+ rc = mcpu_include_paths_add_language( paths, dir,
+ class_type, language );
+ free( dir );
+ if( rc != 0 )
+ return( -1 );
+ }
+ if( *p == 0 )
+ break;
+ start = p + 1;
+ }
+ ++p;
+ }
+
+ return( 0 );
+}
+
+static int
+add_system_root( mcpu_include_paths *paths, const char *root )
+{
+ enum mcpu_language language;
+
+ if( root == NULL || *root == 0 )
+ return( 0 );
+
+ for( language = MCPU_LANG_DIFF;
+ language < MCPU_LANG__COUNT;
+ language = (enum mcpu_language)(language + 1) )
+ {
+ const char *dirname = mcpu_language_directory_name( language );
+ char *dir;
+
+ if( dirname == NULL )
+ continue;
+
+ dir = join_path( root, dirname );
+ if( dir == NULL )
+ return( -1 );
+
+ if( mcpu_include_paths_add_language(paths, dir,
+ MCPU_INCLUDE_SYSTEM_CONFIG_LANGUAGE,
+ language) != 0 )
+ {
+ free( dir );
+ return( -1 );
+ }
+ free( dir );
+ }
+
+ return( mcpu_include_paths_add(paths, root,
+ MCPU_INCLUDE_SYSTEM_CONFIG) );
+}
+
+
+
+int
+mcpu_include_paths_dump( const mcpu_include_paths *paths, FILE *stream )
+{
+ enum mcpu_include_class class_type;
+ const mcpu_include_dir *dir;
+
+ if( paths == NULL || stream == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ for( class_type = MCPU_INCLUDE_USER_EXPLICIT;
+ class_type <= MCPU_INCLUDE_AFTER_CONFIG;
+ class_type = (enum mcpu_include_class)(class_type + 1) )
+ {
+ if( paths->no_standard_includes &&
+ (class_type == MCPU_INCLUDE_SYSTEM_CONFIG_LANGUAGE ||
+ class_type == MCPU_INCLUDE_SYSTEM_CONFIG) )
+ continue;
+
+ for( dir = paths->first; dir; dir = dir->next )
+ {
+ if( dir->class_type != class_type )
+ continue;
+
+ if( fprintf(stream, "search: %s\n", dir->path) < 0 )
+ return( -1 );
+ }
+ }
+
+ return( ferror(stream) ? -1 : 0 );
+}
+
+
+int
+mcpu_include_paths_from_config( mcpu_include_paths *paths,
+ const mcpu_config *config )
+{
+ enum mcpu_language language;
+ const char *value;
+
+ if( paths == NULL || config == NULL )
+ return( 0 );
+
+ /*
+ * Language-specific directories precede the common include root. The
+ * common MCPU_CPP_INCLUDE_PATH itself is language-independent and remains
+ * visible in every language state.
+ */
+ for( language = MCPU_LANG_DIFF;
+ language < MCPU_LANG__COUNT;
+ language = (enum mcpu_language)(language + 1) )
+ {
+ const char *name = mcpu_language_config_variable( language );
+ value = mcpu_config_get( config, name );
+ if( add_language_list(paths, value,
+ MCPU_INCLUDE_USER_CONFIG_LANGUAGE,
+ language) != 0 )
+ return( -1 );
+ }
+
+ value = mcpu_config_get( config, "MCPU_CPP_INCLUDE_PATH" );
+ if( mcpu_include_paths_add_list( paths, value,
+ MCPU_INCLUDE_USER_CONFIG ) != 0 )
+ return( -1 );
+
+ value = mcpu_config_get( config, "MCPU_CPP_SYSTEM_INCLUDE_PATH" );
+ if( add_system_root(paths, value) != 0 )
+ return( -1 );
+
+ value = mcpu_config_get( config, "MCPU_CPP_AFTER_INCLUDE_PATH" );
+ if( mcpu_include_paths_add_list( paths, value,
+ MCPU_INCLUDE_AFTER_CONFIG ) != 0 )
+ return( -1 );
+
+ return( 0 );
+}
+
+char *
+mcpu_include_find( const mcpu_include_paths *paths,
+ const char *source_filename,
+ const char *include_filename,
+ int quoted,
+ enum mcpu_language language,
+ int include_next,
+ const mcpu_include_dir *after_dir,
+ const mcpu_include_dir **found_dir )
+{
+ mcpu_include_dir *dir;
+ int past_after = after_dir == NULL;
+
+ if( found_dir != NULL )
+ *found_dir = NULL;
+
+ if( include_filename == NULL )
+ return( NULL );
+
+ if( include_filename[0] == '/' )
+ {
+ if( file_exists(include_filename) )
+ return( strdup(include_filename) );
+ return( NULL );
+ }
+
+ /*
+ * #include_next never searches the directory of the current source file.
+ * It continues in the configured include chain after the directory entry
+ * that supplied the containing header. If the containing file came from
+ * its source directory (or has no search-chain origin), the configured
+ * chain is searched from its beginning.
+ */
+ if( !include_next && quoted && source_filename )
+ {
+ char *base = source_directory( source_filename );
+ char *path;
+
+ if( base == NULL )
+ return( NULL );
+ path = join_path( base, include_filename );
+ free( base );
+ if( path && file_exists(path) )
+ return( path );
+ free( path );
+ }
+
+ if( paths == NULL )
+ return( NULL );
+
+ {
+ static const enum mcpu_include_class order[] =
+ {
+ MCPU_INCLUDE_USER_EXPLICIT,
+ MCPU_INCLUDE_SYSTEM_EXPLICIT,
+ MCPU_INCLUDE_USER_CONFIG_LANGUAGE,
+ MCPU_INCLUDE_USER_CONFIG,
+ MCPU_INCLUDE_SYSTEM_CONFIG_LANGUAGE,
+ MCPU_INCLUDE_SYSTEM_CONFIG,
+ MCPU_INCLUDE_AFTER_EXPLICIT,
+ MCPU_INCLUDE_AFTER_CONFIG
+ };
+ size_t pass;
+
+ for( pass = 0; pass < sizeof(order) / sizeof(order[0]); ++pass )
+ {
+ for( dir = paths->first; dir; dir = dir->next )
+ {
+ char *path;
+
+ if( dir->class_type != order[pass] )
+ continue;
+
+ if( !past_after )
+ {
+ if( dir == after_dir )
+ past_after = 1;
+ continue;
+ }
+
+ if( dir->language_specific && dir->language != language )
+ continue;
+
+ if( paths->no_standard_includes &&
+ (dir->class_type == MCPU_INCLUDE_SYSTEM_CONFIG_LANGUAGE ||
+ dir->class_type == MCPU_INCLUDE_SYSTEM_CONFIG) )
+ continue;
+
+ path = join_path( dir->path, include_filename );
+ if( path == NULL )
+ return( NULL );
+
+ if( file_exists(path) )
+ {
+ if( found_dir != NULL )
+ *found_dir = dir;
+ return( path );
+ }
+
+ free( path );
+ }
+ }
+ }
+
+ return( NULL );
+}
diff --git a/src/mcpp-include-path.h b/src/mcpp-include-path.h
new file mode 100644
index 0000000..98001f8
--- /dev/null
+++ b/src/mcpp-include-path.h
@@ -0,0 +1,61 @@
+#ifndef __MCPU_CPP_INCLUDE_PATH_H__
+#define __MCPU_CPP_INCLUDE_PATH_H__ 1
+
+#include <defs.h>
+#include <mcpp-config-file.h>
+#include <mcpp-language.h>
+
+enum mcpu_include_class
+{
+ MCPU_INCLUDE_USER_EXPLICIT = 0,
+ MCPU_INCLUDE_SYSTEM_EXPLICIT,
+ MCPU_INCLUDE_USER_CONFIG_LANGUAGE,
+ MCPU_INCLUDE_USER_CONFIG,
+ MCPU_INCLUDE_SYSTEM_CONFIG_LANGUAGE,
+ MCPU_INCLUDE_SYSTEM_CONFIG,
+ MCPU_INCLUDE_AFTER_EXPLICIT,
+ MCPU_INCLUDE_AFTER_CONFIG
+};
+
+typedef struct mcpu_include_dir mcpu_include_dir;
+struct mcpu_include_dir
+{
+ char *path;
+ enum mcpu_include_class class_type;
+ enum mcpu_language language;
+ int language_specific;
+ mcpu_include_dir *next;
+};
+
+typedef struct mcpu_include_paths mcpu_include_paths;
+struct mcpu_include_paths
+{
+ mcpu_include_dir *first;
+ mcpu_include_dir *last;
+ int no_standard_includes;
+};
+
+void mcpu_include_paths_init( mcpu_include_paths *paths );
+void mcpu_include_paths_free( mcpu_include_paths *paths );
+int mcpu_include_paths_add( mcpu_include_paths *paths, const char *path,
+ enum mcpu_include_class class_type );
+int mcpu_include_paths_add_language( mcpu_include_paths *paths,
+ const char *path,
+ enum mcpu_include_class class_type,
+ enum mcpu_language language );
+int mcpu_include_paths_add_list( mcpu_include_paths *paths,
+ const char *list,
+ enum mcpu_include_class class_type );
+int mcpu_include_paths_from_config( mcpu_include_paths *paths,
+ const mcpu_config *config );
+int mcpu_include_paths_dump( const mcpu_include_paths *paths, FILE *stream );
+char *mcpu_include_find( const mcpu_include_paths *paths,
+ const char *source_filename,
+ const char *include_filename,
+ int quoted,
+ enum mcpu_language language,
+ int include_next,
+ const mcpu_include_dir *after_dir,
+ const mcpu_include_dir **found_dir );
+
+#endif /* __MCPU_CPP_INCLUDE_PATH_H__ */
diff --git a/src/mcpp-language.c b/src/mcpp-language.c
new file mode 100644
index 0000000..d224bc8
--- /dev/null
+++ b/src/mcpp-language.c
@@ -0,0 +1,108 @@
+#include <defs.h>
+
+static int
+text_equal_ascii_nocase( const __mpu_char16_t *text, size_t length,
+ const char *ascii )
+{
+ size_t i;
+ size_t ascii_length;
+
+ if( text == NULL || ascii == NULL )
+ return( 0 );
+
+ ascii_length = strlen( ascii );
+ if( length != ascii_length )
+ return( 0 );
+
+ for( i = 0; i < length; ++i )
+ {
+ __mpu_char16_t c = text[i];
+ unsigned char a = (unsigned char)ascii[i];
+
+ if( c >= 'A' && c <= 'Z' )
+ c = (__mpu_char16_t)(c - 'A' + 'a');
+ if( a >= 'A' && a <= 'Z' )
+ a = (unsigned char)(a - 'A' + 'a');
+
+ if( c != (__mpu_char16_t)a )
+ return( 0 );
+ }
+
+ return( 1 );
+}
+
+const char *
+mcpu_language_name( enum mcpu_language language )
+{
+ switch( language )
+ {
+ case MCPU_LANG_0: return( "0" );
+ case MCPU_LANG_DIFF: return( "diff" );
+ case MCPU_LANG_DIFT: return( "dift" );
+ case MCPU_LANG_ALG: return( "alg" );
+ case MCPU_LANG_AS: return( "as" );
+ case MCPU_LANG_AVM: return( "avm" );
+ case MCPU_LANG_ACS: return( "ACS" );
+ default: return( "unknown" );
+ }
+}
+
+int
+mcpu_language_from_text( const __mpu_char16_t *text, size_t length,
+ enum mcpu_language *language )
+{
+ enum mcpu_language candidate;
+
+ if( text == NULL || language == NULL )
+ return( -1 );
+
+ /*
+ * MCPU_LANG_0 is the initial base language state. It is deliberately
+ * not a valid #lang argument. Every accepted name comes directly from
+ * the internal switched-language table and is matched case-insensitively.
+ */
+ for( candidate = MCPU_LANG_DIFF;
+ candidate < MCPU_LANG__COUNT;
+ candidate = (enum mcpu_language)(candidate + 1) )
+ {
+ if( text_equal_ascii_nocase(text, length,
+ mcpu_language_name(candidate)) )
+ {
+ *language = candidate;
+ return( 0 );
+ }
+ }
+
+ return( -1 );
+}
+
+const char *
+mcpu_language_config_variable( enum mcpu_language language )
+{
+ switch( language )
+ {
+ case MCPU_LANG_DIFF: return( "MCPU_CPP_DIFF_INCLUDE_PATH" );
+ case MCPU_LANG_DIFT: return( "MCPU_CPP_DIFT_INCLUDE_PATH" );
+ case MCPU_LANG_ALG: return( "MCPU_CPP_ALG_INCLUDE_PATH" );
+ case MCPU_LANG_AS: return( "MCPU_CPP_AS_INCLUDE_PATH" );
+ case MCPU_LANG_AVM: return( "MCPU_CPP_AVM_INCLUDE_PATH" );
+ case MCPU_LANG_ACS: return( "MCPU_CPP_ACS_INCLUDE_PATH" );
+ default: return( NULL );
+ }
+}
+
+
+const char *
+mcpu_language_directory_name( enum mcpu_language language )
+{
+ switch( language )
+ {
+ case MCPU_LANG_DIFF: return( "diff" );
+ case MCPU_LANG_DIFT: return( "dift" );
+ case MCPU_LANG_ALG: return( "alg" );
+ case MCPU_LANG_AS: return( "as" );
+ case MCPU_LANG_AVM: return( "avm" );
+ case MCPU_LANG_ACS: return( "acs" );
+ default: return( NULL );
+ }
+}
diff --git a/src/mcpp-language.h b/src/mcpp-language.h
new file mode 100644
index 0000000..a758e37
--- /dev/null
+++ b/src/mcpp-language.h
@@ -0,0 +1,24 @@
+#ifndef __MCPU_CPP_LANGUAGE_H__
+#define __MCPU_CPP_LANGUAGE_H__ 1
+
+#include <defs.h>
+
+enum mcpu_language
+{
+ MCPU_LANG_0 = 0,
+ MCPU_LANG_DIFF,
+ MCPU_LANG_DIFT,
+ MCPU_LANG_ALG,
+ MCPU_LANG_AS,
+ MCPU_LANG_AVM,
+ MCPU_LANG_ACS,
+ MCPU_LANG__COUNT
+};
+
+const char *mcpu_language_name( enum mcpu_language language );
+int mcpu_language_from_text( const __mpu_char16_t *text, size_t length,
+ enum mcpu_language *language );
+const char *mcpu_language_config_variable( enum mcpu_language language );
+const char *mcpu_language_directory_name( enum mcpu_language language );
+
+#endif /* __MCPU_CPP_LANGUAGE_H__ */
diff --git a/src/mcpp-lexer.c b/src/mcpp-lexer.c
new file mode 100644
index 0000000..067e1a2
--- /dev/null
+++ b/src/mcpp-lexer.c
@@ -0,0 +1,371 @@
+#include <defs.h>
+
+void
+mcpu_pp_lexer_init( mcpu_pp_lexer_state *state )
+{
+ if( state )
+ {
+ state->in_block_comment = 0;
+ state->options = NULL;
+ state->filename = NULL;
+ state->line_number = 0;
+ state->splice_offsets = NULL;
+ state->splice_count = 0;
+ }
+}
+
+void
+mcpu_pp_lexer_set_diagnostics( mcpu_pp_lexer_state *state,
+ const mcpp_options *options,
+ const char *filename,
+ unsigned line_number,
+ const size_t *splice_offsets,
+ size_t splice_count )
+{
+ if( state == NULL )
+ return;
+
+ state->options = options;
+ state->filename = filename;
+ state->line_number = line_number;
+ state->splice_offsets = splice_offsets;
+ state->splice_count = splice_count;
+}
+
+static unsigned
+lexer_warning_line( const mcpu_pp_lexer_state *state, size_t offset )
+{
+ size_t i;
+ unsigned line;
+
+ line = state->line_number;
+ for( i = 0; i < state->splice_count; ++i )
+ if( state->splice_offsets[i] <= offset )
+ ++line;
+
+ return( line );
+}
+
+static int
+lexer_has_splice_after( const mcpu_pp_lexer_state *state, size_t offset )
+{
+ size_t i;
+
+ for( i = 0; i < state->splice_count; ++i )
+ if( state->splice_offsets[i] >= offset )
+ return( 1 );
+
+ return( 0 );
+}
+
+static int
+lexer_comment_warning( const mcpu_pp_lexer_state *state, size_t offset,
+ const char *message )
+{
+ if( state->options == NULL || !state->options->warn_comments )
+ return( 0 );
+
+ return( mcpp_diagnostic_warning(state->options,
+ state->filename,
+ lexer_warning_line(state, offset),
+ "%s", message) );
+}
+
+int
+mcpu_pp_is_identifier_start( __mpu_char16_t c )
+{
+ return( c == '_' || mpu_ucs2_is_xid_start(c) );
+}
+
+int
+mcpu_pp_is_identifier_char( __mpu_char16_t c )
+{
+ return( c == '_' || c == '$' || mpu_ucs2_is_xid_continue(c) );
+}
+
+static int
+is_quote16( __mpu_char16_t c, enum mcpu_language language )
+{
+ if( c == '"' || c == '`' )
+ return( 1 );
+
+ if( c == '\'' && language != MCPU_LANG_DIFF )
+ return( 1 );
+
+ return( 0 );
+}
+
+static int
+append_space_once( mcpu_text *text )
+{
+ if( text->length != 0 )
+ {
+ __mpu_char16_t last = text->data[text->length - 1];
+ if( last == ' ' || last == '\t' || last == '\f' || last == '\v' ||
+ last == '\r' || last == '\n' )
+ return( 0 );
+ }
+
+ return( mcpu_text_append_char(text, ' ') );
+}
+
+static int
+is_blank16( __mpu_char16_t c )
+{
+ return( c == ' ' || c == '\t' || c == '\f' ||
+ c == '\v' || c == '\r' );
+}
+
+static void
+remove_comment_tail_space( mcpu_text *text )
+{
+ size_t end;
+ int newline;
+
+ if( text == NULL || text->length == 0 )
+ return;
+
+ newline = text->data[text->length - 1] == '\n';
+ end = newline ? text->length - 1 : text->length;
+
+ while( end != 0 && is_blank16(text->data[end - 1]) )
+ --end;
+
+ if( newline )
+ text->data[end++] = '\n';
+
+ text->length = end;
+ text->data[text->length] = 0;
+}
+
+int
+mcpu_pp_prepare_line( mcpu_pp_lexer_state *state,
+ const __mpu_char16_t *line,
+ size_t length,
+ enum mcpu_language language,
+ mcpu_text *prepared )
+{
+ size_t p = 0;
+ __mpu_char16_t quote = 0;
+ int escaped = 0;
+ int first_token = 1;
+ int saw_hash = 0;
+ int reading_directive = 0;
+ int directive_done = 0;
+ int include_directive = 0;
+ int include_argument_pending = 0;
+ int in_include_angle = 0;
+ __mpu_char16_t directive[32];
+ size_t directive_length = 0;
+ int comment_tail;
+
+ if( state == NULL || line == NULL || prepared == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ mcpu_text_free( prepared );
+ mcpu_text_init( prepared );
+ comment_tail = state->in_block_comment;
+
+ while( p < length )
+ {
+ __mpu_char16_t c = line[p];
+
+ if( state->in_block_comment )
+ {
+ if( p + 1 < length && c == '/' && line[p + 1] == '*' )
+ {
+ if( lexer_comment_warning(state, p, "\"/*\" within comment") != 0 )
+ return( -1 );
+ }
+
+ if( p + 1 < length && c == '*' && line[p + 1] == '/' )
+ {
+ state->in_block_comment = 0;
+ p += 2;
+ }
+ else
+ {
+ if( c == '\n' && mcpu_text_append_char(prepared, c) != 0 )
+ return( -1 );
+ ++p;
+ }
+ continue;
+ }
+
+ if( in_include_angle )
+ {
+ if( mcpu_text_append_char(prepared, c) != 0 )
+ return( -1 );
+ if( c == '>' || c == '\n' )
+ in_include_angle = 0;
+ ++p;
+ continue;
+ }
+
+ if( quote )
+ {
+ if( mcpu_text_append_char(prepared, c) != 0 )
+ return( -1 );
+
+ if( escaped )
+ escaped = 0;
+ else if( c == '\\' )
+ escaped = 1;
+ else if( c == quote || c == '\n' )
+ quote = 0;
+
+ ++p;
+ continue;
+ }
+
+ if( p + 1 < length && c == '/' && line[p + 1] == '*' )
+ {
+ if( append_space_once(prepared) != 0 )
+ return( -1 );
+ comment_tail = 1;
+ state->in_block_comment = 1;
+ p += 2;
+ continue;
+ }
+
+ if( p + 1 < length && c == '/' && line[p + 1] == '/' )
+ {
+ if( lexer_has_splice_after(state, p + 2) &&
+ lexer_comment_warning(state, p, "multi-line comment") != 0 )
+ return( -1 );
+
+ if( append_space_once(prepared) != 0 )
+ return( -1 );
+ comment_tail = 1;
+ while( p < length && line[p] != '\n' )
+ ++p;
+ continue;
+ }
+
+ if( comment_tail && !is_blank16(c) && c != '\n' )
+ comment_tail = 0;
+
+ if( first_token )
+ {
+ if( c == ' ' || c == '\t' || c == '\f' || c == '\v' || c == '\r' )
+ {
+ if( mcpu_text_append_char(prepared, c) != 0 )
+ return( -1 );
+ ++p;
+ continue;
+ }
+
+ first_token = 0;
+ if( c == '#' )
+ {
+ saw_hash = 1;
+ reading_directive = 1;
+ }
+ }
+
+ if( reading_directive && !directive_done && saw_hash && c != '#' )
+ {
+ if( directive_length == 0 &&
+ (c == ' ' || c == '\t' || c == '\f' || c == '\v' || c == '\r') )
+ {
+ if( mcpu_text_append_char(prepared, c) != 0 )
+ return( -1 );
+ ++p;
+ continue;
+ }
+
+ if( mcpu_pp_is_identifier_char(c) )
+ {
+ if( directive_length + 1 < sizeof(directive) / sizeof(directive[0]) )
+ directive[directive_length++] = c;
+ if( mcpu_text_append_char(prepared, c) != 0 )
+ return( -1 );
+ ++p;
+ continue;
+ }
+
+ directive_done = 1;
+ reading_directive = 0;
+ if( (directive_length == 7 &&
+ directive[0] == 'i' && directive[1] == 'n' &&
+ directive[2] == 'c' && directive[3] == 'l' &&
+ directive[4] == 'u' && directive[5] == 'd' &&
+ directive[6] == 'e') ||
+ (directive_length == 12 &&
+ directive[0] == 'i' && directive[1] == 'n' &&
+ directive[2] == 'c' && directive[3] == 'l' &&
+ directive[4] == 'u' && directive[5] == 'd' &&
+ directive[6] == 'e' && directive[7] == '_' &&
+ directive[8] == 'n' && directive[9] == 'e' &&
+ directive[10] == 'x' && directive[11] == 't') )
+ {
+ include_directive = 1;
+ include_argument_pending = 1;
+ }
+ }
+
+ if( include_directive && include_argument_pending )
+ {
+ if( c == ' ' || c == '\t' || c == '\f' || c == '\v' || c == '\r' )
+ {
+ if( mcpu_text_append_char(prepared, c) != 0 )
+ return( -1 );
+ ++p;
+ continue;
+ }
+
+ include_argument_pending = 0;
+ if( c == '<' )
+ in_include_angle = 1;
+ }
+
+ if( is_quote16(c, language) )
+ quote = c;
+
+ if( mcpu_text_append_char(prepared, c) != 0 )
+ return( -1 );
+ ++p;
+ }
+
+ if( comment_tail )
+ remove_comment_tail_space( prepared );
+
+ return( 0 );
+}
+
+int
+mcpu_pp_find_directive( const __mpu_char16_t *line,
+ size_t length,
+ size_t *hash_offset )
+{
+ size_t p = 0;
+
+ if( line == NULL || hash_offset == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ while( p < length )
+ {
+ if( line[p] == ' ' || line[p] == '\t' || line[p] == '\f' ||
+ line[p] == '\v' || line[p] == '\r' )
+ {
+ ++p;
+ continue;
+ }
+
+ if( line[p] == '#' )
+ {
+ *hash_offset = p;
+ return( 1 );
+ }
+
+ break;
+ }
+
+ return( 0 );
+}
diff --git a/src/mcpp-lexer.h b/src/mcpp-lexer.h
new file mode 100644
index 0000000..cc18799
--- /dev/null
+++ b/src/mcpp-lexer.h
@@ -0,0 +1,38 @@
+#ifndef __MCPU_CPP_LEXER_H__
+#define __MCPU_CPP_LEXER_H__ 1
+
+#include <defs.h>
+#include <mcpp-language.h>
+#include <mcpp-text.h>
+
+typedef struct mcpp_options mcpp_options;
+typedef struct mcpu_pp_lexer_state mcpu_pp_lexer_state;
+struct mcpu_pp_lexer_state
+{
+ int in_block_comment;
+ const mcpp_options *options;
+ const char *filename;
+ unsigned line_number;
+ const size_t *splice_offsets;
+ size_t splice_count;
+};
+
+void mcpu_pp_lexer_init( mcpu_pp_lexer_state *state );
+void mcpu_pp_lexer_set_diagnostics( mcpu_pp_lexer_state *state,
+ const mcpp_options *options,
+ const char *filename,
+ unsigned line_number,
+ const size_t *splice_offsets,
+ size_t splice_count );
+int mcpu_pp_prepare_line( mcpu_pp_lexer_state *state,
+ const __mpu_char16_t *line,
+ size_t length,
+ enum mcpu_language language,
+ mcpu_text *prepared );
+int mcpu_pp_find_directive( const __mpu_char16_t *line,
+ size_t length,
+ size_t *hash_offset );
+int mcpu_pp_is_identifier_start( __mpu_char16_t c );
+int mcpu_pp_is_identifier_char( __mpu_char16_t c );
+
+#endif /* __MCPU_CPP_LEXER_H__ */
diff --git a/src/mcpp-lib.c b/src/mcpp-lib.c
new file mode 100644
index 0000000..a0879f2
--- /dev/null
+++ b/src/mcpp-lib.c
@@ -0,0 +1,3443 @@
+#include <defs.h>
+
+#include <sys/stat.h>
+
+
+struct mcpp_dependency
+{
+ char *path;
+ dev_t device;
+ ino_t inode;
+ int has_identity;
+ int unresolved;
+ int system_only;
+ mcpp_dependency *next;
+};
+
+
+static int
+include_dir_is_system( const mcpu_include_dir *dir )
+{
+ if( dir == NULL )
+ return( 0 );
+
+ switch( dir->class_type )
+ {
+ case MCPU_INCLUDE_SYSTEM_EXPLICIT:
+ case MCPU_INCLUDE_SYSTEM_CONFIG_LANGUAGE:
+ case MCPU_INCLUDE_SYSTEM_CONFIG:
+ case MCPU_INCLUDE_AFTER_EXPLICIT:
+ case MCPU_INCLUDE_AFTER_CONFIG:
+ return( 1 );
+ default:
+ return( 0 );
+ }
+}
+
+
+static int
+dependency_add_internal( mcpp *cpp, const char *path, int system_header,
+ int unresolved )
+{
+ mcpp_dependency *p;
+ struct stat st;
+ int has_identity;
+
+ if( cpp == NULL || path == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ /*
+ * Resolved dependencies belong to the physical identity domain and are
+ * deduplicated by st_dev/st_ino whenever possible. -MG dependencies have
+ * no physical object yet, so they deliberately bypass stat() and occupy a
+ * separate unresolved identity domain keyed by the exact include operand.
+ * This prevents an unrelated file in the current directory from turning an
+ * unresolved <name> into a false physical match.
+ */
+ has_identity = !unresolved && stat(path, &st) == 0;
+
+ for( p = cpp->dependencies_first; p != NULL; p = p->next )
+ {
+ int same = 0;
+
+ if( p->unresolved != unresolved )
+ continue;
+
+ if( unresolved )
+ same = strcmp(p->path, path) == 0;
+ else if( has_identity && p->has_identity )
+ same = p->device == st.st_dev && p->inode == st.st_ino;
+ else if( !has_identity && !p->has_identity )
+ same = strcmp(p->path, path) == 0;
+
+ if( same )
+ {
+ /*
+ * A resolved physical file may first be reached through a system path
+ * and later through a user path; preserve the established registry rule
+ * that the user reach makes it a user dependency. An unresolved -MG
+ * entry has no physical provenance, so its first classification remains
+ * authoritative, matching GNU CPP's missing-header behavior.
+ */
+ if( !unresolved && !system_header )
+ p->system_only = 0;
+ return( 0 );
+ }
+ }
+
+ p = (mcpp_dependency *)calloc( 1, sizeof(*p) );
+ if( p == NULL )
+ return( -1 );
+
+ p->path = strdup( path );
+ if( p->path == NULL )
+ {
+ free( p );
+ return( -1 );
+ }
+
+ p->has_identity = has_identity;
+ p->unresolved = unresolved != 0;
+ if( has_identity )
+ {
+ p->device = st.st_dev;
+ p->inode = st.st_ino;
+ }
+ p->system_only = system_header != 0;
+
+ if( cpp->dependencies_last != NULL )
+ cpp->dependencies_last->next = p;
+ else
+ cpp->dependencies_first = p;
+ cpp->dependencies_last = p;
+
+ return( 0 );
+}
+
+
+static int
+dependency_add( mcpp *cpp, const char *path, int system_header )
+{
+ return( dependency_add_internal(cpp, path, system_header, 0) );
+}
+
+
+static int
+dependency_add_unresolved( mcpp *cpp, const char *path,
+ int system_header )
+{
+ return( dependency_add_internal(cpp, path, system_header, 1) );
+}
+
+
+static void
+dependencies_free( mcpp *cpp )
+{
+ mcpp_dependency *p;
+
+ if( cpp == NULL )
+ return;
+
+ p = cpp->dependencies_first;
+ while( p != NULL )
+ {
+ mcpp_dependency *next = p->next;
+ free( p->path );
+ free( p );
+ p = next;
+ }
+
+ cpp->dependencies_first = NULL;
+ cpp->dependencies_last = NULL;
+}
+
+struct mcpp_once_file
+{
+ dev_t device;
+ ino_t inode;
+ mcpp_once_file *next;
+};
+
+
+static int
+once_file_identity( const char *filename, dev_t *device, ino_t *inode )
+{
+ struct stat st;
+
+ if( filename == NULL || device == NULL || inode == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ if( stat(filename, &st) != 0 )
+ return( -1 );
+
+ *device = st.st_dev;
+ *inode = st.st_ino;
+ return( 0 );
+}
+
+
+static int
+once_file_seen_identity( const mcpp *cpp, dev_t device, ino_t inode )
+{
+ const mcpp_once_file *p;
+
+ if( cpp == NULL )
+ return( 0 );
+
+ for( p = cpp->once_files; p != NULL; p = p->next )
+ if( p->device == device && p->inode == inode )
+ return( 1 );
+
+ return( 0 );
+}
+
+
+static int
+once_file_seen( const mcpp *cpp, const char *filename )
+{
+ dev_t device;
+ ino_t inode;
+
+ if( once_file_identity(filename, &device, &inode) != 0 )
+ return( 0 );
+
+ return( once_file_seen_identity(cpp, device, inode) );
+}
+
+
+static int
+once_file_mark( mcpp *cpp, const char *filename )
+{
+ mcpp_once_file *entry;
+ dev_t device;
+ ino_t inode;
+
+ if( cpp == NULL || filename == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ /*
+ * A stream such as <stdin> has no stable filesystem identity. In that
+ * case #pragma once is consumed but there is nothing that can be entered
+ * in the physical-file registry.
+ */
+ if( once_file_identity(filename, &device, &inode) != 0 )
+ return( 0 );
+
+ if( once_file_seen_identity(cpp, device, inode) )
+ return( 0 );
+
+ entry = (mcpp_once_file *)calloc( 1, sizeof(*entry) );
+ if( entry == NULL )
+ return( -1 );
+
+ entry->device = device;
+ entry->inode = inode;
+ entry->next = cpp->once_files;
+ cpp->once_files = entry;
+ return( 0 );
+}
+
+
+static void
+once_files_free( mcpp *cpp )
+{
+ mcpp_once_file *p;
+
+ if( cpp == NULL )
+ return;
+
+ p = cpp->once_files;
+ while( p != NULL )
+ {
+ mcpp_once_file *next = p->next;
+ free( p );
+ p = next;
+ }
+ cpp->once_files = NULL;
+}
+
+
+static enum mcpu_language
+current_language( const mcpp *cpp )
+{
+ return( cpp->lang_stack[cpp->lang_depth - 1] );
+}
+
+static int
+install_builtin( mcpp *cpp, const char *name,
+ enum mcpu_macro_builtin builtin )
+{
+ mcpu_text text;
+ int rc;
+
+ mcpu_text_init( &text );
+ if( mcpu_text_from_utf8(&text, name) != 0 )
+ return( -1 );
+
+ rc = mcpu_macro_define_builtin( &cpp->macros, text.data, text.length,
+ builtin );
+ mcpu_text_free( &text );
+ return( rc );
+}
+
+
+static int
+initialize_builtins( mcpp *cpp )
+{
+ if( cpp->builtins_initialized )
+ return( 0 );
+
+ if( install_builtin(cpp, "__FILE__",
+ MCPU_MACRO_BUILTIN_FILE) != 0 ||
+ install_builtin(cpp, "__LINE__",
+ MCPU_MACRO_BUILTIN_LINE) != 0 ||
+ install_builtin(cpp, "__DATE__",
+ MCPU_MACRO_BUILTIN_DATE) != 0 ||
+ install_builtin(cpp, "__TIME__",
+ MCPU_MACRO_BUILTIN_TIME) != 0 ||
+ install_builtin(cpp, "__BASE_FILE__",
+ MCPU_MACRO_BUILTIN_BASE_FILE) != 0 ||
+ install_builtin(cpp, "__INCLUDE_LEVEL__",
+ MCPU_MACRO_BUILTIN_INCLUDE_LEVEL) != 0 ||
+ mcpu_predefined_install(&cpp->macros) != 0 )
+ return( -1 );
+
+ cpp->builtins_initialized = 1;
+ return( 0 );
+}
+
+
+static int
+initialize_timestamp( mcpp *cpp )
+{
+ static const char *months[] =
+ {
+ "Jan", "Feb", "Mar", "Apr", "May", "Jun",
+ "Jul", "Aug", "Sep", "Oct", "Nov", "Dec"
+ };
+ time_t now;
+ struct tm *tm;
+
+ now = time( NULL );
+ if( now == (time_t)-1 )
+ return( -1 );
+
+ tm = localtime( &now );
+ if( tm == NULL || tm->tm_mon < 0 || tm->tm_mon > 11 )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ snprintf( cpp->preprocess_date, sizeof(cpp->preprocess_date),
+ "%s %2d %4d", months[tm->tm_mon], tm->tm_mday,
+ tm->tm_year + 1900 );
+ snprintf( cpp->preprocess_time, sizeof(cpp->preprocess_time),
+ "%02d:%02d:%02d", tm->tm_hour, tm->tm_min, tm->tm_sec );
+
+ return( 0 );
+}
+
+
+static void
+macro_context( const mcpp *cpp, const char *filename,
+ unsigned line_number, mcpu_macro_expansion_context *context )
+{
+ context->filename = filename;
+ context->base_filename = cpp->base_filename;
+ context->line_number = line_number;
+ context->include_level = cpp->include_depth ? cpp->include_depth - 1 : 0;
+ context->date = cpp->preprocess_date;
+ context->time = cpp->preprocess_time;
+ context->options = MCPP_OPTIONS(cpp);
+}
+
+
+static char *
+escape_line_filename( const char *filename )
+{
+ size_t n;
+ size_t i;
+ size_t o = 0;
+ char *escaped;
+
+ if( filename == NULL )
+ {
+ errno = EINVAL;
+ return( NULL );
+ }
+
+ n = strlen( filename );
+ if( n > (SIZE_MAX - 1) / 2 )
+ {
+ errno = EOVERFLOW;
+ return( NULL );
+ }
+
+ escaped = (char *)malloc( n * 2 + 1 );
+ if( escaped == NULL )
+ return( NULL );
+
+ for( i = 0; i < n; ++i )
+ {
+ switch( filename[i] )
+ {
+ case '\\':
+ case '"':
+ escaped[o++] = '\\';
+ escaped[o++] = filename[i];
+ break;
+ case '\n':
+ escaped[o++] = '\\';
+ escaped[o++] = 'n';
+ break;
+ case '\r':
+ escaped[o++] = '\\';
+ escaped[o++] = 'r';
+ break;
+ case '\t':
+ escaped[o++] = '\\';
+ escaped[o++] = 't';
+ break;
+ default:
+ escaped[o++] = filename[i];
+ break;
+ }
+ }
+
+ escaped[o] = 0;
+ return( escaped );
+}
+
+static int
+set_output_position( mcpp *cpp, unsigned line, const char *filename )
+{
+ char *copy;
+
+ if( cpp == NULL || filename == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ copy = strdup( filename );
+ if( copy == NULL )
+ return( -1 );
+
+ free( cpp->output_filename );
+ cpp->output_filename = copy;
+ cpp->output_line = line;
+ return( 0 );
+}
+
+static int
+append_line_marker( mcpp *cpp, unsigned line, const char *filename,
+ int file_change )
+{
+ char number[64];
+ char *escaped;
+ char *position_filename;
+
+ escaped = escape_line_filename( filename );
+ if( escaped == NULL )
+ return( -1 );
+
+ position_filename = strdup( filename );
+ if( position_filename == NULL )
+ {
+ free( escaped );
+ return( -1 );
+ }
+
+ if( mcpu_text_append_ascii( &cpp->output, "# " ) != 0 )
+ {
+ free( position_filename );
+ free( escaped );
+ return( -1 );
+ }
+
+ snprintf( number, sizeof(number), "%u", line );
+ if( mcpu_text_append_ascii( &cpp->output, number ) != 0 ||
+ mcpu_text_append_ascii( &cpp->output, " \"" ) != 0 ||
+ mcpu_text_append_utf8( &cpp->output, escaped ) != 0 ||
+ mcpu_text_append_ascii( &cpp->output, "\"" ) != 0 )
+ {
+ free( position_filename );
+ free( escaped );
+ return( -1 );
+ }
+
+ if( file_change == 1 || file_change == 2 )
+ {
+ if( mcpu_text_append_ascii(&cpp->output,
+ file_change == 1 ? " 1" : " 2") != 0 )
+ {
+ free( position_filename );
+ free( escaped );
+ return( -1 );
+ }
+ }
+
+ if( mcpu_text_append_char(&cpp->output, '\n') != 0 )
+ {
+ free( position_filename );
+ free( escaped );
+ return( -1 );
+ }
+
+ free( cpp->output_filename );
+ cpp->output_filename = position_filename;
+ cpp->output_line = line;
+ free( escaped );
+ return( 0 );
+}
+
+static int
+append_output_newlines( mcpp *cpp, unsigned count )
+{
+ unsigned i;
+
+ for( i = 0; i < count; ++i )
+ {
+ if( mcpu_text_append_char(&cpp->output, '\n') != 0 )
+ return( -1 );
+ }
+
+ if( cpp->output_filename != NULL )
+ cpp->output_line += count;
+
+ return( 0 );
+}
+
+/*
+ * GNU CPP keeps short invisible source gaps as ordinary newlines, but once
+ * the next visible source position is eight or more lines away it emits a
+ * fresh line marker instead. The discarded text is therefore forgotten as
+ * output bytes, but never forgotten as source position: "remove, but
+ * remember".
+ */
+static int
+sync_output_position( mcpp *cpp, unsigned line, const char *filename )
+{
+ if( cpp->output_filename != NULL &&
+ strcmp(cpp->output_filename, filename) == 0 &&
+ line >= cpp->output_line && line - cpp->output_line < 8 )
+ return( append_output_newlines(cpp, line - cpp->output_line) );
+
+ return( append_line_marker(cpp, line, filename, 0) );
+}
+
+static int
+text_has_visible_tokens( const __mpu_char16_t *text, size_t length )
+{
+ size_t i;
+
+ for( i = 0; i < length; ++i )
+ {
+ switch( text[i] )
+ {
+ case ' ':
+ case '\t':
+ case '\f':
+ case '\v':
+ case '\r':
+ case '\n':
+ break;
+ default:
+ return( 1 );
+ }
+ }
+
+ return( 0 );
+}
+
+static void
+advance_output_position( mcpp *cpp,
+ const __mpu_char16_t *text, size_t length )
+{
+ size_t i;
+
+ if( cpp->output_filename == NULL )
+ return;
+
+ for( i = 0; i < length; ++i )
+ if( text[i] == '\n' )
+ ++cpp->output_line;
+}
+
+static int
+append_visible_text( mcpp *cpp, unsigned line, const char *filename,
+ const __mpu_char16_t *text, size_t length )
+{
+ if( !text_has_visible_tokens(text, length) )
+ return( 0 );
+
+ if( sync_output_position(cpp, line, filename) != 0 ||
+ mcpu_text_append(&cpp->output, text, length) != 0 )
+ return( -1 );
+
+ advance_output_position( cpp, text, length );
+ return( 0 );
+}
+
+static int
+is_space16( __mpu_char16_t c )
+{
+ return( c == ' ' || c == '\t' || c == '\f' || c == '\v' || c == '\r' );
+}
+
+static int
+is_ident16( __mpu_char16_t c )
+{
+ return( (c >= 'a' && c <= 'z') ||
+ (c >= 'A' && c <= 'Z') ||
+ (c >= '0' && c <= '9') || c == '_' || c >= 0x80 );
+}
+
+static int
+replacement_quote16( __mpu_char16_t c, enum mcpu_language language )
+{
+ if( c == '"' || c == '`' )
+ return( 1 );
+
+ if( c == '\'' && language != MCPU_LANG_DIFF )
+ return( 1 );
+
+ return( 0 );
+}
+
+
+static int
+normalize_macro_replacement( const __mpu_char16_t *text, size_t length,
+ enum mcpu_language language,
+ mcpu_text *normalized )
+{
+ size_t p = 0;
+ __mpu_char16_t quote = 0;
+ int escaped = 0;
+ int pending_space = 0;
+
+ while( p < length )
+ {
+ __mpu_char16_t c = text[p++];
+
+ if( quote != 0 )
+ {
+ if( mcpu_text_append_char(normalized, c) != 0 )
+ return( -1 );
+
+ if( escaped )
+ escaped = 0;
+ else if( c == '\\' )
+ escaped = 1;
+ else if( c == quote || c == '\n' )
+ quote = 0;
+
+ continue;
+ }
+
+ if( is_space16(c) || c == '\n' )
+ {
+ if( normalized->length != 0 )
+ pending_space = 1;
+ continue;
+ }
+
+ if( pending_space )
+ {
+ if( mcpu_text_append_char(normalized, ' ') != 0 )
+ return( -1 );
+ pending_space = 0;
+ }
+
+ if( mcpu_text_append_char(normalized, c) != 0 )
+ return( -1 );
+
+ if( replacement_quote16(c, language) )
+ quote = c;
+ }
+
+ return( 0 );
+}
+
+
+static void
+skip_space16( const __mpu_char16_t *line, size_t length, size_t *pos )
+{
+ while( *pos < length && is_space16(line[*pos]) )
+ ++*pos;
+}
+
+static int
+parse_directive_name( const __mpu_char16_t *line, size_t length,
+ size_t hash, size_t *name_start, size_t *name_length,
+ size_t *after_name )
+{
+ size_t p = hash + 1;
+ size_t start;
+
+ skip_space16( line, length, &p );
+ start = p;
+ while( p < length && is_ident16(line[p]) )
+ ++p;
+
+ *name_start = start;
+ *name_length = p - start;
+ *after_name = p;
+
+ return( *name_length != 0 );
+}
+
+static int
+handle_lang( mcpp *cpp, const char *filename, unsigned line_number,
+ const __mpu_char16_t *line, size_t length, size_t p,
+ int spliced )
+{
+ size_t start;
+ size_t end;
+ size_t q;
+ enum mcpu_language language;
+ char *name;
+ int newline;
+
+ if( spliced )
+ {
+ fprintf( stderr,
+ "%s:%u: error: #lang must be contained in one physical source line\n",
+ filename, line_number );
+ return( -1 );
+ }
+
+ skip_space16( line, length, &p );
+ if( p >= length || line[p] != '"' )
+ {
+ fprintf( stderr,
+ "%s:%u: error: string constant expected after #lang\n",
+ filename, line_number );
+ return( -1 );
+ }
+
+ start = ++p;
+ while( p < length && line[p] != '"' && line[p] != '\n' )
+ ++p;
+
+ if( p >= length || line[p] != '"' )
+ {
+ fprintf( stderr,
+ "%s:%u: error: unterminated #lang string constant\n",
+ filename, line_number );
+ return( -1 );
+ }
+ end = p++;
+
+ if( start == end )
+ {
+ fprintf( stderr,
+ "%s:%u: error: empty language name in #lang\n",
+ filename, line_number );
+ return( -1 );
+ }
+
+ for( q = start; q < end; ++q )
+ {
+ if( line[q] == ' ' || line[q] == '\t' || line[q] == '\f' ||
+ line[q] == '\v' || line[q] == '\r' || line[q] == '\n' )
+ {
+ fprintf( stderr,
+ "%s:%u: error: #lang language name must be one word without whitespace\n",
+ filename, line_number );
+ return( -1 );
+ }
+ }
+
+ skip_space16( line, length, &p );
+ if( p < length && line[p] != '\n' )
+ {
+ fprintf( stderr,
+ "%s:%u: error: extra text after #lang string constant\n",
+ filename, line_number );
+ return( -1 );
+ }
+
+ if( mcpu_language_from_text( line + start, end - start, &language ) != 0 )
+ {
+ name = mcpu_text_to_utf8( line + start, end - start );
+ fprintf( stderr, "%s:%u: error: unknown language '%s'\n",
+ filename, line_number, name ? name : "?" );
+ free( name );
+ return( -1 );
+ }
+
+ if( cpp->lang_depth >= MCPU_CPP_LANG_STACK_SIZE )
+ {
+ fprintf( stderr, "%s:%u: error: #lang stack overflow\n",
+ filename, line_number );
+ return( -1 );
+ }
+
+ cpp->lang_stack[cpp->lang_depth++] = language;
+
+ if( cpp->verbose )
+ fprintf( stderr, "%s:%u: #lang %s (depth %lu)\n",
+ filename, line_number, mcpu_language_name(language),
+ (unsigned long)cpp->lang_depth );
+
+ newline = length != 0 && line[length - 1] == '\n';
+ if( mcpu_text_append_ascii(&cpp->output, "#lang \"") != 0 ||
+ mcpu_text_append(&cpp->output, line + start, end - start) != 0 ||
+ mcpu_text_append_char(&cpp->output, '"') != 0 ||
+ (newline && mcpu_text_append_char(&cpp->output, '\n') != 0) )
+ return( -1 );
+
+ return( 0 );
+}
+
+static int
+handle_endlang( mcpp *cpp, const char *filename, unsigned line_number,
+ const __mpu_char16_t *line, size_t length )
+{
+ if( cpp->lang_depth <= 1 )
+ {
+ fprintf( stderr, "%s:%u: error: unbalanced #endlang\n",
+ filename, line_number );
+ return( -1 );
+ }
+
+ --cpp->lang_depth;
+
+ if( cpp->verbose )
+ fprintf( stderr, "%s:%u: #endlang -> %s (depth %lu)\n",
+ filename, line_number, mcpu_language_name(current_language(cpp)),
+ (unsigned long)cpp->lang_depth );
+
+ return( mcpu_text_append( &cpp->output, line, length ) );
+}
+
+static int
+parse_include_filename( const __mpu_char16_t *line, size_t length, size_t p,
+ char **filename, int *quoted )
+{
+ __mpu_char16_t open;
+ __mpu_char16_t close;
+ size_t start;
+ size_t end;
+
+ *filename = NULL;
+ *quoted = 0;
+
+ skip_space16( line, length, &p );
+ if( p >= length )
+ return( -1 );
+
+ open = line[p++];
+ if( open == '"' )
+ {
+ close = '"';
+ *quoted = 1;
+ }
+ else if( open == '<' )
+ close = '>';
+ else
+ return( -1 );
+
+ start = p;
+ while( p < length && line[p] != close && line[p] != '\n' )
+ ++p;
+ end = p;
+
+ if( p >= length || line[p] != close || end == start )
+ return( -1 );
+
+ *filename = mcpu_text_to_utf8( line + start, end - start );
+ return( *filename ? 0 : -1 );
+}
+
+
+static size_t
+line_content_end( const __mpu_char16_t *line, size_t length )
+{
+ size_t end = length;
+
+ if( end != 0 && line[end - 1] == '\n' )
+ --end;
+
+ while( end != 0 && is_space16(line[end - 1]) )
+ --end;
+
+ return( end );
+}
+
+static void
+free_define_args( const __mpu_char16_t **argnames,
+ size_t *argname_lengths )
+{
+ free( argnames );
+ free( argname_lengths );
+}
+
+
+static int
+append_define_arg( const __mpu_char16_t ***argnames,
+ size_t **argname_lengths,
+ size_t *nargs, size_t *capacity,
+ const __mpu_char16_t *name, size_t name_length )
+{
+ const __mpu_char16_t **new_names;
+ size_t *new_lengths;
+ size_t new_capacity;
+
+ if( *nargs == *capacity )
+ {
+ new_capacity = *capacity ? *capacity * 2 : 4;
+ if( new_capacity < *capacity ||
+ new_capacity > SIZE_MAX / sizeof(**argnames) ||
+ new_capacity > SIZE_MAX / sizeof(**argname_lengths) )
+ {
+ errno = EOVERFLOW;
+ return( -1 );
+ }
+
+ new_names = (const __mpu_char16_t **)realloc(
+ (void *)*argnames, new_capacity * sizeof(**argnames) );
+ if( new_names == NULL )
+ return( -1 );
+
+ *argnames = new_names;
+
+ new_lengths = (size_t *)realloc( *argname_lengths,
+ new_capacity * sizeof(**argname_lengths) );
+ if( new_lengths == NULL )
+ return( -1 );
+
+ *argname_lengths = new_lengths;
+ *capacity = new_capacity;
+ }
+
+ (*argnames)[*nargs] = name;
+ (*argname_lengths)[*nargs] = name_length;
+ ++*nargs;
+
+ return( 0 );
+}
+
+
+static int
+replacement_va_opt_range_valid( const __mpu_char16_t *text,
+ size_t length,
+ enum mcpu_language language,
+ int function_like, int variadic,
+ int inside_va_opt,
+ const char *filename,
+ unsigned line_number )
+{
+ size_t p = 0;
+ __mpu_char16_t quote = 0;
+ int escaped = 0;
+
+ while( p < length )
+ {
+ __mpu_char16_t c = text[p];
+
+ if( quote )
+ {
+ if( escaped )
+ escaped = 0;
+ else if( c == '\\' )
+ escaped = 1;
+ else if( c == quote || c == '\n' )
+ quote = 0;
+
+ ++p;
+ continue;
+ }
+
+ if( c == '"' || c == '`' ||
+ (c == '\'' && language != MCPU_LANG_DIFF) )
+ {
+ quote = c;
+ ++p;
+ continue;
+ }
+
+ if( mcpu_pp_is_identifier_start(c) )
+ {
+ size_t q = p + 1;
+
+ while( q < length && mcpu_pp_is_identifier_char(text[q]) )
+ ++q;
+
+ if( mcpu_text_equal_ascii(text + p, q - p, "__VA_OPT__") )
+ {
+ size_t open;
+ size_t close;
+ size_t first;
+ size_t last;
+ size_t r;
+ int depth;
+ __mpu_char16_t inner_quote = 0;
+ int inner_escaped = 0;
+
+ if( !function_like || !variadic )
+ {
+ fprintf( stderr,
+ "%s:%u: error: '__VA_OPT__' may appear only in a variadic macro replacement list\n",
+ filename, line_number );
+ return( 0 );
+ }
+
+ if( inside_va_opt )
+ {
+ fprintf( stderr,
+ "%s:%u: error: '__VA_OPT__' may not appear inside another '__VA_OPT__'\n",
+ filename, line_number );
+ return( 0 );
+ }
+
+ open = q;
+ while( open < length && is_space16(text[open]) )
+ ++open;
+
+ if( open >= length || text[open] != '(' )
+ {
+ fprintf( stderr,
+ "%s:%u: error: '__VA_OPT__' must be followed by '('\n",
+ filename, line_number );
+ return( 0 );
+ }
+
+ depth = 1;
+ close = open + 1;
+ while( close < length && depth != 0 )
+ {
+ __mpu_char16_t d = text[close];
+
+ if( inner_quote )
+ {
+ if( inner_escaped )
+ inner_escaped = 0;
+ else if( d == '\\' )
+ inner_escaped = 1;
+ else if( d == inner_quote || d == '\n' )
+ inner_quote = 0;
+ ++close;
+ continue;
+ }
+
+ if( d == '"' || d == '`' ||
+ (d == '\'' && language != MCPU_LANG_DIFF) )
+ {
+ inner_quote = d;
+ ++close;
+ continue;
+ }
+
+ if( d == '(' )
+ ++depth;
+ else if( d == ')' )
+ --depth;
+
+ ++close;
+ }
+
+ if( depth != 0 )
+ {
+ fprintf( stderr,
+ "%s:%u: error: unterminated '__VA_OPT__'\n",
+ filename, line_number );
+ return( 0 );
+ }
+
+ --close;
+ first = open + 1;
+ while( first < close && is_space16(text[first]) )
+ ++first;
+ last = close;
+ while( last > first && is_space16(text[last - 1]) )
+ --last;
+
+ if( first + 1 < last && text[first] == '#' && text[first + 1] == '#' )
+ {
+ fprintf( stderr,
+ "%s:%u: error: '##' cannot appear at the beginning of '__VA_OPT__'\n",
+ filename, line_number );
+ return( 0 );
+ }
+
+ if( last >= first + 2 && text[last - 2] == '#' && text[last - 1] == '#' )
+ {
+ fprintf( stderr,
+ "%s:%u: error: '##' cannot appear at the end of '__VA_OPT__'\n",
+ filename, line_number );
+ return( 0 );
+ }
+
+ if( !replacement_va_opt_range_valid(text + open + 1,
+ close - open - 1,
+ language,
+ function_like, variadic, 1,
+ filename, line_number) )
+ return( 0 );
+
+ r = close + 1;
+ p = r;
+ continue;
+ }
+
+ p = q;
+ continue;
+ }
+
+ ++p;
+ }
+
+ return( 1 );
+}
+
+
+static int
+replacement_va_opt_valid( const __mpu_char16_t *text,
+ size_t length,
+ enum mcpu_language language,
+ int function_like, int variadic,
+ const char *filename,
+ unsigned line_number )
+{
+ return( replacement_va_opt_range_valid(text, length,
+ language,
+ function_like, variadic, 0,
+ filename, line_number) );
+}
+
+
+static int
+replacement_macro_operators_valid( const __mpu_char16_t *text,
+ size_t length,
+ enum mcpu_language language,
+ int function_like, int variadic,
+ const __mpu_char16_t *const *argnames,
+ const size_t *argname_lengths,
+ size_t nargs,
+ const char *filename,
+ unsigned line_number )
+{
+ size_t p = 0;
+ __mpu_char16_t quote = 0;
+ int escaped = 0;
+ int saw_token = 0;
+ int paste_at_end = 0;
+
+ while( p < length )
+ {
+ __mpu_char16_t c = text[p];
+
+ if( quote )
+ {
+ if( escaped )
+ escaped = 0;
+ else if( c == '\\' )
+ escaped = 1;
+ else if( c == quote || c == '\n' )
+ quote = 0;
+
+ ++p;
+ continue;
+ }
+
+ if( c == '"' || c == '`' ||
+ (c == '\'' && language != MCPU_LANG_DIFF) )
+ {
+ quote = c;
+ saw_token = 1;
+ paste_at_end = 0;
+ ++p;
+ continue;
+ }
+
+ if( c == '#' && p + 1 < length && text[p + 1] == '#' )
+ {
+ if( !saw_token )
+ {
+ fprintf( stderr,
+ "%s:%u: error: '##' cannot appear at the beginning of a macro replacement list\n",
+ filename, line_number );
+ return( 0 );
+ }
+
+ paste_at_end = 1;
+ p += 2;
+ continue;
+ }
+
+ if( c == '#' && function_like )
+ {
+ size_t q = p + 1;
+ size_t start;
+ size_t i;
+ int found = 0;
+
+ while( q < length && is_space16(text[q]) )
+ ++q;
+
+ if( q >= length || !mcpu_pp_is_identifier_start(text[q]) )
+ {
+ fprintf( stderr,
+ "%s:%u: error: '#' operator is not followed by a macro argument name\n",
+ filename, line_number );
+ return( 0 );
+ }
+
+ start = q++;
+ while( q < length && mcpu_pp_is_identifier_char(text[q]) )
+ ++q;
+
+ for( i = 0; i < nargs; ++i )
+ {
+ if( argname_lengths[i] == q - start &&
+ memcmp(argnames[i], text + start,
+ (q - start) * sizeof(__mpu_char16_t)) == 0 )
+ {
+ found = 1;
+ break;
+ }
+ }
+
+ if( !found && variadic &&
+ (mcpu_text_equal_ascii(text + start, q - start, "__VA_ARGS__") ||
+ mcpu_text_equal_ascii(text + start, q - start, "__VA_OPT__")) )
+ found = 1;
+
+ if( !found )
+ {
+ fprintf( stderr,
+ "%s:%u: error: '#' operator should be followed by a macro argument name\n",
+ filename, line_number );
+ return( 0 );
+ }
+
+ saw_token = 1;
+ paste_at_end = 0;
+ p = q;
+ continue;
+ }
+
+ if( is_space16(c) )
+ {
+ ++p;
+ continue;
+ }
+
+ saw_token = 1;
+ paste_at_end = 0;
+ ++p;
+ }
+
+ if( paste_at_end )
+ {
+ fprintf( stderr,
+ "%s:%u: error: '##' cannot appear at the end of a macro replacement list\n",
+ filename, line_number );
+ return( 0 );
+ }
+
+ return( 1 );
+}
+
+
+static int
+handle_define( mcpp *cpp, const char *filename, unsigned line_number,
+ const __mpu_char16_t *line, size_t length, size_t p,
+ enum mcpu_macro_origin origin )
+{
+ size_t name_start;
+ size_t name_end;
+ size_t replacement_start;
+ size_t replacement_end;
+ const __mpu_char16_t **argnames = NULL;
+ size_t *argname_lengths = NULL;
+ size_t nargs = 0;
+ size_t capacity = 0;
+ int function_like = 0;
+ int variadic = 0;
+ mcpu_macro *old;
+ mcpu_text replacement;
+ int rc = -1;
+
+ mcpu_text_init( &replacement );
+
+ skip_space16( line, length, &p );
+ if( p >= length || !mcpu_pp_is_identifier_start(line[p]) )
+ {
+ fprintf( stderr, "%s:%u: error: macro name expected after #define\n",
+ filename, line_number );
+ return( -1 );
+ }
+
+ name_start = p++;
+ while( p < length && mcpu_pp_is_identifier_char(line[p]) )
+ ++p;
+ name_end = p;
+
+ /***************************************************************
+ A function-like definition is recognized only when
+ the opening parenthesis immediately follows the macro name.
+ A space changes the definition into an object-like macro whose
+ replacement happens to begin with '('.
+ ***************************************************************/
+ if( p < length && line[p] == '(' )
+ {
+ size_t i;
+
+ function_like = 1;
+ ++p;
+ skip_space16( line, length, &p );
+
+ if( p + 2 < length &&
+ line[p] == '.' && line[p + 1] == '.' && line[p + 2] == '.' )
+ {
+ variadic = 1;
+ p += 3;
+ skip_space16( line, length, &p );
+ if( p >= length || line[p] != ')' )
+ {
+ fprintf( stderr,
+ "%s:%u: error: badly punctuated parameter list in #define\n",
+ filename, line_number );
+ goto done;
+ }
+ }
+ else
+ {
+ while( p < length && line[p] != ')' )
+ {
+ size_t arg_start;
+ size_t arg_end;
+
+ if( !mcpu_pp_is_identifier_start(line[p]) )
+ {
+ fprintf( stderr,
+ "%s:%u: error: invalid macro parameter name in #define\n",
+ filename, line_number );
+ goto done;
+ }
+
+ arg_start = p++;
+ while( p < length && mcpu_pp_is_identifier_char(line[p]) )
+ ++p;
+ arg_end = p;
+
+ if( mcpu_text_equal_ascii(line + arg_start, arg_end - arg_start,
+ "__VA_ARGS__") )
+ {
+ fprintf( stderr,
+ "%s:%u: error: '__VA_ARGS__' cannot be used as a macro parameter name\n",
+ filename, line_number );
+ goto done;
+ }
+
+ for( i = 0; i < nargs; ++i )
+ {
+ if( argname_lengths[i] == arg_end - arg_start &&
+ memcmp( argnames[i], line + arg_start,
+ (arg_end - arg_start) * sizeof(__mpu_char16_t) ) == 0 )
+ {
+ char *arg = mcpu_text_to_utf8( line + arg_start,
+ arg_end - arg_start );
+ fprintf( stderr,
+ "%s:%u: error: duplicate argument name '%s' in #define\n",
+ filename, line_number, arg ? arg : "?" );
+ free( arg );
+ goto done;
+ }
+ }
+
+ if( append_define_arg(&argnames, &argname_lengths,
+ &nargs, &capacity,
+ line + arg_start, arg_end - arg_start) != 0 )
+ goto done;
+
+ skip_space16( line, length, &p );
+ if( p >= length )
+ {
+ fprintf( stderr,
+ "%s:%u: error: unterminated parameter list in #define\n",
+ filename, line_number );
+ goto done;
+ }
+
+ if( line[p] == ',' )
+ {
+ ++p;
+ skip_space16( line, length, &p );
+
+ if( p + 2 < length &&
+ line[p] == '.' && line[p + 1] == '.' && line[p + 2] == '.' )
+ {
+ variadic = 1;
+ p += 3;
+ skip_space16( line, length, &p );
+ if( p >= length || line[p] != ')' )
+ {
+ fprintf( stderr,
+ "%s:%u: error: badly punctuated parameter list in #define\n",
+ filename, line_number );
+ goto done;
+ }
+ break;
+ }
+
+ if( p >= length || !mcpu_pp_is_identifier_start(line[p]) )
+ {
+ fprintf( stderr,
+ "%s:%u: error: badly punctuated parameter list in #define\n",
+ filename, line_number );
+ goto done;
+ }
+ continue;
+ }
+
+ if( line[p] != ')' )
+ {
+ fprintf( stderr,
+ "%s:%u: error: badly punctuated parameter list in #define\n",
+ filename, line_number );
+ goto done;
+ }
+ }
+ }
+
+ if( p >= length || line[p] != ')' )
+ {
+ fprintf( stderr,
+ "%s:%u: error: unterminated parameter list in #define\n",
+ filename, line_number );
+ goto done;
+ }
+
+ ++p;
+ }
+
+ skip_space16( line, length, &p );
+ replacement_start = p;
+ replacement_end = line_content_end( line, length );
+ if( replacement_end < replacement_start )
+ replacement_end = replacement_start;
+
+ if( normalize_macro_replacement(line + replacement_start,
+ replacement_end - replacement_start,
+ current_language(cpp), &replacement) != 0 )
+ goto done;
+
+ if( !replacement_va_opt_valid(
+ replacement.data, replacement.length,
+ current_language(cpp), function_like, variadic,
+ filename, line_number) ||
+ !replacement_macro_operators_valid(
+ replacement.data, replacement.length,
+ current_language(cpp), function_like, variadic,
+ argnames, argname_lengths, nargs,
+ filename, line_number) )
+ goto done;
+
+ old = mcpu_macro_find( &cpp->macros,
+ line + name_start, name_end - name_start );
+ if( old &&
+ !mcpu_macro_definition_equal(old,
+ function_like ? (int)nargs : -1,
+ function_like ? variadic : 0,
+ argnames, argname_lengths,
+ replacement.data,
+ replacement.length) )
+ {
+ char *name = mcpu_text_to_utf8( line + name_start,
+ name_end - name_start );
+ if( mcpp_diagnostic_warning(MCPP_OPTIONS(cpp), filename, line_number,
+ "macro '%s' redefined",
+ name ? name : "?") != 0 )
+ {
+ free( name );
+ goto done;
+ }
+ free( name );
+ }
+
+ if( function_like )
+ rc = mcpu_macro_define_function(&cpp->macros,
+ line + name_start, name_end - name_start,
+ argnames, argname_lengths, nargs, variadic,
+ replacement.data, replacement.length,
+ origin);
+ else
+ rc = mcpu_macro_define_object(&cpp->macros,
+ line + name_start, name_end - name_start,
+ replacement.data, replacement.length,
+ origin);
+
+done:
+ mcpu_text_free( &replacement );
+ free_define_args( argnames, argname_lengths );
+ return( rc );
+}
+
+
+static int
+handle_undef( mcpp *cpp, const char *filename, unsigned line_number,
+ const __mpu_char16_t *line, size_t length, size_t p )
+{
+ size_t name_start;
+ size_t name_end;
+
+ skip_space16( line, length, &p );
+ if( p >= length || !mcpu_pp_is_identifier_start(line[p]) )
+ {
+ fprintf( stderr, "%s:%u: error: macro name expected after #undef\n",
+ filename, line_number );
+ return( -1 );
+ }
+
+ name_start = p++;
+ while( p < length && mcpu_pp_is_identifier_char(line[p]) )
+ ++p;
+ name_end = p;
+
+ skip_space16( line, length, &p );
+ if( p < length && line[p] != '\n' )
+ {
+ fprintf( stderr, "%s:%u: error: extra tokens after #undef\n",
+ filename, line_number );
+ return( -1 );
+ }
+
+ if( mcpu_macro_undef(&cpp->macros,
+ line + name_start, name_end - name_start) < 0 )
+ return( -1 );
+
+ return( 0 );
+}
+
+
+static int
+command_line_text( const mcpp *cpp, const mcpp_option_action *action,
+ mcpu_text *text )
+{
+ if( mcpu_text_from_utf8(text, action->argument) == 0 )
+ return( 0 );
+
+ if( errno == EILSEQ )
+ fprintf( stderr,
+ "%s: %s: invalid UTF-8 or non-UCS-2 character\n",
+ MCPP_OPTIONS(cpp)->progname, action->option );
+ else
+ fprintf( stderr, "%s: %s: %s\n",
+ MCPP_OPTIONS(cpp)->progname, action->option, strerror(errno) );
+
+ return( -1 );
+}
+
+
+static size_t
+command_line_define_declarator_end( const mcpu_text *text, size_t equal )
+{
+ size_t p;
+
+ if( text == NULL || text->length == 0 || equal == 0 )
+ return( 0 );
+
+ if( !mcpu_pp_is_identifier_start(text->data[0]) )
+ return( equal );
+
+ p = 1;
+ while( p < equal && mcpu_pp_is_identifier_char(text->data[p]) )
+ ++p;
+
+ if( p < equal && text->data[p] == '(' )
+ {
+ ++p;
+ while( p < equal && text->data[p] != ')' )
+ ++p;
+ if( p < equal && text->data[p] == ')' )
+ ++p;
+ }
+
+ return( p );
+}
+
+
+static int
+prepare_command_line_define( mcpp *cpp, const mcpp_option_action *action,
+ const mcpu_text *text, mcpu_text *prepared )
+{
+ mcpu_text line;
+ mcpu_pp_lexer_state lexer;
+ size_t equal;
+ size_t declarator_end;
+ int has_equal;
+ int rc = -1;
+
+ mcpu_text_init( &line );
+
+ equal = 0;
+ has_equal = 0;
+ while( equal < text->length )
+ {
+ if( text->data[equal] == '=' )
+ {
+ has_equal = 1;
+ break;
+ }
+ ++equal;
+ }
+
+ declarator_end = command_line_define_declarator_end(
+ text, has_equal ? equal : text->length );
+
+ if( mcpu_text_append(&line, text->data, declarator_end) != 0 )
+ goto done;
+
+ if( has_equal )
+ {
+ if( mcpu_text_append_char(&line, ' ') != 0 ||
+ mcpu_text_append(&line, text->data + equal + 1,
+ text->length - equal - 1) != 0 )
+ goto done;
+ }
+ else
+ {
+ if( mcpu_text_append_ascii(&line, " 1") != 0 )
+ goto done;
+ }
+
+ mcpu_pp_lexer_init( &lexer );
+ if( mcpu_pp_prepare_line(&lexer, line.data, line.length,
+ current_language(cpp), prepared) != 0 )
+ goto done;
+
+ if( lexer.in_block_comment )
+ {
+ fprintf( stderr, "%s: %s: unterminated comment\n",
+ MCPP_OPTIONS(cpp)->progname, action->option );
+ goto done;
+ }
+
+ rc = 0;
+
+done:
+ mcpu_text_free( &line );
+ return( rc );
+}
+
+
+static int
+prepare_command_line_undef( mcpp *cpp, const mcpp_option_action *action,
+ const mcpu_text *text, mcpu_text *prepared )
+{
+ mcpu_text line;
+ mcpu_pp_lexer_state lexer;
+ size_t p;
+ int rc = -1;
+
+ mcpu_text_init( &line );
+
+ if( text->length != 0 && mcpu_pp_is_identifier_start(text->data[0]) )
+ {
+ p = 1;
+ while( p < text->length && mcpu_pp_is_identifier_char(text->data[p]) )
+ ++p;
+ }
+ else
+ p = text->length;
+
+ if( mcpu_text_append(&line, text->data, p) != 0 )
+ goto done;
+
+ mcpu_pp_lexer_init( &lexer );
+ if( mcpu_pp_prepare_line(&lexer, line.data, line.length,
+ current_language(cpp), prepared) != 0 )
+ goto done;
+
+ if( lexer.in_block_comment )
+ {
+ fprintf( stderr, "%s: %s: unterminated comment\n",
+ MCPP_OPTIONS(cpp)->progname, action->option );
+ goto done;
+ }
+
+ rc = 0;
+
+done:
+ mcpu_text_free( &line );
+ return( rc );
+}
+
+
+static int
+apply_command_line_define( mcpp *cpp, const mcpp_option_action *action )
+{
+ mcpu_text text;
+ mcpu_text prepared;
+ int rc;
+
+ mcpu_text_init( &text );
+ mcpu_text_init( &prepared );
+
+ if( command_line_text(cpp, action, &text) != 0 ||
+ prepare_command_line_define(cpp, action, &text, &prepared) != 0 )
+ goto fail;
+
+ rc = handle_define( cpp, "<command-line>", 0,
+ prepared.data, prepared.length, 0,
+ MCPU_MACRO_ORIGIN_COMMAND_LINE );
+ mcpu_text_free( &prepared );
+ mcpu_text_free( &text );
+ return( rc );
+
+fail:
+ mcpu_text_free( &prepared );
+ mcpu_text_free( &text );
+ return( -1 );
+}
+
+
+static int
+apply_command_line_undef( mcpp *cpp, const mcpp_option_action *action )
+{
+ mcpu_text text;
+ mcpu_text prepared;
+ int rc;
+
+ mcpu_text_init( &text );
+ mcpu_text_init( &prepared );
+
+ if( command_line_text(cpp, action, &text) != 0 ||
+ prepare_command_line_undef(cpp, action, &text, &prepared) != 0 )
+ {
+ mcpu_text_free( &prepared );
+ mcpu_text_free( &text );
+ return( -1 );
+ }
+
+ rc = handle_undef( cpp, "<command-line>", 0,
+ prepared.data, prepared.length, 0 );
+ mcpu_text_free( &prepared );
+ mcpu_text_free( &text );
+ return( rc );
+}
+
+
+static int
+apply_command_line_macros( mcpp *cpp )
+{
+ size_t i;
+
+ if( cpp->command_line_macros_applied || MCPP_OPTIONS(cpp) == NULL )
+ return( 0 );
+
+ for( i = 0; i < MCPP_OPTIONS(cpp)->action_count; ++i )
+ {
+ const mcpp_option_action *action = &MCPP_OPTIONS(cpp)->actions[i];
+
+ switch( action->kind )
+ {
+ case MCPP_ACTION_DEFINE:
+ if( apply_command_line_define(cpp, action) != 0 )
+ return( -1 );
+ break;
+ case MCPP_ACTION_UNDEF:
+ if( apply_command_line_undef(cpp, action) != 0 )
+ return( -1 );
+ break;
+ default:
+ break;
+ }
+ }
+
+ cpp->command_line_macros_applied = 1;
+ return( 0 );
+}
+
+
+enum mcpp_conditional_directive
+{
+ MCPP_COND_NONE = 0,
+ MCPP_COND_IF,
+ MCPP_COND_IFDEF,
+ MCPP_COND_IFNDEF,
+ MCPP_COND_ELIF,
+ MCPP_COND_ELSE,
+ MCPP_COND_ENDIF
+};
+
+static int
+diagnostic_directive( const __mpu_char16_t *name, size_t length,
+ int *is_error, const char **directive_name )
+{
+ if( mcpu_text_equal_ascii(name, length, "error") )
+ {
+ *is_error = 1;
+ *directive_name = "error";
+ return( 1 );
+ }
+
+ if( mcpu_text_equal_ascii(name, length, "warning") )
+ {
+ *is_error = 0;
+ *directive_name = "warning";
+ return( 1 );
+ }
+
+ return( 0 );
+}
+
+static int
+append_diagnostic_text( mcpu_text *message,
+ const __mpu_char16_t *line, size_t length,
+ size_t p )
+{
+ int pending_space = 0;
+ __mpu_char16_t quote = 0;
+ int escaped = 0;
+
+ skip_space16( line, length, &p );
+
+ while( p < length && line[p] != '\n' )
+ {
+ __mpu_char16_t c = line[p++];
+
+ if( quote )
+ {
+ if( mcpu_text_append_char(message, c) != 0 )
+ return( -1 );
+
+ if( escaped )
+ escaped = 0;
+ else if( c == '\\' )
+ escaped = 1;
+ else if( c == quote )
+ quote = 0;
+ continue;
+ }
+
+ if( c == ' ' || c == '\t' || c == '\f' || c == '\v' || c == '\r' )
+ {
+ if( message->length != 0 )
+ pending_space = 1;
+ continue;
+ }
+
+ if( pending_space )
+ {
+ if( mcpu_text_append_char(message, ' ') != 0 )
+ return( -1 );
+ pending_space = 0;
+ }
+
+ if( c == '"' || c == '\'' || c == '`' )
+ quote = c;
+
+ if( mcpu_text_append_char(message, c) != 0 )
+ return( -1 );
+ }
+
+ return( 0 );
+}
+
+static int
+handle_diagnostic( mcpp *cpp, const char *filename, unsigned line_number,
+ const __mpu_char16_t *line, size_t length, size_t p,
+ int is_error, const char *directive_name )
+{
+ mcpu_text message;
+ char *utf8;
+
+ mcpu_text_init( &message );
+ if( append_diagnostic_text(&message, line, length, p) != 0 )
+ {
+ mcpu_text_free( &message );
+ return( -1 );
+ }
+
+ utf8 = mcpu_text_to_utf8( message.data, message.length );
+ mcpu_text_free( &message );
+ if( utf8 == NULL )
+ return( -1 );
+
+ if( is_error )
+ {
+ fprintf( stderr, "%s:%u: error: #%s%s%s\n",
+ filename, line_number, directive_name,
+ *utf8 ? " " : "", utf8 );
+ free( utf8 );
+ return( -1 );
+ }
+
+ {
+ int rc = mcpp_diagnostic_warning(MCPP_OPTIONS(cpp), filename, line_number,
+ "#%s%s%s", directive_name,
+ *utf8 ? " " : "", utf8);
+ free( utf8 );
+ return( rc );
+ }
+}
+
+static enum mcpp_conditional_directive
+conditional_directive( const __mpu_char16_t *name, size_t length )
+{
+ if( mcpu_text_equal_ascii(name, length, "if") )
+ return( MCPP_COND_IF );
+
+ if( mcpu_text_equal_ascii(name, length, "ifdef") )
+ return( MCPP_COND_IFDEF );
+
+ if( mcpu_text_equal_ascii(name, length, "ifndef") )
+ return( MCPP_COND_IFNDEF );
+
+ if( mcpu_text_equal_ascii(name, length, "elif") )
+ return( MCPP_COND_ELIF );
+
+ if( mcpu_text_equal_ascii(name, length, "else") )
+ return( MCPP_COND_ELSE );
+
+ if( mcpu_text_equal_ascii(name, length, "endif") )
+ return( MCPP_COND_ENDIF );
+
+ return( MCPP_COND_NONE );
+}
+
+static int
+conditional_active( const mcpp *cpp )
+{
+ if( cpp->conditional_depth == 0 )
+ return( 1 );
+
+ return( cpp->conditional_stack[cpp->conditional_depth - 1].active );
+}
+
+static int
+conditional_push( mcpp *cpp, const char *filename, unsigned line_number,
+ int parent_active, int condition )
+{
+ mcpp_conditional *frame;
+
+ if( cpp->conditional_depth >= MCPU_CPP_CONDITIONAL_STACK_SIZE )
+ {
+ fprintf( stderr, "%s:%u: error: conditional nesting exceeds %d levels\n",
+ filename, line_number, MCPU_CPP_CONDITIONAL_STACK_SIZE );
+ return( -1 );
+ }
+
+ frame = &cpp->conditional_stack[cpp->conditional_depth++];
+ frame->filename = filename;
+ frame->line_number = line_number;
+ frame->parent_active = parent_active;
+ frame->active = parent_active && condition;
+ frame->branch_succeeded = frame->active;
+ frame->seen_else = 0;
+
+ return( 0 );
+}
+
+static int
+conditional_identifier( mcpp *cpp, const char *filename,
+ unsigned line_number,
+ const __mpu_char16_t *line, size_t length,
+ size_t p, int negate, int parent_active )
+{
+ size_t start;
+ size_t end;
+ int defined;
+
+ if( !parent_active )
+ return( conditional_push(cpp, filename, line_number, 0, 0) );
+
+ skip_space16( line, length, &p );
+ if( p >= length || !mcpu_pp_is_identifier_start(line[p]) )
+ {
+ fprintf( stderr, "%s:%u: error: identifier expected after #%s\n",
+ filename, line_number, negate ? "ifndef" : "ifdef" );
+ return( -1 );
+ }
+
+ start = p++;
+ while( p < length && mcpu_pp_is_identifier_char(line[p]) )
+ ++p;
+ end = p;
+
+ skip_space16( line, length, &p );
+ if( p < length && line[p] != '\n' )
+ {
+ fprintf( stderr, "%s:%u: error: extra tokens after #%s\n",
+ filename, line_number, negate ? "ifndef" : "ifdef" );
+ return( -1 );
+ }
+
+ defined = mcpu_macro_find( &cpp->macros, line + start, end - start ) != NULL;
+ if( negate ) defined = !defined;
+
+ return( conditional_push(cpp, filename, line_number, 1, defined) );
+}
+
+static int
+conditional_if( mcpp *cpp, const char *filename, unsigned line_number,
+ const __mpu_char16_t *line, size_t length, size_t p,
+ int parent_active )
+{
+ mcpu_macro_expansion_context context;
+ int value = 0;
+
+ if( parent_active )
+ {
+ macro_context( cpp, filename, line_number, &context );
+ if( mcpp_eval_if_expression(&cpp->macros, line + p, length - p,
+ current_language(cpp), &context,
+ filename, line_number, &value) != 0 )
+ return( -1 );
+ }
+
+ return( conditional_push(cpp, filename, line_number,
+ parent_active, value) );
+}
+
+static int
+conditional_elif( mcpp *cpp, const char *filename, unsigned line_number,
+ const __mpu_char16_t *line, size_t length, size_t p,
+ size_t conditional_base )
+{
+ mcpp_conditional *frame;
+ mcpu_macro_expansion_context context;
+ int value = 0;
+
+ if( cpp->conditional_depth <= conditional_base )
+ {
+ fprintf( stderr, "%s:%u: error: '#elif' not within a conditional\n",
+ filename, line_number );
+ return( -1 );
+ }
+
+ frame = &cpp->conditional_stack[cpp->conditional_depth - 1];
+ if( frame->seen_else )
+ {
+ fprintf( stderr, "%s:%u: error: '#elif' after '#else'\n",
+ filename, line_number );
+ return( -1 );
+ }
+
+ if( !frame->parent_active || frame->branch_succeeded )
+ {
+ frame->active = 0;
+ return( 0 );
+ }
+
+ macro_context( cpp, filename, line_number, &context );
+ if( mcpp_eval_if_expression(&cpp->macros, line + p, length - p,
+ current_language(cpp), &context,
+ filename, line_number, &value) != 0 )
+ return( -1 );
+
+ frame->active = value != 0;
+ if( frame->active )
+ frame->branch_succeeded = 1;
+
+ return( 0 );
+}
+
+static int
+conditional_else( mcpp *cpp, const char *filename, unsigned line_number,
+ size_t conditional_base )
+{
+ mcpp_conditional *frame;
+
+ if( cpp->conditional_depth <= conditional_base )
+ {
+ fprintf( stderr, "%s:%u: error: '#else' not within a conditional\n",
+ filename, line_number );
+ return( -1 );
+ }
+
+ frame = &cpp->conditional_stack[cpp->conditional_depth - 1];
+ if( frame->seen_else )
+ {
+ fprintf( stderr, "%s:%u: error: '#else' after '#else'\n",
+ filename, line_number );
+ return( -1 );
+ }
+
+ frame->seen_else = 1;
+ frame->active = frame->parent_active && !frame->branch_succeeded;
+ if( frame->active )
+ frame->branch_succeeded = 1;
+
+ return( 0 );
+}
+
+static int
+conditional_endif( mcpp *cpp, const char *filename, unsigned line_number,
+ size_t conditional_base )
+{
+ if( cpp->conditional_depth <= conditional_base )
+ {
+ fprintf( stderr, "%s:%u: error: unbalanced '#endif'\n",
+ filename, line_number );
+ return( -1 );
+ }
+
+ --cpp->conditional_depth;
+ return( 0 );
+}
+
+static int
+handle_conditional( mcpp *cpp, enum mcpp_conditional_directive directive,
+ const char *filename, unsigned line_number,
+ const __mpu_char16_t *line, size_t length, size_t p,
+ size_t conditional_base )
+{
+ int parent_active = conditional_active( cpp );
+
+ switch( directive )
+ {
+ case MCPP_COND_IF:
+ return( conditional_if(cpp, filename, line_number,
+ line, length, p, parent_active) );
+
+ case MCPP_COND_IFDEF:
+ return( conditional_identifier(cpp, filename, line_number,
+ line, length, p, 0, parent_active) );
+
+ case MCPP_COND_IFNDEF:
+ return( conditional_identifier(cpp, filename, line_number,
+ line, length, p, 1, parent_active) );
+
+ case MCPP_COND_ELIF:
+ return( conditional_elif(cpp, filename, line_number,
+ line, length, p, conditional_base) );
+
+ case MCPP_COND_ELSE:
+ return( conditional_else(cpp, filename, line_number,
+ conditional_base) );
+
+ case MCPP_COND_ENDIF:
+ return( conditional_endif(cpp, filename, line_number,
+ conditional_base) );
+
+ default:
+ return( 0 );
+ }
+}
+
+static int
+parse_line_filename( const __mpu_char16_t *text, size_t length,
+ size_t *pos, char **filename )
+{
+ mcpu_text decoded;
+ size_t p = *pos;
+
+ *filename = NULL;
+ if( p >= length || text[p] != '"' )
+ return( 0 );
+
+ ++p;
+ mcpu_text_init( &decoded );
+
+ while( p < length && text[p] != '\n' )
+ {
+ __mpu_char16_t c = text[p++];
+
+ if( c == '"' )
+ {
+ *filename = mcpu_text_to_utf8( decoded.data, decoded.length );
+ mcpu_text_free( &decoded );
+ if( *filename == NULL )
+ return( -1 );
+ *pos = p;
+ return( 1 );
+ }
+
+ if( c == '\\' )
+ {
+ unsigned value;
+ int digit;
+ unsigned count;
+
+ if( p >= length || text[p] == '\n' )
+ goto bad;
+
+ c = text[p++];
+ switch( c )
+ {
+ case 'a': c = '\a'; break;
+ case 'b': c = '\b'; break;
+ case 'f': c = '\f'; break;
+ case 'n': c = '\n'; break;
+ case 'r': c = '\r'; break;
+ case 't': c = '\t'; break;
+ case 'v': c = '\v'; break;
+ case '\\': c = '\\'; break;
+ case '\'': c = '\''; break;
+ case '"': c = '"'; break;
+ case '?': c = '?'; break;
+
+ case 'x':
+ case 'X':
+ value = 0;
+ count = 0;
+ while( p < length )
+ {
+ __mpu_char16_t h = text[p];
+
+ if( h >= '0' && h <= '9' ) digit = h - '0';
+ else if( h >= 'a' && h <= 'f' ) digit = h - 'a' + 10;
+ else if( h >= 'A' && h <= 'F' ) digit = h - 'A' + 10;
+ else break;
+
+ if( value > 0xffffU / 16U )
+ goto bad;
+ value = value * 16U + (unsigned)digit;
+ if( value > 0xffffU )
+ goto bad;
+ ++p;
+ ++count;
+ }
+ if( count == 0 || value == 0 ||
+ (value >= 0xd800U && value <= 0xdfffU) )
+ goto bad;
+ c = (__mpu_char16_t)value;
+ break;
+
+ default:
+ if( c >= '0' && c <= '7' )
+ {
+ value = c - '0';
+ for( count = 1; count < 3 && p < length; ++count )
+ {
+ c = text[p];
+ if( c < '0' || c > '7' ) break;
+ value = value * 8U + (unsigned)(c - '0');
+ ++p;
+ }
+ if( value == 0 || value > 0xffffU ||
+ (value >= 0xd800U && value <= 0xdfffU) )
+ goto bad;
+ c = (__mpu_char16_t)value;
+ }
+ break;
+ }
+ }
+
+ if( mcpu_text_append_char(&decoded, c) != 0 )
+ goto fail;
+ }
+
+bad:
+ errno = EINVAL;
+fail:
+ mcpu_text_free( &decoded );
+ return( -1 );
+}
+
+static int
+handle_line( mcpp *cpp, const char *filename, unsigned line_number,
+ unsigned next_physical_line,
+ const __mpu_char16_t *line, size_t length, size_t p,
+ char **nominal_filename, int64_t *line_delta )
+{
+ mcpu_macro_expansion_context context;
+ mcpu_text expanded;
+ const __mpu_char16_t *text;
+ size_t text_length;
+ size_t end;
+ uintmax_t value = 0;
+ char *new_filename = NULL;
+ int parsed_filename;
+
+ macro_context( cpp, *nominal_filename, line_number, &context );
+ mcpu_text_init( &expanded );
+
+ if( mcpu_macro_expand(&cpp->macros, line + p, length - p,
+ current_language(cpp), &context,
+ &expanded) != 0 )
+ goto fail;
+
+ text = expanded.data;
+ text_length = expanded.length;
+ end = line_content_end( text, text_length );
+ p = 0;
+ skip_space16( text, end, &p );
+
+ if( p >= end || text[p] < '0' || text[p] > '9' )
+ {
+ fprintf( stderr, "%s:%u: error: invalid #line directive\n",
+ filename, line_number );
+ goto fail;
+ }
+
+ while( p < end && text[p] >= '0' && text[p] <= '9' )
+ {
+ unsigned digit = text[p++] - '0';
+
+ if( value > (UINT_MAX - digit) / 10U )
+ {
+ fprintf( stderr, "%s:%u: error: line number out of range in #line directive\n",
+ filename, line_number );
+ goto fail;
+ }
+ value = value * 10U + digit;
+ }
+
+ skip_space16( text, end, &p );
+ parsed_filename = parse_line_filename( text, end, &p, &new_filename );
+ if( parsed_filename < 0 )
+ {
+ fprintf( stderr, "%s:%u: error: invalid filename in #line directive\n",
+ filename, line_number );
+ goto fail;
+ }
+
+ skip_space16( text, end, &p );
+ if( p != end )
+ {
+ fprintf( stderr, "%s:%u: error: garbage at end of #line directive\n",
+ filename, line_number );
+ goto fail;
+ }
+
+ if( parsed_filename > 0 )
+ {
+ free( *nominal_filename );
+ *nominal_filename = new_filename;
+ new_filename = NULL;
+ }
+
+ *line_delta = (int64_t)value - (int64_t)next_physical_line;
+
+ if( append_line_marker(cpp, (unsigned)value, *nominal_filename, 0) != 0 )
+ goto fail;
+
+ mcpu_text_free( &expanded );
+ return( 0 );
+
+fail:
+ free( new_filename );
+ mcpu_text_free( &expanded );
+ return( -1 );
+}
+
+static int process_file( mcpp *cpp, const char *filename,
+ const mcpu_include_dir *include_dir,
+ int system_header );
+
+static int
+pragma_once_directive( const __mpu_char16_t *line, size_t length, size_t p )
+{
+ size_t end = line_content_end( line, length );
+
+ skip_space16( line, end, &p );
+ if( p + 4 > end ||
+ !mcpu_text_equal_ascii(line + p, 4, "once") )
+ return( 0 );
+
+ p += 4;
+ skip_space16( line, end, &p );
+ return( p == end );
+}
+
+static int
+handle_include( mcpp *cpp, const char *source_filename,
+ const char *nominal_filename,
+ unsigned line_number, unsigned resume_line_number,
+ const __mpu_char16_t *line, size_t length, size_t p,
+ int include_next, const mcpu_include_dir *current_dir,
+ int source_is_system )
+{
+ char *requested = NULL;
+ char *found = NULL;
+ const mcpu_include_dir *found_dir = NULL;
+ int quoted;
+ int rc = -1;
+ mcpu_macro_expansion_context context;
+
+ macro_context( cpp, nominal_filename, line_number, &context );
+
+ if( parse_include_filename( line, length, p, &requested, &quoted ) != 0 )
+ {
+ mcpu_text expanded;
+
+ mcpu_text_init( &expanded );
+ if( mcpu_macro_expand(&cpp->macros, line + p, length - p,
+ current_language(cpp), &context,
+ &expanded) != 0 ||
+ parse_include_filename(expanded.data, expanded.length, 0,
+ &requested, &quoted) != 0 )
+ {
+ fprintf( stderr,
+ "%s:%u: error: #%s does not expand to \"file\" or <file>\n",
+ nominal_filename, line_number,
+ include_next ? "include_next" : "include" );
+ mcpu_text_free( &expanded );
+ goto done;
+ }
+ mcpu_text_free( &expanded );
+ }
+
+ found = mcpu_include_find( &cpp->include_paths,
+ source_filename,
+ requested,
+ quoted,
+ current_language(cpp),
+ include_next,
+ include_next ? current_dir : NULL,
+ &found_dir );
+ if( found == NULL )
+ {
+ if( MCPP_OPTIONS(cpp) != NULL &&
+ MCPP_OPTIONS(cpp)->print_deps_missing_files )
+ {
+ int dependency_is_system = source_is_system || !quoted;
+
+ if( dependency_add_unresolved(cpp, requested,
+ dependency_is_system) != 0 )
+ goto done;
+
+ if( cpp->verbose )
+ fprintf( stderr, "%s:%u: %s %s -> <generated> [%s]\n",
+ nominal_filename, line_number,
+ include_next ? "include_next" : "include",
+ requested, mcpu_language_name(current_language(cpp)) );
+
+ rc = 0;
+ goto done;
+ }
+
+ fprintf( stderr, "%s:%u: error: cannot find %s file '%s'\n",
+ nominal_filename, line_number,
+ include_next ? "include_next" : "include", requested );
+ goto done;
+ }
+
+ if( cpp->verbose )
+ fprintf( stderr, "%s:%u: %s %s -> %s [%s]\n",
+ nominal_filename, line_number,
+ include_next ? "include_next" : "include",
+ requested, found, mcpu_language_name(current_language(cpp)) );
+
+ {
+ int dependency_is_system = source_is_system || include_dir_is_system(found_dir);
+
+ if( dependency_add(cpp, found, dependency_is_system) != 0 )
+ goto done;
+
+ if( once_file_seen(cpp, found) )
+ {
+ rc = 0;
+ goto done;
+ }
+
+ if( sync_output_position(cpp, line_number, nominal_filename) != 0 ||
+ append_line_marker(cpp, 1, found, 1) != 0 )
+ goto done;
+ if( process_file( cpp, found, found_dir, dependency_is_system ) != 0 )
+ goto done;
+ }
+ if( append_line_marker( cpp, resume_line_number, nominal_filename, 2 ) != 0 )
+ goto done;
+
+ rc = 0;
+
+done:
+ free( requested );
+ free( found );
+ return( rc );
+}
+
+static int
+process_source( mcpp *cpp, mcpu_source *source,
+ const mcpu_include_dir *include_dir,
+ int source_is_system )
+{
+ mcpu_pp_lexer_state lexer;
+ mcpu_text prepared;
+ mcpu_text expanded;
+ const __mpu_char16_t *line;
+ size_t length;
+ unsigned line_number;
+ unsigned next_line_number;
+ int spliced;
+ size_t conditional_base;
+ int status;
+ mcpu_macro_expansion_context context;
+ const char *filename = source->filename;
+ char *nominal_filename;
+ int64_t line_delta = 0;
+
+ if( cpp->include_depth >= MCPU_CPP_INCLUDE_STACK_SIZE )
+ {
+ fprintf( stderr, "%s: error: include nesting exceeds %d files\n",
+ filename, MCPU_CPP_INCLUDE_STACK_SIZE );
+ return( -1 );
+ }
+
+ nominal_filename = strdup( filename );
+ if( nominal_filename == NULL )
+ return( -1 );
+
+ ++cpp->include_depth;
+ conditional_base = cpp->conditional_depth;
+ mcpu_pp_lexer_init( &lexer );
+ mcpu_text_init( &prepared );
+ mcpu_text_init( &expanded );
+
+ while( (status = mcpu_source_next_line( source, &line, &length,
+ &line_number,
+ &next_line_number, &spliced )) > 0 )
+ {
+ int64_t reported = (int64_t)line_number + line_delta;
+ int64_t reported_next = (int64_t)next_line_number + line_delta;
+ unsigned logical_line;
+ unsigned logical_next_line;
+ size_t hash = 0;
+ int directive;
+
+ if( reported < 0 || reported > UINT_MAX ||
+ reported_next < 0 || reported_next > UINT_MAX )
+ {
+ fprintf( stderr, "%s:%u: error: logical line number out of range\n",
+ filename, line_number );
+ goto fail;
+ }
+
+ logical_line = (unsigned)reported;
+ logical_next_line = (unsigned)reported_next;
+
+ mcpu_pp_lexer_set_diagnostics(&lexer, MCPP_OPTIONS(cpp),
+ nominal_filename, logical_line,
+ source->splice_offsets,
+ source->splice_count);
+
+ if( mcpu_pp_prepare_line(&lexer, line, length,
+ current_language(cpp), &prepared) != 0 )
+ goto fail;
+
+ line = prepared.data;
+ length = prepared.length;
+
+ if( !mcpu_source_valid_ucs2(line, length) )
+ {
+ fprintf( stderr, "%s:%u: error: character outside UCS-2\n",
+ nominal_filename, logical_line );
+ goto fail;
+ }
+
+ directive = mcpu_pp_find_directive( line, length, &hash );
+ if( directive < 0 )
+ goto fail;
+
+ if( directive )
+ {
+ size_t name_start;
+ size_t name_length;
+ size_t after_name;
+
+ if( parse_directive_name( line, length, hash,
+ &name_start, &name_length,
+ &after_name ) )
+ {
+ enum mcpp_conditional_directive conditional;
+
+ conditional = conditional_directive( line + name_start, name_length );
+ if( conditional != MCPP_COND_NONE )
+ {
+ if( handle_conditional(cpp, conditional, source->filename,
+ line_number, line, length, after_name,
+ conditional_base) != 0 )
+ goto fail;
+ continue;
+ }
+
+ if( !conditional_active(cpp) )
+ continue;
+
+ {
+ int diagnostic_error;
+ const char *diagnostic_name;
+
+ if( diagnostic_directive(line + name_start, name_length,
+ &diagnostic_error, &diagnostic_name) )
+ {
+ if( handle_diagnostic(cpp, nominal_filename, logical_line,
+ line, length, after_name,
+ diagnostic_error, diagnostic_name) != 0 )
+ goto fail;
+
+ continue;
+ }
+ }
+
+ if( mcpu_text_equal_ascii(line + name_start, name_length, "line") )
+ {
+ if( handle_line( cpp, nominal_filename, logical_line,
+ next_line_number, line, length, after_name,
+ &nominal_filename, &line_delta ) != 0 )
+ goto fail;
+ continue;
+ }
+
+ if( mcpu_text_equal_ascii(line + name_start, name_length, "pragma") )
+ {
+ if( pragma_once_directive(line, length, after_name) )
+ {
+ if( once_file_mark(cpp, source->filename) != 0 )
+ {
+ fprintf( stderr,
+ "%s:%u: error: cannot record #pragma once: %s\n",
+ nominal_filename, logical_line, strerror(errno) );
+ goto fail;
+ }
+
+ continue;
+ }
+ }
+
+ if( mcpu_text_equal_ascii(line + name_start, name_length, "include") ||
+ mcpu_text_equal_ascii(line + name_start, name_length,
+ "include_next") )
+ {
+ int include_next =
+ mcpu_text_equal_ascii( line + name_start, name_length,
+ "include_next" );
+
+ if( handle_include( cpp, source->filename, nominal_filename,
+ logical_line, logical_next_line,
+ line, length, after_name, include_next,
+ include_dir, source_is_system ) != 0 )
+ goto fail;
+ continue;
+ }
+
+ if( mcpu_text_equal_ascii(line + name_start, name_length, "define") )
+ {
+ if( handle_define( cpp, nominal_filename, logical_line,
+ line, length, after_name,
+ MCPU_MACRO_ORIGIN_SOURCE ) != 0 )
+ goto fail;
+
+ if( MCPP_OPTIONS(cpp) != NULL &&
+ MCPP_OPTIONS(cpp)->dump_macros == MCPP_DUMP_DEFINITIONS &&
+ append_visible_text(cpp, logical_line, nominal_filename,
+ line, length) != 0 )
+ goto fail;
+
+ continue;
+ }
+
+ if( mcpu_text_equal_ascii(line + name_start, name_length, "undef") )
+ {
+ if( handle_undef( cpp, nominal_filename, logical_line,
+ line, length, after_name ) != 0 )
+ goto fail;
+ continue;
+ }
+
+ if( mcpu_text_equal_ascii(line + name_start, name_length, "lang") )
+ {
+ {
+ size_t output_start;
+
+ if( sync_output_position(cpp, logical_line, nominal_filename) != 0 )
+ goto fail;
+ output_start = cpp->output.length;
+ if( handle_lang(cpp, nominal_filename, logical_line,
+ line, length, after_name, spliced) != 0 )
+ goto fail;
+ advance_output_position( cpp, cpp->output.data + output_start,
+ cpp->output.length - output_start );
+ }
+ continue;
+ }
+
+ if( mcpu_text_equal_ascii( line + name_start,
+ name_length, "endlang" ) )
+ {
+ {
+ size_t output_start;
+
+ if( sync_output_position(cpp, logical_line, nominal_filename) != 0 )
+ goto fail;
+ output_start = cpp->output.length;
+ if( handle_endlang(cpp, nominal_filename, logical_line,
+ line, length) != 0 )
+ goto fail;
+ advance_output_position( cpp, cpp->output.data + output_start,
+ cpp->output.length - output_start );
+ }
+ continue;
+ }
+ }
+ }
+
+ if( !conditional_active(cpp) )
+ continue;
+
+ macro_context( cpp, nominal_filename, logical_line, &context );
+ if( mcpu_macro_expand(&cpp->macros, line, length,
+ current_language(cpp), &context,
+ &expanded) != 0 ||
+ append_visible_text(cpp, logical_line, nominal_filename,
+ expanded.data, expanded.length) != 0 )
+ goto fail;
+ }
+
+ if( status < 0 )
+ goto fail;
+
+ if( lexer.in_block_comment )
+ {
+ fprintf( stderr, "%s: error: unterminated comment at end of file\n",
+ nominal_filename );
+ goto fail;
+ }
+
+ if( cpp->conditional_depth != conditional_base )
+ {
+ mcpp_conditional *frame =
+ &cpp->conditional_stack[cpp->conditional_depth - 1];
+
+ fprintf( stderr, "%s:%u: error: unterminated conditional directive\n",
+ frame->filename, frame->line_number );
+ goto fail;
+ }
+
+ --cpp->include_depth;
+ mcpu_text_free( &prepared );
+ mcpu_text_free( &expanded );
+ free( nominal_filename );
+ return( 0 );
+
+fail:
+ cpp->conditional_depth = conditional_base;
+ --cpp->include_depth;
+ mcpu_text_free( &prepared );
+ mcpu_text_free( &expanded );
+ free( nominal_filename );
+ return( -1 );
+}
+
+static int
+process_file( mcpp *cpp, const char *filename,
+ const mcpu_include_dir *include_dir,
+ int system_header )
+{
+ mcpu_source source;
+ int rc;
+
+ mcpu_source_init( &source );
+ if( mcpu_source_open( &source, filename ) != 0 )
+ {
+ fprintf( stderr, "%s: %s\n", filename, strerror(errno) );
+ mcpu_source_free( &source );
+ return( -1 );
+ }
+
+ rc = process_source( cpp, &source, include_dir, system_header );
+ mcpu_source_free( &source );
+ return( rc );
+}
+
+static int
+process_stream( mcpp *cpp, FILE *stream, const char *name )
+{
+ mcpu_source source;
+ int rc;
+
+ mcpu_source_init( &source );
+ if( mcpu_source_open_stream( &source, stream, name ) != 0 )
+ {
+ fprintf( stderr, "%s: %s\n", name, strerror(errno) );
+ mcpu_source_free( &source );
+ return( -1 );
+ }
+
+ rc = process_source( cpp, &source, NULL, 0 );
+ mcpu_source_free( &source );
+ return( rc );
+}
+
+void
+mcpp_init( mcpp *cpp )
+{
+ if( cpp == NULL )
+ return;
+
+ cpp->options_data = NULL;
+ mcpu_config_init( &cpp->config );
+ mcpp_runtime_paths_init( &cpp->runtime_paths );
+ mcpu_include_paths_init( &cpp->include_paths );
+ cpp->lang_depth = 1;
+ cpp->lang_stack[0] = MCPU_LANG_0;
+ cpp->include_depth = 0;
+ cpp->conditional_depth = 0;
+ cpp->base_filename = NULL;
+ cpp->preprocess_date[0] = 0;
+ cpp->preprocess_time[0] = 0;
+ cpp->builtins_initialized = 0;
+ cpp->command_line_macros_applied = 0;
+ mcpu_macro_table_init( &cpp->macros );
+ cpp->once_files = NULL;
+ cpp->dependencies_first = NULL;
+ cpp->dependencies_last = NULL;
+ cpp->verbose = 0;
+ cpp->output_filename = NULL;
+ cpp->output_line = 0;
+ mcpu_text_init( &cpp->output );
+}
+
+void
+mcpp_free( mcpp *cpp )
+{
+ if( cpp == NULL )
+ return;
+
+ mcpu_config_free( &cpp->config );
+ mcpp_runtime_paths_free( &cpp->runtime_paths );
+ mcpu_include_paths_free( &cpp->include_paths );
+ free( cpp->base_filename );
+ cpp->base_filename = NULL;
+ mcpu_macro_table_free( &cpp->macros );
+ once_files_free( cpp );
+ dependencies_free( cpp );
+ free( cpp->output_filename );
+ cpp->output_filename = NULL;
+ cpp->output_line = 0;
+ mcpu_text_free( &cpp->output );
+}
+
+int
+mcpp_load_config( mcpp *cpp, const char *explicit_filename,
+ int no_config )
+{
+ const char *home;
+
+ if( cpp == NULL )
+ return( -1 );
+
+ /*
+ * The physical MCPU runtime root is not a configuration-file value. The
+ * sibling <root>/include directory is the lowest-priority default and
+ * therefore remains active even when --no-config suppresses all
+ * configuration-file reads. A subsequently read configuration file may
+ * replace it, including with an empty value that disables the configured
+ * system include tree.
+ */
+ if( mcpu_config_set( &cpp->config,
+ "MCPU_CPP_SYSTEM_INCLUDE_PATH",
+ cpp->runtime_paths.system_include_path ) != 0 )
+ return( -1 );
+
+ if( no_config )
+ return( 0 );
+
+ if( explicit_filename )
+ {
+ if( mcpu_config_read( &cpp->config, explicit_filename, 1 ) != 0 )
+ {
+ fprintf( stderr, "%s: cannot read configuration: %s\n",
+ explicit_filename, strerror(errno) );
+ return( -1 );
+ }
+ }
+ else
+ {
+ char *user = NULL;
+
+ if( cpp->runtime_paths.config_file == NULL ||
+ mcpu_config_read( &cpp->config,
+ cpp->runtime_paths.config_file, 0 ) != 0 )
+ return( -1 );
+
+ if( mcpu_config_read( &cpp->config,
+ MCPU_CPP_SYSTEM_CONFIG_FILE, 0 ) != 0 )
+ return( -1 );
+
+ home = getenv( "HOME" );
+ if( home && *home )
+ {
+ size_t n = strlen(home) + sizeof("/.mcpu/mcpu-cpp.conf");
+ user = (char *)malloc( n );
+ if( user == NULL )
+ return( -1 );
+ snprintf( user, n, "%s/.mcpu/mcpu-cpp.conf", home );
+ if( mcpu_config_read( &cpp->config, user, 0 ) != 0 )
+ {
+ free( user );
+ return( -1 );
+ }
+ free( user );
+ }
+ }
+
+ return( 0 );
+}
+
+
+int
+mcpp_prepare( mcpp *cpp )
+{
+ if( cpp == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ if( cpp->preprocess_date[0] == 0 && initialize_timestamp(cpp) != 0 )
+ return( -1 );
+
+ if( initialize_builtins(cpp) != 0 )
+ return( -1 );
+
+ return( apply_command_line_macros(cpp) );
+}
+
+
+static void
+restore_output_length( mcpp *cpp, size_t length )
+{
+ if( cpp == NULL )
+ return;
+
+ cpp->output.length = length;
+ if( cpp->output.data != NULL )
+ cpp->output.data[length] = 0;
+}
+
+
+static int
+process_forced_file( mcpp *cpp, const mcpp_option_action *action,
+ int discard_output )
+{
+ const mcpu_include_dir *found_dir = NULL;
+ char *found;
+ char *saved_output_filename = NULL;
+ size_t output_length;
+ unsigned saved_output_line = 0;
+ int system_header;
+ int rc = -1;
+
+ if( cpp == NULL || action == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ /*
+ * GNU-style command-line forced files are resolved relative to the current
+ * working directory first. Passing "." as the synthetic containing source
+ * gives mcpu_include_find() exactly that quoted-include first step, followed
+ * by the ordinary configured include chain. The primary input directory is
+ * deliberately not involved.
+ */
+ found = mcpu_include_find( &cpp->include_paths, ".", action->argument, 1,
+ current_language(cpp), 0, NULL, &found_dir );
+ if( found == NULL )
+ {
+ const char *progname = MCPP_OPTIONS(cpp) != NULL ?
+ MCPP_OPTIONS(cpp)->progname : "mcpu-cpp";
+
+ if( MCPP_OPTIONS(cpp) != NULL &&
+ MCPP_OPTIONS(cpp)->print_deps_missing_files )
+ {
+ /*
+ * GNU -MG also tolerates missing command-line -include/-imacros files.
+ * With no successful search there is no physical search provenance, so
+ * the command-line operand enters the unresolved registry as a user
+ * dependency exactly as written.
+ */
+ if( dependency_add_unresolved(cpp, action->argument, 0) != 0 )
+ return( -1 );
+
+ if( cpp->verbose )
+ fprintf( stderr, "%s %s -> <generated> [%s]\n",
+ action->option, action->argument,
+ mcpu_language_name(current_language(cpp)) );
+ return( 0 );
+ }
+
+ fprintf( stderr, "%s: error: %s cannot find file '%s'\n",
+ progname, action->option, action->argument );
+ return( -1 );
+ }
+
+ system_header = include_dir_is_system( found_dir );
+ if( dependency_add(cpp, found, system_header) != 0 )
+ goto done;
+
+ if( once_file_seen(cpp, found) )
+ {
+ rc = 0;
+ goto done;
+ }
+
+ output_length = cpp->output.length;
+ if( discard_output && cpp->output_filename != NULL )
+ {
+ saved_output_filename = strdup( cpp->output_filename );
+ if( saved_output_filename == NULL )
+ goto done;
+ saved_output_line = cpp->output_line;
+ }
+
+ if( cpp->verbose )
+ fprintf( stderr, "%s %s -> %s [%s]\n",
+ action->option, action->argument, found,
+ mcpu_language_name(current_language(cpp)) );
+
+ if( append_line_marker(cpp, 1, found, 1) != 0 )
+ goto forced_fail;
+
+ /*
+ * A forced file is conceptually included by the primary source before its
+ * first line. Reserve the primary-source frame so __INCLUDE_LEVEL__ is 1
+ * in the forced file and increases normally in headers included from it.
+ */
+ if( cpp->include_depth >= MCPU_CPP_INCLUDE_STACK_SIZE )
+ {
+ fprintf( stderr, "%s: error: include nesting exceeds %d files\n",
+ found, MCPU_CPP_INCLUDE_STACK_SIZE );
+ goto forced_fail;
+ }
+
+ ++cpp->include_depth;
+ rc = process_file( cpp, found, found_dir, system_header );
+ --cpp->include_depth;
+ if( rc != 0 )
+ goto forced_fail;
+
+ if( discard_output )
+ {
+ restore_output_length( cpp, output_length );
+ if( saved_output_filename != NULL &&
+ set_output_position(cpp, saved_output_line, saved_output_filename) != 0 )
+ goto done;
+ }
+ else if( append_line_marker(cpp, 1, cpp->base_filename, 2) != 0 )
+ {
+ rc = -1;
+ goto done;
+ }
+
+ rc = 0;
+ goto done;
+
+forced_fail:
+ if( discard_output )
+ {
+ restore_output_length( cpp, output_length );
+ if( saved_output_filename != NULL )
+ (void)set_output_position(cpp, saved_output_line, saved_output_filename);
+ }
+ rc = -1;
+
+done:
+ free( saved_output_filename );
+ free( found );
+ return( rc );
+}
+
+
+static int
+process_forced_files( mcpp *cpp )
+{
+ size_t i;
+ int pass;
+
+ if( cpp == NULL || MCPP_OPTIONS(cpp) == NULL )
+ return( 0 );
+
+ /*
+ * Command-line -D/-U actions have already been applied by mcpp_prepare().
+ * Process every -imacros file before every -include file while preserving
+ * command-line order inside each class.
+ */
+ for( pass = 0; pass < 2; ++pass )
+ {
+ enum mcpp_option_action_kind kind =
+ pass == 0 ? MCPP_ACTION_IMACROS : MCPP_ACTION_INCLUDE;
+
+ for( i = 0; i < MCPP_OPTIONS(cpp)->action_count; ++i )
+ {
+ const mcpp_option_action *action = &MCPP_OPTIONS(cpp)->actions[i];
+
+ if( action->kind != kind )
+ continue;
+
+ if( process_forced_file(cpp, action, kind == MCPP_ACTION_IMACROS) != 0 )
+ return( -1 );
+ }
+ }
+
+ return( 0 );
+}
+
+
+int
+mcpp_dump_macros( mcpp *cpp, FILE *stream )
+{
+ if( cpp == NULL || stream == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ if( mcpp_prepare(cpp) != 0 )
+ return( -1 );
+
+ return( mcpu_macro_dump(&cpp->macros, stream,
+ MCPP_OPTIONS(cpp) != NULL &&
+ MCPP_OPTIONS(cpp)->dump_only_plus_predefined) );
+}
+
+
+int
+mcpp_process( mcpp *cpp, const char *filename )
+{
+ if( cpp == NULL || filename == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ free( cpp->base_filename );
+ cpp->base_filename = strdup( filename );
+ if( cpp->base_filename == NULL || mcpp_prepare(cpp) != 0 )
+ return( -1 );
+
+ if( MCPP_OPTIONS(cpp) != NULL &&
+ MCPP_OPTIONS(cpp)->dump_macros == MCPP_DUMP_DEFINITIONS )
+ {
+ const char *marker = "<built-in>";
+
+ if( append_line_marker(cpp, 0, filename, 0) != 0 ||
+ mcpu_macro_dump_text(&cpp->macros, &cpp->output, marker) != 0 )
+ return( -1 );
+ }
+
+ if( dependency_add(cpp, filename, 0) != 0 )
+ return( -1 );
+
+ if( append_line_marker(cpp, 1, filename, 0) != 0 )
+ return( -1 );
+
+ if( process_forced_files(cpp) != 0 )
+ return( -1 );
+
+ if( process_file( cpp, filename, NULL, 0 ) != 0 )
+ return( -1 );
+
+ if( cpp->lang_depth != 1 )
+ {
+ fprintf( stderr,
+ "%s: error: unbalanced #lang/#endlang at end of translation unit (depth %lu)\n",
+ filename, (unsigned long)cpp->lang_depth );
+ return( -1 );
+ }
+
+ return( 0 );
+}
+
+
+int
+mcpp_process_stream( mcpp *cpp, FILE *stream, const char *name )
+{
+ if( cpp == NULL || stream == NULL || name == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ free( cpp->base_filename );
+ cpp->base_filename = strdup( name );
+ if( cpp->base_filename == NULL || mcpp_prepare(cpp) != 0 )
+ return( -1 );
+
+ if( MCPP_OPTIONS(cpp) != NULL &&
+ MCPP_OPTIONS(cpp)->dump_macros == MCPP_DUMP_DEFINITIONS )
+ {
+ const char *marker = "<built-in>";
+
+ if( append_line_marker(cpp, 0, name, 0) != 0 ||
+ mcpu_macro_dump_text(&cpp->macros, &cpp->output, marker) != 0 )
+ return( -1 );
+ }
+
+ if( dependency_add(cpp, strcmp(name, "<stdin>") == 0 ? "-" : name, 0) != 0 )
+ return( -1 );
+
+ if( append_line_marker(cpp, 1, name, 0) != 0 )
+ return( -1 );
+
+ if( process_forced_files(cpp) != 0 )
+ return( -1 );
+
+ if( process_stream(cpp, stream, name) != 0 )
+ return( -1 );
+
+ if( cpp->lang_depth != 1 )
+ {
+ fprintf( stderr,
+ "%s: error: unbalanced #lang/#endlang at end of translation unit (depth %lu)\n",
+ name, (unsigned long)cpp->lang_depth );
+ return( -1 );
+ }
+
+ return( 0 );
+}
+
+static int
+write_make_escaped( FILE *stream, const char *text )
+{
+ const unsigned char *p = (const unsigned char *)text;
+
+ while( *p )
+ {
+ if( *p == '$' )
+ {
+ if( fputs("$$", stream) == EOF )
+ return( -1 );
+ }
+ else
+ {
+ if( (*p == ' ' || *p == '\t' || *p == '#' || *p == '\\') &&
+ fputc('\\', stream) == EOF )
+ return( -1 );
+ if( fputc(*p, stream) == EOF )
+ return( -1 );
+ }
+ ++p;
+ }
+
+ return( 0 );
+}
+
+
+static int
+write_make_target_quoted( FILE *stream, const char *text )
+{
+ const unsigned char *p = (const unsigned char *)text;
+ size_t slashes = 0;
+
+ while( *p )
+ {
+ unsigned char c = *p++;
+
+ if( c == '\\' )
+ {
+ if( fputc(c, stream) == EOF )
+ return( -1 );
+ ++slashes;
+ continue;
+ }
+
+ if( c == ' ' || c == '\t' )
+ {
+ size_t i;
+
+ for( i = 0; i < slashes; ++i )
+ if( fputc('\\', stream) == EOF )
+ return( -1 );
+ if( fputc('\\', stream) == EOF )
+ return( -1 );
+ }
+ else if( c == '#' )
+ {
+ if( fputc('\\', stream) == EOF )
+ return( -1 );
+ }
+ else if( c == '$' )
+ {
+ if( fputc('$', stream) == EOF )
+ return( -1 );
+ }
+
+ slashes = 0;
+ if( fputc(c, stream) == EOF )
+ return( -1 );
+ }
+
+ return( 0 );
+}
+
+
+static char *
+default_dependency_target( const mcpp *cpp )
+{
+ const char *base;
+ const char *dot;
+ const char *suffix = ".o";
+ size_t stem_length;
+ size_t suffix_length;
+ char *target;
+
+ if( cpp == NULL || cpp->base_filename == NULL )
+ return( NULL );
+
+ if( strcmp(cpp->base_filename, "<stdin>") == 0 ||
+ strcmp(cpp->base_filename, "-") == 0 )
+ return( strdup("-") );
+
+ base = strrchr( cpp->base_filename, '/' );
+ base = base == NULL ? cpp->base_filename : base + 1;
+ dot = strrchr( base, '.' );
+ stem_length = dot != NULL && dot != base ? (size_t)(dot - base) : strlen(base);
+
+ if( MCPP_OPTIONS(cpp) != NULL && MCPP_OPTIONS(cpp)->object_suffix != NULL )
+ suffix = MCPP_OPTIONS(cpp)->object_suffix;
+ suffix_length = strlen( suffix );
+
+ target = (char *)malloc( stem_length + suffix_length + 1 );
+ if( target == NULL )
+ return( NULL );
+
+ memcpy( target, base, stem_length );
+ memcpy( target + stem_length, suffix, suffix_length + 1 );
+ return( target );
+}
+
+
+static int
+write_dependency_targets( mcpp *cpp, FILE *stream )
+{
+ char *target;
+ size_t i;
+ int pass;
+ int written = 0;
+
+ if( cpp == NULL || stream == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ if( MCPP_OPTIONS(cpp) != NULL )
+ for( pass = 0; pass < 2; ++pass )
+ {
+ enum mcpp_option_action_kind kind = pass == 0 ?
+ MCPP_ACTION_DEP_TARGET : MCPP_ACTION_DEP_TARGET_QUOTED;
+
+ for( i = 0; i < MCPP_OPTIONS(cpp)->action_count; ++i )
+ {
+ const mcpp_option_action *action = &MCPP_OPTIONS(cpp)->actions[i];
+
+ if( action->kind != kind )
+ continue;
+
+ if( written && fputc(' ', stream) == EOF )
+ return( -1 );
+
+ if( kind == MCPP_ACTION_DEP_TARGET )
+ {
+ if( fputs(action->argument, stream) == EOF )
+ return( -1 );
+ }
+ else if( write_make_target_quoted(stream, action->argument) != 0 )
+ return( -1 );
+
+ written = 1;
+ }
+ }
+
+ if( written )
+ return( 0 );
+
+ target = default_dependency_target( cpp );
+ if( target == NULL )
+ return( -1 );
+
+ if( write_make_target_quoted(stream, target) != 0 )
+ {
+ free( target );
+ return( -1 );
+ }
+
+ free( target );
+ return( 0 );
+}
+
+
+int
+mcpp_write_dependencies( mcpp *cpp, FILE *stream, int include_system )
+{
+ mcpp_dependency *p;
+
+ if( cpp == NULL || stream == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ if( write_dependency_targets(cpp, stream) != 0 || fputs(":", stream) == EOF )
+ return( -1 );
+
+ for( p = cpp->dependencies_first; p != NULL; p = p->next )
+ {
+ if( !include_system && p->system_only )
+ continue;
+
+ if( fputc(' ', stream) == EOF || write_make_escaped(stream, p->path) != 0 )
+ return( -1 );
+ }
+
+ if( fputc('\n', stream) == EOF )
+ return( -1 );
+
+ return( ferror(stream) ? -1 : 0 );
+}
+
+
+int
+mcpp_write_output( mcpp *cpp, const char *filename )
+{
+ char *utf8;
+ FILE *fp = stdout;
+ int close_fp = 0;
+ size_t n;
+
+ if( cpp == NULL )
+ return( -1 );
+
+ utf8 = mcpu_text_to_utf8( cpp->output.data, cpp->output.length );
+ if( utf8 == NULL )
+ return( -1 );
+
+ if( filename )
+ {
+ fp = fopen( filename, "wb" );
+ if( fp == NULL )
+ {
+ free( utf8 );
+ return( -1 );
+ }
+ close_fp = 1;
+ }
+
+ n = strlen( utf8 );
+ if( fwrite( utf8, 1, n, fp ) != n )
+ {
+ if( close_fp )
+ fclose( fp );
+ free( utf8 );
+ return( -1 );
+ }
+
+ if( close_fp && fclose(fp) != 0 )
+ {
+ free( utf8 );
+ return( -1 );
+ }
+
+ free( utf8 );
+ return( 0 );
+}
diff --git a/src/mcpp-lib.h b/src/mcpp-lib.h
new file mode 100644
index 0000000..6974672
--- /dev/null
+++ b/src/mcpp-lib.h
@@ -0,0 +1,65 @@
+#ifndef __MCPU_CPP_CPP_H__
+#define __MCPU_CPP_CPP_H__ 1
+
+#include <defs.h>
+#include <mcpp-config-file.h>
+#include <mcpp-include-path.h>
+#include <mcpp-language.h>
+#include <mcpp-macro.h>
+#include <mcpp-text.h>
+
+typedef struct mcpp_conditional mcpp_conditional;
+struct mcpp_conditional
+{
+ const char *filename;
+ unsigned line_number;
+ int parent_active;
+ int active;
+ int branch_succeeded;
+ int seen_else;
+};
+
+typedef struct mcpp_once_file mcpp_once_file;
+typedef struct mcpp_dependency mcpp_dependency;
+
+typedef struct mcpp mcpp;
+struct mcpp
+{
+ mcpp_options *options_data;
+ mcpu_config config;
+ mcpp_runtime_paths runtime_paths;
+ mcpu_include_paths include_paths;
+ enum mcpu_language lang_stack[MCPU_CPP_LANG_STACK_SIZE];
+ size_t lang_depth;
+ size_t include_depth;
+ mcpp_conditional conditional_stack[MCPU_CPP_CONDITIONAL_STACK_SIZE];
+ size_t conditional_depth;
+ char *base_filename;
+ char preprocess_date[16];
+ char preprocess_time[16];
+ int builtins_initialized;
+ int command_line_macros_applied;
+ mcpu_macro_table macros;
+ mcpp_once_file *once_files;
+ mcpp_dependency *dependencies_first;
+ mcpp_dependency *dependencies_last;
+ int verbose;
+ char *output_filename;
+ unsigned output_line;
+ mcpu_text output;
+};
+
+void mcpp_init( mcpp *cpp );
+void mcpp_free( mcpp *cpp );
+int mcpp_load_config( mcpp *cpp, const char *explicit_filename,
+ int no_config );
+int mcpp_prepare( mcpp *cpp );
+int mcpp_process( mcpp *cpp, const char *filename );
+int mcpp_process_stream( mcpp *cpp, FILE *stream, const char *name );
+int mcpp_dump_macros( mcpp *cpp, FILE *stream );
+int mcpp_write_output( mcpp *cpp, const char *filename );
+int mcpp_write_dependencies( mcpp *cpp, FILE *stream, int include_system );
+
+#define MCPP_OPTIONS(_cpp_) ((_cpp_)->options_data)
+
+#endif /* __MCPU_CPP_CPP_H__ */
diff --git a/src/mcpp-macro.c b/src/mcpp-macro.c
new file mode 100644
index 0000000..1837548
--- /dev/null
+++ b/src/mcpp-macro.c
@@ -0,0 +1,2878 @@
+#include <defs.h>
+
+
+typedef struct mcpu_macro_arg mcpu_macro_arg;
+struct mcpu_macro_arg
+{
+ const __mpu_char16_t *raw;
+ size_t raw_length;
+ mcpu_text expanded;
+ int expanded_ready;
+};
+
+
+static unsigned
+macro_hash( const __mpu_char16_t *name, size_t length )
+{
+ unsigned h = 5381U;
+ size_t i;
+
+ for( i = 0; i < length; ++i )
+ h = ((h << 5) + h) ^ (unsigned)name[i];
+
+ return( h % MCPU_CPP_MACRO_BUCKETS );
+}
+
+
+static __mpu_char16_t *
+dup_text( const __mpu_char16_t *text, size_t length )
+{
+ __mpu_char16_t *copy;
+
+ copy = (__mpu_char16_t *)calloc( length + 1, sizeof(__mpu_char16_t) );
+ if( copy == NULL )
+ return( NULL );
+
+ if( length != 0 )
+ memcpy( copy, text, length * sizeof(__mpu_char16_t) );
+
+ return( copy );
+}
+
+
+static void
+free_argnames( __mpu_char16_t **argnames, size_t *argname_lengths,
+ int nargs )
+{
+ int i;
+
+ if( argnames )
+ {
+ for( i = 0; i < nargs; ++i )
+ free( argnames[i] );
+ }
+
+ free( argnames );
+ free( argname_lengths );
+}
+
+
+static void
+free_macro_contents( mcpu_macro *macro )
+{
+ if( macro == NULL )
+ return;
+
+ free( macro->name );
+ free_argnames( macro->argnames, macro->argname_lengths, macro->nargs );
+ free( macro->replacement );
+}
+
+
+void
+mcpu_macro_table_init( mcpu_macro_table *table )
+{
+ if( table )
+ memset( table, 0, sizeof(*table) );
+}
+
+
+void
+mcpu_macro_table_free( mcpu_macro_table *table )
+{
+ unsigned i;
+
+ if( table == NULL )
+ return;
+
+ for( i = 0; i < MCPU_CPP_MACRO_BUCKETS; ++i )
+ {
+ mcpu_macro *macro = table->bucket[i];
+
+ while( macro )
+ {
+ mcpu_macro *next = macro->next;
+ free_macro_contents( macro );
+ free( macro );
+ macro = next;
+ }
+
+ table->bucket[i] = NULL;
+ }
+}
+
+
+mcpu_macro *
+mcpu_macro_find( mcpu_macro_table *table,
+ const __mpu_char16_t *name, size_t name_length )
+{
+ mcpu_macro *macro;
+ unsigned h;
+
+ if( table == NULL || name == NULL || name_length == 0 )
+ return( NULL );
+
+ h = macro_hash( name, name_length );
+ for( macro = table->bucket[h]; macro; macro = macro->next )
+ {
+ if( macro->name_length == name_length &&
+ memcmp( macro->name, name,
+ name_length * sizeof(__mpu_char16_t) ) == 0 )
+ return( macro );
+ }
+
+ return( NULL );
+}
+
+
+static int
+copy_argnames( const __mpu_char16_t *const *argnames,
+ const size_t *argname_lengths, size_t nargs,
+ __mpu_char16_t ***copy_names, size_t **copy_lengths )
+{
+ __mpu_char16_t **names = NULL;
+ size_t *lengths = NULL;
+ size_t i;
+
+ *copy_names = NULL;
+ *copy_lengths = NULL;
+
+ if( nargs == 0 )
+ return( 0 );
+
+ names = (__mpu_char16_t **)calloc( nargs, sizeof(*names) );
+ lengths = (size_t *)calloc( nargs, sizeof(*lengths) );
+ if( names == NULL || lengths == NULL )
+ goto fail;
+
+ for( i = 0; i < nargs; ++i )
+ {
+ names[i] = dup_text( argnames[i], argname_lengths[i] );
+ if( names[i] == NULL )
+ goto fail;
+ lengths[i] = argname_lengths[i];
+ }
+
+ *copy_names = names;
+ *copy_lengths = lengths;
+ return( 0 );
+
+fail:
+ if( names )
+ {
+ for( i = 0; i < nargs; ++i )
+ free( names[i] );
+ }
+ free( names );
+ free( lengths );
+ return( -1 );
+}
+
+
+static int
+macro_install( mcpu_macro_table *table,
+ const __mpu_char16_t *name, size_t name_length,
+ int nargs, int variadic,
+ const __mpu_char16_t *const *argnames,
+ const size_t *argname_lengths,
+ const __mpu_char16_t *replacement,
+ size_t replacement_length,
+ enum mcpu_macro_builtin builtin,
+ unsigned builtin_bits,
+ enum mcpu_macro_origin origin )
+{
+ mcpu_macro *macro;
+ __mpu_char16_t *new_replacement;
+ __mpu_char16_t **new_argnames = NULL;
+ size_t *new_argname_lengths = NULL;
+ unsigned h;
+
+ if( table == NULL || name == NULL || name_length == 0 ||
+ (replacement == NULL && replacement_length != 0) || nargs < -1 ||
+ (variadic && nargs < 0) ||
+ (nargs > 0 && (argnames == NULL || argname_lengths == NULL)) )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ new_replacement = dup_text( replacement, replacement_length );
+ if( new_replacement == NULL )
+ return( -1 );
+
+ if( nargs >= 0 &&
+ copy_argnames(argnames, argname_lengths, (size_t)nargs,
+ &new_argnames, &new_argname_lengths) != 0 )
+ {
+ free( new_replacement );
+ return( -1 );
+ }
+
+ macro = mcpu_macro_find( table, name, name_length );
+ if( macro )
+ {
+ free_argnames( macro->argnames, macro->argname_lengths, macro->nargs );
+ free( macro->replacement );
+ macro->nargs = nargs;
+ macro->variadic = variadic;
+ macro->argnames = new_argnames;
+ macro->argname_lengths = new_argname_lengths;
+ macro->replacement = new_replacement;
+ macro->replacement_length = replacement_length;
+ macro->builtin = builtin;
+ macro->builtin_bits = builtin_bits;
+ macro->origin = origin;
+ return( 0 );
+ }
+
+ macro = (mcpu_macro *)calloc( 1, sizeof(*macro) );
+ if( macro == NULL )
+ {
+ free_argnames( new_argnames, new_argname_lengths, nargs );
+ free( new_replacement );
+ return( -1 );
+ }
+
+ macro->name = dup_text( name, name_length );
+ if( macro->name == NULL )
+ {
+ free_argnames( new_argnames, new_argname_lengths, nargs );
+ free( new_replacement );
+ free( macro );
+ return( -1 );
+ }
+
+ macro->name_length = name_length;
+ macro->nargs = nargs;
+ macro->variadic = variadic;
+ macro->argnames = new_argnames;
+ macro->argname_lengths = new_argname_lengths;
+ macro->replacement = new_replacement;
+ macro->replacement_length = replacement_length;
+ macro->builtin = builtin;
+ macro->builtin_bits = builtin_bits;
+ macro->origin = origin;
+
+ h = macro_hash( name, name_length );
+ macro->next = table->bucket[h];
+ table->bucket[h] = macro;
+
+ return( 0 );
+}
+
+
+int
+mcpu_macro_define_object( mcpu_macro_table *table,
+ const __mpu_char16_t *name,
+ size_t name_length,
+ const __mpu_char16_t *replacement,
+ size_t replacement_length,
+ enum mcpu_macro_origin origin )
+{
+ return( macro_install(table, name, name_length, -1, 0, NULL, NULL,
+ replacement, replacement_length,
+ MCPU_MACRO_BUILTIN_NONE, 0, origin) );
+}
+
+
+int
+mcpu_macro_define_builtin( mcpu_macro_table *table,
+ const __mpu_char16_t *name,
+ size_t name_length,
+ enum mcpu_macro_builtin builtin )
+{
+ if( builtin == MCPU_MACRO_BUILTIN_NONE )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ return( macro_install(table, name, name_length, -1, 0, NULL, NULL,
+ NULL, 0, builtin, 0, MCPU_MACRO_ORIGIN_BUILTIN) );
+}
+
+
+int
+mcpu_macro_define_sized_builtin( mcpu_macro_table *table,
+ const __mpu_char16_t *name,
+ size_t name_length,
+ enum mcpu_macro_builtin builtin,
+ unsigned bits )
+{
+ if( builtin < MCPU_MACRO_BUILTIN_REAL_MAX || bits == 0 )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ return( macro_install(table, name, name_length, -1, 0, NULL, NULL,
+ NULL, 0, builtin, bits, MCPU_MACRO_ORIGIN_BUILTIN) );
+}
+
+
+int
+mcpu_macro_define_function( mcpu_macro_table *table,
+ const __mpu_char16_t *name,
+ size_t name_length,
+ const __mpu_char16_t *const *argnames,
+ const size_t *argname_lengths,
+ size_t nargs, int variadic,
+ const __mpu_char16_t *replacement,
+ size_t replacement_length,
+ enum mcpu_macro_origin origin )
+{
+ if( nargs > (size_t)INT_MAX )
+ {
+ errno = EOVERFLOW;
+ return( -1 );
+ }
+
+ return( macro_install(table, name, name_length, (int)nargs, variadic,
+ argnames, argname_lengths,
+ replacement, replacement_length,
+ MCPU_MACRO_BUILTIN_NONE, 0, origin) );
+}
+
+
+int
+mcpu_macro_definition_equal( const mcpu_macro *macro,
+ int nargs, int variadic,
+ const __mpu_char16_t *const *argnames,
+ const size_t *argname_lengths,
+ const __mpu_char16_t *replacement,
+ size_t replacement_length )
+{
+ int i;
+
+ if( macro == NULL || macro->builtin != MCPU_MACRO_BUILTIN_NONE ||
+ macro->nargs != nargs || macro->variadic != variadic ||
+ macro->replacement_length != replacement_length )
+ return( 0 );
+
+ if( replacement_length != 0 &&
+ memcmp( macro->replacement, replacement,
+ replacement_length * sizeof(__mpu_char16_t) ) != 0 )
+ return( 0 );
+
+ if( nargs < 0 )
+ return( 1 );
+
+ for( i = 0; i < nargs; ++i )
+ {
+ if( macro->argname_lengths[i] != argname_lengths[i] ||
+ memcmp( macro->argnames[i], argnames[i],
+ argname_lengths[i] * sizeof(__mpu_char16_t) ) != 0 )
+ return( 0 );
+ }
+
+ return( 1 );
+}
+
+
+int
+mcpu_macro_undef( mcpu_macro_table *table,
+ const __mpu_char16_t *name, size_t name_length )
+{
+ mcpu_macro **link;
+ unsigned h;
+
+ if( table == NULL || name == NULL || name_length == 0 )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ h = macro_hash( name, name_length );
+ link = &table->bucket[h];
+
+ while( *link )
+ {
+ mcpu_macro *macro = *link;
+
+ if( macro->name_length == name_length &&
+ memcmp( macro->name, name,
+ name_length * sizeof(__mpu_char16_t) ) == 0 )
+ {
+ *link = macro->next;
+ free_macro_contents( macro );
+ free( macro );
+ return( 1 );
+ }
+
+ link = &macro->next;
+ }
+
+ return( 0 );
+}
+
+
+static int
+is_quote16( __mpu_char16_t c, enum mcpu_language language )
+{
+ if( c == '"' || c == '`' )
+ return( 1 );
+
+ if( c == '\'' && language != MCPU_LANG_DIFF )
+ return( 1 );
+
+ return( 0 );
+}
+
+
+static int
+is_space16( __mpu_char16_t c )
+{
+ return( c == ' ' || c == '\t' || c == '\f' || c == '\v' ||
+ c == '\r' || c == '\n' );
+}
+
+
+static int
+macro_argument_index( const mcpu_macro *macro,
+ const __mpu_char16_t *name, size_t name_length )
+{
+ int i;
+
+ for( i = 0; i < macro->nargs; ++i )
+ {
+ if( macro->argname_lengths[i] == name_length &&
+ memcmp( macro->argnames[i], name,
+ name_length * sizeof(__mpu_char16_t) ) == 0 )
+ return( i );
+ }
+
+ if( macro->variadic &&
+ mcpu_text_equal_ascii(name, name_length, "__VA_ARGS__") )
+ return( macro->nargs );
+
+ return( -1 );
+}
+
+
+static void
+macro_args_free( mcpu_macro_arg *args, size_t nargs )
+{
+ size_t i;
+
+ if( args == NULL )
+ return;
+
+ for( i = 0; i < nargs; ++i )
+ mcpu_text_free( &args[i].expanded );
+
+ free( args );
+}
+
+
+static int
+append_macro_arg( mcpu_macro_arg **args, size_t *nargs, size_t *capacity,
+ const __mpu_char16_t *raw, size_t raw_length )
+{
+ mcpu_macro_arg *new_args;
+ size_t new_capacity;
+
+ if( *nargs == *capacity )
+ {
+ new_capacity = *capacity ? *capacity * 2 : 4;
+ if( new_capacity < *capacity ||
+ new_capacity > SIZE_MAX / sizeof(**args) )
+ {
+ errno = EOVERFLOW;
+ return( -1 );
+ }
+
+ new_args = (mcpu_macro_arg *)realloc( *args,
+ new_capacity * sizeof(**args) );
+ if( new_args == NULL )
+ return( -1 );
+
+ *args = new_args;
+ *capacity = new_capacity;
+ }
+
+ (*args)[*nargs].raw = raw;
+ (*args)[*nargs].raw_length = raw_length;
+ mcpu_text_init( &(*args)[*nargs].expanded );
+ (*args)[*nargs].expanded_ready = 0;
+ ++*nargs;
+
+ return( 0 );
+}
+
+
+static int
+parse_macro_args( const __mpu_char16_t *input, size_t length,
+ size_t open_paren, enum mcpu_language language,
+ mcpu_macro_arg **args_out, size_t *nargs_out,
+ size_t *after_call )
+{
+ mcpu_macro_arg *args = NULL;
+ size_t nargs = 0;
+ size_t capacity = 0;
+ size_t p = open_paren + 1;
+ size_t start = p;
+ int depth = 0;
+ __mpu_char16_t quote = 0;
+ int escaped = 0;
+
+ while( p < length )
+ {
+ __mpu_char16_t c = input[p];
+
+ if( quote )
+ {
+ if( escaped )
+ escaped = 0;
+ else if( c == '\\' )
+ escaped = 1;
+ else if( c == quote )
+ quote = 0;
+ else if( c == '\n' && quote == '\'' )
+ quote = 0;
+
+ ++p;
+ continue;
+ }
+
+ if( is_quote16(c, language) )
+ {
+ quote = c;
+ ++p;
+ continue;
+ }
+
+ if( c == '(' )
+ {
+ ++depth;
+ ++p;
+ continue;
+ }
+
+ if( c == ')' )
+ {
+ if( depth != 0 )
+ {
+ --depth;
+ ++p;
+ continue;
+ }
+
+ if( append_macro_arg(&args, &nargs, &capacity,
+ input + start, p - start) != 0 )
+ goto fail;
+
+ *args_out = args;
+ *nargs_out = nargs;
+ *after_call = p + 1;
+ return( 0 );
+ }
+
+ if( c == ',' && depth == 0 )
+ {
+ if( append_macro_arg(&args, &nargs, &capacity,
+ input + start, p - start) != 0 )
+ goto fail;
+ start = ++p;
+ continue;
+ }
+
+ ++p;
+ }
+
+ errno = EINVAL;
+
+fail:
+ macro_args_free( args, nargs );
+ return( -1 );
+}
+
+
+static int
+prepare_variadic_args( const mcpu_macro *macro,
+ const __mpu_char16_t *input, size_t after_call,
+ mcpu_macro_arg **args_io, size_t *nargs_io )
+{
+ mcpu_macro_arg *args = *args_io;
+ size_t nargs = *nargs_io;
+ size_t fixed = (size_t)macro->nargs;
+ size_t i;
+
+ if( !macro->variadic || nargs < fixed )
+ return( 0 );
+
+ if( nargs == fixed )
+ {
+ mcpu_macro_arg *new_args;
+
+ new_args = (mcpu_macro_arg *)realloc( args,
+ (fixed + 1) * sizeof(*args) );
+ if( new_args == NULL )
+ return( -1 );
+
+ args = new_args;
+ args[fixed].raw = input + after_call - 1;
+ args[fixed].raw_length = 0;
+ mcpu_text_init( &args[fixed].expanded );
+ args[fixed].expanded_ready = 0;
+ nargs = fixed + 1;
+ }
+ else
+ {
+ const __mpu_char16_t *begin = args[fixed].raw;
+ const __mpu_char16_t *end = args[nargs - 1].raw +
+ args[nargs - 1].raw_length;
+
+ for( i = fixed + 1; i < nargs; ++i )
+ mcpu_text_free( &args[i].expanded );
+
+ args[fixed].raw = begin;
+ args[fixed].raw_length = (size_t)(end - begin);
+ nargs = fixed + 1;
+ }
+
+ *args_io = args;
+ *nargs_io = nargs;
+ return( 0 );
+}
+
+
+static int expand_range( mcpu_macro_table *table,
+ const __mpu_char16_t *input, size_t length,
+ enum mcpu_language language,
+ const mcpu_macro_expansion_context *context,
+ mcpu_text *output, unsigned depth );
+
+
+static int
+append_quoted_utf8( mcpu_text *output, const char *s )
+{
+ mcpu_text text;
+ size_t i;
+ int rc = -1;
+
+ if( output == NULL || s == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ mcpu_text_init( &text );
+ if( mcpu_text_from_utf8(&text, s) != 0 ||
+ mcpu_text_append_char(output, '"') != 0 )
+ goto done;
+
+ for( i = 0; i < text.length; ++i )
+ {
+ switch( text.data[i] )
+ {
+ case '\\':
+ case '"':
+ if( mcpu_text_append_char(output, '\\') != 0 ||
+ mcpu_text_append_char(output, text.data[i]) != 0 )
+ goto done;
+ break;
+
+ case '\n':
+ if( mcpu_text_append_ascii(output, "\\n") != 0 )
+ goto done;
+ break;
+
+ case '\r':
+ if( mcpu_text_append_ascii(output, "\\r") != 0 )
+ goto done;
+ break;
+
+ case '\t':
+ if( mcpu_text_append_ascii(output, "\\t") != 0 )
+ goto done;
+ break;
+
+ default:
+ if( mcpu_text_append_char(output, text.data[i]) != 0 )
+ goto done;
+ break;
+ }
+ }
+
+ if( mcpu_text_append_char(output, '"') != 0 )
+ goto done;
+
+ rc = 0;
+
+done:
+ mcpu_text_free( &text );
+ return( rc );
+}
+
+
+static int
+append_unsigned( mcpu_text *output, size_t value )
+{
+ char buf[64];
+
+ snprintf( buf, sizeof(buf), "%lu", (unsigned long)value );
+ return( mcpu_text_append_ascii(output, buf) );
+}
+
+
+static int
+expand_builtin( const mcpu_macro *macro,
+ const mcpu_macro_expansion_context *context,
+ mcpu_text *output )
+{
+ if( macro == NULL || output == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ if( macro->builtin >= MCPU_MACRO_BUILTIN_FILE &&
+ macro->builtin <= MCPU_MACRO_BUILTIN_INCLUDE_LEVEL && context == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ switch( macro->builtin )
+ {
+ case MCPU_MACRO_BUILTIN_FILE:
+ return( append_quoted_utf8(output, context->filename ?
+ context->filename : "") );
+
+ case MCPU_MACRO_BUILTIN_LINE:
+ return( append_unsigned(output, context->line_number) );
+
+ case MCPU_MACRO_BUILTIN_DATE:
+ return( append_quoted_utf8(output, context->date ? context->date : "") );
+
+ case MCPU_MACRO_BUILTIN_TIME:
+ return( append_quoted_utf8(output, context->time ? context->time : "") );
+
+ case MCPU_MACRO_BUILTIN_BASE_FILE:
+ return( append_quoted_utf8(output, context->base_filename ?
+ context->base_filename : "") );
+
+ case MCPU_MACRO_BUILTIN_INCLUDE_LEVEL:
+ return( append_unsigned(output, context->include_level) );
+
+ case MCPU_MACRO_BUILTIN_REAL_MAX:
+ case MCPU_MACRO_BUILTIN_REAL_MIN:
+ case MCPU_MACRO_BUILTIN_REAL_EPSILON:
+ case MCPU_MACRO_BUILTIN_REAL_MAX_EXP:
+ case MCPU_MACRO_BUILTIN_REAL_MIN_EXP:
+ case MCPU_MACRO_BUILTIN_REAL_MAX_10_EXP:
+ case MCPU_MACRO_BUILTIN_REAL_MIN_10_EXP:
+ return( mcpu_predefined_expand_builtin(macro->builtin,
+ macro->builtin_bits,
+ output) );
+
+ case MCPU_MACRO_BUILTIN_NONE:
+ default:
+ errno = EINVAL;
+ return( -1 );
+ }
+}
+
+
+static int
+ensure_arg_expanded( mcpu_macro_table *table, mcpu_macro_arg *arg,
+ enum mcpu_language language,
+ const mcpu_macro_expansion_context *context,
+ unsigned depth )
+{
+ const __mpu_char16_t *raw;
+ size_t raw_length;
+
+ if( arg->expanded_ready )
+ return( 0 );
+
+ raw = arg->raw;
+ raw_length = arg->raw_length;
+
+ while( raw_length != 0 && is_space16(*raw) )
+ {
+ ++raw;
+ --raw_length;
+ }
+ while( raw_length != 0 && is_space16(raw[raw_length - 1]) )
+ --raw_length;
+
+ if( expand_range(table, raw, raw_length,
+ language, context,
+ &arg->expanded, depth + 1) != 0 )
+ return( -1 );
+
+ arg->expanded_ready = 1;
+ return( 0 );
+}
+
+
+static int
+append_stringified_arg( mcpu_text *output,
+ const __mpu_char16_t *raw, size_t raw_length,
+ enum mcpu_language language )
+{
+ size_t p = 0;
+ __mpu_char16_t quote = 0;
+ int escaped = 0;
+ int pending_space = 0;
+
+ while( raw_length != 0 && is_space16(*raw) )
+ {
+ ++raw;
+ --raw_length;
+ }
+ while( raw_length != 0 && is_space16(raw[raw_length - 1]) )
+ --raw_length;
+
+ if( mcpu_text_append_char(output, '"') != 0 )
+ return( -1 );
+
+ while( p < raw_length )
+ {
+ __mpu_char16_t c = raw[p++];
+
+ if( quote == 0 && is_space16(c) )
+ {
+ pending_space = 1;
+ continue;
+ }
+
+ if( pending_space )
+ {
+ if( mcpu_text_append_char(output, ' ') != 0 )
+ return( -1 );
+ pending_space = 0;
+ }
+
+ if( c == '"' || (quote != 0 && c == '\\') )
+ {
+ if( mcpu_text_append_char(output, '\\') != 0 )
+ return( -1 );
+ }
+
+ if( mcpu_text_append_char(output, c) != 0 )
+ return( -1 );
+
+ if( quote != 0 )
+ {
+ if( escaped )
+ escaped = 0;
+ else if( c == '\\' )
+ escaped = 1;
+ else if( c == quote )
+ quote = 0;
+ }
+ else if( is_quote16(c, language) )
+ {
+ quote = c;
+ escaped = 0;
+ }
+ }
+
+ return( mcpu_text_append_char(output, '"') );
+}
+
+
+/***************************************************************
+ Token concatenation works with
+ preprocessing tokens rather than plain character concatenation.
+ Parameters adjacent to ## are inserted without prescan, empty
+ arguments act as placemarkers, pasted tokens are formed first,
+ and the resulting replacement list is rescanned.
+ ***************************************************************/
+enum mcpu_macro_piece_kind
+{
+ MCPU_MACRO_PIECE_TOKEN = 0,
+ MCPU_MACRO_PIECE_SPACE,
+ MCPU_MACRO_PIECE_PASTE,
+ MCPU_MACRO_PIECE_PLACEMARKER
+};
+
+typedef struct mcpu_macro_piece mcpu_macro_piece;
+struct mcpu_macro_piece
+{
+ enum mcpu_macro_piece_kind kind;
+ mcpu_text text;
+};
+
+typedef struct mcpu_macro_piece_list mcpu_macro_piece_list;
+struct mcpu_macro_piece_list
+{
+ mcpu_macro_piece *piece;
+ size_t count;
+ size_t capacity;
+};
+
+
+static void
+macro_piece_list_init( mcpu_macro_piece_list *list )
+{
+ if( list )
+ memset( list, 0, sizeof(*list) );
+}
+
+
+static void
+macro_piece_list_free( mcpu_macro_piece_list *list )
+{
+ size_t i;
+
+ if( list == NULL )
+ return;
+
+ for( i = 0; i < list->count; ++i )
+ mcpu_text_free( &list->piece[i].text );
+
+ free( list->piece );
+ memset( list, 0, sizeof(*list) );
+}
+
+
+static int
+macro_piece_list_reserve( mcpu_macro_piece_list *list, size_t need )
+{
+ mcpu_macro_piece *p;
+ size_t capacity;
+
+ if( need <= list->capacity )
+ return( 0 );
+
+ capacity = list->capacity ? list->capacity : 16;
+ while( capacity < need )
+ {
+ if( capacity > SIZE_MAX / 2 )
+ {
+ errno = EOVERFLOW;
+ return( -1 );
+ }
+ capacity *= 2;
+ }
+
+ if( capacity > SIZE_MAX / sizeof(*p) )
+ {
+ errno = EOVERFLOW;
+ return( -1 );
+ }
+
+ p = (mcpu_macro_piece *)realloc( list->piece,
+ capacity * sizeof(*p) );
+ if( p == NULL )
+ return( -1 );
+
+ list->piece = p;
+ list->capacity = capacity;
+ return( 0 );
+}
+
+
+static int
+macro_piece_list_append( mcpu_macro_piece_list *list,
+ enum mcpu_macro_piece_kind kind,
+ const __mpu_char16_t *text, size_t length )
+{
+ mcpu_macro_piece *piece;
+
+ if( macro_piece_list_reserve(list, list->count + 1) != 0 )
+ return( -1 );
+
+ piece = &list->piece[list->count];
+ piece->kind = kind;
+ mcpu_text_init( &piece->text );
+
+ if( length != 0 && mcpu_text_append(&piece->text, text, length) != 0 )
+ {
+ mcpu_text_free( &piece->text );
+ return( -1 );
+ }
+
+ ++list->count;
+ return( 0 );
+}
+
+
+static int
+macro_piece_list_append_piece( mcpu_macro_piece_list *list,
+ const mcpu_macro_piece *piece )
+{
+ return( macro_piece_list_append(list, piece->kind,
+ piece->text.data, piece->text.length) );
+}
+
+
+static void
+macro_piece_list_erase( mcpu_macro_piece_list *list,
+ size_t first, size_t count )
+{
+ size_t i;
+
+ if( count == 0 || first >= list->count )
+ return;
+
+ if( count > list->count - first )
+ count = list->count - first;
+
+ for( i = first; i < first + count; ++i )
+ mcpu_text_free( &list->piece[i].text );
+
+ if( first + count < list->count )
+ memmove( list->piece + first,
+ list->piece + first + count,
+ (list->count - first - count) * sizeof(*list->piece) );
+
+ list->count -= count;
+}
+
+
+static int
+macro_is_digit16( __mpu_char16_t c )
+{
+ return( c >= '0' && c <= '9' );
+}
+
+
+static size_t
+macro_pp_number_end( const __mpu_char16_t *text, size_t length, size_t p )
+{
+ size_t q = p + 1;
+
+ while( q < length )
+ {
+ __mpu_char16_t c = text[q];
+
+ if( mcpu_pp_is_identifier_char(c) || c == '.' )
+ {
+ ++q;
+ continue;
+ }
+
+ if( (c == '+' || c == '-') && q != p )
+ {
+ __mpu_char16_t prev = text[q - 1];
+
+ if( prev == 'e' || prev == 'E' || prev == 'p' || prev == 'P' )
+ {
+ ++q;
+ continue;
+ }
+ }
+
+ break;
+ }
+
+ return( q );
+}
+
+
+static size_t
+macro_quote_end( const __mpu_char16_t *text, size_t length, size_t p )
+{
+ __mpu_char16_t quote = text[p++];
+ int escaped = 0;
+
+ while( p < length )
+ {
+ __mpu_char16_t c = text[p++];
+
+ if( escaped )
+ escaped = 0;
+ else if( c == '\\' )
+ escaped = 1;
+ else if( c == quote || c == '\n' )
+ break;
+ }
+
+ return( p );
+}
+
+
+static size_t
+macro_literal_prefix_length( const __mpu_char16_t *text, size_t length,
+ size_t p, enum mcpu_language language )
+{
+ size_t q = p;
+
+ if( q + 2 < length && text[q] == 'u' && text[q + 1] == '8' &&
+ (text[q + 2] == '"' ||
+ (text[q + 2] == '\'' && language != MCPU_LANG_DIFF)) )
+ return( 2 );
+
+ if( q + 1 < length &&
+ (text[q] == 'L' || text[q] == 'u' || text[q] == 'U') &&
+ (text[q + 1] == '"' ||
+ (text[q + 1] == '\'' && language != MCPU_LANG_DIFF)) )
+ return( 1 );
+
+ return( 0 );
+}
+
+
+static int
+macro_ascii_at( const __mpu_char16_t *text, size_t length,
+ size_t p, const char *s )
+{
+ size_t i = 0;
+
+ while( s[i] )
+ {
+ if( p + i >= length ||
+ text[p + i] != (__mpu_char16_t)(unsigned char)s[i] )
+ return( 0 );
+ ++i;
+ }
+
+ return( 1 );
+}
+
+
+static size_t
+macro_punctuator_length( const __mpu_char16_t *text,
+ size_t length, size_t p )
+{
+ static const char *const punctuators[] =
+ {
+ "%:%:",
+ ">>=", "<<=", "...",
+ "->", "++", "--", "<<", ">>", "<=", ">=", "==", "!=",
+ "&&", "||", "*=", "/=", "%=", "+=", "-=", "&=", "^=", "|=",
+ "<:", ":>", "<%", "%>", "%:", "##",
+ "[", "]", "(", ")", "{", "}", ".", "&", "*", "+", "-",
+ "~", "!", "/", "%", "<", ">", "^", "|", "?", ":", ";",
+ "=", ",", "#",
+ NULL
+ };
+ size_t i;
+
+ for( i = 0; punctuators[i] != NULL; ++i )
+ {
+ size_t n = strlen( punctuators[i] );
+
+ if( macro_ascii_at(text, length, p, punctuators[i]) )
+ return( n );
+ }
+
+ return( 0 );
+}
+
+
+static int
+macro_piece_list_tokenize( mcpu_macro_piece_list *list,
+ const __mpu_char16_t *text, size_t length,
+ enum mcpu_language language,
+ int recognize_paste )
+{
+ size_t p = 0;
+
+ while( p < length )
+ {
+ size_t q;
+
+ if( is_space16(text[p]) )
+ {
+ q = p + 1;
+ while( q < length && is_space16(text[q]) )
+ ++q;
+
+ if( macro_piece_list_append(list, MCPU_MACRO_PIECE_SPACE,
+ text + p, q - p) != 0 )
+ return( -1 );
+ p = q;
+ continue;
+ }
+
+ if( recognize_paste && p + 1 < length &&
+ text[p] == '#' && text[p + 1] == '#' )
+ {
+ if( macro_piece_list_append(list, MCPU_MACRO_PIECE_PASTE,
+ NULL, 0) != 0 )
+ return( -1 );
+ p += 2;
+ continue;
+ }
+
+ {
+ size_t prefix_length;
+
+ prefix_length = macro_literal_prefix_length(text, length, p, language);
+ if( prefix_length != 0 )
+ {
+ q = macro_quote_end( text, length, p + prefix_length );
+ if( macro_piece_list_append(list, MCPU_MACRO_PIECE_TOKEN,
+ text + p, q - p) != 0 )
+ return( -1 );
+ p = q;
+ continue;
+ }
+ }
+
+ if( is_quote16(text[p], language) )
+ {
+ q = macro_quote_end( text, length, p );
+ if( macro_piece_list_append(list, MCPU_MACRO_PIECE_TOKEN,
+ text + p, q - p) != 0 )
+ return( -1 );
+ p = q;
+ continue;
+ }
+
+ if( mcpu_pp_is_identifier_start(text[p]) )
+ {
+ q = p + 1;
+ while( q < length && mcpu_pp_is_identifier_char(text[q]) )
+ ++q;
+
+ if( macro_piece_list_append(list, MCPU_MACRO_PIECE_TOKEN,
+ text + p, q - p) != 0 )
+ return( -1 );
+ p = q;
+ continue;
+ }
+
+ if( macro_is_digit16(text[p]) ||
+ (text[p] == '.' && p + 1 < length && macro_is_digit16(text[p + 1])) )
+ {
+ q = macro_pp_number_end( text, length, p );
+ if( macro_piece_list_append(list, MCPU_MACRO_PIECE_TOKEN,
+ text + p, q - p) != 0 )
+ return( -1 );
+ p = q;
+ continue;
+ }
+
+ q = macro_punctuator_length( text, length, p );
+ if( q != 0 )
+ {
+ if( macro_piece_list_append(list, MCPU_MACRO_PIECE_TOKEN,
+ text + p, q) != 0 )
+ return( -1 );
+ p += q;
+ continue;
+ }
+
+ if( macro_piece_list_append(list, MCPU_MACRO_PIECE_TOKEN,
+ text + p, 1) != 0 )
+ return( -1 );
+ ++p;
+ }
+
+ return( 0 );
+}
+
+
+static int
+macro_piece_is_ascii( const mcpu_macro_piece *piece, const char *s )
+{
+ if( piece == NULL || piece->kind != MCPU_MACRO_PIECE_TOKEN )
+ return( 0 );
+
+ return( mcpu_text_equal_ascii(piece->text.data, piece->text.length, s) );
+}
+
+
+static int
+macro_replacement_has_paste( const mcpu_macro *macro,
+ enum mcpu_language language )
+{
+ size_t p = 0;
+ __mpu_char16_t quote = 0;
+ int escaped = 0;
+
+ while( p < macro->replacement_length )
+ {
+ __mpu_char16_t c = macro->replacement[p];
+
+ if( quote )
+ {
+ if( escaped )
+ escaped = 0;
+ else if( c == '\\' )
+ escaped = 1;
+ else if( c == quote || c == '\n' )
+ quote = 0;
+ ++p;
+ continue;
+ }
+
+ if( is_quote16(c, language) )
+ {
+ quote = c;
+ ++p;
+ continue;
+ }
+
+ if( c == '#' && p + 1 < macro->replacement_length &&
+ macro->replacement[p + 1] == '#' )
+ return( 1 );
+
+ ++p;
+ }
+
+ return( 0 );
+}
+
+
+static int
+macro_replacement_has_va_opt( const mcpu_macro *macro,
+ enum mcpu_language language )
+{
+ size_t p = 0;
+ __mpu_char16_t quote = 0;
+ int escaped = 0;
+
+ while( p < macro->replacement_length )
+ {
+ __mpu_char16_t c = macro->replacement[p];
+
+ if( quote )
+ {
+ if( escaped )
+ escaped = 0;
+ else if( c == '\\' )
+ escaped = 1;
+ else if( c == quote || c == '\n' )
+ quote = 0;
+ ++p;
+ continue;
+ }
+
+ if( is_quote16(c, language) )
+ {
+ quote = c;
+ ++p;
+ continue;
+ }
+
+ if( mcpu_pp_is_identifier_start(c) )
+ {
+ size_t q = p + 1;
+
+ while( q < macro->replacement_length &&
+ mcpu_pp_is_identifier_char(macro->replacement[q]) )
+ ++q;
+
+ if( mcpu_text_equal_ascii(macro->replacement + p, q - p,
+ "__VA_OPT__") )
+ return( 1 );
+
+ p = q;
+ continue;
+ }
+
+ ++p;
+ }
+
+ return( 0 );
+}
+
+
+static int
+macro_piece_parameter_index( const mcpu_macro *macro,
+ const mcpu_macro_piece *piece )
+{
+ if( piece == NULL || piece->kind != MCPU_MACRO_PIECE_TOKEN )
+ return( -1 );
+
+ return( macro_argument_index(macro, piece->text.data, piece->text.length) );
+}
+
+
+static int
+macro_piece_parameter_is_pasted( const mcpu_macro_piece_list *replacement,
+ size_t index )
+{
+ size_t p;
+
+ p = index;
+ while( p != 0 )
+ {
+ --p;
+ if( replacement->piece[p].kind == MCPU_MACRO_PIECE_SPACE )
+ continue;
+ if( replacement->piece[p].kind == MCPU_MACRO_PIECE_PASTE )
+ return( 1 );
+ break;
+ }
+
+ p = index + 1;
+ while( p < replacement->count )
+ {
+ if( replacement->piece[p].kind == MCPU_MACRO_PIECE_SPACE )
+ {
+ ++p;
+ continue;
+ }
+ if( replacement->piece[p].kind == MCPU_MACRO_PIECE_PASTE )
+ return( 1 );
+ break;
+ }
+
+ return( 0 );
+}
+
+
+static int
+macro_append_raw_argument( mcpu_macro_piece_list *list,
+ const mcpu_macro_arg *arg,
+ enum mcpu_language language )
+{
+ const __mpu_char16_t *raw = arg->raw;
+ size_t raw_length = arg->raw_length;
+
+ while( raw_length != 0 && is_space16(*raw) )
+ {
+ ++raw;
+ --raw_length;
+ }
+ while( raw_length != 0 && is_space16(raw[raw_length - 1]) )
+ --raw_length;
+
+ if( raw_length == 0 )
+ return( macro_piece_list_append(list, MCPU_MACRO_PIECE_PLACEMARKER,
+ NULL, 0) );
+
+ return( macro_piece_list_tokenize(list, raw, raw_length,
+ language, 0) );
+}
+
+
+static int
+macro_append_expanded_argument( mcpu_macro_table *table,
+ mcpu_macro_piece_list *list,
+ mcpu_macro_arg *arg,
+ enum mcpu_language language,
+ const mcpu_macro_expansion_context *context,
+ unsigned depth )
+{
+ if( ensure_arg_expanded(table, arg, language, context, depth) != 0 )
+ return( -1 );
+
+ return( macro_piece_list_tokenize(list,
+ arg->expanded.data,
+ arg->expanded.length,
+ language, 0) );
+}
+
+
+static int
+macro_append_stringified_argument( mcpu_macro_piece_list *list,
+ const mcpu_macro_arg *arg,
+ enum mcpu_language language )
+{
+ mcpu_text stringified;
+ int rc;
+
+ mcpu_text_init( &stringified );
+ rc = append_stringified_arg( &stringified,
+ arg->raw, arg->raw_length,
+ language );
+ if( rc == 0 )
+ rc = macro_piece_list_append( list, MCPU_MACRO_PIECE_TOKEN,
+ stringified.data, stringified.length );
+
+ mcpu_text_free( &stringified );
+ return( rc );
+}
+
+
+static int macro_resolve_pastes(
+ mcpu_macro_piece_list *list,
+ enum mcpu_language language,
+ const mcpu_macro_expansion_context *context );
+static int macro_pieces_to_text(
+ const mcpu_macro_piece_list *list, mcpu_text *text );
+
+
+static int
+macro_va_opt_bounds( const mcpu_macro_piece_list *replacement,
+ size_t index, size_t limit,
+ size_t *content_start, size_t *content_end,
+ size_t *after )
+{
+ size_t p;
+ int depth;
+
+ if( replacement == NULL || index >= limit ||
+ !macro_piece_is_ascii(&replacement->piece[index], "__VA_OPT__") )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ p = index + 1;
+ while( p < limit && replacement->piece[p].kind == MCPU_MACRO_PIECE_SPACE )
+ ++p;
+
+ if( p >= limit || !macro_piece_is_ascii(&replacement->piece[p], "(") )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ *content_start = ++p;
+ depth = 1;
+ while( p < limit )
+ {
+ if( macro_piece_is_ascii(&replacement->piece[p], "(") )
+ ++depth;
+ else if( macro_piece_is_ascii(&replacement->piece[p], ")") )
+ {
+ --depth;
+ if( depth == 0 )
+ {
+ *content_end = p;
+ *after = p + 1;
+ return( 0 );
+ }
+ }
+ ++p;
+ }
+
+ errno = EINVAL;
+ return( -1 );
+}
+
+
+static int
+macro_va_opt_is_pasted( const mcpu_macro_piece_list *replacement,
+ size_t index, size_t after, size_t limit )
+{
+ size_t p;
+
+ p = index;
+ while( p != 0 )
+ {
+ --p;
+ if( replacement->piece[p].kind == MCPU_MACRO_PIECE_SPACE )
+ continue;
+ if( replacement->piece[p].kind == MCPU_MACRO_PIECE_PASTE )
+ return( 1 );
+ break;
+ }
+
+ p = after;
+ while( p < limit )
+ {
+ if( replacement->piece[p].kind == MCPU_MACRO_PIECE_SPACE )
+ {
+ ++p;
+ continue;
+ }
+ if( replacement->piece[p].kind == MCPU_MACRO_PIECE_PASTE )
+ return( 1 );
+ break;
+ }
+
+ return( 0 );
+}
+
+
+static int
+macro_piece_list_has_token( const mcpu_macro_piece_list *list )
+{
+ size_t i;
+
+ for( i = 0; i < list->count; ++i )
+ {
+ if( list->piece[i].kind == MCPU_MACRO_PIECE_TOKEN )
+ return( 1 );
+ }
+
+ return( 0 );
+}
+
+
+static int
+macro_piece_list_append_range( mcpu_macro_piece_list *dst,
+ const mcpu_macro_piece_list *src )
+{
+ size_t i;
+
+ for( i = 0; i < src->count; ++i )
+ {
+ if( macro_piece_list_append_piece(dst, &src->piece[i]) != 0 )
+ return( -1 );
+ }
+
+ return( 0 );
+}
+
+
+static int
+macro_variadic_has_tokens( mcpu_macro_table *table,
+ const mcpu_macro *macro,
+ mcpu_macro_arg *args,
+ enum mcpu_language language,
+ const mcpu_macro_expansion_context *context,
+ unsigned depth )
+{
+ mcpu_macro_arg *arg;
+ size_t i;
+
+ if( !macro->variadic )
+ return( 0 );
+
+ arg = &args[macro->nargs];
+ if( ensure_arg_expanded(table, arg, language, context, depth) != 0 )
+ return( -1 );
+
+ for( i = 0; i < arg->expanded.length; ++i )
+ {
+ if( !is_space16(arg->expanded.data[i]) )
+ return( 1 );
+ }
+
+ return( 0 );
+}
+
+
+static int macro_substitute_piece_range(
+ mcpu_macro_table *table,
+ mcpu_macro *macro,
+ mcpu_macro_arg *args,
+ const mcpu_macro_piece_list *replacement,
+ size_t begin, size_t end,
+ enum mcpu_language language,
+ const mcpu_macro_expansion_context *context,
+ unsigned depth,
+ mcpu_macro_piece_list *substituted );
+
+
+static int
+macro_append_stringified_va_opt( mcpu_macro_table *table,
+ mcpu_macro *macro,
+ mcpu_macro_arg *args,
+ const mcpu_macro_piece_list *replacement,
+ size_t content_start, size_t content_end,
+ enum mcpu_language language,
+ const mcpu_macro_expansion_context *context,
+ unsigned depth,
+ mcpu_macro_piece_list *substituted )
+{
+ mcpu_macro_piece_list content;
+ mcpu_text text;
+ mcpu_text stringified;
+ int nonempty;
+ int rc = -1;
+
+ macro_piece_list_init( &content );
+ mcpu_text_init( &text );
+ mcpu_text_init( &stringified );
+
+ nonempty = macro_variadic_has_tokens(table, macro, args,
+ language, context, depth);
+ if( nonempty < 0 )
+ goto done;
+
+ if( nonempty )
+ {
+ if( macro_substitute_piece_range(table, macro, args, replacement,
+ content_start, content_end,
+ language, context, depth,
+ &content) != 0 ||
+ macro_resolve_pastes(&content, language, context) != 0 ||
+ macro_pieces_to_text(&content, &text) != 0 )
+ goto done;
+ }
+
+ if( append_stringified_arg(&stringified,
+ text.data, text.length, language) != 0 ||
+ macro_piece_list_append(substituted, MCPU_MACRO_PIECE_TOKEN,
+ stringified.data, stringified.length) != 0 )
+ goto done;
+
+ rc = 0;
+
+done:
+ mcpu_text_free( &stringified );
+ mcpu_text_free( &text );
+ macro_piece_list_free( &content );
+ return( rc );
+}
+
+
+static int
+macro_substitute_piece_range( mcpu_macro_table *table,
+ mcpu_macro *macro,
+ mcpu_macro_arg *args,
+ const mcpu_macro_piece_list *replacement,
+ size_t begin, size_t end,
+ enum mcpu_language language,
+ const mcpu_macro_expansion_context *context,
+ unsigned depth,
+ mcpu_macro_piece_list *substituted )
+{
+ size_t i = begin;
+
+ while( i < end )
+ {
+ const mcpu_macro_piece *piece = &replacement->piece[i];
+ int argno;
+
+ if( piece->kind == MCPU_MACRO_PIECE_SPACE ||
+ piece->kind == MCPU_MACRO_PIECE_PASTE )
+ {
+ if( macro_piece_list_append_piece(substituted, piece) != 0 )
+ return( -1 );
+ ++i;
+ continue;
+ }
+
+ if( macro_piece_is_ascii(piece, "#") )
+ {
+ size_t j = i + 1;
+
+ while( j < end &&
+ replacement->piece[j].kind == MCPU_MACRO_PIECE_SPACE )
+ ++j;
+
+ if( j >= end )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ argno = macro_piece_parameter_index( macro, &replacement->piece[j] );
+ if( argno >= 0 )
+ {
+ if( macro_append_stringified_argument(substituted,
+ &args[argno], language) != 0 )
+ return( -1 );
+ i = j + 1;
+ continue;
+ }
+
+ if( macro_piece_is_ascii(&replacement->piece[j], "__VA_OPT__") )
+ {
+ size_t content_start;
+ size_t content_end;
+ size_t after;
+
+ if( macro_va_opt_bounds(replacement, j, end,
+ &content_start, &content_end, &after) != 0 ||
+ macro_append_stringified_va_opt(table, macro, args, replacement,
+ content_start, content_end,
+ language, context, depth,
+ substituted) != 0 )
+ return( -1 );
+
+ i = after;
+ continue;
+ }
+
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ if( macro_piece_is_ascii(piece, "__VA_OPT__") )
+ {
+ mcpu_macro_piece_list content;
+ size_t content_start;
+ size_t content_end;
+ size_t after;
+ int nonempty;
+ int pasted;
+ int rc = -1;
+
+ macro_piece_list_init( &content );
+ if( macro_va_opt_bounds(replacement, i, end,
+ &content_start, &content_end, &after) != 0 )
+ goto va_done;
+
+ nonempty = macro_variadic_has_tokens(table, macro, args,
+ language, context, depth);
+ if( nonempty < 0 )
+ goto va_done;
+
+ pasted = macro_va_opt_is_pasted(replacement, i, after, end);
+ if( nonempty )
+ {
+ if( macro_substitute_piece_range(table, macro, args, replacement,
+ content_start, content_end,
+ language, context, depth,
+ &content) != 0 ||
+ macro_resolve_pastes(&content, language, context) != 0 )
+ goto va_done;
+
+ if( pasted && !macro_piece_list_has_token(&content) )
+ {
+ if( macro_piece_list_append(substituted,
+ MCPU_MACRO_PIECE_PLACEMARKER,
+ NULL, 0) != 0 )
+ goto va_done;
+ }
+ else if( macro_piece_list_append_range(substituted, &content) != 0 )
+ goto va_done;
+ }
+ else if( pasted &&
+ macro_piece_list_append(substituted,
+ MCPU_MACRO_PIECE_PLACEMARKER,
+ NULL, 0) != 0 )
+ goto va_done;
+
+ rc = 0;
+
+va_done:
+ macro_piece_list_free( &content );
+ if( rc != 0 )
+ return( -1 );
+ i = after;
+ continue;
+ }
+
+ argno = macro_piece_parameter_index( macro, piece );
+ if( argno >= 0 )
+ {
+ if( macro_piece_parameter_is_pasted(replacement, i) )
+ {
+ if( macro_append_raw_argument(substituted,
+ &args[argno], language) != 0 )
+ return( -1 );
+ }
+ else
+ {
+ if( macro_append_expanded_argument(table, substituted,
+ &args[argno], language,
+ context, depth) != 0 )
+ return( -1 );
+ }
+
+ ++i;
+ continue;
+ }
+
+ if( macro_piece_list_append_piece(substituted, piece) != 0 )
+ return( -1 );
+ ++i;
+ }
+
+ return( 0 );
+}
+
+
+static int
+macro_build_function_pieces( mcpu_macro_table *table,
+ mcpu_macro *macro,
+ mcpu_macro_arg *args,
+ enum mcpu_language language,
+ const mcpu_macro_expansion_context *context,
+ unsigned depth,
+ mcpu_macro_piece_list *substituted )
+{
+ mcpu_macro_piece_list replacement;
+ int rc = -1;
+
+ macro_piece_list_init( &replacement );
+ if( macro_piece_list_tokenize(&replacement,
+ macro->replacement,
+ macro->replacement_length,
+ language, 1) != 0 )
+ goto done;
+
+ if( macro_substitute_piece_range(table, macro, args, &replacement,
+ 0, replacement.count,
+ language, context, depth,
+ substituted) != 0 )
+ goto done;
+
+ rc = 0;
+
+done:
+ macro_piece_list_free( &replacement );
+ return( rc );
+}
+
+
+static void
+macro_normalize_paste_spacing( mcpu_macro_piece_list *list )
+{
+ size_t i = 0;
+
+ while( i < list->count )
+ {
+ if( list->piece[i].kind == MCPU_MACRO_PIECE_SPACE &&
+ ((i != 0 && list->piece[i - 1].kind == MCPU_MACRO_PIECE_PASTE) ||
+ (i + 1 < list->count &&
+ list->piece[i + 1].kind == MCPU_MACRO_PIECE_PASTE)) )
+ {
+ macro_piece_list_erase( list, i, 1 );
+ if( i != 0 )
+ --i;
+ continue;
+ }
+ ++i;
+ }
+
+ i = 1;
+ while( i < list->count )
+ {
+ if( list->piece[i - 1].kind == MCPU_MACRO_PIECE_PASTE &&
+ list->piece[i].kind == MCPU_MACRO_PIECE_PASTE )
+ {
+ /***********************************************************
+ GNU CPP accepts a run of adjacent ## operators as one paste
+ operator. Preserve that long-standing behavior.
+ ***********************************************************/
+ macro_piece_list_erase( list, i, 1 );
+ continue;
+ }
+ ++i;
+ }
+}
+
+
+static int
+macro_paste_is_single_token( const mcpu_text *text,
+ enum mcpu_language language )
+{
+ mcpu_macro_piece_list tokens;
+ int valid;
+
+ macro_piece_list_init( &tokens );
+ if( macro_piece_list_tokenize(&tokens,
+ text->data, text->length,
+ language, 0) != 0 )
+ {
+ macro_piece_list_free( &tokens );
+ return( -1 );
+ }
+
+ valid = tokens.count == 1 &&
+ tokens.piece[0].kind == MCPU_MACRO_PIECE_TOKEN;
+ macro_piece_list_free( &tokens );
+ return( valid );
+}
+
+
+static int
+macro_warn_invalid_paste( const mcpu_macro_piece *left,
+ const mcpu_macro_piece *right,
+ const mcpu_macro_expansion_context *context )
+{
+ char *l = mcpu_text_to_utf8( left->text.data, left->text.length );
+ char *r = mcpu_text_to_utf8( right->text.data, right->text.length );
+
+ int rc;
+
+ rc = mcpp_diagnostic_warning(
+ context ? context->options : NULL,
+ context && context->filename ? context->filename : "<input>",
+ context ? context->line_number : 0,
+ "pasting \"%s\" and \"%s\" does not give a valid preprocessing token",
+ l ? l : "?", r ? r : "?" );
+
+ free( l );
+ free( r );
+ return( rc );
+}
+
+
+static int
+macro_resolve_pastes( mcpu_macro_piece_list *list,
+ enum mcpu_language language,
+ const mcpu_macro_expansion_context *context )
+{
+ size_t i;
+
+ macro_normalize_paste_spacing( list );
+
+ for( ;; )
+ {
+ for( i = 0; i < list->count; ++i )
+ {
+ if( list->piece[i].kind == MCPU_MACRO_PIECE_PASTE )
+ break;
+ }
+
+ if( i == list->count )
+ break;
+
+ if( i == 0 || i + 1 >= list->count ||
+ list->piece[i - 1].kind == MCPU_MACRO_PIECE_SPACE ||
+ list->piece[i + 1].kind == MCPU_MACRO_PIECE_SPACE ||
+ list->piece[i - 1].kind == MCPU_MACRO_PIECE_PASTE ||
+ list->piece[i + 1].kind == MCPU_MACRO_PIECE_PASTE )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ if( list->piece[i - 1].kind == MCPU_MACRO_PIECE_PLACEMARKER &&
+ list->piece[i + 1].kind == MCPU_MACRO_PIECE_PLACEMARKER )
+ {
+ macro_piece_list_erase( list, i, 2 );
+ continue;
+ }
+
+ if( list->piece[i - 1].kind == MCPU_MACRO_PIECE_PLACEMARKER )
+ {
+ macro_piece_list_erase( list, i - 1, 2 );
+ continue;
+ }
+
+ if( list->piece[i + 1].kind == MCPU_MACRO_PIECE_PLACEMARKER )
+ {
+ macro_piece_list_erase( list, i, 2 );
+ continue;
+ }
+
+ {
+ mcpu_text pasted;
+ int valid;
+
+ mcpu_text_init( &pasted );
+ if( mcpu_text_append(&pasted,
+ list->piece[i - 1].text.data,
+ list->piece[i - 1].text.length) != 0 ||
+ mcpu_text_append(&pasted,
+ list->piece[i + 1].text.data,
+ list->piece[i + 1].text.length) != 0 )
+ {
+ mcpu_text_free( &pasted );
+ return( -1 );
+ }
+
+ valid = macro_paste_is_single_token( &pasted, language );
+ if( valid < 0 )
+ {
+ mcpu_text_free( &pasted );
+ return( -1 );
+ }
+
+ if( !valid )
+ {
+ /***********************************************************
+ GNU CPP documents this case as a diagnostic followed by
+ emission of the two original tokens. Whitespace between
+ them is unspecified, so keep them adjacent and discard ##.
+ ***********************************************************/
+ if( macro_warn_invalid_paste(&list->piece[i - 1],
+ &list->piece[i + 1], context) != 0 )
+ {
+ mcpu_text_free( &pasted );
+ return( -1 );
+ }
+ mcpu_text_free( &pasted );
+ macro_piece_list_erase( list, i, 1 );
+ continue;
+ }
+
+ mcpu_text_free( &list->piece[i - 1].text );
+ list->piece[i - 1].text = pasted;
+ list->piece[i - 1].kind = MCPU_MACRO_PIECE_TOKEN;
+ macro_piece_list_erase( list, i, 2 );
+ }
+ }
+
+ i = 0;
+ while( i < list->count )
+ {
+ if( list->piece[i].kind == MCPU_MACRO_PIECE_PLACEMARKER )
+ {
+ macro_piece_list_erase( list, i, 1 );
+ continue;
+ }
+ ++i;
+ }
+
+ return( 0 );
+}
+
+
+static int
+macro_pieces_to_text( const mcpu_macro_piece_list *list, mcpu_text *text )
+{
+ size_t i;
+ int previous_space = 0;
+
+ for( i = 0; i < list->count; ++i )
+ {
+ if( list->piece[i].kind == MCPU_MACRO_PIECE_PASTE ||
+ list->piece[i].kind == MCPU_MACRO_PIECE_PLACEMARKER )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ if( list->piece[i].kind == MCPU_MACRO_PIECE_SPACE )
+ {
+ /***********************************************************
+ Removal of an empty parameter, placemarker or __VA_OPT__
+ fragment can make two replacement-list space pieces
+ adjacent. They are one logical whitespace separator.
+
+ Preserve the text of a single space piece: it may come from
+ inside an actual macro argument, whose formatting is not part
+ of replacement-list normalization.
+ ***********************************************************/
+ if( previous_space )
+ continue;
+ previous_space = 1;
+ }
+ else
+ previous_space = 0;
+
+ if( mcpu_text_append(text,
+ list->piece[i].text.data,
+ list->piece[i].text.length) != 0 )
+ return( -1 );
+ }
+
+ return( 0 );
+}
+
+
+static int
+expand_function_macro_with_paste( mcpu_macro_table *table,
+ mcpu_macro *macro,
+ mcpu_macro_arg *args,
+ enum mcpu_language language,
+ const mcpu_macro_expansion_context *context,
+ mcpu_text *output, unsigned depth )
+{
+ mcpu_macro_piece_list pieces;
+ mcpu_text substituted;
+ int rc = -1;
+
+ macro_piece_list_init( &pieces );
+ mcpu_text_init( &substituted );
+
+ if( macro_build_function_pieces(table, macro, args,
+ language, context, depth,
+ &pieces) != 0 ||
+ macro_resolve_pastes(&pieces, language, context) != 0 ||
+ macro_pieces_to_text(&pieces, &substituted) != 0 )
+ goto done;
+
+ macro->expanding = 1;
+ rc = expand_range( table, substituted.data, substituted.length,
+ language, context, output, depth + 1 );
+ macro->expanding = 0;
+
+done:
+ macro->expanding = 0;
+ mcpu_text_free( &substituted );
+ macro_piece_list_free( &pieces );
+ return( rc );
+}
+
+
+static int
+expand_object_macro_with_paste( mcpu_macro_table *table,
+ mcpu_macro *macro,
+ enum mcpu_language language,
+ const mcpu_macro_expansion_context *context,
+ mcpu_text *output, unsigned depth )
+{
+ mcpu_macro_piece_list pieces;
+ mcpu_text substituted;
+ int rc = -1;
+
+ macro_piece_list_init( &pieces );
+ mcpu_text_init( &substituted );
+
+ if( macro_piece_list_tokenize(&pieces,
+ macro->replacement,
+ macro->replacement_length,
+ language, 1) != 0 ||
+ macro_resolve_pastes(&pieces, language, context) != 0 ||
+ macro_pieces_to_text(&pieces, &substituted) != 0 )
+ goto done;
+
+ macro->expanding = 1;
+ rc = expand_range( table, substituted.data, substituted.length,
+ language, context, output, depth + 1 );
+ macro->expanding = 0;
+
+done:
+ macro->expanding = 0;
+ mcpu_text_free( &substituted );
+ macro_piece_list_free( &pieces );
+ return( rc );
+}
+
+
+static int
+expand_function_macro( mcpu_macro_table *table, mcpu_macro *macro,
+ mcpu_macro_arg *args, size_t nargs,
+ enum mcpu_language language,
+ const mcpu_macro_expansion_context *context,
+ mcpu_text *output, unsigned depth )
+{
+ mcpu_text substituted;
+ size_t p = 0;
+ size_t i;
+ int rc = -1;
+
+ if( !macro->variadic && macro->nargs == 0 && nargs == 1 )
+ {
+ int all_space = 1;
+
+ for( i = 0; i < args[0].raw_length; ++i )
+ {
+ if( !is_space16(args[0].raw[i]) )
+ {
+ all_space = 0;
+ break;
+ }
+ }
+
+ if( all_space )
+ nargs = 0;
+ }
+
+ if( nargs < (size_t)macro->nargs + (macro->variadic ? 1U : 0U) )
+ {
+ char *name = mcpu_text_to_utf8( macro->name, macro->name_length );
+ fprintf( stderr, "%s:%u: error: macro '%s' used with too few arguments\n",
+ context && context->filename ? context->filename : "<input>",
+ context ? context->line_number : 0,
+ name ? name : "?" );
+ free( name );
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ if( !macro->variadic && nargs > (size_t)macro->nargs )
+ {
+ char *name = mcpu_text_to_utf8( macro->name, macro->name_length );
+ fprintf( stderr, "%s:%u: error: macro '%s' used with too many arguments\n",
+ context && context->filename ? context->filename : "<input>",
+ context ? context->line_number : 0,
+ name ? name : "?" );
+ free( name );
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ if( macro_replacement_has_paste(macro, language) ||
+ macro_replacement_has_va_opt(macro, language) )
+ return( expand_function_macro_with_paste(table, macro, args,
+ language, context,
+ output, depth) );
+
+ mcpu_text_init( &substituted );
+
+ while( p < macro->replacement_length )
+ {
+ if( is_quote16(macro->replacement[p], language) )
+ {
+ __mpu_char16_t quote = macro->replacement[p];
+ int escaped = 0;
+ int first = 1;
+
+ do
+ {
+ __mpu_char16_t c = macro->replacement[p++];
+
+ if( mcpu_text_append_char(&substituted, c) != 0 )
+ goto substituted_fail;
+
+ if( first )
+ first = 0;
+ else if( escaped )
+ escaped = 0;
+ else if( c == '\\' )
+ escaped = 1;
+ else if( c == quote )
+ break;
+ else if( c == '\n' )
+ break;
+ }
+ while( p < macro->replacement_length );
+
+ continue;
+ }
+
+ if( macro->replacement[p] == '#' )
+ {
+ size_t q = p + 1;
+ size_t start;
+ int argno;
+
+ if( q < macro->replacement_length && macro->replacement[q] == '#' )
+ {
+ errno = EINVAL;
+ goto substituted_fail;
+ }
+
+ while( q < macro->replacement_length && is_space16(macro->replacement[q]) )
+ ++q;
+
+ if( q >= macro->replacement_length ||
+ !mcpu_pp_is_identifier_start(macro->replacement[q]) )
+ {
+ errno = EINVAL;
+ goto substituted_fail;
+ }
+
+ start = q++;
+ while( q < macro->replacement_length &&
+ mcpu_pp_is_identifier_char(macro->replacement[q]) )
+ ++q;
+
+ argno = macro_argument_index( macro,
+ macro->replacement + start, q - start );
+ if( argno < 0 )
+ {
+ errno = EINVAL;
+ goto substituted_fail;
+ }
+
+ if( append_stringified_arg(&substituted,
+ args[argno].raw,
+ args[argno].raw_length,
+ language) != 0 )
+ goto substituted_fail;
+
+ p = q;
+ continue;
+ }
+
+ if( mcpu_pp_is_identifier_start(macro->replacement[p]) )
+ {
+ size_t start = p++;
+ int argno;
+
+ while( p < macro->replacement_length &&
+ mcpu_pp_is_identifier_char(macro->replacement[p]) )
+ ++p;
+
+ argno = macro_argument_index( macro,
+ macro->replacement + start, p - start );
+ if( argno >= 0 )
+ {
+ /***********************************************************
+ Ordinary parameter occurrences use
+ the macro-expanded actual argument. Stringified occurrences
+ above deliberately use the raw spelling and therefore do not
+ force this expansion.
+ ***********************************************************/
+ if( ensure_arg_expanded(table, &args[argno], language,
+ context, depth) != 0 )
+ goto substituted_fail;
+
+ if( mcpu_text_append(&substituted,
+ args[argno].expanded.data,
+ args[argno].expanded.length) != 0 )
+ goto substituted_fail;
+ }
+ else if( mcpu_text_append(&substituted,
+ macro->replacement + start, p - start) != 0 )
+ goto substituted_fail;
+
+ continue;
+ }
+
+ if( mcpu_text_append_char(&substituted, macro->replacement[p++]) != 0 )
+ goto substituted_fail;
+ }
+
+ macro->expanding = 1;
+ rc = expand_range( table, substituted.data, substituted.length,
+ language, context,
+ output, depth + 1 );
+
+substituted_fail:
+ mcpu_text_free( &substituted );
+
+ macro->expanding = 0;
+ return( rc );
+}
+
+
+static int
+expand_range( mcpu_macro_table *table,
+ const __mpu_char16_t *input, size_t length,
+ enum mcpu_language language,
+ const mcpu_macro_expansion_context *context,
+ mcpu_text *output, unsigned depth )
+{
+ size_t p = 0;
+
+ if( depth > MCPU_CPP_MACRO_EXPANSION_LIMIT )
+ {
+ errno = ELOOP;
+ return( -1 );
+ }
+
+ while( p < length )
+ {
+ if( is_quote16(input[p], language) )
+ {
+ __mpu_char16_t quote = input[p];
+ int escaped = 0;
+ int first = 1;
+
+ do
+ {
+ __mpu_char16_t c = input[p++];
+
+ if( mcpu_text_append_char(output, c) != 0 )
+ return( -1 );
+
+ if( first )
+ first = 0;
+ else if( escaped )
+ escaped = 0;
+ else if( c == '\\' )
+ escaped = 1;
+ else if( c == quote )
+ break;
+ else if( c == '\n' )
+ break;
+ }
+ while( p < length );
+
+ continue;
+ }
+
+ if( mcpu_pp_is_identifier_start(input[p]) )
+ {
+ size_t start = p++;
+ mcpu_macro *macro;
+
+ while( p < length && mcpu_pp_is_identifier_char(input[p]) )
+ ++p;
+
+ macro = mcpu_macro_find( table, input + start, p - start );
+ if( macro && !macro->expanding )
+ {
+ if( macro->builtin != MCPU_MACRO_BUILTIN_NONE )
+ {
+ if( expand_builtin(macro, context, output) != 0 )
+ return( -1 );
+ continue;
+ }
+
+ if( macro->nargs < 0 )
+ {
+ int rc;
+
+ if( macro_replacement_has_paste(macro, language) )
+ rc = expand_object_macro_with_paste(table, macro,
+ language, context,
+ output, depth);
+ else
+ {
+ macro->expanding = 1;
+ rc = expand_range( table,
+ macro->replacement,
+ macro->replacement_length,
+ language, context,
+ output, depth + 1 );
+ macro->expanding = 0;
+ }
+
+ if( rc != 0 )
+ return( -1 );
+ continue;
+ }
+ else
+ {
+ size_t q = p;
+
+ while( q < length && is_space16(input[q]) )
+ ++q;
+
+ if( q < length && input[q] == '(' )
+ {
+ mcpu_macro_arg *args = NULL;
+ size_t nargs = 0;
+ size_t after_call;
+ int rc;
+
+ if( parse_macro_args(input, length, q, language,
+ &args, &nargs, &after_call) != 0 )
+ {
+ char *name = mcpu_text_to_utf8( macro->name,
+ macro->name_length );
+ fprintf( stderr,
+ "%s:%u: error: unterminated argument list for macro '%s'\n",
+ context && context->filename ? context->filename : "<input>",
+ context ? context->line_number : 0,
+ name ? name : "?" );
+ free( name );
+ return( -1 );
+ }
+
+ if( prepare_variadic_args(macro, input, after_call,
+ &args, &nargs) != 0 )
+ {
+ macro_args_free( args, nargs );
+ return( -1 );
+ }
+
+ rc = expand_function_macro( table, macro, args, nargs,
+ language, context,
+ output, depth );
+ macro_args_free( args, nargs );
+ if( rc != 0 )
+ return( -1 );
+
+ p = after_call;
+ continue;
+ }
+ }
+ }
+
+ if( mcpu_text_append(output, input + start, p - start) != 0 )
+ return( -1 );
+
+ continue;
+ }
+
+ if( mcpu_text_append_char(output, input[p++]) != 0 )
+ return( -1 );
+ }
+
+ return( 0 );
+}
+
+
+
+static int
+macro_name_compare( const void *a, const void *b )
+{
+ const mcpu_macro *ma = *(const mcpu_macro * const *)a;
+ const mcpu_macro *mb = *(const mcpu_macro * const *)b;
+ size_t n;
+ size_t i;
+
+ n = ma->name_length < mb->name_length ? ma->name_length : mb->name_length;
+ for( i = 0; i < n; ++i )
+ {
+ if( ma->name[i] < mb->name[i] )
+ return( -1 );
+ if( ma->name[i] > mb->name[i] )
+ return( 1 );
+ }
+
+ if( ma->name_length < mb->name_length )
+ return( -1 );
+ if( ma->name_length > mb->name_length )
+ return( 1 );
+ return( 0 );
+}
+
+
+int
+mcpu_macro_dump( const mcpu_macro_table *table, FILE *stream,
+ int include_predefined )
+{
+ mcpu_macro **list;
+ size_t predefined_count = 0;
+ size_t other_count = 0;
+ size_t count;
+ size_t i;
+ size_t k = 0;
+ size_t other_begin;
+
+ if( table == NULL || stream == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ for( i = 0; i < MCPU_CPP_MACRO_BUCKETS; ++i )
+ {
+ const mcpu_macro *macro;
+ for( macro = table->bucket[i]; macro; macro = macro->next )
+ {
+ /* Context-dependent special macros are not useful in a static dump. */
+ if( macro->builtin != MCPU_MACRO_BUILTIN_NONE &&
+ macro->builtin < MCPU_MACRO_BUILTIN_REAL_MAX )
+ continue;
+
+ if( macro->origin == MCPU_MACRO_ORIGIN_BUILTIN )
+ ++predefined_count;
+ else
+ ++other_count;
+ }
+ }
+
+ count = other_count + (include_predefined ? predefined_count : 0);
+ list = count ? (mcpu_macro **)calloc( count, sizeof(*list) ) : NULL;
+ if( count != 0 && list == NULL )
+ return( -1 );
+
+ if( include_predefined )
+ {
+ for( i = 0; i < MCPU_CPP_MACRO_BUCKETS; ++i )
+ {
+ mcpu_macro *macro;
+ for( macro = table->bucket[i]; macro; macro = macro->next )
+ {
+ if( (macro->builtin == MCPU_MACRO_BUILTIN_NONE ||
+ macro->builtin >= MCPU_MACRO_BUILTIN_REAL_MAX) &&
+ macro->origin == MCPU_MACRO_ORIGIN_BUILTIN )
+ list[k++] = macro;
+ }
+ }
+ }
+
+ other_begin = k;
+ for( i = 0; i < MCPU_CPP_MACRO_BUCKETS; ++i )
+ {
+ mcpu_macro *macro;
+ for( macro = table->bucket[i]; macro; macro = macro->next )
+ {
+ if( (macro->builtin == MCPU_MACRO_BUILTIN_NONE ||
+ macro->builtin >= MCPU_MACRO_BUILTIN_REAL_MAX) &&
+ macro->origin != MCPU_MACRO_ORIGIN_BUILTIN )
+ list[k++] = macro;
+ }
+ }
+
+ if( include_predefined && predefined_count > 1 )
+ qsort( list, predefined_count, sizeof(*list), macro_name_compare );
+ if( other_count > 1 )
+ qsort( list + other_begin, other_count, sizeof(*list), macro_name_compare );
+
+ for( i = 0; i < count; ++i )
+ {
+ mcpu_macro *macro = list[i];
+ char *name = mcpu_text_to_utf8( macro->name, macro->name_length );
+ char *replacement = NULL;
+ mcpu_text generated;
+ size_t replacement_length;
+ int j;
+
+ mcpu_text_init( &generated );
+ if( macro->builtin == MCPU_MACRO_BUILTIN_NONE )
+ {
+ replacement = mcpu_text_to_utf8( macro->replacement,
+ macro->replacement_length );
+ replacement_length = macro->replacement_length;
+ }
+ else
+ {
+ if( expand_builtin(macro, NULL, &generated) != 0 )
+ {
+ free( name );
+ mcpu_text_free( &generated );
+ free( list );
+ return( -1 );
+ }
+ replacement = mcpu_text_to_utf8( generated.data, generated.length );
+ replacement_length = generated.length;
+ }
+
+ if( name == NULL || replacement == NULL )
+ {
+ free( name );
+ free( replacement );
+ mcpu_text_free( &generated );
+ free( list );
+ return( -1 );
+ }
+
+ if( fprintf(stream, "#define %s", name) < 0 )
+ goto io_error;
+
+ if( macro->nargs >= 0 )
+ {
+ if( fputc('(', stream) == EOF )
+ goto io_error;
+ for( j = 0; j < macro->nargs; ++j )
+ {
+ char *arg = mcpu_text_to_utf8( macro->argnames[j],
+ macro->argname_lengths[j] );
+ if( arg == NULL )
+ goto io_error;
+ if( j != 0 && fputc(',', stream) == EOF )
+ {
+ free( arg );
+ goto io_error;
+ }
+ if( fputs(arg, stream) == EOF )
+ {
+ free( arg );
+ goto io_error;
+ }
+ free( arg );
+ }
+ if( macro->variadic )
+ {
+ if( macro->nargs != 0 && fputc(',', stream) == EOF )
+ goto io_error;
+ if( fputs("...", stream) == EOF )
+ goto io_error;
+ }
+ if( fputc(')', stream) == EOF )
+ goto io_error;
+ }
+
+ if( replacement_length != 0 )
+ {
+ if( fputc(' ', stream) == EOF || fputs(replacement, stream) == EOF )
+ goto io_error;
+ }
+
+ if( fputc('\n', stream) == EOF )
+ goto io_error;
+
+ free( name );
+ free( replacement );
+ mcpu_text_free( &generated );
+ continue;
+
+io_error:
+ free( name );
+ free( replacement );
+ mcpu_text_free( &generated );
+ free( list );
+ return( -1 );
+ }
+
+ free( list );
+ return( ferror(stream) ? -1 : 0 );
+}
+
+
+int
+mcpu_macro_dump_text( const mcpu_macro_table *table, mcpu_text *output,
+ const char *line_filename )
+{
+ mcpu_macro **list;
+ size_t count = 0;
+ size_t i;
+ size_t k = 0;
+
+ if( table == NULL || output == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ for( i = 0; i < MCPU_CPP_MACRO_BUCKETS; ++i )
+ {
+ const mcpu_macro *macro;
+
+ for( macro = table->bucket[i]; macro; macro = macro->next )
+ if( macro->builtin == MCPU_MACRO_BUILTIN_NONE ||
+ macro->builtin >= MCPU_MACRO_BUILTIN_REAL_MAX )
+ ++count;
+ }
+
+ list = count ? (mcpu_macro **)calloc( count, sizeof(*list) ) : NULL;
+ if( count != 0 && list == NULL )
+ return( -1 );
+
+ for( i = 0; i < MCPU_CPP_MACRO_BUCKETS; ++i )
+ {
+ mcpu_macro *macro;
+
+ for( macro = table->bucket[i]; macro; macro = macro->next )
+ if( macro->builtin == MCPU_MACRO_BUILTIN_NONE ||
+ macro->builtin >= MCPU_MACRO_BUILTIN_REAL_MAX )
+ list[k++] = macro;
+ }
+
+ if( count > 1 )
+ qsort( list, count, sizeof(*list), macro_name_compare );
+
+ for( i = 0; i < count; ++i )
+ {
+ mcpu_macro *macro = list[i];
+ mcpu_text generated;
+ const __mpu_char16_t *replacement;
+ size_t replacement_length;
+ int j;
+
+ mcpu_text_init( &generated );
+
+ if( line_filename != NULL )
+ {
+ const char *marker = macro->origin == MCPU_MACRO_ORIGIN_COMMAND_LINE ?
+ "<command-line>" : "<built-in>";
+
+ if( mcpu_text_append_ascii(output, "# 0 \"" ) != 0 ||
+ mcpu_text_append_utf8(output, marker) != 0 ||
+ mcpu_text_append_ascii(output, "\"\n") != 0 )
+ goto fail;
+ }
+
+ if( mcpu_text_append_ascii(output, "#define ") != 0 ||
+ mcpu_text_append(output, macro->name, macro->name_length) != 0 )
+ goto fail;
+
+ if( macro->nargs >= 0 )
+ {
+ if( mcpu_text_append_char(output, '(') != 0 )
+ goto fail;
+
+ for( j = 0; j < macro->nargs; ++j )
+ {
+ if( j != 0 && mcpu_text_append_char(output, ',') != 0 )
+ goto fail;
+ if( mcpu_text_append(output, macro->argnames[j],
+ macro->argname_lengths[j]) != 0 )
+ goto fail;
+ }
+
+ if( macro->variadic )
+ {
+ if( macro->nargs != 0 && mcpu_text_append_char(output, ',') != 0 )
+ goto fail;
+ if( mcpu_text_append_ascii(output, "...") != 0 )
+ goto fail;
+ }
+
+ if( mcpu_text_append_char(output, ')') != 0 )
+ goto fail;
+ }
+
+ if( macro->builtin == MCPU_MACRO_BUILTIN_NONE )
+ {
+ replacement = macro->replacement;
+ replacement_length = macro->replacement_length;
+ }
+ else
+ {
+ if( expand_builtin(macro, NULL, &generated) != 0 )
+ goto fail;
+ replacement = generated.data;
+ replacement_length = generated.length;
+ }
+
+ if( replacement_length != 0 &&
+ (mcpu_text_append_char(output, ' ') != 0 ||
+ mcpu_text_append(output, replacement, replacement_length) != 0) )
+ goto fail;
+
+ if( mcpu_text_append_char(output, '\n') != 0 )
+ goto fail;
+
+ mcpu_text_free( &generated );
+ continue;
+
+fail:
+ mcpu_text_free( &generated );
+ free( list );
+ return( -1 );
+ }
+
+ free( list );
+ return( 0 );
+}
+
+
+int
+mcpu_macro_expand( mcpu_macro_table *table,
+ const __mpu_char16_t *input, size_t length,
+ enum mcpu_language language,
+ const mcpu_macro_expansion_context *context,
+ mcpu_text *output )
+{
+ if( table == NULL || input == NULL || output == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ mcpu_text_free( output );
+ mcpu_text_init( output );
+
+ return( expand_range(table, input, length, language,
+ context, output, 0) );
+}
diff --git a/src/mcpp-macro.h b/src/mcpp-macro.h
new file mode 100644
index 0000000..719972c
--- /dev/null
+++ b/src/mcpp-macro.h
@@ -0,0 +1,126 @@
+#ifndef __MCPU_CPP_MACRO_H__
+#define __MCPU_CPP_MACRO_H__ 1
+
+#include <defs.h>
+#include <mcpp-language.h>
+#include <mcpp-text.h>
+
+#define MCPU_CPP_MACRO_BUCKETS 257
+#define MCPU_CPP_MACRO_EXPANSION_LIMIT 400
+
+enum mcpu_macro_origin
+{
+ MCPU_MACRO_ORIGIN_SOURCE = 0,
+ MCPU_MACRO_ORIGIN_BUILTIN,
+ MCPU_MACRO_ORIGIN_COMMAND_LINE
+};
+
+
+enum mcpu_macro_builtin
+{
+ MCPU_MACRO_BUILTIN_NONE = 0,
+ MCPU_MACRO_BUILTIN_FILE,
+ MCPU_MACRO_BUILTIN_LINE,
+ MCPU_MACRO_BUILTIN_DATE,
+ MCPU_MACRO_BUILTIN_TIME,
+ MCPU_MACRO_BUILTIN_BASE_FILE,
+ MCPU_MACRO_BUILTIN_INCLUDE_LEVEL,
+ MCPU_MACRO_BUILTIN_REAL_MAX,
+ MCPU_MACRO_BUILTIN_REAL_MIN,
+ MCPU_MACRO_BUILTIN_REAL_EPSILON,
+ MCPU_MACRO_BUILTIN_REAL_MAX_EXP,
+ MCPU_MACRO_BUILTIN_REAL_MIN_EXP,
+ MCPU_MACRO_BUILTIN_REAL_MAX_10_EXP,
+ MCPU_MACRO_BUILTIN_REAL_MIN_10_EXP
+};
+
+typedef struct mcpp_options mcpp_options;
+typedef struct mcpu_macro_expansion_context mcpu_macro_expansion_context;
+struct mcpu_macro_expansion_context
+{
+ const char *filename;
+ const char *base_filename;
+ unsigned line_number;
+ size_t include_level;
+ const char *date;
+ const char *time;
+ const mcpp_options *options;
+};
+
+typedef struct mcpu_macro mcpu_macro;
+struct mcpu_macro
+{
+ __mpu_char16_t *name;
+ size_t name_length;
+
+ int nargs;
+ int variadic;
+ __mpu_char16_t **argnames;
+ size_t *argname_lengths;
+
+ __mpu_char16_t *replacement;
+ size_t replacement_length;
+
+ enum mcpu_macro_builtin builtin;
+ enum mcpu_macro_origin origin;
+ unsigned builtin_bits;
+ int expanding;
+ mcpu_macro *next;
+};
+
+typedef struct mcpu_macro_table mcpu_macro_table;
+struct mcpu_macro_table
+{
+ mcpu_macro *bucket[MCPU_CPP_MACRO_BUCKETS];
+};
+
+void mcpu_macro_table_init( mcpu_macro_table *table );
+void mcpu_macro_table_free( mcpu_macro_table *table );
+mcpu_macro *mcpu_macro_find( mcpu_macro_table *table,
+ const __mpu_char16_t *name,
+ size_t name_length );
+int mcpu_macro_define_object( mcpu_macro_table *table,
+ const __mpu_char16_t *name,
+ size_t name_length,
+ const __mpu_char16_t *replacement,
+ size_t replacement_length,
+ enum mcpu_macro_origin origin );
+int mcpu_macro_define_builtin( mcpu_macro_table *table,
+ const __mpu_char16_t *name,
+ size_t name_length,
+ enum mcpu_macro_builtin builtin );
+int mcpu_macro_define_sized_builtin( mcpu_macro_table *table,
+ const __mpu_char16_t *name,
+ size_t name_length,
+ enum mcpu_macro_builtin builtin,
+ unsigned bits );
+int mcpu_macro_define_function( mcpu_macro_table *table,
+ const __mpu_char16_t *name,
+ size_t name_length,
+ const __mpu_char16_t *const *argnames,
+ const size_t *argname_lengths,
+ size_t nargs, int variadic,
+ const __mpu_char16_t *replacement,
+ size_t replacement_length,
+ enum mcpu_macro_origin origin );
+int mcpu_macro_definition_equal( const mcpu_macro *macro,
+ int nargs, int variadic,
+ const __mpu_char16_t *const *argnames,
+ const size_t *argname_lengths,
+ const __mpu_char16_t *replacement,
+ size_t replacement_length );
+int mcpu_macro_undef( mcpu_macro_table *table,
+ const __mpu_char16_t *name,
+ size_t name_length );
+int mcpu_macro_dump( const mcpu_macro_table *table, FILE *stream,
+ int include_predefined );
+int mcpu_macro_dump_text( const mcpu_macro_table *table, mcpu_text *output,
+ const char *line_filename );
+int mcpu_macro_expand( mcpu_macro_table *table,
+ const __mpu_char16_t *input,
+ size_t length,
+ enum mcpu_language language,
+ const mcpu_macro_expansion_context *context,
+ mcpu_text *output );
+
+#endif /* __MCPU_CPP_MACRO_H__ */
diff --git a/src/mcpp-options.c b/src/mcpp-options.c
new file mode 100644
index 0000000..62fafc6
--- /dev/null
+++ b/src/mcpp-options.c
@@ -0,0 +1,365 @@
+#include <defs.h>
+
+static const char *
+option_argument( mcpp_options *opts, int argc, char **argv, int *i,
+ const char *attached, const char *option )
+{
+ if( attached != NULL && *attached != 0 )
+ return( attached );
+
+ if( *i + 1 >= argc )
+ {
+ fprintf( stderr, "%s: argument missing after '%s'\n",
+ opts->progname, option );
+ return( NULL );
+ }
+
+ ++*i;
+ return( argv[*i] );
+}
+
+static int
+append_action( mcpp_options *opts, enum mcpp_option_action_kind kind,
+ const char *option, const char *argument )
+{
+ if( opts->action_count >= opts->action_capacity )
+ {
+ errno = EOVERFLOW;
+ return( -1 );
+ }
+
+ opts->actions[opts->action_count].kind = kind;
+ opts->actions[opts->action_count].option = option;
+ opts->actions[opts->action_count].argument = argument;
+ ++opts->action_count;
+
+ return( 0 );
+}
+
+static int
+set_filename( mcpp_options *opts, const char *name )
+{
+ if( opts->in_fname == NULL )
+ {
+ opts->in_fname = strcmp(name, "-") == 0 ? "" : name;
+ return( 0 );
+ }
+
+ if( opts->out_fname == NULL )
+ {
+ opts->out_fname = strcmp(name, "-") == 0 ? "" : name;
+ return( 0 );
+ }
+
+ fprintf( stderr, "%s: too many filenames\n", opts->progname );
+ return( -1 );
+}
+
+static int
+parse_dump_option( mcpp_options *opts, const char *arg )
+{
+ const char *p = arg + 2;
+
+ if( *p == 0 )
+ return( -1 );
+
+ while( *p )
+ {
+ switch( *p++ )
+ {
+ case 'M':
+ opts->dump_macros = MCPP_DUMP_ONLY;
+ opts->no_output = 1;
+ while( *p == 'P' )
+ {
+ opts->dump_only_plus_predefined = 1;
+ ++p;
+ }
+ break;
+ case 'D':
+ opts->dump_macros = MCPP_DUMP_DEFINITIONS;
+ break;
+ default:
+ return( -1 );
+ }
+ }
+
+ return( 0 );
+}
+
+void
+mcpp_options_init( mcpp_options *opts, int argc )
+{
+ if( opts == NULL )
+ return;
+
+ memset( opts, 0, sizeof(*opts) );
+ opts->progname = "mcpu-cpp";
+ opts->object_suffix = ".o";
+ opts->action_capacity = argc > 0 ? (size_t)argc * 2 : 2;
+ opts->actions = (mcpp_option_action *)calloc( opts->action_capacity,
+ sizeof(*opts->actions) );
+}
+
+void
+mcpp_options_free( mcpp_options *opts )
+{
+ if( opts == NULL )
+ return;
+
+ free( opts->actions );
+ opts->actions = NULL;
+ opts->action_count = 0;
+ opts->action_capacity = 0;
+}
+
+int
+mcpp_handle_options( mcpp_options *opts, int argc, char **argv )
+{
+ size_t j;
+ int i;
+
+ if( opts == NULL || argv == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ for( i = 0; i < argc; ++i )
+ {
+ const char *arg = argv[i];
+ const char *value;
+
+ if( arg[0] != '-' || arg[1] == 0 )
+ {
+ if( set_filename(opts, arg) != 0 ) return( -1 );
+ continue;
+ }
+
+ if( strcmp(arg, "--help") == 0 ) opts->help = 1;
+ else if( strcmp(arg, "--version") == 0 ) opts->version = 1;
+ else if( strcmp(arg, "--no-config") == 0 ) opts->no_config = 1;
+ else if( strcmp(arg, "--config-file") == 0 )
+ {
+ value = option_argument(opts, argc, argv, &i, NULL, arg);
+ if( value == NULL ) return( -1 );
+ opts->config_fname = value;
+ }
+ else if( strcmp(arg, "--object-suffix") == 0 )
+ {
+ value = option_argument(opts, argc, argv, &i, NULL, arg);
+ if( value == NULL ) return( -1 );
+ opts->object_suffix = value;
+ }
+ else if( strcmp(arg, "-dconfig") == 0 ) opts->dump_config = 1;
+ else if( strcmp(arg, "-dsearch-dirs") == 0 ) opts->dump_search_dirs = 1;
+ else if( arg[0] == '-' && arg[1] == 'd' && arg[2] != 0 )
+ {
+ if( parse_dump_option(opts, arg) != 0 ) goto unknown;
+ }
+ else if( strcmp(arg, "-v") == 0 || strcmp(arg, "--verbose") == 0 ) opts->verbose = 1;
+ else if( strcmp(arg, "-nostdinc") == 0 ) opts->no_standard_includes = 1;
+ else if( strcmp(arg, "-nostdinc++") == 0 ) opts->no_standard_kplusplus_includes = 1;
+ else if( strcmp(arg, "-E") == 0 ) { }
+ else if( strcmp(arg, "-w") == 0 ) opts->inhibit_warnings = 1;
+ else if( strcmp(arg, "-Wcomment") == 0 || strcmp(arg, "-Wcomments") == 0 )
+ {
+ opts->warn_comments = 1;
+ opts->warn_comments_explicit = 1;
+ }
+ else if( strcmp(arg, "-Wno-comment") == 0 || strcmp(arg, "-Wno-comments") == 0 )
+ {
+ opts->warn_comments = 0;
+ opts->warn_comments_explicit = 1;
+ }
+ else if( strcmp(arg, "-Wall") == 0 )
+ {
+ if( !opts->warn_comments_explicit )
+ opts->warn_comments = 1;
+ }
+ else if( strcmp(arg, "-Werror") == 0 ) opts->warnings_are_errors = 1;
+ else if( strcmp(arg, "-Wno-error") == 0 ) opts->warnings_are_errors = 0;
+ else if( strcmp(arg, "-lint") == 0 ) opts->for_lint = 1;
+ else if( strcmp(arg, "-M") == 0 )
+ {
+ opts->print_deps = 2;
+ opts->inhibit_output = 1;
+ }
+ else if( strcmp(arg, "-MM") == 0 )
+ {
+ opts->print_deps = 1;
+ opts->inhibit_output = 1;
+ }
+ else if( strcmp(arg, "-MD") == 0 )
+ {
+ opts->print_deps = 2;
+ opts->inhibit_output = 0;
+ }
+ else if( strcmp(arg, "-MMD") == 0 )
+ {
+ opts->print_deps = 1;
+ opts->inhibit_output = 0;
+ }
+ else if( arg[0] == '-' && arg[1] == 'M' && arg[2] == 'F' )
+ {
+ value = option_argument(opts, argc, argv, &i, arg + 3, "-MF");
+ if( value == NULL ) return( -1 );
+ opts->deps_file = value;
+ }
+ else if( arg[0] == '-' && arg[1] == 'M' &&
+ (arg[2] == 'T' || arg[2] == 'Q') )
+ {
+ enum mcpp_option_action_kind kind = arg[2] == 'T' ?
+ MCPP_ACTION_DEP_TARGET : MCPP_ACTION_DEP_TARGET_QUOTED;
+ const char *option = arg[2] == 'T' ? "-MT" : "-MQ";
+
+ value = option_argument(opts, argc, argv, &i, arg + 3, option);
+ if( value == NULL || append_action(opts, kind, option, value) != 0 )
+ return( -1 );
+ }
+ else if( strcmp(arg, "-MG") == 0 ) opts->print_deps_missing_files = 1;
+ else if( strcmp(arg, "-o") == 0 )
+ {
+ if( opts->out_fname != NULL )
+ {
+ fprintf(stderr, "%s: output filename specified twice\n", opts->progname);
+ return( -1 );
+ }
+ value = option_argument(opts, argc, argv, &i, NULL, arg);
+ if( value == NULL ) return( -1 );
+ opts->out_fname = strcmp(value, "-") == 0 ? "" : value;
+ }
+ else if( arg[1] == 'D' )
+ {
+ value = option_argument(opts, argc, argv, &i, arg + 2, "-D");
+ if( value == NULL || append_action(opts, MCPP_ACTION_DEFINE, "-D", value) != 0 ) return( -1 );
+ }
+ else if( arg[1] == 'U' )
+ {
+ value = option_argument(opts, argc, argv, &i, arg + 2, "-U");
+ if( value == NULL || append_action(opts, MCPP_ACTION_UNDEF, "-U", value) != 0 ) return( -1 );
+ }
+ else if( arg[1] == 'I' && arg[2] == '-' && arg[3] == 0 ) goto unknown;
+ else if( arg[1] == 'I' )
+ {
+ value = option_argument(opts, argc, argv, &i, arg + 2, "-I");
+ if( value == NULL || append_action(opts, MCPP_ACTION_INCLUDE_USER, "-I", value) != 0 ) return( -1 );
+ }
+ else if( strcmp(arg, "-include") == 0 || strcmp(arg, "-imacros") == 0 ||
+ strcmp(arg, "-isystem") == 0 || strcmp(arg, "-idirafter") == 0 )
+ {
+ enum mcpp_option_action_kind kind = MCPP_ACTION_INCLUDE;
+ if( strcmp(arg, "-imacros") == 0 ) kind = MCPP_ACTION_IMACROS;
+ else if( strcmp(arg, "-isystem") == 0 ) kind = MCPP_ACTION_INCLUDE_SYSTEM;
+ else if( strcmp(arg, "-idirafter") == 0 ) kind = MCPP_ACTION_INCLUDE_AFTER;
+ value = option_argument(opts, argc, argv, &i, NULL, arg);
+ if( value == NULL || append_action(opts, kind, arg, value) != 0 ) return( -1 );
+ }
+ else
+ {
+unknown:
+ fprintf( stderr, "%s: unknown option '%s'\n", opts->progname, arg );
+ return( -1 );
+ }
+ }
+
+ if( opts->in_fname == NULL ) opts->in_fname = "";
+ if( opts->out_fname == NULL ) opts->out_fname = "";
+
+ if( opts->deps_file != NULL && opts->print_deps == 0 )
+ {
+ fprintf( stderr,
+ "%s: -MF requires one of -M, -MM, -MD or -MMD\n",
+ opts->progname );
+ return( -1 );
+ }
+
+ if( opts->print_deps == 0 )
+ {
+ for( j = 0; j < opts->action_count; ++j )
+ if( opts->actions[j].kind == MCPP_ACTION_DEP_TARGET ||
+ opts->actions[j].kind == MCPP_ACTION_DEP_TARGET_QUOTED )
+ {
+ fprintf( stderr,
+ "%s: -MT/-MQ require one of -M, -MM, -MD or -MMD\n",
+ opts->progname );
+ return( -1 );
+ }
+ }
+
+ if( opts->print_deps_missing_files &&
+ (opts->print_deps == 0 || !opts->inhibit_output) )
+ {
+ fprintf( stderr, "%s: -MG may only be used with -M or -MM\n",
+ opts->progname );
+ return( -1 );
+ }
+
+ return( 0 );
+}
+
+static const char *
+action_name( enum mcpp_option_action_kind kind )
+{
+ switch( kind )
+ {
+ case MCPP_ACTION_DEFINE: return( "define" );
+ case MCPP_ACTION_UNDEF: return( "undef" );
+ case MCPP_ACTION_INCLUDE: return( "include" );
+ case MCPP_ACTION_IMACROS: return( "imacros" );
+ case MCPP_ACTION_INCLUDE_USER: return( "include-user" );
+ case MCPP_ACTION_INCLUDE_SYSTEM: return( "include-system" );
+ case MCPP_ACTION_INCLUDE_AFTER: return( "include-after" );
+ case MCPP_ACTION_DEP_TARGET: return( "dep-target" );
+ case MCPP_ACTION_DEP_TARGET_QUOTED: return( "dep-target-quoted" );
+ }
+ return( "unknown" );
+}
+
+int
+mcpp_options_dump( const mcpp_options *opts, FILE *stream )
+{
+ size_t i;
+
+ if( opts == NULL || stream == NULL ) return( -1 );
+
+ fprintf(stream, "input=%s\n", *opts->in_fname ? opts->in_fname : "<stdin>");
+ fprintf(stream, "output=%s\n", *opts->out_fname ? opts->out_fname : "<stdout>");
+ fprintf(stream, "verbose=%d\n", opts->verbose);
+ fprintf(stream, "nostdinc=%d\n", opts->no_standard_includes);
+ fprintf(stream, "dump_config=%d\n", opts->dump_config);
+ fprintf(stream, "dump_search_dirs=%d\n", opts->dump_search_dirs);
+ fprintf(stream, "dump_macros=%d\n", (int)opts->dump_macros);
+ fprintf(stream, "actions=%lu\n", (unsigned long)opts->action_count);
+ for( i = 0; i < opts->action_count; ++i )
+ fprintf(stream, "action[%lu]=%s:%s\n", (unsigned long)i,
+ action_name(opts->actions[i].kind), opts->actions[i].argument);
+ return( 0 );
+}
+
+int
+mcpp_options_has_unimplemented( const mcpp_options *opts )
+{
+ size_t i;
+ if( opts == NULL ) return( 0 );
+ if( opts->print_deps && opts->inhibit_output &&
+ opts->dump_macros != MCPP_DUMP_NONE )
+ return( 1 );
+ if( opts->print_deps && !opts->inhibit_output &&
+ opts->dump_macros == MCPP_DUMP_ONLY )
+ return( 1 );
+ if( opts->no_standard_kplusplus_includes || opts->for_lint )
+ return( 1 );
+ for( i = 0; i < opts->action_count; ++i )
+ if( opts->actions[i].kind != MCPP_ACTION_DEFINE &&
+ opts->actions[i].kind != MCPP_ACTION_UNDEF &&
+ opts->actions[i].kind != MCPP_ACTION_INCLUDE &&
+ opts->actions[i].kind != MCPP_ACTION_IMACROS &&
+ opts->actions[i].kind != MCPP_ACTION_INCLUDE_USER &&
+ opts->actions[i].kind != MCPP_ACTION_INCLUDE_SYSTEM &&
+ opts->actions[i].kind != MCPP_ACTION_INCLUDE_AFTER &&
+ opts->actions[i].kind != MCPP_ACTION_DEP_TARGET &&
+ opts->actions[i].kind != MCPP_ACTION_DEP_TARGET_QUOTED )
+ return( 1 );
+ return( 0 );
+}
diff --git a/src/mcpp-options.h b/src/mcpp-options.h
new file mode 100644
index 0000000..910db92
--- /dev/null
+++ b/src/mcpp-options.h
@@ -0,0 +1,74 @@
+#ifndef __MCPU_CPP_OPTIONS_H__
+#define __MCPU_CPP_OPTIONS_H__ 1
+
+enum mcpp_dump_macros
+{
+ MCPP_DUMP_NONE = 0,
+ MCPP_DUMP_ONLY,
+ MCPP_DUMP_DEFINITIONS
+};
+
+enum mcpp_option_action_kind
+{
+ MCPP_ACTION_DEFINE = 0,
+ MCPP_ACTION_UNDEF,
+ MCPP_ACTION_INCLUDE,
+ MCPP_ACTION_IMACROS,
+ MCPP_ACTION_INCLUDE_USER,
+ MCPP_ACTION_INCLUDE_SYSTEM,
+ MCPP_ACTION_INCLUDE_AFTER,
+ MCPP_ACTION_DEP_TARGET,
+ MCPP_ACTION_DEP_TARGET_QUOTED
+};
+
+typedef struct mcpp_option_action mcpp_option_action;
+struct mcpp_option_action
+{
+ enum mcpp_option_action_kind kind;
+ const char *option;
+ const char *argument;
+};
+
+typedef struct mcpp_options mcpp_options;
+struct mcpp_options
+{
+ const char *progname;
+ const char *in_fname;
+ const char *out_fname;
+ const char *config_fname;
+ const char *deps_file;
+ const char *deps_target;
+ const char *object_suffix;
+
+ int help;
+ int version;
+ int verbose;
+ int no_config;
+ int dump_config;
+ int dump_search_dirs;
+ int no_standard_includes;
+ int no_standard_kplusplus_includes;
+ int inhibit_output;
+ int no_output;
+ int print_deps;
+ int print_deps_missing_files;
+ int inhibit_warnings;
+ int warn_comments;
+ int warn_comments_explicit;
+ int warnings_are_errors;
+ int for_lint;
+ enum mcpp_dump_macros dump_macros;
+ int dump_only_plus_predefined;
+
+ mcpp_option_action *actions;
+ size_t action_count;
+ size_t action_capacity;
+};
+
+void mcpp_options_init( mcpp_options *opts, int argc );
+void mcpp_options_free( mcpp_options *opts );
+int mcpp_handle_options( mcpp_options *opts, int argc, char **argv );
+int mcpp_options_dump( const mcpp_options *opts, FILE *stream );
+int mcpp_options_has_unimplemented( const mcpp_options *opts );
+
+#endif
diff --git a/src/mcpp-predefined.c b/src/mcpp-predefined.c
new file mode 100644
index 0000000..556232c
--- /dev/null
+++ b/src/mcpp-predefined.c
@@ -0,0 +1,627 @@
+#include <defs.h>
+
+
+static int
+define_utf8( mcpu_macro_table *table, const char *name,
+ const char *replacement )
+{
+ mcpu_text n;
+ mcpu_text r;
+ int rc = -1;
+
+ mcpu_text_init( &n );
+ mcpu_text_init( &r );
+
+ if( mcpu_text_from_utf8(&n, name) != 0 ||
+ mcpu_text_from_utf8(&r, replacement ? replacement : "") != 0 )
+ goto done;
+
+ rc = mcpu_macro_define_object( table, n.data, n.length,
+ r.data, r.length,
+ MCPU_MACRO_ORIGIN_BUILTIN );
+
+done:
+ mcpu_text_free( &n );
+ mcpu_text_free( &r );
+ return( rc );
+}
+
+
+static int
+define_unsigned( mcpu_macro_table *table, const char *name,
+ unsigned long value )
+{
+ char text[64];
+
+ snprintf( text, sizeof(text), "%lu", value );
+ return( define_utf8(table, name, text) );
+}
+
+
+static int
+define_quoted( mcpu_macro_table *table, const char *name,
+ const char *value )
+{
+ size_t n;
+ char *text;
+ size_t i;
+ size_t o = 0;
+ int rc;
+
+ if( value == NULL )
+ value = "";
+
+ n = strlen( value );
+ if( n > (SIZE_MAX - 3) / 2 )
+ {
+ errno = EOVERFLOW;
+ return( -1 );
+ }
+
+ text = (char *)malloc( n * 2 + 3 );
+ if( text == NULL )
+ return( -1 );
+
+ text[o++] = '"';
+ for( i = 0; i < n; ++i )
+ {
+ if( value[i] == '\\' || value[i] == '"' )
+ text[o++] = '\\';
+ text[o++] = value[i];
+ }
+ text[o++] = '"';
+ text[o] = 0;
+
+ rc = define_utf8( table, name, text );
+ free( text );
+ return( rc );
+}
+
+
+static int
+define_from_macro( mcpu_macro_table *table, const char *name,
+ const char *source_name )
+{
+ mcpu_text source;
+ mcpu_text target;
+ mcpu_macro *macro;
+ int rc = -1;
+
+ mcpu_text_init( &source );
+ mcpu_text_init( &target );
+
+ if( mcpu_text_from_utf8(&source, source_name) != 0 ||
+ mcpu_text_from_utf8(&target, name) != 0 )
+ goto done;
+
+ macro = mcpu_macro_find( table, source.data, source.length );
+ if( macro == NULL || macro->builtin != MCPU_MACRO_BUILTIN_NONE )
+ {
+ errno = EINVAL;
+ goto done;
+ }
+
+ rc = mcpu_macro_define_object( table, target.data, target.length,
+ macro->replacement,
+ macro->replacement_length,
+ MCPU_MACRO_ORIGIN_BUILTIN );
+
+done:
+ mcpu_text_free( &source );
+ mcpu_text_free( &target );
+ return( rc );
+}
+
+
+static int
+define_sized_builtin_utf8( mcpu_macro_table *table, const char *name,
+ enum mcpu_macro_builtin builtin, unsigned bits )
+{
+ mcpu_text n;
+ int rc;
+
+ mcpu_text_init( &n );
+ if( mcpu_text_from_utf8(&n, name) != 0 )
+ return( -1 );
+
+ rc = mcpu_macro_define_sized_builtin( table, n.data, n.length,
+ builtin, bits );
+ mcpu_text_free( &n );
+ return( rc );
+}
+
+
+static unsigned long
+mcpp_uint_decimal_digs( unsigned bits )
+{
+ uint64_t n;
+
+ if( bits < 1 )
+ return( 0 );
+
+ /*
+ * Number of decimal digits in 2^bits - 1. This follows the integer-only
+ * scheme used by LibMPU _int_digs(), but does not include a terminating
+ * NUL. The tighter upper approximation keeps the result exact for every
+ * integer width supported by the current LibMPU family through 65536 bits.
+ *
+ * log10(2) < 301029995664 / 1000000000000
+ */
+ n = ((uint64_t)bits * UINT64_C(301029995664) +
+ UINT64_C(999999999999)) / UINT64_C(1000000000000);
+
+ return( (unsigned long)n );
+}
+
+
+static unsigned long
+mcpp_int_decimal_digs( unsigned bits )
+{
+ if( bits < 2 )
+ return( 0 );
+
+ /* Signed maximum is 2^(bits - 1) - 1; sign is not part of DECIMAL_DIG. */
+ return( mcpp_uint_decimal_digs(bits - 1) );
+}
+
+
+static int
+define_fixed_integer_family( mcpu_macro_table *table, unsigned bits )
+{
+ size_t nb = (size_t)bits / 8;
+ mpu_int *value = NULL;
+ char *text = NULL;
+ char name[96];
+ size_t text_size;
+ unsigned long int_digs;
+ unsigned long uint_digs;
+ int rc = -1;
+
+ if( bits == 0 || (bits % 8) != 0 )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ snprintf( name, sizeof(name), "__INT%u_TYPE__", bits );
+ {
+ char type[64];
+ snprintf( type, sizeof(type), "int%u", bits );
+ if( define_utf8(table, name, type) != 0 )
+ return( -1 );
+ }
+
+ snprintf( name, sizeof(name), "__UINT%u_TYPE__", bits );
+ {
+ char type[64];
+ snprintf( type, sizeof(type), "uint%u", bits );
+ if( define_utf8(table, name, type) != 0 )
+ return( -1 );
+ }
+
+ snprintf( name, sizeof(name), "__INT%u_WIDTH__", bits );
+ if( define_unsigned(table, name, bits) != 0 )
+ return( -1 );
+ snprintf( name, sizeof(name), "__UINT%u_WIDTH__", bits );
+ if( define_unsigned(table, name, bits) != 0 )
+ return( -1 );
+
+ /*
+ * TYPE/WIDTH/SIZEOF and decimal-digit metadata remain available for the
+ * complete LibMPU integer family through NB_I_MAX * 8. Only textual MAX
+ * values are capped at 256 bits so -dMP stays compact and useful.
+ */
+ snprintf( name, sizeof(name), "__SIZEOF_INT%u__", bits );
+ if( define_unsigned(table, name, bits / 8) != 0 )
+ return( -1 );
+ snprintf( name, sizeof(name), "__SIZEOF_UINT%u__", bits );
+ if( define_unsigned(table, name, bits / 8) != 0 )
+ return( -1 );
+
+ int_digs = mcpp_int_decimal_digs( bits );
+ uint_digs = mcpp_uint_decimal_digs( bits );
+ if( int_digs == 0 || uint_digs == 0 )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+ snprintf( name, sizeof(name), "__INT%u_DECIMAL_DIG__", bits );
+ if( define_unsigned(table, name, int_digs) != 0 )
+ return( -1 );
+ snprintf( name, sizeof(name), "__UINT%u_DECIMAL_DIG__", bits );
+ if( define_unsigned(table, name, uint_digs) != 0 )
+ return( -1 );
+
+ if( bits > 256 )
+ return( 0 );
+
+ value = (mpu_int *)malloc( nb );
+ if( value == NULL )
+ return( -1 );
+
+ text_size = (size_t)bits + 3;
+ text = (char *)calloc( text_size, 1 );
+ if( text == NULL )
+ goto done;
+
+ memset( value, 0xff, nb );
+ iuitoa( (__mpu_char8_t *)text, value, RADIX_HEX, LOWERCASE, (int)nb );
+
+ snprintf( name, sizeof(name), "__UINT%u_MAX__", bits );
+ if( define_utf8(table, name, text) != 0 )
+ goto done;
+
+ /*
+ * iuitoa() is the canonical LibMPU conversion. For the signed maximum
+ * the all-ones unsigned bit pattern has only its top bit cleared.
+ */
+ if( strlen(text) < 3 || text[0] != '0' || text[1] != 'x' )
+ {
+ errno = EINVAL;
+ goto done;
+ }
+ text[2] = '7';
+
+ snprintf( name, sizeof(name), "__INT%u_MAX__", bits );
+ if( define_utf8(table, name, text) != 0 )
+ goto done;
+
+ rc = 0;
+
+done:
+ free( text );
+ free( value );
+ return( rc );
+}
+
+typedef void (*real_value_fn)( mpu_real *, unsigned int, int );
+
+static int
+append_real_value( mcpu_text *output, unsigned bits, real_value_fn fn )
+{
+ int nb = (int)(bits / 8);
+ int max_string;
+ mpu_real *value;
+ char *text;
+ int rc = -1;
+
+ max_string = _real_max_string( nb );
+ if( max_string <= 0 )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ value = (mpu_real *)calloc( (size_t)nb, 1 );
+ text = (char *)calloc( (size_t)max_string + 4096, 1 );
+ if( value == NULL || text == NULL )
+ goto done;
+
+ fn( value, 0, nb );
+ real_to_ascii( (__mpu_char8_t *)text, value,
+ _real_mant_digs(nb), 'e', 0, 0, nb );
+ if( text[0] == 0 )
+ {
+ errno = EINVAL;
+ goto done;
+ }
+
+ rc = mcpu_text_append_ascii( output, text );
+
+done:
+ free( text );
+ free( value );
+ return( rc );
+}
+
+
+static void
+real_epsilon_wrapper( mpu_real *value, unsigned int sign, int nb )
+{
+ (void)sign;
+ _real_epsilon( value, nb );
+}
+
+
+static int
+append_real_exponent( mcpu_text *output, unsigned bits,
+ void (*fn)(mpu_int *, int, int) )
+{
+ int nb_r = (int)(bits / 8);
+ int nb_e = _sizeof_exp( nb_r );
+ mpu_int *value;
+ char *text;
+ size_t text_size;
+ int rc = -1;
+
+ if( nb_e <= 0 )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ value = (mpu_int *)calloc( (size_t)nb_e, 1 );
+ text_size = (size_t)nb_e * 8 + 3;
+ text = (char *)calloc( text_size, 1 );
+ if( value == NULL || text == NULL )
+ goto done;
+
+ fn( value, nb_e, nb_r );
+ iitoa( (__mpu_char8_t *)text, value, RADIX_DEC, LOWERCASE, nb_e );
+ if( text[0] == 0 )
+ {
+ errno = EINVAL;
+ goto done;
+ }
+
+ rc = mcpu_text_append_ascii( output, text );
+
+done:
+ free( text );
+ free( value );
+ return( rc );
+}
+
+
+static int
+define_real_family( mcpu_macro_table *table, unsigned bits )
+{
+ char name[96];
+ char type[64];
+ int nb = (int)(bits / 8);
+
+ /*
+ * Real and Complex language types exist only through the configured
+ * MPU_REAL_IO_LIMIT. Complex WIDTH is the width parameter in the language
+ * type name (complex128 -> 128), while SIZEOF is twice the Real storage.
+ */
+ if( bits > MCPU_CPP_MPU_REAL_IO_LIMIT )
+ return( 0 );
+
+ snprintf( name, sizeof(name), "__REAL%u_TYPE__", bits );
+ snprintf( type, sizeof(type), "real%u", bits );
+ if( define_utf8(table, name, type) != 0 )
+ return( -1 );
+
+ snprintf( name, sizeof(name), "__COMPLEX%u_TYPE__", bits );
+ snprintf( type, sizeof(type), "complex%u", bits );
+ if( define_utf8(table, name, type) != 0 )
+ return( -1 );
+
+ snprintf( name, sizeof(name), "__REAL%u_WIDTH__", bits );
+ if( define_unsigned(table, name, bits) != 0 )
+ return( -1 );
+ snprintf( name, sizeof(name), "__COMPLEX%u_WIDTH__", bits );
+ if( define_unsigned(table, name, bits) != 0 )
+ return( -1 );
+
+ snprintf( name, sizeof(name), "__SIZEOF_REAL%u__", bits );
+ if( define_unsigned(table, name, bits / 8) != 0 )
+ return( -1 );
+ snprintf( name, sizeof(name), "__SIZEOF_COMPLEX%u__", bits );
+ if( define_unsigned(table, name, bits / 4) != 0 )
+ return( -1 );
+
+ /*
+ * These are compact structural/text-conversion properties. They are
+ * available for every configured Real type through MPU_REAL_IO_LIMIT.
+ * _real_max_string() returns a character count, not a byte count.
+ */
+ {
+ int exp_size = _sizeof_exp( nb );
+ int max_strlen = _real_max_string( nb );
+
+ if( exp_size <= 0 || max_strlen <= 0 )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ snprintf( name, sizeof(name), "__SIZEOF_REAL%u_EXP__", bits );
+ if( define_unsigned(table, name, (unsigned long)exp_size) != 0 )
+ return( -1 );
+
+ snprintf( name, sizeof(name), "__REAL%u_MAX_STRLEN__", bits );
+ if( define_unsigned(table, name, (unsigned long)max_strlen) != 0 )
+ return( -1 );
+ }
+
+ /*
+ * Precision metadata is compact and is useful for every configured Real
+ * width. Only large textual numeric values are capped at 256 bits.
+ */
+ snprintf( name, sizeof(name), "__REAL%u_MANT_DIG__", bits );
+ if( define_unsigned(table, name, (unsigned long)_real_mant_digs(nb)) != 0 )
+ return( -1 );
+
+ snprintf( name, sizeof(name), "__REAL%u_DECIMAL_DIG__", bits );
+ if( define_unsigned(table, name, (unsigned long)_real_digs(nb)) != 0 )
+ return( -1 );
+
+ if( bits > 256 )
+ return( 0 );
+
+ snprintf( name, sizeof(name), "__REAL%u_MAX__", bits );
+ if( define_sized_builtin_utf8(table, name,
+ MCPU_MACRO_BUILTIN_REAL_MAX, bits) != 0 )
+ return( -1 );
+
+ snprintf( name, sizeof(name), "__REAL%u_MIN__", bits );
+ if( define_sized_builtin_utf8(table, name,
+ MCPU_MACRO_BUILTIN_REAL_MIN, bits) != 0 )
+ return( -1 );
+
+ snprintf( name, sizeof(name), "__REAL%u_EPSILON__", bits );
+ if( define_sized_builtin_utf8(table, name,
+ MCPU_MACRO_BUILTIN_REAL_EPSILON, bits) != 0 )
+ return( -1 );
+
+ snprintf( name, sizeof(name), "__REAL%u_MAX_EXP__", bits );
+ if( define_sized_builtin_utf8(table, name,
+ MCPU_MACRO_BUILTIN_REAL_MAX_EXP, bits) != 0 )
+ return( -1 );
+ snprintf( name, sizeof(name), "__REAL%u_MIN_EXP__", bits );
+ if( define_sized_builtin_utf8(table, name,
+ MCPU_MACRO_BUILTIN_REAL_MIN_EXP, bits) != 0 )
+ return( -1 );
+ snprintf( name, sizeof(name), "__REAL%u_MAX_10_EXP__", bits );
+ if( define_sized_builtin_utf8(table, name,
+ MCPU_MACRO_BUILTIN_REAL_MAX_10_EXP, bits) != 0 )
+ return( -1 );
+ snprintf( name, sizeof(name), "__REAL%u_MIN_10_EXP__", bits );
+ if( define_sized_builtin_utf8(table, name,
+ MCPU_MACRO_BUILTIN_REAL_MIN_10_EXP, bits) != 0 )
+ return( -1 );
+
+ return( 0 );
+}
+
+int
+mcpu_predefined_expand_builtin( enum mcpu_macro_builtin builtin,
+ unsigned bits,
+ mcpu_text *output )
+{
+ if( output == NULL || bits == 0 || bits > 256 ||
+ bits > MCPU_CPP_MPU_REAL_IO_LIMIT )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ switch( builtin )
+ {
+ case MCPU_MACRO_BUILTIN_REAL_MAX:
+ return( append_real_value(output, bits, _real_max) );
+ case MCPU_MACRO_BUILTIN_REAL_MIN:
+ return( append_real_value(output, bits, _real_min) );
+ case MCPU_MACRO_BUILTIN_REAL_EPSILON:
+ return( append_real_value(output, bits, real_epsilon_wrapper) );
+ case MCPU_MACRO_BUILTIN_REAL_MAX_EXP:
+ return( append_real_exponent(output, bits, _real_max_exp) );
+ case MCPU_MACRO_BUILTIN_REAL_MIN_EXP:
+ return( append_real_exponent(output, bits, _real_min_exp) );
+ case MCPU_MACRO_BUILTIN_REAL_MAX_10_EXP:
+ return( append_real_exponent(output, bits, _real_max_10_exp) );
+ case MCPU_MACRO_BUILTIN_REAL_MIN_10_EXP:
+ return( append_real_exponent(output, bits, _real_min_10_exp) );
+ default:
+ errno = EINVAL;
+ return( -1 );
+ }
+}
+
+
+int
+mcpu_predefined_install( mcpu_macro_table *table )
+{
+ unsigned bits;
+ char type[64];
+ char ref[96];
+
+ if( table == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ if( define_utf8(table, "_ARCH_MCPU", "1") != 0 ||
+ define_quoted(table, "__MCPU_CPP_VERSION__", PACKAGE_VERSION) != 0 ||
+ define_utf8(table, "__REGISTER_PREFIX__", "") != 0 ||
+ define_utf8(table, "__LOCAL_LABEL_PREFIX__", "") != 0 ||
+ define_utf8(table, "__USER_LABEL_PREFIX__", "") != 0 ||
+ define_utf8(table, "__IMMEDIATE_PREFIX__", "") != 0 ||
+ define_utf8(table, "__ORDER_LITTLE_ENDIAN__", "1234") != 0 ||
+ define_utf8(table, "__ORDER_BIG_ENDIAN__", "4321") != 0 ||
+ define_utf8(table, "__ORDER_PDP_ENDIAN__", "3412") != 0 )
+ return( -1 );
+
+#if MCPU_CPP_MPU_BYTE_ORDER == 1234
+ if( define_utf8(table, "__MCPU_BYTE_ORDER__", "__ORDER_LITTLE_ENDIAN__") != 0 )
+ return( -1 );
+#else
+ if( define_utf8(table, "__MCPU_BYTE_ORDER__", "__ORDER_BIG_ENDIAN__") != 0 )
+ return( -1 );
+#endif
+ if( define_utf8(table, "__BYTE_ORDER__", "__MCPU_BYTE_ORDER__") != 0 )
+ return( -1 );
+
+#if MCPU_CPP_MPU_WORD_ORDER == 1234
+ if( define_utf8(table, "__MCPU_WORD_ORDER__", "__ORDER_LITTLE_ENDIAN__") != 0 )
+ return( -1 );
+#else
+ if( define_utf8(table, "__MCPU_WORD_ORDER__", "__ORDER_BIG_ENDIAN__") != 0 )
+ return( -1 );
+#endif
+
+ if( define_unsigned(table, "__MCPU_MACHINE_REGISTER_WIDTH__",
+ MCPU_CPP_MPU_REGISTER_WIDTH) != 0 ||
+ define_unsigned(table, "__MCPU_REAL_IO_LIMIT__",
+ MCPU_CPP_MPU_REAL_IO_LIMIT) != 0 ||
+ define_unsigned(table, "__MCPU_MATH_FN_LIMIT__",
+ MCPU_CPP_MPU_MATH_FN_LIMIT) != 0 ||
+ define_unsigned(table, "__MCPU_INT_MAX_WIDTH__",
+ MCPU_CPP_MPU_INT_MAX_WIDTH) != 0 ||
+ define_unsigned(table, "__MCPU_REAL_MAX_WIDTH__",
+ MCPU_CPP_MPU_REAL_IO_LIMIT) != 0 ||
+ define_unsigned(table, "__MCPU_COMPLEX_MAX_WIDTH__",
+ MCPU_CPP_MPU_REAL_IO_LIMIT) != 0 ||
+ define_unsigned(table, "__SIZEOF_POINTER__", 8) != 0 ||
+ define_unsigned(table, "__MCPU_POINTER_WIDTH__", 64) != 0 ||
+ define_unsigned(table, "__MCPU_SIZEOF_SIZE__", MCPU_CPP_SIZEOF_SIZE_T) != 0 ||
+ define_unsigned(table, "__MCPU_SIZE_WIDTH__", MCPU_CPP_SIZE_WIDTH) != 0 ||
+ define_unsigned(table, "__SIZEOF_PTRDIFF__", 8) != 0 ||
+ define_unsigned(table, "__PTRDIFF_WIDTH__", 64) != 0 ||
+ define_utf8(table, "__CHAR8_TYPE__", "char8") != 0 ||
+ define_utf8(table, "__CHAR16_TYPE__", "char16") != 0 ||
+ define_unsigned(table, "__CHAR8_WIDTH__", 8) != 0 ||
+ define_unsigned(table, "__CHAR16_WIDTH__", 16) != 0 ||
+ define_unsigned(table, "__SIZEOF_CHAR8__", 1) != 0 ||
+ define_unsigned(table, "__SIZEOF_CHAR16__", 2) != 0 ||
+ define_utf8(table, "__CHAR8_MAX__", "0xff") != 0 ||
+ define_utf8(table, "__CHAR16_MAX__", "0xffff") != 0 )
+ return( -1 );
+
+ for( bits = 8; bits != 0 && bits <= MCPU_CPP_MPU_INT_MAX_WIDTH; bits *= 2 )
+ {
+ if( define_fixed_integer_family(table, bits) != 0 )
+ return( -1 );
+ }
+
+ snprintf( type, sizeof(type), "uint%u", (unsigned)MCPU_CPP_SIZE_WIDTH );
+ if( define_utf8(table, "__MCPU_SIZE_TYPE__", type) != 0 )
+ return( -1 );
+ snprintf( ref, sizeof(ref), "__UINT%u_MAX__", (unsigned)MCPU_CPP_SIZE_WIDTH );
+ if( define_from_macro(table, "__MCPU_SIZE_MAX__", ref) != 0 )
+ return( -1 );
+
+ snprintf( type, sizeof(type), "int%u", (unsigned)MCPU_CPP_SSIZE_WIDTH );
+ if( define_utf8(table, "__MCPU_SSIZE_TYPE__", type) != 0 ||
+ define_unsigned(table, "__MCPU_SSIZE_WIDTH__", MCPU_CPP_SSIZE_WIDTH) != 0 ||
+ define_unsigned(table, "__MCPU_SIZEOF_SSIZE__", MCPU_CPP_SIZEOF_SSIZE_T) != 0 )
+ return( -1 );
+ if( MCPU_CPP_SSIZE_WIDTH <= 256 )
+ {
+ snprintf( ref, sizeof(ref), "__INT%u_MAX__", (unsigned)MCPU_CPP_SSIZE_WIDTH );
+ if( define_from_macro(table, "__MCPU_SSIZE_MAX__", ref) != 0 )
+ return( -1 );
+ }
+
+ if( define_utf8(table, "__PTRDIFF_TYPE__", "int64") != 0 ||
+ define_unsigned(table, "__PTRDIFF_WIDTH__", 64) != 0 ||
+ define_utf8(table, "__PTRDIFF_MAX__", "0x7fffffffffffffff") != 0 ||
+ define_utf8(table, "__INTPTR_TYPE__", "int64") != 0 ||
+ define_utf8(table, "__UINTPTR_TYPE__", "uint64") != 0 ||
+ define_unsigned(table, "__INTPTR_WIDTH__", 64) != 0 ||
+ define_unsigned(table, "__UINTPTR_WIDTH__", 64) != 0 ||
+ define_utf8(table, "__INTPTR_MAX__", "0x7fffffffffffffff") != 0 ||
+ define_utf8(table, "__UINTPTR_MAX__", "0xffffffffffffffff") != 0 )
+ return( -1 );
+
+ for( bits = 32; bits != 0 && bits <= MCPU_CPP_MPU_REAL_IO_LIMIT; bits *= 2 )
+ {
+ if( define_real_family(table, bits) != 0 )
+ return( -1 );
+ }
+
+ return( 0 );
+}
diff --git a/src/mcpp-predefined.h b/src/mcpp-predefined.h
new file mode 100644
index 0000000..4010ae0
--- /dev/null
+++ b/src/mcpp-predefined.h
@@ -0,0 +1,12 @@
+#ifndef __MCPU_CPP_PREDEFINED_H__
+#define __MCPU_CPP_PREDEFINED_H__ 1
+
+#include <defs.h>
+#include <mcpp-macro.h>
+
+int mcpu_predefined_install( mcpu_macro_table *table );
+int mcpu_predefined_expand_builtin( enum mcpu_macro_builtin builtin,
+ unsigned bits,
+ mcpu_text *output );
+
+#endif /* __MCPU_CPP_PREDEFINED_H__ */
diff --git a/src/mcpp-runtime.c b/src/mcpp-runtime.c
new file mode 100644
index 0000000..f498b20
--- /dev/null
+++ b/src/mcpp-runtime.c
@@ -0,0 +1,268 @@
+#include <defs.h>
+
+#include <unistd.h>
+
+
+static char *
+path_dirname( const char *path )
+{
+ const char *slash;
+ size_t length;
+ char *result;
+
+ if( path == NULL || *path == 0 )
+ {
+ errno = EINVAL;
+ return( NULL );
+ }
+
+ slash = strrchr( path, '/' );
+ if( slash == NULL )
+ {
+ errno = EINVAL;
+ return( NULL );
+ }
+
+ if( slash == path )
+ length = 1;
+ else
+ length = (size_t)(slash - path);
+
+ result = (char *)malloc( length + 1 );
+ if( result == NULL )
+ return( NULL );
+
+ memcpy( result, path, length );
+ result[length] = 0;
+ return( result );
+}
+
+
+static char *
+path_join( const char *directory, const char *name )
+{
+ size_t directory_length;
+ size_t name_length;
+ int need_slash;
+ char *result;
+
+ if( directory == NULL || name == NULL )
+ {
+ errno = EINVAL;
+ return( NULL );
+ }
+
+ directory_length = strlen( directory );
+ name_length = strlen( name );
+ need_slash = directory_length != 0 && directory[directory_length - 1] != '/';
+
+ if( directory_length > SIZE_MAX - name_length - (size_t)need_slash - 1 )
+ {
+ errno = EOVERFLOW;
+ return( NULL );
+ }
+
+ result = (char *)malloc( directory_length + name_length +
+ (size_t)need_slash + 1 );
+ if( result == NULL )
+ return( NULL );
+
+ memcpy( result, directory, directory_length );
+ if( need_slash )
+ result[directory_length++] = '/';
+ memcpy( result + directory_length, name, name_length );
+ result[directory_length + name_length] = 0;
+ return( result );
+}
+
+
+static int
+resolve_proc_self_exe( char path[PATH_MAX] )
+{
+ ssize_t length;
+
+ length = readlink( "/proc/self/exe", path, PATH_MAX - 1 );
+ if( length < 0 )
+ return( -1 );
+
+ if( length >= PATH_MAX - 1 )
+ {
+ errno = ENAMETOOLONG;
+ return( -1 );
+ }
+
+ path[length] = 0;
+ return( 0 );
+}
+
+
+static int
+resolve_candidate( const char *candidate, char path[PATH_MAX] )
+{
+ if( candidate == NULL || *candidate == 0 )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ if( realpath(candidate, path) == NULL )
+ return( -1 );
+
+ if( access(path, X_OK) != 0 )
+ return( -1 );
+
+ return( 0 );
+}
+
+
+static int
+resolve_argv0( const char *argv0, char path[PATH_MAX] )
+{
+ const char *search;
+ const char *entry;
+
+ if( argv0 == NULL || *argv0 == 0 )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ if( strchr(argv0, '/') != NULL )
+ return( resolve_candidate(argv0, path) );
+
+ search = getenv( "PATH" );
+ if( search == NULL )
+ {
+ errno = ENOENT;
+ return( -1 );
+ }
+
+ entry = search;
+ for( ;; )
+ {
+ const char *colon = strchr( entry, ':' );
+ size_t length = colon ? (size_t)(colon - entry) : strlen(entry);
+ const char *directory = entry;
+ char directory_buffer[PATH_MAX];
+ char *candidate;
+ int rc;
+
+ if( length == 0 )
+ directory = ".";
+ else
+ {
+ if( length >= sizeof(directory_buffer) )
+ {
+ errno = ENAMETOOLONG;
+ return( -1 );
+ }
+ memcpy( directory_buffer, entry, length );
+ directory_buffer[length] = 0;
+ directory = directory_buffer;
+ }
+
+ candidate = path_join( directory, argv0 );
+ if( candidate == NULL )
+ return( -1 );
+ rc = resolve_candidate( candidate, path );
+ free( candidate );
+ if( rc == 0 )
+ return( 0 );
+
+ if( colon == NULL )
+ break;
+ entry = colon + 1;
+ }
+
+ errno = ENOENT;
+ return( -1 );
+}
+
+
+void
+mcpp_runtime_paths_init( mcpp_runtime_paths *paths )
+{
+ if( paths == NULL )
+ return;
+
+ paths->executable_path = NULL;
+ paths->executable_dir = NULL;
+ paths->installation_root = NULL;
+ paths->config_file = NULL;
+ paths->system_include_path = NULL;
+}
+
+
+void
+mcpp_runtime_paths_free( mcpp_runtime_paths *paths )
+{
+ if( paths == NULL )
+ return;
+
+ free( paths->executable_path );
+ free( paths->executable_dir );
+ free( paths->installation_root );
+ free( paths->config_file );
+ free( paths->system_include_path );
+ mcpp_runtime_paths_init( paths );
+}
+
+
+int
+mcpp_runtime_paths_discover( mcpp_runtime_paths *paths, const char *argv0 )
+{
+ char resolved[PATH_MAX];
+ char *executable_path = NULL;
+ char *executable_dir = NULL;
+ char *installation_root = NULL;
+ char *config_file = NULL;
+ char *system_include_path = NULL;
+
+ if( paths == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ if( resolve_proc_self_exe(resolved) != 0 )
+ {
+ if( resolve_argv0(argv0, resolved) != 0 )
+ return( -1 );
+ }
+
+ executable_path = strdup( resolved );
+ if( executable_path == NULL )
+ goto fail;
+
+ executable_dir = path_dirname( executable_path );
+ if( executable_dir == NULL )
+ goto fail;
+
+ installation_root = path_dirname( executable_dir );
+ if( installation_root == NULL )
+ goto fail;
+
+ config_file = path_join( installation_root, "etc/mcpu-cpp.conf" );
+ if( config_file == NULL )
+ goto fail;
+
+ system_include_path = path_join( installation_root, "include" );
+ if( system_include_path == NULL )
+ goto fail;
+
+ mcpp_runtime_paths_free( paths );
+ paths->executable_path = executable_path;
+ paths->executable_dir = executable_dir;
+ paths->installation_root = installation_root;
+ paths->config_file = config_file;
+ paths->system_include_path = system_include_path;
+ return( 0 );
+
+fail:
+ free( executable_path );
+ free( executable_dir );
+ free( installation_root );
+ free( config_file );
+ free( system_include_path );
+ return( -1 );
+}
diff --git a/src/mcpp-runtime.h b/src/mcpp-runtime.h
new file mode 100644
index 0000000..75d1ddf
--- /dev/null
+++ b/src/mcpp-runtime.h
@@ -0,0 +1,19 @@
+#ifndef __MCPU_CPP_RUNTIME_H__
+#define __MCPU_CPP_RUNTIME_H__ 1
+
+typedef struct mcpp_runtime_paths mcpp_runtime_paths;
+struct mcpp_runtime_paths
+{
+ char *executable_path;
+ char *executable_dir;
+ char *installation_root;
+ char *config_file;
+ char *system_include_path;
+};
+
+void mcpp_runtime_paths_init( mcpp_runtime_paths *paths );
+void mcpp_runtime_paths_free( mcpp_runtime_paths *paths );
+int mcpp_runtime_paths_discover( mcpp_runtime_paths *paths,
+ const char *argv0 );
+
+#endif /* __MCPU_CPP_RUNTIME_H__ */
diff --git a/src/mcpp-semantic.c b/src/mcpp-semantic.c
new file mode 100644
index 0000000..3063139
--- /dev/null
+++ b/src/mcpp-semantic.c
@@ -0,0 +1,349 @@
+#include <defs.h>
+
+#define MCPP_INTEGER_BITS 64U
+
+static __mpu_int64_t
+mcpp_semantic_signed( mcpp_integer value )
+{
+ if( value.value <= (__mpu_uint64_t)INT64_MAX )
+ return( (__mpu_int64_t)value.value );
+
+ return( -1 - (__mpu_int64_t)(UINT64_MAX - value.value) );
+}
+
+static __mpu_uint64_t
+mcpp_semantic_unsigned_from_signed( __mpu_int64_t value )
+{
+ if( value >= 0 )
+ return( (__mpu_uint64_t)value );
+
+ return( UINT64_MAX - (__mpu_uint64_t)(-(value + 1)) );
+}
+
+static unsigned
+mcpp_semantic_shift_count( mcpp_integer value, int *negative )
+{
+ __mpu_uint64_t magnitude;
+
+ *negative = 0;
+ if( !value.unsignedp && mcpp_semantic_signed(value) < 0 )
+ {
+ *negative = 1;
+ magnitude = (__mpu_uint64_t)0 - value.value;
+ }
+ else
+ magnitude = value.value;
+
+ if( magnitude >= MCPP_INTEGER_BITS )
+ return( MCPP_INTEGER_BITS );
+
+ return( (unsigned)magnitude );
+}
+
+static mcpp_integer
+mcpp_semantic_shift_left_count( mcpp_integer value, unsigned shift )
+{
+ mcpp_integer result;
+
+ result.unsignedp = value.unsignedp;
+ if( shift >= MCPP_INTEGER_BITS )
+ result.value = 0;
+ else
+ result.value = value.value << shift;
+
+ return( result );
+}
+
+static mcpp_integer
+mcpp_semantic_shift_right_count( mcpp_integer value, unsigned shift )
+{
+ mcpp_integer result;
+ int negative;
+
+ result.unsignedp = value.unsignedp;
+ negative = !value.unsignedp && mcpp_semantic_signed(value) < 0;
+
+ if( shift >= MCPP_INTEGER_BITS )
+ {
+ result.value = negative ? UINT64_MAX : 0;
+ return( result );
+ }
+
+ if( shift == 0 || value.unsignedp || !negative )
+ {
+ result.value = value.value >> shift;
+ return( result );
+ }
+
+ result.value = (value.value >> shift) |
+ (UINT64_MAX << (MCPP_INTEGER_BITS - shift));
+ return( result );
+}
+
+void
+mcpp_semantic_context_init( mcpp_semantic_context *context,
+ const char *filename,
+ unsigned line_number,
+ const mcpp_options *options )
+{
+ context->filename = filename;
+ context->line_number = line_number;
+ context->options = options;
+ context->failed = 0;
+}
+
+void
+mcpp_semantic_error( mcpp_semantic_context *context, const char *message )
+{
+ if( context->failed )
+ return;
+
+ fprintf( stderr, "%s:%u: error: %s\n",
+ context->filename, context->line_number, message );
+ context->failed = 1;
+}
+
+void
+mcpp_semantic_warning( mcpp_semantic_context *context,
+ const char *message )
+{
+ if( mcpp_diagnostic_warning(context->options,
+ context->filename, context->line_number,
+ "%s", message) != 0 )
+ context->failed = 1;
+}
+
+int
+mcpp_semantic_failed( const mcpp_semantic_context *context )
+{
+ return( context->failed );
+}
+
+mcpp_integer
+mcpp_semantic_make( __mpu_uint64_t value, int unsignedp )
+{
+ mcpp_integer result;
+
+ result.value = value;
+ result.unsignedp = unsignedp ? 1 : 0;
+ return( result );
+}
+
+int
+mcpp_semantic_true( mcpp_integer value )
+{
+ return( value.value != 0 );
+}
+
+mcpp_integer
+mcpp_semantic_neg( mcpp_integer value )
+{
+ value.value = (__mpu_uint64_t)0 - value.value;
+ return( value );
+}
+
+mcpp_integer
+mcpp_semantic_not( mcpp_integer value )
+{
+ return( mcpp_semantic_make(!mcpp_semantic_true(value), 0) );
+}
+
+mcpp_integer
+mcpp_semantic_compl( mcpp_integer value )
+{
+ value.value = ~value.value;
+ return( value );
+}
+
+mcpp_integer
+mcpp_semantic_mul( mcpp_integer left, mcpp_integer right )
+{
+ return( mcpp_semantic_make(left.value * right.value,
+ left.unsignedp || right.unsignedp) );
+}
+
+mcpp_integer
+mcpp_semantic_div( mcpp_semantic_context *context,
+ mcpp_integer left, mcpp_integer right, int evaluate )
+{
+ int unsignedp = left.unsignedp || right.unsignedp;
+
+ if( right.value == 0 )
+ {
+ if( evaluate )
+ mcpp_semantic_error( context, "division by zero in #if expression" );
+ return( mcpp_semantic_make(0, unsignedp) );
+ }
+
+ if( unsignedp )
+ return( mcpp_semantic_make(left.value / right.value, 1) );
+
+ if( mcpp_semantic_signed(left) == INT64_MIN &&
+ mcpp_semantic_signed(right) == -1 )
+ return( mcpp_semantic_make((__mpu_uint64_t)1 << 63, 0) );
+
+ return( mcpp_semantic_make(
+ mcpp_semantic_unsigned_from_signed(
+ mcpp_semantic_signed(left) / mcpp_semantic_signed(right)), 0) );
+}
+
+mcpp_integer
+mcpp_semantic_mod( mcpp_semantic_context *context,
+ mcpp_integer left, mcpp_integer right, int evaluate )
+{
+ int unsignedp = left.unsignedp || right.unsignedp;
+
+ if( right.value == 0 )
+ {
+ if( evaluate )
+ mcpp_semantic_error( context, "division by zero in #if expression" );
+ return( mcpp_semantic_make(0, unsignedp) );
+ }
+
+ if( unsignedp )
+ return( mcpp_semantic_make(left.value % right.value, 1) );
+
+ if( mcpp_semantic_signed(left) == INT64_MIN &&
+ mcpp_semantic_signed(right) == -1 )
+ return( mcpp_semantic_make(0, 0) );
+
+ return( mcpp_semantic_make(
+ mcpp_semantic_unsigned_from_signed(
+ mcpp_semantic_signed(left) % mcpp_semantic_signed(right)), 0) );
+}
+
+mcpp_integer
+mcpp_semantic_add( mcpp_integer left, mcpp_integer right )
+{
+ return( mcpp_semantic_make(left.value + right.value,
+ left.unsignedp || right.unsignedp) );
+}
+
+mcpp_integer
+mcpp_semantic_sub( mcpp_integer left, mcpp_integer right )
+{
+ return( mcpp_semantic_make(left.value - right.value,
+ left.unsignedp || right.unsignedp) );
+}
+
+mcpp_integer
+mcpp_semantic_lshift( mcpp_integer left, mcpp_integer right )
+{
+ unsigned shift;
+ int negative;
+
+ shift = mcpp_semantic_shift_count( right, &negative );
+ if( negative )
+ return( mcpp_semantic_shift_right_count(left, shift) );
+
+ return( mcpp_semantic_shift_left_count(left, shift) );
+}
+
+mcpp_integer
+mcpp_semantic_rshift( mcpp_integer left, mcpp_integer right )
+{
+ unsigned shift;
+ int negative;
+
+ shift = mcpp_semantic_shift_count( right, &negative );
+ if( negative )
+ return( mcpp_semantic_shift_left_count(left, shift) );
+
+ return( mcpp_semantic_shift_right_count(left, shift) );
+}
+
+mcpp_integer
+mcpp_semantic_equal( mcpp_integer left, mcpp_integer right )
+{
+ int unsignedp = left.unsignedp || right.unsignedp;
+ int value = unsignedp ? left.value == right.value :
+ mcpp_semantic_signed(left) == mcpp_semantic_signed(right);
+ return( mcpp_semantic_make((__mpu_uint64_t)value, 0) );
+}
+
+mcpp_integer
+mcpp_semantic_notequal( mcpp_integer left, mcpp_integer right )
+{
+ mcpp_integer result = mcpp_semantic_equal( left, right );
+ result.value = !result.value;
+ return( result );
+}
+
+mcpp_integer
+mcpp_semantic_less( mcpp_integer left, mcpp_integer right )
+{
+ int unsignedp = left.unsignedp || right.unsignedp;
+ int value = unsignedp ? left.value < right.value :
+ mcpp_semantic_signed(left) < mcpp_semantic_signed(right);
+ return( mcpp_semantic_make((__mpu_uint64_t)value, 0) );
+}
+
+mcpp_integer
+mcpp_semantic_greater( mcpp_integer left, mcpp_integer right )
+{
+ int unsignedp = left.unsignedp || right.unsignedp;
+ int value = unsignedp ? left.value > right.value :
+ mcpp_semantic_signed(left) > mcpp_semantic_signed(right);
+ return( mcpp_semantic_make((__mpu_uint64_t)value, 0) );
+}
+
+mcpp_integer
+mcpp_semantic_leq( mcpp_integer left, mcpp_integer right )
+{
+ int unsignedp = left.unsignedp || right.unsignedp;
+ int value = unsignedp ? left.value <= right.value :
+ mcpp_semantic_signed(left) <= mcpp_semantic_signed(right);
+ return( mcpp_semantic_make((__mpu_uint64_t)value, 0) );
+}
+
+mcpp_integer
+mcpp_semantic_geq( mcpp_integer left, mcpp_integer right )
+{
+ int unsignedp = left.unsignedp || right.unsignedp;
+ int value = unsignedp ? left.value >= right.value :
+ mcpp_semantic_signed(left) >= mcpp_semantic_signed(right);
+ return( mcpp_semantic_make((__mpu_uint64_t)value, 0) );
+}
+
+mcpp_integer
+mcpp_semantic_bitand( mcpp_integer left, mcpp_integer right )
+{
+ return( mcpp_semantic_make(left.value & right.value,
+ left.unsignedp || right.unsignedp) );
+}
+
+mcpp_integer
+mcpp_semantic_bitxor( mcpp_integer left, mcpp_integer right )
+{
+ return( mcpp_semantic_make(left.value ^ right.value,
+ left.unsignedp || right.unsignedp) );
+}
+
+mcpp_integer
+mcpp_semantic_bitor( mcpp_integer left, mcpp_integer right )
+{
+ return( mcpp_semantic_make(left.value | right.value,
+ left.unsignedp || right.unsignedp) );
+}
+
+mcpp_integer
+mcpp_semantic_and( mcpp_integer left, mcpp_integer right )
+{
+ return( mcpp_semantic_make(mcpp_semantic_true(left) && mcpp_semantic_true(right), 0) );
+}
+
+mcpp_integer
+mcpp_semantic_or( mcpp_integer left, mcpp_integer right )
+{
+ return( mcpp_semantic_make(mcpp_semantic_true(left) || mcpp_semantic_true(right), 0) );
+}
+
+mcpp_integer
+mcpp_semantic_conditional( mcpp_integer condition,
+ mcpp_integer yes, mcpp_integer no )
+{
+ mcpp_integer result = mcpp_semantic_true(condition) ? yes : no;
+
+ result.unsignedp = yes.unsignedp || no.unsignedp;
+ return( result );
+}
diff --git a/src/mcpp-semantic.h b/src/mcpp-semantic.h
new file mode 100644
index 0000000..efa793e
--- /dev/null
+++ b/src/mcpp-semantic.h
@@ -0,0 +1,67 @@
+#ifndef __MCPU_CPP_SEMANTIC_H__
+#define __MCPU_CPP_SEMANTIC_H__ 1
+
+#include <defs.h>
+
+typedef struct mcpp_options mcpp_options;
+typedef struct mcpp_integer mcpp_integer;
+struct mcpp_integer
+{
+ __mpu_uint64_t value;
+ int unsignedp;
+};
+
+typedef struct mcpp_semantic_context mcpp_semantic_context;
+struct mcpp_semantic_context
+{
+ const char *filename;
+ unsigned line_number;
+ const mcpp_options *options;
+ int failed;
+};
+
+void mcpp_semantic_context_init( mcpp_semantic_context *context,
+ const char *filename,
+ unsigned line_number,
+ const mcpp_options *options );
+void mcpp_semantic_error( mcpp_semantic_context *context,
+ const char *message );
+void mcpp_semantic_warning( mcpp_semantic_context *context,
+ const char *message );
+int mcpp_semantic_failed( const mcpp_semantic_context *context );
+
+mcpp_integer mcpp_semantic_make( __mpu_uint64_t value, int unsignedp );
+int mcpp_semantic_true( mcpp_integer value );
+
+mcpp_integer mcpp_semantic_neg( mcpp_integer value );
+mcpp_integer mcpp_semantic_not( mcpp_integer value );
+mcpp_integer mcpp_semantic_compl( mcpp_integer value );
+
+mcpp_integer mcpp_semantic_mul( mcpp_integer left, mcpp_integer right );
+mcpp_integer mcpp_semantic_div( mcpp_semantic_context *context,
+ mcpp_integer left, mcpp_integer right,
+ int evaluate );
+mcpp_integer mcpp_semantic_mod( mcpp_semantic_context *context,
+ mcpp_integer left, mcpp_integer right,
+ int evaluate );
+mcpp_integer mcpp_semantic_add( mcpp_integer left, mcpp_integer right );
+mcpp_integer mcpp_semantic_sub( mcpp_integer left, mcpp_integer right );
+mcpp_integer mcpp_semantic_lshift( mcpp_integer left, mcpp_integer right );
+mcpp_integer mcpp_semantic_rshift( mcpp_integer left, mcpp_integer right );
+
+mcpp_integer mcpp_semantic_equal( mcpp_integer left, mcpp_integer right );
+mcpp_integer mcpp_semantic_notequal( mcpp_integer left, mcpp_integer right );
+mcpp_integer mcpp_semantic_less( mcpp_integer left, mcpp_integer right );
+mcpp_integer mcpp_semantic_greater( mcpp_integer left, mcpp_integer right );
+mcpp_integer mcpp_semantic_leq( mcpp_integer left, mcpp_integer right );
+mcpp_integer mcpp_semantic_geq( mcpp_integer left, mcpp_integer right );
+mcpp_integer mcpp_semantic_bitand( mcpp_integer left, mcpp_integer right );
+mcpp_integer mcpp_semantic_bitxor( mcpp_integer left, mcpp_integer right );
+mcpp_integer mcpp_semantic_bitor( mcpp_integer left, mcpp_integer right );
+mcpp_integer mcpp_semantic_and( mcpp_integer left, mcpp_integer right );
+mcpp_integer mcpp_semantic_or( mcpp_integer left, mcpp_integer right );
+mcpp_integer mcpp_semantic_conditional( mcpp_integer condition,
+ mcpp_integer yes,
+ mcpp_integer no );
+
+#endif /* __MCPU_CPP_SEMANTIC_H__ */
diff --git a/src/mcpp-source.c b/src/mcpp-source.c
new file mode 100644
index 0000000..c40e370
--- /dev/null
+++ b/src/mcpp-source.c
@@ -0,0 +1,332 @@
+#include <defs.h>
+
+static char *
+read_stream_bytes( FILE *fp, size_t *file_length )
+{
+ char *data = NULL;
+ size_t length = 0;
+ size_t capacity = 0;
+
+ if( fp == NULL || file_length == NULL )
+ {
+ errno = EINVAL;
+ return( NULL );
+ }
+
+ *file_length = 0;
+
+ for( ;; )
+ {
+ size_t n;
+
+ if( capacity - length < 4096 )
+ {
+ size_t new_capacity = capacity ? capacity * 2 : 8192;
+ char *q = (char *)realloc( data, new_capacity + 1 );
+ if( q == NULL )
+ {
+ free( data );
+ return( NULL );
+ }
+ data = q;
+ capacity = new_capacity;
+ }
+
+ n = fread( data + length, 1, capacity - length, fp );
+ length += n;
+
+ if( n == 0 )
+ {
+ if( ferror(fp) )
+ {
+ free( data );
+ return( NULL );
+ }
+ break;
+ }
+ }
+
+ if( data == NULL )
+ {
+ data = (char *)calloc( 1, 1 );
+ if( data == NULL )
+ return( NULL );
+ }
+ else
+ data[length] = 0;
+
+ *file_length = length;
+ return( data );
+}
+
+static int
+normalize_newlines( mcpu_text *text )
+{
+ size_t r;
+ size_t w = 0;
+
+ if( text == NULL || text->data == NULL )
+ return( 0 );
+
+ for( r = 0; r < text->length; ++r )
+ {
+ if( text->data[r] == '\r' )
+ {
+ if( r + 1 < text->length && text->data[r + 1] == '\n' )
+ ++r;
+ text->data[w++] = '\n';
+ }
+ else
+ text->data[w++] = text->data[r];
+ }
+
+ text->length = w;
+ text->data[w] = 0;
+
+ return( 0 );
+}
+
+void
+mcpu_source_init( mcpu_source *source )
+{
+ if( source == NULL )
+ return;
+
+ source->filename = NULL;
+ mcpu_text_init( &source->text );
+ mcpu_text_init( &source->logical_line );
+ source->splice_offsets = NULL;
+ source->splice_count = 0;
+ source->splice_capacity = 0;
+ source->offset = 0;
+ source->line = 1;
+}
+
+void
+mcpu_source_free( mcpu_source *source )
+{
+ if( source == NULL )
+ return;
+
+ free( source->filename );
+ source->filename = NULL;
+ mcpu_text_free( &source->text );
+ mcpu_text_free( &source->logical_line );
+ free( source->splice_offsets );
+ source->splice_offsets = NULL;
+ source->splice_count = 0;
+ source->splice_capacity = 0;
+ source->offset = 0;
+ source->line = 1;
+}
+
+static int
+source_from_bytes( mcpu_source *source, char *bytes, size_t byte_length,
+ const char *name )
+{
+ const __mpu_char8_t *input = (const __mpu_char8_t *)bytes;
+ const __mpu_char8_t *p;
+
+ if( memchr( bytes, 0, byte_length ) != NULL )
+ {
+ fprintf( stderr, "%s: NUL character is not allowed in source text\n", name );
+ return( -1 );
+ }
+
+ if( byte_length >= 3 &&
+ (unsigned char)bytes[0] == 0xef &&
+ (unsigned char)bytes[1] == 0xbb &&
+ (unsigned char)bytes[2] == 0xbf )
+ input += 3;
+
+ if( !mpu_utf8valid(input) )
+ {
+ fprintf( stderr, "%s: invalid UTF-8\n", name );
+ return( -1 );
+ }
+
+ p = input;
+ while( *p != 0 )
+ {
+ __mpu_char32_t value;
+ const __mpu_char8_t *next;
+ __mpu_char16_t c;
+
+ next = mpu_utf8get( p, &value );
+ if( next == NULL )
+ {
+ fprintf( stderr, "%s: invalid UTF-8\n", name );
+ return( -1 );
+ }
+
+ if( value <= 0xffff )
+ c = (__mpu_char16_t)value;
+ else
+ c = MCPU_CPP_NON_UCS2_SENTINEL;
+
+ if( mcpu_text_append_char(&source->text, c) != 0 )
+ return( -1 );
+
+ p = next;
+ }
+
+ if( normalize_newlines(&source->text) != 0 )
+ return( -1 );
+
+ source->filename = strdup( name );
+ if( source->filename == NULL )
+ return( -1 );
+
+ return( 0 );
+}
+
+int
+mcpu_source_valid_ucs2( const __mpu_char16_t *text, size_t length )
+{
+ size_t i;
+
+ if( text == NULL && length != 0 )
+ return( 0 );
+
+ for( i = 0; i < length; ++i )
+ if( text[i] == MCPU_CPP_NON_UCS2_SENTINEL )
+ return( 0 );
+
+ return( 1 );
+}
+
+int
+mcpu_source_open_stream( mcpu_source *source, FILE *stream, const char *name )
+{
+ char *bytes;
+ size_t byte_length;
+ int rc;
+
+ if( source == NULL || stream == NULL || name == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ mcpu_source_free( source );
+ mcpu_source_init( source );
+
+ bytes = read_stream_bytes( stream, &byte_length );
+ if( bytes == NULL )
+ return( -1 );
+
+ rc = source_from_bytes( source, bytes, byte_length, name );
+ free( bytes );
+ return( rc );
+}
+
+int
+mcpu_source_open( mcpu_source *source, const char *filename )
+{
+ FILE *fp;
+ int rc;
+
+ if( source == NULL || filename == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ fp = fopen( filename, "rb" );
+ if( fp == NULL )
+ return( -1 );
+
+ rc = mcpu_source_open_stream( source, fp, filename );
+ if( fclose(fp) != 0 && rc == 0 )
+ rc = -1;
+
+ return( rc );
+}
+
+static int
+append_splice_offset( mcpu_source *source, size_t offset )
+{
+ size_t *offsets;
+ size_t capacity;
+
+ if( source->splice_count < source->splice_capacity )
+ {
+ source->splice_offsets[source->splice_count++] = offset;
+ return( 0 );
+ }
+
+ capacity = source->splice_capacity ? source->splice_capacity * 2 : 8;
+ offsets = (size_t *)realloc( source->splice_offsets,
+ capacity * sizeof(*offsets) );
+ if( offsets == NULL )
+ return( -1 );
+
+ source->splice_offsets = offsets;
+ source->splice_capacity = capacity;
+ source->splice_offsets[source->splice_count++] = offset;
+ return( 0 );
+}
+
+
+int
+mcpu_source_next_line( mcpu_source *source,
+ const __mpu_char16_t **line,
+ size_t *length,
+ unsigned *line_number,
+ unsigned *next_line_number,
+ int *spliced )
+{
+ unsigned first_line;
+
+ if( source == NULL || line == NULL || length == NULL ||
+ line_number == NULL || next_line_number == NULL || spliced == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ if( source->offset >= source->text.length )
+ return( 0 );
+
+ mcpu_text_free( &source->logical_line );
+ mcpu_text_init( &source->logical_line );
+ source->splice_count = 0;
+ first_line = source->line;
+ *spliced = 0;
+
+ while( source->offset < source->text.length )
+ {
+ __mpu_char16_t c = source->text.data[source->offset++];
+
+ /*
+ * Backslash-newline deletion is preprocessing phase 2 and therefore
+ * happens before comments, directives and macro recognition.
+ */
+ if( c == '\\' && source->offset < source->text.length &&
+ source->text.data[source->offset] == '\n' )
+ {
+ if( append_splice_offset(source, source->logical_line.length) != 0 )
+ return( -1 );
+ ++source->offset;
+ ++source->line;
+ *spliced = 1;
+ continue;
+ }
+
+ if( mcpu_text_append_char( &source->logical_line, c ) != 0 )
+ return( -1 );
+
+ if( c == '\n' )
+ {
+ ++source->line;
+ break;
+ }
+ }
+
+ *line = source->logical_line.data;
+ *length = source->logical_line.length;
+ *line_number = first_line;
+ *next_line_number = source->line;
+
+ return( 1 );
+}
diff --git a/src/mcpp-source.h b/src/mcpp-source.h
new file mode 100644
index 0000000..9a7973a
--- /dev/null
+++ b/src/mcpp-source.h
@@ -0,0 +1,35 @@
+#ifndef __MCPU_CPP_SOURCE_H__
+#define __MCPU_CPP_SOURCE_H__ 1
+
+#include <defs.h>
+#include <mcpp-text.h>
+
+#define MCPU_CPP_NON_UCS2_SENTINEL 0xd800
+
+typedef struct mcpu_source mcpu_source;
+struct mcpu_source
+{
+ char *filename;
+ mcpu_text text;
+ mcpu_text logical_line;
+ size_t *splice_offsets;
+ size_t splice_count;
+ size_t splice_capacity;
+ size_t offset;
+ unsigned line;
+};
+
+void mcpu_source_init( mcpu_source *source );
+void mcpu_source_free( mcpu_source *source );
+int mcpu_source_open( mcpu_source *source, const char *filename );
+int mcpu_source_open_stream( mcpu_source *source, FILE *stream,
+ const char *name );
+int mcpu_source_valid_ucs2( const __mpu_char16_t *text, size_t length );
+int mcpu_source_next_line( mcpu_source *source,
+ const __mpu_char16_t **line,
+ size_t *length,
+ unsigned *line_number,
+ unsigned *next_line_number,
+ int *spliced );
+
+#endif /* __MCPU_CPP_SOURCE_H__ */
diff --git a/src/mcpp-text.c b/src/mcpp-text.c
new file mode 100644
index 0000000..febbd3b
--- /dev/null
+++ b/src/mcpp-text.c
@@ -0,0 +1,237 @@
+#include <defs.h>
+
+void
+mcpu_text_init( mcpu_text *text )
+{
+ if( text == NULL )
+ return;
+
+ text->data = NULL;
+ text->length = 0;
+ text->capacity = 0;
+}
+
+void
+mcpu_text_free( mcpu_text *text )
+{
+ if( text == NULL )
+ return;
+
+ free( text->data );
+ text->data = NULL;
+ text->length = 0;
+ text->capacity = 0;
+}
+
+int
+mcpu_text_reserve( mcpu_text *text, size_t need )
+{
+ __mpu_char16_t *p;
+ size_t capacity;
+
+ if( text == NULL )
+ return( -1 );
+
+ if( need <= text->capacity )
+ return( 0 );
+
+ capacity = text->capacity ? text->capacity : 256;
+ while( capacity < need )
+ {
+ if( capacity > (SIZE_MAX / 2) )
+ {
+ errno = EOVERFLOW;
+ return( -1 );
+ }
+ capacity *= 2;
+ }
+
+ if( capacity > (SIZE_MAX / sizeof(__mpu_char16_t)) )
+ {
+ errno = EOVERFLOW;
+ return( -1 );
+ }
+
+ p = (__mpu_char16_t *)realloc( text->data,
+ capacity * sizeof(__mpu_char16_t) );
+ if( p == NULL )
+ return( -1 );
+
+ text->data = p;
+ text->capacity = capacity;
+
+ return( 0 );
+}
+
+int
+mcpu_text_append( mcpu_text *text,
+ const __mpu_char16_t *data, size_t length )
+{
+ if( text == NULL || (data == NULL && length != 0) )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ if( length > SIZE_MAX - text->length - 1 )
+ {
+ errno = EOVERFLOW;
+ return( -1 );
+ }
+
+ if( mcpu_text_reserve( text, text->length + length + 1 ) != 0 )
+ return( -1 );
+
+ if( length != 0 )
+ memcpy( text->data + text->length, data,
+ length * sizeof(__mpu_char16_t) );
+
+ text->length += length;
+ text->data[text->length] = 0;
+
+ return( 0 );
+}
+
+int
+mcpu_text_append_char( mcpu_text *text, __mpu_char16_t ch )
+{
+ return( mcpu_text_append( text, &ch, 1 ) );
+}
+
+int
+mcpu_text_append_ascii( mcpu_text *text, const char *s )
+{
+ if( text == NULL || s == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ while( *s )
+ {
+ unsigned char c = (unsigned char)*s++;
+
+ if( c > 0x7f )
+ {
+ errno = EILSEQ;
+ return( -1 );
+ }
+
+ if( mcpu_text_append_char( text, (__mpu_char16_t)c ) != 0 )
+ return( -1 );
+ }
+
+ return( 0 );
+}
+
+int
+mcpu_text_append_utf8( mcpu_text *text, const char *s )
+{
+ __mpu_char16_t *u;
+ __mpu_size_t n;
+ int rc;
+
+ if( text == NULL || s == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ n = mpu_utf8_to_ucs2( NULL, (const __mpu_char8_t *)s, 0 );
+ if( n == (__mpu_size_t)-1 )
+ return( -1 );
+
+ u = (__mpu_char16_t *)calloc( (size_t)n + 1, sizeof(__mpu_char16_t) );
+ if( u == NULL )
+ return( -1 );
+
+ if( mpu_utf8_to_ucs2( u, (const __mpu_char8_t *)s,
+ (size_t)n + 1 ) == (__mpu_size_t)-1 )
+ {
+ free( u );
+ return( -1 );
+ }
+
+ rc = mcpu_text_append( text, u, (size_t)n );
+ free( u );
+
+ return( rc );
+}
+
+int
+mcpu_text_from_utf8( mcpu_text *text, const char *s )
+{
+ if( text == NULL || s == NULL )
+ {
+ errno = EINVAL;
+ return( -1 );
+ }
+
+ mcpu_text_free( text );
+ mcpu_text_init( text );
+
+ return( mcpu_text_append_utf8( text, s ) );
+}
+
+char *
+mcpu_text_to_utf8( const __mpu_char16_t *data, size_t length )
+{
+ __mpu_char16_t *tmp;
+ __mpu_char8_t *out;
+ __mpu_size_t n;
+
+ if( data == NULL && length != 0 )
+ {
+ errno = EINVAL;
+ return( NULL );
+ }
+
+ tmp = (__mpu_char16_t *)calloc( length + 1, sizeof(__mpu_char16_t) );
+ if( tmp == NULL )
+ return( NULL );
+
+ if( length != 0 )
+ memcpy( tmp, data, length * sizeof(__mpu_char16_t) );
+
+ n = mpu_ucs2_to_utf8( NULL, tmp, 0 );
+ if( n == (__mpu_size_t)-1 )
+ {
+ free( tmp );
+ return( NULL );
+ }
+
+ out = (__mpu_char8_t *)calloc( (size_t)n + 1, 1 );
+ if( out == NULL )
+ {
+ free( tmp );
+ return( NULL );
+ }
+
+ if( mpu_ucs2_to_utf8( out, tmp, (size_t)n + 1 ) == (__mpu_size_t)-1 )
+ {
+ free( out );
+ free( tmp );
+ return( NULL );
+ }
+
+ free( tmp );
+ return( (char *)out );
+}
+
+int
+mcpu_text_equal_ascii( const __mpu_char16_t *data,
+ size_t length, const char *s )
+{
+ size_t i;
+
+ if( data == NULL || s == NULL )
+ return( 0 );
+
+ for( i = 0; i < length && s[i]; ++i )
+ {
+ if( data[i] != (__mpu_char16_t)(unsigned char)s[i] )
+ return( 0 );
+ }
+
+ return( i == length && s[i] == 0 );
+}
diff --git a/src/mcpp-text.h b/src/mcpp-text.h
new file mode 100644
index 0000000..dce651c
--- /dev/null
+++ b/src/mcpp-text.h
@@ -0,0 +1,27 @@
+#ifndef __MCPU_CPP_TEXT_H__
+#define __MCPU_CPP_TEXT_H__ 1
+
+#include <defs.h>
+
+typedef struct mcpu_text mcpu_text;
+struct mcpu_text
+{
+ __mpu_char16_t *data;
+ size_t length;
+ size_t capacity;
+};
+
+void mcpu_text_init( mcpu_text *text );
+void mcpu_text_free( mcpu_text *text );
+int mcpu_text_reserve( mcpu_text *text, size_t need );
+int mcpu_text_append( mcpu_text *text,
+ const __mpu_char16_t *data, size_t length );
+int mcpu_text_append_char( mcpu_text *text, __mpu_char16_t ch );
+int mcpu_text_append_ascii( mcpu_text *text, const char *s );
+int mcpu_text_append_utf8( mcpu_text *text, const char *s );
+int mcpu_text_from_utf8( mcpu_text *text, const char *s );
+char *mcpu_text_to_utf8( const __mpu_char16_t *data, size_t length );
+int mcpu_text_equal_ascii( const __mpu_char16_t *data,
+ size_t length, const char *s );
+
+#endif /* __MCPU_CPP_TEXT_H__ */
diff --git a/tests/Makefile.am b/tests/Makefile.am
new file mode 100644
index 0000000..ade9bae
--- /dev/null
+++ b/tests/Makefile.am
@@ -0,0 +1,101 @@
+
+TESTS = \
+ t0001-utf8.sh \
+ t0002-lang.sh \
+ t0003-include.sh \
+ t0004-config.sh \
+ t0005-errors.sh \
+ t0006-path-order.sh \
+ t0007-language-path.sh \
+ t0008-comments.sh \
+ t0009-text-scanner.sh \
+ t0010-language-names.sh \
+ t0011-base-include-path.sh \
+ t0012-user-config-path.sh \
+ t0013-include-precedence.sh \
+ t0014-preprocessing-phases.sh \
+ t0015-object-macros.sh \
+ t0016-macro-include.sh \
+ t0017-macro-errors.sh \
+ t0018-include-lexing.sh \
+ t0019-recursive-macro.sh \
+ t0020-function-macros.sh \
+ t0021-function-macro-errors.sh \
+ t0022-predefined-macros.sh \
+ t0023-predefined-redefine.sh \
+ t0024-abi-predefined.sh \
+ t0025-dump-macros.sh \
+ t0026-dump-config.sh \
+ t0027-system-language-path.sh \
+ t0028-predefined-ranges.sh \
+ t0029-predefined-abi-names.sh \
+ t0030-integer-decimal-digits.sh \
+ t0031-stringification.sh \
+ t0032-command-line.sh \
+ t0033-dump-definitions.sh \
+ t0034-conditionals.sh \
+ t0035-line-control.sh \
+ t0036-lang-string.sh \
+ t0037-token-concatenation.sh \
+ t0038-ucs2-identifiers.sh \
+ t0039-command-line-macros.sh \
+ t0040-zubr-expression.sh \
+ t0041-integer-width-suffix.sh \
+ t0042-diagnostics.sh \
+ t0043-include-next.sh \
+ t0044-configured-search-order.sh \
+ t0045-search-dirs-verbose.sh \
+ t0046-pragma-once.sh \
+ t0047-dependencies.sh \
+ t0048-no-config-defaults.sh \
+ t0049-relocatable-root.sh \
+ t0050-dependency-side-effects.sh \
+ t0051-dump-macro-origin.sh \
+ t0052-warning-control.sh \
+ t0053-interface-cleanup.sh \
+ t0054-inhibit-warnings.sh \
+ t0055-forced-files.sh \
+ t0056-dependency-targets.sh \
+ t0057-missing-generated-dependencies.sh \
+ t0058-variadic-macros.sh \
+ t0059-va-opt.sh \
+ t0060-macro-whitespace.sh \
+ t0061-output-line-compaction.sh
+
+noinst_PROGRAMS = mcpp-options-test
+
+mcpp_options_test_SOURCES = \
+ mcpp-options-test.c \
+ ../src/mcpp-options.c
+
+mcpp_options_test_CPPFLAGS = -I$(top_srcdir)/src $(LIBMPUIO_CFLAGS)
+mcpp_options_test_LDADD = $(LIBMPUIO_LIBS)
+
+EXTRA_DIST = $(TESTS) \
+ data/utf8.c \
+ data/lang.c \
+ data/include-main.c \
+ data/include/one.h \
+ data/include/bridge.h \
+ data/system/system.h \
+ data/config.conf \
+ data/config-main.c \
+ data/unbalanced.c \
+ data/nonucs2.c \
+ data/order-main.c \
+ data/user/order.h \
+ data/system/order.h \
+ data/lang-path-main.c \
+ data/lang-as/lang.h \
+ data/lang-diff/lang.h \
+ data/base-path-main.c \
+ data/lang-base/base.h \
+ data/lang-base/diff/explicit.h \
+ data/common-precedence/same.h \
+ data/lang-as-precedence/same.h
+
+TESTS_ENVIRONMENT = MCPU_CPP='$(abs_top_builddir)/src/mcpu-cpp'; export MCPU_CPP; \
+ MCPP_OPTIONS_TEST='$(abs_top_builddir)/tests/mcpp-options-test'; export MCPP_OPTIONS_TEST;
+
+distclean-local:
+ -rm -rf $(DEPDIR)
diff --git a/tests/data/base-path-main.c b/tests/data/base-path-main.c
new file mode 100644
index 0000000..6ca1ffb
--- /dev/null
+++ b/tests/data/base-path-main.c
@@ -0,0 +1 @@
+#include <base.h>
diff --git a/tests/data/common-precedence/same.h b/tests/data/common-precedence/same.h
new file mode 100644
index 0000000..8a7b2c8
--- /dev/null
+++ b/tests/data/common-precedence/same.h
@@ -0,0 +1 @@
+int from_common_include_root;
diff --git a/tests/data/config-main.c b/tests/data/config-main.c
new file mode 100644
index 0000000..892eb5e
--- /dev/null
+++ b/tests/data/config-main.c
@@ -0,0 +1,2 @@
+#include "one.h"
+#include <system.h>
diff --git a/tests/data/config.conf b/tests/data/config.conf
new file mode 100644
index 0000000..e395802
--- /dev/null
+++ b/tests/data/config.conf
@@ -0,0 +1,2 @@
+MCPU_CPP_INCLUDE_PATH = ./include;
+MCPU_CPP_SYSTEM_INCLUDE_PATH = ./system;
diff --git a/tests/data/include-main.c b/tests/data/include-main.c
new file mode 100644
index 0000000..93bca59
--- /dev/null
+++ b/tests/data/include-main.c
@@ -0,0 +1,6 @@
+#include "one.h"
+#include <system.h>
+#lang "diff"
+#include "bridge.h"
+y' = 1;
+#endlang
diff --git a/tests/data/include/bridge.h b/tests/data/include/bridge.h
new file mode 100644
index 0000000..a878df1
--- /dev/null
+++ b/tests/data/include/bridge.h
@@ -0,0 +1,4 @@
+/* Entered while current language is diff. */
+#endlang
+int bridge_c;
+#lang "diff"
diff --git a/tests/data/include/one.h b/tests/data/include/one.h
new file mode 100644
index 0000000..6200094
--- /dev/null
+++ b/tests/data/include/one.h
@@ -0,0 +1 @@
+int from_one;
diff --git a/tests/data/lang-as-precedence/same.h b/tests/data/lang-as-precedence/same.h
new file mode 100644
index 0000000..f139324
--- /dev/null
+++ b/tests/data/lang-as-precedence/same.h
@@ -0,0 +1 @@
+int from_as_specific_include_path;
diff --git a/tests/data/lang-as/lang.h b/tests/data/lang-as/lang.h
new file mode 100644
index 0000000..f1f9283
--- /dev/null
+++ b/tests/data/lang-as/lang.h
@@ -0,0 +1 @@
+MCPU_AS_LANGUAGE_PATH from_as_language_path;
diff --git a/tests/data/lang-base/base.h b/tests/data/lang-base/base.h
new file mode 100644
index 0000000..1d8a123
--- /dev/null
+++ b/tests/data/lang-base/base.h
@@ -0,0 +1 @@
+int from_base_language_include_root;
diff --git a/tests/data/lang-base/diff/explicit.h b/tests/data/lang-base/diff/explicit.h
new file mode 100644
index 0000000..fa9157f
--- /dev/null
+++ b/tests/data/lang-base/diff/explicit.h
@@ -0,0 +1 @@
+int from_explicit_diff_subdirectory;
diff --git a/tests/data/lang-diff/lang.h b/tests/data/lang-diff/lang.h
new file mode 100644
index 0000000..54603aa
--- /dev/null
+++ b/tests/data/lang-diff/lang.h
@@ -0,0 +1 @@
+y' = from_diff_language_path;
diff --git a/tests/data/lang-path-main.c b/tests/data/lang-path-main.c
new file mode 100644
index 0000000..864bc07
--- /dev/null
+++ b/tests/data/lang-path-main.c
@@ -0,0 +1,6 @@
+#lang "as"
+#include "lang.h"
+#endlang
+#lang "diff"
+#include "lang.h"
+#endlang
diff --git a/tests/data/lang.c b/tests/data/lang.c
new file mode 100644
index 0000000..75cdc70
--- /dev/null
+++ b/tests/data/lang.c
@@ -0,0 +1,9 @@
+int before;
+#lang "diff"
+y' = omega;
+#lang "alg"
+f = x + y;
+#endlang
+z' = y;
+#endlang
+int after;
diff --git a/tests/data/nonucs2.c b/tests/data/nonucs2.c
new file mode 100644
index 0000000..d8cc27c
--- /dev/null
+++ b/tests/data/nonucs2.c
@@ -0,0 +1 @@
+int x; /* 😀 */
diff --git a/tests/data/order-main.c b/tests/data/order-main.c
new file mode 100644
index 0000000..04cac0c
--- /dev/null
+++ b/tests/data/order-main.c
@@ -0,0 +1 @@
+#include <order.h>
diff --git a/tests/data/system/order.h b/tests/data/system/order.h
new file mode 100644
index 0000000..8e7f307
--- /dev/null
+++ b/tests/data/system/order.h
@@ -0,0 +1 @@
+int system_order;
diff --git a/tests/data/system/system.h b/tests/data/system/system.h
new file mode 100644
index 0000000..c94d2ab
--- /dev/null
+++ b/tests/data/system/system.h
@@ -0,0 +1 @@
+int from_system;
diff --git a/tests/data/unbalanced.c b/tests/data/unbalanced.c
new file mode 100644
index 0000000..1bf56c1
--- /dev/null
+++ b/tests/data/unbalanced.c
@@ -0,0 +1,2 @@
+#lang "diff"
+y' = 1;
diff --git a/tests/data/user/order.h b/tests/data/user/order.h
new file mode 100644
index 0000000..f6313ca
--- /dev/null
+++ b/tests/data/user/order.h
@@ -0,0 +1 @@
+int user_order;
diff --git a/tests/data/utf8.c b/tests/data/utf8.c
new file mode 100644
index 0000000..d5153ea
--- /dev/null
+++ b/tests/data/utf8.c
@@ -0,0 +1,2 @@
+/* UTF-8 source, UCS-2 internal representation. */
+const char *message = "Привет, MCPU";
diff --git a/tests/mcpp-options-test.c b/tests/mcpp-options-test.c
new file mode 100644
index 0000000..a0e8311
--- /dev/null
+++ b/tests/mcpp-options-test.c
@@ -0,0 +1,24 @@
+#include <defs.h>
+
+int
+main( int argc, char **argv )
+{
+ mcpp_options options;
+ const char *p;
+ int rc = 1;
+
+ mcpp_options_init( &options, argc );
+ if( options.actions == NULL )
+ return( 1 );
+
+ p = argv[0] + strlen( argv[0] );
+ while( p != argv[0] && p[-1] != '/' ) --p;
+ options.progname = p;
+
+ if( mcpp_handle_options(&options, argc - 1, argv + 1) == 0 &&
+ mcpp_options_dump(&options, stdout) == 0 )
+ rc = 0;
+
+ mcpp_options_free( &options );
+ return( rc );
+}
diff --git a/tests/t0001-utf8.sh b/tests/t0001-utf8.sh
new file mode 100755
index 0000000..6726cfe
--- /dev/null
+++ b/tests/t0001-utf8.sh
@@ -0,0 +1,6 @@
+#!/bin/sh
+set -eu
+out="${TMPDIR-/tmp}/mcpu-cpp-utf8-$$.out"
+trap 'rm -f "$out"' EXIT HUP INT TERM
+"$MCPU_CPP" --no-config "$srcdir/data/utf8.c" -o "$out"
+grep 'Привет, MCPU' "$out" >/dev/null
diff --git a/tests/t0002-lang.sh b/tests/t0002-lang.sh
new file mode 100755
index 0000000..1d0102e
--- /dev/null
+++ b/tests/t0002-lang.sh
@@ -0,0 +1,8 @@
+#!/bin/sh
+set -eu
+out="${TMPDIR-/tmp}/mcpu-cpp-lang-$$.out"
+trap 'rm -f "$out"' EXIT HUP INT TERM
+"$MCPU_CPP" --no-config "$srcdir/data/lang.c" -o "$out"
+grep '#lang "diff"' "$out" >/dev/null
+grep '#lang "alg"' "$out" >/dev/null
+grep '#endlang' "$out" >/dev/null
diff --git a/tests/t0003-include.sh b/tests/t0003-include.sh
new file mode 100755
index 0000000..b8ddcec
--- /dev/null
+++ b/tests/t0003-include.sh
@@ -0,0 +1,11 @@
+#!/bin/sh
+set -eu
+out="${TMPDIR-/tmp}/mcpu-cpp-include-$$.out"
+trap 'rm -f "$out"' EXIT HUP INT TERM
+"$MCPU_CPP" --no-config -I "$srcdir/data/include" \
+ -isystem "$srcdir/data/system" \
+ "$srcdir/data/include-main.c" -o "$out"
+grep 'from_one' "$out" >/dev/null
+grep 'from_system' "$out" >/dev/null
+grep 'bridge_c' "$out" >/dev/null
+grep "y' = 1" "$out" >/dev/null
diff --git a/tests/t0004-config.sh b/tests/t0004-config.sh
new file mode 100755
index 0000000..d025d99
--- /dev/null
+++ b/tests/t0004-config.sh
@@ -0,0 +1,12 @@
+#!/bin/sh
+set -eu
+out="${TMPDIR-/tmp}/mcpu-cpp-config-$$.out"
+conf="${TMPDIR-/tmp}/mcpu-cpp-config-$$.conf"
+trap 'rm -f "$out" "$conf"' EXIT HUP INT TERM
+cat > "$conf" <<EOT
+MCPU_CPP_INCLUDE_PATH = $srcdir/data/include;
+MCPU_CPP_SYSTEM_INCLUDE_PATH = $srcdir/data/system;
+EOT
+"$MCPU_CPP" --config-file "$conf" "$srcdir/data/config-main.c" -o "$out"
+grep 'from_one' "$out" >/dev/null
+grep 'from_system' "$out" >/dev/null
diff --git a/tests/t0005-errors.sh b/tests/t0005-errors.sh
new file mode 100755
index 0000000..458b4c8
--- /dev/null
+++ b/tests/t0005-errors.sh
@@ -0,0 +1,22 @@
+#!/bin/sh
+set -eu
+out="${TMPDIR-/tmp}/mcpu-cpp-error-$$.out"
+err="${TMPDIR-/tmp}/mcpu-cpp-error-$$.err"
+bad="${TMPDIR-/tmp}/mcpu-cpp-nonucs2-$$.c"
+trap 'rm -f "$out" "$err" "$bad"' EXIT HUP INT TERM
+
+if "$MCPU_CPP" --no-config "$srcdir/data/unbalanced.c" -o "$out" 2>"$err"; then
+ exit 1
+fi
+grep 'unbalanced #lang/#endlang' "$err" >/dev/null
+
+# A valid UTF-8 scalar outside UCS-2 is harmless inside a comment.
+"$MCPU_CPP" --no-config "$srcdir/data/nonucs2.c" -o "$out"
+grep '^int x;$' "$out" >/dev/null
+
+# The same scalar in program text is rejected after comment removal.
+printf 'int x = \360\237\230\200;\n' > "$bad"
+if "$MCPU_CPP" --no-config "$bad" -o "$out" 2>"$err"; then
+ exit 1
+fi
+grep 'character outside UCS-2' "$err" >/dev/null
diff --git a/tests/t0006-path-order.sh b/tests/t0006-path-order.sh
new file mode 100755
index 0000000..0b9f76e
--- /dev/null
+++ b/tests/t0006-path-order.sh
@@ -0,0 +1,13 @@
+#!/bin/sh
+set -eu
+out="${TMPDIR-/tmp}/mcpu-cpp-order-$$.out"
+trap 'rm -f "$out"' EXIT HUP INT TERM
+# Deliberately put -isystem first: -I must still have higher semantic priority.
+"$MCPU_CPP" --no-config \
+ -isystem "$srcdir/data/system" \
+ -I "$srcdir/data/user" \
+ "$srcdir/data/order-main.c" -o "$out"
+grep 'user_order' "$out" >/dev/null
+if grep 'system_order' "$out" >/dev/null; then
+ exit 1
+fi
diff --git a/tests/t0007-language-path.sh b/tests/t0007-language-path.sh
new file mode 100755
index 0000000..6a75797
--- /dev/null
+++ b/tests/t0007-language-path.sh
@@ -0,0 +1,12 @@
+#!/bin/sh
+set -eu
+out="${TMPDIR-/tmp}/mcpu-cpp-langpath-$$.out"
+conf="${TMPDIR-/tmp}/mcpu-cpp-langpath-$$.conf"
+trap 'rm -f "$out" "$conf"' EXIT HUP INT TERM
+cat > "$conf" <<EOT
+MCPU_CPP_AS_INCLUDE_PATH = $srcdir/data/lang-as;
+MCPU_CPP_DIFF_INCLUDE_PATH = $srcdir/data/lang-diff;
+EOT
+"$MCPU_CPP" --config-file "$conf" "$srcdir/data/lang-path-main.c" -o "$out"
+grep 'from_as_language_path' "$out" >/dev/null
+grep 'from_diff_language_path' "$out" >/dev/null
diff --git a/tests/t0008-comments.sh b/tests/t0008-comments.sh
new file mode 100755
index 0000000..26386ba
--- /dev/null
+++ b/tests/t0008-comments.sh
@@ -0,0 +1,41 @@
+#!/bin/sh
+set -eu
+src="${TMPDIR-/tmp}/mcpu-cpp-comments-$$.c"
+out="${TMPDIR-/tmp}/mcpu-cpp-comments-$$.out"
+trap 'rm -f "$src" "$out"' EXIT HUP INT TERM
+cat > "$src" <<'EOT'
+/*
+#lang "diff"
+*/
+int c_side;
+/* comment before directive */ #lang "diff"
+y' = 1;
+#endlang
+ // one-line comment
+ /*
+ multi-line comment
+ */
+LEFT/**/RIGHT
+int block_tail; /* block tail */
+int line_tail; // line tail
+int multi_tail; /*
+ block tail
+ */
+EOT
+"$MCPU_CPP" --no-config "$src" -o "$out"
+grep 'int c_side' "$out" >/dev/null
+grep "y' = 1" "$out" >/dev/null
+grep '^LEFT RIGHT$' "$out" >/dev/null
+
+# Comment-only lines must be really empty; no synthetic or leading spaces remain.
+if grep '^[[:blank:]][[:blank:]]*$' "$out" >/dev/null; then
+ exit 1
+fi
+
+# A comment ending a non-empty line must not leave trailing whitespace.
+grep '^int block_tail;$' "$out" >/dev/null
+grep '^int line_tail;$' "$out" >/dev/null
+grep '^int multi_tail;$' "$out" >/dev/null
+if grep '[[:blank:]]$' "$out" >/dev/null; then
+ exit 1
+fi
diff --git a/tests/t0009-text-scanner.sh b/tests/t0009-text-scanner.sh
new file mode 100755
index 0000000..773968f
--- /dev/null
+++ b/tests/t0009-text-scanner.sh
@@ -0,0 +1,37 @@
+#!/bin/sh
+set -eu
+src="${TMPDIR-/tmp}/mcpu-cpp-scan-$$.c"
+out="${TMPDIR-/tmp}/mcpu-cpp-scan-$$.out"
+err="${TMPDIR-/tmp}/mcpu-cpp-scan-$$.err"
+bom="${TMPDIR-/tmp}/mcpu-cpp-bom-$$.c"
+nul="${TMPDIR-/tmp}/mcpu-cpp-nul-$$.c"
+trap 'rm -f "$src" "$out" "$err" "$bom" "$nul"' EXIT HUP INT TERM
+
+cat > "$src" <<'EOT'
+const char *a = "/* not a comment";
+#lang "diff"
+y' = 1;
+const char *b = "*/ still not a comment";
+#endlang
+const char *c = "// not a comment";
+/* real comment starts here
+#lang "alg"
+*/
+int c_side;
+EOT
+"$MCPU_CPP" --no-config "$src" -o "$out"
+grep '#lang "diff"' "$out" >/dev/null
+grep "y' = 1" "$out" >/dev/null
+grep 'int c_side' "$out" >/dev/null
+
+# UTF-8 BOM is accepted and removed at the external text boundary.
+printf '\357\273\277int bom_ok;\n' > "$bom"
+"$MCPU_CPP" --no-config "$bom" -o "$out"
+grep 'int bom_ok' "$out" >/dev/null
+
+# Embedded NUL is not text and must never be silently truncated.
+printf 'int before;\000int after;\n' > "$nul"
+if "$MCPU_CPP" --no-config "$nul" -o "$out" 2>"$err"; then
+ exit 1
+fi
+grep 'NUL character is not allowed' "$err" >/dev/null
diff --git a/tests/t0010-language-names.sh b/tests/t0010-language-names.sh
new file mode 100755
index 0000000..83d4325
--- /dev/null
+++ b/tests/t0010-language-names.sh
@@ -0,0 +1,38 @@
+#!/bin/sh
+set -eu
+dir="${TMPDIR-/tmp}/mcpu-cpp-languages-$$"
+mkdir -p "$dir"
+trap 'rm -rf "$dir"' EXIT HUP INT TERM
+
+cat > "$dir/ok.c" <<'EOT'
+#lang "diff"
+#endlang
+#lang "DIFT"
+#endlang
+#lang "Alg"
+#endlang
+#lang "aS"
+#endlang
+#lang "AvM"
+#endlang
+#lang "acs"
+#endlang
+EOT
+"$MCPU_CPP" --no-config "$dir/ok.c" -o "$dir/ok.out"
+grep '^#lang "diff"$' "$dir/ok.out" >/dev/null
+grep '^#lang "DIFT"$' "$dir/ok.out" >/dev/null
+grep '^#lang "Alg"$' "$dir/ok.out" >/dev/null
+grep '^#lang "aS"$' "$dir/ok.out" >/dev/null
+grep '^#lang "AvM"$' "$dir/ok.out" >/dev/null
+grep '^#lang "acs"$' "$dir/ok.out" >/dev/null
+
+for lang in 0 c vasm unknown
+do
+ printf '#lang "%s"\n#endlang\n' "$lang" > "$dir/bad.c"
+ if "$MCPU_CPP" --no-config "$dir/bad.c" -o "$dir/bad.out" 2> "$dir/bad.err"
+ then
+ echo "#lang \"$lang\" unexpectedly accepted" >&2
+ exit 1
+ fi
+ grep "unknown language '$lang'" "$dir/bad.err" >/dev/null
+done
diff --git a/tests/t0011-base-include-path.sh b/tests/t0011-base-include-path.sh
new file mode 100755
index 0000000..f862e88
--- /dev/null
+++ b/tests/t0011-base-include-path.sh
@@ -0,0 +1,19 @@
+#!/bin/sh
+set -eu
+dir="${TMPDIR-/tmp}/mcpu-cpp-basepath-$$"
+mkdir -p "$dir"
+trap 'rm -rf "$dir"' EXIT HUP INT TERM
+cat > "$dir/config" <<EOT
+MCPU_CPP_INCLUDE_PATH = $srcdir/data/lang-base;
+EOT
+"$MCPU_CPP" --config-file "$dir/config" "$srcdir/data/base-path-main.c" -o "$dir/base.out"
+grep 'from_base_language_include_root' "$dir/base.out" >/dev/null
+cat > "$dir/as.c" <<'EOT'
+#lang "as"
+#include <base.h>
+#include <diff/explicit.h>
+#endlang
+EOT
+"$MCPU_CPP" --config-file "$dir/config" "$dir/as.c" -o "$dir/as.out"
+grep 'from_base_language_include_root' "$dir/as.out" >/dev/null
+grep 'from_explicit_diff_subdirectory' "$dir/as.out" >/dev/null
diff --git a/tests/t0012-user-config-path.sh b/tests/t0012-user-config-path.sh
new file mode 100755
index 0000000..54b6e64
--- /dev/null
+++ b/tests/t0012-user-config-path.sh
@@ -0,0 +1,26 @@
+#!/bin/sh
+set -eu
+dir="${TMPDIR-/tmp}/mcpu-cpp-userconf-$$"
+mkdir -p "$dir/home/.mcpu" "$dir/include" "$dir/system/as"
+trap 'rm -rf "$dir"' EXIT HUP INT TERM
+cat > "$dir/home/.mcpu/mcpu-cpp.conf" <<EOT
+MCPU_CPP_INCLUDE_PATH = $dir/include;
+MCPU_CPP_SYSTEM_INCLUDE_PATH = $dir/system;
+EOT
+cat > "$dir/include/userconf.h" <<'EOT'
+int from_user_mcpu_configuration;
+EOT
+cat > "$dir/system/as/system-userconf.h" <<'EOT'
+int from_user_system_root;
+EOT
+cat > "$dir/main.c" <<'EOT'
+#include <userconf.h>
+#lang "as"
+#include <system-userconf.h>
+#endlang
+EOT
+HOME="$dir/home" "$MCPU_CPP" "$dir/main.c" -o "$dir/out"
+grep 'from_user_mcpu_configuration' "$dir/out" >/dev/null
+grep 'from_user_system_root' "$dir/out" >/dev/null
+HOME="$dir/home" "$MCPU_CPP" -dconfig > "$dir/config.out"
+grep "^MCPU_CPP_SYSTEM_INCLUDE_PATH = $dir/system;$" "$dir/config.out" >/dev/null
diff --git a/tests/t0013-include-precedence.sh b/tests/t0013-include-precedence.sh
new file mode 100755
index 0000000..0bfc1e5
--- /dev/null
+++ b/tests/t0013-include-precedence.sh
@@ -0,0 +1,18 @@
+#!/bin/sh
+set -eu
+dir="${TMPDIR-/tmp}/mcpu-cpp-precedence-$$"
+mkdir -p "$dir"
+trap 'rm -rf "$dir"' EXIT HUP INT TERM
+cat > "$dir/config" <<EOT
+MCPU_CPP_INCLUDE_PATH = $srcdir/data/common-precedence;
+MCPU_CPP_AS_INCLUDE_PATH = $srcdir/data/lang-as-precedence;
+EOT
+cat > "$dir/main.c" <<'EOT'
+#include <same.h>
+#lang "as"
+#include <same.h>
+#endlang
+EOT
+"$MCPU_CPP" --config-file "$dir/config" "$dir/main.c" -o "$dir/out"
+grep 'from_common_include_root' "$dir/out" >/dev/null
+grep 'from_as_specific_include_path' "$dir/out" >/dev/null
diff --git a/tests/t0014-preprocessing-phases.sh b/tests/t0014-preprocessing-phases.sh
new file mode 100755
index 0000000..5cc98c8
--- /dev/null
+++ b/tests/t0014-preprocessing-phases.sh
@@ -0,0 +1,21 @@
+#!/bin/sh
+set -eu
+src="${TMPDIR-/tmp}/mcpu-cpp-phases-$$.c"
+out="${TMPDIR-/tmp}/mcpu-cpp-phases-$$.out"
+trap 'rm -f "$src" "$out"' EXIT HUP INT TERM
+cat > "$src" <<'EOT'
+#defi\
+ne VALUE 10\
+20
+/* comment */ VALUE // trailing comment
+const char *s = "VALUE /* not a comment */";
+EOT
+"$MCPU_CPP" --no-config "$src" -o "$out"
+grep '1020' "$out" >/dev/null
+grep 'VALUE /\* not a comment \*/' "$out" >/dev/null
+if grep 'trailing comment' "$out" >/dev/null; then
+ exit 1
+fi
+if grep '/\* comment \*/' "$out" >/dev/null; then
+ exit 1
+fi
diff --git a/tests/t0015-object-macros.sh b/tests/t0015-object-macros.sh
new file mode 100755
index 0000000..a59bd23
--- /dev/null
+++ b/tests/t0015-object-macros.sh
@@ -0,0 +1,27 @@
+#!/bin/sh
+set -eu
+src="${TMPDIR-/tmp}/mcpu-cpp-macro-$$.c"
+out="${TMPDIR-/tmp}/mcpu-cpp-macro-$$.out"
+trap 'rm -f "$src" "$out"' EXIT HUP INT TERM
+cat > "$src" <<'EOT'
+#define A B
+#define B 42
+#define EMPTY
+A
+x EMPTY y
+"A B"
+#define SELF SELF
+SELF
+#undef B
+A
+#lang "diff"
+y' = A;
+#endlang
+EOT
+"$MCPU_CPP" --no-config "$src" -o "$out"
+grep '^42$' "$out" >/dev/null
+grep '^x y$' "$out" >/dev/null
+grep '^"A B"$' "$out" >/dev/null
+grep '^SELF$' "$out" >/dev/null
+grep '^B$' "$out" >/dev/null
+grep "y' = B;" "$out" >/dev/null
diff --git a/tests/t0016-macro-include.sh b/tests/t0016-macro-include.sh
new file mode 100755
index 0000000..0c1b56e
--- /dev/null
+++ b/tests/t0016-macro-include.sh
@@ -0,0 +1,16 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-minclude-$$"
+mkdir -p "$base/inc"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+cat > "$base/main.c" <<'EOT'
+#define HEADER <macro.h>
+#include HEADER
+VALUE
+EOT
+cat > "$base/inc/macro.h" <<'EOT'
+#define VALUE 77
+EOT
+"$MCPU_CPP" --no-config -I "$base/inc" "$base/main.c" -o "$base/out"
+grep '^77$' "$base/out" >/dev/null
+grep "^# 3 \"$base/main.c\" 2$" "$base/out" >/dev/null
diff --git a/tests/t0017-macro-errors.sh b/tests/t0017-macro-errors.sh
new file mode 100755
index 0000000..f54f74c
--- /dev/null
+++ b/tests/t0017-macro-errors.sh
@@ -0,0 +1,13 @@
+#!/bin/sh
+set -eu
+src="${TMPDIR-/tmp}/mcpu-cpp-macro-errors-$$.c"
+out="${TMPDIR-/tmp}/mcpu-cpp-macro-errors-$$.out"
+err="${TMPDIR-/tmp}/mcpu-cpp-macro-errors-$$.err"
+trap 'rm -f "$src" "$out" "$err"' EXIT HUP INT TERM
+cat > "$src" <<'EOT'
+#define
+EOT
+if "$MCPU_CPP" --no-config "$src" -o "$out" 2>"$err"; then
+ exit 1
+fi
+grep 'macro name expected after #define' "$err" >/dev/null
diff --git a/tests/t0018-include-lexing.sh b/tests/t0018-include-lexing.sh
new file mode 100755
index 0000000..ed1561d
--- /dev/null
+++ b/tests/t0018-include-lexing.sh
@@ -0,0 +1,17 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-include-lex-$$"
+mkdir -p "$base/inc/x"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+cat > "$base/main.c" <<'EOT'
+#include \
+<x/*y>
+AFTER
+EOT
+cat > "$base/inc/x/*y" <<'EOT'
+INSIDE
+EOT
+"$MCPU_CPP" --no-config -I "$base/inc" "$base/main.c" -o "$base/out"
+grep '^INSIDE$' "$base/out" >/dev/null
+grep '^AFTER$' "$base/out" >/dev/null
+grep "^# 3 \"$base/main.c\" 2$" "$base/out" >/dev/null
diff --git a/tests/t0019-recursive-macro.sh b/tests/t0019-recursive-macro.sh
new file mode 100755
index 0000000..ffba144
--- /dev/null
+++ b/tests/t0019-recursive-macro.sh
@@ -0,0 +1,13 @@
+#!/bin/sh
+set -eu
+src="${TMPDIR-/tmp}/mcpu-cpp-recursive-$$.c"
+out="${TMPDIR-/tmp}/mcpu-cpp-recursive-$$.out"
+trap 'rm -f "$src" "$out"' EXIT HUP INT TERM
+cat > "$src" <<'EOT'
+int foo;
+#define foo (4 + foo)
+f = foo;
+EOT
+"$MCPU_CPP" --no-config "$src" -o "$out"
+grep '^int foo;$' "$out" >/dev/null
+grep '^f = (4 + foo);$' "$out" >/dev/null
diff --git a/tests/t0020-function-macros.sh b/tests/t0020-function-macros.sh
new file mode 100755
index 0000000..aba3fc5
--- /dev/null
+++ b/tests/t0020-function-macros.sh
@@ -0,0 +1,42 @@
+#!/bin/sh
+set -eu
+src="${TMPDIR-/tmp}/mcpu-cpp-fmacro-$$.c"
+out="${TMPDIR-/tmp}/mcpu-cpp-fmacro-$$.out"
+trap 'rm -f "$src" "$out"' EXIT HUP INT TERM
+cat > "$src" <<'EOT'
+#define min(X, Y) ((X) < (Y) ? (X) : (Y))
+#define A 7
+#define ZERO() zero-text
+#define INNER(X) [X]
+#define OUTER(X, Y) INNER(X) + INNER(Y)
+#define OBJECT (argument)
+#define WRAP(X) <X>
+#define TWICE(X) (X) + \
+(X)
+min(1, 2)
+min(A, 3)
+min(min(a, b), c)
+OUTER(A, min(4, 5))
+ZERO ()
+ZERO
+OBJECT
+WRAP("a,b")
+TWICE(3)
+"min(A, 3)"
+#lang "diff"
+#define DERIV(X) X' + X
+DERIV(y)
+#endlang
+EOT
+"$MCPU_CPP" --no-config "$src" -o "$out"
+grep '^((1) < (2) ? (1) : (2))$' "$out" >/dev/null
+grep '^((7) < (3) ? (7) : (3))$' "$out" >/dev/null
+grep '^((((a) < (b) ? (a) : (b))) < (c) ? (((a) < (b) ? (a) : (b))) : (c))$' "$out" >/dev/null
+grep '^\[7\] + \[((4) < (5) ? (4) : (5))\]$' "$out" >/dev/null
+grep '^zero-text$' "$out" >/dev/null
+grep '^ZERO$' "$out" >/dev/null
+grep '^(argument)$' "$out" >/dev/null
+grep '^<"a,b">$' "$out" >/dev/null
+grep '^(3) + (3)$' "$out" >/dev/null
+grep '^"min(A, 3)"$' "$out" >/dev/null
+grep "^y' + y$" "$out" >/dev/null
diff --git a/tests/t0021-function-macro-errors.sh b/tests/t0021-function-macro-errors.sh
new file mode 100755
index 0000000..2353b35
--- /dev/null
+++ b/tests/t0021-function-macro-errors.sh
@@ -0,0 +1,89 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-fmacro-errors-$$"
+trap 'rm -f "$base".*' EXIT HUP INT TERM
+
+cat > "$base.dup.c" <<'EOT'
+#define F(X, X) X
+EOT
+if "$MCPU_CPP" --no-config "$base.dup.c" -o "$base.out" 2>"$base.err"; then
+ echo "duplicate parameter name accepted" >&2
+ exit 1
+fi
+grep "duplicate argument name 'X'" "$base.err" >/dev/null
+
+cat > "$base.few.c" <<'EOT'
+#define F(X, Y) X + Y
+F(1)
+EOT
+if "$MCPU_CPP" --no-config "$base.few.c" -o "$base.out" 2>"$base.err"; then
+ echo "too few macro arguments accepted" >&2
+ exit 1
+fi
+grep "used with too few arguments" "$base.err" >/dev/null
+
+cat > "$base.many.c" <<'EOT'
+#define F(X) X
+F(1, 2)
+EOT
+if "$MCPU_CPP" --no-config "$base.many.c" -o "$base.out" 2>"$base.err"; then
+ echo "too many macro arguments accepted" >&2
+ exit 1
+fi
+grep "used with too many arguments" "$base.err" >/dev/null
+
+cat > "$base.unterm.c" <<'EOT'
+#define F(X) X
+F(1
+EOT
+if "$MCPU_CPP" --no-config "$base.unterm.c" -o "$base.out" 2>"$base.err"; then
+ echo "unterminated macro call accepted" >&2
+ exit 1
+fi
+grep "unterminated argument list" "$base.err" >/dev/null
+
+cat > "$base.concat-first.c" <<'EOT'
+#define CAT(X, Y) ## X
+EOT
+if "$MCPU_CPP" --no-config "$base.concat-first.c" -o "$base.out" 2>"$base.err"; then
+ echo "leading ## in replacement list accepted" >&2
+ exit 1
+fi
+grep "'##' cannot appear at the beginning of a macro replacement list" "$base.err" >/dev/null
+
+cat > "$base.concat-last.c" <<'EOT'
+#define CAT(X, Y) Y ##
+EOT
+if "$MCPU_CPP" --no-config "$base.concat-last.c" -o "$base.out" 2>"$base.err"; then
+ echo "trailing ## in replacement list accepted" >&2
+ exit 1
+fi
+grep "'##' cannot appear at the end of a macro replacement list" "$base.err" >/dev/null
+
+cat > "$base.sharp-name.c" <<'EOT'
+#define S(X) #Y
+EOT
+if "$MCPU_CPP" --no-config "$base.sharp-name.c" -o "$base.out" 2>"$base.err"; then
+ echo "stringification of a non-parameter accepted" >&2
+ exit 1
+fi
+grep "'#' operator should be followed by a macro argument name" "$base.err" >/dev/null
+
+cat > "$base.sharp-end.c" <<'EOT'
+#define S(X) #
+EOT
+if "$MCPU_CPP" --no-config "$base.sharp-end.c" -o "$base.out" 2>"$base.err"; then
+ echo "unterminated stringification operator accepted" >&2
+ exit 1
+fi
+grep "'#' operator is not followed by a macro argument name" "$base.err" >/dev/null
+
+cat > "$base.bracket.c" <<'EOT'
+#define F(X) X
+F(array[x = y, x + 1])
+EOT
+if "$MCPU_CPP" --no-config "$base.bracket.c" -o "$base.out" 2>"$base.err"; then
+ echo "comma in brackets incorrectly protected macro argument" >&2
+ exit 1
+fi
+grep "used with too many arguments" "$base.err" >/dev/null
diff --git a/tests/t0022-predefined-macros.sh b/tests/t0022-predefined-macros.sh
new file mode 100755
index 0000000..960e1e8
--- /dev/null
+++ b/tests/t0022-predefined-macros.sh
@@ -0,0 +1,56 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-predef-$$"
+mkdir -p "$base"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+cat > "$base/main.c" <<'EOT'
+root_file = __FILE__;
+root_base = __BASE_FILE__;
+root_level = __INCLUDE_LEVEL__;
+#define HERE() __LINE__
+root_line = HERE();
+cpp_version = __MCPU_CPP_VERSION__;
+old_version_name = __VERSION__;
+date = __DATE__;
+time = __TIME__;
+#include "one.h"
+after_line = __LINE__;
+EOT
+
+cat > "$base/one.h" <<'EOT'
+one_file = __FILE__;
+one_base = __BASE_FILE__;
+one_level = __INCLUDE_LEVEL__;
+one_line = __LINE__;
+#include "two.h"
+EOT
+
+cat > "$base/two.h" <<'EOT'
+two_file = __FILE__;
+two_base = __BASE_FILE__;
+two_level = __INCLUDE_LEVEL__;
+two_line = __LINE__;
+EOT
+
+"$MCPU_CPP" --no-config "$base/main.c" -o "$base/out"
+
+grep -F "root_file = \"$base/main.c\";" "$base/out" >/dev/null
+grep -F "root_base = \"$base/main.c\";" "$base/out" >/dev/null
+grep '^root_level = 0;$' "$base/out" >/dev/null
+grep '^root_line = 5;$' "$base/out" >/dev/null
+grep '^cpp_version = "1\.0\.2";$' "$base/out" >/dev/null
+grep '^old_version_name = __VERSION__;$' "$base/out" >/dev/null
+grep '^date = "[A-Z][a-z][a-z] [0-9] [0-9][0-9][0-9][0-9]";$\|^date = "[A-Z][a-z][a-z] [12][0-9] [0-9][0-9][0-9][0-9]";$\|^date = "[A-Z][a-z][a-z] 3[01] [0-9][0-9][0-9][0-9]";$' "$base/out" >/dev/null
+grep '^time = "[0-2][0-9]:[0-5][0-9]:[0-5][0-9]";$' "$base/out" >/dev/null
+
+grep -F "one_file = \"$base/one.h\";" "$base/out" >/dev/null
+grep -F "one_base = \"$base/main.c\";" "$base/out" >/dev/null
+grep '^one_level = 1;$' "$base/out" >/dev/null
+grep '^one_line = 4;$' "$base/out" >/dev/null
+
+grep -F "two_file = \"$base/two.h\";" "$base/out" >/dev/null
+grep -F "two_base = \"$base/main.c\";" "$base/out" >/dev/null
+grep '^two_level = 2;$' "$base/out" >/dev/null
+grep '^two_line = 4;$' "$base/out" >/dev/null
+grep '^after_line = 11;$' "$base/out" >/dev/null
diff --git a/tests/t0023-predefined-redefine.sh b/tests/t0023-predefined-redefine.sh
new file mode 100755
index 0000000..b30a1d9
--- /dev/null
+++ b/tests/t0023-predefined-redefine.sh
@@ -0,0 +1,20 @@
+#!/bin/sh
+set -eu
+src="${TMPDIR-/tmp}/mcpu-cpp-predef-redef-$$.c"
+out="${TMPDIR-/tmp}/mcpu-cpp-predef-redef-$$.out"
+err="${TMPDIR-/tmp}/mcpu-cpp-predef-redef-$$.err"
+trap 'rm -f "$src" "$out" "$err"' EXIT HUP INT TERM
+cat > "$src" <<'EOT'
+before = __LINE__;
+#define __LINE__ 77
+after = __LINE__;
+#undef __LINE__
+literal = __LINE__;
+"__FILE__ __LINE__ __MCPU_CPP_VERSION__"
+EOT
+"$MCPU_CPP" --no-config "$src" -o "$out" 2>"$err"
+grep '^before = 1;$' "$out" >/dev/null
+grep '^literal = __LINE__;$' "$out" >/dev/null
+grep '^after = 77;$' "$out" >/dev/null
+grep '^"__FILE__ __LINE__ __MCPU_CPP_VERSION__"$' "$out" >/dev/null
+grep "warning: macro '__LINE__' redefined" "$err" >/dev/null
diff --git a/tests/t0024-abi-predefined.sh b/tests/t0024-abi-predefined.sh
new file mode 100755
index 0000000..984f0a2
--- /dev/null
+++ b/tests/t0024-abi-predefined.sh
@@ -0,0 +1,191 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-abi-predef-$$"
+mkdir -p "$base"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+cat > "$base/main.c" <<'EOT'
+arch = _ARCH_MCPU;
+cpp_version = __MCPU_CPP_VERSION__;
+old_version = __VERSION__;
+byte_order = __BYTE_ORDER__;
+mcpu_byte_order = __MCPU_BYTE_ORDER__;
+word_order = __MCPU_WORD_ORDER__;
+old_float_word_order = __FLOAT_WORD_ORDER__;
+register_width = __MCPU_MACHINE_REGISTER_WIDTH__;
+real_io_limit = __MCPU_REAL_IO_LIMIT__;
+math_fn_limit = __MCPU_MATH_FN_LIMIT__;
+pointer_width = __MCPU_POINTER_WIDTH__;
+int_max_width = __MCPU_INT_MAX_WIDTH__;
+real_max_width = __MCPU_REAL_MAX_WIDTH__;
+complex_max_width = __MCPU_COMPLEX_MAX_WIDTH__;
+size_type = __MCPU_SIZE_TYPE__;
+size_width = __MCPU_SIZE_WIDTH__;
+size_sizeof = __MCPU_SIZEOF_SIZE__;
+size_max = __MCPU_SIZE_MAX__;
+ssize_type = __MCPU_SSIZE_TYPE__;
+ssize_width = __MCPU_SSIZE_WIDTH__;
+ssize_sizeof = __MCPU_SIZEOF_SSIZE__;
+ssize_max = __MCPU_SSIZE_MAX__;
+ptrdiff_type = __PTRDIFF_TYPE__;
+ptrdiff_width = __PTRDIFF_WIDTH__;
+ptrdiff_max = __PTRDIFF_MAX__;
+sizeof_ptrdiff = __SIZEOF_PTRDIFF__;
+old_sizeof_ptrdiff_t = __SIZEOF_PTRDIFF_T__;
+intptr_type = __INTPTR_TYPE__;
+uintptr_type = __UINTPTR_TYPE__;
+intptr_width = __INTPTR_WIDTH__;
+uintptr_width = __UINTPTR_WIDTH__;
+intptr_max = __INTPTR_MAX__;
+uintptr_max = __UINTPTR_MAX__;
+char8_type = __CHAR8_TYPE__;
+char16_type = __CHAR16_TYPE__;
+char8_width = __CHAR8_WIDTH__;
+char16_width = __CHAR16_WIDTH__;
+sizeof_char8 = __SIZEOF_CHAR8__;
+sizeof_char16 = __SIZEOF_CHAR16__;
+register_prefix = [__REGISTER_PREFIX__];
+local_label_prefix = [__LOCAL_LABEL_PREFIX__];
+user_label_prefix = [__USER_LABEL_PREFIX__];
+immediate_prefix = [__IMMEDIATE_PREFIX__];
+int128_type = __INT128_TYPE__;
+uint128_type = __UINT128_TYPE__;
+int128_width = __INT128_WIDTH__;
+int128_decimal = __INT128_DECIMAL_DIG__;
+uint128_decimal = __UINT128_DECIMAL_DIG__;
+sizeof_int128 = __SIZEOF_INT128__;
+sizeof_uint128 = __SIZEOF_UINT128__;
+int128_max = __INT128_MAX__;
+int128_min_must_not_exist = __INT128_MIN__;
+uint128_max = __UINT128_MAX__;
+int1024_type = __INT1024_TYPE__;
+int1024_width = __INT1024_WIDTH__;
+int1024_max_must_not_exist = __INT1024_MAX__;
+int1024_decimal = __INT1024_DECIMAL_DIG__;
+sizeof_int1024 = __SIZEOF_INT1024__;
+real128_type = __REAL128_TYPE__;
+complex128_type = __COMPLEX128_TYPE__;
+real128_width = __REAL128_WIDTH__;
+complex128_width = __COMPLEX128_WIDTH__;
+sizeof_real128 = __SIZEOF_REAL128__;
+sizeof_real128_exp = __SIZEOF_REAL128_EXP__;
+real128_max_strlen = __REAL128_MAX_STRLEN__;
+sizeof_complex128 = __SIZEOF_COMPLEX128__;
+real128_mant = __REAL128_MANT_DIG__;
+real128_decimal = __REAL128_DECIMAL_DIG__;
+real128_old_dig_must_not_exist = __REAL128_DIG__;
+real128_max = __REAL128_MAX__;
+real128_min = __REAL128_MIN__;
+real128_epsilon = __REAL128_EPSILON__;
+real128_max_exp = __REAL128_MAX_EXP__;
+real128_min_exp = __REAL128_MIN_EXP__;
+real1024_type = __REAL1024_TYPE__;
+real1024_width = __REAL1024_WIDTH__;
+real1024_max_must_not_exist = __REAL1024_MAX__;
+real1024_mant = __REAL1024_MANT_DIG__;
+real1024_decimal = __REAL1024_DECIMAL_DIG__;
+sizeof_real1024 = __SIZEOF_REAL1024__;
+sizeof_real1024_exp = __SIZEOF_REAL1024_EXP__;
+real1024_max_strlen = __REAL1024_MAX_STRLEN__;
+complex1024_type = __COMPLEX1024_TYPE__;
+complex1024_width = __COMPLEX1024_WIDTH__;
+sizeof_complex1024 = __SIZEOF_COMPLEX1024__;
+int65536_type = __INT65536_TYPE__;
+int65536_width = __INT65536_WIDTH__;
+sizeof_int65536 = __SIZEOF_INT65536__;
+char_type_must_not_exist = __CHAR_TYPE__;
+wchar_type_must_not_exist = __WCHAR_TYPE__;
+EOT
+
+"$MCPU_CPP" --no-config "$base/main.c" -o "$base/out"
+
+grep '^arch = 1;$' "$base/out" >/dev/null
+grep '^cpp_version = "1\.0\.2";$' "$base/out" >/dev/null
+grep '^old_version = __VERSION__;$' "$base/out" >/dev/null
+grep '^byte_order = \(1234\|4321\);$' "$base/out" >/dev/null
+grep '^mcpu_byte_order = \(1234\|4321\);$' "$base/out" >/dev/null
+grep '^word_order = \(1234\|4321\);$' "$base/out" >/dev/null
+grep '^old_float_word_order = __FLOAT_WORD_ORDER__;$' "$base/out" >/dev/null
+grep '^register_width = 64;$' "$base/out" >/dev/null
+grep '^pointer_width = 64;$' "$base/out" >/dev/null
+grep '^int_max_width = [1-9][0-9]*;$' "$base/out" >/dev/null
+grep '^real_max_width = [1-9][0-9]*;$' "$base/out" >/dev/null
+grep '^complex_max_width = [1-9][0-9]*;$' "$base/out" >/dev/null
+grep '^size_type = uint[0-9][0-9]*;$' "$base/out" >/dev/null
+grep '^size_width = [1-9][0-9]*;$' "$base/out" >/dev/null
+grep '^size_sizeof = [1-9][0-9]*;$' "$base/out" >/dev/null
+grep '^size_max = 0x[0-9a-f][0-9a-f]*;$' "$base/out" >/dev/null
+grep '^ssize_type = int[0-9][0-9]*;$' "$base/out" >/dev/null
+grep '^ssize_width = [1-9][0-9]*;$' "$base/out" >/dev/null
+grep '^ssize_sizeof = [1-9][0-9]*;$' "$base/out" >/dev/null
+grep '^ssize_max = 0x[0-9a-f][0-9a-f]*;$' "$base/out" >/dev/null
+grep '^ptrdiff_type = int64;$' "$base/out" >/dev/null
+grep '^ptrdiff_width = 64;$' "$base/out" >/dev/null
+grep '^ptrdiff_max = 0x7fffffffffffffff;$' "$base/out" >/dev/null
+grep '^sizeof_ptrdiff = 8;$' "$base/out" >/dev/null
+grep '^old_sizeof_ptrdiff_t = __SIZEOF_PTRDIFF_T__;$' "$base/out" >/dev/null
+grep '^intptr_type = int64;$' "$base/out" >/dev/null
+grep '^uintptr_type = uint64;$' "$base/out" >/dev/null
+grep '^intptr_width = 64;$' "$base/out" >/dev/null
+grep '^uintptr_width = 64;$' "$base/out" >/dev/null
+grep '^intptr_max = 0x7fffffffffffffff;$' "$base/out" >/dev/null
+grep '^uintptr_max = 0xffffffffffffffff;$' "$base/out" >/dev/null
+grep '^char8_type = char8;$' "$base/out" >/dev/null
+grep '^char16_type = char16;$' "$base/out" >/dev/null
+grep '^char8_width = 8;$' "$base/out" >/dev/null
+grep '^char16_width = 16;$' "$base/out" >/dev/null
+grep '^sizeof_char8 = 1;$' "$base/out" >/dev/null
+grep '^sizeof_char16 = 2;$' "$base/out" >/dev/null
+grep '^register_prefix = \[\];$' "$base/out" >/dev/null
+grep '^local_label_prefix = \[\];$' "$base/out" >/dev/null
+grep '^user_label_prefix = \[\];$' "$base/out" >/dev/null
+grep '^immediate_prefix = \[\];$' "$base/out" >/dev/null
+grep '^int128_type = int128;$' "$base/out" >/dev/null
+grep '^uint128_type = uint128;$' "$base/out" >/dev/null
+grep '^int128_width = 128;$' "$base/out" >/dev/null
+grep '^int128_decimal = 39;$' "$base/out" >/dev/null
+grep '^uint128_decimal = 39;$' "$base/out" >/dev/null
+grep '^sizeof_int128 = 16;$' "$base/out" >/dev/null
+grep '^sizeof_uint128 = 16;$' "$base/out" >/dev/null
+grep '^sizeof_real128_exp = [1-9][0-9]*;$' "$base/out" >/dev/null
+grep '^real128_max_strlen = [1-9][0-9]*;$' "$base/out" >/dev/null
+grep '^int128_max = 0x7fffffffffffffffffffffffffffffff;$' "$base/out" >/dev/null
+grep '^int128_min_must_not_exist = __INT128_MIN__;$' "$base/out" >/dev/null
+grep '^uint128_max = 0xffffffffffffffffffffffffffffffff;$' "$base/out" >/dev/null
+grep '^int1024_type = int1024;$' "$base/out" >/dev/null
+grep '^int1024_width = 1024;$' "$base/out" >/dev/null
+grep '^int1024_max_must_not_exist = __INT1024_MAX__;$' "$base/out" >/dev/null
+grep '^int1024_decimal = 308;$' "$base/out" >/dev/null
+grep '^sizeof_int1024 = 128;$' "$base/out" >/dev/null
+grep '^real128_type = real128;$' "$base/out" >/dev/null
+grep '^complex128_type = complex128;$' "$base/out" >/dev/null
+grep '^real128_width = 128;$' "$base/out" >/dev/null
+grep '^complex128_width = 128;$' "$base/out" >/dev/null
+grep '^sizeof_real128 = 16;$' "$base/out" >/dev/null
+grep '^sizeof_complex128 = 32;$' "$base/out" >/dev/null
+grep '^real128_mant = [1-9][0-9]*;$' "$base/out" >/dev/null
+grep '^real128_decimal = [1-9][0-9]*;$' "$base/out" >/dev/null
+grep '^real128_old_dig_must_not_exist = __REAL128_DIG__;$' "$base/out" >/dev/null
+grep '^real128_max = [0-9][0-9.]*e+[0-9][0-9]*;$' "$base/out" >/dev/null
+grep '^real128_min = [0-9][0-9.]*e-[0-9][0-9]*;$' "$base/out" >/dev/null
+grep '^real128_epsilon = [0-9][0-9.]*e-[0-9][0-9]*;$' "$base/out" >/dev/null
+grep '^real128_max_exp = [1-9][0-9]*;$' "$base/out" >/dev/null
+grep '^real128_min_exp = -[1-9][0-9]*;$' "$base/out" >/dev/null
+grep '^real1024_type = real1024;$' "$base/out" >/dev/null
+grep '^real1024_width = 1024;$' "$base/out" >/dev/null
+grep '^real1024_max_must_not_exist = __REAL1024_MAX__;$' "$base/out" >/dev/null
+grep '^real1024_mant = [1-9][0-9]*;$' "$base/out" >/dev/null
+grep '^real1024_decimal = [1-9][0-9]*;$' "$base/out" >/dev/null
+grep '^sizeof_real1024 = 128;$' "$base/out" >/dev/null
+grep '^sizeof_real1024_exp = [1-9][0-9]*;$' "$base/out" >/dev/null
+grep '^real1024_max_strlen = [1-9][0-9]*;$' "$base/out" >/dev/null
+grep '^complex1024_type = complex1024;$' "$base/out" >/dev/null
+grep '^complex1024_width = 1024;$' "$base/out" >/dev/null
+grep '^sizeof_complex1024 = 256;$' "$base/out" >/dev/null
+grep '^int65536_type = int65536;$' "$base/out" >/dev/null
+grep '^int65536_width = 65536;$' "$base/out" >/dev/null
+grep '^sizeof_int65536 = 8192;$' "$base/out" >/dev/null
+grep '^real_io_limit = [1-9][0-9]*;$' "$base/out" >/dev/null
+grep '^math_fn_limit = [1-9][0-9]*;$' "$base/out" >/dev/null
+grep '^char_type_must_not_exist = __CHAR_TYPE__;$' "$base/out" >/dev/null
+grep '^wchar_type_must_not_exist = __WCHAR_TYPE__;$' "$base/out" >/dev/null
diff --git a/tests/t0025-dump-macros.sh b/tests/t0025-dump-macros.sh
new file mode 100755
index 0000000..ea0ede5
--- /dev/null
+++ b/tests/t0025-dump-macros.sh
@@ -0,0 +1,99 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-dm-$$"
+mkdir -p "$base"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+"$MCPU_CPP" --no-config -dMP < /dev/null > "$base/macros"
+
+test -s "$base/macros"
+LC_ALL=C sort -c "$base/macros"
+grep '^#define _ARCH_MCPU 1$' "$base/macros" >/dev/null
+grep '^#define __MCPU_CPP_VERSION__ "1\.0\.2"$' "$base/macros" >/dev/null
+grep '^#define __MCPU_MACHINE_REGISTER_WIDTH__ 64$' "$base/macros" >/dev/null
+grep '^#define __MCPU_REAL_IO_LIMIT__ [0-9][0-9]*$' "$base/macros" >/dev/null
+grep '^#define __MCPU_MATH_FN_LIMIT__ [0-9][0-9]*$' "$base/macros" >/dev/null
+grep '^#define __MCPU_BYTE_ORDER__ __ORDER_\(LITTLE\|BIG\)_ENDIAN__$' "$base/macros" >/dev/null
+grep '^#define __MCPU_WORD_ORDER__ __ORDER_\(LITTLE\|BIG\)_ENDIAN__$' "$base/macros" >/dev/null
+grep '^#define __BYTE_ORDER__ __MCPU_BYTE_ORDER__$' "$base/macros" >/dev/null
+grep '^#define __MCPU_INT_MAX_WIDTH__ [1-9][0-9]*$' "$base/macros" >/dev/null
+grep '^#define __MCPU_REAL_MAX_WIDTH__ [1-9][0-9]*$' "$base/macros" >/dev/null
+grep '^#define __MCPU_COMPLEX_MAX_WIDTH__ [1-9][0-9]*$' "$base/macros" >/dev/null
+grep '^#define __REGISTER_PREFIX__$' "$base/macros" >/dev/null
+grep '^#define __USER_LABEL_PREFIX__$' "$base/macros" >/dev/null
+grep '^#define __MCPU_SIZE_TYPE__ uint[0-9][0-9]*$' "$base/macros" >/dev/null
+grep '^#define __MCPU_SIZE_WIDTH__ [1-9][0-9]*$' "$base/macros" >/dev/null
+grep '^#define __MCPU_SIZEOF_SIZE__ [1-9][0-9]*$' "$base/macros" >/dev/null
+grep '^#define __MCPU_SIZE_MAX__ 0x[0-9a-f][0-9a-f]*$' "$base/macros" >/dev/null
+grep '^#define __PTRDIFF_TYPE__ int64$' "$base/macros" >/dev/null
+grep '^#define __PTRDIFF_WIDTH__ 64$' "$base/macros" >/dev/null
+grep '^#define __PTRDIFF_MAX__ 0x7fffffffffffffff$' "$base/macros" >/dev/null
+grep '^#define __SIZEOF_PTRDIFF__ 8$' "$base/macros" >/dev/null
+grep '^#define __INTPTR_TYPE__ int64$' "$base/macros" >/dev/null
+grep '^#define __INTPTR_WIDTH__ 64$' "$base/macros" >/dev/null
+grep '^#define __INTPTR_MAX__ 0x7fffffffffffffff$' "$base/macros" >/dev/null
+grep '^#define __MCPU_SSIZE_TYPE__ int[0-9][0-9]*$' "$base/macros" >/dev/null
+grep '^#define __MCPU_SSIZE_WIDTH__ [1-9][0-9]*$' "$base/macros" >/dev/null
+grep '^#define __MCPU_SIZEOF_SSIZE__ [1-9][0-9]*$' "$base/macros" >/dev/null
+grep '^#define __MCPU_SSIZE_MAX__ 0x[0-9a-f][0-9a-f]*$' "$base/macros" >/dev/null
+grep '^#define __UINTPTR_TYPE__ uint64$' "$base/macros" >/dev/null
+grep '^#define __UINTPTR_WIDTH__ 64$' "$base/macros" >/dev/null
+grep '^#define __UINTPTR_MAX__ 0xffffffffffffffff$' "$base/macros" >/dev/null
+grep '^#define __CHAR8_TYPE__ char8$' "$base/macros" >/dev/null
+grep '^#define __CHAR16_TYPE__ char16$' "$base/macros" >/dev/null
+grep '^#define __SIZEOF_CHAR8__ 1$' "$base/macros" >/dev/null
+grep '^#define __SIZEOF_CHAR16__ 2$' "$base/macros" >/dev/null
+grep '^#define __INT128_DECIMAL_DIG__ 39$' "$base/macros" >/dev/null
+grep '^#define __UINT128_DECIMAL_DIG__ 39$' "$base/macros" >/dev/null
+grep '^#define __REAL128_DECIMAL_DIG__ [1-9][0-9]*$' "$base/macros" >/dev/null
+grep '^#define __REAL128_MANT_DIG__ [1-9][0-9]*$' "$base/macros" >/dev/null
+grep '^#define __SIZEOF_REAL128_EXP__ 4$' "$base/macros" >/dev/null
+grep '^#define __REAL128_MAX_STRLEN__ 60$' "$base/macros" >/dev/null
+grep '^#define __REAL128_MAX__ [0-9][0-9.]*e+[0-9][0-9]*$' "$base/macros" >/dev/null
+grep '^#define __INT1024_TYPE__ int1024$' "$base/macros" >/dev/null
+grep '^#define __INT1024_WIDTH__ 1024$' "$base/macros" >/dev/null
+grep '^#define __SIZEOF_INT1024__ 128$' "$base/macros" >/dev/null
+grep '^#define __INT1024_DECIMAL_DIG__ 308$' "$base/macros" >/dev/null
+grep '^#define __UINT1024_DECIMAL_DIG__ 309$' "$base/macros" >/dev/null
+grep '^#define __INT65536_TYPE__ int65536$' "$base/macros" >/dev/null
+grep '^#define __INT65536_WIDTH__ 65536$' "$base/macros" >/dev/null
+grep '^#define __SIZEOF_INT65536__ 8192$' "$base/macros" >/dev/null
+grep '^#define __REAL1024_TYPE__ real1024$' "$base/macros" >/dev/null
+grep '^#define __REAL1024_WIDTH__ 1024$' "$base/macros" >/dev/null
+grep '^#define __SIZEOF_REAL1024__ 128$' "$base/macros" >/dev/null
+grep '^#define __REAL1024_DECIMAL_DIG__ [1-9][0-9]*$' "$base/macros" >/dev/null
+grep '^#define __REAL1024_MANT_DIG__ [1-9][0-9]*$' "$base/macros" >/dev/null
+grep '^#define __COMPLEX128_TYPE__ complex128$' "$base/macros" >/dev/null
+grep '^#define __COMPLEX128_WIDTH__ 128$' "$base/macros" >/dev/null
+grep '^#define __SIZEOF_COMPLEX128__ 32$' "$base/macros" >/dev/null
+real_limit=`sed -n 's/^#define __MCPU_REAL_IO_LIMIT__ \([0-9][0-9]*\)$/\1/p' "$base/macros"`
+test -n "$real_limit"
+real_size=`expr "$real_limit" / 8`
+complex_size=`expr "$real_limit" / 4`
+grep "^#define __REAL${real_limit}_TYPE__ real${real_limit}$" "$base/macros" >/dev/null
+grep "^#define __REAL${real_limit}_WIDTH__ ${real_limit}$" "$base/macros" >/dev/null
+grep "^#define __SIZEOF_REAL${real_limit}__ ${real_size}$" "$base/macros" >/dev/null
+grep "^#define __SIZEOF_REAL${real_limit}_EXP__ [1-9][0-9]*$" "$base/macros" >/dev/null
+grep "^#define __REAL${real_limit}_MAX_STRLEN__ [1-9][0-9]*$" "$base/macros" >/dev/null
+grep "^#define __REAL${real_limit}_DECIMAL_DIG__ [1-9][0-9]*$" "$base/macros" >/dev/null
+grep "^#define __REAL${real_limit}_MANT_DIG__ [1-9][0-9]*$" "$base/macros" >/dev/null
+grep "^#define __COMPLEX${real_limit}_TYPE__ complex${real_limit}$" "$base/macros" >/dev/null
+grep "^#define __COMPLEX${real_limit}_WIDTH__ ${real_limit}$" "$base/macros" >/dev/null
+grep "^#define __SIZEOF_COMPLEX${real_limit}__ ${complex_size}$" "$base/macros" >/dev/null
+
+for name in \
+ __VERSION__ __MPU_MACHINE_REGISTER_WIDTH__ __MPU_REAL_IO_LIMIT__ \
+ __MPU_MATH_FN_LIMIT__ __INT128_MIN__ __INT1024_MAX__ \
+ __REAL128_DIG__ __REAL1024_MAX__ __FLOAT_WORD_ORDER__ \
+ __SIZE_TYPE__ __SIZE_WIDTH__ __SIZE_MAX__ __SIZEOF_SIZE_T__ \
+ __MCPU_SIZEOF_SSIZE_T__ __SIZEOF_CHAR8_T__ __SIZEOF_CHAR16_T__ \
+ __SIZEOF_PTRDIFF_T__; do
+ if grep "^#define $name\>" "$base/macros" >/dev/null; then exit 1; fi
+done
+
+if grep '^#define __FILE__' "$base/macros" >/dev/null; then exit 1; fi
+if grep '^#define __LINE__' "$base/macros" >/dev/null; then exit 1; fi
+if grep '^#define __DATE__' "$base/macros" >/dev/null; then exit 1; fi
+if grep '^#define __TIME__' "$base/macros" >/dev/null; then exit 1; fi
+if grep '^#define __BASE_FILE__' "$base/macros" >/dev/null; then exit 1; fi
+if grep '^#define __INCLUDE_LEVEL__' "$base/macros" >/dev/null; then exit 1; fi
diff --git a/tests/t0026-dump-config.sh b/tests/t0026-dump-config.sh
new file mode 100755
index 0000000..0778f28
--- /dev/null
+++ b/tests/t0026-dump-config.sh
@@ -0,0 +1,22 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-dconfig-$$"
+mkdir -p "$base"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+cat > "$base/test.conf" <<'EOT'
+ZETA = /zeta;
+ROOT = /opt/mcpu;
+ALPHA = $ROOT/include;
+MCPU_CPP_INCLUDE_PATH = $ROOT/include/mcpu;
+EOT
+
+"$MCPU_CPP" --config-file "$base/test.conf" -dconfig > "$base/config"
+
+test -s "$base/config"
+LC_ALL=C sort -c "$base/config"
+grep '^ALPHA = /opt/mcpu/include;$' "$base/config" >/dev/null
+grep '^MCPU_CPP_INCLUDE_PATH = /opt/mcpu/include/mcpu;$' "$base/config" >/dev/null
+grep '^ROOT = /opt/mcpu;$' "$base/config" >/dev/null
+grep '^ZETA = /zeta;$' "$base/config" >/dev/null
+grep '^MCPU_CPP_SYSTEM_INCLUDE_PATH = ' "$base/config" >/dev/null
diff --git a/tests/t0027-system-language-path.sh b/tests/t0027-system-language-path.sh
new file mode 100755
index 0000000..0947c07
--- /dev/null
+++ b/tests/t0027-system-language-path.sh
@@ -0,0 +1,44 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-system-lang-$$"
+mkdir -p "$base/system/diff" "$base/system/as"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+cat > "$base/test.conf" <<EOT
+MCPU_CPP_SYSTEM_INCLUDE_PATH = $base/system;
+EOT
+
+cat > "$base/system/common.h" <<'EOT'
+common_system_header
+EOT
+cat > "$base/system/diff/lang.h" <<'EOT'
+diff_system_header
+EOT
+cat > "$base/system/as/lang.h" <<'EOT'
+as_system_header
+EOT
+cat > "$base/system/diff/explicit.h" <<'EOT'
+explicit_diff_system_header
+EOT
+
+cat > "$base/main.c" <<'EOT'
+#include <common.h>
+#lang "diff"
+#include <lang.h>
+#endlang
+#lang "as"
+#include <lang.h>
+#include <diff/explicit.h>
+#endlang
+EOT
+
+"$MCPU_CPP" --config-file "$base/test.conf" "$base/main.c" -o "$base/out"
+
+grep '^common_system_header$' "$base/out" >/dev/null
+grep '^diff_system_header$' "$base/out" >/dev/null
+grep '^as_system_header$' "$base/out" >/dev/null
+grep '^explicit_diff_system_header$' "$base/out" >/dev/null
+
+if "$MCPU_CPP" --config-file "$base/test.conf" -nostdinc "$base/main.c" -o "$base/no" 2>/dev/null; then
+ exit 1
+fi
diff --git a/tests/t0028-predefined-ranges.sh b/tests/t0028-predefined-ranges.sh
new file mode 100755
index 0000000..72693a2
--- /dev/null
+++ b/tests/t0028-predefined-ranges.sh
@@ -0,0 +1,95 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-predef-ranges-$$"
+mkdir -p "$base"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+"$MCPU_CPP" --no-config -dMP < /dev/null > "$base/macros"
+
+# Integer families: derive the largest advertised width, then verify every
+# power-of-two family from int8 through that LibMPU limit.
+int_limit=`sed -n 's/^#define __INT\([0-9][0-9]*\)_TYPE__ int[0-9][0-9]*$/\1/p' "$base/macros" | sort -n | tail -1`
+test -n "$int_limit"
+grep "^#define __MCPU_INT_MAX_WIDTH__ ${int_limit}$" "$base/macros" >/dev/null
+
+bits=8
+while test "$bits" -le "$int_limit"; do
+ bytes=`expr "$bits" / 8`
+ grep "^#define __INT${bits}_TYPE__ int${bits}$" "$base/macros" >/dev/null
+ grep "^#define __UINT${bits}_TYPE__ uint${bits}$" "$base/macros" >/dev/null
+ grep "^#define __INT${bits}_WIDTH__ ${bits}$" "$base/macros" >/dev/null
+ grep "^#define __UINT${bits}_WIDTH__ ${bits}$" "$base/macros" >/dev/null
+ grep "^#define __SIZEOF_INT${bits}__ ${bytes}$" "$base/macros" >/dev/null
+ grep "^#define __SIZEOF_UINT${bits}__ ${bytes}$" "$base/macros" >/dev/null
+
+ grep "^#define __INT${bits}_DECIMAL_DIG__ [1-9][0-9]*$" "$base/macros" >/dev/null
+ grep "^#define __UINT${bits}_DECIMAL_DIG__ [1-9][0-9]*$" "$base/macros" >/dev/null
+
+ if test "$bits" -le 256; then
+ grep "^#define __INT${bits}_MAX__ 0x[0-9a-f][0-9a-f]*$" "$base/macros" >/dev/null
+ grep "^#define __UINT${bits}_MAX__ 0x[0-9a-f][0-9a-f]*$" "$base/macros" >/dev/null
+ else
+ for name in "__INT${bits}_MAX__" "__UINT${bits}_MAX__"; do
+ if grep "^#define ${name}\\>" "$base/macros" >/dev/null; then exit 1; fi
+ done
+ fi
+
+ if grep "^#define __INT${bits}_MIN__\\>" "$base/macros" >/dev/null; then exit 1; fi
+ bits=`expr "$bits" \* 2`
+done
+
+# Real/Complex families: their complete structural metadata extends through
+# MPU_REAL_IO_LIMIT. Complex WIDTH is the language parameter, not storage bits.
+real_limit=`sed -n 's/^#define __MCPU_REAL_IO_LIMIT__ \([0-9][0-9]*\)$/\1/p' "$base/macros"`
+test -n "$real_limit"
+grep "^#define __MCPU_REAL_MAX_WIDTH__ ${real_limit}$" "$base/macros" >/dev/null
+grep "^#define __MCPU_COMPLEX_MAX_WIDTH__ ${real_limit}$" "$base/macros" >/dev/null
+
+bits=32
+while test "$bits" -le "$real_limit"; do
+ real_bytes=`expr "$bits" / 8`
+ complex_bytes=`expr "$bits" / 4`
+ grep "^#define __REAL${bits}_TYPE__ real${bits}$" "$base/macros" >/dev/null
+ grep "^#define __REAL${bits}_WIDTH__ ${bits}$" "$base/macros" >/dev/null
+ grep "^#define __SIZEOF_REAL${bits}__ ${real_bytes}$" "$base/macros" >/dev/null
+ case "$bits" in
+ 32) exp_bytes=1; max_strlen=20 ;;
+ 64) exp_bytes=2; max_strlen=40 ;;
+ 128) exp_bytes=4; max_strlen=60 ;;
+ 256) exp_bytes=4; max_strlen=80 ;;
+ 512) exp_bytes=8; max_strlen=160 ;;
+ 1024) exp_bytes=8; max_strlen=320 ;;
+ 2048) exp_bytes=16; max_strlen=640 ;;
+ 4096) exp_bytes=16; max_strlen=1280 ;;
+ 8192) exp_bytes=32; max_strlen=2560 ;;
+ 16384) exp_bytes=32; max_strlen=5120 ;;
+ 32768) exp_bytes=64; max_strlen=10240 ;;
+ 65536) exp_bytes=64; max_strlen=20480 ;;
+ *) exit 1 ;;
+ esac
+ grep "^#define __SIZEOF_REAL${bits}_EXP__ ${exp_bytes}$" "$base/macros" >/dev/null
+ grep "^#define __REAL${bits}_MAX_STRLEN__ ${max_strlen}$" "$base/macros" >/dev/null
+ grep "^#define __COMPLEX${bits}_TYPE__ complex${bits}$" "$base/macros" >/dev/null
+ grep "^#define __COMPLEX${bits}_WIDTH__ ${bits}$" "$base/macros" >/dev/null
+ grep "^#define __SIZEOF_COMPLEX${bits}__ ${complex_bytes}$" "$base/macros" >/dev/null
+
+ grep "^#define __REAL${bits}_DECIMAL_DIG__ [1-9][0-9]*$" "$base/macros" >/dev/null
+ grep "^#define __REAL${bits}_MANT_DIG__ [1-9][0-9]*$" "$base/macros" >/dev/null
+
+ if test "$bits" -le 256; then
+ grep "^#define __REAL${bits}_MAX__ " "$base/macros" >/dev/null
+ grep "^#define __REAL${bits}_MIN__ " "$base/macros" >/dev/null
+ grep "^#define __REAL${bits}_EPSILON__ " "$base/macros" >/dev/null
+ grep "^#define __REAL${bits}_MAX_EXP__ " "$base/macros" >/dev/null
+ grep "^#define __REAL${bits}_MIN_EXP__ " "$base/macros" >/dev/null
+ grep "^#define __REAL${bits}_MAX_10_EXP__ " "$base/macros" >/dev/null
+ grep "^#define __REAL${bits}_MIN_10_EXP__ " "$base/macros" >/dev/null
+ else
+ for suffix in MAX MIN EPSILON MAX_EXP MIN_EXP MAX_10_EXP MIN_10_EXP; do
+ if grep "^#define __REAL${bits}_${suffix}__\\>" "$base/macros" >/dev/null; then exit 1; fi
+ done
+ fi
+
+ if grep "^#define __REAL${bits}_DIG__\\>" "$base/macros" >/dev/null; then exit 1; fi
+ bits=`expr "$bits" \* 2`
+done
diff --git a/tests/t0029-predefined-abi-names.sh b/tests/t0029-predefined-abi-names.sh
new file mode 100755
index 0000000..37b2f55
--- /dev/null
+++ b/tests/t0029-predefined-abi-names.sh
@@ -0,0 +1,38 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-predef-names-$$"
+mkdir -p "$base"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+"$MCPU_CPP" --no-config -dMP < /dev/null > "$base/macros"
+
+grep '^#define __MCPU_SIZE_TYPE__ uint[0-9][0-9]*$' "$base/macros" >/dev/null
+grep '^#define __MCPU_SIZE_WIDTH__ [1-9][0-9]*$' "$base/macros" >/dev/null
+grep '^#define __MCPU_SIZEOF_SIZE__ [1-9][0-9]*$' "$base/macros" >/dev/null
+grep '^#define __MCPU_SIZE_MAX__ 0x[0-9a-f][0-9a-f]*$' "$base/macros" >/dev/null
+grep '^#define __MCPU_SSIZE_TYPE__ int[0-9][0-9]*$' "$base/macros" >/dev/null
+grep '^#define __MCPU_SSIZE_WIDTH__ [1-9][0-9]*$' "$base/macros" >/dev/null
+grep '^#define __MCPU_SIZEOF_SSIZE__ [1-9][0-9]*$' "$base/macros" >/dev/null
+grep '^#define __MCPU_SSIZE_MAX__ 0x[0-9a-f][0-9a-f]*$' "$base/macros" >/dev/null
+grep '^#define __SIZEOF_PTRDIFF__ 8$' "$base/macros" >/dev/null
+grep '^#define __BYTE_ORDER__ __MCPU_BYTE_ORDER__$' "$base/macros" >/dev/null
+grep '^#define __MCPU_BYTE_ORDER__ __ORDER_\(LITTLE\|BIG\)_ENDIAN__$' "$base/macros" >/dev/null
+grep '^#define __MCPU_WORD_ORDER__ __ORDER_\(LITTLE\|BIG\)_ENDIAN__$' "$base/macros" >/dev/null
+grep '^#define __MCPU_INT_MAX_WIDTH__ [1-9][0-9]*$' "$base/macros" >/dev/null
+grep '^#define __MCPU_REAL_MAX_WIDTH__ [1-9][0-9]*$' "$base/macros" >/dev/null
+grep '^#define __MCPU_COMPLEX_MAX_WIDTH__ [1-9][0-9]*$' "$base/macros" >/dev/null
+
+grep '^#define __SIZEOF_CHAR8__ 1$' "$base/macros" >/dev/null
+grep '^#define __SIZEOF_CHAR16__ 2$' "$base/macros" >/dev/null
+
+real_limit=`sed -n 's/^#define __MCPU_REAL_IO_LIMIT__ \([0-9][0-9]*\)$/\1/p' "$base/macros"`
+test -n "$real_limit"
+grep "^#define __SIZEOF_REAL${real_limit}_EXP__ [1-9][0-9]*$" "$base/macros" >/dev/null
+grep "^#define __REAL${real_limit}_MAX_STRLEN__ [1-9][0-9]*$" "$base/macros" >/dev/null
+
+for name in \
+ __SIZE_TYPE__ __SIZE_WIDTH__ __SIZE_MAX__ __SIZEOF_SIZE_T__ \
+ __MCPU_SIZEOF_SSIZE_T__ __SIZEOF_CHAR8_T__ __SIZEOF_CHAR16_T__ \
+ __SIZEOF_PTRDIFF_T__ __FLOAT_WORD_ORDER__; do
+ if grep "^#define ${name}\\>" "$base/macros" >/dev/null; then exit 1; fi
+done
diff --git a/tests/t0030-integer-decimal-digits.sh b/tests/t0030-integer-decimal-digits.sh
new file mode 100755
index 0000000..08f7fe2
--- /dev/null
+++ b/tests/t0030-integer-decimal-digits.sh
@@ -0,0 +1,33 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-int-digs-$$"
+mkdir -p "$base"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+"$MCPU_CPP" --no-config -dMP < /dev/null > "$base/macros"
+
+check_digs()
+{
+ bits="$1"
+ int_digs="$2"
+ uint_digs="$3"
+
+ grep "^#define __INT${bits}_DECIMAL_DIG__ ${int_digs}$" "$base/macros" >/dev/null
+ grep "^#define __UINT${bits}_DECIMAL_DIG__ ${uint_digs}$" "$base/macros" >/dev/null
+}
+
+# Decimal digits of the numeric maxima only: no sign and no terminating NUL.
+check_digs 8 3 3
+check_digs 16 5 5
+check_digs 32 10 10
+check_digs 64 19 20
+check_digs 128 39 39
+check_digs 256 77 78
+check_digs 512 154 155
+check_digs 1024 308 309
+check_digs 2048 617 617
+check_digs 4096 1233 1234
+check_digs 8192 2466 2467
+check_digs 16384 4932 4933
+check_digs 32768 9864 9865
+check_digs 65536 19729 19729
diff --git a/tests/t0031-stringification.sh b/tests/t0031-stringification.sh
new file mode 100755
index 0000000..213d787
--- /dev/null
+++ b/tests/t0031-stringification.sh
@@ -0,0 +1,50 @@
+#!/bin/sh
+set -eu
+src="${TMPDIR-/tmp}/mcpu-cpp-stringify-$$.c"
+out="${TMPDIR-/tmp}/mcpu-cpp-stringify-$$.out"
+dump="${TMPDIR-/tmp}/mcpu-cpp-stringify-$$.dump"
+trap 'rm -f "$src" "$out" "$dump"' EXIT HUP INT TERM
+
+cat > "$src" <<'EOT'
+#define A 7
+#define STR(X) #X
+#define WSTR(X) # X
+#define XSTR(X) STR(X)
+#define MIX(X) #X | X
+#define BAD(X, Y) X + Y
+#define HASH(X) "#" X
+#define EMPTY(X) <X>
+STR(A)
+WSTR(alpha)
+XSTR(A)
+STR( a + b )
+STR(Привет мир)
+STR("a\\b")
+STR()
+STR(BAD(1))
+MIX(A)
+HASH(A)
+EMPTY()
+#lang "diff"
+#define DSTR(X) #X
+DSTR(y' + z)
+#endlang
+EOT
+
+"$MCPU_CPP" --no-config "$src" -o "$out"
+
+grep -Fx '"A"' "$out" >/dev/null
+grep -Fx '"alpha"' "$out" >/dev/null
+grep -Fx '"7"' "$out" >/dev/null
+grep -Fx '"a + b"' "$out" >/dev/null
+grep -Fx '"Привет мир"' "$out" >/dev/null
+grep -Fx '"\"a\\\\b\""' "$out" >/dev/null
+grep -Fx '""' "$out" >/dev/null
+grep -Fx '"BAD(1)"' "$out" >/dev/null
+grep -Fx '"A" | 7' "$out" >/dev/null
+grep -Fx '"#" 7' "$out" >/dev/null
+grep -Fx '<>' "$out" >/dev/null
+grep -Fx '"y'"'"' + z"' "$out" >/dev/null
+
+"$MCPU_CPP" --no-config -dM "$src" > "$dump"
+grep -Fx '#define STR(X) #X' "$dump" >/dev/null
diff --git a/tests/t0032-command-line.sh b/tests/t0032-command-line.sh
new file mode 100755
index 0000000..e0deb91
--- /dev/null
+++ b/tests/t0032-command-line.sh
@@ -0,0 +1,119 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-options-$$"
+mkdir -p "$base"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+check()
+{
+ name=$1
+ shift
+ "$MCPP_OPTIONS_TEST" "$@" > "$base/$name"
+}
+
+check pipe
+cat > "$base/pipe.expected" <<'EOT'
+input=<stdin>
+output=<stdout>
+verbose=0
+nostdinc=0
+dump_config=0
+dump_search_dirs=0
+dump_macros=0
+actions=0
+EOT
+cmp "$base/pipe.expected" "$base/pipe"
+
+check files input.S output.s
+sed -n '1,2p' "$base/files" > "$base/files.io"
+printf '%s\n' 'input=input.S' 'output=output.s' > "$base/files.expected"
+cmp "$base/files.expected" "$base/files.io"
+
+check dash - -
+sed -n '1,2p' "$base/dash" > "$base/dash.io"
+printf '%s\n' 'input=<stdin>' 'output=<stdout>' > "$base/dash.expected"
+cmp "$base/dash.expected" "$base/dash.io"
+
+check outfile input.S -o result.s
+sed -n '1,2p' "$base/outfile" > "$base/outfile.io"
+printf '%s\n' 'input=input.S' 'output=result.s' > "$base/outfile.expected"
+cmp "$base/outfile.expected" "$base/outfile.io"
+
+check options -v -nostdinc -Iinc1 -I inc2 -isystem sys -idirafter after input.S -
+grep '^verbose=1$' "$base/options" >/dev/null
+grep '^nostdinc=1$' "$base/options" >/dev/null
+grep '^actions=4$' "$base/options" >/dev/null
+grep '^action\[0\]=include-user:inc1$' "$base/options" >/dev/null
+grep '^action\[1\]=include-user:inc2$' "$base/options" >/dev/null
+grep '^action\[2\]=include-system:sys$' "$base/options" >/dev/null
+grep '^action\[3\]=include-after:after$' "$base/options" >/dev/null
+
+check ordered -DA=1 -UA -DA=2
+sed -n '/^actions=/,$p' "$base/ordered" > "$base/ordered.actions"
+cat > "$base/ordered.expected" <<'EOT'
+actions=3
+action[0]=define:A=1
+action[1]=undef:A
+action[2]=define:A=2
+EOT
+cmp "$base/ordered.expected" "$base/ordered.actions"
+
+check searchdirs -dsearch-dirs
+grep '^dump_search_dirs=1$' "$base/searchdirs" >/dev/null
+
+check dump -dMP input.S -o macros.txt
+grep '^dump_macros=1$' "$base/dump" >/dev/null
+
+check retained -include force.h -imacros macros.h -DA=1 -UOLD -M -MG -w -Wall -Werror input.S output.s
+grep '^input=input.S$' "$base/retained" >/dev/null
+grep '^output=output.s$' "$base/retained" >/dev/null
+grep '^action\[[0-9][0-9]*\]=include:force.h$' "$base/retained" >/dev/null
+grep '^action\[[0-9][0-9]*\]=imacros:macros.h$' "$base/retained" >/dev/null
+grep '^action\[[0-9][0-9]*\]=define:A=1$' "$base/retained" >/dev/null
+grep '^action\[[0-9][0-9]*\]=undef:OLD$' "$base/retained" >/dev/null
+
+check dumpD -dD input.S
+grep '^dump_macros=2$' "$base/dumpD" >/dev/null
+
+check md -MD input.S
+grep '^input=input.S$' "$base/md" >/dev/null
+check mmd -MMD input.S
+grep '^input=input.S$' "$base/mmd" >/dev/null
+check mf -MD -MF deps.mk input.S
+grep '^input=input.S$' "$base/mf" >/dev/null
+check mfattached -MMD -MFdeps.mk input.S
+grep '^input=input.S$' "$base/mfattached" >/dev/null
+check mt -M -MT target.o input.S
+grep '^action\[[0-9][0-9]*\]=dep-target:target.o$' "$base/mt" >/dev/null
+check mtattached -MM -MTtarget.o input.S
+grep '^action\[[0-9][0-9]*\]=dep-target:target.o$' "$base/mtattached" >/dev/null
+check mq -MD -MQ '$(OBJDIR)/target.o' input.S
+grep '^action\[[0-9][0-9]*\]=dep-target-quoted:$(OBJDIR)/target.o$' "$base/mq" >/dev/null
+check mqattached -MMD '-MQ$(OBJDIR)/target.o' input.S
+grep '^action\[[0-9][0-9]*\]=dep-target-quoted:$(OBJDIR)/target.o$' "$base/mqattached" >/dev/null
+if "$MCPP_OPTIONS_TEST" a b c > /dev/null 2>&1; then exit 1; fi
+if "$MCPP_OPTIONS_TEST" -o > /dev/null 2>&1; then exit 1; fi
+if "$MCPP_OPTIONS_TEST" -isystem > /dev/null 2>&1; then exit 1; fi
+if "$MCPP_OPTIONS_TEST" --bad-option > /dev/null 2>&1; then exit 1; fi
+if "$MCPP_OPTIONS_TEST" -MF deps.mk input.S > /dev/null 2>&1; then exit 1; fi
+if "$MCPP_OPTIONS_TEST" -MT target.o input.S > /dev/null 2>&1; then exit 1; fi
+if "$MCPP_OPTIONS_TEST" -MQ target.o input.S > /dev/null 2>&1; then exit 1; fi
+if "$MCPP_OPTIONS_TEST" -M -MT > /dev/null 2>&1; then exit 1; fi
+if "$MCPP_OPTIONS_TEST" -M -MQ > /dev/null 2>&1; then exit 1; fi
+
+# Unsupported legacy assertion options remain rejected.
+if "$MCPP_OPTIONS_TEST" -Aquestion input.S > /dev/null 2>&1; then exit 1; fi
+if "$MCPP_OPTIONS_TEST" -dMA input.S > /dev/null 2>&1; then exit 1; fi
+
+# Removed compatibility options are not recognized by mcpu-cpp.
+for opt in \
+ -x -X -out-unix-mode -I- -H -C -P '-$' -r -R -undef -u -dN -g3 -dark \
+ -Wimport -Wno-import -pedantic-ANSI-C -pedantic-ANSI-C-errors \
+ -lang-k -lang-k++ -lang-k-k++-comments -lang-asm -+ \
+ -iprefix -iwithprefix -iwithprefixbefore
+do
+ if "$MCPP_OPTIONS_TEST" "$opt" input.S > /dev/null 2>&1; then
+ echo "removed option unexpectedly accepted: $opt" >&2
+ exit 1
+ fi
+done
diff --git a/tests/t0033-dump-definitions.sh b/tests/t0033-dump-definitions.sh
new file mode 100755
index 0000000..2637e53
--- /dev/null
+++ b/tests/t0033-dump-definitions.sh
@@ -0,0 +1,51 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-dD-$$"
+mkdir -p "$base/inc"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+cat > "$base/inc/defs.h" <<'EOT'
+#ifndef MCPP_TEST_DEFS_H
+#define MCPP_TEST_DEFS_H 1
+#define FROM_HEADER 17 /* header definition */
+#endif /* MCPP_TEST_DEFS_H */
+EOT
+
+cat > "$base/main.c" <<'EOT'
+#include <defs.h>
+#define LOCAL_VALUE 25 // local definition
+LOCAL_VALUE + FROM_HEADER
+EOT
+
+"$MCPU_CPP" --no-config -I"$base/inc" -dD "$base/main.c" -o "$base/out"
+
+# -dD starts with an input marker, emits predefined definitions with
+# <built-in> markers, keeps source #define directives, and still emits the
+# ordinary preprocessed result.
+grep '^# 0 "'"$base/main.c"'"$' "$base/out" >/dev/null
+grep '^# 0 "<built-in>"$' "$base/out" >/dev/null
+grep '^#define __MCPU_CPP_VERSION__ "1.0.2"$' "$base/out" >/dev/null
+grep '^#define FROM_HEADER 17$' "$base/out" >/dev/null
+grep '^#define LOCAL_VALUE 25$' "$base/out" >/dev/null
+if grep '^[[:blank:]]*#[[:blank:]]*\(if\|ifdef\|ifndef\|elif\|else\|endif\)\>' "$base/out" >/dev/null; then
+ exit 1
+fi
+grep '^25 + 17$' "$base/out" >/dev/null
+if grep '[[:blank:]]$' "$base/out" >/dev/null; then
+ exit 1
+fi
+
+# Every predefined definition in the initial block has its own built-in marker.
+awk '
+ /^# 1 / { exit }
+ /^#define / {
+ if( previous != "# 0 \"<built-in>\"") exit 1
+ }
+ { previous = $0 }
+' "$base/out"
+
+# Without -dD the same source definitions are removed from normal output.
+"$MCPU_CPP" --no-config -I"$base/inc" "$base/main.c" -o "$base/normal"
+if grep '^#define LOCAL_VALUE 25$' "$base/normal" >/dev/null; then
+ exit 1
+fi
diff --git a/tests/t0034-conditionals.sh b/tests/t0034-conditionals.sh
new file mode 100755
index 0000000..85f3d66
--- /dev/null
+++ b/tests/t0034-conditionals.sh
@@ -0,0 +1,136 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-conditionals-$$"
+mkdir -p "$base/inc"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+cat > "$base/inc/guard.h" <<'EOT'
+#ifndef MCPP_TEST_GUARD_H
+#define MCPP_TEST_GUARD_H 1
+#define HEADER_VALUE 17
+header_line
+#endif
+EOT
+
+cat > "$base/main.c" <<'EOT'
+#define A 1
+#define B 2
+#define F(x) ((x) + 1)
+
+#include <guard.h>
+#include <guard.h>
+
+#if A + B * 3 == 7 && defined(A) && !defined(NO_SUCH_MACRO)
+if_line
+#else
+bad_if
+#endif
+
+#if 0
+#define BAD_SKIPPED 1
+bad_elif_0
+#elif F(B) == 3
+elif_line
+#else
+bad_elif_else
+#endif
+
+#ifdef A
+ifdef_line
+#else
+bad_ifdef
+#endif
+
+#ifndef NO_SUCH_MACRO
+ifndef_line
+#endif
+
+#if 0
+# if 1 / 0
+bad_nested
+# endif
+#else
+nested_skip_line
+#endif
+
+#if 0 && (1 / 0)
+bad_short_and
+#endif
+
+#if 1 || (1 / 0)
+short_or_line
+#endif
+
+#if 1 ? 2 : 1 / 0
+ternary_true_line
+#endif
+
+#if 0 ? 1 / 0 : 3
+ternary_false_line
+#endif
+
+#if UNKNOWN_IDENTIFIER
+bad_unknown
+#else
+unknown_zero_line
+#endif
+EOT
+
+"$MCPU_CPP" --no-config -I"$base/inc" -dD "$base/main.c" -o "$base/out"
+
+for word in \
+ header_line if_line elif_line ifdef_line ifndef_line nested_skip_line \
+ short_or_line ternary_true_line ternary_false_line unknown_zero_line; do
+ test "`grep -c "^${word}$" "$base/out"`" -eq 1
+done
+
+for word in \
+ bad_if bad_elif_0 bad_elif_else bad_ifdef bad_nested bad_short_and bad_unknown; do
+ if grep "^${word}$" "$base/out" >/dev/null; then exit 1; fi
+done
+
+# Conditional-control directives are consumed by the preprocessor and never
+# survive in the normal output, including -dD output.
+if grep '^[[:blank:]]*#[[:blank:]]*\(if\|ifdef\|ifndef\|elif\|else\|endif\)\>' "$base/out" >/dev/null; then
+ exit 1
+fi
+
+# Active definitions are retained by -dD, skipped ones are not.
+grep '^#define A 1$' "$base/out" >/dev/null
+grep '^#define HEADER_VALUE 17$' "$base/out" >/dev/null
+if grep '^#define BAD_' "$base/out" >/dev/null; then exit 1; fi
+
+cat > "$base/bad-endif.c" <<'EOT'
+#endif
+EOT
+if "$MCPU_CPP" --no-config "$base/bad-endif.c" -o "$base/bad-out" 2>/dev/null; then
+ exit 1
+fi
+
+cat > "$base/bad-else.c" <<'EOT'
+#if 1
+#else
+#else
+#endif
+EOT
+if "$MCPU_CPP" --no-config "$base/bad-else.c" -o "$base/bad-out" 2>/dev/null; then
+ exit 1
+fi
+
+cat > "$base/bad-elif.c" <<'EOT'
+#if 0
+#else
+#elif 1
+#endif
+EOT
+if "$MCPU_CPP" --no-config "$base/bad-elif.c" -o "$base/bad-out" 2>/dev/null; then
+ exit 1
+fi
+
+cat > "$base/unterminated.c" <<'EOT'
+#if 1
+unterminated
+EOT
+if "$MCPU_CPP" --no-config "$base/unterminated.c" -o "$base/bad-out" 2>/dev/null; then
+ exit 1
+fi
diff --git a/tests/t0035-line-control.sh b/tests/t0035-line-control.sh
new file mode 100755
index 0000000..b892015
--- /dev/null
+++ b/tests/t0035-line-control.sh
@@ -0,0 +1,41 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-line-control-$$"
+mkdir -p "$base/inc"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+cat > "$base/inc/one.h" <<'EOT'
+int header_line = __LINE__;
+EOT
+
+cat > "$base/main.c" <<'EOT'
+int before = __LINE__;
+#line 62 "main.y"
+int after = __LINE__;
+const char *file_after = __FILE__;
+#include "one.h"
+int returned = __LINE__;
+#define LINE_NUMBER 90
+#line LINE_NUMBER "macro-line.y"
+int macro_line = __LINE__;
+EOT
+
+"$MCPU_CPP" --no-config -I"$base/inc" "$base/main.c" -o "$base/out"
+
+# Generated line information uses GNU linemarker syntax, not input #line syntax.
+if grep '^#line[[:blank:]]' "$base/out" >/dev/null; then exit 1; fi
+
+grep "^# 1 \"$base/main.c\"$" "$base/out" >/dev/null
+grep '^# 62 "main.y"$' "$base/out" >/dev/null
+grep '^int after = 62;$' "$base/out" >/dev/null
+grep '^const char \*file_after = "main.y";$' "$base/out" >/dev/null
+
+# Flag 1 means entering an included file; flag 2 means returning to its includer.
+grep "^# 1 \"$base/inc/one.h\" 1$" "$base/out" >/dev/null
+grep '^# 65 "main.y" 2$' "$base/out" >/dev/null
+grep '^int returned = 65;$' "$base/out" >/dev/null
+
+# Arguments of #line are macro-expanded before interpretation.
+grep '^# 90 "macro-line.y"$' "$base/out" >/dev/null
+grep '^int macro_line = 90;$' "$base/out" >/dev/null
+
diff --git a/tests/t0036-lang-string.sh b/tests/t0036-lang-string.sh
new file mode 100755
index 0000000..8e40022
--- /dev/null
+++ b/tests/t0036-lang-string.sh
@@ -0,0 +1,120 @@
+#!/bin/sh
+set -eu
+dir="${TMPDIR-/tmp}/mcpu-cpp-lang-string-$$"
+mkdir -p "$dir"
+trap 'rm -rf "$dir"' EXIT HUP INT TERM
+
+cat > "$dir/normalize.c" <<'EOT'
+ # lang "DiFf"
+#endlang
+ #lang "dIfT"
+#endlang
+EOT
+"$MCPU_CPP" --no-config "$dir/normalize.c" -o "$dir/normalize.out"
+grep '^#lang "DiFf"$' "$dir/normalize.out" >/dev/null
+grep '^#lang "dIfT"$' "$dir/normalize.out" >/dev/null
+if grep '^[[:space:]][[:space:]]*#lang' "$dir/normalize.out" >/dev/null
+then
+ echo '#lang leading whitespace was not normalized' >&2
+ exit 1
+fi
+
+cat > "$dir/case.c" <<'EOT'
+#lang "diff"
+#endlang
+#lang "Diff"
+#endlang
+#lang "DIFF"
+#endlang
+#lang "dIfF"
+y' = 1;
+#endlang
+#lang "dIfT"
+#endlang
+#lang "ALG"
+#endlang
+#lang "As"
+#endlang
+#lang "aVm"
+#endlang
+#lang "aCs"
+#endlang
+EOT
+"$MCPU_CPP" --no-config "$dir/case.c" -o "$dir/case.out"
+for spelling in diff Diff DIFF dIfF dIfT ALG As aVm aCs
+do
+ grep "^#lang \"$spelling\"$" "$dir/case.out" >/dev/null
+done
+grep "^y' = 1;$" "$dir/case.out" >/dev/null
+
+check_bad()
+{
+ name=$1
+ pattern=$2
+ if "$MCPU_CPP" --no-config "$dir/$name.c" -o "$dir/$name.out" 2> "$dir/$name.err"
+ then
+ echo "$name unexpectedly accepted" >&2
+ exit 1
+ fi
+ grep -F "$pattern" "$dir/$name.err" >/dev/null
+}
+
+cat > "$dir/unquoted.c" <<'EOT'
+#lang diff
+EOT
+check_bad unquoted 'string constant expected after #lang'
+
+cat > "$dir/empty.c" <<'EOT'
+#lang ""
+EOT
+check_bad empty 'empty language name in #lang'
+
+cat > "$dir/leading-space.c" <<'EOT'
+#lang " diff"
+EOT
+check_bad leading-space 'one word without whitespace'
+
+cat > "$dir/trailing-space.c" <<'EOT'
+#lang "diff "
+EOT
+check_bad trailing-space 'one word without whitespace'
+
+cat > "$dir/internal-space.c" <<'EOT'
+#lang "di ff"
+EOT
+check_bad internal-space 'one word without whitespace'
+
+printf '#lang "di\tff"\n' > "$dir/internal-tab.c"
+check_bad internal-tab 'one word without whitespace'
+
+cat > "$dir/unknown.c" <<'EOT'
+#lang "cplusplus"
+EOT
+check_bad unknown "unknown language 'cplusplus'"
+
+cat > "$dir/zero.c" <<'EOT'
+#lang "0"
+EOT
+check_bad zero "unknown language '0'"
+
+cat > "$dir/extra.c" <<'EOT'
+#lang "diff" extra
+EOT
+check_bad extra 'extra text after #lang string constant'
+
+cat > "$dir/multiline.c" <<'EOT'
+#lang "di
+ff"
+EOT
+check_bad multiline 'unterminated #lang string constant'
+
+cat > "$dir/spliced.c" <<'EOT'
+#lang "di\
+ff"
+EOT
+check_bad spliced 'one physical source line'
+
+cat > "$dir/no-escape.c" <<'EOT'
+#lang "d\iff"
+EOT
+check_bad no-escape "unknown language 'd\\iff'"
diff --git a/tests/t0037-token-concatenation.sh b/tests/t0037-token-concatenation.sh
new file mode 100755
index 0000000..f2d2c3f
--- /dev/null
+++ b/tests/t0037-token-concatenation.sh
@@ -0,0 +1,122 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-concat-$$"
+mkdir -p "$base"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+cat > "$base/main.c" <<'EOT'
+#define A left
+#define B right
+#define AB raw_result
+#define leftright expanded_result
+#define CAT(X, Y) X ## Y
+#define XCAT(X, Y) CAT(X, Y)
+#define AFTERX(X) X_ ## X
+#define XAFTERX(X) AFTERX(X)
+#define TABLESIZE 1024
+#define BUFSIZE TABLESIZE
+#define MADE 91
+#define MK(A, B) A ## B
+#define L(X) X ## tail
+#define R(X) head ## X
+#define CHAIN(A, B, C) A ## B ## C
+#define OBJECT obj ## ect
+#define object object_result
+#define COMMAND(NAME) #NAME | NAME ## _command
+#define quit_command command_result
+#define CALL(NAME) NAME ## _fn(1)
+#define run_fn(X) called_result
+#define ONE1 1
+#define PCAT(A, B) A ## B
+#define RUN(A, B) A ## ## B
+#define ab adjacent_result
+#define TEXT "a ## b"
+#define COMMENT_CAT(A, B) A /* left */ ## /* right */ B
+#define DIRECT_EMPTY
+#define xDIRECT_EMPTY direct_empty_raw
+#define EXPAND_EMPTY(A, B) CAT(A, B)
+
+CAT(A, B)
+XCAT(A, B)
+AFTERX(BUFSIZE)
+XAFTERX(BUFSIZE)
+MK(MA, DE)
+CAT(x y, z)
+CAT(x, y z)
+L()
+R()
+CHAIN(x, , z)
+OBJECT
+CAT(1.5, e3)
+CAT(+, =)
+CAT(L, "wide")
+CAT(u8, "utf8")
+COMMAND(quit)
+CALL(run)
+RUN(a, b)
+CAT(x, DIRECT_EMPTY)
+EXPAND_EMPTY(x, DIRECT_EMPTY)
+TEXT
+COMMENT_CAT(c, d)
+CAT(e /* argument comment */, f)
+#if PCAT(ONE, 1)
+conditional_result
+#endif
+EOT
+
+"$MCPU_CPP" --no-config "$base/main.c" -o "$base/out"
+
+grep -Fx 'raw_result' "$base/out" >/dev/null
+grep -Fx 'expanded_result' "$base/out" >/dev/null
+grep -Fx 'X_BUFSIZE' "$base/out" >/dev/null
+grep -Fx 'X_1024' "$base/out" >/dev/null
+grep -Fx '91' "$base/out" >/dev/null
+grep -Fx 'x yz' "$base/out" >/dev/null
+grep -Fx 'xy z' "$base/out" >/dev/null
+grep -Fx 'tail' "$base/out" >/dev/null
+grep -Fx 'head' "$base/out" >/dev/null
+grep -Fx 'xz' "$base/out" >/dev/null
+grep -Fx 'object_result' "$base/out" >/dev/null
+grep -Fx '1.5e3' "$base/out" >/dev/null
+grep -Fx '+=' "$base/out" >/dev/null
+grep -Fx 'L"wide"' "$base/out" >/dev/null
+grep -Fx 'u8"utf8"' "$base/out" >/dev/null
+grep -Fx '"quit" | command_result' "$base/out" >/dev/null
+grep -Fx 'called_result' "$base/out" >/dev/null
+grep -Fx 'adjacent_result' "$base/out" >/dev/null
+grep -Fx 'direct_empty_raw' "$base/out" >/dev/null
+grep -Fx 'x' "$base/out" >/dev/null
+grep -Fx '"a ## b"' "$base/out" >/dev/null
+grep -Fx 'cd' "$base/out" >/dev/null
+grep -Fx 'ef' "$base/out" >/dev/null
+grep -Fx 'conditional_result' "$base/out" >/dev/null
+
+"$MCPU_CPP" --no-config -dM "$base/main.c" > "$base/dump"
+grep -Fx '#define CAT(X,Y) X ## Y' "$base/dump" >/dev/null
+grep -Fx '#define COMMAND(NAME) #NAME | NAME ## _command' "$base/dump" >/dev/null
+
+cat > "$base/invalid.c" <<'EOT'
+#define BAD(X) X ## +
+BAD(foo)
+EOT
+"$MCPU_CPP" --no-config "$base/invalid.c" -o "$base/invalid.out" 2>"$base/invalid.err"
+grep 'pasting "foo" and "+" does not give a valid preprocessing token' "$base/invalid.err" >/dev/null
+grep 'foo' "$base/invalid.out" | tr -d '[:space:]' | grep -Fx 'foo+' >/dev/null
+
+cat > "$base/first.c" <<'EOT'
+#define FIRST(X) ## X
+EOT
+if "$MCPU_CPP" --no-config "$base/first.c" -o "$base/first.out" 2>"$base/first.err"; then
+ echo "leading ## accepted" >&2
+ exit 1
+fi
+grep "'##' cannot appear at the beginning of a macro replacement list" "$base/first.err" >/dev/null
+
+cat > "$base/last.c" <<'EOT'
+#define LAST(X) X ##
+EOT
+if "$MCPU_CPP" --no-config "$base/last.c" -o "$base/last.out" 2>"$base/last.err"; then
+ echo "trailing ## accepted" >&2
+ exit 1
+fi
+grep "'##' cannot appear at the end of a macro replacement list" "$base/last.err" >/dev/null
diff --git a/tests/t0038-ucs2-identifiers.sh b/tests/t0038-ucs2-identifiers.sh
new file mode 100755
index 0000000..d1ae5db
--- /dev/null
+++ b/tests/t0038-ucs2-identifiers.sh
@@ -0,0 +1,122 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-ucs2-identifiers-$$"
+mkdir -p "$base"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+# The source file is UTF-8 externally. mcpu-cpp decodes it to strict UCS-2,
+# and identifier classification is then performed with the locale-independent
+# Unicode XID_Start/XID_Continue predicates provided by LibMPUIO.
+cat > "$base/main.c" <<'EOT'
+#define АНДРЕЙ 11
+#define résumé 12
+#define Élève 13
+#define ΩΜΕΓΑ 14
+#define переменная2 15
+#define a١ 16
+#define é 17
+#define СУММА(лево, право) лево + право
+#define СТРОКА(имя) #имя
+#define СКЛЕИТЬ(лево, право) лево ## право
+#define АНДРЕЙ2 18
+#define ВРЕМЕННЫЙ 19
+#undef ВРЕМЕННЫЙ
+#define VALUE$OLD 20
+#define ДОЛЛАР$ИМЯ 21
+#define DOLLAR_PARAM(x$old) x$old
+#define CAT_DOLLAR(a,b) a ## b
+#define LEFT$RIGHT 22
+
+АНДРЕЙ
+résumé
+Élève
+ΩΜΕΓΑ
+переменная2
+a١
+é
+СУММА(АНДРЕЙ, résumé)
+СТРОКА(АНДРЕЙ)
+СКЛЕИТЬ(АНДР, ЕЙ)
+СКЛЕИТЬ(АНДРЕЙ, 2)
+андрей
+ВРЕМЕННЫЙ
+VALUE$OLD
+ДОЛЛАР$ИМЯ
+DOLLAR_PARAM(23)
+CAT_DOLLAR(LEFT$, RIGHT)
+
+#ifdef АНДРЕЙ
+ifdef_ok
+#endif
+
+#ifndef ВРЕМЕННЫЙ
+ifndef_ok
+#endif
+
+#if defined(résumé) && defined(ΩΜΕΓΑ)
+defined_ok
+#endif
+
+#ifdef VALUE$OLD
+dollar_ifdef_ok
+#endif
+
+#if defined(ДОЛЛАР$ИМЯ)
+dollar_defined_ok
+#endif
+EOT
+
+"$MCPU_CPP" --no-config "$base/main.c" -o "$base/out"
+
+grep -Fx '11' "$base/out" >/dev/null
+grep -Fx '12' "$base/out" >/dev/null
+grep -Fx '13' "$base/out" >/dev/null
+grep -Fx '14' "$base/out" >/dev/null
+grep -Fx '15' "$base/out" >/dev/null
+grep -Fx '16' "$base/out" >/dev/null
+grep -Fx '17' "$base/out" >/dev/null
+grep -Fx '11 + 12' "$base/out" >/dev/null
+grep -Fx '"АНДРЕЙ"' "$base/out" >/dev/null
+# Raw operands of ## are concatenated, then the result is rescanned.
+grep -Fx '11' "$base/out" >/dev/null
+grep -Fx '18' "$base/out" >/dev/null
+grep -Fx 'андрей' "$base/out" >/dev/null
+grep -Fx 'ВРЕМЕННЫЙ' "$base/out" >/dev/null
+grep -Fx 'ifdef_ok' "$base/out" >/dev/null
+grep -Fx 'ifndef_ok' "$base/out" >/dev/null
+grep -Fx 'defined_ok' "$base/out" >/dev/null
+grep -Fx '20' "$base/out" >/dev/null
+grep -Fx '21' "$base/out" >/dev/null
+grep -Fx '23' "$base/out" >/dev/null
+grep -Fx '22' "$base/out" >/dev/null
+grep -Fx 'dollar_ifdef_ok' "$base/out" >/dev/null
+grep -Fx 'dollar_defined_ok' "$base/out" >/dev/null
+
+# XID_Continue admits combining marks after an identifier start. The same
+# combining mark is not XID_Start and therefore cannot begin a macro name.
+printf '#define \314\201bad 1\n' > "$base/bad-start.c"
+if "$MCPU_CPP" --no-config "$base/bad-start.c" -o "$base/bad-start.out" 2>"$base/bad-start.err"; then
+ echo "combining mark accepted as identifier start" >&2
+ exit 1
+fi
+
+grep 'macro name expected after #define' "$base/bad-start.err" >/dev/null
+
+# A decimal digit is XID_Continue but not XID_Start. Non-ASCII digits are
+# therefore legal after the first character but cannot start an identifier.
+printf '#define \331\241bad 1\n' > "$base/bad-digit-start.c"
+if "$MCPU_CPP" --no-config "$base/bad-digit-start.c" -o "$base/bad-digit-start.out" 2>"$base/bad-digit-start.err"; then
+ echo "non-ASCII digit accepted as identifier start" >&2
+ exit 1
+fi
+
+grep 'macro name expected after #define' "$base/bad-digit-start.err" >/dev/null
+
+
+# Dollar is an identifier-continuation extension, never an identifier start.
+printf '#define $BAD 1\n' > "$base/bad-dollar-start.c"
+if "$MCPU_CPP" --no-config "$base/bad-dollar-start.c" -o "$base/bad-dollar-start.out" 2>"$base/bad-dollar-start.err"; then
+ echo "dollar accepted as identifier start" >&2
+ exit 1
+fi
+grep 'macro name expected after #define' "$base/bad-dollar-start.err" >/dev/null
diff --git a/tests/t0039-command-line-macros.sh b/tests/t0039-command-line-macros.sh
new file mode 100755
index 0000000..94a5ec3
--- /dev/null
+++ b/tests/t0039-command-line-macros.sh
@@ -0,0 +1,174 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-command-line-macros-$$"
+mkdir -p "$base"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+cat > "$base/main.c" <<'EOT'
+OBJECT
+DEFAULT
+EMPTY marker
+СУММА(10, 7)
+АНДРЕЙ
+résumé
+ПОРЯДОК
+СТРОКА(АНДРЕЙ)
+СКЛЕИТЬ(АНД,РЕЙ)
+#ifdef __MCPU_CPP_VERSION__
+BUILTIN_PRESENT
+#else
+BUILTIN_REMOVED
+#endif
+#if defined(КОМАНДА) && КОМАНДА == 31
+CONDITIONAL_OK
+#else
+CONDITIONAL_BAD
+#endif
+EOT
+
+"$MCPU_CPP" --no-config \
+ -D OBJECT=17 \
+ -DDEFAULT \
+ -DEMPTY= \
+ '-DСУММА(лево,право)=лево + право' \
+ -DАНДРЕЙ=23 \
+ -Drésumé=29 \
+ -DПОРЯДОК=1 -U ПОРЯДОК -DПОРЯДОК=2 \
+ '-DСТРОКА(x)=#x' \
+ '-DСКЛЕИТЬ(a,b)=a ## b' \
+ -DАНДРЕЙ=23 \
+ -U__MCPU_CPP_VERSION__ \
+ -DКОМАНДА=31 \
+ "$base/main.c" -o "$base/out"
+
+grep '^17$' "$base/out" >/dev/null
+grep '^1$' "$base/out" >/dev/null
+grep '^ marker$' "$base/out" >/dev/null
+grep '^10 + 7$' "$base/out" >/dev/null
+test "`grep -c '^23$' "$base/out"`" -eq 2
+grep '^29$' "$base/out" >/dev/null
+grep '^2$' "$base/out" >/dev/null
+grep '^"АНДРЕЙ"$' "$base/out" >/dev/null
+grep '^BUILTIN_REMOVED$' "$base/out" >/dev/null
+if grep '^BUILTIN_PRESENT$' "$base/out" >/dev/null; then exit 1; fi
+grep '^CONDITIONAL_OK$' "$base/out" >/dev/null
+if grep '^CONDITIONAL_BAD$' "$base/out" >/dev/null; then exit 1; fi
+
+# A function-like -D without '=' has the GNU meaning "replacement is 1".
+printf '%s\n' 'F(99)' > "$base/default-func.c"
+"$MCPU_CPP" --no-config '-DF(x)' "$base/default-func.c" -o "$base/default-func.out"
+grep '^1$' "$base/default-func.out" >/dev/null
+
+# Unicode XID continuation works on the command line as it does in source.
+combining=$(printf 'e\314\201')
+printf '%s\n' "$combining" > "$base/combining.c"
+"$MCPU_CPP" --no-config "-D$combining=37" "$base/combining.c" -o "$base/combining.out"
+grep '^37$' "$base/combining.out" >/dev/null
+
+arabic_three=$(printf '\331\243')
+name="ИМЯ$arabic_three"
+printf '%s\n' "$name" > "$base/digit.c"
+"$MCPU_CPP" --no-config "-D$name=43" "$base/digit.c" -o "$base/digit.out"
+grep '^43$' "$base/digit.out" >/dev/null
+
+# Command-line definitions pass through the same phase-3 comment cleanup as
+# source #define directives.
+printf '%s\n' 'COMMENTED' > "$base/comment.c"
+"$MCPU_CPP" --no-config '-DCOMMENTED=left/* comment */right' \
+ "$base/comment.c" -o "$base/comment.out"
+grep '^left right$' "$base/comment.out" >/dev/null
+
+# -dM observes the final command-line macro state.
+printf '%s\n' '' > "$base/empty.c"
+"$MCPU_CPP" --no-config -dM -DПОРЯДОК=1 -UПОРЯДОК -DПОРЯДОК=2 \
+ "$base/empty.c" -o "$base/macros"
+grep '^#define ПОРЯДОК 2$' "$base/macros" >/dev/null
+
+# '$' is accepted after the first identifier character. Single quotes protect
+# it from shell parameter expansion; without quotes a backslash can do the same.
+"$MCPU_CPP" --no-config -dD '-DАНДРЕЙ$_Y=62' -DОЛЬГА\$OLD=37 \
+ "$base/empty.c" -o "$base/dollar-command-line"
+grep '^#define АНДРЕЙ$_Y 62$' "$base/dollar-command-line" >/dev/null
+grep '^#define ОЛЬГА$OLD 37$' "$base/dollar-command-line" >/dev/null
+
+# '$' is not an identifier-start character.
+if "$MCPU_CPP" --no-config '-D$BAD=1' "$base/empty.c" -o "$base/bad" \
+ >"$base/dollar-start.stdout" 2>"$base/dollar-start.stderr"; then
+ exit 1
+fi
+grep 'macro name expected after #define' "$base/dollar-start.stderr" >/dev/null
+
+# For -D/-U an invalid tail after a valid identifier is discarded rather than
+# becoming part of the replacement list. The value after '=' is preserved.
+"$MCPU_CPP" --no-config -dD '-DАНДРЕЙ@XYZ=62' '-DABC+TAIL=17' \
+ '-UREMOVE@TAIL' -DREMOVE=99 "$base/empty.c" -o "$base/invalid-tail"
+grep '^#define АНДРЕЙ 62$' "$base/invalid-tail" >/dev/null
+grep '^#define ABC 17$' "$base/invalid-tail" >/dev/null
+grep '^#define REMOVE 99$' "$base/invalid-tail" >/dev/null
+if grep '^#define АНДРЕЙ @XYZ 62$' "$base/invalid-tail" >/dev/null; then
+ exit 1
+fi
+
+# Explicit '=' in an -U tail is just another invalid suffix and is discarded.
+"$MCPU_CPP" --no-config -DNAME=1 '-UNAME=bad' -dM "$base/empty.c" -o "$base/undef-tail"
+if grep '^#define NAME ' "$base/undef-tail" >/dev/null; then
+ exit 1
+fi
+
+# In -dD output command-line definitions have their own GNU-style origin
+# marker and must not be reported as predefined/builtin macros.
+"$MCPU_CPP" --no-config -dD -DАНДРЕЙ_К=62 -DОЛЬГА=37 \
+ "$base/empty.c" -o "$base/dD-command-line"
+awk '
+ /^#define АНДРЕЙ_К 62$/ {
+ if( previous != "# 0 \"<command-line>\"") exit 1
+ andrey = 1
+ }
+ /^#define ОЛЬГА 37$/ {
+ if( previous != "# 0 \"<command-line>\"") exit 1
+ olga = 1
+ }
+ /^#define __MCPU_CPP_VERSION__ / {
+ if( previous != "# 0 \"<built-in>\"") exit 1
+ builtin = 1
+ }
+ { previous = $0 }
+ END { if( !andrey || !olga || !builtin ) exit 1 }
+' "$base/dD-command-line"
+
+# An unterminated block comment in a command-line definition is diagnosed.
+if "$MCPU_CPP" --no-config '-DBROKEN=left/*' "$base/empty.c" -o "$base/bad" \
+ >"$base/comment-bad.stdout" 2>"$base/comment-bad.stderr"; then
+ exit 1
+fi
+grep -- '-D: unterminated comment' "$base/comment-bad.stderr" >/dev/null
+
+# Malformed macro names are rejected by the ordinary #define/#undef parsers.
+if "$MCPU_CPP" --no-config -D1ABC=2 "$base/empty.c" -o "$base/bad" \
+ >"$base/bad.stdout" 2>"$base/bad.stderr"; then
+ exit 1
+fi
+grep 'macro name expected after #define' "$base/bad.stderr" >/dev/null
+
+
+# Only -D/-U payloads are decoded. Invalid UTF-8 and valid supplementary-plane
+# UTF-8 are rejected because the internal text model is strict UCS-2.
+bad_utf8=$(printf '\377')
+if "$MCPU_CPP" --no-config "-D${bad_utf8}=1" "$base/empty.c" -o "$base/bad" \
+ >"$base/utf8.stdout" 2>"$base/utf8.stderr"; then
+ exit 1
+fi
+grep -- '-D: invalid UTF-8 or non-UCS-2 character' "$base/utf8.stderr" >/dev/null
+
+non_ucs2=$(printf '\360\237\230\200')
+if "$MCPU_CPP" --no-config "-D${non_ucs2}=1" "$base/empty.c" -o "$base/bad" \
+ >"$base/nonucs2.stdout" 2>"$base/nonucs2.stderr"; then
+ exit 1
+fi
+grep -- '-D: invalid UTF-8 or non-UCS-2 character' "$base/nonucs2.stderr" >/dev/null
+
+if "$MCPU_CPP" --no-config "-U${bad_utf8}" "$base/empty.c" -o "$base/bad" \
+ >"$base/utf8u.stdout" 2>"$base/utf8u.stderr"; then
+ exit 1
+fi
+grep -- '-U: invalid UTF-8 or non-UCS-2 character' "$base/utf8u.stderr" >/dev/null
diff --git a/tests/t0040-zubr-expression.sh b/tests/t0040-zubr-expression.sh
new file mode 100755
index 0000000..d95c629
--- /dev/null
+++ b/tests/t0040-zubr-expression.sh
@@ -0,0 +1,79 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-zubr-expression-$$"
+mkdir -p "$base"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+cat > "$base/main.c" <<'EOT'
+#if 0b1010 == 10 && 077 == 63 && 0x2a == 42
+radix_ok
+#else
+bad_radix
+#endif
+
+#if 9223372036854775807 == 0x7fffffffffffffff
+int64_ok
+#else
+bad_int64
+#endif
+
+#if 0xffffffffffffffffU > 9223372036854775807
+uint64_ok
+#else
+bad_uint64
+#endif
+
+#if 'A' == 65 && 'Я' == 0x42f && '\x42f' == 0x42f && '\101' == 65
+ucs2_char_ok
+#else
+bad_ucs2_char
+#endif
+
+#if -1 < 0 && (8 << -1) == 4 && (4 >> -1) == 8
+signed_shift_ok
+#else
+bad_signed_shift
+#endif
+
+#if 0 && (1 / 0)
+bad_short_and
+#endif
+
+#if 1 || (1 / 0)
+short_or_ok
+#endif
+
+#if 1 ? 7 : 1 / 0
+ternary_ok
+#endif
+EOT
+
+"$MCPU_CPP" --no-config "$base/main.c" -o "$base/out"
+
+for word in \
+ radix_ok int64_ok uint64_ok ucs2_char_ok signed_shift_ok short_or_ok ternary_ok; do
+ test "`grep -c "^${word}$" "$base/out"`" -eq 1
+done
+
+for word in \
+ bad_radix bad_int64 bad_uint64 bad_ucs2_char bad_signed_shift bad_short_and; do
+ if grep "^${word}$" "$base/out" >/dev/null; then exit 1; fi
+done
+
+cat > "$base/bad-float.c" <<'EOT'
+#if 1.0
+bad
+#endif
+EOT
+if "$MCPU_CPP" --no-config "$base/bad-float.c" -o "$base/bad-out" 2>/dev/null; then
+ exit 1
+fi
+
+cat > "$base/bad-inc.c" <<'EOT'
+#if 1++2
+bad
+#endif
+EOT
+if "$MCPU_CPP" --no-config "$base/bad-inc.c" -o "$base/bad-out" 2>/dev/null; then
+ exit 1
+fi
diff --git a/tests/t0041-integer-width-suffix.sh b/tests/t0041-integer-width-suffix.sh
new file mode 100755
index 0000000..a59483f
--- /dev/null
+++ b/tests/t0041-integer-width-suffix.sh
@@ -0,0 +1,95 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-integer-width-suffix-$$"
+mkdir -p "$base"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+cat > "$base/main.c" <<'EOT'
+#if 0x7fz8 == 127
+z8_positive
+#endif
+#if 0x80z8 == -128
+z8_sign_extend
+#endif
+#if 0xffz8 == -1
+z8_minus_one
+#endif
+#if 0xffz8u == 255
+z8_zero_extend
+#endif
+#if 0x1ffz8 == -1 && 0x1ffz8U == 255
+z8_truncate
+#endif
+#if 0x8000z16 == -32768 && 0xffffz16U == 65535
+z16_ok
+#endif
+#if 0x80000000z32 < 0 && 0xffffffffz32u == 4294967295U
+z32_ok
+#endif
+#if 0xffffffffffffffffz64 == -1
+z64_signed
+#endif
+#if 0xffffffffffffffffZ064U > 0
+z64_unsigned_leading_zero_width
+#endif
+#if 0x7fz8 + 1 == 128
+operations_are_64_bit
+#endif
+#if (~0xffz8u) == 0xffffffffffffff00U
+compl_is_64_bit
+#endif
+#if (0x80z8 >> 1) == -64 && (0x80z8u >> 1) == 64
+shift_uses_promoted_value
+#endif
+#if 1z24 == 1
+invalid_width_ignored
+#endif
+#if 1z024U == 1U
+invalid_width_u_preserved
+#endif
+#if 1U == 1 && 1u == 1
+plain_u_ok
+#endif
+EOT
+
+"$MCPU_CPP" --no-config "$base/main.c" -o "$base/out" 2>"$base/err"
+
+for word in \
+ z8_positive z8_sign_extend z8_minus_one z8_zero_extend z8_truncate \
+ z16_ok z32_ok z64_signed z64_unsigned_leading_zero_width \
+ operations_are_64_bit compl_is_64_bit shift_uses_promoted_value \
+ invalid_width_ignored invalid_width_u_preserved plain_u_ok; do
+ test "`grep -c "^${word}$" "$base/out"`" -eq 1
+done
+
+test "`grep -c 'warning: invalid zNNN integer-width suffix; suffix ignored' "$base/err"`" -eq 2
+
+for bad in \
+ '1z128' \
+ '1z256u' \
+ '1z32undefined' \
+ '1z32ufoo' \
+ '1z32$foo' \
+ '1L' \
+ '1LL' \
+ '18446744073709551616'; do
+ cat > "$base/bad.c" <<EOT
+#if $bad
+bad
+#endif
+EOT
+ if "$MCPU_CPP" --no-config "$base/bad.c" -o "$base/bad-out" 2>"$base/bad-err"; then
+ echo "accepted invalid integer constant: $bad" >&2
+ exit 1
+ fi
+done
+
+cat > "$base/wide.c" <<'EOT'
+#if 1z128
+bad
+#endif
+EOT
+if "$MCPU_CPP" --no-config "$base/wide.c" -o "$base/bad-out" 2>"$base/wide.err"; then
+ exit 1
+fi
+grep 'error: integer constants wider than 64 bits are not allowed in conditional directives' "$base/wide.err" >/dev/null
diff --git a/tests/t0042-diagnostics.sh b/tests/t0042-diagnostics.sh
new file mode 100755
index 0000000..9510332
--- /dev/null
+++ b/tests/t0042-diagnostics.sh
@@ -0,0 +1,68 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-diagnostics-$$"
+mkdir -p "$base"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+cat > "$base/warning.c" <<'EOT'
+#define MESSAGE expanded_message
+before
+#warning MESSAGE is not expanded
+#warning one /* removed comment */ two
+#warning "spaces inside string"
+#warning Привет мир
+#if 0
+#warning hidden warning
+#error hidden error
+#endif
+after
+EOT
+
+"$MCPU_CPP" --no-config "$base/warning.c" -o "$base/warning.out" 2>"$base/warning.err"
+
+grep '^before$' "$base/warning.out" >/dev/null
+grep '^after$' "$base/warning.out" >/dev/null
+! grep '#warning' "$base/warning.out" >/dev/null
+
+grep 'warning: #warning MESSAGE is not expanded$' "$base/warning.err" >/dev/null
+grep 'warning: #warning one two$' "$base/warning.err" >/dev/null
+grep 'warning: #warning "spaces inside string"$' "$base/warning.err" >/dev/null
+grep 'warning: #warning Привет мир$' "$base/warning.err" >/dev/null
+! grep 'expanded_message' "$base/warning.err" >/dev/null
+! grep 'hidden warning' "$base/warning.err" >/dev/null
+! grep 'hidden error' "$base/warning.err" >/dev/null
+
+"$MCPU_CPP" --no-config -dD "$base/warning.c" -o "$base/warning-dd.out" 2>"$base/warning-dd.err"
+grep '^#define MESSAGE expanded_message$' "$base/warning-dd.out" >/dev/null
+! grep '#warning' "$base/warning-dd.out" >/dev/null
+grep 'warning: #warning MESSAGE is not expanded$' "$base/warning-dd.err" >/dev/null
+
+cat > "$base/line.c" <<'EOT'
+#line 62 "generated.mc"
+#warning logical location
+ok
+EOT
+"$MCPU_CPP" --no-config "$base/line.c" -o "$base/line.out" 2>"$base/line.err"
+grep '^generated\.mc:62: warning: #warning logical location$' "$base/line.err" >/dev/null
+grep '^ok$' "$base/line.out" >/dev/null
+
+cat > "$base/error.c" <<'EOT'
+#define WHY replacement
+#error WHY failed
+never
+EOT
+if "$MCPU_CPP" --no-config "$base/error.c" -o "$base/error.out" 2>"$base/error.err"; then
+ echo '#error did not stop preprocessing' >&2
+ exit 1
+fi
+grep 'error: #error WHY failed$' "$base/error.err" >/dev/null
+! grep 'replacement' "$base/error.err" >/dev/null
+
+
+cat > "$base/empty.c" <<'EOT'
+#warning
+ok
+EOT
+"$MCPU_CPP" --no-config "$base/empty.c" -o "$base/empty.out" 2>"$base/empty.err"
+grep 'warning: #warning$' "$base/empty.err" >/dev/null
+grep '^ok$' "$base/empty.out" >/dev/null
diff --git a/tests/t0043-include-next.sh b/tests/t0043-include-next.sh
new file mode 100755
index 0000000..a2d86c3
--- /dev/null
+++ b/tests/t0043-include-next.sh
@@ -0,0 +1,87 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-include-next-$$"
+mkdir -p "$base/wrapper" "$base/system" "$base/after" "$base/local"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+cat > "$base/config" <<EOT
+MCPU_CPP_SYSTEM_INCLUDE_PATH = $base/system;
+EOT
+
+cat > "$base/main.c" <<'EOT'
+#include <wrapped.h>
+#include "local/local.h"
+#if 0
+#include_next <never.h>
+#endif
+EOT
+
+cat > "$base/wrapper/wrapped.h" <<'EOT'
+wrapper_begin
+#line 500 "virtual/wrapped.h"
+#define NEXT_WRAPPED <wrapped.h>
+#include_next NEXT_WRAPPED
+wrapper_end
+EOT
+
+cat > "$base/system/wrapped.h" <<'EOT'
+system_begin
+#include_next "wrapped.h"
+system_end
+EOT
+
+cat > "$base/after/wrapped.h" <<'EOT'
+after_header
+EOT
+
+cat > "$base/local/local.h" <<'EOT'
+local_header
+#include_next <from_chain.h>
+EOT
+
+cat > "$base/wrapper/from_chain.h" <<'EOT'
+from_chain_wrapper
+EOT
+
+"$MCPU_CPP" --config-file "$base/config" \
+ -isystem "$base/wrapper" -idirafter "$base/after" \
+ "$base/main.c" -o "$base/out"
+
+# The wrapper header found through -isystem continues into the configured
+# system include root, then that header continues into -idirafter. The quote
+# form of #include_next must not re-search the current source directory.
+awk '
+ /wrapper_begin/ { a = NR }
+ /system_begin/ { b = NR }
+ /after_header/ { c = NR }
+ /system_end/ { d = NR }
+ /wrapper_end/ { e = NR }
+ END { exit !(a && a < b && b < c && c < d && d < e) }
+' "$base/out"
+
+# A header found by ordinary quoted-source-directory lookup has no configured
+# search entry to skip, so include_next begins with the configured chain.
+grep '^local_header$' "$base/out" >/dev/null
+grep '^from_chain_wrapper$' "$base/out" >/dev/null
+
+# Macro-expanded include_next operands use the same include operand parser.
+grep '^system_begin$' "$base/out" >/dev/null
+
+# Inactive conditional branches must not execute include_next.
+! grep 'never.h' "$base/out" >/dev/null
+
+# If no later search entry contains the file, include_next is an error rather
+# than falling back to the directory that supplied the current header.
+cat > "$base/wrapper/last.h" <<'EOT'
+#include_next <last.h>
+EOT
+cat > "$base/last-main.c" <<'EOT'
+#include <last.h>
+EOT
+if "$MCPU_CPP" --config-file "$base/config" \
+ -isystem "$base/wrapper" "$base/last-main.c" \
+ -o "$base/last.out" 2>"$base/last.err"; then
+ echo '#include_next unexpectedly found the current header again' >&2
+ exit 1
+fi
+grep "cannot find include_next file 'last.h'" "$base/last.err" >/dev/null
diff --git a/tests/t0044-configured-search-order.sh b/tests/t0044-configured-search-order.sh
new file mode 100755
index 0000000..980790f
--- /dev/null
+++ b/tests/t0044-configured-search-order.sh
@@ -0,0 +1,114 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-config-order-$$"
+mkdir -p \
+ "$base/cmd-I" \
+ "$base/cmd-isystem" \
+ "$base/config-as" \
+ "$base/config-user" \
+ "$base/system/as" \
+ "$base/cmd-after" \
+ "$base/config-after"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+cat > "$base/config" <<EOT
+MCPU_CPP_AS_INCLUDE_PATH = $base/config-as;
+MCPU_CPP_INCLUDE_PATH = $base/config-user;
+MCPU_CPP_SYSTEM_INCLUDE_PATH = $base/system;
+MCPU_CPP_AFTER_INCLUDE_PATH = $base/config-after;
+EOT
+
+cat > "$base/main.c" <<'EOT'
+#lang "as"
+#include <same.h>
+#endlang
+EOT
+
+cat > "$base/cmd-I/same.h" <<'EOT'
+from_explicit_I
+#include_next <same.h>
+EOT
+cat > "$base/cmd-isystem/same.h" <<'EOT'
+from_explicit_isystem
+#include_next <same.h>
+EOT
+cat > "$base/config-as/same.h" <<'EOT'
+from_config_language
+#include_next <same.h>
+EOT
+cat > "$base/config-user/same.h" <<'EOT'
+from_config_user
+#include_next <same.h>
+EOT
+cat > "$base/system/as/same.h" <<'EOT'
+from_system_language
+#include_next <same.h>
+EOT
+cat > "$base/system/same.h" <<'EOT'
+from_system_root
+#include_next <same.h>
+EOT
+cat > "$base/cmd-after/same.h" <<'EOT'
+from_explicit_idirafter
+#include_next <same.h>
+EOT
+cat > "$base/config-after/same.h" <<'EOT'
+from_config_after
+EOT
+
+# Deliberately scramble command-line option order. Semantic classes, not
+# argv order, define precedence.
+"$MCPU_CPP" --config-file "$base/config" \
+ -idirafter "$base/cmd-after" \
+ -isystem "$base/cmd-isystem" \
+ -I "$base/cmd-I" \
+ "$base/main.c" -o "$base/out"
+
+awk '
+ /from_explicit_I/ { a = NR }
+ /from_explicit_isystem/ { b = NR }
+ /from_config_language/ { c = NR }
+ /from_config_user/ { d = NR }
+ /from_system_language/ { e = NR }
+ /from_system_root/ { f = NR }
+ /from_explicit_idirafter/ { g = NR }
+ /from_config_after/ { h = NR }
+ END { exit !(a && a < b && b < c && c < d && d < e && e < f && f < g && g < h) }
+' "$base/out"
+
+# An empty MCPU_CPP_SYSTEM_INCLUDE_PATH disables the complete configured
+# system tree, including the automatically derived <lang> subdirectory.
+cat > "$base/no-system.conf" <<EOT
+MCPU_CPP_SYSTEM_INCLUDE_PATH = ;
+EOT
+cat > "$base/system-only.c" <<'EOT'
+#lang "as"
+#include <system-only.h>
+#endlang
+EOT
+cat > "$base/system/as/system-only.h" <<'EOT'
+system_only_header
+EOT
+if "$MCPU_CPP" --config-file "$base/no-system.conf" \
+ "$base/system-only.c" -o "$base/no-system.out" 2>"$base/no-system.err"; then
+ echo 'empty MCPU_CPP_SYSTEM_INCLUDE_PATH did not disable system includes' >&2
+ exit 1
+fi
+
+# -nostdinc suppresses the configured system tree for one invocation but does
+# not suppress explicit -isystem or either after class.
+cat > "$base/nostdinc.c" <<'EOT'
+#lang "as"
+#include <nostdinc.h>
+#endlang
+EOT
+cat > "$base/system/as/nostdinc.h" <<'EOT'
+wrong_configured_system
+EOT
+cat > "$base/cmd-after/nostdinc.h" <<'EOT'
+from_after_with_nostdinc
+EOT
+"$MCPU_CPP" --config-file "$base/config" -nostdinc \
+ -idirafter "$base/cmd-after" "$base/nostdinc.c" -o "$base/nostdinc.out"
+grep '^from_after_with_nostdinc$' "$base/nostdinc.out" >/dev/null
+! grep '^wrong_configured_system$' "$base/nostdinc.out" >/dev/null
diff --git a/tests/t0045-search-dirs-verbose.sh b/tests/t0045-search-dirs-verbose.sh
new file mode 100755
index 0000000..03a6393
--- /dev/null
+++ b/tests/t0045-search-dirs-verbose.sh
@@ -0,0 +1,89 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-search-dirs-$$"
+mkdir -p \
+ "$base/cmd-user" \
+ "$base/cmd-system" \
+ "$base/config-as" \
+ "$base/config-user" \
+ "$base/system" \
+ "$base/cmd-after" \
+ "$base/config-after"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+runtime_root=$(cd "$(dirname "$MCPU_CPP")/.." && pwd -P)
+
+cat > "$base/test.conf" <<EOT
+MCPU_CPP_INCLUDE_PATH = $base/old-user;
+MCPU_CPP_AS_INCLUDE_PATH = $base/config-as;
+MCPU_CPP_INCLUDE_PATH = $base/config-user;
+MCPU_CPP_SYSTEM_INCLUDE_PATH = $base/system;
+MCPU_CPP_AFTER_INCLUDE_PATH = $base/config-after;
+EOT
+
+"$MCPU_CPP" \
+ --config-file "$base/test.conf" \
+ -I "$base/cmd-user" \
+ -isystem "$base/cmd-system" \
+ -idirafter "$base/cmd-after" \
+ -v -dsearch-dirs \
+ > "$base/search.out" 2> "$base/verbose.out"
+
+cat > "$base/search.expected" <<EOT
+search: $base/cmd-user
+search: $base/cmd-system
+search: $base/config-as
+search: $base/config-user
+search: $base/system/diff
+search: $base/system/dift
+search: $base/system/alg
+search: $base/system/as
+search: $base/system/avm
+search: $base/system/acs
+search: $base/system
+search: $base/cmd-after
+search: $base/config-after
+EOT
+cmp "$base/search.expected" "$base/search.out"
+
+# -v prints one effective value, not every assignment encountered while
+# reading configuration files.
+test "$(grep -c '^config: MCPU_CPP_INCLUDE_PATH=' "$base/verbose.out")" -eq 1
+grep "^config: MCPU_CPP_INCLUDE_PATH=$base/config-user$" "$base/verbose.out" >/dev/null
+if grep "$base/old-user" "$base/verbose.out" >/dev/null; then
+ echo 'verbose output contains an overridden configuration value' >&2
+ exit 1
+fi
+grep "^config: MCPU_CPP_AS_INCLUDE_PATH=$base/config-as$" "$base/verbose.out" >/dev/null
+grep "^config: MCPU_CPP_SYSTEM_INCLUDE_PATH=$base/system$" "$base/verbose.out" >/dev/null
+grep "^config: MCPU_CPP_AFTER_INCLUDE_PATH=$base/config-after$" "$base/verbose.out" >/dev/null
+
+# Even when an explicitly selected configuration file is empty, -v and
+# -dsearch-dirs must expose the runtime-derived installation system root.
+: > "$base/empty.conf"
+"$MCPU_CPP" --config-file "$base/empty.conf" -v -dsearch-dirs \
+ > "$base/default-search.out" 2> "$base/default-verbose.out"
+grep '^config: MCPU_CPP_SYSTEM_INCLUDE_PATH=' "$base/default-verbose.out" >/dev/null
+grep "^search: $runtime_root/include$" "$base/default-search.out" >/dev/null
+
+# -nostdinc removes both the generated <lang> system directories and the
+# configured system root from the directory dump, but keeps explicit -isystem.
+"$MCPU_CPP" \
+ --config-file "$base/test.conf" \
+ -isystem "$base/cmd-system" \
+ -nostdinc -dsearch-dirs > "$base/nostdinc.out"
+grep "^search: $base/cmd-system$" "$base/nostdinc.out" >/dev/null
+if grep "^search: $base/system" "$base/nostdinc.out" >/dev/null; then
+ echo '-nostdinc left configured system directories in -dsearch-dirs' >&2
+ exit 1
+fi
+
+# A dump action stops preprocessing. An input name may be present but need
+# not exist.
+"$MCPU_CPP" --config-file "$base/test.conf" -dsearch-dirs \
+ "$base/does-not-exist.c" > /dev/null
+
+if "$MCPU_CPP" --config-file "$base/test.conf" -dsearch-dirs -o "$base/out" \
+ > /dev/null 2>&1; then
+ echo '-dsearch-dirs accepted an output file' >&2
+ exit 1
+fi
diff --git a/tests/t0046-pragma-once.sh b/tests/t0046-pragma-once.sh
new file mode 100755
index 0000000..9429dd7
--- /dev/null
+++ b/tests/t0046-pragma-once.sh
@@ -0,0 +1,67 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-pragma-once-$$"
+mkdir -p "$base"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+cat > "$base/once.h" <<'EOT'
+#pragma once
+ONCE_BODY
+EOT
+ln -s once.h "$base/once-link.h"
+ln "$base/once.h" "$base/once-hard.h"
+
+cat > "$base/self.h" <<'EOT'
+#pragma once
+SELF_BEGIN
+#include "self.h"
+SELF_END
+EOT
+
+cat > "$base/inactive.h" <<'EOT'
+#if 0
+#pragma once
+#endif
+INACTIVE_BODY
+EOT
+
+cat > "$base/main.c" <<'EOT'
+#include "once.h"
+#include "once.h"
+#include "once-link.h"
+#include "once-hard.h"
+#include "self.h"
+#include "inactive.h"
+#include "inactive.h"
+#pragma pack(push, 1)
+EOT
+
+"$MCPU_CPP" --no-config "$base/main.c" -o "$base/out"
+
+# The same physical file is processed only once, even through a symbolic link
+# and a hard link. #pragma once itself is consumed by the preprocessor.
+test "$(grep -c '^ONCE_BODY$' "$base/out")" -eq 1
+! grep '^#pragma once$' "$base/out" >/dev/null
+
+# Marking happens as soon as the active directive is seen, so a file may safely
+# include itself after #pragma once without recursive inclusion.
+test "$(grep -c '^SELF_BEGIN$' "$base/out")" -eq 1
+test "$(grep -c '^SELF_END$' "$base/out")" -eq 1
+
+# An inactive #pragma once has no effect.
+test "$(grep -c '^INACTIVE_BODY$' "$base/out")" -eq 2
+
+# Other pragmas are not consumed; they remain available to the later compiler.
+grep '^#pragma pack(push, 1)$' "$base/out" >/dev/null
+
+# -dD does not make #pragma once reappear; unrelated pragmas are still kept.
+"$MCPU_CPP" --no-config -dD "$base/main.c" -o "$base/dd.out"
+! grep '^#pragma once$' "$base/dd.out" >/dev/null
+grep '^#pragma pack(push, 1)$' "$base/dd.out" >/dev/null
+
+# A non-file input stream has no physical identity. The directive is simply
+# consumed and must not turn stdin preprocessing into an error.
+printf '#pragma once\nSTDIN_BODY\n' | \
+ "$MCPU_CPP" --no-config > "$base/stdin.out"
+grep '^STDIN_BODY$' "$base/stdin.out" >/dev/null
+! grep '^#pragma once$' "$base/stdin.out" >/dev/null
diff --git a/tests/t0047-dependencies.sh b/tests/t0047-dependencies.sh
new file mode 100755
index 0000000..87ecb63
--- /dev/null
+++ b/tests/t0047-dependencies.sh
@@ -0,0 +1,140 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-deps-$$"
+mkdir -p "$base/user" "$base/sys" "$base/cfg-user" "$base/cfg-sys" "$base/after" "$base/cfg-after"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+cat > "$base/config.conf" <<EOT
+MCPU_CPP_INCLUDE_PATH = $base/cfg-user;
+MCPU_CPP_SYSTEM_INCLUDE_PATH = $base/cfg-sys;
+MCPU_CPP_AFTER_INCLUDE_PATH = $base/cfg-after;
+EOT
+
+cat > "$base/main.c" <<'EOT'
+#include "local.h"
+#include "local-alias.h"
+#include <user.h>
+#include <cfg-user.h>
+#include "sys-explicit.h"
+#include <sys-config.h>
+#include <after-explicit.h>
+#include <after-config.h>
+#include <same.h>
+#include <shared.h>
+SHOULD_NOT_APPEAR_IN_DEP_OUTPUT
+EOT
+
+cat > "$base/local.h" <<'EOT'
+#pragma once
+#line 100 "logical-local.h"
+LOCAL_BODY
+EOT
+ln "$base/local.h" "$base/local-alias.h"
+
+cat > "$base/user/user.h" <<'EOT'
+USER_HEADER
+EOT
+cat > "$base/cfg-user/cfg-user.h" <<'EOT'
+CONFIG_USER_HEADER
+EOT
+cat > "$base/user/shared.h" <<'EOT'
+SHARED_USER_HEADER
+EOT
+cat > "$base/user/transitive-only.h" <<'EOT'
+TRANSITIVE_USER_HEADER
+EOT
+
+cat > "$base/sys/sys-explicit.h" <<'EOT'
+#include <transitive-only.h>
+#include <shared.h>
+#include "sys-child.h"
+EOT
+cat > "$base/sys/sys-child.h" <<'EOT'
+SYSTEM_CHILD
+EOT
+cat > "$base/cfg-sys/sys-config.h" <<'EOT'
+SYSTEM_CONFIG
+EOT
+cat > "$base/after/after-explicit.h" <<'EOT'
+AFTER_EXPLICIT
+EOT
+cat > "$base/cfg-after/after-config.h" <<'EOT'
+AFTER_CONFIG
+EOT
+
+cat > "$base/user/same.h" <<'EOT'
+USER_WRAPPER
+#include_next <same.h>
+EOT
+cat > "$base/sys/same.h" <<'EOT'
+SYSTEM_WRAPPED
+EOT
+
+"$MCPU_CPP" --config-file "$base/config.conf" \
+ -I "$base/user" -isystem "$base/sys" -idirafter "$base/after" \
+ -M "$base/main.c" > "$base/M.out"
+
+# -M emits only a make rule and includes every physical dependency.
+grep "^main\\.o:" "$base/M.out" >/dev/null
+! grep 'SHOULD_NOT_APPEAR_IN_DEP_OUTPUT' "$base/M.out" >/dev/null
+for f in \
+ "$base/main.c" "$base/local.h" "$base/user/user.h" \
+ "$base/cfg-user/cfg-user.h" "$base/sys/sys-explicit.h" \
+ "$base/user/transitive-only.h" "$base/user/shared.h" \
+ "$base/sys/sys-child.h" "$base/cfg-sys/sys-config.h" \
+ "$base/after/after-explicit.h" "$base/cfg-after/after-config.h" \
+ "$base/user/same.h" "$base/sys/same.h"
+do
+ grep -F "$f" "$base/M.out" >/dev/null || {
+ echo "-M omitted dependency: $f" >&2
+ exit 1
+ }
+done
+
+# local-alias.h is a hard link to local.h and must not create a duplicate
+# physical dependency. The logical #line name is not a dependency either.
+! grep -F "$base/local-alias.h" "$base/M.out" >/dev/null
+! grep -F 'logical-local.h' "$base/M.out" >/dev/null
+
+"$MCPU_CPP" --config-file "$base/config.conf" \
+ -I "$base/user" -isystem "$base/sys" -idirafter "$base/after" \
+ -MM "$base/main.c" > "$base/MM.out"
+
+grep "^main\\.o:" "$base/MM.out" >/dev/null
+for f in "$base/main.c" "$base/local.h" "$base/user/user.h" \
+ "$base/cfg-user/cfg-user.h" "$base/user/same.h" "$base/user/shared.h"
+do
+ grep -F "$f" "$base/MM.out" >/dev/null || {
+ echo "-MM omitted user dependency: $f" >&2
+ exit 1
+ }
+done
+
+# Quoting style does not define system-ness. sys-explicit.h was included
+# with quotes but found in -isystem, and all descendants reachable only from
+# that system header are excluded. -idirafter/config AFTER are system class.
+for f in "$base/sys/sys-explicit.h" "$base/user/transitive-only.h" \
+ "$base/sys/sys-child.h" "$base/cfg-sys/sys-config.h" \
+ "$base/after/after-explicit.h" "$base/cfg-after/after-config.h" \
+ "$base/sys/same.h"
+do
+ if grep -F "$f" "$base/MM.out" >/dev/null; then
+ echo "-MM kept system dependency: $f" >&2
+ exit 1
+ fi
+done
+
+# The same physical shared.h is first reached transitively from a system
+# header, then directly from main. The direct user inclusion makes it a user
+# dependency, so -MM must retain it (checked above).
+
+# Normal mcpu-cpp output-file syntax remains usable for dependency output.
+"$MCPU_CPP" --config-file "$base/config.conf" -I "$base/user" \
+ -isystem "$base/sys" -idirafter "$base/after" \
+ -MM "$base/main.c" -o "$base/deps.mk"
+test -s "$base/deps.mk"
+grep '^main\.o:' "$base/deps.mk" >/dev/null
+
+# stdin follows GNU cpp's useful default target convention.
+printf 'int x;\n' | "$MCPU_CPP" --no-config -M > "$base/stdin.dep"
+grep '^-: -$' "$base/stdin.dep" >/dev/null
diff --git a/tests/t0048-no-config-defaults.sh b/tests/t0048-no-config-defaults.sh
new file mode 100755
index 0000000..a6d2c75
--- /dev/null
+++ b/tests/t0048-no-config-defaults.sh
@@ -0,0 +1,73 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-no-config-defaults-$$"
+mkdir -p "$base/home/.mcpu" "$base/I" "$base/isystem" "$base/after"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+runtime_root=$(cd "$(dirname "$MCPU_CPP")/.." && pwd -P)
+
+cat > "$base/home/.mcpu/mcpu-cpp.conf" <<EOT
+MCPU_CPP_INCLUDE_PATH = $base/forbidden-user;
+MCPU_CPP_SYSTEM_INCLUDE_PATH = $base/forbidden-system;
+MCPU_CPP_AFTER_INCLUDE_PATH = $base/forbidden-after;
+EOT
+
+# --no-config skips every configuration file, but the runtime-derived installation
+# default remains the effective system root.
+HOME="$base/home" "$MCPU_CPP" --no-config -dsearch-dirs > "$base/search.out"
+grep "^search: $runtime_root/include/diff$" "$base/search.out" >/dev/null
+grep "^search: $runtime_root/include/dift$" "$base/search.out" >/dev/null
+grep "^search: $runtime_root/include/alg$" "$base/search.out" >/dev/null
+grep "^search: $runtime_root/include/as$" "$base/search.out" >/dev/null
+grep "^search: $runtime_root/include/avm$" "$base/search.out" >/dev/null
+grep "^search: $runtime_root/include/acs$" "$base/search.out" >/dev/null
+grep "^search: $runtime_root/include$" "$base/search.out" >/dev/null
+if grep "$base/forbidden" "$base/search.out" >/dev/null; then
+ echo '--no-config read the user configuration file' >&2
+ exit 1
+fi
+
+# The same runtime-derived value is visible through -dconfig and -v.
+HOME="$base/home" "$MCPU_CPP" --no-config -dconfig > "$base/config.out"
+grep "^MCPU_CPP_SYSTEM_INCLUDE_PATH = $runtime_root/include;$" "$base/config.out" >/dev/null
+if grep "$base/forbidden" "$base/config.out" >/dev/null; then
+ echo '--no-config leaked user configuration into -dconfig' >&2
+ exit 1
+fi
+
+HOME="$base/home" "$MCPU_CPP" --no-config -v -dsearch-dirs \
+ > "$base/verbose-search.out" 2> "$base/verbose.out"
+grep "^config: MCPU_CPP_SYSTEM_INCLUDE_PATH=$runtime_root/include$" "$base/verbose.out" >/dev/null
+if grep "$base/forbidden" "$base/verbose.out" >/dev/null; then
+ echo '--no-config leaked user configuration into -v' >&2
+ exit 1
+fi
+
+# Explicit command-line classes retain their normative priority around the
+# runtime-derived system root.
+HOME="$base/home" "$MCPU_CPP" --no-config \
+ -I "$base/I" -isystem "$base/isystem" -idirafter "$base/after" \
+ -dsearch-dirs > "$base/ordered.out"
+first=$(sed -n '1p' "$base/ordered.out")
+second=$(sed -n '2p' "$base/ordered.out")
+last=$(tail -n 1 "$base/ordered.out")
+test "$first" = "search: $base/I"
+test "$second" = "search: $base/isystem"
+test "$last" = "search: $base/after"
+grep "^search: $runtime_root/include$" "$base/ordered.out" >/dev/null
+
+# -nostdinc, not --no-config, is the switch that suppresses the runtime-derived
+# standard-system include tree. Explicit classes remain visible.
+HOME="$base/home" "$MCPU_CPP" --no-config -nostdinc \
+ -I "$base/I" -isystem "$base/isystem" -idirafter "$base/after" \
+ -dsearch-dirs > "$base/nostdinc.out"
+cat > "$base/nostdinc.expected" <<EOT
+search: $base/I
+search: $base/isystem
+search: $base/after
+EOT
+cmp "$base/nostdinc.expected" "$base/nostdinc.out"
+
+# With no explicit directories, --no-config -nostdinc has no search entries.
+HOME="$base/home" "$MCPU_CPP" --no-config -nostdinc -dsearch-dirs \
+ > "$base/empty.out"
+test ! -s "$base/empty.out"
diff --git a/tests/t0049-relocatable-root.sh b/tests/t0049-relocatable-root.sh
new file mode 100755
index 0000000..d64a9b8
--- /dev/null
+++ b/tests/t0049-relocatable-root.sh
@@ -0,0 +1,55 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-relocatable-root-$$"
+root_a="$base/mcpu-a"
+root_b="$base/mcpu-b"
+mkdir -p "$root_a/bin" "$root_a/etc" "$root_a/include/diff" "$base/home"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+cp "$MCPU_CPP" "$root_a/bin/mcpu-cpp"
+chmod +x "$root_a/bin/mcpu-cpp"
+
+cat > "$root_a/include/diff/relocated.h" <<'EOT'
+int relocated_header = 37;
+EOT
+
+cat > "$root_a/etc/mcpu-cpp.conf" <<'EOT'
+MCPU_CPP_RELOCATION_SENTINEL = runtime-config;
+EOT
+
+# The copied executable first derives its defaults from its new physical root.
+HOME="$base/home" "$root_a/bin/mcpu-cpp" --no-config -dsearch-dirs \
+ > "$base/root-a.search"
+grep "^search: $root_a/include/diff$" "$base/root-a.search" >/dev/null
+grep "^search: $root_a/include$" "$base/root-a.search" >/dev/null
+
+# Move the complete MCPU tree without reconfiguring or rebuilding anything.
+mv "$root_a" "$root_b"
+
+HOME="$base/home" "$root_b/bin/mcpu-cpp" --no-config -dsearch-dirs \
+ > "$base/root-b.search"
+grep "^search: $root_b/include/diff$" "$base/root-b.search" >/dev/null
+grep "^search: $root_b/include$" "$base/root-b.search" >/dev/null
+if grep "$root_a" "$base/root-b.search" >/dev/null; then
+ echo 'relocated mcpu-cpp retained its previous installation root' >&2
+ exit 1
+fi
+
+cat > "$base/main.c" <<'EOT'
+#lang "diff"
+#include <relocated.h>
+#endlang
+EOT
+HOME="$base/home" "$root_b/bin/mcpu-cpp" --no-config \
+ "$base/main.c" -o "$base/main.out"
+grep '^int relocated_header = 37;$' "$base/main.out" >/dev/null
+
+# The packaged configuration is found relative to the relocated root as well.
+HOME="$base/home" "$root_b/bin/mcpu-cpp" -dconfig > "$base/config.out"
+grep '^MCPU_CPP_RELOCATION_SENTINEL = runtime-config;$' "$base/config.out" >/dev/null
+
+# Invoking through a symbolic link must still resolve the physical executable
+# and therefore the same relocated MCPU root.
+ln -s "$root_b/bin/mcpu-cpp" "$base/mcpu-cpp"
+HOME="$base/home" "$base/mcpu-cpp" --no-config -dconfig > "$base/link.out"
+grep "^MCPU_CPP_SYSTEM_INCLUDE_PATH = $root_b/include;$" "$base/link.out" >/dev/null
diff --git a/tests/t0050-dependency-side-effects.sh b/tests/t0050-dependency-side-effects.sh
new file mode 100755
index 0000000..67cbcf6
--- /dev/null
+++ b/tests/t0050-dependency-side-effects.sh
@@ -0,0 +1,136 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-dep-side-effects-$$"
+mkdir -p "$base/src" "$base/user" "$base/sys" "$base/out" "$base/stdin"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+cat > "$base/src/main.c" <<'EOT'
+#include "user.h"
+#include <system.h>
+MAIN_BODY
+EOT
+cat > "$base/user/user.h" <<'EOT'
+USER_BODY
+EOT
+cat > "$base/sys/system.h" <<'EOT'
+#include "system-child.h"
+SYSTEM_BODY
+EOT
+cat > "$base/sys/system-child.h" <<'EOT'
+SYSTEM_CHILD_BODY
+EOT
+
+# -MD keeps preprocessing output and writes basename.d in the current
+# directory when no -o/-MF is present. The input directory is not copied to
+# the default dependency filename.
+(
+ cd "$base"
+ "$MCPU_CPP" --no-config -I "$base/user" -isystem "$base/sys" \
+ -MD "$base/src/main.c" > md.out
+)
+test -s "$base/md.out"
+grep 'MAIN_BODY' "$base/md.out" >/dev/null
+test -s "$base/main.d"
+grep '^main\.o:' "$base/main.d" >/dev/null
+for f in "$base/src/main.c" "$base/user/user.h" \
+ "$base/sys/system.h" "$base/sys/system-child.h"
+do
+ grep -F "$f" "$base/main.d" >/dev/null || {
+ echo "-MD omitted dependency: $f" >&2
+ exit 1
+ }
+done
+
+# -MMD has the same side-effect behavior but applies the already established
+# -MM user-dependency filter.
+rm -f "$base/main.d"
+(
+ cd "$base"
+ "$MCPU_CPP" --no-config -I "$base/user" -isystem "$base/sys" \
+ -MMD "$base/src/main.c" > mmd.out
+)
+test -s "$base/mmd.out"
+grep 'MAIN_BODY' "$base/mmd.out" >/dev/null
+test -s "$base/main.d"
+grep -F "$base/src/main.c" "$base/main.d" >/dev/null
+grep -F "$base/user/user.h" "$base/main.d" >/dev/null
+! grep -F "$base/sys/system.h" "$base/main.d" >/dev/null
+! grep -F "$base/sys/system-child.h" "$base/main.d" >/dev/null
+
+# With ordinary -o, preprocessing goes to that file and the automatic
+# dependency filename is obtained by replacing its suffix with .d.
+rm -f "$base/main.d" "$base/out/result.i" "$base/out/result.d"
+"$MCPU_CPP" --no-config -I "$base/user" -isystem "$base/sys" \
+ -MD "$base/src/main.c" -o "$base/out/result.i"
+test -s "$base/out/result.i"
+grep 'MAIN_BODY' "$base/out/result.i" >/dev/null
+test -s "$base/out/result.d"
+grep '^main\.o:' "$base/out/result.d" >/dev/null
+test ! -e "$base/main.d"
+
+# -MF overrides automatic dependency naming while leaving normal preprocessing
+# output semantics untouched.
+rm -f "$base/custom.d" "$base/main.d"
+(
+ cd "$base"
+ "$MCPU_CPP" --no-config -I "$base/user" -isystem "$base/sys" \
+ -MD -MF custom.d "$base/src/main.c" > mf.out
+)
+test -s "$base/mf.out"
+grep 'MAIN_BODY' "$base/mf.out" >/dev/null
+test -s "$base/custom.d"
+test ! -e "$base/main.d"
+
+# Attached -MFfile spelling is accepted as by GNU cpp.
+rm -f "$base/attached.d"
+(
+ cd "$base"
+ "$MCPU_CPP" --no-config -I "$base/user" -isystem "$base/sys" \
+ -MMD -MFattached.d "$base/src/main.c" > attached.out
+)
+test -s "$base/attached.d"
+! grep -F "$base/sys/system.h" "$base/attached.d" >/dev/null
+
+# -MF also redirects the already implemented dependency-only -M/-MM modes.
+rm -f "$base/only.d" "$base/only.out"
+"$MCPU_CPP" --no-config -I "$base/user" -isystem "$base/sys" \
+ -M -MF "$base/only.d" "$base/src/main.c" > "$base/only.out"
+test ! -s "$base/only.out"
+test -s "$base/only.d"
+grep -F "$base/sys/system.h" "$base/only.d" >/dev/null
+
+rm -f "$base/only-user.d" "$base/only-user.out"
+"$MCPU_CPP" --no-config -I "$base/user" -isystem "$base/sys" \
+ -MM -MF "$base/only-user.d" "$base/src/main.c" > "$base/only-user.out"
+test ! -s "$base/only-user.out"
+test -s "$base/only-user.d"
+! grep -F "$base/sys/system.h" "$base/only-user.d" >/dev/null
+
+# -MF - means dependency stdout. Put normal preprocessing in a regular -o
+# file so the two streams are easy to verify independently.
+rm -f "$base/out/stdout.i" "$base/out/stdout.d"
+"$MCPU_CPP" --no-config -I "$base/user" -isystem "$base/sys" \
+ -MMD -MF - "$base/src/main.c" -o "$base/out/stdout.i" \
+ > "$base/mf-dash.out"
+test -s "$base/out/stdout.i"
+grep 'MAIN_BODY' "$base/out/stdout.i" >/dev/null
+grep '^main\.o:' "$base/mf-dash.out" >/dev/null
+test ! -e "$base/out/stdout.d"
+
+# stdin follows the existing '-' target convention and gets '-.d' when -MD
+# needs an automatic dependency filename.
+(
+ cd "$base/stdin"
+ printf 'STDIN_BODY\n' | "$MCPU_CPP" --no-config -MD > stdout.i
+)
+test -s "$base/stdin/stdout.i"
+grep 'STDIN_BODY' "$base/stdin/stdout.i" >/dev/null
+test -s "$base/stdin/-.d"
+grep '^-: -$' "$base/stdin/-.d" >/dev/null
+
+# -MF alone is a command-line error: there is no dependency mode to redirect.
+if "$MCPU_CPP" --no-config -MF "$base/orphan.d" "$base/src/main.c" \
+ > /dev/null 2>&1; then
+ echo "-MF without a dependency mode unexpectedly succeeded" >&2
+ exit 1
+fi
diff --git a/tests/t0051-dump-macro-origin.sh b/tests/t0051-dump-macro-origin.sh
new file mode 100755
index 0000000..1a2c504
--- /dev/null
+++ b/tests/t0051-dump-macro-origin.sh
@@ -0,0 +1,63 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-dump-origin-$$"
+mkdir -p "$base"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+cat > "$base/main.c" <<'EOT'
+#define Z_USER 62
+#define A_USER 37
+#define ФУНКЦИЯ(x) x + 1
+#undef __MCPU_CPP_VERSION__
+EOT
+
+# -dM contains only ordinary final-state macros. Source and command-line
+# definitions are ordinary; built-ins are not.
+"$MCPU_CPP" --no-config -dM -DCMD_USER=91 "$base/main.c" > "$base/dM"
+grep -Fx '#define A_USER 37' "$base/dM" >/dev/null
+grep -Fx '#define CMD_USER 91' "$base/dM" >/dev/null
+grep -Fx '#define Z_USER 62' "$base/dM" >/dev/null
+grep -Fx '#define ФУНКЦИЯ(x) x + 1' "$base/dM" >/dev/null
+if grep '^#define _ARCH_MCPU ' "$base/dM" >/dev/null; then exit 1; fi
+if grep '^#define __MCPU_' "$base/dM" >/dev/null; then exit 1; fi
+LC_ALL=C sort -c "$base/dM"
+
+# Empty input has no ordinary definitions.
+"$MCPU_CPP" --no-config -dM < /dev/null > "$base/empty-dM"
+test ! -s "$base/empty-dM"
+
+# -dMP contains predefined macros first and ordinary macros second. Both
+# groups are deterministic and the #undef above removes the version builtin.
+"$MCPU_CPP" --no-config -dMP -DCMD_USER=91 "$base/main.c" > "$base/dMP"
+grep -Fx '#define _ARCH_MCPU 1' "$base/dMP" >/dev/null
+grep -Fx '#define A_USER 37' "$base/dMP" >/dev/null
+grep -Fx '#define CMD_USER 91' "$base/dMP" >/dev/null
+grep -Fx '#define Z_USER 62' "$base/dMP" >/dev/null
+grep -Fx '#define ФУНКЦИЯ(x) x + 1' "$base/dMP" >/dev/null
+if grep '^#define __MCPU_CPP_VERSION__ ' "$base/dMP" >/dev/null; then exit 1; fi
+
+predefined_line=`grep -n '^#define _ARCH_MCPU 1$' "$base/dMP" | sed 's/:.*//'`
+ordinary_line=`grep -n '^#define A_USER 37$' "$base/dMP" | sed 's/:.*//'`
+test "$predefined_line" -lt "$ordinary_line"
+
+# Verify ordering inside the ordinary tail explicitly.
+tail -n +"$ordinary_line" "$base/dMP" > "$base/ordinary-tail"
+LC_ALL=C sort -c "$base/ordinary-tail"
+
+# A former builtin name, if explicitly redefined after #undef, is an ordinary
+# macro because origin follows the current definition rather than the spelling.
+cat > "$base/redefine.c" <<'EOT'
+#undef _ARCH_MCPU
+#define _ARCH_MCPU 777
+#define LOCAL 1
+EOT
+"$MCPU_CPP" --no-config -dM "$base/redefine.c" > "$base/redefine-dM"
+grep -Fx '#define _ARCH_MCPU 777' "$base/redefine-dM" >/dev/null
+"$MCPU_CPP" --no-config -dMP "$base/redefine.c" > "$base/redefine-dMP"
+grep -Fx '#define _ARCH_MCPU 777' "$base/redefine-dMP" >/dev/null
+test "`grep -c '^#define _ARCH_MCPU ' "$base/redefine-dMP"`" -eq 1
+
+# Context-dependent special builtins remain outside the static dump.
+for name in __FILE__ __LINE__ __DATE__ __TIME__ __BASE_FILE__ __INCLUDE_LEVEL__; do
+ if grep "^#define ${name}\\>" "$base/dMP" >/dev/null; then exit 1; fi
+done
diff --git a/tests/t0052-warning-control.sh b/tests/t0052-warning-control.sh
new file mode 100755
index 0000000..8214d96
--- /dev/null
+++ b/tests/t0052-warning-control.sh
@@ -0,0 +1,108 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-warning-control-$$"
+mkdir -p "$base"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+cat > "$base/comments.c" <<'EOT'
+/* outer /* nested */
+// continued \
+comment
+ok
+EOT
+
+# Comment warnings are optional and disabled by default.
+"$MCPU_CPP" --no-config "$base/comments.c" -o "$base/default.out" 2>"$base/default.err"
+test ! -s "$base/default.err"
+grep '^ok$' "$base/default.out" >/dev/null
+
+# -Wcomment and -Wcomments are synonyms.
+for option in -Wcomment -Wcomments -Wall; do
+ "$MCPU_CPP" --no-config "$option" "$base/comments.c" \
+ -o "$base/comment.out" 2>"$base/comment.err"
+ test "`grep -c 'warning: \"/\*\" within comment' "$base/comment.err"`" -eq 1
+ test "`grep -c 'warning: multi-line comment' "$base/comment.err"`" -eq 1
+done
+
+# A specific -Wno-comment is stronger than the -Wall group regardless of order.
+for options in '-Wall -Wno-comment' '-Wno-comment -Wall' \
+ '-Wall -Wno-comments' '-Wno-comments -Wall'; do
+ # shellcheck disable=SC2086
+ "$MCPU_CPP" --no-config $options "$base/comments.c" \
+ -o "$base/no-comment.out" 2>"$base/no-comment.err"
+ test ! -s "$base/no-comment.err"
+done
+
+# Same-specificity comment options use the last occurrence.
+"$MCPU_CPP" --no-config -Wno-comment -Wcomment "$base/comments.c" \
+ -o "$base/comment-last.out" 2>"$base/comment-last.err"
+grep 'warning: multi-line comment' "$base/comment-last.err" >/dev/null
+
+"$MCPU_CPP" --no-config -Wcomment -Wno-comment "$base/comments.c" \
+ -o "$base/no-comment-last.out" 2>"$base/no-comment-last.err"
+test ! -s "$base/no-comment-last.err"
+
+# -Werror does not enable optional comment diagnostics by itself.
+"$MCPU_CPP" --no-config -Werror "$base/comments.c" \
+ -o "$base/werror-only.out" 2>"$base/werror-only.err"
+test ! -s "$base/werror-only.err"
+
+# When comment diagnostics are enabled, -Werror promotes them to errors.
+if "$MCPU_CPP" --no-config -Wcomment -Werror "$base/comments.c" \
+ -o "$base/comment-error.out" 2>"$base/comment-error.err"; then
+ echo '-Wcomment -Werror did not fail' >&2
+ exit 1
+fi
+grep 'error: "/\*" within comment' "$base/comment-error.err" >/dev/null
+
+# -Werror/-Wno-error have equal specificity; the last occurrence wins.
+cat > "$base/directive.c" <<'EOT'
+#warning requested warning
+ok
+EOT
+"$MCPU_CPP" --no-config -Werror -Wno-error "$base/directive.c" \
+ -o "$base/no-error.out" 2>"$base/no-error.err"
+grep 'warning: #warning requested warning$' "$base/no-error.err" >/dev/null
+
+if "$MCPU_CPP" --no-config -Wno-error -Werror "$base/directive.c" \
+ -o "$base/error-last.out" 2>"$base/error-last.err"; then
+ echo '-Wno-error -Werror did not fail' >&2
+ exit 1
+fi
+grep 'error: #warning requested warning$' "$base/error-last.err" >/dev/null
+
+# Existing always-issued warnings are also promoted by -Werror.
+cat > "$base/width.c" <<'EOT'
+#if 1z24 == 1
+ok
+#endif
+EOT
+if "$MCPU_CPP" --no-config -Werror "$base/width.c" \
+ -o "$base/width.out" 2>"$base/width.err"; then
+ echo 'zNNN warning was not promoted by -Werror' >&2
+ exit 1
+fi
+grep 'error: invalid zNNN integer-width suffix; suffix ignored$' "$base/width.err" >/dev/null
+
+cat > "$base/redefine.c" <<'EOT'
+#define VALUE 1
+#define VALUE 2
+EOT
+if "$MCPU_CPP" --no-config -Werror "$base/redefine.c" \
+ -o "$base/redefine.out" 2>"$base/redefine.err"; then
+ echo 'macro redefinition warning was not promoted by -Werror' >&2
+ exit 1
+fi
+grep "error: macro 'VALUE' redefined$" "$base/redefine.err" >/dev/null
+
+cat > "$base/paste.c" <<'EOT'
+#define BAD(X) X ## +
+BAD(foo)
+EOT
+if "$MCPU_CPP" --no-config -Werror "$base/paste.c" \
+ -o "$base/paste.out" 2>"$base/paste.err"; then
+ echo 'invalid paste warning was not promoted by -Werror' >&2
+ exit 1
+fi
+grep 'error: pasting "foo" and "+" does not give a valid preprocessing token$' \
+ "$base/paste.err" >/dev/null
diff --git a/tests/t0053-interface-cleanup.sh b/tests/t0053-interface-cleanup.sh
new file mode 100755
index 0000000..80d5410
--- /dev/null
+++ b/tests/t0053-interface-cleanup.sh
@@ -0,0 +1,34 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-interface-cleanup-$$"
+mkdir -p "$base/wrapper" "$base/after"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+# --version identifies the independent MCPU languages preprocessor.
+"$MCPU_CPP" --version > "$base/version"
+printf '%s\n' 'mcpu-cpp 1.0.2' 'MCPU languages preprocessor' > "$base/version.expected"
+cmp "$base/version.expected" "$base/version"
+
+# --help contains the current public interface, including --object-suffix,
+# and puts the stdin/stdout notes after the option list.
+"$MCPU_CPP" --help > "$base/help"
+grep '^mcpu-cpp 1\.0\.2$' "$base/help" >/dev/null
+grep '^MCPU languages preprocessor$' "$base/help" >/dev/null
+grep '^ --object-suffix SFX set object suffix used for dependency targets$' "$base/help" >/dev/null
+help_version_line=$(grep -n '^ --version ' "$base/help" | cut -d: -f1)
+input_note_line=$(grep -n "^If input is omitted or '-', read standard input\.$" "$base/help" | cut -d: -f1)
+test "$input_note_line" -gt "$help_version_line"
+
+grep "^If output is omitted or '-', write standard output\.$" "$base/help" >/dev/null
+
+# --object-suffix requires an argument and changes the default dependency target.
+cat > "$base/object.c" <<'SRC'
+int x;
+SRC
+if "$MCPU_CPP" --no-config --object-suffix > /dev/null 2> "$base/object.err"; then
+ echo '--object-suffix unexpectedly accepted without an argument' >&2
+ exit 1
+fi
+grep "argument missing after '--object-suffix'" "$base/object.err" >/dev/null
+"$MCPU_CPP" --no-config --object-suffix .obj -M "$base/object.c" > "$base/object.d"
+grep '^object\.obj:' "$base/object.d" >/dev/null
diff --git a/tests/t0054-inhibit-warnings.sh b/tests/t0054-inhibit-warnings.sh
new file mode 100755
index 0000000..339d098
--- /dev/null
+++ b/tests/t0054-inhibit-warnings.sh
@@ -0,0 +1,81 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-inhibit-warnings-$$"
+mkdir -p "$base"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+cat > "$base/directive.c" <<'EOT'
+#warning requested warning
+ok
+EOT
+
+# -w suppresses #warning completely.
+"$MCPU_CPP" --no-config -w "$base/directive.c" \
+ -o "$base/directive.out" 2>"$base/directive.err"
+test ! -s "$base/directive.err"
+grep '^ok$' "$base/directive.out" >/dev/null
+
+# -w is a global warning gate, so command-line order relative to -Werror
+# does not matter and there is no warning left to promote.
+for options in '-w -Werror' '-Werror -w'; do
+ # shellcheck disable=SC2086
+ "$MCPU_CPP" --no-config $options "$base/directive.c" \
+ -o "$base/werror.out" 2>"$base/werror.err"
+ test ! -s "$base/werror.err"
+done
+
+cat > "$base/width.c" <<'EOT'
+#if 1z24 == 1
+ok
+#endif
+EOT
+"$MCPU_CPP" --no-config -w "$base/width.c" \
+ -o "$base/width.out" 2>"$base/width.err"
+test ! -s "$base/width.err"
+grep '^ok$' "$base/width.out" >/dev/null
+
+cat > "$base/redefine.c" <<'EOT'
+#define VALUE 1
+#define VALUE 2
+VALUE
+EOT
+"$MCPU_CPP" --no-config -w "$base/redefine.c" \
+ -o "$base/redefine.out" 2>"$base/redefine.err"
+test ! -s "$base/redefine.err"
+grep '^2$' "$base/redefine.out" >/dev/null
+
+cat > "$base/paste.c" <<'EOT'
+#define BAD(X) X ## +
+BAD(foo)
+EOT
+"$MCPU_CPP" --no-config -w "$base/paste.c" \
+ -o "$base/paste.out" 2>"$base/paste.err"
+test ! -s "$base/paste.err"
+
+cat > "$base/comments.c" <<'EOT'
+/* outer /* nested */
+// continued \
+comment
+ok
+EOT
+
+# A specific warning class may be enabled, but -w still suppresses its output.
+# Command-line order does not matter.
+for options in '-w -Wcomment' '-Wcomment -w' '-w -Wall' '-Wall -w'; do
+ # shellcheck disable=SC2086
+ "$MCPU_CPP" --no-config $options "$base/comments.c" \
+ -o "$base/comments.out" 2>"$base/comments.err"
+ test ! -s "$base/comments.err"
+ grep '^ok$' "$base/comments.out" >/dev/null
+done
+
+# Errors are not warnings and must remain visible/fatal under -w.
+cat > "$base/error.c" <<'EOT'
+#error requested error
+EOT
+if "$MCPU_CPP" --no-config -w "$base/error.c" \
+ -o "$base/error.out" 2>"$base/error.err"; then
+ echo '-w suppressed a real preprocessing error' >&2
+ exit 1
+fi
+grep 'error: #error requested error$' "$base/error.err" >/dev/null
diff --git a/tests/t0055-forced-files.sh b/tests/t0055-forced-files.sh
new file mode 100755
index 0000000..454fa55
--- /dev/null
+++ b/tests/t0055-forced-files.sh
@@ -0,0 +1,242 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-forced-files-$$"
+mkdir -p "$base/project/src" "$base/project/user" "$base/project/sys"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+cat > "$base/project/defs1.h" <<'EOT'
+IMACROS1_BODY_MUST_NOT_APPEAR
+#if CLI_VALUE != 9
+#error command-line macros were not applied before -imacros
+#endif
+#if __INCLUDE_LEVEL__ != 1
+#error -imacros must run at include level 1
+#endif
+#define ORDER_VALUE 1
+#define FROM_IMACROS1 11
+#include "imacros-child.h"
+#lang "as"
+#define FROM_AS_LANG 13
+#endlang
+EOT
+
+cat > "$base/project/imacros-child.h" <<'EOT'
+IMACROS_CHILD_BODY_MUST_NOT_APPEAR
+#if __INCLUDE_LEVEL__ != 2
+#error nested include from -imacros must run at include level 2
+#endif
+#define FROM_IMACROS_CHILD 12
+EOT
+
+cat > "$base/project/defs2.h" <<'EOT'
+IMACROS2_BODY_MUST_NOT_APPEAR
+#if ORDER_VALUE != 1
+#error -imacros command-line order is broken
+#endif
+#if FROM_IMACROS_CHILD != 12 || FROM_AS_LANG != 13
+#error -imacros did not retain preprocessing state
+#endif
+#undef ORDER_VALUE
+#define ORDER_VALUE 2
+EOT
+
+cat > "$base/project/a.h" <<'EOT'
+#if ORDER_VALUE != 2
+#error all -imacros files must precede all -include files
+#endif
+INCLUDE_A_BODY
+FORCED_LEVEL __INCLUDE_LEVEL__
+FORCED_BASE __BASE_FILE__
+#undef ORDER_VALUE
+#define ORDER_VALUE 3
+EOT
+
+cat > "$base/project/b.h" <<'EOT'
+#if ORDER_VALUE != 3
+#error -include command-line order is broken
+#endif
+INCLUDE_B_BODY
+#undef ORDER_VALUE
+#define ORDER_VALUE 4
+EOT
+
+cat > "$base/project/src/main.c" <<'EOT'
+#if ORDER_VALUE != 4
+#error forced files did not complete before primary input
+#endif
+MAIN_BODY
+MAIN_LEVEL __INCLUDE_LEVEL__
+FROM_IMACROS1 FROM_IMACROS_CHILD FROM_AS_LANG CLI_VALUE
+EOT
+
+# Deliberately interleave -include and -imacros. The semantic order is all
+# command-line -D/-U first, then all -imacros in their own argv order, then all
+# -include in their own argv order, and only then the primary input.
+(
+ cd "$base/project"
+ "$MCPU_CPP" --no-config \
+ -include a.h -DCLI_VALUE=9 -imacros defs1.h \
+ -include b.h -imacros defs2.h \
+ src/main.c -o "$base/order.out"
+)
+
+! grep 'IMACROS1_BODY_MUST_NOT_APPEAR' "$base/order.out" >/dev/null
+! grep 'IMACROS2_BODY_MUST_NOT_APPEAR' "$base/order.out" >/dev/null
+! grep 'IMACROS_CHILD_BODY_MUST_NOT_APPEAR' "$base/order.out" >/dev/null
+! grep '^#lang ' "$base/order.out" >/dev/null
+! grep '^#endlang' "$base/order.out" >/dev/null
+
+awk '
+ /^INCLUDE_A_BODY$/ { a = NR }
+ /^INCLUDE_B_BODY$/ { b = NR }
+ /^MAIN_BODY$/ { c = NR }
+ END { exit !(a && a < b && b < c) }
+' "$base/order.out"
+grep '^11 12 13 9$' "$base/order.out" >/dev/null
+grep '^FORCED_LEVEL 1$' "$base/order.out" >/dev/null
+grep '^FORCED_BASE "src/main.c"$' "$base/order.out" >/dev/null
+grep '^MAIN_LEVEL 0$' "$base/order.out" >/dev/null
+# Each visible forced include returns to the primary file in the line-marker
+# stack before another forced include or the primary body continues.
+test `grep -c '^# 1 "src/main.c" 2$' "$base/order.out"` -eq 2
+
+# A forced command-line file is searched from the current working directory,
+# not from the physical directory of the primary input.
+cat > "$base/project/cwd-first.h" <<'EOT'
+CWD_FORCED_HEADER
+EOT
+cat > "$base/project/src/cwd-first.h" <<'EOT'
+PRIMARY_DIRECTORY_MUST_NOT_WIN
+EOT
+cat > "$base/project/src/cwd-main.c" <<'EOT'
+CWD_MAIN
+EOT
+(
+ cd "$base/project"
+ "$MCPU_CPP" --no-config -include cwd-first.h src/cwd-main.c \
+ -o "$base/cwd.out"
+)
+grep '^CWD_FORCED_HEADER$' "$base/cwd.out" >/dev/null
+! grep 'PRIMARY_DIRECTORY_MUST_NOT_WIN' "$base/cwd.out" >/dev/null
+
+# After CWD, the ordinary quoted include chain is used. A forced file found
+# through -I retains its physical provenance, so quoted includes inside it are
+# resolved relative to that file's own directory.
+cat > "$base/project/user/forced-user.h" <<'EOT'
+FORCED_USER_BEGIN
+#include "forced-child.h"
+FORCED_USER_END
+EOT
+cat > "$base/project/user/forced-child.h" <<'EOT'
+FORCED_USER_CHILD
+EOT
+(
+ cd "$base/project"
+ "$MCPU_CPP" --no-config -I user -include forced-user.h src/cwd-main.c \
+ -o "$base/user.out"
+)
+grep '^FORCED_USER_BEGIN$' "$base/user.out" >/dev/null
+grep '^FORCED_USER_CHILD$' "$base/user.out" >/dev/null
+grep '^FORCED_USER_END$' "$base/user.out" >/dev/null
+
+# Search-chain provenance of a forced file is retained for #include_next.
+cat > "$base/project/user/wrapped.h" <<'EOT'
+FORCED_WRAPPER
+#include_next <wrapped.h>
+EOT
+cat > "$base/project/sys/wrapped.h" <<'EOT'
+FORCED_INCLUDE_NEXT
+EOT
+(
+ cd "$base/project"
+ "$MCPU_CPP" --no-config -I user -isystem sys -include wrapped.h \
+ src/cwd-main.c -o "$base/next.out"
+)
+grep '^FORCED_WRAPPER$' "$base/next.out" >/dev/null
+grep '^FORCED_INCLUDE_NEXT$' "$base/next.out" >/dev/null
+
+# -dD does not make normal output from -imacros visible; macro state still
+# reaches the primary input.
+cat > "$base/project/only-macros.h" <<'EOT'
+IMACROS_DD_BODY_MUST_NOT_APPEAR
+#define DD_VALUE 77
+EOT
+cat > "$base/project/src/dd-main.c" <<'EOT'
+DD_VALUE
+EOT
+(
+ cd "$base/project"
+ "$MCPU_CPP" --no-config -dD -imacros only-macros.h src/dd-main.c \
+ -o "$base/dd.out"
+)
+! grep 'IMACROS_DD_BODY_MUST_NOT_APPEAR' "$base/dd.out" >/dev/null
+! grep '^#define DD_VALUE 77$' "$base/dd.out" >/dev/null
+grep '^77$' "$base/dd.out" >/dev/null
+
+# Forced files participate in physical dependency tracking. The primary input
+# remains the first dependency even though forced files are preprocessed first.
+(
+ cd "$base/project"
+ "$MCPU_CPP" --no-config -M -DCLI_VALUE=9 \
+ -imacros defs1.h -imacros defs2.h -include a.h -include b.h src/main.c \
+ > "$base/deps.out"
+)
+grep '^main\.o:' "$base/deps.out" >/dev/null
+first_dep=`sed 's/^[^:]*: //' "$base/deps.out" | awk '{print $1}'`
+test "$first_dep" = 'src/main.c' || {
+ echo 'primary input is not the first forced-file dependency' >&2
+ exit 1
+}
+for f in defs1.h imacros-child.h defs2.h a.h b.h; do
+ grep -E "(^|[[:space:]])(\./)?$f([[:space:]]|$)" "$base/deps.out" >/dev/null || {
+ echo "forced dependency missing: $f" >&2
+ exit 1
+ }
+done
+
+# A forced file found through -isystem has system dependency class and is
+# therefore omitted by -MM while the primary user source remains.
+cat > "$base/project/sys/system-forced.h" <<'EOT'
+SYSTEM_FORCED_BODY
+EOT
+(
+ cd "$base/project"
+ "$MCPU_CPP" --no-config -MM -isystem sys -include system-forced.h \
+ src/cwd-main.c > "$base/mm.out"
+)
+grep -E '(^|[[:space:]])src/cwd-main\.c([[:space:]]|$)' "$base/mm.out" >/dev/null
+! grep -F "$base/project/sys/system-forced.h" "$base/mm.out" >/dev/null
+
+# Public help advertises both implemented forced-file options.
+"$MCPU_CPP" --help > "$base/help"
+grep '^ -imacros FILE ' "$base/help" >/dev/null
+grep '^ -include FILE ' "$base/help" >/dev/null
+
+# Both option families require an argument.
+for opt in -include -imacros; do
+ if "$MCPU_CPP" --no-config "$opt" > /dev/null 2>"$base/missing-arg.err"; then
+ echo "$opt accepted a missing argument" >&2
+ exit 1
+ fi
+ grep -- "argument missing after '$opt'" "$base/missing-arg.err" >/dev/null
+done
+
+# Missing forced files are preprocessing errors for both option families.
+for opt in -include -imacros; do
+ if (
+ cd "$base/project"
+ "$MCPU_CPP" --no-config "$opt" no-such-forced-file.h src/cwd-main.c \
+ -o "$base/missing.out" 2>"$base/missing.err"
+ ); then
+ echo "$opt accepted a missing forced file" >&2
+ exit 1
+ fi
+ grep -- "$opt cannot find file 'no-such-forced-file.h'" "$base/missing.err" >/dev/null
+done
+
+# The same forced-file pipeline also applies when the primary source is stdin.
+printf 'DD_VALUE\n' | (
+ cd "$base/project"
+ "$MCPU_CPP" --no-config -imacros only-macros.h -
+) > "$base/stdin.out"
+grep '^77$' "$base/stdin.out" >/dev/null
diff --git a/tests/t0056-dependency-targets.sh b/tests/t0056-dependency-targets.sh
new file mode 100755
index 0000000..ea5f4b9
--- /dev/null
+++ b/tests/t0056-dependency-targets.sh
@@ -0,0 +1,103 @@
+#!/bin/sh
+set -eu
+
+base="${TMPDIR-/tmp}/mcpu-cpp-dependency-targets-$$"
+mkdir -p "$base"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+cd "$base"
+
+cat > main.c <<'EOT'
+#include "local.h"
+int main_value;
+EOT
+
+cat > local.h <<'EOT'
+#define LOCAL_VALUE 1
+EOT
+
+# The established default target remains make-quoted and uses the configured
+# object suffix only when no explicit -MT/-MQ target is present.
+"$MCPU_CPP" --no-config -M main.c > default.d
+grep '^main\.o: main\.c \./local\.h$' default.d >/dev/null
+
+"$MCPU_CPP" --no-config --object-suffix .obj -M main.c > suffix.d
+grep '^main\.obj: main\.c \./local\.h$' suffix.d >/dev/null
+
+# -MT replaces the default target and is emitted exactly as supplied.
+"$MCPU_CPP" --no-config -M -MT build/main.o main.c > mt.d
+grep '^build/main\.o: main\.c \./local\.h$' mt.d >/dev/null
+
+"$MCPU_CPP" --no-config -M -MTattached.o main.c > mt-attached.d
+grep '^attached\.o: main\.c \./local\.h$' mt-attached.d >/dev/null
+
+"$MCPU_CPP" --no-config -M -MT 'raw one raw#two$' main.c > mt-raw.d
+grep '^raw one raw#two\$: main\.c \./local\.h$' mt-raw.d >/dev/null
+
+# -MQ applies Make quoting to the target.
+"$MCPU_CPP" --no-config -M -MQ '$(OBJDIR)/main.o' main.c > mq.d
+grep '^\$\$(OBJDIR)/main\.o: main\.c \./local\.h$' mq.d >/dev/null
+
+"$MCPU_CPP" --no-config -M '-MQ$(OBJDIR)/attached.o' main.c > mq-attached.d
+grep '^\$\$(OBJDIR)/attached\.o: main\.c \./local\.h$' mq-attached.d >/dev/null
+
+"$MCPU_CPP" --no-config -M -MQ 'dir name/#x$.o' main.c > mq-special.d
+grep '^dir\\ name/\\#x\$\$\.o: main\.c \./local\.h$' mq-special.d >/dev/null
+
+"$MCPU_CPP" --no-config -M -MQ 'path\name.o' main.c > mq-backslash.d
+grep -F 'path\name.o: main.c ./local.h' mq-backslash.d >/dev/null
+
+# Multiple targets produce one rule. As in GNU CPP, all -MT targets precede
+# all -MQ targets; command-line order is preserved inside each class.
+"$MCPU_CPP" --no-config -M \
+ -MQ '$(DIR)/quoted1.o' \
+ -MT raw1.o \
+ -MQ 'quoted two.o' \
+ -MT 'raw2.o raw3.o' \
+ main.c > multiple.d
+grep '^raw1\.o raw2\.o raw3\.o \$\$(DIR)/quoted1\.o quoted\\ two\.o: main\.c \./local\.h$' multiple.d >/dev/null
+
+# Explicit targets override the automatic target even when an object suffix was
+# requested.
+"$MCPU_CPP" --no-config --object-suffix .obj -M -MT explicit.target main.c > explicit.d
+grep '^explicit\.target: main\.c \./local\.h$' explicit.d >/dev/null
+
+# -MT/-MQ work in side-effect dependency modes and with -MF.
+"$MCPU_CPP" --no-config -MD -MF md.d -MT md-target.o main.c > md.i
+grep '^md-target\.o: main\.c \./local\.h$' md.d >/dev/null
+grep '^int main_value;$' md.i >/dev/null
+
+"$MCPU_CPP" --no-config -MMD -MF mmd.d -MQ '$(OUT)/mmd.o' main.c > mmd.i
+grep '^\$\$(OUT)/mmd\.o: main\.c \./local\.h$' mmd.d >/dev/null
+grep '^int main_value;$' mmd.i >/dev/null
+
+# Both options require an active dependency-generation mode.
+if "$MCPU_CPP" --no-config -MT orphan.o main.c > orphan.out 2> orphan.err; then
+ echo "-MT without dependency generation unexpectedly succeeded" >&2
+ exit 1
+fi
+grep -- '-MT/-MQ require one of -M, -MM, -MD or -MMD' orphan.err >/dev/null
+
+if "$MCPU_CPP" --no-config -MQ orphan.o main.c > orphan-q.out 2> orphan-q.err; then
+ echo "-MQ without dependency generation unexpectedly succeeded" >&2
+ exit 1
+fi
+grep -- '-MT/-MQ require one of -M, -MM, -MD or -MMD' orphan-q.err >/dev/null
+
+# Missing arguments remain command-line errors.
+if "$MCPU_CPP" --no-config -M -MT > missing-mt.out 2> missing-mt.err; then
+ echo "-MT without an argument unexpectedly succeeded" >&2
+ exit 1
+fi
+grep "argument missing after '-MT'" missing-mt.err >/dev/null
+
+if "$MCPU_CPP" --no-config -M -MQ > missing-mq.out 2> missing-mq.err; then
+ echo "-MQ without an argument unexpectedly succeeded" >&2
+ exit 1
+fi
+grep "argument missing after '-MQ'" missing-mq.err >/dev/null
+
+# Public help documents both options.
+"$MCPU_CPP" --help > help.out
+grep '^ -MT TARGET set unquoted make dependency target$' help.out >/dev/null
+grep '^ -MQ TARGET set make-quoted dependency target$' help.out >/dev/null
diff --git a/tests/t0057-missing-generated-dependencies.sh b/tests/t0057-missing-generated-dependencies.sh
new file mode 100755
index 0000000..7ad5001
--- /dev/null
+++ b/tests/t0057-missing-generated-dependencies.sh
@@ -0,0 +1,166 @@
+#!/bin/sh
+set -eu
+
+base="${TMPDIR-/tmp}/mcpu-cpp-missing-generated-deps-$$"
+mkdir -p "$base/user" "$base/sys"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+cd "$base"
+
+cat > main.c <<'EOT'
+#include "generated-user.h"
+#include <generated-system.h>
+#define GENERATED_MACRO_HEADER "generated-macro.h"
+#include GENERATED_MACRO_HEADER
+EOT
+
+# Without -MG, the first unresolved include remains a hard error.
+if "$MCPU_CPP" --no-config -M main.c > no-mg.out 2> no-mg.err; then
+ echo "missing include unexpectedly succeeded without -MG" >&2
+ exit 1
+fi
+grep "cannot find include file 'generated-user.h'" no-mg.err >/dev/null
+
+# -M -MG records unresolved operands exactly as written and does not diagnose
+# them as missing files. Macro-expanded include operands use their expanded
+# filename.
+"$MCPU_CPP" --no-config -M -MG main.c > M.d 2> M.err
+test ! -s M.err
+grep '^main\.o:' M.d >/dev/null
+for f in main.c generated-user.h generated-system.h generated-macro.h
+do
+ grep -F "$f" M.d >/dev/null || {
+ echo "-M -MG omitted unresolved dependency: $f" >&2
+ exit 1
+ }
+done
+
+# -MM retains unresolved quoted/user includes but filters unresolved angle
+# includes as system-class dependencies.
+"$MCPU_CPP" --no-config -MM -MG main.c > MM.d
+grep -F 'generated-user.h' MM.d >/dev/null
+grep -F 'generated-macro.h' MM.d >/dev/null
+if grep -F 'generated-system.h' MM.d >/dev/null; then
+ echo "-MM -MG kept unresolved angle/system dependency" >&2
+ exit 1
+fi
+
+# Missing dependencies reached from a real system header inherit system
+# context, even when the nested directive uses quotes. User-header quoted
+# descendants remain user dependencies.
+cat > sys/system-parent.h <<'EOT'
+#include "generated-from-system.h"
+EOT
+cat > user/user-parent.h <<'EOT'
+#include "generated-from-user.h"
+EOT
+cat > nested.c <<'EOT'
+#include <system-parent.h>
+#include <user-parent.h>
+EOT
+
+"$MCPU_CPP" --no-config -MM -MG -I user -isystem sys nested.c > nested-MM.d
+grep -F 'user/user-parent.h' nested-MM.d >/dev/null
+grep -F 'generated-from-user.h' nested-MM.d >/dev/null
+if grep -F 'sys/system-parent.h' nested-MM.d >/dev/null ||
+ grep -F 'generated-from-system.h' nested-MM.d >/dev/null; then
+ echo "-MM -MG kept system-context dependency" >&2
+ exit 1
+fi
+
+"$MCPU_CPP" --no-config -M -MG -I user -isystem sys nested.c > nested-M.d
+grep -F 'sys/system-parent.h' nested-M.d >/dev/null
+grep -F 'generated-from-system.h' nested-M.d >/dev/null
+
+# Unresolved dependency identity is textual and separate from physical
+# st_dev/st_ino identity. shadow.h exists in CWD, but <shadow.h> is not found
+# through the angle search chain under --no-config; a later quoted ./shadow.h
+# resolves that physical file. Both registry entries must survive under -M.
+cat > shadow.h <<'EOT'
+#define SHADOW_VALUE 1
+EOT
+cat > identity.c <<'EOT'
+#include <shadow.h>
+#include "./shadow.h"
+EOT
+"$MCPU_CPP" --no-config -M -MG identity.c > identity-M.d
+grep '^identity\.o: identity\.c shadow\.h \./\./shadow\.h$' identity-M.d >/dev/null || {
+ echo "unresolved and physical dependency identities were incorrectly merged" >&2
+ cat identity-M.d >&2
+ exit 1
+}
+
+# In -MM the unresolved angle entry is filtered, while the resolved quoted
+# physical dependency remains.
+"$MCPU_CPP" --no-config -MM -MG identity.c > identity-MM.d
+grep -F '././shadow.h' identity-MM.d >/dev/null
+case "$(cat identity-MM.d)" in
+ *' shadow.h '*)
+ echo "-MM kept the unresolved angle dependency" >&2
+ exit 1
+ ;;
+esac
+
+# Repeated unresolved operands are deduplicated textually. Their first
+# classification is retained, matching GNU CPP: angle-first stays system-only,
+# while quote-first stays visible to -MM.
+cat > first-system.c <<'EOT'
+#include <same-missing.h>
+#include "same-missing.h"
+EOT
+"$MCPU_CPP" --no-config -MM -MG first-system.c > first-system.d
+if grep -F 'same-missing.h' first-system.d >/dev/null; then
+ echo "angle-first unresolved dependency lost its first classification" >&2
+ exit 1
+fi
+
+cat > first-user.c <<'EOT'
+#include "same-missing.h"
+#include <same-missing.h>
+EOT
+"$MCPU_CPP" --no-config -MM -MG first-user.c > first-user.d
+grep -F 'same-missing.h' first-user.d >/dev/null
+
+# Different unresolved spellings are distinct because no physical identity is
+# available to canonicalize them.
+cat > spelling.c <<'EOT'
+#include "generated.h"
+#include "./generated.h"
+EOT
+"$MCPU_CPP" --no-config -M -MG spelling.c > spelling.d
+grep -F ' generated.h' spelling.d >/dev/null
+grep -F ' ./generated.h' spelling.d >/dev/null
+
+# -MG also applies to missing command-line forced files. They enter the same
+# unresolved registry as user dependencies and preserve the command-line
+# operand; no synthetic path is prepended.
+: > forced.c
+"$MCPU_CPP" --no-config -M -MG \
+ -imacros missing-macros.h -include missing-include.h forced.c > forced-M.d
+grep -F 'missing-macros.h' forced-M.d >/dev/null
+grep -F 'missing-include.h' forced-M.d >/dev/null
+
+"$MCPU_CPP" --no-config -MM -MG \
+ -imacros missing-macros.h -include missing-include.h forced.c > forced-MM.d
+grep -F 'missing-macros.h' forced-MM.d >/dev/null
+grep -F 'missing-include.h' forced-MM.d >/dev/null
+
+# -MG is dependency-only: GNU permits it with -M/-MM, not side-effect -MD/-MMD
+# and not without dependency generation.
+for opts in '-MG' '-MD -MG' '-MMD -MG'
+do
+ if "$MCPU_CPP" --no-config $opts forced.c > invalid.out 2> invalid.err; then
+ echo "$opts unexpectedly accepted" >&2
+ exit 1
+ fi
+ grep -- '-MG may only be used with -M or -MM' invalid.err >/dev/null
+done
+
+# Explicit dependency targets continue to compose with -MG.
+"$MCPU_CPP" --no-config -M -MG -MT generated-target.o main.c > target.d
+grep '^generated-target\.o:' target.d >/dev/null
+grep -F 'generated-user.h' target.d >/dev/null
+
+# Public help exposes the implemented option.
+"$MCPU_CPP" --help > help.out
+grep '^ -MG treat missing headers as generated dependencies$' help.out >/dev/null
diff --git a/tests/t0058-variadic-macros.sh b/tests/t0058-variadic-macros.sh
new file mode 100755
index 0000000..584e980
--- /dev/null
+++ b/tests/t0058-variadic-macros.sh
@@ -0,0 +1,135 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-variadic-$$"
+mkdir -p "$base"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+cat > "$base/main.c" <<'EOT'
+#define A 7
+#define SUM(X, Y) X + Y
+#define V(...) <__VA_ARGS__>
+#define F(first, ...) first | __VA_ARGS__
+#define STRV(...) #__VA_ARGS__
+#define LEFT(...) pre ## __VA_ARGS__
+#define RIGHT(...) __VA_ARGS__ ## post
+#define CALL(FN, ...) FN(__VA_ARGS__)
+#define WRAP(...) V(__VA_ARGS__)
+#define ID(X) X
+#define РУССКИЙ(первый, ...) первый : __VA_ARGS__
+
+V(alpha, beta, gamma)
+V()
+F(one, two, three)
+F(one)
+F(one,)
+F(A, SUM(1, 2))
+STRV(A, b + c)
+LEFT(fix)
+LEFT()
+RIGHT(fix)
+RIGHT()
+CALL(ID, A)
+WRAP(A, SUM(3, 4))
+РУССКИЙ(один, два, три)
+EOT
+
+"$MCPU_CPP" --no-config "$base/main.c" -o "$base/out"
+
+grep -Fx '<alpha, beta, gamma>' "$base/out" >/dev/null
+grep -Fx '<>' "$base/out" >/dev/null
+grep -Fx 'one | two, three' "$base/out" >/dev/null
+# Both an omitted variadic tail and an explicitly empty one expand to no tokens.
+test "$(grep -Fxc 'one | ' "$base/out")" -eq 2
+grep -Fx '7 | 1 + 2' "$base/out" >/dev/null
+grep -Fx '"A, b + c"' "$base/out" >/dev/null
+grep -Fx 'prefix' "$base/out" >/dev/null
+grep -Fx 'pre' "$base/out" >/dev/null
+grep -Fx 'fixpost' "$base/out" >/dev/null
+grep -Fx 'post' "$base/out" >/dev/null
+grep -Fx '7' "$base/out" >/dev/null
+grep -Fx '<7, 3 + 4>' "$base/out" >/dev/null
+grep -Fx 'один : два, три' "$base/out" >/dev/null
+
+# The variable argument is preserved as such when macro definitions are dumped.
+"$MCPU_CPP" --no-config -dM "$base/main.c" > "$base/dump"
+grep -Fx '#define V(...) <__VA_ARGS__>' "$base/dump" >/dev/null
+grep -Fx '#define F(first,...) first | __VA_ARGS__' "$base/dump" >/dev/null
+grep -Fx '#define STRV(...) #__VA_ARGS__' "$base/dump" >/dev/null
+grep -Fx '#define РУССКИЙ(первый,...) первый : __VA_ARGS__' "$base/dump" >/dev/null
+
+
+# Variadic-ness is part of a macro definition. Repeating the same variadic
+# definition is harmless, while changing a non-variadic definition into a
+# variadic one is a real redefinition and therefore participates in -Werror.
+cat > "$base/redefine-same.c" <<'EOT'
+#define SAME(x, ...) x | __VA_ARGS__
+#define SAME(x, ...) x | __VA_ARGS__
+SAME(1, 2)
+EOT
+"$MCPU_CPP" --no-config "$base/redefine-same.c" \
+ -o "$base/redefine-same.out" 2>"$base/redefine-same.err"
+test ! -s "$base/redefine-same.err"
+grep -Fx '1 | 2' "$base/redefine-same.out" >/dev/null
+
+cat > "$base/redefine-kind.c" <<'EOT'
+#define CHANGE(x) x
+#define CHANGE(x, ...) x | __VA_ARGS__
+EOT
+if "$MCPU_CPP" --no-config -Werror "$base/redefine-kind.c" \
+ -o "$base/redefine-kind.out" 2>"$base/redefine-kind.err"; then
+ echo 'variadic/non-variadic redefinition escaped -Werror' >&2
+ exit 1
+fi
+grep "error: macro 'CHANGE' redefined" "$base/redefine-kind.err" >/dev/null
+
+# A variadic macro may have fixed parameters, but those fixed parameters are
+# still required. An empty spelling counts as an argument, just as it does
+# for ordinary function-like macros.
+cat > "$base/few.c" <<'EOT'
+#define NEEDS_TWO(a, b, ...) a + b + __VA_ARGS__
+NEEDS_TWO(1)
+EOT
+if "$MCPU_CPP" --no-config "$base/few.c" -o "$base/few.out" 2>"$base/few.err"; then
+ echo 'variadic macro accepted too few fixed arguments' >&2
+ exit 1
+fi
+grep "macro 'NEEDS_TWO' used with too few arguments" "$base/few.err" >/dev/null
+
+# The old GNU named variadic form is deliberately outside the 0.0.46
+# contract. Only the C99-style ... + __VA_ARGS__ form is accepted.
+cat > "$base/named.c" <<'EOT'
+#define OLD(args...) args
+EOT
+if "$MCPU_CPP" --no-config "$base/named.c" -o "$base/named.out" 2>"$base/named.err"; then
+ echo 'GNU named variadic macro form unexpectedly accepted' >&2
+ exit 1
+fi
+grep 'badly punctuated parameter list in #define' "$base/named.err" >/dev/null
+
+cat > "$base/malformed.c" <<'EOT'
+#define BAD(..., x) x
+EOT
+if "$MCPU_CPP" --no-config "$base/malformed.c" -o "$base/malformed.out" 2>"$base/malformed.err"; then
+ echo 'parameters after ... unexpectedly accepted' >&2
+ exit 1
+fi
+grep 'badly punctuated parameter list in #define' "$base/malformed.err" >/dev/null
+
+cat > "$base/reserved.c" <<'EOT'
+#define BAD(__VA_ARGS__) __VA_ARGS__
+EOT
+if "$MCPU_CPP" --no-config "$base/reserved.c" -o "$base/reserved.out" 2>"$base/reserved.err"; then
+ echo '__VA_ARGS__ unexpectedly accepted as an ordinary parameter name' >&2
+ exit 1
+fi
+grep "'__VA_ARGS__' cannot be used as a macro parameter name" "$base/reserved.err" >/dev/null
+
+# A replacement using #__VA_ARGS__ is valid only for a variadic macro.
+cat > "$base/not-variadic.c" <<'EOT'
+#define BAD(x) #__VA_ARGS__
+EOT
+if "$MCPU_CPP" --no-config "$base/not-variadic.c" -o "$base/not-variadic.out" 2>"$base/not-variadic.err"; then
+ echo '#__VA_ARGS__ unexpectedly accepted in a non-variadic macro' >&2
+ exit 1
+fi
+grep "'#' operator should be followed by a macro argument name" "$base/not-variadic.err" >/dev/null
diff --git a/tests/t0059-va-opt.sh b/tests/t0059-va-opt.sh
new file mode 100755
index 0000000..052df6c
--- /dev/null
+++ b/tests/t0059-va-opt.sh
@@ -0,0 +1,150 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-va-opt-$$"
+mkdir -p "$base"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+cat > "$base/main.c" <<'EOT'
+#define EMPTY
+#define X 123
+#define xy RESCANNED
+#define HAS(...) __VA_OPT__(yes)
+#define COMMA(format, ...) call(format __VA_OPT__(,) __VA_ARGS__)
+#define STR(...) #__VA_OPT__(__VA_ARGS__)
+#define STRFIX(a, ...) #__VA_OPT__(a __VA_ARGS__)
+#define STRPASTE(a, ...) #__VA_OPT__(a ## z)
+#define LEFT(...) pre ## __VA_OPT__(__VA_ARGS__)
+#define RIGHT(...) __VA_OPT__(__VA_ARGS__) ## post
+#define INNER(a, ...) __VA_OPT__(a ## y)
+#define BALANCED(...) __VA_OPT__((a, (b, c)))
+#define EMPTY_LEFT(...) x ## __VA_OPT__()
+#define EMPTY_RIGHT(...) __VA_OPT__() ## y
+#define TEXT(...) "__VA_OPT__(not syntax)" __VA_OPT__(ok)
+
+[HAS()]
+[HAS(EMPTY)]
+[HAS(token)]
+COMMA("zero")
+COMMA("two", X, 7)
+STR()
+STR(X)
+STRFIX(X, y)
+STRPASTE(X, y)
+LEFT()
+LEFT(X)
+RIGHT()
+RIGHT(X)
+INNER(x, token)
+BALANCED(token)
+EMPTY_LEFT(token)
+EMPTY_RIGHT(token)
+TEXT()
+TEXT(token)
+EOT
+
+"$MCPU_CPP" --no-config "$base/main.c" -o "$base/out"
+sed '/^#/d; /^[[:space:]]*$/d; s/[[:space:]]//g' "$base/out" > "$base/norm"
+
+cat > "$base/expected" <<'EOT'
+[]
+[]
+[yes]
+call("zero")
+call("two",123,7)
+""
+"123"
+"123y"
+"Xz"
+pre
+pre123
+post
+123post
+RESCANNED
+(a,(b,c))
+x
+y
+"__VA_OPT__(notsyntax)"
+"__VA_OPT__(notsyntax)"ok
+EOT
+
+diff -u "$base/expected" "$base/norm"
+
+# __VA_OPT__ remains visible in macro dumps as part of the replacement list.
+"$MCPU_CPP" --no-config -dM "$base/main.c" > "$base/dump"
+grep -Fx '#define HAS(...) __VA_OPT__(yes)' "$base/dump" >/dev/null
+grep -Fx '#define COMMA(format,...) call(format __VA_OPT__(,) __VA_ARGS__)' "$base/dump" >/dev/null
+grep -Fx '#define STR(...) #__VA_OPT__(__VA_ARGS__)' "$base/dump" >/dev/null
+
+# __VA_OPT__ is structural syntax, not a general identifier-like extension.
+cat > "$base/nonvariadic.c" <<'EOT'
+#define BAD(x) __VA_OPT__(x)
+EOT
+if "$MCPU_CPP" --no-config "$base/nonvariadic.c" -o "$base/nonvariadic.out" 2>"$base/nonvariadic.err"; then
+ echo '__VA_OPT__ unexpectedly accepted in a non-variadic macro' >&2
+ exit 1
+fi
+grep "'__VA_OPT__' may appear only in a variadic macro replacement list" "$base/nonvariadic.err" >/dev/null
+
+cat > "$base/object.c" <<'EOT'
+#define BAD __VA_OPT__(x)
+EOT
+if "$MCPU_CPP" --no-config "$base/object.c" -o "$base/object.out" 2>"$base/object.err"; then
+ echo '__VA_OPT__ unexpectedly accepted in an object-like macro' >&2
+ exit 1
+fi
+grep "'__VA_OPT__' may appear only in a variadic macro replacement list" "$base/object.err" >/dev/null
+
+cat > "$base/nested.c" <<'EOT'
+#define BAD(...) __VA_OPT__(a __VA_OPT__(b))
+EOT
+if "$MCPU_CPP" --no-config "$base/nested.c" -o "$base/nested.out" 2>"$base/nested.err"; then
+ echo 'nested __VA_OPT__ unexpectedly accepted' >&2
+ exit 1
+fi
+grep "'__VA_OPT__' may not appear inside another '__VA_OPT__'" "$base/nested.err" >/dev/null
+
+cat > "$base/no-open.c" <<'EOT'
+#define BAD(...) __VA_OPT__ x
+EOT
+if "$MCPU_CPP" --no-config "$base/no-open.c" -o "$base/no-open.out" 2>"$base/no-open.err"; then
+ echo '__VA_OPT__ without opening parenthesis unexpectedly accepted' >&2
+ exit 1
+fi
+grep "'__VA_OPT__' must be followed by '('" "$base/no-open.err" >/dev/null
+
+cat > "$base/unterminated.c" <<'EOT'
+#define BAD(...) __VA_OPT__((x)
+EOT
+if "$MCPU_CPP" --no-config "$base/unterminated.c" -o "$base/unterminated.out" 2>"$base/unterminated.err"; then
+ echo 'unterminated __VA_OPT__ unexpectedly accepted' >&2
+ exit 1
+fi
+grep "unterminated '__VA_OPT__'" "$base/unterminated.err" >/dev/null
+
+cat > "$base/paste-first.c" <<'EOT'
+#define BAD(...) __VA_OPT__(## x)
+EOT
+if "$MCPU_CPP" --no-config "$base/paste-first.c" -o "$base/paste-first.out" 2>"$base/paste-first.err"; then
+ echo 'leading ## inside __VA_OPT__ unexpectedly accepted' >&2
+ exit 1
+fi
+grep "'##' cannot appear at the beginning of '__VA_OPT__'" "$base/paste-first.err" >/dev/null
+
+cat > "$base/paste-last.c" <<'EOT'
+#define BAD(...) __VA_OPT__(x ##)
+EOT
+if "$MCPU_CPP" --no-config "$base/paste-last.c" -o "$base/paste-last.out" 2>"$base/paste-last.err"; then
+ echo 'trailing ## inside __VA_OPT__ unexpectedly accepted' >&2
+ exit 1
+fi
+grep "'##' cannot appear at the end of '__VA_OPT__'" "$base/paste-last.err" >/dev/null
+
+# The historical GNU comma-swallow extension remains deliberately absent.
+# __VA_OPT__ is the supported way to make a separator conditional.
+cat > "$base/no-gnu-comma.c" <<'EOT'
+#define OLD(format, ...) call(format, ## __VA_ARGS__)
+OLD("x")
+EOT
+"$MCPU_CPP" --no-config "$base/no-gnu-comma.c" -o "$base/no-gnu-comma.out" 2>"$base/no-gnu-comma.err"
+sed '/^#/d; /^[[:space:]]*$/d; s/[[:space:]]//g' "$base/no-gnu-comma.out" > "$base/no-gnu-comma.norm"
+grep -Fx 'call("x",)' "$base/no-gnu-comma.norm" >/dev/null
diff --git a/tests/t0060-macro-whitespace.sh b/tests/t0060-macro-whitespace.sh
new file mode 100755
index 0000000..8f870ba
--- /dev/null
+++ b/tests/t0060-macro-whitespace.sh
@@ -0,0 +1,72 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-macro-whitespace-$$"
+mkdir -p "$base"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+cat > "$base/main.c" <<'EOT'
+#define MULTI(fmt, ...) \
+ do \
+ { \
+ output(fmt __VA_OPT__(,) __VA_ARGS__); \
+ done(); \
+ } \
+ while( 0 )
+
+#define TEXT "left right"
+#define OPS + + - > < <
+#define ID(x) x
+#define V(...) __VA_OPT__(__VA_ARGS__)
+
+void
+f( int x )
+{
+ MULTI("x=%d", x);
+ MULTI("hello");
+ ID(alpha + beta);
+ V(gamma + delta);
+ const char *s = TEXT;
+ OPS
+}
+EOT
+
+"$MCPU_CPP" --no-config "$base/main.c" -o "$base/out"
+sed '/^#/d; /^[[:space:]]*$/d' "$base/out" > "$base/norm"
+
+cat > "$base/expected" <<'EOT'
+void
+f( int x )
+{
+ do { output("x=%d" , x); done(); } while( 0 );
+ do { output("hello" ); done(); } while( 0 );
+ alpha + beta;
+ gamma + delta;
+ const char *s = "left right";
+ + + - > < <
+}
+EOT
+
+diff -u "$base/expected" "$base/norm"
+
+# Macro dumps use the same normalized replacement-list whitespace.
+"$MCPU_CPP" --no-config -dM "$base/main.c" > "$base/dump"
+grep -Fx '#define MULTI(fmt,...) do { output(fmt __VA_OPT__(,) __VA_ARGS__); done(); } while( 0 )' "$base/dump" >/dev/null
+grep -Fx '#define TEXT "left right"' "$base/dump" >/dev/null
+grep -Fx '#define OPS + + - > < <' "$base/dump" >/dev/null
+grep -Fx '#define ID(x) x' "$base/dump" >/dev/null
+grep -Fx '#define V(...) __VA_OPT__(__VA_ARGS__)' "$base/dump" >/dev/null
+
+# Ordinary source text that is not replacement-list formatting stays untouched.
+cat > "$base/plain.c" <<'EOT'
+int plain = 1;
+#define ID(x) x
+#define V(...) __VA_OPT__(__VA_ARGS__)
+ID(a + b)
+EOT
+"$MCPU_CPP" --no-config "$base/plain.c" -o "$base/plain.out"
+sed '/^#/d; /^[[:space:]]*$/d' "$base/plain.out" > "$base/plain.norm"
+cat > "$base/plain.expected" <<'EOT'
+int plain = 1;
+a + b
+EOT
+diff -u "$base/plain.expected" "$base/plain.norm"
diff --git a/tests/t0061-output-line-compaction.sh b/tests/t0061-output-line-compaction.sh
new file mode 100755
index 0000000..e40b1ee
--- /dev/null
+++ b/tests/t0061-output-line-compaction.sh
@@ -0,0 +1,131 @@
+#!/bin/sh
+set -eu
+base="${TMPDIR-/tmp}/mcpu-cpp-output-line-compaction-$$"
+mkdir -p "$base"
+trap 'rm -rf "$base"' EXIT HUP INT TERM
+
+# Seven invisible lines stay as seven ordinary newlines. The next visible
+# source line is line 9, and no corrective marker is needed.
+cat > "$base/gap7.c" <<'EOT'
+int a;
+// one
+#define A 1
+#if 0
+hidden
+#endif
+/* six */
+
+int b;
+EOT
+"$MCPU_CPP" --no-config "$base/gap7.c" -o "$base/gap7.out"
+if grep -F "# 9 \"$base/gap7.c\"" "$base/gap7.out" >/dev/null; then
+ echo "unexpected line marker for seven-line invisible gap" >&2
+ exit 1
+fi
+gap7_count=`awk '
+ $0 == "int a;" { inside = 1; next }
+ $0 == "int b;" { print count; exit }
+ inside { ++count }
+' "$base/gap7.out"`
+test "$gap7_count" -eq 7
+
+# Eight invisible lines cross the GNU CPP threshold. They disappear from the
+# byte stream and are represented by a line marker for the next visible line.
+cat > "$base/gap8.c" <<'EOT'
+int a;
+// one
+#define A 1
+#if 0
+hidden
+#endif
+/* six */
+// seven
+
+int b;
+EOT
+"$MCPU_CPP" --no-config "$base/gap8.c" -o "$base/gap8.out"
+grep -F "# 10 \"$base/gap8.c\"" "$base/gap8.out" >/dev/null
+awk '
+ /^# 10 "/ { getline; if( $0 == "int b;" ) ok = 1 }
+ END { exit ok ? 0 : 1 }
+' "$base/gap8.out"
+
+# A completely invisible header has no synthetic end-of-header marker. Only
+# the structural enter/return markers survive. A long invisible gap in the
+# parent is then represented by the parent's next visible source position.
+cat > "$base/silent.h" <<'EOT'
+#ifndef SILENT_H
+#define SILENT_H 1
+#define H1 1
+#define H2 2
+#define H3 3
+#define H4 4
+#define H5 5
+#define H6 6
+#define H7 7
+#define H8 8
+#define H9 9
+#endif
+EOT
+cat > "$base/include.c" <<'EOT'
+// leading comment
+#include "silent.h"
+// one
+#define P1 1
+#if 0
+hidden
+#endif
+// six
+// seven
+
+int visible;
+EOT
+"$MCPU_CPP" --no-config -I"$base" "$base/include.c" -o "$base/include.out"
+grep -F "# 1 \"$base/silent.h\" 1" "$base/include.out" >/dev/null
+grep -F "# 3 \"$base/include.c\" 2" "$base/include.out" >/dev/null
+grep -F "# 11 \"$base/include.c\"" "$base/include.out" >/dev/null
+if grep -E "^# (2|3|4|5|6|7|8|9|10|11|12) \"$base/silent.h\"" "$base/include.out" >/dev/null; then
+ echo "silent header acquired a synthetic progress marker" >&2
+ exit 1
+fi
+awk '
+ /\/silent\.h" 1$/ { getline; if( $0 ~ /\/include\.c" 2$/ ) adjacent = 1 }
+ /^# 11 "/ { getline; if( $0 == "int visible;" ) visible = 1 }
+ END { exit adjacent && visible ? 0 : 1 }
+' "$base/include.out"
+
+# The threshold is based on source position, not on why the lines are
+# invisible. A long run of comments alone is compacted in exactly the same
+# way as directives or inactive conditional text.
+cat > "$base/comments.c" <<'EOT'
+int before;
+// 1
+// 2
+// 3
+// 4
+// 5
+// 6
+// 7
+// 8
+int after;
+EOT
+"$MCPU_CPP" --no-config "$base/comments.c" -o "$base/comments.out"
+grep -F "# 10 \"$base/comments.c\"" "$base/comments.out" >/dev/null
+
+# __LINE__ observes source coordinates, not the compacted output layout.
+cat > "$base/line.c" <<'EOT'
+int a = __LINE__;
+// 1
+// 2
+// 3
+// 4
+// 5
+// 6
+// 7
+// 8
+int b = __LINE__;
+EOT
+"$MCPU_CPP" --no-config "$base/line.c" -o "$base/line.out"
+grep -Fx 'int a = 1;' "$base/line.out" >/dev/null
+grep -Fx 'int b = 10;' "$base/line.out" >/dev/null
+grep -F "# 10 \"$base/line.c\"" "$base/line.out" >/dev/null