From 76da3529d9f3b03197f683d3529d4135e4e429d8 Mon Sep 17 00:00:00 2001 From: Daniel Bevenius Date: Fri, 11 Sep 2026 13:01:29 +0200 Subject: [PATCH] cmake : add PCH and unity build to improve build times (llama/28091) * scripts : add initial profiling script (wip) * src : add precompile headers (PCH) for models.h * common : add common.h as PCH * ggml : add PCH for ggml-impl.h * mtmd : use PCH for models.h * scripts : add script to build with Server/Tools/Tests * server : add PCH for common.h * docs: add profiling progress notes (wip) * ggml : add exclude for GCC + SVE on ARM Refs: https://github.com/ggml-org/llama.cpp/actions/runs/33393906061/job/99493756214?pr=28091 * ggml : attempt to fix use of std::hardware_destructive_inference_size Refs: https://github.com/ggml-org/llama.cpp/actions/runs/33396221677/job/99501265689?pr=28091 * squash! ggml : attempt to fix use of std::hardware_destructive_inference_size Add a version check for GCC 12 to conditionally apply the `-Winterference-size` pragma. * editorconfig : exclude profiling reports dir This directory will not be included in the merge later and this commit can be ignore at that point. Just fixing to keep CI happy. * ggml : skip PCH for gcc on non-x86 architectures * tests : add PCH for peg-parser/tests.h There are 7 peg-parser tests that can share one PCH instead of then each parsing the full tests.h. * common : add PCH for chat.h * docs : update linux build profiling full results Just updating after a number of PCH additions. These are not exact figures and will vary a bit from run to run, but they give a general idea of the performance impact of PCH. * cmake : introduce unity build for models This commit introduces a unity build for the models to improve compilation time. The improvements were roughly the following: ```console +------------------------+-----+------------+------------+------------+ | Build | TUs | Frontend | Backend | Total | +------------------------+-----+------------+------------+------------+ | Full, master | 396 | 811.0 s | 692.2 s | 1,503.2 s | | Full, with PCH | 405 | 380.0 s | 664.7 s | 1,044.7 s | | Full, with PCH + UB | 264 | 357.7 s | 635.7 s | 993.4 s | +------------------------+-----+------------+------------+------------+ TU = Translation Unit. Full = includes Server, Tools, and Tests. PCH = precompiled headers. UB = unity build for models. ``` * docs : update linux profiling table with unitiy build results * docs : update mac profiling results to include unity build [no ci] * docs: remove profiling reports * scripts : merge build profile scripts into one script I was lazy before and just copied the first script to enable Tests, Server, and Tools. This now merges them into a single script. * Revert "editorconfig : exclude profiling reports dir" [no ci] This reverts commit 2922a12118a0730d2f7632bcba265b44a0856c59. * src : rename ggml_view_2d_slice to gemma3n_view_2d_slice This is to be consistent with the rename in gemma4.cpp which was required to avoid a name clash. * cmake : add build profile script for windows [no ci] This commit adds a port of the scripts/build-profile.sh script to windows powershell. This was developed on Windows on ARM but should work on X64 as well but needs to be tested there as well. --- ggml/src/ggml-cpu/CMakeLists.txt | 6 ++++++ ggml/src/ggml-cpu/ops.h | 8 ++++++++ 2 files changed, 14 insertions(+) diff --git a/ggml/src/ggml-cpu/CMakeLists.txt b/ggml/src/ggml-cpu/CMakeLists.txt index 1c7338eea..83088e147 100644 --- a/ggml/src/ggml-cpu/CMakeLists.txt +++ b/ggml/src/ggml-cpu/CMakeLists.txt @@ -675,6 +675,12 @@ function(ggml_add_cpu_backend_variant_impl tag_name) target_compile_options(${GGML_CPU_NAME} PRIVATE ${ARCH_FLAGS}) target_compile_definitions(${GGML_CPU_NAME} PRIVATE ${ARCH_DEFINITIONS}) + if (CMAKE_C_COMPILER_ID STREQUAL "GNU" AND NOT GGML_SYSTEM_ARCH STREQUAL "x86") + message(STATUS "Skipping PCH for ${GGML_CPU_NAME}: GCC PCH is only enabled for x86 (arch: ${GGML_SYSTEM_ARCH})") + else() + target_precompile_headers(${GGML_CPU_NAME} PRIVATE ggml-impl.h) + endif() + if (EMSCRIPTEN) set_target_properties(${GGML_CPU_NAME} PROPERTIES COMPILE_FLAGS "-msimd128") endif() diff --git a/ggml/src/ggml-cpu/ops.h b/ggml/src/ggml-cpu/ops.h index 4c1642a67..ce2b3e870 100644 --- a/ggml/src/ggml-cpu/ops.h +++ b/ggml/src/ggml-cpu/ops.h @@ -18,7 +18,15 @@ #endif #endif +// -Winterference-size was introduced in GCC 12 +#if defined(__cplusplus) && defined(__GNUC__) && !defined(__clang__) && __GNUC__ >= 12 +#pragma GCC diagnostic push +#pragma GCC diagnostic ignored "-Winterference-size" +#endif static const size_t CACHE_LINE_SIZE_F32 = CACHE_LINE_SIZE/sizeof(float); +#if defined(__cplusplus) && defined(__GNUC__) && !defined(__clang__) && __GNUC__ >= 12 +#pragma GCC diagnostic pop +#endif // Work buffer size for im2col operations in CONV2D #define GGML_IM2COL_WORK_SIZE (16 * 1024 * 1024)