mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-09-14 18:02:52 +02:00
* scripts : add initial profiling script (wip)
* src : add precompile headers (PCH) for models.h
* common : add common.h as PCH
* ggml : add PCH for ggml-impl.h
* mtmd : use PCH for models.h
* scripts : add script to build with Server/Tools/Tests
* server : add PCH for common.h
* docs: add profiling progress notes (wip)
* ggml : add exclude for GCC + SVE on ARM
Refs: https://github.com/ggml-org/llama.cpp/actions/runs/33393906061/job/99493756214?pr=28091
* ggml : attempt to fix use of std::hardware_destructive_inference_size
Refs: https://github.com/ggml-org/llama.cpp/actions/runs/33396221677/job/99501265689?pr=28091
* squash! ggml : attempt to fix use of std::hardware_destructive_inference_size
Add a version check for GCC 12 to conditionally apply the `-Winterference-size`
pragma.
* editorconfig : exclude profiling reports dir
This directory will not be included in the merge later and this commit
can be ignore at that point. Just fixing to keep CI happy.
* ggml : skip PCH for gcc on non-x86 architectures
* tests : add PCH for peg-parser/tests.h
There are 7 peg-parser tests that can share one PCH instead of then each
parsing the full tests.h.
* common : add PCH for chat.h
* docs : update linux build profiling full results
Just updating after a number of PCH additions. These are not exact
figures and will vary a bit from run to run, but they give a general idea
of the performance impact of PCH.
* cmake : introduce unity build for models
This commit introduces a unity build for the models to improve
compilation time.
The improvements were roughly the following:
```console
+------------------------+-----+------------+------------+------------+
| Build | TUs | Frontend | Backend | Total |
+------------------------+-----+------------+------------+------------+
| Full, master | 396 | 811.0 s | 692.2 s | 1,503.2 s |
| Full, with PCH | 405 | 380.0 s | 664.7 s | 1,044.7 s |
| Full, with PCH + UB | 264 | 357.7 s | 635.7 s | 993.4 s |
+------------------------+-----+------------+------------+------------+
TU = Translation Unit.
Full = includes Server, Tools, and Tests.
PCH = precompiled headers.
UB = unity build for models.
```
* docs : update linux profiling table with unitiy build results
* docs : update mac profiling results to include unity build [no ci]
* docs: remove profiling reports
* scripts : merge build profile scripts into one script
I was lazy before and just copied the first script to enable Tests,
Server, and Tools. This now merges them into a single script.
* Revert "editorconfig : exclude profiling reports dir" [no ci]
This reverts commit 2922a12118.
* src : rename ggml_view_2d_slice to gemma3n_view_2d_slice
This is to be consistent with the rename in gemma4.cpp which was
required to avoid a name clash.
* cmake : add build profile script for windows [no ci]
This commit adds a port of the scripts/build-profile.sh script to
windows powershell.
This was developed on Windows on ARM but should work on X64 as well but
needs to be tested there as well.
187 lines
5.0 KiB
CMake
187 lines
5.0 KiB
CMake
find_package(Threads REQUIRED)
|
|
|
|
llama_add_compile_flags()
|
|
|
|
#
|
|
# llama-common-base
|
|
#
|
|
|
|
# Build info header
|
|
|
|
if(EXISTS "${PROJECT_SOURCE_DIR}/.git")
|
|
set(GIT_DIR "${PROJECT_SOURCE_DIR}/.git")
|
|
|
|
# Is git submodule
|
|
if(NOT IS_DIRECTORY "${GIT_DIR}")
|
|
file(READ ${GIT_DIR} REAL_GIT_DIR_LINK)
|
|
string(REGEX REPLACE "gitdir: (.*)\n$" "\\1" REAL_GIT_DIR ${REAL_GIT_DIR_LINK})
|
|
string(FIND "${REAL_GIT_DIR}" "/" SLASH_POS)
|
|
if (SLASH_POS EQUAL 0)
|
|
set(GIT_DIR "${REAL_GIT_DIR}")
|
|
else()
|
|
set(GIT_DIR "${PROJECT_SOURCE_DIR}/${REAL_GIT_DIR}")
|
|
endif()
|
|
endif()
|
|
|
|
if(EXISTS "${GIT_DIR}/index")
|
|
# For build-info.cpp below
|
|
set_property(DIRECTORY APPEND PROPERTY CMAKE_CONFIGURE_DEPENDS "${GIT_DIR}/index")
|
|
else()
|
|
message(WARNING "Git index not found in git repository.")
|
|
endif()
|
|
else()
|
|
message(WARNING "Git repository not found; to enable automatic generation of build info, make sure Git is installed and the project is a Git repository.")
|
|
endif()
|
|
|
|
set(TEMPLATE_FILE "${CMAKE_CURRENT_SOURCE_DIR}/build-info.cpp.in")
|
|
set(OUTPUT_FILE "${CMAKE_CURRENT_BINARY_DIR}/build-info.cpp")
|
|
|
|
configure_file(${TEMPLATE_FILE} ${OUTPUT_FILE})
|
|
|
|
set(TARGET llama-common-base)
|
|
add_library(${TARGET} STATIC ${OUTPUT_FILE})
|
|
|
|
target_include_directories(${TARGET} PUBLIC .)
|
|
|
|
if (BUILD_SHARED_LIBS)
|
|
set_target_properties(${TARGET} PROPERTIES POSITION_INDEPENDENT_CODE ON)
|
|
endif()
|
|
|
|
#
|
|
# llama-common
|
|
#
|
|
|
|
set(TARGET llama-common)
|
|
|
|
include(parsers/sources.cmake)
|
|
|
|
add_library(${TARGET}
|
|
${LLAMA_CHAT_PARSERS_SOURCES}
|
|
arg.cpp
|
|
arg.h
|
|
base64.hpp
|
|
chat-auto-parser-generator.cpp
|
|
chat-auto-parser-helpers.cpp
|
|
chat-auto-parser.h
|
|
chat-diff-analyzer.cpp
|
|
chat-peg-parser.cpp
|
|
chat-peg-parser.h
|
|
chat.cpp
|
|
chat.h
|
|
common.cpp
|
|
common.h
|
|
console.cpp
|
|
console.h
|
|
debug.cpp
|
|
debug.h
|
|
download.cpp
|
|
download.h
|
|
fit.cpp
|
|
fit.h
|
|
hf-cache.cpp
|
|
hf-cache.h
|
|
http.h
|
|
imatrix-loader.cpp
|
|
imatrix-loader.h
|
|
json-schema-to-grammar.cpp
|
|
json.cpp
|
|
json.h
|
|
llguidance.cpp
|
|
log.cpp
|
|
log.h
|
|
ngram-cache.cpp
|
|
ngram-cache.h
|
|
ngram-map.cpp
|
|
ngram-map.h
|
|
ngram-mod.cpp
|
|
ngram-mod.h
|
|
peg-parser.cpp
|
|
peg-parser.h
|
|
preset.cpp
|
|
preset.h
|
|
reasoning-budget.cpp
|
|
reasoning-budget.h
|
|
sampling.cpp
|
|
sampling.h
|
|
speculative.cpp
|
|
speculative.h
|
|
subproc.cpp
|
|
subproc.h
|
|
trie.cpp
|
|
trie.h
|
|
unicode.cpp
|
|
unicode.h
|
|
jinja/lexer.cpp
|
|
jinja/lexer.h
|
|
jinja/parser.cpp
|
|
jinja/parser.h
|
|
jinja/runtime.cpp
|
|
jinja/runtime.h
|
|
jinja/value.cpp
|
|
jinja/value.h
|
|
jinja/string.cpp
|
|
jinja/string.h
|
|
jinja/caps.cpp
|
|
jinja/caps.h
|
|
)
|
|
|
|
set_target_properties(${TARGET} PROPERTIES
|
|
VERSION ${LLAMA_VERSION_BASE}
|
|
SOVERSION ${LLAMA_VERSION_MAJOR}
|
|
MACHO_CURRENT_VERSION 0 # keep macOS linker from seeing oversized version number
|
|
)
|
|
|
|
target_include_directories(${TARGET} PUBLIC .)
|
|
target_link_libraries (${TARGET} PUBLIC vendor::nlohmann vendor::sheredom)
|
|
target_compile_features (${TARGET} PUBLIC cxx_std_17)
|
|
target_precompile_headers (${TARGET} PRIVATE common.h)
|
|
target_precompile_headers (${TARGET} PRIVATE chat.h)
|
|
|
|
if (LLAMA_SUBPROCESS)
|
|
target_compile_definitions(${TARGET} PUBLIC LLAMA_SUBPROCESS)
|
|
endif()
|
|
|
|
if (BUILD_SHARED_LIBS)
|
|
set_target_properties(${TARGET} PROPERTIES POSITION_INDEPENDENT_CODE ON)
|
|
|
|
# TODO: make fine-grained exports in the future
|
|
set_target_properties(${TARGET} PROPERTIES WINDOWS_EXPORT_ALL_SYMBOLS ON)
|
|
endif()
|
|
|
|
target_link_libraries(${TARGET} PUBLIC llama-common-base)
|
|
target_link_libraries(${TARGET} PRIVATE cpp-httplib)
|
|
|
|
if (LLAMA_LLGUIDANCE)
|
|
include(ExternalProject)
|
|
set(LLGUIDANCE_SRC ${CMAKE_BINARY_DIR}/llguidance/source)
|
|
set(LLGUIDANCE_PATH ${LLGUIDANCE_SRC}/target/release)
|
|
set(LLGUIDANCE_LIB_NAME "${CMAKE_STATIC_LIBRARY_PREFIX}llguidance${CMAKE_STATIC_LIBRARY_SUFFIX}")
|
|
|
|
ExternalProject_Add(llguidance_ext
|
|
GIT_REPOSITORY https://github.com/guidance-ai/llguidance
|
|
# v1.0.1:
|
|
GIT_TAG d795912fedc7d393de740177ea9ea761e7905774
|
|
PREFIX ${CMAKE_BINARY_DIR}/llguidance
|
|
SOURCE_DIR ${LLGUIDANCE_SRC}
|
|
BUILD_IN_SOURCE TRUE
|
|
CONFIGURE_COMMAND ""
|
|
BUILD_COMMAND cargo build --release --package llguidance
|
|
INSTALL_COMMAND ""
|
|
BUILD_BYPRODUCTS ${LLGUIDANCE_PATH}/${LLGUIDANCE_LIB_NAME} ${LLGUIDANCE_PATH}/llguidance.h
|
|
UPDATE_COMMAND ""
|
|
)
|
|
target_compile_definitions(${TARGET} PUBLIC LLAMA_USE_LLGUIDANCE)
|
|
|
|
add_library(llguidance STATIC IMPORTED)
|
|
set_target_properties(llguidance PROPERTIES IMPORTED_LOCATION ${LLGUIDANCE_PATH}/${LLGUIDANCE_LIB_NAME})
|
|
add_dependencies(llguidance llguidance_ext)
|
|
|
|
target_include_directories(${TARGET} PRIVATE ${LLGUIDANCE_PATH})
|
|
target_link_libraries(${TARGET} PRIVATE llguidance)
|
|
if (WIN32)
|
|
target_link_libraries(${TARGET} PRIVATE ws2_32 userenv ntdll bcrypt)
|
|
endif()
|
|
endif()
|
|
|
|
target_link_libraries(${TARGET} PUBLIC llama Threads::Threads)
|