diff --git a/packages/embed-llamacpp/vcpkg-configuration.json b/packages/embed-llamacpp/vcpkg-configuration.json index e894485aaa..c383e60f96 100644 --- a/packages/embed-llamacpp/vcpkg-configuration.json +++ b/packages/embed-llamacpp/vcpkg-configuration.json @@ -14,5 +14,8 @@ "spirv-headers" ] } + ], + "overlay-ports": [ + "../../vcpkg-overlays/ports" ] } diff --git a/packages/fabric/vcpkg-configuration.json b/packages/fabric/vcpkg-configuration.json index 39265c0d89..37335a2da5 100644 --- a/packages/fabric/vcpkg-configuration.json +++ b/packages/fabric/vcpkg-configuration.json @@ -13,5 +13,8 @@ "spirv-headers" ] } + ], + "overlay-ports": [ + "../../vcpkg-overlays/ports" ] } diff --git a/packages/llm-llamacpp/vcpkg-configuration.json b/packages/llm-llamacpp/vcpkg-configuration.json index 74bbaaec5a..98bd001d6b 100644 --- a/packages/llm-llamacpp/vcpkg-configuration.json +++ b/packages/llm-llamacpp/vcpkg-configuration.json @@ -17,5 +17,8 @@ "spirv-headers" ] } + ], + "overlay-ports": [ + "../../vcpkg-overlays/ports" ] } diff --git a/packages/model-fit/vcpkg-configuration.json b/packages/model-fit/vcpkg-configuration.json index 74bbaaec5a..98bd001d6b 100644 --- a/packages/model-fit/vcpkg-configuration.json +++ b/packages/model-fit/vcpkg-configuration.json @@ -17,5 +17,8 @@ "spirv-headers" ] } + ], + "overlay-ports": [ + "../../vcpkg-overlays/ports" ] } diff --git a/packages/ocr-ggml/vcpkg-configuration.json b/packages/ocr-ggml/vcpkg-configuration.json index 7531c31bb3..cbab73712f 100644 --- a/packages/ocr-ggml/vcpkg-configuration.json +++ b/packages/ocr-ggml/vcpkg-configuration.json @@ -63,5 +63,8 @@ "spirv-headers" ] } + ], + "overlay-ports": [ + "../../vcpkg-overlays/ports" ] } diff --git a/packages/translation-nmtcpp/vcpkg-configuration.json b/packages/translation-nmtcpp/vcpkg-configuration.json index b49381752c..084da01bf4 100644 --- a/packages/translation-nmtcpp/vcpkg-configuration.json +++ b/packages/translation-nmtcpp/vcpkg-configuration.json @@ -80,5 +80,8 @@ "spirv-headers" ] } + ], + "overlay-ports": [ + "../../vcpkg-overlays/ports" ] } diff --git a/packages/vla-ggml/vcpkg-configuration.json b/packages/vla-ggml/vcpkg-configuration.json index 38cb9910d8..0386c1cef4 100644 --- a/packages/vla-ggml/vcpkg-configuration.json +++ b/packages/vla-ggml/vcpkg-configuration.json @@ -14,5 +14,8 @@ "spirv-headers" ] } + ], + "overlay-ports": [ + "../../vcpkg-overlays/ports" ] } diff --git a/vcpkg-overlays/ports/qvac-fabric/portfile.cmake b/vcpkg-overlays/ports/qvac-fabric/portfile.cmake new file mode 100644 index 0000000000..68eb7706c4 --- /dev/null +++ b/vcpkg-overlays/ports/qvac-fabric/portfile.cmake @@ -0,0 +1,237 @@ +vcpkg_from_github( + OUT_SOURCE_PATH SOURCE_PATH + REPO tetherto/qvac-fabric-llm.cpp + # QVAC-23075 rollout Phase A: validate the 7 consumers against the VisionPsy fabric + # branch head before the tag exists. A branch SHA, not v${VERSION}, on purpose. + # Replaced by the published tag at the registry publish, so this overlay is temporary. + REF 4ef2b3fdc0788d38e4e176030c07241ede40c5d0 + SHA512 d37f50c9097fdbbb5af4c26ff130d1725aa431611220195a0a7f9fbbfb8acb0663593b09084f499145cc164e18ba2db96230005b885a38b942b58c7c79c675d2 + HEAD_REF main +) + +# Upstream CMake options only, passed through to vcpkg_cmake_configure. +vcpkg_check_features( + OUT_FEATURE_OPTIONS FEATURE_OPTIONS + FEATURES + force-profiler FORCE_GGML_VK_PERF_LOGGER + llama BUILD_LLAMA +) + +# Portfile-only feature flags (drive PLATFORM_OPTIONS; not upstream cache vars). +vcpkg_check_features( + OUT_FEATURE_OPTIONS _PORTFILE_FEATURE_OPTIONS + FEATURES + gpu-backends BUILD_GPU_BACKENDS + kleidiai BUILD_KLEIDIAI + openmp BUILD_OPENMP + hip-backend BUILD_HIP_BACKEND +) + +# gpu-backends is default-on via default-features in vcpkg.json. CPU-only +# consumers (e.g. @qvac/classification-ggml) disable it with +# default-features:false (and re-add 'llama' if needed). +if(NOT BUILD_GPU_BACKENDS) + message(STATUS "qvac-fabric: gpu-backends feature OFF, building CPU-only ggml (no Metal/Vulkan/CUDA/OpenCL)") +endif() + +set(PLATFORM_OPTIONS) + +if (VCPKG_TARGET_IS_ANDROID AND BUILD_GPU_BACKENDS) + # The Android NDK ships only the C Vulkan headers; the ggml Vulkan backend + # additionally needs the C++ bindings (vulkan.hpp) and SPIRV-Headers, which as + # of b9840 ggml fetches itself via FetchContent (ggml/src/ggml-vulkan/CMakeLists.txt, + # `if (ANDROID)` block). The registry vcpkg-cmake sets FETCHCONTENT_FULLY_DISCONNECTED=ON + # globally, so allow the fetch here (same as the kleidiai path below). + list(APPEND PLATFORM_OPTIONS -DFETCHCONTENT_FULLY_DISCONNECTED=OFF) +endif() + +if(NOT BUILD_GPU_BACKENDS) + # Force every GPU backend off explicitly, in case upstream defaults change. + list(APPEND PLATFORM_OPTIONS + -DGGML_METAL=OFF + -DGGML_VULKAN=OFF + -DGGML_CUDA=OFF + -DGGML_OPENCL=OFF + ) + if (VCPKG_TARGET_IS_IOS) + # Same iOS BLAS/Accelerate gating as the GPU-on path; unrelated to the + # CPU-vs-GPU split, an iOS-toolchain workaround for missing frameworks. + list(APPEND PLATFORM_OPTIONS -DGGML_BLAS=OFF -DGGML_ACCELERATE=OFF) + endif() +elseif (VCPKG_TARGET_IS_OSX OR VCPKG_TARGET_IS_IOS) + list(APPEND PLATFORM_OPTIONS -DGGML_METAL=ON) + if (VCPKG_TARGET_IS_IOS) + list(APPEND PLATFORM_OPTIONS -DGGML_BLAS=OFF -DGGML_ACCELERATE=OFF) + endif() +else() + list(APPEND PLATFORM_OPTIONS -DGGML_VULKAN=ON) +endif() + +# Android: always build CPU variants (NEON_DOTPROD, NEON_I8MM, etc.) and CPU +# repacking. These are CPU-only runtime optimizations selected based on the +# device's SIMD capabilities at load time, completely orthogonal to the GPU +# backends. Bundling them is essential for good CPU inference performance on +# the wide range of arm64 devices the addons ship to. Requires GGML_BACKEND_DL +# to dispatch the variants at runtime; the existing #ifdef guard around +# `ggml_backend_load_all_from_path()` in ggml-backend-reg.cpp keeps the search +# scoped to the consumer's own prebuilds dir. +if(VCPKG_TARGET_IS_ANDROID OR (VCPKG_TARGET_IS_LINUX AND BUILD_GPU_BACKENDS)) + # Desktop Linux also needs GGML_BACKEND_DL=ON so that multiple GPU backends + # (Vulkan + HIP/ROCm) can coexist as separately-loaded modules, the same way + # Android dispatches CPU variants at runtime. Without DL the Linux build links + # a single static GPU backend and a second one (HIP) cannot be stacked. + # GGML_NATIVE is incompatible with DL, so CPU variants are dispatched via + # GGML_CPU_ALL_VARIANTS instead. Consumers must ship the core ggml/llama libs + # alongside their backend modules so the dynamically-linked .bare can resolve + # them at load time. + set(DL_BACKENDS ON) + list(APPEND PLATFORM_OPTIONS + -DGGML_BACKEND_DL=ON + -DGGML_CPU_ALL_VARIANTS=ON + -DGGML_CPU_REPACK=ON) +else() + set(DL_BACKENDS OFF) +endif() + +# HIP/ROCm backend, opt-in via the 'hip-backend' feature (Linux + AMD only). +# Only @qvac/vla-ggml requests it, so every other consumer builds with no HIP +# and gains no ROCm dependency. Builds libqvac-ggml-hip.so as a standalone DL +# module alongside Vulkan (GGML_BACKEND_DL is already ON above), so the addon +# dlopen's whichever GPU backend BackendSelection picks at runtime. The `hip` +# feature-dependency port forwards the system ROCm's find_package() configs. +# +# FAIL-SAFE: enable GGML_HIP only when a ROCm SDK is actually present. On a build +# host without ROCm we skip HIP and build Vulkan/CPU only, so the build never +# hard-fails, and at runtime a missing HIP module just isn't loaded (the DL +# loader skips it) so BackendSelection falls back to Vulkan/CPU. Targets gfx1151 +# (Strix Halo / Radeon 8060S); the HIP compiler + ROCM_PATH come from the build env. +# linux-x64 only: AMD GPU hosts (Strix Halo / gfx1151) are x86_64, and the ROCm +# dist is x64. On other arches (e.g. linux-arm64) HIP is skipped even if the +# feature is requested, so no ROCm requirement and no build break. +if(VCPKG_TARGET_IS_LINUX AND VCPKG_TARGET_ARCHITECTURE STREQUAL "x64" AND BUILD_GPU_BACKENDS AND BUILD_HIP_BACKEND) + # DETERMINISTIC: requesting hip-backend REQUIRES a ROCm SDK at build time. We + # must NOT silently skip when ROCm is absent, because a host-dependent skip yields a + # no-HIP package with the SAME vcpkg ABI as a real HIP build, which the binary + # cache then conflates (cache poisoning: a no-ROCm build caches a no-HIP + # package that ROCm-equipped builds then restore). So ROCm present => HIP; + # ROCm absent => hard error (don't request hip-backend on a host without ROCm). + # The RUNTIME fail-safe is unchanged: an absent HIP module / non-AMD target is + # simply not loaded and BackendSelection falls back to Vulkan/CPU. + if(NOT (DEFINED ENV{ROCM_PATH} AND EXISTS "$ENV{ROCM_PATH}/lib/cmake/hip/hip-config.cmake")) + message(FATAL_ERROR "qvac-fabric: hip-backend feature requires a ROCm SDK. Set ROCM_PATH to a ROCm/TheRock install containing lib/cmake/hip/hip-config.cmake. Do not request hip-backend on a host without ROCm.") + endif() + message(STATUS "qvac-fabric: hip-backend ON, building GGML_HIP (gfx1151)") + list(APPEND PLATFORM_OPTIONS + -DGGML_HIP=ON + -DAMDGPU_TARGETS=gfx1151 + -DCMAKE_HIP_ARCHITECTURES=gfx1151) +endif() + +if(VCPKG_TARGET_IS_ANDROID AND BUILD_KLEIDIAI) + message(STATUS "qvac-fabric: kleidiai feature ON, building with ARM KleidiAI optimized kernels") + # ggml only vendors KleidiAI via FetchContent; registry vcpkg-cmake sets + # FETCHCONTENT_FULLY_DISCONNECTED=ON globally, so allow the download here. + list(APPEND PLATFORM_OPTIONS + -DGGML_CPU_KLEIDIAI=ON + -DFETCHCONTENT_FULLY_DISCONNECTED=OFF + ) +endif() + +if(VCPKG_TARGET_IS_ANDROID AND BUILD_OPENMP) + message(STATUS "qvac-fabric: OpenMP for Android enabled") + list(APPEND PLATFORM_OPTIONS -DGGML_OPENMP=ON) +else() + message(STATUS "qvac-fabric: OpenMP Disabled") + list(APPEND PLATFORM_OPTIONS -DGGML_OPENMP=OFF) +endif() + +if (VCPKG_TARGET_IS_ANDROID AND BUILD_GPU_BACKENDS) + list(APPEND PLATFORM_OPTIONS -DGGML_OPENCL=ON) +endif() + +if(BUILD_GPU_BACKENDS AND NOT VCPKG_TARGET_IS_OSX AND NOT VCPKG_TARGET_IS_IOS) + if(VCPKG_TARGET_IS_WINDOWS AND NOT VCPKG_TARGET_IS_MINGW) + string(APPEND VCPKG_C_FLAGS " /I${CURRENT_INSTALLED_DIR}/include") + string(APPEND VCPKG_CXX_FLAGS " /I${CURRENT_INSTALLED_DIR}/include") + else() + string(APPEND VCPKG_C_FLAGS " -isystem ${CURRENT_INSTALLED_DIR}/include") + string(APPEND VCPKG_CXX_FLAGS " -isystem ${CURRENT_INSTALLED_DIR}/include") + endif() +endif() + +# Under GGML_BACKEND_DL the per-microarch backends ship as standalone +# libqvac-ggml-*.so modules that the consumer dlopen's at runtime. Built with +# -stdlib=libc++ they otherwise carry a runtime NEEDED dependency on the system +# libc++.so.1 / libc++abi.so.1, so they silently fail to dlopen on any target +# without libc++ installed (e.g. stock ubuntu-24.04, no CPU backend registers, +# inference aborts). Statically link the C++ runtime into the modules so they +# are self-contained, matching how the addons link themselves. The module<->addon +# boundary is the C ggml-backend ABI, so per-module libc++ copies never exchange +# C++ objects. Linux only: Apple/iOS use Metal frameworks, Android ships +# libc++_shared via the NDK STL, Windows uses the MSVC runtime. +if(VCPKG_TARGET_IS_LINUX AND DL_BACKENDS) + string(APPEND VCPKG_LINKER_FLAGS " -static-libstdc++") +endif() + +set(LLAMA_OPTIONS) +if("llama" IN_LIST FEATURES) + list(APPEND LLAMA_OPTIONS -DLLAMA_MTMD=ON) +else() + list(APPEND LLAMA_OPTIONS + -DLLAMA_MTMD=OFF + -DLLAMA_BUILD_COMMON=OFF + ) +endif() + +vcpkg_cmake_configure( + SOURCE_PATH "${SOURCE_PATH}" + DISABLE_PARALLEL_CONFIGURE + OPTIONS + -DGGML_NATIVE=OFF + -DGGML_CCACHE=OFF + -DGGML_LLAMAFILE=OFF + -DLLAMA_CURL=OFF + -DLLAMA_BUILD_TESTS=OFF + -DLLAMA_BUILD_TOOLS=OFF + -DLLAMA_BUILD_EXAMPLES=OFF + -DLLAMA_BUILD_SERVER=OFF + -DLLAMA_BUILD_APP=OFF + -DMTMD_VIDEO=OFF + -DLLAMA_ALL_WARNINGS=OFF + ${LLAMA_OPTIONS} + ${PLATFORM_OPTIONS} + ${FEATURE_OPTIONS} +) + +vcpkg_cmake_install() +vcpkg_cmake_config_fixup( + PACKAGE_NAME ggml) + +if(BUILD_LLAMA) + vcpkg_cmake_config_fixup(PACKAGE_NAME llama) +endif() + +vcpkg_copy_pdbs() +vcpkg_fixup_pkgconfig() + + +if(BUILD_LLAMA) + file(MAKE_DIRECTORY "${CURRENT_PACKAGES_DIR}/tools/${PORT}") + file(RENAME "${CURRENT_PACKAGES_DIR}/bin/convert_hf_to_gguf.py" "${CURRENT_PACKAGES_DIR}/tools/${PORT}/convert-hf-to-gguf.py") + file(INSTALL "${SOURCE_PATH}/gguf-py" DESTINATION "${CURRENT_PACKAGES_DIR}/tools/${PORT}") + file(RENAME "${CURRENT_PACKAGES_DIR}/bin/vulkan_profiling_analyzer.py" "${CURRENT_PACKAGES_DIR}/tools/${PORT}/vulkan_profiling_analyzer.py") +endif() + +if (NOT VCPKG_BUILD_TYPE) + file(REMOVE "${CURRENT_PACKAGES_DIR}/debug/bin/convert_hf_to_gguf.py") +endif() + +file(REMOVE_RECURSE "${CURRENT_PACKAGES_DIR}/debug/include") +file(REMOVE_RECURSE "${CURRENT_PACKAGES_DIR}/debug/share") + +if (VCPKG_LIBRARY_LINKAGE MATCHES "static") + file(REMOVE_RECURSE "${CURRENT_PACKAGES_DIR}/bin") + file(REMOVE_RECURSE "${CURRENT_PACKAGES_DIR}/debug/bin") +endif() + +vcpkg_install_copyright(FILE_LIST "${SOURCE_PATH}/LICENSE") diff --git a/vcpkg-overlays/ports/qvac-fabric/vcpkg.json b/vcpkg-overlays/ports/qvac-fabric/vcpkg.json new file mode 100644 index 0000000000..ecd3f5b9e9 --- /dev/null +++ b/vcpkg-overlays/ports/qvac-fabric/vcpkg.json @@ -0,0 +1,58 @@ +{ + "name": "qvac-fabric", + "version": "10069.1.0", + "description": "LLM inference in C/C++", + "homepage": "https://github.com/tetherto/qvac-fabric-llm.cpp", + "license": "MIT", + "dependencies": [ + { + "name": "opencl", + "platform": "android" + }, + { + "name": "vcpkg-cmake", + "host": true + }, + { + "name": "vcpkg-cmake-config", + "host": true + } + ], + "default-features": [ + "gpu-backends", + "llama" + ], + "features": { + "force-profiler": { + "description": "Force vk performance logging in ggml" + }, + "gpu-backends": { + "description": "Build the GPU backends ggml ships per platform: Metal on Apple, Vulkan on Linux/Windows/Android, plus the Android backend-DL hybrid mode and OpenCL. Default-on so existing consumers (llamacpp-llm, llamacpp-embed, nmtcpp, diffusion-cpp) keep their current behaviour with default features. Disable to produce a CPU-only ggml build (useful for consumers like @qvac/classification-ggml that don't need GPU paths and want to skip the vulkan-sdk / metal / opencl build cost). Orthogonal to the 'llama' feature.", + "dependencies": [ + { + "name": "spirv-headers", + "platform": "!osx & !ios", + "version>=": "1.4.341.0" + } + ] + }, + "hip-backend": { + "description": "Build the ROCm/HIP GPU backend (libqvac-ggml-hip.so) for AMD GPUs on Linux as a DL module alongside Vulkan. Opt-in (only @qvac/vla-ggml requests it). Fail-safe: if no ROCm SDK is present at build time the HIP backend is skipped (Vulkan/CPU only). Targets gfx1151 (Strix Halo). The 'hip' dependency forwards the system ROCm find_package configs.", + "dependencies": [ + { + "name": "hip", + "platform": "linux & x64" + } + ] + }, + "kleidiai": { + "description": "Enable ARM KleidiAI optimized kernels on Android." + }, + "llama": { + "description": "Build llama components." + }, + "openmp": { + "description": "Enable openmp on Android." + } + } +}