7ca5991d2b
* Faster tensors (#8) Add fast matrix and matrix/vector multiplication. * Use map for shader replacements instead of pair of strings * Wasm (#9) * webgpu : fix build on emscripten * more debugging stuff * test-backend-ops: force single thread on wasm * fix single-thread case for init_tensor_uniform * use jspi * add pthread * test: remember to set n_thread for cpu backend * Add buffer label and enable dawn-specific toggles to turn off some checks * Intermediate state * Fast working f16/f32 vec4 * Working float fast mul mat * Clean up naming of mul_mat to match logical model, start work on q mul_mat * Setup for subgroup matrix mat mul * Basic working subgroup matrix * Working subgroup matrix tiling * Handle weirder sg matrix sizes (but still % sg matrix size) * Working start to gemv * working f16 accumulation with shared memory staging * Print out available subgroup matrix configurations * Vectorize dst stores for sg matrix shader * Gemv working scalar * Minor set_rows optimization (#4) * updated optimization, fixed errors * non vectorized version now dispatches one thread per element * Simplify * Change logic for set_rows pipelines --------- Co-authored-by: Neha Abbas <nehaabbas@macbookpro.lan> Co-authored-by: Neha Abbas <nehaabbas@ReeseLevines-MacBook-Pro.local> Co-authored-by: Reese Levine <reeselevine1@gmail.com> * Comment on dawn toggles * Working subgroup matrix code for (semi)generic sizes * Remove some comments * Cleanup code * Update dawn version and move to portable subgroup size * Try to fix new dawn release * Update subgroup size comment * Only check for subgroup matrix configs if they are supported * Add toggles for subgroup matrix/f16 support on nvidia+vulkan * Make row/col naming consistent * Refactor shared memory loading * Move sg matrix stores to correct file * Working q4_0 * Formatting * Work with emscripten builds * Fix test-backend-ops emscripten for f16/quantized types * Use emscripten memory64 to support get_memory * Add build flags and try ci --------- Co-authored-by: Xuan Son Nguyen <son@huggingface.co> * Remove extra whitespace * Move wasm single-thread logic out of test-backend-ops for cpu backend * Disable multiple threads for emscripten single-thread builds in ggml_graph_plan * Fix .gitignore * Add memory64 option and remove unneeded macros for setting threads to 1 --------- Co-authored-by: Xuan Son Nguyen <son@huggingface.co>
81 lines
2.8 KiB
CMake
81 lines
2.8 KiB
CMake
cmake_minimum_required(VERSION 3.13)
|
|
|
|
find_package(Python3 REQUIRED)
|
|
|
|
# Shader locations
|
|
set(SHADER_DIR "${CMAKE_CURRENT_SOURCE_DIR}/wgsl-shaders")
|
|
set(SHADER_OUTPUT_DIR "${CMAKE_CURRENT_BINARY_DIR}/generated")
|
|
set(SHADER_HEADER "${SHADER_OUTPUT_DIR}/ggml-wgsl-shaders.hpp")
|
|
file(MAKE_DIRECTORY ${SHADER_OUTPUT_DIR})
|
|
|
|
message(STATUS "Shader output dir: ${SHADER_OUTPUT_DIR}")
|
|
|
|
# Find all WGSL files
|
|
file(GLOB WGSL_SHADER_FILES "${SHADER_DIR}/*.wgsl")
|
|
|
|
# Generate the header using a Python script
|
|
add_custom_command(
|
|
OUTPUT ${SHADER_HEADER}
|
|
COMMAND ${CMAKE_COMMAND} -E echo "Embedding WGSL shaders to ggml-wgsl-shaders.hpp"
|
|
COMMAND ${CMAKE_COMMAND} -E make_directory ${SHADER_OUTPUT_DIR}
|
|
COMMAND ${CMAKE_COMMAND} -E env PYTHONIOENCODING=utf-8
|
|
${Python3_EXECUTABLE} ${CMAKE_CURRENT_SOURCE_DIR}/wgsl-shaders/embed_wgsl.py
|
|
--input_dir "${SHADER_DIR}"
|
|
--output_file "${SHADER_HEADER}"
|
|
DEPENDS ${WGSL_SHADER_FILES} ${CMAKE_CURRENT_SOURCE_DIR}/wgsl-shaders/embed_wgsl.py
|
|
VERBATIM
|
|
)
|
|
|
|
add_custom_target(generate_shaders DEPENDS ${SHADER_HEADER})
|
|
|
|
ggml_add_backend_library(ggml-webgpu
|
|
ggml-webgpu.cpp
|
|
${SHADER_HEADER}
|
|
../../include/ggml-webgpu.h
|
|
)
|
|
|
|
add_dependencies(ggml-webgpu generate_shaders)
|
|
|
|
if(EMSCRIPTEN)
|
|
set(EMDAWNWEBGPU_DIR "" CACHE PATH "Path to emdawnwebgpu_pkg")
|
|
|
|
if(NOT EMDAWNWEBGPU_DIR)
|
|
# default built-in port
|
|
target_compile_options(ggml-webgpu PRIVATE "--use-port=emdawnwebgpu")
|
|
target_link_options(ggml-webgpu INTERFACE "--use-port=emdawnwebgpu")
|
|
else()
|
|
# custom port
|
|
target_compile_options(ggml-webgpu PRIVATE "--use-port=${EMDAWNWEBGPU_DIR}/emdawnwebgpu.port.py")
|
|
target_link_options(ggml-webgpu INTERFACE "--use-port=${EMDAWNWEBGPU_DIR}/emdawnwebgpu.port.py")
|
|
endif()
|
|
|
|
if (GGML_WEBGPU_JSPI)
|
|
target_compile_options(ggml-webgpu PRIVATE "-fwasm-exceptions")
|
|
target_link_options(ggml-webgpu INTERFACE "-sJSPI" "-fwasm-exceptions")
|
|
else()
|
|
target_compile_options(ggml-webgpu PRIVATE "-fexceptions")
|
|
target_link_options(ggml-webgpu INTERFACE "-sASYNCIFY" "-exceptions")
|
|
endif()
|
|
else()
|
|
find_package(Dawn REQUIRED)
|
|
set(DawnWebGPU_TARGET dawn::webgpu_dawn)
|
|
endif()
|
|
|
|
if (GGML_WEBGPU_DEBUG)
|
|
target_compile_definitions(ggml-webgpu PRIVATE GGML_WEBGPU_DEBUG=1)
|
|
if(EMSCRIPTEN)
|
|
target_link_options(ggml-webgpu INTERFACE "-sASSERTIONS=2")
|
|
endif()
|
|
endif()
|
|
|
|
if (GGML_WEBGPU_CPU_PROFILE)
|
|
target_compile_definitions(ggml-webgpu PRIVATE GGML_WEBGPU_CPU_PROFILE=1)
|
|
endif()
|
|
|
|
if (GGML_WEBGPU_GPU_PROFILE)
|
|
target_compile_definitions(ggml-webgpu PRIVATE GGML_WEBGPU_GPU_PROFILE=1)
|
|
endif()
|
|
|
|
target_include_directories(ggml-webgpu PRIVATE ${SHADER_OUTPUT_DIR})
|
|
target_link_libraries(ggml-webgpu PRIVATE ${DawnWebGPU_TARGET})
|